aboutsummaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--.ai/metrics/work-the-backlog.jsonl15
-rw-r--r--.ai/notes.org16
-rw-r--r--.ai/protocols.org95
-rw-r--r--.ai/references/calendar-reference.org66
-rwxr-xr-x.ai/scripts/agent-lock248
-rwxr-xr-x.ai/scripts/agent-roster84
-rwxr-xr-x.ai/scripts/apkg-to-orgdrill.py251
-rwxr-xr-x.ai/scripts/capture-guard91
-rwxr-xr-x.ai/scripts/cj-remove-block.py81
-rwxr-xr-x.ai/scripts/cross-agent-comms/cross-agent-discover230
-rw-r--r--.ai/scripts/cross-agent-comms/cross-agent-discover.md155
-rwxr-xr-x.ai/scripts/cross-agent-comms/cross-agent-halt134
-rw-r--r--.ai/scripts/cross-agent-comms/cross-agent-halt.md134
-rwxr-xr-x.ai/scripts/cross-agent-comms/cross-agent-recv250
-rw-r--r--.ai/scripts/cross-agent-comms/cross-agent-recv.md218
-rwxr-xr-x.ai/scripts/cross-agent-comms/cross-agent-resume145
-rw-r--r--.ai/scripts/cross-agent-comms/cross-agent-resume.md117
-rwxr-xr-x.ai/scripts/cross-agent-comms/cross-agent-send356
-rw-r--r--.ai/scripts/cross-agent-comms/cross-agent-send.md199
-rwxr-xr-x.ai/scripts/cross-agent-comms/cross-agent-status185
-rw-r--r--.ai/scripts/cross-agent-comms/cross-agent-status.md139
-rwxr-xr-x.ai/scripts/cross-agent-comms/cross-agent-watch106
-rw-r--r--.ai/scripts/cross-agent-comms/cross-agent-watch.md130
-rwxr-xr-x.ai/scripts/flashcard-stats.py12
-rwxr-xr-x.ai/scripts/flashcard-to-anki.py133
-rwxr-xr-x.ai/scripts/inbox-send.py105
-rwxr-xr-x.ai/scripts/inbox-status1
-rw-r--r--.ai/scripts/lint-org.el388
-rwxr-xr-x.ai/scripts/route-batch175
-rw-r--r--.ai/scripts/route_recommend.py145
-rwxr-xr-x.ai/scripts/self-inject.sh68
-rwxr-xr-x.ai/scripts/session-context-path8
-rwxr-xr-x.ai/scripts/spec-sort715
-rwxr-xr-x.ai/scripts/task-review-staleness.sh40
-rw-r--r--.ai/scripts/tests/agent-lock.bats214
-rw-r--r--.ai/scripts/tests/agent-roster.bats141
-rw-r--r--.ai/scripts/tests/capture-guard.bats130
-rw-r--r--.ai/scripts/tests/flashcard-sync.bats25
-rw-r--r--.ai/scripts/tests/inbox-status.bats12
-rw-r--r--.ai/scripts/tests/lint-org-cli.bats18
-rw-r--r--.ai/scripts/tests/route-batch.bats202
-rw-r--r--.ai/scripts/tests/self-inject.bats78
-rw-r--r--.ai/scripts/tests/spec-sort.bats453
-rw-r--r--.ai/scripts/tests/task-review-staleness.bats53
-rw-r--r--.ai/scripts/tests/test-lint-org.el437
-rw-r--r--.ai/scripts/tests/test-todo-cleanup.el610
-rw-r--r--.ai/scripts/tests/test-wrap-org-table.el42
-rw-r--r--.ai/scripts/tests/test_apkg_to_orgdrill.py301
-rw-r--r--.ai/scripts/tests/test_cj_remove_block.py167
-rw-r--r--.ai/scripts/tests/test_cross_agent_discover.py204
-rw-r--r--.ai/scripts/tests/test_cross_agent_halt.py204
-rw-r--r--.ai/scripts/tests/test_cross_agent_recv.py176
-rw-r--r--.ai/scripts/tests/test_cross_agent_send.py210
-rw-r--r--.ai/scripts/tests/test_cross_agent_status.py165
-rw-r--r--.ai/scripts/tests/test_cross_agent_watch.py155
-rw-r--r--.ai/scripts/tests/test_flashcard_stats.py25
-rw-r--r--.ai/scripts/tests/test_flashcard_to_anki.py92
-rw-r--r--.ai/scripts/tests/test_inbox_send.py235
-rw-r--r--.ai/scripts/tests/test_route_recommend.py152
-rw-r--r--.ai/scripts/tests/test_upcoming_birthdays.py168
-rw-r--r--.ai/scripts/todo-cleanup.el582
-rwxr-xr-x.ai/scripts/upcoming_birthdays.py182
-rw-r--r--.ai/scripts/wrap-org-table.el58
-rw-r--r--.ai/sessions/2026-06-13-12-11-inbox-zero-build-and-session-title.org57
-rw-r--r--.ai/sessions/2026-06-13-15-29-codex-inbox-ack.org30
-rw-r--r--.ai/sessions/2026-06-15-08-43-czsusp-epoch-helper-instance-slices.org61
-rw-r--r--.ai/sessions/2026-06-16-23-37-cross-agent-comms-removal-and-batch-specs.org194
-rw-r--r--.ai/sessions/2026-06-21-02-44-launcher-fix-kb-feature-wrapup-routing.org110
-rw-r--r--.ai/sessions/2026-06-22-01-33-spec-review-fold-coverage-fix-inbox-triage.org57
-rw-r--r--.ai/sessions/2026-06-23-22-36-inbox-guard-bash-bundle-consolidation-spec.org164
-rw-r--r--.ai/sessions/2026-06-24-00-14-inbox-consolidation-wrap-teardown-roam-fix.org85
-rw-r--r--.ai/sessions/2026-06-24-09-27-task-audit-blocked-deps-anki-wrap-teardown.org75
-rw-r--r--.ai/sessions/2026-06-28-15-57-inbox-proposals-shipped-and-task-audit.org278
-rw-r--r--.ai/sessions/2026-06-29-03-56-spec-lifecycle-decision-and-speedrun-ratified.org107
-rw-r--r--.ai/sessions/2026-06-30-13-55-pager-mcp-ssh-alias-and-emacsd-proposals.org101
-rw-r--r--.ai/sessions/2026-07-02-09-29-docs-lifecycle-speedrun-autonomous-loop.org110
-rw-r--r--.ai/sessions/2026-07-04-12-57-audit-closeouts-and-startup-sync-guard.org68
-rw-r--r--.ai/sessions/2026-07-11-02-30-inbox-clearout-six-proposals-shipped.org74
-rw-r--r--.ai/sessions/2026-07-13-23-48-runtime-portability-pager-local-llm.org191
-rw-r--r--.ai/sessions/2026-07-14-02-50-sentry-spec-review-lint-org-fix.org84
-rw-r--r--.ai/sessions/2026-07-17-10-55-inbox-fixes-emacs-wayland-sentry-prep.org238
-rw-r--r--.ai/sessions/2026-07-19-21-15-launcher-hardening-inbox-hook-colloquialisms.org238
-rw-r--r--.ai/sessions/2026-07-20-23-36-signal-pager-and-triage-phone-push.org185
-rw-r--r--.ai/sessions/2026-07-24-18-05-sentry-implement-pass-review-and-hook-fixes.org386
-rw-r--r--.ai/sessions/2026-07-25-09-24-invalid-block-filter-push-task-review.org83
-rw-r--r--.ai/sessions/2026-07-25-15-34-clean-wrap-and-inbox-safe-sync.org85
-rw-r--r--.ai/sessions/2026-07-27-17-02-context-engineering-rightsizing.org176
-rw-r--r--.ai/sessions/2026-07-29-06-36-adversarial-review-flow-and-telegram-fixes.org514
-rw-r--r--.ai/workflows/INDEX.org45
-rw-r--r--.ai/workflows/add-calendar-event.org2
-rw-r--r--.ai/workflows/broadcast.org6
-rw-r--r--.ai/workflows/clean-todo.org21
-rw-r--r--.ai/workflows/code-quality.org90
-rw-r--r--.ai/workflows/create-workflow.org8
-rw-r--r--.ai/workflows/cross-agent-comms.org334
-rw-r--r--.ai/workflows/daily-prep.org7
-rw-r--r--.ai/workflows/delete-calendar-event.org2
-rw-r--r--.ai/workflows/edit-calendar-event.org2
-rw-r--r--.ai/workflows/email-assembly.org2
-rw-r--r--.ai/workflows/extract-email.org2
-rw-r--r--.ai/workflows/find-email.org2
-rw-r--r--.ai/workflows/first-session.org2
-rw-r--r--.ai/workflows/flashcard-review.org2
-rw-r--r--.ai/workflows/helper-mode.org101
-rw-r--r--.ai/workflows/inbox.org524
-rw-r--r--.ai/workflows/journal-entry.org2
-rw-r--r--.ai/workflows/meeting-prep.org2
-rw-r--r--.ai/workflows/meeting-prep.pre-wire.org2
-rw-r--r--.ai/workflows/monitor-inbox.org94
-rw-r--r--.ai/workflows/no-approvals.org12
-rw-r--r--.ai/workflows/open-tasks.org34
-rw-r--r--.ai/workflows/page-me.org53
-rw-r--r--.ai/workflows/process-inbox.org215
-rw-r--r--.ai/workflows/process-meeting-transcript.org26
-rw-r--r--.ai/workflows/read-calendar-events.org2
-rw-r--r--.ai/workflows/readability-audit.org242
-rw-r--r--.ai/workflows/rename-artifact.org2
-rw-r--r--.ai/workflows/send-email.org2
-rw-r--r--.ai/workflows/sentry.org227
-rw-r--r--.ai/workflows/session-harvest.org2
-rw-r--r--.ai/workflows/spec-create.org22
-rw-r--r--.ai/workflows/spec-response.org67
-rw-r--r--.ai/workflows/spec-review.org113
-rw-r--r--.ai/workflows/startup.org115
-rw-r--r--.ai/workflows/status-check.org2
-rw-r--r--.ai/workflows/summarize-emails.org2
-rw-r--r--.ai/workflows/suspend.org143
-rw-r--r--.ai/workflows/sync-email.org2
-rw-r--r--.ai/workflows/task-audit.org40
-rw-r--r--.ai/workflows/task-review.org12
-rw-r--r--.ai/workflows/triage-intake.cmail.org2
-rw-r--r--.ai/workflows/triage-intake.github-prs.org2
-rw-r--r--.ai/workflows/triage-intake.org237
-rw-r--r--.ai/workflows/triage-intake.personal-calendar.org2
-rw-r--r--.ai/workflows/triage-intake.personal-gmail.org23
-rw-r--r--.ai/workflows/triage-intake.telegram.org143
-rw-r--r--.ai/workflows/work-the-backlog.org266
-rw-r--r--.ai/workflows/wrap-it-up.org217
-rw-r--r--.claude/commands/lint-org.md5
-rw-r--r--.claude/commands/refactor.md39
-rw-r--r--.claude/commands/respond-to-cj-comments.md8
-rw-r--r--.claude/commands/start-work.md23
-rw-r--r--.claude/settings.json29
-rw-r--r--.codex/hooks.json30
-rw-r--r--.gitignore8
l---------AGENTS.md1
-rw-r--r--Makefile46
-rw-r--r--README.org22
-rw-r--r--archive/task-archive.org2009
-rw-r--r--claude-rules/commits.md446
-rw-r--r--claude-rules/cross-project.md12
-rw-r--r--claude-rules/daily-drivers.md66
-rw-r--r--claude-rules/desktop-capture.md60
-rw-r--r--claude-rules/docs-lifecycle.md75
-rw-r--r--claude-rules/emacs.md29
-rw-r--r--claude-rules/host-identity.md20
-rw-r--r--claude-rules/interaction.md56
-rw-r--r--claude-rules/knowledge-base.md29
-rw-r--r--claude-rules/locating-craig.md48
-rw-r--r--claude-rules/org-tables.md5
-rw-r--r--claude-rules/subagents.md72
-rw-r--r--claude-rules/testing.md344
-rw-r--r--claude-rules/todo-format.md315
-rw-r--r--claude-rules/triggers.md7
-rw-r--r--claude-rules/ui-prototyping.md93
-rw-r--r--claude-rules/verification.md16
-rw-r--r--claude-rules/working-files.md24
-rw-r--r--claude-templates/.ai/notes.org14
-rw-r--r--claude-templates/.ai/protocols.org95
-rw-r--r--claude-templates/.ai/references/calendar-reference.org66
-rwxr-xr-xclaude-templates/.ai/scripts/agent-lock248
-rwxr-xr-xclaude-templates/.ai/scripts/agent-roster84
-rwxr-xr-xclaude-templates/.ai/scripts/apkg-to-orgdrill.py251
-rwxr-xr-xclaude-templates/.ai/scripts/capture-guard91
-rwxr-xr-xclaude-templates/.ai/scripts/cj-remove-block.py81
-rwxr-xr-xclaude-templates/.ai/scripts/cross-agent-comms/cross-agent-discover230
-rw-r--r--claude-templates/.ai/scripts/cross-agent-comms/cross-agent-discover.md155
-rwxr-xr-xclaude-templates/.ai/scripts/cross-agent-comms/cross-agent-halt134
-rw-r--r--claude-templates/.ai/scripts/cross-agent-comms/cross-agent-halt.md134
-rwxr-xr-xclaude-templates/.ai/scripts/cross-agent-comms/cross-agent-recv250
-rw-r--r--claude-templates/.ai/scripts/cross-agent-comms/cross-agent-recv.md218
-rwxr-xr-xclaude-templates/.ai/scripts/cross-agent-comms/cross-agent-resume145
-rw-r--r--claude-templates/.ai/scripts/cross-agent-comms/cross-agent-resume.md117
-rwxr-xr-xclaude-templates/.ai/scripts/cross-agent-comms/cross-agent-send356
-rw-r--r--claude-templates/.ai/scripts/cross-agent-comms/cross-agent-send.md199
-rwxr-xr-xclaude-templates/.ai/scripts/cross-agent-comms/cross-agent-status185
-rw-r--r--claude-templates/.ai/scripts/cross-agent-comms/cross-agent-status.md139
-rwxr-xr-xclaude-templates/.ai/scripts/cross-agent-comms/cross-agent-watch106
-rw-r--r--claude-templates/.ai/scripts/cross-agent-comms/cross-agent-watch.md130
-rwxr-xr-xclaude-templates/.ai/scripts/flashcard-stats.py12
-rwxr-xr-xclaude-templates/.ai/scripts/flashcard-to-anki.py133
-rwxr-xr-xclaude-templates/.ai/scripts/inbox-send.py105
-rwxr-xr-xclaude-templates/.ai/scripts/inbox-status1
-rw-r--r--claude-templates/.ai/scripts/lint-org.el388
-rwxr-xr-xclaude-templates/.ai/scripts/route-batch175
-rw-r--r--claude-templates/.ai/scripts/route_recommend.py145
-rwxr-xr-xclaude-templates/.ai/scripts/self-inject.sh68
-rwxr-xr-xclaude-templates/.ai/scripts/session-context-path8
-rwxr-xr-xclaude-templates/.ai/scripts/spec-sort715
-rwxr-xr-xclaude-templates/.ai/scripts/task-review-staleness.sh40
-rw-r--r--claude-templates/.ai/scripts/tests/agent-lock.bats214
-rw-r--r--claude-templates/.ai/scripts/tests/agent-roster.bats141
-rw-r--r--claude-templates/.ai/scripts/tests/capture-guard.bats130
-rw-r--r--claude-templates/.ai/scripts/tests/flashcard-sync.bats25
-rw-r--r--claude-templates/.ai/scripts/tests/inbox-status.bats12
-rw-r--r--claude-templates/.ai/scripts/tests/lint-org-cli.bats18
-rw-r--r--claude-templates/.ai/scripts/tests/route-batch.bats202
-rw-r--r--claude-templates/.ai/scripts/tests/self-inject.bats78
-rw-r--r--claude-templates/.ai/scripts/tests/spec-sort.bats453
-rw-r--r--claude-templates/.ai/scripts/tests/task-review-staleness.bats53
-rw-r--r--claude-templates/.ai/scripts/tests/test-lint-org.el437
-rw-r--r--claude-templates/.ai/scripts/tests/test-todo-cleanup.el610
-rw-r--r--claude-templates/.ai/scripts/tests/test-wrap-org-table.el42
-rw-r--r--claude-templates/.ai/scripts/tests/test_apkg_to_orgdrill.py301
-rw-r--r--claude-templates/.ai/scripts/tests/test_cj_remove_block.py167
-rw-r--r--claude-templates/.ai/scripts/tests/test_cross_agent_discover.py204
-rw-r--r--claude-templates/.ai/scripts/tests/test_cross_agent_halt.py204
-rw-r--r--claude-templates/.ai/scripts/tests/test_cross_agent_recv.py176
-rw-r--r--claude-templates/.ai/scripts/tests/test_cross_agent_send.py210
-rw-r--r--claude-templates/.ai/scripts/tests/test_cross_agent_status.py165
-rw-r--r--claude-templates/.ai/scripts/tests/test_cross_agent_watch.py155
-rw-r--r--claude-templates/.ai/scripts/tests/test_flashcard_stats.py25
-rw-r--r--claude-templates/.ai/scripts/tests/test_flashcard_to_anki.py92
-rw-r--r--claude-templates/.ai/scripts/tests/test_inbox_send.py235
-rw-r--r--claude-templates/.ai/scripts/tests/test_route_recommend.py152
-rw-r--r--claude-templates/.ai/scripts/tests/test_upcoming_birthdays.py168
-rw-r--r--claude-templates/.ai/scripts/todo-cleanup.el582
-rwxr-xr-xclaude-templates/.ai/scripts/upcoming_birthdays.py182
-rw-r--r--claude-templates/.ai/scripts/wrap-org-table.el58
-rw-r--r--claude-templates/.ai/workflows/INDEX.org45
-rw-r--r--claude-templates/.ai/workflows/add-calendar-event.org2
-rw-r--r--claude-templates/.ai/workflows/broadcast.org6
-rw-r--r--claude-templates/.ai/workflows/clean-todo.org21
-rw-r--r--claude-templates/.ai/workflows/code-quality.org90
-rw-r--r--claude-templates/.ai/workflows/create-workflow.org8
-rw-r--r--claude-templates/.ai/workflows/cross-agent-comms.org334
-rw-r--r--claude-templates/.ai/workflows/daily-prep.org7
-rw-r--r--claude-templates/.ai/workflows/delete-calendar-event.org2
-rw-r--r--claude-templates/.ai/workflows/edit-calendar-event.org2
-rw-r--r--claude-templates/.ai/workflows/email-assembly.org2
-rw-r--r--claude-templates/.ai/workflows/extract-email.org2
-rw-r--r--claude-templates/.ai/workflows/find-email.org2
-rw-r--r--claude-templates/.ai/workflows/first-session.org2
-rw-r--r--claude-templates/.ai/workflows/flashcard-review.org2
-rw-r--r--claude-templates/.ai/workflows/helper-mode.org101
-rw-r--r--claude-templates/.ai/workflows/inbox-zero.org97
-rw-r--r--claude-templates/.ai/workflows/inbox.org524
-rw-r--r--claude-templates/.ai/workflows/journal-entry.org2
-rw-r--r--claude-templates/.ai/workflows/meeting-prep.org2
-rw-r--r--claude-templates/.ai/workflows/meeting-prep.pre-wire.org2
-rw-r--r--claude-templates/.ai/workflows/monitor-inbox.org94
-rw-r--r--claude-templates/.ai/workflows/no-approvals.org12
-rw-r--r--claude-templates/.ai/workflows/open-tasks.org34
-rw-r--r--claude-templates/.ai/workflows/page-me.org53
-rw-r--r--claude-templates/.ai/workflows/process-inbox.org215
-rw-r--r--claude-templates/.ai/workflows/process-meeting-transcript.org26
-rw-r--r--claude-templates/.ai/workflows/read-calendar-events.org2
-rw-r--r--claude-templates/.ai/workflows/readability-audit.org242
-rw-r--r--claude-templates/.ai/workflows/rename-artifact.org2
-rw-r--r--claude-templates/.ai/workflows/send-email.org2
-rw-r--r--claude-templates/.ai/workflows/sentry.org227
-rw-r--r--claude-templates/.ai/workflows/session-harvest.org2
-rw-r--r--claude-templates/.ai/workflows/spec-create.org22
-rw-r--r--claude-templates/.ai/workflows/spec-response.org67
-rw-r--r--claude-templates/.ai/workflows/spec-review.org113
-rw-r--r--claude-templates/.ai/workflows/startup.org115
-rw-r--r--claude-templates/.ai/workflows/status-check.org2
-rw-r--r--claude-templates/.ai/workflows/summarize-emails.org2
-rw-r--r--claude-templates/.ai/workflows/suspend.org143
-rw-r--r--claude-templates/.ai/workflows/sync-email.org2
-rw-r--r--claude-templates/.ai/workflows/task-audit.org40
-rw-r--r--claude-templates/.ai/workflows/task-review.org12
-rw-r--r--claude-templates/.ai/workflows/triage-intake.cmail.org2
-rw-r--r--claude-templates/.ai/workflows/triage-intake.github-prs.org2
-rw-r--r--claude-templates/.ai/workflows/triage-intake.org237
-rw-r--r--claude-templates/.ai/workflows/triage-intake.personal-calendar.org2
-rw-r--r--claude-templates/.ai/workflows/triage-intake.personal-gmail.org23
-rw-r--r--claude-templates/.ai/workflows/triage-intake.telegram.org143
-rw-r--r--claude-templates/.ai/workflows/work-the-backlog.org266
-rw-r--r--claude-templates/.ai/workflows/wrap-it-up.org217
-rw-r--r--claude-templates/AGENTS.md19
-rwxr-xr-xclaude-templates/bin/agent-page12
-rwxr-xr-xclaude-templates/bin/agent-text57
-rwxr-xr-xclaude-templates/bin/ai402
-rwxr-xr-xclaude-templates/bin/git-worktree-gate185
-rwxr-xr-xclaude-templates/bin/install-ai23
-rw-r--r--docs/design/2026-05-28-generic-agent-runtime-spec.org4
-rw-r--r--docs/design/2026-05-28-pattern-catalog-prompt-labels-and-defaults.org2
-rw-r--r--docs/design/2026-06-02-flush-promotion.org2
-rw-r--r--docs/design/2026-06-02-pattern-catalog-spec.org2
-rw-r--r--docs/design/2026-06-15-auto-triage-intake-spec.org47
-rw-r--r--docs/design/2026-06-15-fix-speedrun-workflow-proposal.org21
-rw-r--r--docs/design/2026-06-15-spec-storage-lifecycle-proposal.org44
-rw-r--r--docs/design/2026-06-16-inbox-zero-phase-e-proposal.diff65
-rw-r--r--docs/design/2026-06-16-inbox-zero-phase-e-proposal.org (renamed from .ai/workflows/inbox-zero.org)44
-rw-r--r--docs/design/2026-06-16-inbox-zero-phase-e-sender-note.org27
-rw-r--r--docs/design/2026-06-17-flashcard-multitag-note.md28
-rwxr-xr-xdocs/design/2026-06-17-flashcard-multitag-stats.py332
-rwxr-xr-xdocs/design/2026-06-17-flashcard-multitag-to-anki.py294
-rw-r--r--docs/design/2026-06-17-ntfy-agent-comms-proposal.org89
-rw-r--r--docs/design/2026-06-18-triage-intake-phone-push-note.org11
-rw-r--r--docs/design/2026-06-18-triage-intake-phone-push-workflow.org427
-rw-r--r--docs/design/2026-06-21-anki-titlefix-proposal.org57
-rw-r--r--docs/design/2026-06-21-apkg-to-orgdrill-buildreq.org68
-rw-r--r--docs/design/2026-06-21-flashcard-stats-refutation-proposal.org57
-rw-r--r--docs/design/2026-06-21-host-identity-guard-proposal.org54
-rw-r--r--docs/design/2026-06-22-inbox-zero-capture-hardening.org39
-rw-r--r--docs/design/2026-06-23-install-lang-claude-md-gap.org31
-rw-r--r--docs/design/2026-06-23-wrap-teardown-shutdown-proposal.org124
-rw-r--r--docs/design/2026-06-27-bug-priority-matrix-cover-note.org5
-rw-r--r--docs/design/2026-06-27-bug-priority-matrix-proposal.md70
-rw-r--r--docs/design/2026-06-29-green-baseline-proposal.org72
-rw-r--r--docs/design/2026-06-29-lint-org-structural-checkers-proposal.org55
-rw-r--r--docs/design/2026-06-29-todo-cleanup-aging-proposal.org64
-rw-r--r--docs/design/2026-06-30-daily-drivers-tailscale-correction.org9
-rw-r--r--docs/design/2026-07-02-auto-flush-mechanism-note.org20
-rw-r--r--docs/design/2026-07-13-runtime-portability-inventories.org84
-rw-r--r--docs/design/2026-07-14-sentry-workflow-proposal.org150
-rw-r--r--docs/design/2026-07-15-subproject-pattern-proposal.org22
-rw-r--r--docs/design/2026-07-15-subprojects-convention-home-instance.org282
-rw-r--r--docs/design/2026-07-16-polyglot-bundle-collision.txt30
-rw-r--r--docs/design/2026-07-17-dated-log-planning-line-strip-proposal.md15
-rw-r--r--docs/design/2026-07-17-todo-cleanup-dated-seal-proposal.md38
-rw-r--r--docs/design/2026-07-18-colloquialisms-and-the-list-proposal.md22
-rw-r--r--docs/design/2026-07-20-signal-pager-runbook.org198
-rw-r--r--docs/design/task-review.org2
-rw-r--r--docs/design/wrapup-routing-spec.org181
-rw-r--r--docs/specs/2026-06-16-autonomous-batch-execution-spec.org393
-rw-r--r--docs/specs/2026-06-16-encourage-kb-contribution-spec.org206
-rw-r--r--docs/specs/2026-07-01-docs-lifecycle-spec.org361
-rw-r--r--docs/specs/2026-07-14-sentry-workflow-spec.org278
-rw-r--r--docs/specs/2026-07-20-silent-until-signal-monitors-spec.org126
-rw-r--r--docs/specs/2026-07-20-triage-source-activation-spec.org127
-rw-r--r--docs/specs/agent-knowledge-base-spec.org (renamed from docs/agent-knowledge-base-spec.org)16
-rw-r--r--docs/specs/inbox-workflow-consolidation-spec.org199
-rw-r--r--docs/specs/wrapup-routing-spec.org226
-rw-r--r--flush/SKILL.md32
-rw-r--r--hooks/README.md3
-rw-r--r--hooks/_common.py18
-rwxr-xr-xhooks/ai-wrap-teardown.sh110
-rwxr-xr-xhooks/git-commit-confirm.py57
-rwxr-xr-xhooks/inbox-boundary-check.sh52
-rwxr-xr-xhooks/rulesets-write-boundary.py94
-rwxr-xr-xhooks/session-start-disarm.sh42
-rw-r--r--hooks/settings-snippet.json14
-rw-r--r--hooks/tests/test_git_commit_confirm.py58
-rw-r--r--hooks/tests/test_rulesets_write_boundary.py110
-rw-r--r--inbox/PROCESSED-2026-06-11-1703-from-home-consolidation-handoff-rulesets.org46
-rw-r--r--inbox/PROCESSED-2026-06-11-1703-from-home-project-consolidation-spec.org350
-rw-r--r--inbox/PROCESSED-2026-06-11-1705-from-home-addendum-to-today-s-consolidation.org5
-rw-r--r--inbox/PROCESSED-2026-06-11-1755-from-work-from-the-work-project-2026-06-11-craig.org7
-rw-r--r--inbox/PROCESSED-2026-06-11-1823-from-.emacs.d-memory-sweep-phase-1-5-complete-for.org5
-rw-r--r--inbox/PROCESSED-2026-06-11-1909-from-home-inbox-response-consolidation-and-todo.org25
-rw-r--r--inbox/PROCESSED-2026-06-11-1951-from-home-inbox-response-jr-estate-memory-sweep.org12
-rw-r--r--inbox/PROCESSED-2026-06-11-2154-from-home-inbox-response-finances-memory-sweep.org8
-rw-r--r--inbox/PROCESSED-2026-06-11-2308-from-home-lint-org-el-false-positive-mu4e-msgid.org5
-rw-r--r--inbox/PROCESSED-2026-06-12-0101-from-.emacs.d-page-signal-is-broken-the-dedicated.org5
-rw-r--r--inbox/PROCESSED-2026-06-12-0207-from-home-memory-sweep-reply-for-the-2026-06-10.org5
-rw-r--r--inbox/lint-followups.org18
-rw-r--r--languages/bash/CLAUDE.md71
-rwxr-xr-xlanguages/bash/claude/hooks/validate-bash.sh66
-rw-r--r--languages/bash/claude/rules/bash-testing.md71
-rw-r--r--languages/bash/claude/rules/bash.md83
-rw-r--r--languages/bash/claude/settings.json68
-rwxr-xr-xlanguages/bash/githooks/pre-commit76
-rw-r--r--languages/bash/gitignore-add.txt4
-rw-r--r--languages/bash/tests/validate-bash.bats96
-rw-r--r--languages/default-CLAUDE.md64
-rwxr-xr-xlanguages/elisp/claude/hooks/validate-el.sh7
-rw-r--r--languages/elisp/claude/rules/elisp-testing.md2
-rw-r--r--languages/elisp/claude/scripts/coverage-summary.el21
-rwxr-xr-xlanguages/elisp/githooks/pre-commit41
-rw-r--r--languages/elisp/tests/test-coverage-summary.el45
-rw-r--r--languages/elisp/tests/test-pre-commit-hook.bats126
-rw-r--r--languages/elisp/tests/test-validate-el-hook.bats100
-rwxr-xr-xlanguages/go/githooks/pre-commit42
-rw-r--r--languages/python/CLAUDE.md80
-rwxr-xr-xlanguages/python/claude/hooks/validate-python.sh95
-rw-r--r--languages/python/claude/settings.json79
-rwxr-xr-xlanguages/python/githooks/pre-commit95
-rw-r--r--languages/python/tests/pre-commit.bats138
-rw-r--r--languages/python/tests/validate-python.bats117
-rw-r--r--languages/typescript/CLAUDE.md82
-rwxr-xr-xlanguages/typescript/claude/hooks/validate-typescript.sh94
-rw-r--r--languages/typescript/claude/settings.json80
-rwxr-xr-xlanguages/typescript/githooks/pre-commit107
-rw-r--r--languages/typescript/tests/pre-commit.bats150
-rw-r--r--languages/typescript/tests/validate-typescript.bats125
-rw-r--r--publish/SKILL.md471
-rw-r--r--publish/references/pull-requests.md105
-rw-r--r--review-code/SKILL.md53
-rwxr-xr-xscripts/audit.sh4
-rwxr-xr-xscripts/install-ai.sh25
-rwxr-xr-xscripts/install-lang.sh129
-rwxr-xr-xscripts/lint.sh47
-rwxr-xr-x[-rw-r--r--]scripts/remove.sh0
-rwxr-xr-xscripts/roam-sync.sh9
-rwxr-xr-xscripts/signal-receive.sh51
-rwxr-xr-xscripts/sweep-gitignore-tooling.sh91
-rwxr-xr-xscripts/sync-language-bundle.sh23
-rw-r--r--scripts/systemd/signal-receive.service10
-rw-r--r--scripts/systemd/signal-receive.timer10
-rw-r--r--scripts/tests/agent-text.bats78
-rw-r--r--scripts/tests/ai-launcher-characterization.bats365
-rw-r--r--scripts/tests/ai-launcher-runtime.bats97
-rw-r--r--scripts/tests/ai-wrap-teardown-hook.bats211
-rw-r--r--scripts/tests/audit.bats36
-rw-r--r--scripts/tests/before-close-queue.bats44
-rwxr-xr-xscripts/tests/git-worktree-gate.bats146
-rw-r--r--scripts/tests/inbox-boundary-check-hook.bats83
-rw-r--r--scripts/tests/install-agents-entry.bats51
-rw-r--r--scripts/tests/install-ai.bats87
-rw-r--r--scripts/tests/install-hooks-link.bats15
-rw-r--r--scripts/tests/install-lang-collision.bats143
-rw-r--r--scripts/tests/install-lang-completeness.bats93
-rw-r--r--scripts/tests/install-lang.bats99
-rw-r--r--scripts/tests/lint-coverage.bats54
-rw-r--r--scripts/tests/pre-commit-secret-scan.bats197
-rw-r--r--scripts/tests/rename-ai-artifact.bats12
-rw-r--r--scripts/tests/signal-receive.bats66
-rw-r--r--scripts/tests/sweep-gitignore-tooling.bats122
-rw-r--r--scripts/tests/sync-language-bundle.bats75
-rw-r--r--testing-standards/SKILL.md391
-rw-r--r--todo.org2881
-rw-r--r--voice/SKILL.md40
-rw-r--r--voice/references/voice-profile.org96
-rw-r--r--working/context-engineering-rightsizing/metrics.org269
-rw-r--r--working/context-engineering-rightsizing/proposals.org350
-rw-r--r--working/context-engineering-rightsizing/rollout.org371
-rw-r--r--working/lint-org-example-block/report-from-home.org21
-rw-r--r--working/question-capture-pattern/proposal-from-archsetup.org15
-rw-r--r--working/triage-account-guard/companion-note-from-home.org5
-rw-r--r--working/triage-account-guard/proposed.diff11
-rw-r--r--working/triage-account-guard/triage-intake.personal-gmail.org.proposed74
-rw-r--r--working/voice-term-density/SKILL.md.proposed511
-rw-r--r--working/voice-term-density/profile.diff36
-rw-r--r--working/voice-term-density/proposal-from-work.org42
-rw-r--r--working/voice-term-density/skill.diff58
-rw-r--r--working/voice-term-density/voice-profile.org.proposed1643
-rw-r--r--working/working-dir-orphan-check/proposal-from-work.org12
-rw-r--r--working/working-dir-orphan-check/proposed.diff24
-rw-r--r--working/working-dir-orphan-check/wrap-it-up.org.proposed651
442 files changed, 43000 insertions, 12905 deletions
diff --git a/.ai/metrics/work-the-backlog.jsonl b/.ai/metrics/work-the-backlog.jsonl
new file mode 100644
index 0000000..88c6764
--- /dev/null
+++ b/.ai/metrics/work-the-backlog.jsonl
@@ -0,0 +1,15 @@
+{"ts":"2026-07-02T05:14:42-04:00","run_id":"c726f526-2e35-4513-b25c-18ef61061333","project":"rulesets","caller":"speedrun","task":"id-link-conversion-pass","outcome":"implemented-committed","defer_reason":"","upfront_decision":false,"wall_clock_s":284,"commit_sha":"78bbaae","review_findings":0}
+{"ts":"2026-07-02T05:19:03-04:00","run_id":"c726f526-2e35-4513-b25c-18ef61061333","project":"rulesets","caller":"speedrun","task":"host-identity-guard-rule-plus-startup-lint","outcome":"implemented-committed","defer_reason":"","upfront_decision":true,"wall_clock_s":261,"commit_sha":"b6a977c","review_findings":0}
+{"ts":"2026-07-02T05:22:11-04:00","run_id":"c726f526-2e35-4513-b25c-18ef61061333","project":"rulesets","caller":"speedrun","task":"template-sync-gitignored-only-changes","outcome":"implemented-committed","defer_reason":"","upfront_decision":true,"wall_clock_s":188,"commit_sha":"ed75d3c","review_findings":0}
+{"ts":"2026-07-02T05:58:16-04:00","run_id":"a48f2977-4493-48a3-9238-9b2f5ff5383b","project":"rulesets","caller":"loop","task":"inbox-send-filename-collision-fix","outcome":"implemented-committed","defer_reason":"","upfront_decision":false,"wall_clock_s":300,"commit_sha":"8099377","review_findings":0}
+{"ts":"2026-07-02T05:58:16-04:00","run_id":"a48f2977-4493-48a3-9238-9b2f5ff5383b","project":"rulesets","caller":"loop","task":"page-me-notify-info-level","outcome":"implemented-committed","defer_reason":"","upfront_decision":false,"wall_clock_s":120,"commit_sha":"a6b534f","review_findings":0}
+{"ts":"2026-07-23T23:55:16-05:00","run_id":"711400a5-c6bc-4684-b928-6b818d0d33f2","project":"rulesets","caller":"speedrun","task":"inbox-send phantom empty handoff","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"0f91a8e","review_findings":1}
+{"ts":"2026-07-23T23:57:59-05:00","run_id":"711400a5-c6bc-4684-b928-6b818d0d33f2","project":"rulesets","caller":"speedrun","task":"inbox-send two smaller defects","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"a053e9d","review_findings":0}
+{"ts":"2026-07-24T00:02:14-05:00","run_id":"711400a5-c6bc-4684-b928-6b818d0d33f2","project":"rulesets","caller":"speedrun","task":"lint-org todo-format checkers fire on specs","outcome":"implemented-committed","upfront_decision":true,"commit_sha":"c38bab9","review_findings":0}
+{"ts":"2026-07-24T00:02:49-05:00","run_id":"711400a5-c6bc-4684-b928-6b818d0d33f2","project":"rulesets","caller":"speedrun","task":"notes.org template four lint flags","outcome":"already-satisfied","defer_reason":"already-satisfied","upfront_decision":false,"commit_sha":"","review_findings":0}
+{"ts":"2026-07-24T01:41:14-05:00","run_id":"sentry-fire2-1784875274","project":"rulesets","caller":"loop","task":"cj-remove-block over-deletion + unsafe write","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"17f5d48","review_findings":0}
+{"ts":"2026-07-24T01:43:52-05:00","run_id":"sentry-fire2","project":"rulesets","caller":"loop","task":"route_recommend duplicate-name tier downgrade","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"1b0f284","review_findings":0}
+{"ts":"2026-07-24T02:38:48-05:00","run_id":"sentry-fire3","project":"rulesets","caller":"loop","task":"audit.bats flaky teardown","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"7f45d4b","review_findings":0}
+{"ts":"2026-07-24T03:39:16-05:00","run_id":"sentry-fire4","project":"rulesets","caller":"loop","task":"todo-cleanup missing backup","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"0686784","review_findings":0}
+{"ts":"2026-07-24T04:36:09-05:00","run_id":"sentry-fire5","project":"rulesets","caller":"loop","task":"attachment filename sanitization","outcome":"deferred-verify","defer_reason":"needs-deliberation","upfront_decision":false,"commit_sha":"","review_findings":0}
+{"ts":"2026-07-24T07:38:47-05:00","run_id":"sentry-fire8","project":"rulesets","caller":"loop","task":"bin/ lint coverage gap","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"f91feef","review_findings":1}
diff --git a/.ai/notes.org b/.ai/notes.org
index 4790af9..828fde3 100644
--- a/.ai/notes.org
+++ b/.ai/notes.org
@@ -1,5 +1,5 @@
#+TITLE: Claude Code Notes - Rulesets
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-06
* About This File
@@ -32,7 +32,7 @@ See [[file:../README.org][README.org]] for the full layout, install modes, and l
- =claude-rules/= — generic rules (=commits.md=, =testing.md=, =verification.md=, =subagents.md=) symlinked into =~/.claude/rules/= and applied to every Claude Code session on the machine.
- Top-level skill directories (=add-tests/=, =debug/=, =five-whys/=, =frontend-design/=, =pairwise-tests/=, =playwright-js/=, =playwright-py/=, =root-cause-trace/=, =voice/=) — each a Claude Code skill, symlinked into =~/.claude/skills/= by =make install=.
-- =languages/= — per-language bundles (rules + hooks + settings) copied into target projects via =make install-lang LANG=<name> PROJECT=<path>=. Both =LANG= and =PROJECT= are optional — fzf picks them interactively when omitted. Bundles currently shipping: =elisp=, =python=.
+- =languages/= — per-language bundles (rules + hooks + settings) copied into target projects via =make install-lang LANG=<name> PROJECT=<path>=. Both =LANG= and =PROJECT= are optional — fzf picks them interactively when omitted. Bundles currently shipping: =bash=, =elisp=, =go=, =python=, =typescript=.
- =.claude/= — repo-local Claude Code config: =settings.json= and =commands/=.
- =hooks/=, =scripts/= — install helpers and PostToolUse validators that ride along with bundles.
- =Makefile= — install / uninstall / list entry points.
@@ -61,7 +61,9 @@ This section tracks decisions that need Craig's input before work can proceed.
** Current Reminders
-(None currently — will be added as needed)
+- =[2026-07-27]= Finish the context-engineering rightsizing — Craig's explicit ask at wrap. Surface is 57,800 → 28,949 tokens; the remaining work needs *his decisions*, not execution: =verification.md= (C1 — its honesty core vs the Opus 5 over-verification warning), =interaction.md= (3,828 tok, largest remaining), the TDD rationalization table (cut or keep), and D3 the gate separation (which approval gates are preference vs guardrail). Task: "Finish context-engineering rightsizing" in todo.org. Docs in =working/context-engineering-rightsizing/= are one commit behind — reconcile them first.
+
+- =[2026-07-14]= Review the sentry spec (docs/specs/2026-07-14-sentry-workflow-spec.org) — Craig's explicit ask at wrap: strongly suggest he reviews it before ending the next session. All 12 review findings and 10 decisions are resolved and folded in; the spec is open in his Emacs; the READY flip and the [#B] build task both wait on his deep read.
** Instructions for This Section
@@ -76,9 +78,13 @@ Format:
* Workflow State
+:COMMIT_AUTONOMY: yes
+:LOOP_MAY_COMMIT: yes
+:SENTRY_MAY_IMPLEMENT: yes
+:LAST_SPEC_SORT: 2026-07-02
Markers maintained by workflows to record when they last ran. Read by other workflows that gate their behavior on freshness.
-:LAST_AUDIT: 2026-05-28
-:LAST_INBOX_PROCESS: 2026-06-12 (evening: spec-create decisions-as-TODO convention from .emacs.d, applied + companions reconciled)
+:LAST_AUDIT: 2026-07-20 (open set current — this session's shipped work (working/temp, triage-source-activation, silent-until-signal, suspend detach) closed as it went; sentry cluster consolidated (merged the /schedule tasks, added cross-host-coordination); nothing shipped-but-open per git reconcile. Live finding: the Polyglot + Subprojects scouting tasks are SCHEDULED 2026-07-20 and due.)
+:LAST_INBOX_PROCESS: 2026-07-25 (consolidated home + work Claude-to-Codex MCP registry proposals into one [#B] parked spec decision; memory auditor split from the registry work)
Format: one =:MARKER: YYYY-MM-DD= line per workflow. Workflows overwrite their own marker on completion.
diff --git a/.ai/protocols.org b/.ai/protocols.org
index 6b1d873..bf9f420 100644
--- a/.ai/protocols.org
+++ b/.ai/protocols.org
@@ -1,5 +1,5 @@
#+TITLE: Claude Code Protocols
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2025-11-05
* About This File
@@ -84,7 +84,7 @@ Do NOT estimate, guess, or rely on memory. Just run the command. It takes one se
Every session pulls rulesets first, then the local project repo. Rulesets carries the canonical behavioral rules and =.ai/= templates (the old =claude-templates= repo is folded in as a subtree at =rulesets/claude-templates/=); the project pull lands commits pushed from other machines or teammates since the last session.
-Resolve any dirty-tree or merge issue at each step before moving on. Both pulls run as =git pull --ff-only= (or =git merge --ff-only= against the fetched upstream), so anything non-trivial — non-fast-forward history, dirty working tree, diverged branches — aborts. Surface the state and stop. Never auto-stash, auto-merge, or auto-rebase; the user resolves the conflict before further work.
+Resolve any sync-blocking tree or merge issue at each step before moving on. The shared =git-worktree-gate sync-safe= policy permits untracked deliveries beneath =inbox/= so receiving a handoff never prevents another project from refreshing rulesets; every staged or tracked change, dirty submodule, Git operation in progress, or untracked path outside =inbox/= blocks. Both pulls run as =git pull --ff-only= (or =git merge --ff-only= against the fetched upstream), so non-fast-forward history and diverged branches also abort. Surface the state and stop. Never auto-stash, auto-merge, or auto-rebase; the user resolves the conflict before further work.
Mechanics live in =startup.org= Phase A.0. The rule lives here because it governs the very first action of every session: load the freshest behavioral rules and templates before anything else runs.
@@ -100,8 +100,14 @@ When two agents share one project at the same time, a single =session-context.or
- =AI_AGENT_ID= unset or empty (the normal one-agent-per-project case): =.ai/session-context.org=, exactly as before.
- =AI_AGENT_ID= set: =.ai/session-context.d/<id>.org= (id sanitized to filename-safe chars). Archived at wrap-up to =.ai/sessions/YYYY-MM-DD-HH-MM-<id>-<description>.org= so concurrent agents don't collide on the archive name either.
+The id must be unique per run, and the spawner makes it so by appending an epoch on the tail. The recommended shape is =host.project.runtime.<epoch>= (e.g. =velox.rulesets.claude.1718400000=); a fresh run of the same logical agent then resolves to a fresh anchor. A bare, reused id (just =codex=) makes the next run resolve to the previous run's leftover =codex.org= and either mistake it for a crashed session to recover or clobber it if both run at once. This happened on 2026-06-13: a =codex= run left =session-context.d/codex.org= behind, and the next session had to clean it up by hand.
+
+The epoch is baked into the id by the spawner, never minted inside =session-context-path=. That resolver is called many times per session (startup existence check, every log write, the wrap-up rename) and must return the same path on every call, so a self-generated =date +%s= would fragment the anchor across calls. The shell that runs each call doesn't carry env between calls either, so the agent can't export the epoch once at startup. The only stable source is the =AI_AGENT_ID= the spawner injects on every call.
+
Resolve the path with =.ai/scripts/session-context-path= rather than hardcoding =.ai/session-context.org=; it prints the right path for the current =AI_AGENT_ID=. Fall back to =.ai/session-context.org= if the script isn't present (older checkouts mid-sync). Everything below — the record/recovery purpose, the update triggers, the startup existence check, the wrap-up rename — operates on that resolved path. The prose says "session-context.org" as the default name; read it as "the resolved active path" when =AI_AGENT_ID= is set.
+A helper instance (a second agent running in this project while a primary session is live) follows a different contract: it skips the pulls and rsync, makes only scoped single-heading edits to shared files, leaves all git mutation to the primary, and wraps up by archiving its own context file without committing. The full rules — read/write tiers, data-integrity, light startup, helper wrap-up — live in [[file:workflows/helper-mode.org][workflows/helper-mode.org]]. A session is a helper only when something routes it there (the =ai --helper= launcher, startup's roster check, or an explicit "you are a helper" instruction); the routing itself ships behind the helper-instance feature gate and isn't live yet.
+
This file serves two purposes with one mechanism:
1. *Crash recovery* — if the session dies mid-work, the live file is all that's left. On 2026-01-22 a session crashed during a 20-minute design discussion and all context was lost because this file wasn't being updated.
2. *Session archive* — at wrap-up the file is renamed into =.ai/sessions/=, becoming the permanent record. No transcription to notes.org; the file IS the record.
@@ -181,6 +187,8 @@ Canonical rule: =~/code/rulesets/claude-rules/cross-project.md=.
Every in-progress task that produces files (drafts, source documents, diagrams, scripts, sub-deliverables) gets a dedicated subdirectory under =<project-root>/working/=, named after the task. All artifacts for that task live in that subdirectory until the task is marked done.
+=working/= is version-controlled from creation — it's the tracked home of in-progress work, never gitignored. Filing on completion *reorganizes* durable artifacts into permanent homes; it doesn't mark when they became durable (they were durable, and tracked, from the start). Genuinely disposable artifacts go in a gitignored =temp/= (or =/tmp=), never =working/=; the install tooling ignores =temp/= in both track and gitignore modes.
+
When the task ships, files are **renamed individually** (standard form: =YYYY-MM-DD-<task-slug>-<descriptor>.<ext>=) and **moved flat** into the appropriate permanent home (typically =assets/= or an area-specific =<area>/assets/=). The working subdirectory is then empty and gets deleted.
***Never rename the directory itself as a substitute for filing.*** The point is to keep =assets/= flat-searchable — a nested =assets/old-tech-deck-2026/slide.png= is harder to find than =assets/2026-05-18-tech-deck-vol2-slide-04-diagram.png=.
@@ -197,7 +205,9 @@ Check =inbox/= at every task boundary (after finishing a unit of work, before re
.ai/scripts/inbox-status -q
#+end_src
-Exit 1 means handoffs are pending — process them per =process-inbox.org=. For each accepted handoff, the act-vs-file rule: *act now* when it's clear, bounded, low-risk, in-scope, and cheaper than deferring — just do it, no asking; *file* otherwise — ask first, with filing as option 1 and "do it now" as option 2; *ask* if unsure. Exception: a proposal to change a shared asset (template workflow, rule, skill, synced script) or a substantive convention never silently acts now — it goes through process-inbox's Skeptical Review and its approval (or park) step. Always reply to a handoff's sender (confirm on accept, the why on reject). Full process, the reply discipline, and the opt-in background-monitor =/loop= recipe live in =monitor-inbox.org=.
+Exit 1 means handoffs are pending — process them per =inbox.org= process mode. For each accepted handoff, the act-vs-file rule: *act now* when it's clear, bounded, low-risk, in-scope, and cheaper than deferring — just do it, no asking; *file* otherwise — ask first, with filing as option 1 and "do it now" as option 2; *ask* if unsure. Exception: a proposal to change a shared asset (template workflow, rule, skill, synced script) or a substantive convention never silently acts now — it goes through the inbox engine's skeptical review and its approval (or park) step. Always reply to a handoff's sender (confirm on accept, the why on reject). Full process, the reply discipline, and the opt-in background-monitor =/loop= recipe live in =inbox.org= monitor mode.
+
+A machine-global =Stop= hook (=inbox-boundary-check.sh=) backs this rule so it isn't prose-only. When handoffs are pending it blocks the turn once and injects the count, so a task boundary can't pass with items unseen. It soft-nudges rather than hard-blocks: on the harness re-entry it steps aside, so a mid-task pause to ask "what's next" is never hijacked into inbox processing. The rule above still governs what to do with the items; the hook only makes sure you look.
** Recursive Reads — Honor =.aiignore=
@@ -236,13 +246,33 @@ Execute the wrap-up workflow (details in Session Protocols section below):
2. Git commit and push all changes
3. Valediction summary
+** "Suspend the session" / "Suspend" / "I need to go" / "Stick a pin in everything"
+
+Execute the suspend workflow ([[file:workflows/suspend.org][suspend.org]]): a capture-only mid-session pause for an abrupt departure. It appends a resume-weighted =SUSPENDED= entry to the Session Log, notes uncommitted work, and LEAVES =.ai/session-context.org= in place so the next startup resumes from it — no archive, no teardown, no valediction. The capture-only counterpart to "wrap it up" (which ends + archives + tears down) and to =/flush= (which prompts =/clear= and resumes the same session). "I need to go" is broad — if it reads as a conversational aside, confirm before suspending.
+
+* Colloquialisms and Expansions
+
+Shorthand phrases Craig uses that expand to a defined action the agent applies without asking. The set is extensible: a project may add its own entries, and new shared shorthands land here.
+
+** "the list": the Before-Close Queue
+
+"Put X on the list" or "add X to the list" appends X to the Before-Close Queue, a FIFO queue of tasks and actions to finish before the session closes. Work it oldest-first at wrap-up, before teardown (=wrap-it-up.org= Step 1 works it before finalizing the Summary), and surface anything unfinished in the valediction rather than dropping it.
+
+The queue lives in the session anchor (=.ai/session-context.org=) under a =* Before-Close Queue= heading. Create the heading on the first "put it on the list" if it's absent, then append one line per item. It's session-scoped: it resets when the anchor is archived at wrap. Anything that must outlive the session is a =todo.org= task instead, not a list item.
+
+** "tell <project> <message>": cross-project handoff
+
+"Tell <project> <message>" drops the message in that project's =inbox/= via =inbox-send= (=python3 .ai/scripts/inbox-send.py <project> --text "<message>"=), the sanctioned cross-project handoff. Never write another project's =todo.org= or =inbox/= directly. Resolve =<project>= the way =inbox-send= does (basename match, dots stripped); if it's ambiguous, ask which project rather than guessing.
+
* User Information
** Calendar Management
Three ways to access Craig's calendars: Google Calendar MCP (preferred, both personal + work accounts), gcalcli (fallback, personal only), Emacs org files (read-only viewer).
-For tool recipes, authentication details, and credentials, see [[file:references/calendar-reference.org][calendar-reference.org]].
+For tool recipes and account details, read the calendar workflows in =.ai/workflows/=: =add-calendar-event.org=, =edit-calendar-event.org=, =delete-calendar-event.org=, =read-calendar-events.org=. They carry the MCP tool names, both account ids, the gcalcli fallback, and the conflict-check discipline.
+
+Credentials are needed only for a re-auth Craig performs himself. The MCP bundle's =mcp/README.org= in the rulesets repo is the authority: =gcp-oauth.keys.json= is gitignored and regenerated at install from a base64 var in the bundle, never committed. Named in prose rather than linked, because that path isn't synced into consuming projects.
** GPG Keys
@@ -346,9 +376,19 @@ Craig runs a pure Wayland setup (Hyprland) and avoids XWayland/Xorg apps.
- Clipboard: Use =wl-copy= and =wl-paste= (NOT =xclip= or =xsel=)
- Window management: Use Hyprland commands (NOT =xkill=, =xdotool=, etc.)
- Prefer Wayland-native tools over X11 equivalents
-- Open URLs in browser: Use =google-chrome-stable "URL" &>/dev/null &=
- - The =&>/dev/null &= is required to detach the process and suppress output
- - Without it, the command may appear to hang or produce no result
+- Open URLs in browser: invoke Chrome directly — never =xdg-open=, which returned success in a home session on 2026-07-26 while no tab appeared.
+
+ Chrome is normally already running, and in that case it hands the URL to the live session and exits immediately (rc 0), printing =Opening in existing browser session.= on *stdout*. So run it in the foreground and read that line as the confirmation the tab actually opened:
+
+ #+begin_src bash
+ google-chrome-stable --new-tab "URL"
+ #+end_src
+
+ Don't redirect stdout away while checking for that line — verified 2026-07-27 on ratio: with =2>/dev/null= the message still appears (it isn't stderr), and with =>/dev/null= it vanishes.
+
+ Several URLs in one invocation open as separate tabs (=google-chrome-stable --new-tab "URL1" "URL2"=). Pass them as separate words or an array — the Bash tool runs zsh, which does not word-split an unquoted =$urls= variable, so a space-joined string arrives as one malformed argument (see the zsh note below).
+
+ *Cold start.* If Chrome is *not* already running, the command becomes the browser process and blocks. Detach that case with =&>/dev/null &=, accepting that the confirmation line is discarded — there is no session to confirm into. Don't apply the detach form unconditionally: it suppresses the very output the warm path is verified by.
*** Shell aliases (=ls= → =exa=)
Craig's shell aliases =ls= to =exa=, which prints nothing to non-TTY pipes (e.g. when capturing =ls= output in a Bash tool call). The result looks like the directory is empty when it isn't.
@@ -357,6 +397,12 @@ Craig's shell aliases =ls= to =exa=, which prints nothing to non-TTY pipes (e.g.
- Applies to =ls -la=, =ls -t=, glob expansions piped through =ls=, and any =ls= invocation whose output gets read programmatically.
- Symptom if forgotten: the Bash tool returns empty output and you mistakenly conclude the directory is empty.
+*** zsh does not word-split unquoted variables
+The Bash tool runs zsh, which (unlike bash) does not split an unquoted =$var= on whitespace. =chrome $urls= passes all the space-joined URLs as one malformed argument.
+
+- Loop over the values, use an array, or force the split with =${=var}=.
+- Symptom if forgotten: a command that "works in bash" gets one garbled argument and fails, often silently (from the takuzu session, 2026-07-11).
+
** Miscellaneous Information
- Craig currently lives in New Orleans, LA
- Craig's phone number: 510-316-9357
@@ -396,6 +442,30 @@ Full usage: =notify --help= or see =~/.local/bin/notify=
- =atq= - list all scheduled alarms
- =atrm [number]= - remove an alarm by its queue number
+** Reaching Craig — the notification vocabulary
+
+Two channels, two trigger words. "page me" is the desktop, "text me" is the phone, "text and page me" is both. Pick by where Craig is, and default to both when a run can't tell. Both work from any agent runtime (nothing here is Claude-specific). The words are what Craig says; a run deciding on its own maps the same way (away run texts, at-desk run pages, unsure does both).
+
+- *"page me" — at his laptop/desktop.* A desktop =notify ... --persist= that reaches him on the machine and stays up until dismissed.
+
+ #+begin_src bash
+ notify info "Title" "Message" --persist
+ #+end_src
+
+- *"text me" — away from his machine.* A Signal push to his phone via =agent-text=:
+
+ #+begin_src bash
+ agent-text "Message for Craig's phone"
+ #+end_src
+
+ =agent-text= (in =~/.local/bin= via the rulesets install) sends from the dedicated Signal identity (+15045173983) to Craig's Signal account UUID, firing a normal mobile push. The account is registered on velox (primary) and ratio (linked device), so either sends directly; a machine without it ssh-relays to velox. Verified end to end 2026-07-13 (velox) and 2026-07-20 (ratio). Never target Craig's phone *number* (it reads as unregistered in Signal's directory); the script targets the UUID.
+
+ Caveats: a relay from a non-linked machine needs velox up on the tailnet, and each device holding the account wants a periodic =receive= (the signal-receive timer handles that). The full runbook lives in rulesets =docs/design/=.
+
+- *"text and page me" — both.* Fire =agent-text= and =notify= together. The phone reaches him now, the desktop note waits for his return. This is the default when a run can't tell whether he's away.
+
+On velox, Claude sessions may also have the *signal-mcp* tool (=send_message_to_user=, same identity), fine to use there, but it exists only in velox's local MCP config, so =agent-text= is the portable habit. The tool was named =agent-page= before 2026-07-20; a deprecated =agent-page= shim still delegates to =agent-text=. Do *not* use the old =page-signal= shell script (removed 2026-06-12).
+
* Session Protocols
** CRITICAL: Git Commit Requirements
@@ -421,7 +491,7 @@ When creating commits:
- Keep messages clear and informative
3. **No Claude-tooling artifacts**: Commit messages describe project changes only — the meta-process of how work got shipped stays out of public git history.
- - **ABSOLUTELY NO** mentions of =notes.org=, =session-context.org=, =.ai/sessions/=, =todo.org=, "session wrap-up", or session timestamps (e.g., "Session YYYY-MM-DD HH:MM → ...")
+ - **ABSOLUTELY NO** mentions of =notes.org=, =session-context.org=, =.ai/= (including =.ai/sessions/=), =.claude/=, =CLAUDE.md=, =todo.org=, "session wrap-up", or session timestamps (e.g., "Session YYYY-MM-DD HH:MM → ..."), except when one of those files is itself the change — then name what changed by category, not the surrounding tooling layer
- Subject lines must NEVER start with =session:= as a conventional-commit type — use =docs:=, =refactor:=, =fix:=, =feat:=, =chore:=, etc. (real change categories)
- When a wrap-up commit bundles many changes from a session, describe what /shipped/ (e.g., =refactor: extract RAID logic + add bats testing infrastructure=), not that a session happened
- Same spirit as the no-Claude-attribution rule: the tooling stays invisible in =git log=
@@ -454,7 +524,7 @@ When Craig says this phrase:
- If exact match found: Read and guide through process
3. **Fuzzy match across both directories:** Ask for clarification
- - Example: User says "empty inbox" but we have "inbox-zero.org"
+ - Example: User says "empty inbox" but we have "inbox.org" (roam mode)
- Ask: "Did you mean the 'inbox zero' workflow, or create new 'empty inbox'?"
4. **No match at all:** Offer to create it
@@ -509,12 +579,13 @@ When monitoring a long-running process (rsync, large downloads, builds, VM tests
** "Wrap it up" / "That's a wrap" / "Let's call it a wrap"
-When Craig says any of these phrases (or variations), execute the wrap-up workflow: [[file:workflows/wrap-it-up.org][wrap-it-up.org]]. Four steps:
+When Craig says any of these phrases (or variations), execute the wrap-up workflow: [[file:workflows/wrap-it-up.org][wrap-it-up.org]]. Five load-bearing steps:
1. *Finalize the Summary* in =.ai/session-context.org= (populate the 5 subsections from the Session Log)
2. *Rename* =.ai/session-context.org= → =.ai/sessions/YYYY-MM-DD-HH-MM-description.org=
3. *Git commit + push* to all remotes (see Git Commit Requirements)
-4. *Valediction* — brief, warm, specific closing
+4. *Certify the clean tree* with =git-worktree-gate certify=. Any remaining staged, unstaged, untracked, submodule, or in-progress-operation state blocks wrap entirely; report each path and the exact decision needed. There is no dirty-file deferral.
+5. *Valediction* — brief, warm, specific closing, reachable only after certification
The absence of =.ai/session-context.org= after wrap-up is the signal that the session ended cleanly. If the file is still there at the next session start, the previous session was interrupted.
@@ -533,6 +604,8 @@ Claude needs to add information to =.ai/notes.org=. For large amounts of informa
**The gitignore set follows that same decision.** A project that gitignores =.ai/= (the code-project case) gitignores the whole personal-tooling set: =.ai/=, =.claude/=, =CLAUDE.md=, =AGENTS.md=. =.claude/= is rulesets-owned — copies of =claude-rules/*.md= plus the language bundle's rules, hooks, and settings — and re-synced from rulesets on every startup, so git isn't how it travels between machines; ignoring it also keeps those private rule copies out of the repo, which ignoring =CLAUDE.md= alone would miss. A track-mode project (personal/doc repos, or a team repo that shares config with teammates who don't run rulesets) tracks the set instead. =install-ai.sh= writes the full set at bootstrap in gitignore mode; =scripts/sweep-gitignore-tooling.sh= backfills it idempotently across existing gitignore-mode projects when the set grows.
+**Public reachability decides harder than project type.** Any repo whose remotes include a non-cjennings.net host gitignores the tooling set, whatever kind of project it is — the only exception is a team repo that deliberately shares the config, decided explicitly, never by default. And a private remote is not proof of privacy: a server-side =post-receive --mirror= hook republishes invisibly from the client (the 2026-06-30 =.emacs.d= exposure rode exactly that — a cjennings.net remote mirroring to public GitHub). The sweep recognizes both the anchored (=/.ai/=) and unanchored (=.ai/=) ignore styles — an anchored-style project used to be misread as track-mode and silently skipped — and warns when tracked tooling can reach a non-cjennings.net remote.
+
**Credential-leak concern: gate it on project type, not on the credential itself.** A tracked secret, token, or credentials doc is only a public-leak risk where the repo can reach a public remote — that is, *code projects pushed to public GitHub*, which is exactly why those gitignore =.ai/= and =.claude/=. For *personal / documentation projects* (the =~/projects/= set: elibrary, home, finances, health, philosophy, etc.), the git remote is a private single-user repo on =cjennings.net=, so tracked credentials inside =.ai/= files are fine — that's the design, the project history IS the project. Do NOT raise a leak warning or suggest gitignoring a secret for these. When the question "is this a leak / should we gitignore this secret?" comes up, decide it on *which kind of project and remote* this is, never on the mere presence of a credential in a tracked file.
**When to break out documents:**
diff --git a/.ai/references/calendar-reference.org b/.ai/references/calendar-reference.org
deleted file mode 100644
index b44c0f1..0000000
--- a/.ai/references/calendar-reference.org
+++ /dev/null
@@ -1,66 +0,0 @@
-#+TITLE: Calendar Reference
-#+AUTHOR: Craig Jennings & Claude
-
-Tool recipes, authentication, and credentials for Craig's calendar
-setup. Three access methods, in order of preference.
-
-* Google Calendar MCP Server (preferred for all calendar operations)
-
-Craig has the =@cocal/google-calendar-mcp= MCP server configured at user scope (=~/.claude.json=). It provides full read/write access to Google Calendar via MCP tools.
-
-Two accounts are authenticated:
-- *personal* — craigmartinjennings@gmail.com (primary: "Craig Google")
-- *work* — craig.jennings@deepsat.com (primary: "Craig Deepsat")
-
-MCP tools available:
-- =list-events=, =search-events=, =get-event= — read events
-- =create-event=, =create-events= — add events
-- =update-event= — modify events
-- =delete-event= — remove events
-- =list-calendars=, =list-colors= — calendar metadata
-- =get-freebusy= — check availability
-- =manage-accounts= — add/remove/list authenticated accounts
-- =respond-to-event= — accept/decline invitations
-- =get-current-time= — current time in any timezone
-
-Use =account_id: "personal"= or =account_id: "work"= to specify which account.
-
-Default calendar for adding events: "Craig Google" (personal account).
-
-Calendar workflows are available alongside this reference: add-calendar-event, edit-calendar-event, delete-calendar-event, read-calendar-events.
-
-If re-authentication is needed:
-- Use the =manage-accounts= MCP tool with =action: "add"= and the account nickname
-- OAuth credentials: =~/projects/homelab/assets/gcp-oauth.keys.json=
-- Google Cloud app is in production mode (tokens don't expire after 7 days)
-- See =~/projects/homelab/.ai/gcalcli-setup.org= for Google Cloud project details
-
-* gcalcli (fallback for personal account only)
-
-Craig has =gcalcli= installed via pipx, authenticated to his personal Google account only.
-
-#+begin_src bash
-gcalcli agenda # upcoming events
-gcalcli calw # weekly view
-gcalcli add --title "..." --when "..." --duration "60" # add event
-gcalcli search "..." # search events
-gcalcli delete "..." # delete event
-#+end_src
-
-Use =--calendar "Craig Google"= when adding events.
-
-gcalcli does NOT have access to the work (DeepSat) calendar. Use the MCP server for work calendar operations.
-
-If gcalcli needs re-authentication, credentials are stored in the homelab project: =~/projects/homelab/assets/gcalcli-client-secret.json.gpg= (GPG encrypted).
-
-* Emacs org files (read-only, for viewing schedules)
-
-Craig's calendars are at: =~/.emacs.d/data/*cal.org= (gcal.org, dcal.org, pcal.org)
-
-These files are **READ-ONLY** — NEVER add anything to them.
-
-Use this to:
-- Check meeting times and schedules
-- Verify when events occurred
-- See what's upcoming
-- Note: only updated periodically when Emacs is running — may be stale
diff --git a/.ai/scripts/agent-lock b/.ai/scripts/agent-lock
new file mode 100755
index 0000000..634412c
--- /dev/null
+++ b/.ai/scripts/agent-lock
@@ -0,0 +1,248 @@
+#!/usr/bin/env bash
+# agent-lock — a mkdir-atomic advisory lock for agent workflows.
+#
+# Why not flock: every Bash call an agent makes is its own short-lived shell,
+# so an flock taken in one /loop turn is gone by the next. This helper persists
+# the lock on disk between calls (an atomic mkdir is the acquire), and a crashed
+# holder's lock self-clears via age-based staleness reclaim instead of wedging
+# every later acquire.
+#
+# Serves both of sentry's locks (the single-runner lock and the roam-write
+# lock); callers pass a name, never a path — the helper owns the path scheme.
+#
+# Usage:
+# agent-lock acquire <name> [--ttl=SECONDS] [--wait[=SECONDS]]
+# Atomic acquire. exit 0 on win (fresh, or reclaimed from a stale holder);
+# exit 1 when a live lock already holds <name> (deferred — a note names the
+# holder on stderr). --wait polls up to SECONDS (default 30) before
+# deferring; without it, acquire is single-shot win-or-lose. --ttl records
+# the staleness horizon in the lock's metadata (default below).
+# agent-lock refresh <name>
+# Heartbeat: re-touch a held lock's mtime so it stays young. A runner
+# refreshes its own lock between passes, so a live run's lock is never older
+# than one pass and the TTL sizes to the longest single pass. exit 1 if the
+# lock is absent (nothing to refresh).
+# agent-lock release <name>
+# Remove the lock. Idempotent: exit 0 even if already free.
+# agent-lock status <name>
+# Print "free" | "held ..." | "stale ..." plus metadata. exit 0 (a query
+# never fails on lock state).
+# agent-lock path <name>
+# Print the resolved lock-directory path without creating it.
+#
+# Lock home (the helper owns this; callers pass names only):
+# $AGENT_LOCK_DIR/<name>/ when AGENT_LOCK_DIR is set (tests / advanced)
+# $XDG_RUNTIME_DIR/agent-locks/<name>/ the tmpfs runtime dir /run/user/<uid>
+# (host-local, out of every repo,
+# cleared on reboot). XDG_RUNTIME_DIR is
+# the standard handle for it and is set
+# in sentry's interactive launch.
+# ${XDG_CACHE_HOME:-~/.cache}/agent-locks/<name>/ fallback where no runtime
+# dir exists (XDG_RUNTIME_DIR unset or
+# unwritable — a headless/container box)
+#
+# tmpfs residence is deliberate: a lock under ~/org/roam would ride roam-sync's
+# `git add -A` to the other machine as a phantom hold. Host-locality is by
+# construction, and reboot clears any lock a crash left behind for free.
+#
+# Staleness is age-based on the metadata file's mtime versus the lock's own
+# recorded TTL. Heartbeat re-touches the mtime; a reclaim is always surfaced,
+# never silent.
+
+set -euo pipefail
+
+DEFAULT_TTL=600 # 10 min: sized to the longest single sentry pass, since a
+ # live runner heartbeats between passes and stays young.
+DEFAULT_WAIT=30 # bounded-wait budget for --wait (capture-guard's shape).
+WAIT_INTERVAL=3 # poll cadence while waiting on a busy lock.
+
+usage() {
+ echo "usage: agent-lock {acquire|refresh|release|status|path} <name> [--ttl=N] [--wait[=N]]" >&2
+ exit 2
+}
+
+# Resolve the base directory that holds all lock dirs, per the home scheme above.
+lock_base() {
+ if [ -n "${AGENT_LOCK_DIR:-}" ]; then
+ printf '%s\n' "$AGENT_LOCK_DIR"
+ elif [ -n "${XDG_RUNTIME_DIR:-}" ] && [ -d "$XDG_RUNTIME_DIR" ] && [ -w "$XDG_RUNTIME_DIR" ]; then
+ printf '%s/agent-locks\n' "$XDG_RUNTIME_DIR"
+ else
+ printf '%s/agent-locks\n' "${XDG_CACHE_HOME:-$HOME/.cache}"
+ fi
+}
+
+# Validate a lock name: non-empty, no path separators (so a name can never
+# escape the base dir).
+valid_name() {
+ case "$1" in
+ ''|*/*|.|..) return 1 ;;
+ *) return 0 ;;
+ esac
+}
+
+lock_dir() { printf '%s/%s\n' "$(lock_base)" "$1"; }
+meta_path() { printf '%s/meta\n' "$(lock_dir "$1")"; }
+
+# Read a key from a lock's metadata file; empty if absent.
+meta_get() {
+ local key="$1" file="$2"
+ [ -f "$file" ] || return 0
+ sed -n "s/^${key}=//p" "$file" | head -n1
+}
+
+# Age of a lock in whole seconds, from the metadata mtime.
+lock_age() {
+ local file="$1" mtime now
+ mtime=$(stat -c %Y "$file" 2>/dev/null) || return 1
+ now=$(date +%s)
+ printf '%s\n' "$((now - mtime))"
+}
+
+# True when a lock dir exists but its age exceeds its recorded TTL.
+is_stale() {
+ local name="$1" file age ttl
+ file="$(meta_path "$name")"
+ [ -f "$file" ] || return 1
+ age="$(lock_age "$file")" || return 1
+ ttl="$(meta_get ttl "$file")"
+ [ -n "$ttl" ] || ttl="$DEFAULT_TTL"
+ [ "$age" -gt "$ttl" ]
+}
+
+# Write the metadata file for a freshly-taken lock.
+write_meta() {
+ local name="$1" ttl="$2" file
+ file="$(meta_path "$name")"
+ {
+ printf 'pid=%s\n' "$$"
+ printf 'host=%s\n' "$(uname -n)"
+ printf 'acquired=%s\n' "$(date +%Y-%m-%dT%H:%M:%S%z)"
+ printf 'ttl=%s\n' "$ttl"
+ } > "$file"
+}
+
+# One-line holder description for surfaced notes.
+holder_desc() {
+ local file="$1"
+ printf "pid=%s host=%s age=%ss ttl=%ss" \
+ "$(meta_get pid "$file")" "$(meta_get host "$file")" \
+ "$(lock_age "$file" 2>/dev/null || echo '?')" "$(meta_get ttl "$file")"
+}
+
+# Attempt a single atomic acquire. exit 0 win, 1 busy (live holder).
+try_acquire() {
+ local name="$1" ttl="$2" dir file
+ dir="$(lock_dir "$name")"
+ file="$(meta_path "$name")"
+ mkdir -p "$(lock_base)"
+
+ if mkdir "$dir" 2>/dev/null; then
+ write_meta "$name" "$ttl"
+ return 0
+ fi
+
+ # Directory exists. Reclaim it if the holder is stale; otherwise it's busy.
+ if is_stale "$name"; then
+ # Claim the stale dir atomically before removing it. `mv` of a directory is
+ # atomic, so when two acquirers both see the lock stale, only one's rename
+ # of $dir succeeds — the other's fails because $dir is already gone, and it
+ # falls through to busy. Never `rm -rf $dir` directly: a plain remove lets
+ # the loser delete the winner's freshly-created lock and double-acquire.
+ local claimed="$dir.stale.$$"
+ if mv "$dir" "$claimed" 2>/dev/null; then
+ echo "agent-lock: reclaimed stale lock '$name' ($(holder_desc "$claimed/meta"))" >&2
+ rm -rf "$claimed"
+ # mkdir stays the sole grant: a concurrent fresh acquirer may win here,
+ # in which case our mkdir fails and we correctly defer to it.
+ if mkdir "$dir" 2>/dev/null; then
+ write_meta "$name" "$ttl"
+ return 0
+ fi
+ fi
+ fi
+ return 1
+}
+
+cmd_acquire() {
+ local name="$1"; shift
+ local ttl="$DEFAULT_TTL" wait_total=0
+ while [ $# -gt 0 ]; do
+ case "$1" in
+ --ttl=*) ttl="${1#--ttl=}" ;;
+ --ttl) shift; ttl="${1:-}" ;;
+ --wait) wait_total="$DEFAULT_WAIT" ;;
+ --wait=*) wait_total="${1#--wait=}" ;;
+ *) usage ;;
+ esac
+ shift
+ done
+ case "$ttl" in ''|*[!0-9]*) usage ;; esac
+ case "$wait_total" in *[!0-9]*) usage ;; esac
+
+ local elapsed=0
+ while :; do
+ if try_acquire "$name" "$ttl"; then
+ exit 0
+ fi
+ if [ "$elapsed" -ge "$wait_total" ]; then
+ echo "agent-lock: '$name' busy ($(holder_desc "$(meta_path "$name")")); deferring" >&2
+ exit 1
+ fi
+ local remaining=$((wait_total - elapsed)) step
+ step=$(( remaining < WAIT_INTERVAL ? remaining : WAIT_INTERVAL ))
+ sleep "$step"
+ elapsed=$((elapsed + step))
+ done
+}
+
+cmd_refresh() {
+ local name="$1" file
+ file="$(meta_path "$name")"
+ [ -f "$file" ] || exit 1
+ # Re-stamp acquired and bump mtime so the age clock restarts.
+ local ttl; ttl="$(meta_get ttl "$file")"; [ -n "$ttl" ] || ttl="$DEFAULT_TTL"
+ write_meta "$name" "$ttl"
+ exit 0
+}
+
+cmd_release() {
+ local name="$1" dir
+ dir="$(lock_dir "$name")"
+ rm -rf "$dir"
+ exit 0
+}
+
+cmd_status() {
+ local name="$1" dir file
+ dir="$(lock_dir "$name")"
+ file="$(meta_path "$name")"
+ if [ ! -d "$dir" ]; then
+ echo "free $name"
+ exit 0
+ fi
+ local state="held"
+ is_stale "$name" && state="stale"
+ echo "$state $name pid=$(meta_get pid "$file") host=$(meta_get host "$file") acquired=$(meta_get acquired "$file") ttl=$(meta_get ttl "$file") age=$(lock_age "$file" 2>/dev/null || echo '?')s"
+ exit 0
+}
+
+cmd_path() {
+ lock_dir "$1"
+ exit 0
+}
+
+[ $# -ge 1 ] || usage
+subcmd="$1"; shift
+[ $# -ge 1 ] || usage
+name="$1"; shift
+valid_name "$name" || usage
+
+case "$subcmd" in
+ acquire) cmd_acquire "$name" "$@" ;;
+ refresh) cmd_refresh "$name" ;;
+ release) cmd_release "$name" ;;
+ status) cmd_status "$name" ;;
+ path) cmd_path "$name" ;;
+ *) usage ;;
+esac
diff --git a/.ai/scripts/agent-roster b/.ai/scripts/agent-roster
new file mode 100755
index 0000000..f32b744
--- /dev/null
+++ b/.ai/scripts/agent-roster
@@ -0,0 +1,84 @@
+#!/usr/bin/env bash
+# agent-roster — list other live Claude agents working in this project.
+#
+# The single source of "who else is live in this project." Both launchers
+# (ai --helper) and the in-session startup check call this rather than
+# reimplementing the scan, so concurrent-agent detection has one definition.
+#
+# Scan (stateless): enumerate running Claude processes (pgrep -x claude), read
+# each one's working directory from /proc/<pid>/cwd, keep those whose cwd is
+# the project root or inside it, and drop the scanner's own process ancestry
+# (walk parent pids from /proc/self up). What remains is the set of *other*
+# live agents in this project.
+#
+# Usage: agent-roster [project-root] (default: $PWD)
+# Output: one "pid<TAB>cwd" line per other agent
+# Exit: 0 = alone (no other agents)
+# 1 = one or more other agents (and printed)
+# 2 = roster unavailable (no /proc; non-Linux or absent)
+#
+# Known limits, accepted for v1: a session not running as a local process on
+# this machine (a cloud session against the same checkout) is invisible, and
+# the match is on process cwd, so an agent started from outside the project
+# tree isn't seen. Both are edge shapes the operator created deliberately.
+#
+# The boundary (pgrep, /proc, self pid) is injectable so the filtering logic
+# is testable without spawning real agents: ROSTER_PGREP, ROSTER_PROC,
+# ROSTER_SELF_PID. Production defaults need no environment.
+set -euo pipefail
+
+PGREP="${ROSTER_PGREP:-pgrep}"
+PROC="${ROSTER_PROC:-/proc}"
+SELF_PID="${ROSTER_SELF_PID:-$$}"
+
+root="${1:-$PWD}"
+root="${root%/}"
+
+# Linux /proc is the substrate. Absent (non-Linux, or unreadable) means the
+# scan can't run; say so explicitly rather than reporting a false "alone".
+if [ ! -d "$PROC" ]; then
+ echo "agent-roster: roster unavailable (no $PROC; non-Linux or absent)" >&2
+ exit 2
+fi
+
+# pgrep is the enumeration boundary. Without it the scan can't run, and the
+# no-match exit code (1) below is indistinguishable from "tool missing" once
+# swallowed, so check up front rather than report a false "alone".
+if ! command -v "$PGREP" >/dev/null 2>&1; then
+ echo "agent-roster: roster unavailable ($PGREP not found)" >&2
+ exit 2
+fi
+
+# Build the scanner's ancestry set: SELF_PID and every parent up to init.
+# A Claude found by pgrep that lands in this set is the current session (or its
+# launcher chain), not another agent.
+ancestry=" "
+pid="$SELF_PID"
+while [ -n "$pid" ] && [ "$pid" != "0" ] && [ "$pid" != "1" ]; do
+ ancestry="${ancestry}${pid} "
+ status="$PROC/$pid/status"
+ [ -r "$status" ] || break
+ pid="$(awk '/^PPid:/{print $2; exit}' "$status")"
+done
+
+found=0
+while read -r candidate; do
+ [ -n "$candidate" ] || continue
+ case "$ancestry" in
+ *" $candidate "*) continue ;; # scanner's own ancestry
+ esac
+ # cwd may be gone if the process exited between pgrep and here; skip it.
+ cwd="$(readlink "$PROC/$candidate/cwd" 2>/dev/null)" || continue
+ [ -n "$cwd" ] || continue
+ # Keep only agents at or inside the project root. The trailing slashes make
+ # the prefix test exact, so /foo/project-other doesn't match /foo/project.
+ case "$cwd/" in
+ "$root"/*) ;;
+ *) continue ;;
+ esac
+ printf '%s\t%s\n' "$candidate" "$cwd"
+ found=1
+done < <("$PGREP" -x claude 2>/dev/null || true)
+
+[ "$found" -eq 1 ] && exit 1
+exit 0
diff --git a/.ai/scripts/apkg-to-orgdrill.py b/.ai/scripts/apkg-to-orgdrill.py
new file mode 100755
index 0000000..79e24a4
--- /dev/null
+++ b/.ai/scripts/apkg-to-orgdrill.py
@@ -0,0 +1,251 @@
+#!/usr/bin/env -S uv run --script
+# /// script
+# requires-python = ">=3.11"
+# dependencies = []
+# ///
+"""Convert an Anki .apkg deck into an org-drill file (inverse of flashcard-to-anki.py).
+
+The flashcard pipeline is otherwise one-directional (org-drill -> apkg).
+Decks curated on the phone, and orphaned apkgs whose .org source was never
+saved, can't get back into the org source of truth. This recovers them.
+
+Reading needs no third-party library: an apkg is a zip holding
+collection.anki2 / .anki21 (an Anki sqlite db) plus a media blob, so stdlib
+zipfile + sqlite3 suffice. genanki is only needed to write apkgs, not read
+them.
+
+Mapping (mirrors flashcard-to-anki.py's parse/build, inverted):
+ - Deck name (from the apkg) -> #+TITLE:
+ - Note Front -> ** <Front> :drill:
+ - Note Back (HTML) -> entry body (<br> -> newlines,
+ &amp;/&lt;/&gt; unescaped,
+ <hr id="answer"> stripped)
+ - Note tag -> * <tag> section grouping
+ (best-effort: the tag is a slug,
+ so it won't round-trip to the exact
+ original section title — a human
+ retitles)
+ - A fresh :ID: UUID per card -> so the output is org-drill-valid
+
+GUIDs in flashcard-to-anki.py are derived from the Front text, not the
+:ID:, so a deck regenerated from recovered org still matches existing phone
+cards by Front. Only Front/Back (Basic) note types convert; other models
+(cloze, etc.) are skipped with a warning rather than silently dropped.
+
+Usage:
+ apkg-to-orgdrill.py <input.apkg> # one <deck-slug>.org per deck in cwd
+ apkg-to-orgdrill.py <input.apkg> --output-dir DIR
+ apkg-to-orgdrill.py <input.apkg> --deck "Name" --output deck.org
+"""
+from __future__ import annotations
+
+import argparse
+import json
+import re
+import sqlite3
+import sys
+import tempfile
+import uuid
+import zipfile
+from collections import OrderedDict
+from dataclasses import dataclass
+from pathlib import Path
+
+# Collection member names Anki uses, newest schema first.
+COLLECTION_NAMES = ("collection.anki21", "collection.anki2")
+
+_BR_RE = re.compile(r"<br\s*/?>", re.IGNORECASE)
+_ANSWER_HR_RE = re.compile(r'<hr id="answer">', re.IGNORECASE)
+_MEDIA_RE = re.compile(r"<img\b|\[sound:|<audio\b|<video\b", re.IGNORECASE)
+
+
+@dataclass
+class Note:
+ deck: str
+ front: str
+ back_html: str
+ tag: str
+
+
+def html_to_org_body(back_html: str) -> list[str]:
+ """Invert flashcard-to-anki.py's back-of-card HTML into org body lines.
+
+ <br> (all spellings) and a stray answer <hr> become line breaks; the
+ entity unescape undoes escape_html, which escaped ``&`` first — so ``&``
+ is unescaped last here, or a literally-escaped ``&lt;`` in the source
+ would wrongly collapse to ``<``.
+ """
+ if not back_html:
+ return []
+ s = _ANSWER_HR_RE.sub("\n", back_html)
+ s = _BR_RE.sub("\n", s)
+ s = s.replace("&lt;", "<").replace("&gt;", ">").replace("&amp;", "&")
+ return s.split("\n")
+
+
+def _slug(title: str) -> str:
+ return re.sub(r"[^a-z0-9]+", "-", title.lower()).strip("-")
+
+
+def _read_collection(db_path: Path) -> list[Note]:
+ con = sqlite3.connect(db_path)
+ try:
+ row = con.execute("SELECT decks, models FROM col LIMIT 1").fetchone()
+ if row is None:
+ raise ValueError("collection has no col row")
+ decks_json, models_json = row
+ decks = {int(k): v["name"] for k, v in json.loads(decks_json).items()}
+ models = {
+ int(k): [f["name"] for f in v["flds"]]
+ for k, v in json.loads(models_json).items()
+ }
+
+ # A note's deck comes from its card; the Default deck (id 1) carries
+ # no cards from this pipeline, so it never shows up here.
+ nid_to_did: dict[int, int] = {}
+ for nid, did in con.execute("SELECT nid, did FROM cards"):
+ nid_to_did.setdefault(nid, did)
+
+ notes: list[Note] = []
+ for nid, mid, flds, tags in con.execute(
+ "SELECT id, mid, flds, tags FROM notes"
+ ):
+ field_names = models.get(mid)
+ if not field_names or "Front" not in field_names or "Back" not in field_names:
+ print(
+ f"apkg-to-orgdrill: skip note {nid} — model is not a Front/Back "
+ f"type (fields={field_names})",
+ file=sys.stderr,
+ )
+ continue
+ fields = flds.split("\x1f")
+ fi, bi = field_names.index("Front"), field_names.index("Back")
+ front = fields[fi] if fi < len(fields) else ""
+ back_html = fields[bi] if bi < len(fields) else ""
+
+ did = nid_to_did.get(nid)
+ if did is None:
+ continue # note with no card — orphan
+ deck = decks.get(did)
+ if deck is None:
+ continue
+
+ tag_list = tags.split()
+ tag = tag_list[0] if tag_list else "drill"
+
+ if _MEDIA_RE.search(back_html):
+ print(
+ f"apkg-to-orgdrill: note {nid} references media; org has no "
+ f"media path (left inline for a human to resolve)",
+ file=sys.stderr,
+ )
+ notes.append(Note(deck=deck, front=front, back_html=back_html, tag=tag))
+ return notes
+ finally:
+ con.close()
+
+
+def read_apkg(path: Path) -> list[Note]:
+ """Read an .apkg and return its Front/Back notes. Raises on a malformed file."""
+ with zipfile.ZipFile(path) as z: # BadZipFile if it isn't a zip
+ names = set(z.namelist())
+ col_name = next((n for n in COLLECTION_NAMES if n in names), None)
+ if col_name is None:
+ raise ValueError(f"{path}: no collection.anki2/.anki21 inside the apkg")
+ with tempfile.TemporaryDirectory() as td:
+ db_path = Path(td) / col_name
+ db_path.write_bytes(z.read(col_name))
+ return _read_collection(db_path)
+
+
+def notes_to_org(notes: list[Note], deck_name: str, *, new_id=None) -> str:
+ """Render one deck's notes as an org-drill file in the house shape."""
+ if new_id is None:
+ new_id = lambda: str(uuid.uuid4()) # noqa: E731
+ groups: "OrderedDict[str, list[Note]]" = OrderedDict()
+ for n in notes:
+ groups.setdefault(n.tag, []).append(n)
+
+ lines: list[str] = [f"#+TITLE: {deck_name}", ""]
+ for tag, group in groups.items():
+ lines.append(f"* {tag}")
+ for n in group:
+ lines.append(f"** {n.front} :drill:")
+ lines.append(":PROPERTIES:")
+ lines.append(f":ID: {new_id()}")
+ lines.append(":END:")
+ lines.extend(html_to_org_body(n.back_html))
+ lines.append("")
+ return "\n".join(lines).rstrip("\n") + "\n"
+
+
+def convert(apkg_path: Path, *, new_id=None) -> "OrderedDict[str, str]":
+ """apkg -> {deck_name: org_text}, one entry per deck that has Front/Back cards."""
+ by_deck: "OrderedDict[str, list[Note]]" = OrderedDict()
+ for n in read_apkg(apkg_path):
+ by_deck.setdefault(n.deck, []).append(n)
+ out: "OrderedDict[str, str]" = OrderedDict()
+ for deck, deck_notes in by_deck.items():
+ out[deck] = notes_to_org(deck_notes, deck, new_id=new_id)
+ return out
+
+
+def main() -> int:
+ parser = argparse.ArgumentParser(
+ description="Convert an Anki .apkg deck into an org-drill file.",
+ )
+ parser.add_argument("input", type=Path, help="Path to the .apkg file.")
+ parser.add_argument("--deck", help="Only convert the deck with this exact name.")
+ parser.add_argument(
+ "--output",
+ type=Path,
+ help="Output .org path. Requires a single deck (use --deck to pick one).",
+ )
+ parser.add_argument(
+ "--output-dir",
+ type=Path,
+ help="Directory for per-deck .org files (default: current directory).",
+ )
+ args = parser.parse_args()
+
+ input_path = args.input.expanduser().resolve()
+ if not input_path.is_file():
+ print(f"error: {input_path} not found", file=sys.stderr)
+ return 1
+
+ by_deck = convert(input_path)
+ if args.deck:
+ by_deck = OrderedDict((k, v) for k, v in by_deck.items() if k == args.deck)
+ if not by_deck:
+ print(f"error: no deck named {args.deck!r} in {input_path}", file=sys.stderr)
+ return 1
+ if not by_deck:
+ print(f"error: no Front/Back cards found in {input_path}", file=sys.stderr)
+ return 1
+
+ if args.output:
+ if len(by_deck) != 1:
+ print(
+ f"error: --output needs a single deck; {input_path} has "
+ f"{len(by_deck)} ({', '.join(by_deck)}). Use --deck or --output-dir.",
+ file=sys.stderr,
+ )
+ return 1
+ out = args.output.expanduser().resolve()
+ out.parent.mkdir(parents=True, exist_ok=True)
+ deck, org = next(iter(by_deck.items()))
+ out.write_text(org, encoding="utf-8")
+ print(f"wrote {out} ({org.count(':drill:')} cards, deck {deck!r})")
+ return 0
+
+ out_dir = (args.output_dir or Path.cwd()).expanduser().resolve()
+ out_dir.mkdir(parents=True, exist_ok=True)
+ for deck, org in by_deck.items():
+ out = out_dir / f"{_slug(deck) or 'deck'}.org"
+ out.write_text(org, encoding="utf-8")
+ print(f"wrote {out} ({org.count(':drill:')} cards, deck {deck!r})")
+ return 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/.ai/scripts/capture-guard b/.ai/scripts/capture-guard
new file mode 100755
index 0000000..6c01f2f
--- /dev/null
+++ b/.ai/scripts/capture-guard
@@ -0,0 +1,91 @@
+#!/usr/bin/env bash
+# capture-guard — detect live org-capture buffers visiting a target file
+# before a workflow edits that file on disk.
+#
+# Editing a file on disk while Emacs has an indirect org-capture buffer
+# cloned from it reverts the base buffer underneath the capture, wedging it:
+# the capture can no longer finalize cleanly with C-c C-c, and a freshly-typed
+# item can be lost or written back against post-edit content. inbox.org
+# roam mode Phase D edits ~/org/roam/inbox.org, the file Craig captures into constantly,
+# so it calls this guard first. See claude-rules/emacs.md.
+#
+# Usage: capture-guard [--wait[=SECONDS]] [TARGET_FILE] (default ~/org/roam/inbox.org)
+#
+# Single-shot (default): check once.
+# exit 0 — safe to edit: no Emacs, daemon unreachable, or no capture buffer
+# visits TARGET_FILE.
+# exit 1 — a live capture buffer visits TARGET_FILE; its name(s) printed to
+# stdout, comma-separated.
+#
+# --wait[=SECONDS]: poll until the capture clears or SECONDS elapse (default
+# 30), re-checking every ~10s. Org captures are usually transient — a few
+# seconds of mid-finalize state — so a short wait clears most false alarms
+# before a caller has to surface or skip. Same exit codes: exit 0 the moment
+# it's clear, exit 1 if still blocked at the deadline (last buffer list on
+# stdout). The common case (nothing capturing) returns instantly without
+# sleeping.
+#
+# Conservative by construction: any uncertainty (no Emacs, query failure)
+# resolves to "safe," so the guard never blocks a workflow that would have
+# been fine. It only stops the one case it can positively confirm.
+
+set -euo pipefail
+
+WAIT_TOTAL=0
+case "${1:-}" in
+ --wait) WAIT_TOTAL=30; shift ;;
+ --wait=*) WAIT_TOTAL="${1#--wait=}"; shift ;;
+esac
+
+TARGET="${1:-$HOME/org/roam/inbox.org}"
+INTERVAL=10
+
+# Names of capture buffers whose base buffer visits TARGET. file-equal-p
+# normalizes symlinks and ./.. so the match survives path spelling; it also
+# returns nil when TARGET doesn't exist, which collapses to "safe" below.
+lisp='(let ((target (expand-file-name "'"$TARGET"'")))
+ (mapconcat (function buffer-name)
+ (seq-filter
+ (lambda (b)
+ (and (string-prefix-p "CAPTURE" (buffer-name b))
+ (let* ((base (or (buffer-base-buffer b) b))
+ (f (buffer-file-name base)))
+ (and f (file-equal-p f target)))))
+ (buffer-list))
+ ","))'
+
+LAST_BUFS=""
+
+# detect — return 0 (safe) or 1 (blocked, name(s) in LAST_BUFS). Any
+# uncertainty resolves to safe, matching the single-shot contract.
+detect() {
+ command -v emacsclient >/dev/null 2>&1 || return 0
+ emacsclient -e t >/dev/null 2>&1 || return 0
+ local bufs
+ bufs="$(emacsclient -e "$lisp" 2>/dev/null)" || return 0
+ bufs="${bufs#\"}"
+ bufs="${bufs%\"}"
+ if [ -n "$bufs" ]; then
+ LAST_BUFS="$bufs"
+ return 1
+ fi
+ return 0
+}
+
+# Poll loop. With WAIT_TOTAL=0 (single-shot) it checks once and falls straight
+# through to the exit-1 branch on a block, never sleeping. Each sleep is capped
+# to the remaining budget so a short --wait never overshoots its deadline.
+elapsed=0
+while :; do
+ if detect; then
+ exit 0
+ fi
+ if [ "$elapsed" -ge "$WAIT_TOTAL" ]; then
+ echo "$LAST_BUFS"
+ exit 1
+ fi
+ remaining=$((WAIT_TOTAL - elapsed))
+ step=$((remaining < INTERVAL ? remaining : INTERVAL))
+ sleep "$step"
+ elapsed=$((elapsed + step))
+done
diff --git a/.ai/scripts/cj-remove-block.py b/.ai/scripts/cj-remove-block.py
index 71c7b3d..d5137a3 100755
--- a/.ai/scripts/cj-remove-block.py
+++ b/.ai/scripts/cj-remove-block.py
@@ -16,8 +16,12 @@ Companion to the /respond-to-cj-comments skill and to cj-scan.py.
from __future__ import annotations
import argparse
+import os
import re
+import shutil
import sys
+import tempfile
+from datetime import datetime
from pathlib import Path
SRC_OPEN_RE = re.compile(r"^\s*#\+begin_src\s+cj:", re.IGNORECASE)
@@ -57,12 +61,83 @@ def looks_like_cj_range(lines: list[str], start: int, end: int) -> tuple[bool, s
f"Line {end} does not look like a #+end_src closing fence "
f"(got: {last[:60]!r})"
)
+
+ # The range must hold exactly ONE block. Checking only the first and last
+ # lines let a drifted range run from one block's opener to a *later* block's
+ # closer: validation passed and the removal silently deleted everything
+ # between, prose and headings included. Drift is the case this check exists
+ # for, so it has to look inside the range, not just at its ends.
+ for offset, line in enumerate(lines[start:end - 1], start=start + 1):
+ if SRC_CLOSE_RE.match(line):
+ return False, (
+ f"Range {start}..{end} covers more than one cj block — "
+ f"a #+end_src appears at line {offset}, before the range ends. "
+ f"Re-scan for current line numbers; removing this range would "
+ f"delete everything between the two blocks."
+ )
+ if SRC_OPEN_RE.match(line):
+ return False, (
+ f"Range {start}..{end} covers more than one cj block — "
+ f"a second #+begin_src cj: appears at line {offset}. "
+ f"Re-scan for current line numbers."
+ )
return True, ""
+def _backup(path: Path) -> Path:
+ """Copy path to /tmp before mutating it, mirroring lint-org.el's convention.
+
+ These are Craig's org files. lint-org.el, the other tool that rewrites them,
+ leaves a /tmp copy before touching anything; this matches it so a bad edit is
+ always recoverable without reaching for git (which only reaches the last
+ commit, losing intra-session work).
+ """
+ stamp = datetime.now().strftime("%Y%m%d-%H%M%S")
+ base = Path(tempfile.gettempdir()) / f"{path.name}.before-cj-remove.{stamp}"
+ # Never overwrite an earlier backup. The skill removes several annotations
+ # in quick succession, so a second-resolution stamp collides and the later
+ # copy would replace the earlier one with already-mutated content — losing
+ # the pre-session original the backup exists to preserve.
+ dest = base
+ n = 2
+ while dest.exists():
+ dest = base.with_name(f"{base.name}-{n}")
+ n += 1
+ shutil.copy2(path, dest)
+ return dest
+
+
+def _atomic_write(path: Path, text: str) -> None:
+ """Write text to path via a temp sibling and os.replace.
+
+ A bare write_text truncates the target on open, so a mid-write failure left
+ the org file truncated with no complete copy on disk. Writing a temp sibling
+ and renaming means the file is either its old content or its new content,
+ never a partial.
+ """
+ # Follow a symlink to the file it names. os.replace would otherwise swap the
+ # symlink itself for a regular file, leaving the real target holding the old
+ # content — the edit silently goes nowhere. Resolving also puts the temp
+ # sibling on the same filesystem as the real file, which os.replace needs.
+ path = path.resolve()
+ fd, tmp = tempfile.mkstemp(dir=path.parent, prefix=f".{path.name}.", suffix=".tmp")
+ os.close(fd)
+ tmp_path = Path(tmp)
+ # Carry the original's permissions across. mkstemp creates 0600, and
+ # defaulting to the umask instead widened a deliberately-restricted file
+ # (a 0600 org file came back 0644).
+ shutil.copymode(path, tmp_path)
+ try:
+ tmp_path.write_text(text, encoding="utf-8")
+ os.replace(tmp_path, path)
+ except BaseException:
+ tmp_path.unlink(missing_ok=True)
+ raise
+
+
def remove_range(path: Path, start: int, end: int) -> None:
"""Read path, validate range looks like cj content, remove the range, write back."""
- text = path.read_text()
+ text = path.read_text(encoding="utf-8")
had_trailing_newline = text.endswith("\n")
lines = text.splitlines(keepends=False)
@@ -77,7 +152,9 @@ def remove_range(path: Path, start: int, end: int) -> None:
new_text += "\n"
elif not new_lines and had_trailing_newline:
new_text = ""
- path.write_text(new_text)
+
+ _backup(path)
+ _atomic_write(path, new_text)
def main() -> int:
diff --git a/.ai/scripts/cross-agent-comms/cross-agent-discover b/.ai/scripts/cross-agent-comms/cross-agent-discover
deleted file mode 100755
index 152cf27..0000000
--- a/.ai/scripts/cross-agent-comms/cross-agent-discover
+++ /dev/null
@@ -1,230 +0,0 @@
-#!/usr/bin/env python3
-"""Enumerate cross-agent destinations: local projects + tailnet peers.
-
-See cross-agent-discover.md. Local: scan ~/projects/*/.ai/. Peers: read
-peers.toml, SSH-probe each for reachability. --enumerate-remote optionally
-runs `ls -d ~/projects/*/.ai/` over SSH to list remote projects.
-
-Cache results for 5 min at ~/.cache/cross-agent-comms/discovery.json so
-repeated invocations don't re-probe.
-
-HALT: prints a banner; otherwise continues.
-"""
-
-from __future__ import annotations
-
-import argparse
-import datetime as _dt
-import json
-import os
-import subprocess
-import sys
-import time
-import tomllib
-from pathlib import Path
-
-CONFIG_DIR = Path.home() / ".config" / "cross-agent-comms"
-PEERS_TOML = CONFIG_DIR / "peers.toml"
-HALT_FILE = CONFIG_DIR / "HALT"
-CACHE_DIR = Path.home() / ".cache" / "cross-agent-comms"
-CACHE_FILE = CACHE_DIR / "discovery.json"
-CACHE_TTL_SECONDS = 300
-
-EXIT_OK = 0
-EXIT_GENERAL = 1
-EXIT_PEERS_TOML = 1
-
-
-def err(msg: str) -> None:
- print(msg, file=sys.stderr)
-
-
-def render_banner_if_halt() -> None:
- if not HALT_FILE.exists():
- return
- try:
- reason = HALT_FILE.read_text().strip()
- except OSError:
- reason = "(HALT file unreadable; treated as halted)"
- print("⚠ HALT ACTIVE — cross-agent comms paused")
- if reason:
- print(f" reason: {reason}")
- print()
-
-
-def enumerate_local_projects() -> list[str]:
- projects_dir = Path.home() / "projects"
- if not projects_dir.is_dir():
- return []
- found = []
- for child in sorted(projects_dir.iterdir()):
- if child.is_dir() and (child / ".ai").is_dir():
- found.append(child.name)
- return found
-
-
-def load_peers() -> dict:
- if not PEERS_TOML.exists():
- return {"peers": {}}
- try:
- return tomllib.loads(PEERS_TOML.read_text())
- except (tomllib.TOMLDecodeError, OSError) as e:
- err(f"cannot parse peers.toml: {e}")
- sys.exit(EXIT_PEERS_TOML)
-
-
-def probe_peer_reachability(host: str, ssh_user: str | None) -> tuple[bool, str | None]:
- """Run a short SSH probe with BatchMode=yes (no interactive prompt)."""
- target = f"{ssh_user}@{host}" if ssh_user else host
- try:
- result = subprocess.run(
- ["ssh", "-o", "ConnectTimeout=2", "-o", "BatchMode=yes", target, "true"],
- capture_output=True,
- text=True,
- timeout=5,
- )
- except (FileNotFoundError, subprocess.TimeoutExpired):
- return False, "ssh probe failed"
- if result.returncode == 0:
- return True, None
- return False, (result.stderr.strip().splitlines() or [f"exit {result.returncode}"])[-1]
-
-
-def enumerate_remote_projects(host: str, ssh_user: str | None) -> list[str] | None:
- target = f"{ssh_user}@{host}" if ssh_user else host
- try:
- result = subprocess.run(
- [
- "ssh", "-o", "ConnectTimeout=3", "-o", "BatchMode=yes", target,
- "ls -d ~/projects/*/.ai/ 2>/dev/null",
- ],
- capture_output=True,
- text=True,
- timeout=10,
- )
- except (FileNotFoundError, subprocess.TimeoutExpired):
- return None
- if result.returncode != 0:
- return None
- projects = []
- for line in result.stdout.splitlines():
- # Each line looks like /home/<user>/projects/<name>/.ai/
- parts = line.rstrip("/").split("/")
- if len(parts) >= 2 and parts[-1] == ".ai":
- projects.append(parts[-2])
- return projects
-
-
-def read_cache() -> dict | None:
- if not CACHE_FILE.exists():
- return None
- try:
- age = time.time() - CACHE_FILE.stat().st_mtime
- if age > CACHE_TTL_SECONDS:
- return None
- return json.loads(CACHE_FILE.read_text())
- except (OSError, json.JSONDecodeError):
- return None
-
-
-def write_cache(payload: dict) -> None:
- CACHE_DIR.mkdir(parents=True, exist_ok=True)
- CACHE_FILE.write_text(json.dumps(payload, indent=2))
-
-
-def discover(peer_filter: str | None, enumerate_remote: bool) -> dict:
- local = enumerate_local_projects()
- peers_cfg = load_peers().get("peers", {})
-
- peers_out = []
- for name, cfg in sorted(peers_cfg.items()):
- if peer_filter and name != peer_filter:
- continue
- host = cfg.get("host", name)
- ssh_user = cfg.get("ssh_user")
- reachable, error = probe_peer_reachability(host, ssh_user)
- entry = {
- "name": name,
- "host": host,
- "reachable": reachable,
- }
- if not reachable:
- entry["error"] = error
- if enumerate_remote and reachable:
- entry["projects"] = enumerate_remote_projects(host, ssh_user) or []
- peers_out.append(entry)
-
- return {
- "scanned_at": _dt.datetime.now(_dt.timezone.utc).isoformat(),
- "halt_active": HALT_FILE.exists(),
- "local": local,
- "peers": peers_out,
- }
-
-
-def render_table(payload: dict, enumerate_remote: bool) -> None:
- local = payload.get("local", [])
- print(f"Local ({_local_hostname()}):")
- if local:
- wrapped = ", ".join(local)
- print(f" {wrapped} [{len(local)} project{'s' if len(local) != 1 else ''}]")
- else:
- print(" (no projects with .ai/ found)")
- print()
-
- peers = payload.get("peers", [])
- if not peers:
- print("Peers (from peers.toml):")
- print(" (no peers configured)")
- return
-
- print("Peers (from ~/.config/cross-agent-comms/peers.toml):")
- for p in peers:
- marker = "✓ reachable" if p.get("reachable") else f"✗ UNREACHABLE ({p.get('error', 'unknown')})"
- print(f" {p['name']:<16} {p['host']:<24} {marker}")
- if enumerate_remote and p.get("projects"):
- wrapped = ", ".join(p["projects"])
- print(f" projects: {wrapped}")
-
-
-def _local_hostname() -> str:
- import socket
- return socket.gethostname().split(".")[0]
-
-
-def main() -> int:
- parser = argparse.ArgumentParser(description="Discover cross-agent destinations.")
- parser.add_argument("--enumerate-remote", action="store_true",
- help="SSH into each peer and list ~/projects/*/.ai/")
- parser.add_argument("--no-cache", action="store_true", help="Skip cache; force fresh probe")
- parser.add_argument("--peer", help="Limit to a single peer name from peers.toml")
- parser.add_argument("--json", action="store_true", help="Machine-readable output")
- args = parser.parse_args()
-
- render_banner_if_halt()
-
- payload = None
- if not args.no_cache:
- cached = read_cache()
- if cached is not None:
- # Honor --peer filter on cached payload.
- if args.peer:
- cached["peers"] = [p for p in cached.get("peers", []) if p["name"] == args.peer]
- payload = cached
-
- if payload is None:
- payload = discover(args.peer, args.enumerate_remote)
- if not args.no_cache and not args.peer:
- # Only cache full (unfiltered) discoveries.
- write_cache(payload)
-
- if args.json:
- print(json.dumps(payload, indent=2))
- return EXIT_OK
-
- render_table(payload, args.enumerate_remote)
- return EXIT_OK
-
-
-if __name__ == "__main__":
- sys.exit(main())
diff --git a/.ai/scripts/cross-agent-comms/cross-agent-discover.md b/.ai/scripts/cross-agent-comms/cross-agent-discover.md
deleted file mode 100644
index 95134bb..0000000
--- a/.ai/scripts/cross-agent-comms/cross-agent-discover.md
+++ /dev/null
@@ -1,155 +0,0 @@
-# cross-agent-discover
-
-**Purpose.** Enumerate available cross-agent destinations — local projects on
-this machine and remote projects on tailnet peers. Validates SSH reachability
-for cross-machine destinations before reporting them as usable.
-
-## Usage
-
-```
-cross-agent-discover [--enumerate-remote] [--no-cache] [--peer <name>]
-```
-
-No args required for the common case (local enumeration + peer reachability).
-
-### Flags
-
-| Flag | Default | Purpose |
-|---|---|---|
-| `--enumerate-remote` | off | SSH into each peer and list projects under `~/projects/*/.ai/`. Off by default because SSH adds latency; turn on when you want to see what's available on a remote machine you haven't fully configured. |
-| `--no-cache` | off | Skip the 5-minute cache; force fresh discovery. |
-| `--peer <name>` | (all) | Limit to a single peer from `peers.toml`. |
-| `--json` | off | Machine-readable output. |
-
-## Output
-
-### Default
-
-```
-$ cross-agent-discover
-Local (ratio):
- career, claude-templates, clipper, danneel, documents, elibrary,
- finances, health, homelab, jr-estate, kit, little-elisper,
- philosophy, website [14 projects]
-
-Peers (from ~/.config/cross-agent-comms/peers.toml):
- velox.local reachable (last seen 2 sec ago)
- bastion.local UNREACHABLE (ssh exit 255: connection refused)
-```
-
-### With `--enumerate-remote`
-
-```
-$ cross-agent-discover --enumerate-remote
-Local (ratio):
- ... (as above)
-
-velox.local (reachable):
- career, homelab [2 projects]
-```
-
-## Configuration
-
-Reads `~/.config/cross-agent-comms/peers.toml`:
-
-```toml
-# Each peer is a remote machine reachable via SSH (typically over Tailscale).
-
-[peers.velox]
-host = "velox.local"
-ssh_user = "cjennings"
-
-[peers.bastion]
-host = "bastion.local"
-ssh_user = "cjennings"
-```
-
-Peers entries describe machines, NOT projects. Projects are enumerated
-on-demand under `~/projects/*/.ai/` either locally or via SSH.
-
-## Cache
-
-Successful discovery results are cached at
-`~/.cache/cross-agent-comms/discovery.json` for 5 minutes. Repeated invocations
-within the window read from cache.
-
-`--no-cache` forces a fresh probe. Useful when adding a new peer or after a
-network change.
-
-## SSH reachability check
-
-For each peer, runs:
-
-```
-ssh -o ConnectTimeout=2 -o BatchMode=yes <user>@<host> true
-```
-
-`BatchMode=yes` prevents interactive password prompts — peers that don't have
-key-based auth set up are reported as UNREACHABLE.
-
-If `--enumerate-remote` is set, on success runs:
-
-```
-ssh <user>@<host> 'ls -d ~/projects/*/.ai/ 2>/dev/null'
-```
-
-## Failure modes
-
-| Symptom | Likely cause | Fix |
-|---|---|---|
-| Peer reported UNREACHABLE | Tailscale not connected, SSH key not authorized, host firewalled | `tailscale status`; `ssh -v <peer>` to debug. |
-| Local list is empty | Glob misresolved, or `~/projects/` doesn't exist | Check `ls -d ~/projects/*/.ai/`. |
-| `--enumerate-remote` slow | Cold cache, slow tailnet, many peers | First run is slow, subsequent runs hit cache. Use `--peer <name>` to scope. |
-| Peer unexpectedly missing from output | Not in `peers.toml`, or `peers.toml` malformed | `cat ~/.config/cross-agent-comms/peers.toml` and validate. |
-
-## HALT awareness
-
-Checks `~/.config/cross-agent-comms/HALT` at start. If HALT exists, prints a
-prominent banner before normal output:
-
-```
-$ cross-agent-discover
-⚠ HALT ACTIVE — cross-agent comms paused
- Reason: <reason from HALT file body, if any>
- Resume with: cross-agent-resume
-
-(enumeration continues normally — HALT does not suppress visibility)
-
-Local (ratio):
- career, claude-templates, ...
-
-Peers:
- velox.local reachable
-```
-
-Discover is read-only. Like `cross-agent-status`, it always runs so the user
-keeps visibility into what destinations exist regardless of halt state. The
-banner makes the halt state impossible to miss.
-
-If the HALT file exists but is unreadable, print a warning banner and
-continue.
-
-See `cross-agent-halt.md` for the full halt mechanism.
-
-## Examples
-
-```bash
-# Common: see what's available
-cross-agent-discover
-
-# Force fresh probe after network change
-cross-agent-discover --no-cache
-
-# What's on velox specifically
-cross-agent-discover --peer velox --enumerate-remote
-
-# Pipe to grep
-cross-agent-discover --json | jq '.peers[] | select(.reachable)'
-```
-
-## See also
-
-- `cross-agent-send` — uses `peers.toml` for routing destinations.
-- `cross-agent-status` — local pending messages.
-- `cross-agent-comms.org` — protocol spec, `* Limitations` section
- explains the cross-machine model.
diff --git a/.ai/scripts/cross-agent-comms/cross-agent-halt b/.ai/scripts/cross-agent-comms/cross-agent-halt
deleted file mode 100755
index df25115..0000000
--- a/.ai/scripts/cross-agent-comms/cross-agent-halt
+++ /dev/null
@@ -1,134 +0,0 @@
-#!/usr/bin/env python3
-"""Failsafe halt for cross-agent comms.
-
-See cross-agent-halt.md. Touches ~/.config/cross-agent-comms/HALT and stops
-the cross-agent-watch systemd user service. With --tailnet, propagates the
-HALT file to every peer in peers.toml via SSH; reports per-peer status with
-non-zero exit on partial halt.
-
-Does NOT pkill in-flight scripts — they detect HALT on next iteration and
-stop themselves.
-"""
-
-from __future__ import annotations
-
-import argparse
-import subprocess
-import sys
-import tomllib
-from pathlib import Path
-
-CONFIG_DIR = Path.home() / ".config" / "cross-agent-comms"
-HALT_FILE = CONFIG_DIR / "HALT"
-PEERS_TOML = CONFIG_DIR / "peers.toml"
-
-EXIT_OK = 0
-EXIT_PARTIAL = 1
-
-
-def err(msg: str) -> None:
- print(msg, file=sys.stderr)
-
-
-def write_halt_file(reason: str) -> None:
- CONFIG_DIR.mkdir(parents=True, exist_ok=True)
- HALT_FILE.write_text((reason + "\n") if reason else "")
-
-
-def stop_watcher_service() -> None:
- """Best-effort stop of the systemd watcher service. Failures are logged but not fatal."""
- try:
- subprocess.run(
- ["systemctl", "--user", "stop", "cross-agent-watch.path"],
- capture_output=True, text=True, timeout=5,
- )
- except (FileNotFoundError, subprocess.TimeoutExpired):
- # Watcher service may not be installed — fine.
- pass
-
-
-def load_peers() -> dict:
- if not PEERS_TOML.exists():
- return {}
- try:
- return tomllib.loads(PEERS_TOML.read_text())
- except (tomllib.TOMLDecodeError, OSError) as e:
- err(f"cannot parse peers.toml: {e}")
- return {}
-
-
-def ssh_touch_halt(host: str, ssh_user: str | None, reason: str) -> tuple[bool, str]:
- target = f"{ssh_user}@{host}" if ssh_user else host
- # Build the remote command. Quote the reason carefully.
- remote_cmd = (
- f"mkdir -p ~/.config/cross-agent-comms && "
- f"printf %s {_sh_quote(reason)} > ~/.config/cross-agent-comms/HALT"
- )
- try:
- result = subprocess.run(
- ["ssh", "-o", "ConnectTimeout=3", "-o", "BatchMode=yes", target, remote_cmd],
- capture_output=True, text=True, timeout=10,
- )
- except (FileNotFoundError, subprocess.TimeoutExpired):
- return False, "ssh unavailable or timed out"
- if result.returncode == 0:
- return True, "HALT file written"
- return False, (result.stderr.strip().splitlines() or [f"exit {result.returncode}"])[-1]
-
-
-def _sh_quote(s: str) -> str:
- return "'" + s.replace("'", "'\"'\"'") + "'"
-
-
-def main() -> int:
- parser = argparse.ArgumentParser(description="Halt all cross-agent comms on this machine (and optionally tailnet).")
- parser.add_argument("reason", nargs="?", default="", help="Optional human-readable reason")
- parser.add_argument("--tailnet", action="store_true",
- help="Propagate HALT to every peer in peers.toml")
- args = parser.parse_args()
-
- # Local halt.
- write_halt_file(args.reason)
- stop_watcher_service()
- print("Halting locally ✓ (HALT file written)")
-
- if not args.tailnet:
- print()
- print(f"Halt active. Remove {HALT_FILE} or run cross-agent-resume to clear.")
- print("Agent polling will stop within ~5 min (one cadence cycle).")
- return EXIT_OK
-
- peers = load_peers().get("peers", {})
- if not peers:
- print()
- print("No peers configured in peers.toml — local-only halt complete.")
- return EXIT_OK
-
- print()
- successes = 1 # local already counted
- failures = []
- for name, cfg in sorted(peers.items()):
- host = cfg.get("host", name)
- ssh_user = cfg.get("ssh_user")
- ok, detail = ssh_touch_halt(host, ssh_user, args.reason)
- marker = "✓" if ok else "✗"
- print(f"Halting {host:<28} {marker} ({detail})")
- if ok:
- successes += 1
- else:
- failures.append(f"{name} ({host}): {detail}")
-
- print()
- total = len(peers) + 1
- if failures:
- print(f"PARTIAL HALT: {successes}/{total} machines halted.")
- for f in failures:
- print(f" - {f}")
- print("Resolve the failures or manually halt each machine.")
- return EXIT_PARTIAL
- print(f"Halt active across {total} machine(s).")
- return EXIT_OK
-
-
-if __name__ == "__main__":
- sys.exit(main())
diff --git a/.ai/scripts/cross-agent-comms/cross-agent-halt.md b/.ai/scripts/cross-agent-comms/cross-agent-halt.md
deleted file mode 100644
index b817fbc..0000000
--- a/.ai/scripts/cross-agent-comms/cross-agent-halt.md
+++ /dev/null
@@ -1,134 +0,0 @@
-# cross-agent-halt
-
-**Purpose.** Failsafe stop for all cross-agent activity on the local machine
-(or, with `--tailnet`, across all configured peers). Creates the HALT file
-that every component in the protocol checks; within one polling cadence
-(~5 min) all polling, sending, watching, and receiving stops.
-
-This is the user's emergency brake. Use when something is misbehaving and
-visiting individual sessions is too slow.
-
-## Usage
-
-```
-cross-agent-halt [reason] [--tailnet] [--no-stop-watcher]
-```
-
-### Positional argument
-
-| Position | Meaning | Example |
-|---|---|---|
-| 1 | Optional human-readable reason for the halt. Written into the HALT file's body. Helps future-you remember why you stopped things. | `"investigating runaway poll loop, 2026-04-27"` |
-
-### Flags
-
-| Flag | Default | Purpose |
-|---|---|---|
-| `--tailnet` | local only | Propagate halt to every peer in `peers.toml` via SSH over Tailscale. |
-| `--no-stop-watcher` | (stops watcher) | Skip stopping the `cross-agent-watch.path` systemd unit. Useful if the watcher is intentionally separate from comms (rare). |
-
-## Behavior
-
-### Local halt (default)
-
-1. Write the HALT file: `~/.config/cross-agent-comms/HALT`. If a `[reason]` was
- passed, write it as the file's body. Otherwise the file is empty (existence
- alone triggers halt).
-2. Stop the watcher service: `systemctl --user stop cross-agent-watch.path`
- (and the corresponding `.service` if running).
-3. Print a summary:
- ```
- ✓ HALT file written: ~/.config/cross-agent-comms/HALT
- ✓ Watcher service stopped (cross-agent-watch.path)
- - In-flight sends will complete their current rsync step (~seconds), then
- stop. New sends are blocked.
- - Active agent polling sessions stop within one cadence (~5 min).
- - Use `cross-agent-resume` to clear HALT.
- Per-session polling does NOT auto-resume — you re-engage each session by
- telling its agent to resume polling.
- ```
-4. Exit 0.
-
-### Cross-tailnet halt (`--tailnet`)
-
-1. Apply local halt steps 1-2 first.
-2. Read `peers.toml` for the list of remote machines.
-3. For each peer, SSH and write the HALT file:
- ```
- ssh <user>@<host> "echo '<reason>' > ~/.config/cross-agent-comms/HALT && \
- systemctl --user stop cross-agent-watch.path"
- ```
-4. Track per-peer success/failure. Print results:
- ```
- Halting velox.local ✓ (HALT file written)
- Halting bastion.local ✗ (ssh exit 255: no route to host)
- Halting locally ✓ (HALT file written)
-
- PARTIAL HALT: 2/3 machines halted. bastion.local needs manual halt.
- ```
-5. Exit 0 if all peers halted; exit 1 if any peer failed (so scripts can
- detect partial halt). The local halt always succeeds — even on `--tailnet`,
- if remote peers fail, local is still halted.
-
-## What "halt active" means for each component
-
-| Component | Behavior under HALT |
-|---|---|
-| `cross-agent-send` | Refuses to send. Exits 5 with "halt active; remove ~/.config/cross-agent-comms/HALT to resume." Checks HALT at start AND between each retry/rsync step, so an in-flight send completes its current step then stops. |
-| `cross-agent-recv` | Refuses to verify or dedup. Exits 5 with same message. Inbound files are **left in place** — not moved, not rejected — so resume picks them up cleanly via cold-start. |
-| `cross-agent-watch` | Continues running but suppresses notifications. Logs each event with `(suppressed by HALT)` so the operator can see what would have fired. |
-| `cross-agent-status` | Prints prominent `⚠ HALT ACTIVE` banner before normal output. Continues to enumerate (read-only). |
-| `cross-agent-discover` | Same banner. Continues (read-only). |
-| Agent polling loops | Check HALT on every wake. If set: write a final `progress` note to any active conversation ("HALT fired locally; pausing"), surface "(HALT active; cross-agent comms paused)" in every user response, and stop rescheduling. Polling decays naturally within one cadence. |
-| Conversation initiator | Refuses to write sequence 1 of any new conversation. Surfaces refusal to user. |
-| Startup workflow (Phase A) | Checks HALT at session boot. If set, surfaces immediately and skips cross-agent inbox checks. |
-
-## Failure modes
-
-| Symptom | Cause | Fix |
-|---|---|---|
-| `~/.config/cross-agent-comms/HALT` already exists | Halt was already active | OK — running halt again refreshes the reason text. Safe. |
-| `systemctl --user stop` fails | Watcher service not installed, or systemd not available | The HALT file is still written — components that check HALT will still stop. The systemctl failure surfaces as a non-fatal warning. |
-| `--tailnet` halts some peers but not others | One or more peers unreachable | Exit 1 with per-peer status. Manually halt the unreachable peers (visit each machine, `touch ~/.config/cross-agent-comms/HALT`), or fix the network and re-run. |
-| Permission denied writing the HALT file | `~/.config/cross-agent-comms/` doesn't exist or is owned by another user | `mkdir -p ~/.config/cross-agent-comms/`; check ownership. |
-
-## What halt does NOT do
-
-- Does not kill running Claude sessions. Polling stops within ~5 min, but the
- session itself stays alive and can be re-engaged after resume.
-- Does not delete pending messages. Inbound files in `inbox/from-agents/`
- remain; they get processed when polling resumes.
-- Does not abort in-flight rsync push mid-byte. Atomic-write semantics
- guarantee in-flight messages either complete cleanly or leave only `.tmp.*`
- files (which receivers ignore).
-
-## Examples
-
-```bash
-# Quick halt with no reason
-cross-agent-halt
-
-# Halt with a memo
-cross-agent-halt "runaway poll loop in homelab session, debugging"
-
-# Halt all tailnet peers + local
-cross-agent-halt --tailnet "shutting down for system update"
-
-# Halt protocol comms but leave the watcher service running
-cross-agent-halt --no-stop-watcher
-```
-
-## Recovery
-
-Always pair with `cross-agent-resume` when the situation is resolved:
-
-```bash
-cross-agent-resume # local
-cross-agent-resume --tailnet # all peers
-```
-
-## See also
-
-- `cross-agent-resume` — counterpart that clears HALT.
-- `cross-agent-status` — see HALT state at a glance.
-- `cross-agent-comms.org` — protocol spec, `* Halt mechanism` section.
diff --git a/.ai/scripts/cross-agent-comms/cross-agent-recv b/.ai/scripts/cross-agent-comms/cross-agent-recv
deleted file mode 100755
index b67533a..0000000
--- a/.ai/scripts/cross-agent-comms/cross-agent-recv
+++ /dev/null
@@ -1,250 +0,0 @@
-#!/usr/bin/env python3
-"""Cross-agent message receiver.
-
-See cross-agent-recv.md for the full contract. Reads one message file and
-emits a structured decision the agent acts on:
-
- process | dedup | query | reject
-
-Decision exit codes:
- 0 = process 1 = dedup 2 = query 3 = reject
-
-When HALT is set, the script refuses to verify or dedup and leaves the
-inbound file in place — resume picks it up via cold-start.
-"""
-
-from __future__ import annotations
-
-import argparse
-import hashlib
-import json
-import re
-import shutil
-import subprocess
-import sys
-from pathlib import Path
-
-CONFIG_DIR = Path.home() / ".config" / "cross-agent-comms"
-HALT_FILE = CONFIG_DIR / "HALT"
-EXPECTED_PROTOCOL_VERSION = "5"
-
-REQUIRED_FRONTMATTER = ["TITLE", "CONVERSATION_ID", "MESSAGE_TYPE", "SEQUENCE", "TIMESTAMP", "PROTOCOL_VERSION"]
-VALID_MESSAGE_TYPES = {"request", "progress", "query", "pushback", "complete", "release", "escalate"}
-
-DEC_PROCESS = "process"
-DEC_DEDUP = "dedup"
-DEC_QUERY = "query"
-DEC_REJECT = "reject"
-
-EXIT_FOR_DECISION = {
- DEC_PROCESS: 0,
- DEC_DEDUP: 1,
- DEC_QUERY: 2,
- DEC_REJECT: 3,
-}
-
-EXIT_HALT = 5
-
-
-def err(msg: str) -> None:
- print(msg, file=sys.stderr)
-
-
-def check_halt() -> None:
- if HALT_FILE.exists():
- try:
- reason = HALT_FILE.read_text().strip()
- except OSError:
- err("halt active (HALT file present but unreadable; treated as halted)")
- sys.exit(EXIT_HALT)
- msg = "halt active; leaving inbound message in place (resume will pick up)"
- if reason:
- msg = f"{msg}: {reason}"
- err(msg)
- sys.exit(EXIT_HALT)
-
-
-def parse_frontmatter(path: Path) -> dict[str, str]:
- try:
- text = path.read_text()
- except OSError as e:
- return {"_parse_error": f"cannot read: {e}"}
- fm: dict[str, str] = {}
- for line in text.splitlines():
- line = line.rstrip()
- if not line:
- if fm:
- break
- continue
- m = re.match(r"#\+([A-Z_]+):\s*(.*)", line)
- if m:
- fm[m.group(1)] = m.group(2).strip()
- elif fm:
- break
- return fm
-
-
-def emit_decision(
- decision: str,
- reason: str | None,
- fm: dict[str, str],
- sha256: str | None,
- args: argparse.Namespace,
-) -> int:
- payload = {
- "decision": decision,
- "reason": reason,
- "message_type": fm.get("MESSAGE_TYPE"),
- "conversation_id": fm.get("CONVERSATION_ID"),
- "sequence": fm.get("SEQUENCE"),
- "timestamp": fm.get("TIMESTAMP"),
- "sha256": sha256,
- }
- if args.json:
- print(json.dumps(payload, indent=None if args.compact_json else 2))
- else:
- print(f"decision: {decision}")
- if reason:
- print(f"reason: {reason}")
- for k in ("message_type", "conversation_id", "sequence", "timestamp"):
- v = payload[k]
- if v is not None:
- print(f"{k}: {v}")
- if sha256:
- print(f"sha256: {sha256}")
- return EXIT_FOR_DECISION[decision]
-
-
-def gpg_verify(message_path: Path, sig_path: Path) -> tuple[bool, str]:
- try:
- result = subprocess.run(
- ["gpg", "--verify", str(sig_path), str(message_path)],
- capture_output=True,
- text=True,
- )
- except FileNotFoundError:
- return False, "gpg not installed"
- if result.returncode == 0:
- return True, ""
- return False, result.stderr.strip().splitlines()[-1] if result.stderr.strip() else f"exit {result.returncode}"
-
-
-def sha256_of(path: Path) -> str:
- h = hashlib.sha256()
- with path.open("rb") as f:
- for chunk in iter(lambda: f.read(65536), b""):
- h.update(chunk)
- return h.hexdigest()
-
-
-def find_dedup_match(message_path: Path, fm: dict[str, str], my_hash: str) -> tuple[str, str | None]:
- """Scan the message's directory for same-CONVERSATION_ID/SEQUENCE files.
-
- Returns (decision, reason) — decision is DEC_DEDUP for an exact-hash match,
- or DEC_PROCESS when no match or hash differs (sequence collision is OK).
- """
- parent = message_path.parent
- conv_id = fm["CONVERSATION_ID"]
- sequence = fm["SEQUENCE"]
- for sibling in parent.iterdir():
- if sibling == message_path or not sibling.is_file() or sibling.suffix != ".org":
- continue
- sib_fm = parse_frontmatter(sibling)
- if sib_fm.get("CONVERSATION_ID") != conv_id or sib_fm.get("SEQUENCE") != sequence:
- continue
- # Same conv-id + same sequence — check hash.
- if sha256_of(sibling) == my_hash:
- return DEC_DEDUP, f"identical retry of {sibling.name}"
- return DEC_PROCESS, None
-
-
-def check_requires_tools(fm: dict[str, str]) -> tuple[bool, list[str]]:
- """REQUIRES_TOOLS is a comma-separated list of tool names.
-
- For v5, "tool available" is a heuristic: an executable on PATH whose name
- matches the tool slug. MCP availability is currently out of scope (no
- portable way to query it from a CLI).
- """
- tools_field = fm.get("REQUIRES_TOOLS")
- if not tools_field:
- return True, []
- tools = [t.strip() for t in tools_field.split(",") if t.strip()]
- missing = [t for t in tools if shutil.which(t) is None]
- return len(missing) == 0, missing
-
-
-def main() -> int:
- parser = argparse.ArgumentParser(description="Receive and decide on a cross-agent message.")
- parser.add_argument("message_file", type=Path)
- parser.add_argument("--no-verify", action="store_true", help="Skip GPG verification (testing only)")
- parser.add_argument("--no-dedup", action="store_true", help="Skip SHA-256 dedup against existing files")
- parser.add_argument("--protocol-version", default=EXPECTED_PROTOCOL_VERSION,
- help="Override expected protocol version (default: 5)")
- parser.add_argument("--json", action="store_true", help="Emit JSON output")
- parser.add_argument("--compact-json", action="store_true", help="Compact JSON (no indent)")
- args = parser.parse_args()
-
- check_halt()
-
- if not args.message_file.is_file():
- err(f"message file not found: {args.message_file}")
- return EXIT_FOR_DECISION[DEC_REJECT]
-
- fm = parse_frontmatter(args.message_file)
- if "_parse_error" in fm:
- return emit_decision(DEC_REJECT, fm["_parse_error"], {}, None, args)
-
- # Step 1: frontmatter sanity-check.
- missing = [k for k in REQUIRED_FRONTMATTER if k not in fm]
- if missing:
- return emit_decision(
- DEC_REJECT, f"frontmatter missing required fields: {', '.join(missing)}", fm, None, args
- )
- if fm["MESSAGE_TYPE"] not in VALID_MESSAGE_TYPES:
- return emit_decision(
- DEC_REJECT, f"invalid MESSAGE_TYPE: {fm['MESSAGE_TYPE']!r}", fm, None, args
- )
-
- # Step 2: PROTOCOL_VERSION check.
- if fm["PROTOCOL_VERSION"] != args.protocol_version:
- return emit_decision(
- DEC_QUERY,
- f"PROTOCOL_VERSION mismatch: expected {args.protocol_version}, got {fm['PROTOCOL_VERSION']}",
- fm,
- None,
- args,
- )
-
- # Step 3: GPG verify.
- if not args.no_verify:
- sig_path = args.message_file.with_suffix(args.message_file.suffix + ".asc")
- if not sig_path.is_file():
- return emit_decision(DEC_REJECT, f"signature file missing: {sig_path.name}", fm, None, args)
- ok, gpg_err = gpg_verify(args.message_file, sig_path)
- if not ok:
- return emit_decision(DEC_REJECT, f"gpg verify failed: {gpg_err}", fm, None, args)
-
- # Step 4: SHA-256 dedup.
- my_hash = sha256_of(args.message_file)
- if not args.no_dedup:
- decision, reason = find_dedup_match(args.message_file, fm, my_hash)
- if decision == DEC_DEDUP:
- return emit_decision(DEC_DEDUP, reason, fm, my_hash, args)
-
- # Step 5: REQUIRES_TOOLS check.
- ok, missing_tools = check_requires_tools(fm)
- if not ok:
- return emit_decision(
- DEC_QUERY,
- f"required tools unavailable: {', '.join(missing_tools)}",
- fm,
- my_hash,
- args,
- )
-
- # Step 6: process.
- return emit_decision(DEC_PROCESS, None, fm, my_hash, args)
-
-
-if __name__ == "__main__":
- sys.exit(main())
diff --git a/.ai/scripts/cross-agent-comms/cross-agent-recv.md b/.ai/scripts/cross-agent-comms/cross-agent-recv.md
deleted file mode 100644
index 247a27a..0000000
--- a/.ai/scripts/cross-agent-comms/cross-agent-recv.md
+++ /dev/null
@@ -1,218 +0,0 @@
-# cross-agent-recv
-
-**Purpose.** The canonical receiver-side processor. Reads a single incoming
-message file and reports a structured decision the agent acts on:
-process / dedup / query / reject.
-
-The script handles only mechanical checks (frontmatter, signature, dedup,
-version, tools). Substance-level decisions like `pushback` ("I disagree with
-this request") happen one layer up — after the agent reads the message body
-the script returns as `process`-able.
-
-This is the read-side counterpart to `cross-agent-send`. Together they are the
-two halves of the per-message contract. The agent's polling loop calls
-`cross-agent-recv` on every new file in `inbox/from-agents/` and dispatches on
-the decision.
-
-Without this script, every receiver implementation re-invents GPG verify +
-frontmatter sanity-check + SHA-256 dedup. With it, behavior is consistent
-across projects.
-
-## Usage
-
-```
-cross-agent-recv <message-file>
-```
-
-Single positional argument: a `.org` file in `inbox/from-agents/`. The matching
-`.asc` signature file must be present alongside it.
-
-### Flags
-
-| Flag | Default | Purpose |
-|---|---|---|
-| `--no-verify` | (verify on) | Skip GPG verification. Testing only. |
-| `--no-dedup` | (dedup on) | Skip SHA-256 dedup against existing files. Testing only. |
-| `--protocol-version <N>` | 5 | Override the expected protocol version. Useful for testing forward-compatibility checks. |
-| `--json` | off | Output decision as JSON for easier parsing by the agent. |
-
-## Behavior
-
-Runs the receiver checks in order. First failure determines the decision.
-
-### Step 1 — Frontmatter sanity-check
-
-Parse the message's org-mode frontmatter. Required fields:
-
-- `#+TITLE`
-- `#+CONVERSATION_ID`
-- `#+MESSAGE_TYPE` (must be one of: `request`, `progress`, `query`, `pushback`,
- `complete`, `release`, `escalate`)
-- `#+SEQUENCE` (integer)
-- `#+TIMESTAMP` (ISO 8601 with explicit offset)
-- `#+PROTOCOL_VERSION` (must match the expected version; default 5)
-
-Any required field missing, malformed, or the protocol version mismatched →
-decision = `reject` (frontmatter) or `query` (version mismatch — see below).
-
-### Step 2 — Protocol-version check
-
-If `PROTOCOL_VERSION` doesn't match the expected:
-
-- Decision = `query`. Action: receiver should write a `query` reply asking the
- sender to upgrade to the expected protocol version.
-
-### Step 3 — Signature verification
-
-Look for `<message-file>.asc` alongside the `.org`. If missing or `gpg
---verify` fails:
-
-- Decision = `reject` (signature). Surface to user; do not act.
-
-The `.asc` file MUST be present when the `.org` is — `cross-agent-send`
-guarantees this with its strict ordering (`.asc` lands first). If the `.asc`
-is missing despite the `.org` being present, the sender violated atomic-write
-ordering or the file was tampered with in transit.
-
-### Step 4 — SHA-256 dedup
-
-Compute SHA-256 of the message file. Scan the same directory for existing
-files matching `CONVERSATION_ID + SEQUENCE`:
-
-- No match → decision = `process` (new message, dispatch by type).
-- Match with **identical** SHA-256 → decision = `dedup` (silent retry; do not
- reprocess).
-- Match with **different** SHA-256 → decision = `process` (sequence collision
- with non-identical content; both are legitimate, ordered by `#+TIMESTAMP`).
-
-### Step 5 — REQUIRES_TOOLS optional check
-
-If the message has a `#+REQUIRES_TOOLS` field, verify each named tool/MCP is
-available in the receiver's environment.
-
-- All available → `process`.
-- One or more missing → decision = `query`. The agent should write a `query`
- reply naming the missing tools, asking the sender to reframe the request to
- avoid them.
-
-### Step 6 — Dispatch decision
-
-If all checks pass, decision = `process` with the parsed `MESSAGE_TYPE` so the
-agent's main loop knows which handler to invoke.
-
-## Output
-
-### Default (human-readable)
-
-```
-$ cross-agent-recv inbox/from-agents/20260427T091015Z-from-homelab-prep-fixup.org
-decision: process
-message_type: request
-conversation_id: prep-fixup
-sequence: 6
-sha256: a1b2c3d4...
-```
-
-### `--json`
-
-```json
-{
- "decision": "process",
- "reason": null,
- "message_type": "request",
- "conversation_id": "prep-fixup",
- "sequence": 6,
- "timestamp": "2026-04-27T04:11:42-05:00",
- "sha256": "a1b2c3d4..."
-}
-```
-
-For decisions other than `process`, `reason` carries a human-readable
-explanation:
-
-```json
-{
- "decision": "query",
- "reason": "PROTOCOL_VERSION mismatch: expected 5, got 4",
- "conversation_id": "prep-fixup",
- "sequence": 6
-}
-```
-
-## Decision exit codes
-
-| Decision | Exit code | Agent action |
-|---|---|---|
-| `process` | 0 | Dispatch to the message-type handler |
-| `dedup` | 1 | Silent — do nothing further |
-| `query` | 2 | Write a `query` reply (see `reason` for what to ask) |
-| `reject` | 3 | Surface to user; do not auto-reply |
-
-The agent reads stdout/JSON to learn the decision; it can also key off exit
-code for simpler bash-style dispatching.
-
-## Failure modes
-
-| Symptom | Cause | Fix |
-|---|---|---|
-| `decision: reject (frontmatter)` | Required field missing or malformed | Open the message; fix or surface to user. The sender should not have produced this file. |
-| `decision: reject (signature)` | `.asc` missing, GPG verify failed, or signer unknown | Check that `.asc` exists alongside `.org`. If yes, run `gpg --verify <msg>.asc <msg>` manually for diagnostic output. |
-| `decision: query (PROTOCOL_VERSION)` | Sender on older/newer protocol | Reply with a `query` asking sender to upgrade. Both sides should align before continuing. |
-| `decision: query (REQUIRES_TOOLS)` | Receiver lacks one of the named tools | Reply with a `query` naming the missing tools; sender should reframe to avoid. |
-| `decision: dedup` | Already-processed identical retry | No action. The script handled it correctly. |
-
-## HALT awareness
-
-Checks `~/.config/cross-agent-comms/HALT` at the start of every invocation. If
-HALT exists, exits with code 5 ("halt active; remove
-~/.config/cross-agent-comms/HALT to resume") without verifying, deduping, or
-returning a decision.
-
-**The inbound file is left in place** — not moved, not rejected, not
-deduped. When HALT clears and polling resumes, the file gets picked up via
-the normal cold-start handling (whichever surfaces first: watcher
-notification, startup workflow check, or the next agent poll). Reversibility
-is preserved.
-
-If the HALT file exists but is unreadable, fail-closed — treat as if HALT is
-set.
-
-See `cross-agent-halt.md` for the full halt mechanism.
-
-## Examples
-
-```bash
-# Basic invocation in an agent's polling loop
-for msg in inbox/from-agents/*.org; do
- decision=$(cross-agent-recv --json "$msg")
- case "$(echo "$decision" | jq -r '.decision')" in
- process) handle_message "$msg" ;;
- dedup) ;; # silent
- query) write_query_reply "$msg" "$decision" ;;
- reject) surface_to_user "$msg" "$decision" ;;
- esac
-done
-
-# Test signature verification only
-cross-agent-recv --no-dedup inbox/from-agents/test-msg.org
-
-# Test against a future protocol version
-cross-agent-recv --protocol-version 6 inbox/from-agents/future-msg.org
-```
-
-## Performance
-
-The script is fast (single SHA-256 compute, single GPG verify, frontmatter
-parse). For typical messages (single-digit KB), runs in well under 100ms.
-Dedup-scan is O(N) over files in the directory; if a project's
-`inbox/from-agents/` accumulates hundreds of files, archive released
-conversations to keep the scan fast.
-
-## See also
-
-- `cross-agent-send` — counterpart writer.
-- `cross-agent-watch` — fires when a new message arrives; agent then calls
- `cross-agent-recv` to process it.
-- `cross-agent-status` — pending-message snapshot (uses similar
- released-vs-unreleased logic, but doesn't process individual messages).
-- `cross-agent-comms.org` — protocol spec, the "what" the script implements.
diff --git a/.ai/scripts/cross-agent-comms/cross-agent-resume b/.ai/scripts/cross-agent-comms/cross-agent-resume
deleted file mode 100755
index 1fb83bc..0000000
--- a/.ai/scripts/cross-agent-comms/cross-agent-resume
+++ /dev/null
@@ -1,145 +0,0 @@
-#!/usr/bin/env python3
-"""Resume cross-agent comms after a halt.
-
-See cross-agent-resume.md. Removes ~/.config/cross-agent-comms/HALT and
-restarts the cross-agent-watch systemd user service. With --tailnet,
-propagates the removal to every peer in peers.toml via SSH; reports
-per-peer status with non-zero exit on partial resume.
-
-Per the asymmetry rule: clearing HALT does NOT auto-resume agent polling.
-Each session must explicitly re-engage.
-"""
-
-from __future__ import annotations
-
-import argparse
-import subprocess
-import sys
-import tomllib
-from pathlib import Path
-
-CONFIG_DIR = Path.home() / ".config" / "cross-agent-comms"
-HALT_FILE = CONFIG_DIR / "HALT"
-PEERS_TOML = CONFIG_DIR / "peers.toml"
-
-EXIT_OK = 0
-EXIT_PARTIAL = 1
-
-
-def err(msg: str) -> None:
- print(msg, file=sys.stderr)
-
-
-def remove_halt_file() -> bool:
- """Returns True if HALT was removed, False if it didn't exist."""
- if HALT_FILE.exists():
- try:
- HALT_FILE.unlink()
- return True
- except OSError as e:
- err(f"could not remove HALT: {e}")
- return False
- return False
-
-
-def start_watcher_service() -> None:
- """Best-effort start of the systemd watcher path unit."""
- try:
- subprocess.run(
- ["systemctl", "--user", "start", "cross-agent-watch.path"],
- capture_output=True, text=True, timeout=5,
- )
- except (FileNotFoundError, subprocess.TimeoutExpired):
- pass
-
-
-def load_peers() -> dict:
- if not PEERS_TOML.exists():
- return {}
- try:
- return tomllib.loads(PEERS_TOML.read_text())
- except (tomllib.TOMLDecodeError, OSError) as e:
- err(f"cannot parse peers.toml: {e}")
- return {}
-
-
-def ssh_remove_halt(host: str, ssh_user: str | None) -> tuple[bool, str]:
- target = f"{ssh_user}@{host}" if ssh_user else host
- remote_cmd = "rm -f ~/.config/cross-agent-comms/HALT"
- try:
- result = subprocess.run(
- ["ssh", "-o", "ConnectTimeout=3", "-o", "BatchMode=yes", target, remote_cmd],
- capture_output=True, text=True, timeout=10,
- )
- except (FileNotFoundError, subprocess.TimeoutExpired):
- return False, "ssh unavailable or timed out"
- if result.returncode == 0:
- return True, "HALT cleared"
- return False, (result.stderr.strip().splitlines() or [f"exit {result.returncode}"])[-1]
-
-
-def print_re_engage_instructions() -> None:
- print()
- print("Halt cleared. Watcher restarted.")
- print()
- print("Agent polling does NOT auto-resume — per the failsafe asymmetry rule,")
- print("agents stay paused until you explicitly re-engage each session.")
- print("Open the relevant Claude session and tell the agent to resume polling")
- print("for its conversation.")
-
-
-def main() -> int:
- parser = argparse.ArgumentParser(description="Resume cross-agent comms after a halt.")
- parser.add_argument("--tailnet", action="store_true",
- help="Propagate HALT removal to every peer in peers.toml")
- args = parser.parse_args()
-
- removed = remove_halt_file()
- start_watcher_service()
- if removed:
- print("Resuming locally ✓ (HALT cleared)")
- else:
- print("Resuming locally ✓ (no HALT was active)")
-
- if not args.tailnet:
- print_re_engage_instructions()
- return EXIT_OK
-
- peers = load_peers().get("peers", {})
- if not peers:
- print()
- print("No peers configured in peers.toml — local-only resume complete.")
- print_re_engage_instructions()
- return EXIT_OK
-
- print()
- successes = 1
- failures = []
- for name, cfg in sorted(peers.items()):
- host = cfg.get("host", name)
- ssh_user = cfg.get("ssh_user")
- ok, detail = ssh_remove_halt(host, ssh_user)
- marker = "✓" if ok else "✗"
- print(f"Resuming {host:<27} {marker} ({detail})")
- if ok:
- successes += 1
- else:
- failures.append(f"{name} ({host}): {detail}")
-
- print()
- total = len(peers) + 1
- if failures:
- print(f"PARTIAL RESUME: {successes}/{total} machines cleared.")
- for f in failures:
- print(f" - {f}")
- print("Resolve the failures or manually clear HALT on each machine.")
- print_re_engage_instructions()
- return EXIT_PARTIAL
-
- print(f"Resume complete across {total} machine(s).")
- print_re_engage_instructions()
- return EXIT_OK
-
-
-if __name__ == "__main__":
- sys.exit(main())
diff --git a/.ai/scripts/cross-agent-comms/cross-agent-resume.md b/.ai/scripts/cross-agent-comms/cross-agent-resume.md
deleted file mode 100644
index 8aa8357..0000000
--- a/.ai/scripts/cross-agent-comms/cross-agent-resume.md
+++ /dev/null
@@ -1,117 +0,0 @@
-# cross-agent-resume
-
-**Purpose.** Clear the HALT file and restart the watcher service. Counterpart
-to `cross-agent-halt`. Resuming agent polling is **explicit per-session** —
-this script doesn't auto-revive halted polling loops; you tell each session
-to re-engage.
-
-## Usage
-
-```
-cross-agent-resume [--tailnet]
-```
-
-### Flags
-
-| Flag | Default | Purpose |
-|---|---|---|
-| `--tailnet` | local only | Clear HALT on every peer in `peers.toml` via SSH over Tailscale. |
-
-## Behavior
-
-### Local resume (default)
-
-1. Remove the HALT file: `rm -f ~/.config/cross-agent-comms/HALT`. (Use `-f`
- so a missing file isn't an error — running resume when not halted is safe.)
-2. Restart the watcher service: `systemctl --user start cross-agent-watch.path`.
-3. Print a summary:
- ```
- ✓ HALT file removed
- ✓ Watcher service started (cross-agent-watch.path)
- - cross-agent-send and cross-agent-recv will accept new operations.
- - Inbound messages held during halt will be picked up by the watcher.
- - Agent polling does NOT auto-resume. To re-engage polling in a paused
- session, open that Claude session and tell the agent to resume.
- ```
-4. Exit 0.
-
-### Cross-tailnet resume (`--tailnet`)
-
-1. Apply local resume steps 1-2 first.
-2. Read `peers.toml` for the list of remote machines.
-3. For each peer, SSH:
- ```
- ssh <user>@<host> "rm -f ~/.config/cross-agent-comms/HALT && \
- systemctl --user start cross-agent-watch.path"
- ```
-4. Track per-peer success/failure:
- ```
- Resuming velox.local ✓ (HALT cleared, watcher started)
- Resuming bastion.local ✗ (ssh exit 255: no route to host)
- Resuming locally ✓
-
- PARTIAL RESUME: 2/3 machines resumed. bastion.local still halted.
- ```
-5. Exit 0 if all peers resumed; exit 1 on any failure.
-
-## Why agent polling doesn't auto-resume
-
-Two reasons the asymmetry is deliberate:
-
-1. *Auto-resume could silently invert intentional kills.* If you halted
- because a session was misbehaving, removing HALT shouldn't quietly revive
- that session's polling. You re-engage explicitly so you're aware of which
- sessions came back online.
-
-2. *You may want to inspect before resuming.* After a halt, you might want to
- read pending messages, fix configuration, or kill a particular Claude
- session entirely. Per-session resume forces that pause.
-
-## Re-engaging polling in a Claude session
-
-After `cross-agent-resume`, open the relevant Claude session and say something
-like:
-
-```
-HALT is cleared; resume polling.
-```
-
-The agent will check the HALT file (now absent), re-create its polling
-schedule, and continue the in-flight conversation from wherever it left off.
-The conversation file is intact; the receiver will pick up any new messages
-that arrived during the halt window.
-
-## Failure modes
-
-| Symptom | Cause | Fix |
-|---|---|---|
-| HALT file doesn't exist | Already resumed (or never halted) | OK — `-f` makes this a no-op. |
-| `systemctl --user start` fails | Watcher service not installed | Install per `cross-agent-watch.md`'s systemd recipe. |
-| `--tailnet` resumes some peers but not others | Same as halt: peer unreachable | Per-peer status reported; resolve manually for unreachable peers. |
-| Permission denied removing HALT file | File owned by another user | Check ownership; HALT files should be owned by the running user. |
-
-## Examples
-
-```bash
-# Local resume after a halt
-cross-agent-resume
-
-# Resume all tailnet peers + local
-cross-agent-resume --tailnet
-```
-
-## Recovery flow
-
-After a halt:
-
-1. Investigate whatever caused the halt (runaway loop, bad config, etc.).
-2. Fix the underlying issue.
-3. Run `cross-agent-resume`.
-4. Open each Claude session that was polling and tell its agent to re-engage.
-5. Confirm operation with `cross-agent-status`.
-
-## See also
-
-- `cross-agent-halt` — counterpart that creates the HALT file.
-- `cross-agent-status` — verify HALT cleared and see pending messages.
-- `cross-agent-comms.org` — protocol spec, `* Halt mechanism` section.
diff --git a/.ai/scripts/cross-agent-comms/cross-agent-send b/.ai/scripts/cross-agent-comms/cross-agent-send
deleted file mode 100755
index 68c010a..0000000
--- a/.ai/scripts/cross-agent-comms/cross-agent-send
+++ /dev/null
@@ -1,356 +0,0 @@
-#!/usr/bin/env python3
-"""Cross-agent message sender.
-
-See cross-agent-send.md for the full contract. Briefly:
-
-- Destination as <machine>.<project>; resolved via peers.toml.
-- Same-machine: cp to receiver's inbox/from-agents/ with atomic rename.
-- Cross-machine: rsync over SSH (typically Tailscale) with retry+backoff.
-- GPG-signs by default; .asc renames before .org so receivers never see
- a .org without its sibling signature.
-- Generates the canonical filename; user's input filename is ignored.
-- Honors the HALT file: refuses to send and exits with code 5 when set.
-"""
-
-from __future__ import annotations
-
-import argparse
-import datetime as _dt
-import json
-import os
-import re
-import shutil
-import socket
-import subprocess
-import sys
-import tempfile
-import time
-import tomllib
-from pathlib import Path
-
-CONFIG_DIR = Path.home() / ".config" / "cross-agent-comms"
-PEERS_TOML = CONFIG_DIR / "peers.toml"
-HALT_FILE = CONFIG_DIR / "HALT"
-STATE_DIR = Path.home() / ".local" / "state" / "cross-agent-comms"
-FAILED_SENDS_DIR = STATE_DIR / "failed-sends"
-
-EXIT_OK = 0
-EXIT_GENERAL = 1
-EXIT_DEST_NOT_FOUND = 2
-EXIT_CROSS_MACHINE_FAILED = 3
-EXIT_FRONTMATTER = 4
-EXIT_HALT = 5
-
-REQUIRED_FRONTMATTER = ["CONVERSATION_ID", "MESSAGE_TYPE", "SEQUENCE", "TIMESTAMP", "PROTOCOL_VERSION"]
-VALID_MESSAGE_TYPES = {"request", "progress", "query", "pushback", "complete", "release", "escalate"}
-
-
-def err(msg: str) -> None:
- print(msg, file=sys.stderr)
-
-
-def check_halt() -> None:
- """Exit with code 5 if HALT file exists."""
- if HALT_FILE.exists():
- try:
- reason = HALT_FILE.read_text().strip()
- except OSError:
- # Fail-closed on unreadable HALT.
- err("halt active (HALT file present but unreadable; treated as halted)")
- err(f"remove {HALT_FILE} to resume")
- sys.exit(EXIT_HALT)
- msg = "halt active"
- if reason:
- msg += f": {reason}"
- err(msg)
- err(f"remove {HALT_FILE} to resume")
- sys.exit(EXIT_HALT)
-
-
-def parse_frontmatter(path: Path) -> dict[str, str]:
- """Extract org-mode #+KEY: value frontmatter from the top of the file."""
- try:
- text = path.read_text()
- except OSError as e:
- err(f"cannot read message file: {e}")
- sys.exit(EXIT_GENERAL)
-
- frontmatter: dict[str, str] = {}
- for line in text.splitlines():
- line = line.rstrip()
- if not line:
- # Blank line ends the frontmatter block.
- if frontmatter:
- break
- continue
- m = re.match(r"#\+([A-Z_]+):\s*(.*)", line)
- if m:
- frontmatter[m.group(1)] = m.group(2).strip()
- else:
- # First non-frontmatter line ends parsing.
- if frontmatter:
- break
- return frontmatter
-
-
-def validate_frontmatter(fm: dict[str, str]) -> None:
- missing = [k for k in REQUIRED_FRONTMATTER if k not in fm]
- if missing:
- err(f"frontmatter missing required fields: {', '.join(missing)}")
- sys.exit(EXIT_FRONTMATTER)
- if fm["MESSAGE_TYPE"] not in VALID_MESSAGE_TYPES:
- err(f"invalid MESSAGE_TYPE: {fm['MESSAGE_TYPE']!r}; expected one of {sorted(VALID_MESSAGE_TYPES)}")
- sys.exit(EXIT_FRONTMATTER)
- try:
- int(fm["SEQUENCE"])
- except ValueError:
- err(f"SEQUENCE must be an integer; got {fm['SEQUENCE']!r}")
- sys.exit(EXIT_FRONTMATTER)
-
-
-def load_peers() -> dict:
- if not PEERS_TOML.exists():
- return {}
- try:
- return tomllib.loads(PEERS_TOML.read_text())
- except (tomllib.TOMLDecodeError, OSError) as e:
- err(f"cannot read {PEERS_TOML}: {e}")
- sys.exit(EXIT_GENERAL)
-
-
-def resolve_destination(dest: str, peers: dict) -> tuple[str, str, str | None, str | None]:
- """Resolve <machine>.<project> to (machine, project, host, ssh_user).
-
- host is None for same-machine destinations.
- """
- if "." not in dest:
- err(f"destination must be <machine>.<project>; got {dest!r}")
- sys.exit(EXIT_DEST_NOT_FOUND)
- machine, project = dest.split(".", 1)
-
- local_hostname = socket.gethostname().split(".")[0]
- is_local = machine == local_hostname or machine == "local"
-
- host = None
- ssh_user = None
- if not is_local:
- peer_cfg = peers.get("peers", {}).get(machine)
- if peer_cfg is None:
- available = list(peers.get("peers", {}).keys())
- err(f"destination not found in peers.toml; available peers: {available or '(none)'}")
- sys.exit(EXIT_DEST_NOT_FOUND)
- host = peer_cfg.get("host", machine)
- ssh_user = peer_cfg.get("ssh_user", os.environ.get("USER"))
-
- return machine, project, host, ssh_user
-
-
-def resolve_inbox_path(project: str, peers: dict) -> str:
- """Inbox path on the receiver. Defaults to ~/projects/<project>/inbox/from-agents."""
- proj_cfg = peers.get("projects", {}).get(project)
- if proj_cfg and "inbox_path" in proj_cfg:
- return os.path.expanduser(proj_cfg["inbox_path"])
- return f"~/projects/{project}/inbox/from-agents"
-
-
-def derive_sender_project() -> str:
- """Walk up from CWD looking for ~/projects/<name>/.
-
- Returns the project name if found; falls back to the basename of CWD.
- """
- cwd = Path.cwd().resolve()
- projects_root = (Path.home() / "projects").resolve()
- try:
- rel = cwd.relative_to(projects_root)
- return rel.parts[0]
- except ValueError:
- return cwd.name
-
-
-def generate_canonical_filename(sender: str, conv_id: str) -> str:
- """YYYYMMDDTHHMMSSZ-from-<sender>-<conv-id>.org"""
- now = _dt.datetime.now(_dt.timezone.utc)
- timestamp = now.strftime("%Y%m%dT%H%M%SZ")
- return f"{timestamp}-from-{sender}-{conv_id}.org"
-
-
-def sign(message_path: Path, sig_path: Path, key: str | None) -> None:
- """gpg --detach-sign --armor --output <sig> [--local-user <key>] <message>"""
- cmd = ["gpg", "--detach-sign", "--armor", "--yes", "--output", str(sig_path)]
- if key:
- cmd.extend(["--local-user", key])
- cmd.append(str(message_path))
- try:
- result = subprocess.run(cmd, capture_output=True, text=True)
- except FileNotFoundError:
- err("gpg not found; install gnupg or use --no-sign for testing")
- sys.exit(EXIT_GENERAL)
- if result.returncode != 0:
- err(f"signing failed: {result.stderr.strip()}")
- sys.exit(EXIT_GENERAL)
-
-
-def same_machine_deliver(message_path: Path, sig_path: Path | None, target_dir: Path, canonical_name: str) -> None:
- """Atomic-write delivery: stage .asc, mv to final, then stage .org, mv to final."""
- target_dir.mkdir(parents=True, exist_ok=True)
- final_msg = target_dir / canonical_name
- final_sig = target_dir / f"{canonical_name}.asc"
-
- if sig_path is not None:
- # Stage .asc first, mv to final, THEN stage .org and mv to final.
- with tempfile.NamedTemporaryFile(
- mode="wb", dir=target_dir, prefix=f".tmp.{canonical_name}.asc.", delete=False
- ) as tmp:
- tmp.write(sig_path.read_bytes())
- tmp_sig_path = Path(tmp.name)
- os.replace(tmp_sig_path, final_sig)
-
- # Re-check HALT between .asc and .org per the layered-checks rule.
- check_halt()
-
- with tempfile.NamedTemporaryFile(
- mode="wb", dir=target_dir, prefix=f".tmp.{canonical_name}.", delete=False
- ) as tmp:
- tmp.write(message_path.read_bytes())
- tmp_msg_path = Path(tmp.name)
- os.replace(tmp_msg_path, final_msg)
-
-
-def cross_machine_deliver(
- message_path: Path,
- sig_path: Path | None,
- canonical_name: str,
- host: str,
- ssh_user: str,
- inbox_path: str,
- retries: int,
-) -> bool:
- """rsync push the .asc first (if signed), re-check HALT, then push the .org.
-
- Returns True on success, False on persistent failure (after retries).
- """
- # Stage local copies with the canonical name so rsync sets the right
- # destination filename.
- with tempfile.TemporaryDirectory(prefix="cross-agent-send-") as staging:
- staging_dir = Path(staging)
- local_msg = staging_dir / canonical_name
- local_msg.write_bytes(message_path.read_bytes())
- local_sig = None
- if sig_path is not None:
- local_sig = staging_dir / f"{canonical_name}.asc"
- local_sig.write_bytes(sig_path.read_bytes())
-
- backoffs = [5, 30, 120]
- # Step 1: push .asc first if signed.
- if local_sig is not None:
- if not _rsync_with_retries(local_sig, host, ssh_user, inbox_path, retries, backoffs):
- return False
-
- # Re-check HALT between .asc and .org per the layered-checks rule.
- check_halt()
-
- # Step 2: push .org.
- if not _rsync_with_retries(local_msg, host, ssh_user, inbox_path, retries, backoffs):
- return False
-
- return True
-
-
-def _rsync_with_retries(
- src: Path, host: str, ssh_user: str, inbox_path: str, retries: int, backoffs: list[int]
-) -> bool:
- target = f"{ssh_user}@{host}:{inbox_path}/"
- last_err = ""
- for attempt in range(retries + 1):
- if attempt > 0:
- check_halt()
- wait = backoffs[min(attempt - 1, len(backoffs) - 1)]
- err(f"rsync attempt {attempt} failed: {last_err}; retrying in {wait}s")
- time.sleep(wait)
- try:
- result = subprocess.run(
- ["rsync", "-a", str(src), target],
- capture_output=True,
- text=True,
- )
- except FileNotFoundError:
- err("rsync not found; install rsync")
- return False
- if result.returncode == 0:
- return True
- last_err = result.stderr.strip() or f"exit {result.returncode}"
- err(f"rsync failed after {retries + 1} attempts: {last_err}")
- return False
-
-
-def write_failed_send_marker(dest: str, message_path: Path, error: str, retry_log: list[str]) -> None:
- FAILED_SENDS_DIR.mkdir(parents=True, exist_ok=True)
- timestamp = _dt.datetime.now(_dt.timezone.utc).strftime("%Y%m%dT%H%M%SZ")
- safe_basename = re.sub(r"[^A-Za-z0-9._-]", "_", message_path.name)
- marker = FAILED_SENDS_DIR / f"{timestamp}-{dest.replace('.', '-')}-{safe_basename}.json"
- marker.write_text(json.dumps(
- {
- "timestamp": timestamp,
- "destination": dest,
- "message_path": str(message_path),
- "error": error,
- "retry_log": retry_log,
- },
- indent=2,
- ))
- err(f"marker written: {marker}")
-
-
-def main() -> int:
- parser = argparse.ArgumentParser(description="Send a cross-agent message.")
- parser.add_argument("destination", help="Destination as <machine>.<project>")
- parser.add_argument("message_file", type=Path, help="Path to the message body file")
- parser.add_argument("--no-sign", action="store_true", help="Skip GPG signing (testing only)")
- parser.add_argument("--retries", type=int, default=3, help="Retry count for cross-machine sends")
- parser.add_argument("--key", help="GPG key id to sign with (default: user's primary)")
- args = parser.parse_args()
-
- check_halt()
-
- if not args.message_file.is_file():
- err(f"message file not found: {args.message_file}")
- return EXIT_GENERAL
-
- fm = parse_frontmatter(args.message_file)
- validate_frontmatter(fm)
-
- peers = load_peers()
- machine, project, host, ssh_user = resolve_destination(args.destination, peers)
- inbox_path = resolve_inbox_path(project, peers)
-
- sender = derive_sender_project()
- canonical_name = generate_canonical_filename(sender, fm["CONVERSATION_ID"])
-
- sig_tmp = None
- if not args.no_sign:
- sig_tmp = args.message_file.with_suffix(args.message_file.suffix + ".asc.tmp")
- sign(args.message_file, sig_tmp, args.key)
-
- try:
- if host is None:
- # Same-machine delivery.
- target_dir = Path(os.path.expanduser(inbox_path))
- same_machine_deliver(args.message_file, sig_tmp, target_dir, canonical_name)
- print(f"sent: {target_dir}/{canonical_name}")
- return EXIT_OK
- else:
- ok = cross_machine_deliver(
- args.message_file, sig_tmp, canonical_name, host, ssh_user, inbox_path, args.retries
- )
- if ok:
- print(f"sent: {ssh_user}@{host}:{inbox_path}/{canonical_name}")
- return EXIT_OK
- write_failed_send_marker(args.destination, args.message_file, "rsync failed after retries", [])
- return EXIT_CROSS_MACHINE_FAILED
- finally:
- if sig_tmp is not None and sig_tmp.exists():
- sig_tmp.unlink()
-
-
-if __name__ == "__main__":
- sys.exit(main())
diff --git a/.ai/scripts/cross-agent-comms/cross-agent-send.md b/.ai/scripts/cross-agent-comms/cross-agent-send.md
deleted file mode 100644
index 29bfb24..0000000
--- a/.ai/scripts/cross-agent-comms/cross-agent-send.md
+++ /dev/null
@@ -1,199 +0,0 @@
-# cross-agent-send
-
-**Purpose.** Send a cross-agent message file to a specific destination. Handles
-peer-config lookup, GPG signing, atomic write (same-machine) or rsync push
-(cross-machine), retry-with-backoff, and failure surfacing.
-
-This is the canonical writer. The protocol spec defers all writer mechanics to
-this script.
-
-## Usage
-
-```
-cross-agent-send <destination> <message-file> [--no-sign] [--retries N]
-```
-
-### Positional arguments
-
-| Position | Meaning | Example |
-|---|---|---|
-| 1 | Destination as `<machine>.<project>` | `homelab.career`, `velox.career` |
-| 2 | Message file (already-formatted `.org`) | `/tmp/my-message.org` |
-
-### Flags
-
-| Flag | Default | Purpose |
-|---|---|---|
-| `--no-sign` | (signing on) | Skip GPG signing. Use only for testing; receivers reject unsigned messages by default. |
-| `--retries N` | 3 | Override retry count for cross-machine sends. |
-| `--key <key-id>` | (user's primary key) | GPG key to sign with. Resolution order: `--key` flag, `GPG_USER` env, `git config user.signingkey`, then the first secret key in the keyring. |
-
-## Behavior
-
-### Filename generation (script-controlled)
-
-The script generates the canonical destination filename from the message's
-frontmatter and sender context. The user's input filename is ignored — pass any
-path, the script names the destination correctly:
-
-```
-<UTC-now>T<HHMMSS>Z-from-<sender-slug>-<short-conv-id>.org
-```
-
-`<sender-slug>` comes from the sender machine's project name (config or
-hostname-based). `<short-conv-id>` is read from the message's
-`#+CONVERSATION_ID` frontmatter field. UTC timestamp is generated at send time.
-
-The script also performs the **sender-side max-seen scan** before writing: it
-reads the receiver's `from-agents/` directory, finds the highest existing
-sequence in this conversation across both sender prefixes, and (best-effort)
-suggests `max(seen) + 1` for the next sequence. The user/agent is responsible
-for setting `#+SEQUENCE` in the message body; the script only advises.
-
-### Same-machine destinations
-
-Resolved when the destination's machine matches the current hostname (or is
-not in `peers.toml` as a remote). Steps:
-
-1. Parse frontmatter; extract `CONVERSATION_ID` and `TIMESTAMP`. Validate per
- the *Validation before send* section below.
-2. Generate canonical filename per *Filename generation* above.
-3. Sign: `gpg --detach-sign --armor --output <canonical>.asc --local-user <key> <input>`.
-4. Compute target: read `peers.toml` for the project's `inbox_path`. If
- missing, fall back to `~/projects/<project>/inbox/from-agents/`.
-5. **Atomic write with strict ordering** (signature must precede message):
- - Stage `.asc`: write to `<target>/.tmp.XXXXXX-<canonical>.asc`,
- then `mv` to `<target>/<canonical>.asc`.
- - **Then** stage `.org`: write to `<target>/.tmp.XXXXXX-<canonical>`,
- then `mv` to `<target>/<canonical>`.
- - Receivers only act on `.org` files; staging the `.asc` first guarantees
- the signature is present when the receiver opens the message. Out-of-order
- would race: receiver could read the `.org` before the `.asc` lands and
- fail GPG verify even though the sender did everything right.
-6. Exit 0 on success. Exit non-zero if any step fails.
-
-### Cross-machine destinations
-
-Steps:
-
-1. Parse + generate canonical filename, as same-machine steps 1-2.
-2. Sign locally to `<input>.asc` (or a tmp staging file).
-3. rsync push **with the same .asc-first ordering**:
- - `rsync -a <input>.asc <ssh-user>@<host>:<inbox_path>/<canonical>.asc`
- - **Then** `rsync -a <input> <ssh-user>@<host>:<inbox_path>/<canonical>`
- rsync writes to a hidden temp file then renames atomically by default
- (`--inplace` would defeat this; do not pass it).
-4. Retry on failure: 5s, 30s, 120s backoff, then surface error.
-5. On persistent failure: write a marker file to
- `~/.local/state/cross-agent-comms/failed-sends/<timestamp>-<dest>-<canonical>.json`
- containing the destination, message path, error, and retry log. Exit non-zero.
-
-### Validation before send
-
-- Destination resolves via `peers.toml` (or local fallback). If neither, exit
- immediately with `destination not found in peers.toml; available: <list>`.
-- Message file must be readable, non-empty, and have valid org-mode frontmatter
- with **all** of the following required fields:
- - `#+TITLE`
- - `#+CONVERSATION_ID`
- - `#+MESSAGE_TYPE`
- - `#+SEQUENCE`
- - `#+TIMESTAMP`
- - `#+PROTOCOL_VERSION` (must equal `5` for v5)
-
- If any required field is missing or malformed, exit immediately with a parse
- error naming the offending field.
-
-- Optional fields the script recognizes and passes through (no special
- handling beyond preservation):
- - `#+REQUIRES_TOOLS` — comma-separated tool/MCP slugs the receiver needs.
- - `#+RELEASE_STATUS` — valid only on `MESSAGE_TYPE: release`. Values per
- spec: `complete`, `cancelled`, `withdrawn-after-pushback`,
- `abandoned-after-escalation`.
- - `#+WORKFLOW_VERSION` — sender's version of the cross-agent-comms workflow
- file. Currently advisory; receiver may warn on mismatch but does not block.
-
-## Configuration
-
-Reads `~/.config/cross-agent-comms/peers.toml` for peer routing:
-
-```toml
-[peers.velox]
-host = "velox.local"
-ssh_user = "cjennings"
-
-# Optional: per-project inbox-path overrides for non-default layouts.
-[projects.work]
-inbox_path = "~/projects/work/inbox/from-agents"
-
-[projects.homelab]
-inbox_path = "~/projects/homelab/inbox/from-agents"
-```
-
-If a project entry is omitted, defaults to `~/projects/<project>/inbox/from-agents`.
-
-## Failure modes
-
-| Symptom | Cause | Fix |
-|---|---|---|
-| `destination not found in peers.toml` | Misspelled destination, or peer not configured | Run `cross-agent-discover` to see available destinations. |
-| `signing failed: no secret key` | GPG key missing or not in keyring | `gpg --list-secret-keys` to confirm. Override with `--key <id>`. |
-| `signing failed: pinentry timed out` | Headless session, GUI pinentry unavailable | Confirm `pinentry-program` in `gpg-agent.conf` matches available pinentry. Per protocols.org, GUI pinentry works from Claude Code. |
-| `rsync exit 255` | SSH unreachable | `cross-agent-discover --peer <name>` to confirm reachability. |
-| `rsync exit 23` | Permission denied at destination | Check destination directory perms (`chmod 700`) and ownership. |
-| Marker file written to `failed-sends/` | Persistent cross-machine failure | Inspect the marker's `error` field. After fixing, retry: `cross-agent-send <dest> <msg>` (the marker is for visibility; it does not auto-retry). |
-| Receiver complains "unsigned message" | `--no-sign` was used in production | Don't use `--no-sign` outside testing. |
-
-## HALT awareness
-
-Checks `~/.config/cross-agent-comms/HALT` at the start of every send AND
-between the `.asc` and `.org` rsync calls AND between each retry iteration.
-On HALT exists, exits with code 5 ("halt active; remove
-~/.config/cross-agent-comms/HALT to resume") without writing or pushing
-further.
-
-Worst case: one in-flight send completes its current rsync step within a few
-seconds before halt kicks in for the next step. New sends are blocked
-immediately. No `pkill` needed — the per-iteration check stops things
-naturally.
-
-If the HALT file exists but is unreadable (permissions wrong), fail-closed —
-treat as if HALT is set. Safer than fail-open.
-
-See `cross-agent-halt.md` for the full halt mechanism.
-
-## Examples
-
-```bash
-# Same-machine send
-cross-agent-send homelab.career /tmp/my-message.org
-
-# Cross-machine send via Tailscale
-cross-agent-send velox.career /tmp/my-message.org
-
-# Test send without signing (receiver will reject)
-cross-agent-send homelab.career /tmp/test.org --no-sign
-
-# Override retry count for a flaky link
-cross-agent-send velox.career /tmp/my-message.org --retries 10
-
-# After a delivery failure, inspect the marker
-cat ~/.local/state/cross-agent-comms/failed-sends/*.json | jq .
-```
-
-## Exit codes
-
-| Code | Meaning |
-|---|---|
-| 0 | Sent successfully. |
-| 1 | General error (parse failure, signing failure, etc.). |
-| 2 | Destination not found in peers.toml. |
-| 3 | Cross-machine delivery failed after retries. Marker file written. |
-| 4 | Frontmatter validation failed. |
-
-## See also
-
-- `cross-agent-discover` — validate destinations before sending.
-- `cross-agent-watch` — receiver-side notification.
-- `cross-agent-status` — see what's queued.
-- `cross-agent-comms.org` — protocol spec, the "what" the script implements.
diff --git a/.ai/scripts/cross-agent-comms/cross-agent-status b/.ai/scripts/cross-agent-comms/cross-agent-status
deleted file mode 100755
index 4eee75b..0000000
--- a/.ai/scripts/cross-agent-comms/cross-agent-status
+++ /dev/null
@@ -1,185 +0,0 @@
-#!/usr/bin/env python3
-"""Point-in-time snapshot of pending cross-agent messages across local projects.
-
-See cross-agent-status.md. Pending = messages in inbox/from-agents/ whose
-CONVERSATION_ID has no MESSAGE_TYPE: release at a later #+TIMESTAMP.
-
-HALT: prints a prominent banner before normal output, but continues to enumerate.
-"""
-
-from __future__ import annotations
-
-import argparse
-import glob
-import json
-import os
-import re
-import sys
-from pathlib import Path
-
-CONFIG_DIR = Path.home() / ".config" / "cross-agent-comms"
-HALT_FILE = CONFIG_DIR / "HALT"
-DEFAULT_GLOB = str(Path.home() / "projects" / "*" / "inbox" / "from-agents") + "/"
-
-
-def parse_frontmatter(path: Path) -> dict[str, str]:
- try:
- text = path.read_text()
- except OSError:
- return {}
- fm: dict[str, str] = {}
- for line in text.splitlines():
- line = line.rstrip()
- if not line:
- if fm:
- break
- continue
- m = re.match(r"#\+([A-Z_]+):\s*(.*)", line)
- if m:
- fm[m.group(1)] = m.group(2).strip()
- elif fm:
- break
- return fm
-
-
-def project_name_from_path(path: str) -> str:
- """Walk up from path to find ~/projects/<name>/..."""
- home = str(Path.home())
- parts = Path(path).parts
- for i, part in enumerate(parts):
- if part == "projects" and i + 1 < len(parts) and str(Path(*parts[: i + 1])) == os.path.join(home, "projects"):
- return parts[i + 1]
- # Fallback: dir three levels up from the .org file (project/inbox/from-agents/file.org)
- return Path(path).parent.parent.parent.name
-
-
-def scan_project(inbox_dir: Path) -> tuple[int, str | None, int | None]:
- """Return (pending_count, most_recent_filename_or_None, most_recent_age_seconds_or_None)."""
- if not inbox_dir.is_dir():
- return 0, None, None
-
- # Group .org files by CONVERSATION_ID, also collect release timestamps per conv.
- org_files = sorted(inbox_dir.glob("*.org"))
- if not org_files:
- return 0, None, None
-
- by_conv: dict[str, list[tuple[str, str, Path]]] = {} # conv_id -> [(timestamp, msg_type, path)]
- for f in org_files:
- fm = parse_frontmatter(f)
- conv = fm.get("CONVERSATION_ID")
- ts = fm.get("TIMESTAMP")
- mt = fm.get("MESSAGE_TYPE")
- if not conv or not ts or not mt:
- # Malformed file: count as pending under conv "_unparseable".
- by_conv.setdefault("_unparseable", []).append(("", "request", f))
- continue
- by_conv.setdefault(conv, []).append((ts, mt, f))
-
- pending_files: list[Path] = []
- for conv, entries in by_conv.items():
- entries.sort(key=lambda e: e[0])
- # Find the latest release timestamp.
- release_ts = None
- for ts, mt, _f in entries:
- if mt == "release" and (release_ts is None or ts > release_ts):
- release_ts = ts
- for ts, mt, f in entries:
- if mt == "release":
- continue
- if release_ts is not None and ts <= release_ts:
- continue
- pending_files.append(f)
-
- if not pending_files:
- return 0, None, None
-
- # Most-recent by mtime (proxy for arrival order).
- most_recent = max(pending_files, key=lambda p: p.stat().st_mtime)
- import time
- age = int(time.time() - most_recent.stat().st_mtime)
- return len(pending_files), most_recent.name, age
-
-
-def fmt_age(seconds: int | None) -> str:
- if seconds is None:
- return "—"
- if seconds < 60:
- return f"{seconds}s ago"
- if seconds < 3600:
- return f"{seconds // 60} min ago"
- if seconds < 86400:
- return f"{seconds // 3600} hr ago"
- return f"{seconds // 86400} day(s) ago"
-
-
-def render_banner_if_halt() -> None:
- if not HALT_FILE.exists():
- return
- try:
- reason = HALT_FILE.read_text().strip()
- except OSError:
- reason = "(HALT file unreadable; treated as halted)"
- print("⚠ HALT ACTIVE — cross-agent comms paused")
- if reason:
- print(f" reason: {reason}")
- print(f" clear: rm {HALT_FILE} (or: cross-agent-resume)")
- print()
-
-
-def main() -> int:
- parser = argparse.ArgumentParser(description="Snapshot of pending cross-agent messages across local projects.")
- parser.add_argument("--json", action="store_true", help="Emit JSON output")
- parser.add_argument("--projects-glob", default=DEFAULT_GLOB,
- help=f"Glob for project from-agents dirs (default: {DEFAULT_GLOB})")
- args = parser.parse_args()
-
- render_banner_if_halt()
-
- matched = sorted(glob.glob(args.projects_glob))
- rows = []
- for path in matched:
- inbox = Path(path)
- if not inbox.is_dir():
- continue
- proj = project_name_from_path(path)
- count, most_recent, age = scan_project(inbox)
- rows.append({
- "name": proj,
- "pending_count": count,
- "most_recent": (
- {"filename": most_recent, "age_seconds": age}
- if most_recent else None
- ),
- })
-
- # Sort: pending-first, then alphabetical by name.
- rows.sort(key=lambda r: (-r["pending_count"], r["name"]))
-
- if args.json:
- import datetime as _dt
- payload = {
- "scanned_at": _dt.datetime.now(_dt.timezone.utc).isoformat(),
- "halt_active": HALT_FILE.exists(),
- "projects": rows,
- }
- print(json.dumps(payload, indent=2))
- return 0
-
- if not rows:
- print("No projects with inbox/from-agents/ found — 0 pending.")
- return 0
-
- # Human-readable table.
- name_w = max(len("project"), max(len(r["name"]) for r in rows))
- print(f"{'project':<{name_w}} pending most-recent")
- for r in rows:
- most_recent_str = "—"
- if r["most_recent"]:
- most_recent_str = f"{r['most_recent']['filename']} ({fmt_age(r['most_recent']['age_seconds'])})"
- print(f"{r['name']:<{name_w}} {r['pending_count']:<7} {most_recent_str}")
-
- return 0
-
-
-if __name__ == "__main__":
- sys.exit(main())
diff --git a/.ai/scripts/cross-agent-comms/cross-agent-status.md b/.ai/scripts/cross-agent-comms/cross-agent-status.md
deleted file mode 100644
index 070330c..0000000
--- a/.ai/scripts/cross-agent-comms/cross-agent-status.md
+++ /dev/null
@@ -1,139 +0,0 @@
-# cross-agent-status
-
-**Purpose.** Point-in-time snapshot of pending cross-agent messages across
-every project on this machine. Run from any terminal. No daemon required.
-
-This is the user-pull layer of the cold-start story — `cross-agent-watch`
-pushes notifications, `cross-agent-status` lets the user query.
-
-## Usage
-
-```
-cross-agent-status [--json] [--projects-glob <glob>]
-```
-
-No args required.
-
-### Flags
-
-| Flag | Default | Purpose |
-|---|---|---|
-| `--json` | off (table) | Output as JSON for scripting. |
-| `--projects-glob <glob>` | `~/projects/*/inbox/from-agents/` | Override which directories to scan. |
-
-## Output
-
-### Default (table)
-
-```
-$ cross-agent-status
-project pending most-recent
-career 0 —
-claude-templates 0 —
-clipper 0 —
-homelab 1 20260427T085611Z-from-career-question.org (3 min ago)
-finances 0 —
-... (other 9 projects)
-```
-
-Sort: pending-first, then alphabetical.
-
-### `--json`
-
-```json
-{
- "scanned_at": "2026-04-27T04:13:00-05:00",
- "projects": [
- {
- "name": "homelab",
- "pending_count": 1,
- "most_recent": {
- "filename": "20260427T085611Z-from-career-question.org",
- "age_seconds": 180
- }
- },
- ...
- ]
-}
-```
-
-## Pending semantics
-
-A message is "pending" if it sits in `inbox/from-agents/` AND no
-`MESSAGE_TYPE: release` exists for the same `CONVERSATION_ID` after it.
-
-Concretely:
-
-1. Scan each project's `inbox/from-agents/` for `.org` files.
-2. Group by `CONVERSATION_ID` from frontmatter.
-3. For each conversation, find the highest-`#+TIMESTAMP` message with
- `MESSAGE_TYPE: release`.
-4. Messages with `#+TIMESTAMP` after that release (or in conversations with no
- release) count as pending.
-
-Files without parseable frontmatter are counted as pending and noted in the
-output (single warning row per project).
-
-## Failure modes
-
-| Symptom | Likely cause | Fix |
-|---|---|---|
-| Project missing from output | Project's `.ai/` directory exists but `inbox/from-agents/` does not | Created lazily on first cross-agent message; `mkdir -p` to surface in output. |
-| All projects show "0 pending" but you know one has messages | Glob misresolved, OR all messages are post-release | `cross-agent-status --projects-glob` with explicit path to confirm. |
-| Warning row "N files unparseable in <project>" | Message file has invalid frontmatter | Open the file, fix or move out. |
-
-## Performance
-
-Scans every `.org` file in every watched directory. For Craig's setup (14
-projects, single-digit messages each), runs in <100ms. If a project
-accumulates hundreds of post-release messages, archive them per the persistence
-guidance in the protocol spec.
-
-## HALT awareness
-
-Checks `~/.config/cross-agent-comms/HALT` at start. If HALT exists, prints a
-prominent banner before normal output:
-
-```
-$ cross-agent-status
-⚠ HALT ACTIVE — cross-agent comms paused
- Reason: investigating runaway poll loop, 2026-04-27
- HALT file: ~/.config/cross-agent-comms/HALT
- Resume with: cross-agent-resume
-
-(snapshot continues normally — HALT does not suppress visibility)
-
-project pending most-recent
-career 0 —
-homelab 1 20260427T085611Z-from-career-question.org (3 min ago)
-...
-```
-
-Status is read-only, so it always runs. The banner ensures the user can't
-miss that halt is active when checking inbox state. Reason text comes from
-the HALT file's body; if empty, omit the reason line.
-
-If the HALT file exists but is unreadable, print a warning banner ("HALT
-file present but unreadable; treat as halted") and continue with normal
-output.
-
-See `cross-agent-halt.md` for the full halt mechanism.
-
-## Examples
-
-```bash
-# Snapshot
-cross-agent-status
-
-# JSON for piping
-cross-agent-status --json | jq '.projects[] | select(.pending_count > 0)'
-
-# Single-project query
-cross-agent-status --projects-glob ~/projects/work/inbox/from-agents/
-```
-
-## See also
-
-- `cross-agent-watch` — push notifications on new arrivals.
-- `cross-agent-discover` — enumerate available agents (cross-machine).
-- `cross-agent-comms.org` — protocol spec.
diff --git a/.ai/scripts/cross-agent-comms/cross-agent-watch b/.ai/scripts/cross-agent-comms/cross-agent-watch
deleted file mode 100755
index f50ba26..0000000
--- a/.ai/scripts/cross-agent-comms/cross-agent-watch
+++ /dev/null
@@ -1,106 +0,0 @@
-#!/usr/bin/env bash
-# cross-agent-watch — desktop-notify on new cross-agent messages.
-#
-# See cross-agent-watch.md. Watches every ~/projects/*/inbox/from-agents/ by
-# default. inotifywait fires create + moved_to events; .tmp.* files are
-# filtered out. HALT suppresses notifications but the watcher keeps running
-# and logs each event with "(suppressed by HALT)".
-
-set -uo pipefail
-
-# Defaults.
-PROJECTS_GLOB="${HOME}/projects/*/inbox/from-agents/"
-LOG_FILE="${HOME}/.local/state/cross-agent-comms/watch.log"
-HALT_FILE="${HOME}/.config/cross-agent-comms/HALT"
-QUIET=0
-NO_NOTIFY=0
-
-# Arg parsing.
-while [[ $# -gt 0 ]]; do
- case "$1" in
- --projects-glob)
- PROJECTS_GLOB="$2"; shift 2 ;;
- --log)
- LOG_FILE="$2"; shift 2 ;;
- --quiet)
- QUIET=1; shift ;;
- --no-notify)
- NO_NOTIFY=1; shift ;;
- -h|--help)
- cat <<EOF
-Usage: cross-agent-watch [--projects-glob GLOB] [--log PATH] [--quiet] [--no-notify]
-
-Watches inbox/from-agents/ directories for new cross-agent messages and fires
-desktop notifications. See cross-agent-watch.md for details.
-EOF
- exit 0 ;;
- *)
- echo "unknown flag: $1" >&2; exit 1 ;;
- esac
-done
-
-# Resolve glob to a concrete list of directories.
-# shellcheck disable=SC2086
-DIRS=( $PROJECTS_GLOB )
-# Filter out non-existent paths (glob may include literal pattern when no match).
-EXISTING=()
-for d in "${DIRS[@]}"; do
- if [[ -d "$d" ]]; then
- EXISTING+=( "$d" )
- fi
-done
-
-if [[ ${#EXISTING[@]} -eq 0 ]]; then
- echo "cross-agent-watch: glob resolved 0 directories: $PROJECTS_GLOB" >&2
- exit 1
-fi
-
-# Ensure log dir exists.
-mkdir -p "$(dirname "$LOG_FILE")"
-
-[[ $QUIET -eq 0 ]] && echo "cross-agent-watch: watching ${#EXISTING[@]} dir(s); log: $LOG_FILE"
-
-# Helper: project name from path like /home/.../projects/<name>/inbox/from-agents/...
-project_name() {
- local path="$1"
- # Match ~/projects/<name>/...
- if [[ "$path" =~ ${HOME}/projects/([^/]+)/ ]]; then
- echo "${BASH_REMATCH[1]}"
- else
- basename "$(dirname "$(dirname "$path")")"
- fi
-}
-
-# Main loop. inotifywait emits one line per event in the format
-# "<full-path>" because we passed --format '%w%f'.
-inotifywait -m -e create,moved_to --format '%w%f' "${EXISTING[@]}" 2>/dev/null \
- | while IFS= read -r path; do
- filename="$(basename "$path")"
-
- # Filter .tmp.* staging files.
- case "$filename" in
- .tmp.*) continue ;;
- esac
-
- # Filter .asc sidecars — they land first per the atomic-write ordering;
- # the .org event will fire after.
- case "$filename" in
- *.asc) continue ;;
- esac
-
- proj="$(project_name "$path")"
- iso="$(date -u "+%Y-%m-%dT%H:%M:%SZ")"
-
- if [[ -e "$HALT_FILE" ]]; then
- printf '%s\t%s\t%s\t(suppressed by HALT)\n' "$iso" "$proj" "$filename" >> "$LOG_FILE"
- [[ $QUIET -eq 0 ]] && echo "[$iso] $proj: $filename (suppressed by HALT)"
- continue
- fi
-
- printf '%s\t%s\t%s\n' "$iso" "$proj" "$filename" >> "$LOG_FILE"
- [[ $QUIET -eq 0 ]] && echo "[$iso] $proj: $filename"
-
- if [[ $NO_NOTIFY -eq 0 ]]; then
- notify info "Cross-agent message" "${proj}: ${filename}" --persist 2>/dev/null || true
- fi
- done
diff --git a/.ai/scripts/cross-agent-comms/cross-agent-watch.md b/.ai/scripts/cross-agent-comms/cross-agent-watch.md
deleted file mode 100644
index 04e8005..0000000
--- a/.ai/scripts/cross-agent-comms/cross-agent-watch.md
+++ /dev/null
@@ -1,130 +0,0 @@
-# cross-agent-watch
-
-**Purpose.** Long-running watcher that fires desktop notifications when new
-cross-agent messages land in any project's `inbox/from-agents/` directory.
-This is the primary cold-start mechanism: messages get noticed even when no
-Claude session is active.
-
-## Usage
-
-```
-cross-agent-watch [--projects-glob <glob>] [--log <path>]
-```
-
-No args required. Defaults:
-
-- Watches `~/projects/*/inbox/from-agents/` (matches every project with the
- cross-agent-comms convention).
-- Logs each event to `~/.local/state/cross-agent-comms/watch.log`.
-
-### Flags
-
-| Flag | Default | Purpose |
-|---|---|---|
-| `--projects-glob <glob>` | `~/projects/*/inbox/from-agents/` | Override which directories to watch. Useful for testing on a single project. |
-| `--log <path>` | `~/.local/state/cross-agent-comms/watch.log` | Override log location. Set to `/dev/null` to disable logging. |
-| `--quiet` | off | Suppress stdout output. Notifications still fire. |
-| `--no-notify` | off | Skip `notify` calls. Useful for testing the watcher loop without spamming notifications. |
-
-## Behavior
-
-1. Resolves the projects-glob to a concrete list of directories at startup.
- New projects added to `~/projects/` after startup are NOT picked up — restart
- the watcher to re-resolve.
-2. Runs `inotifywait -m -e create,moved_to --format '%w%f'` against each
- watched directory.
-3. For each event, calls
- `notify info "Cross-agent message" "<project>: <filename>" --persist`. The
- `--persist` flag keeps the page on screen until dismissed, so an inbound
- message that arrives while Craig is away from the desk isn't missed.
-4. Appends an event line to the log:
- `<ISO-8601-timestamp>\t<project>\t<filename>`.
-
-## Event filtering
-
-- Watches `create` AND `moved_to` events. The `moved_to` part is critical for
- the atomic-write convention (`mktemp` + `mv` produces a `moved_to`, not a
- `create`).
-- Files starting with `.tmp.` are ignored — they're staging files from
- in-progress writes that should never produce a notification.
-
-## Installation
-
-### Option A — tmux pane (personal, easy)
-
-Run in a tmux pane that survives session disconnects:
-
-```
-tmux new -d -s cross-agent-watch 'cross-agent-watch'
-```
-
-### Option B — systemd user service (production)
-
-Provided files:
-
-- `~/.config/systemd/user/cross-agent-watch.service`
-- `~/.config/systemd/user/cross-agent-watch.path`
-
-Enable with:
-
-```
-systemctl --user enable --now cross-agent-watch.path
-```
-
-The path unit triggers the service unit on filesystem changes; the service
-unit re-execs `cross-agent-watch` if it dies. Survives reboot.
-
-## Failure modes
-
-| Symptom | Likely cause | Fix |
-|---|---|---|
-| No notifications fire on new files | inotifywait not running, or glob resolved to zero dirs | Check `cross-agent-watch --projects-glob ... --quiet` exits non-zero immediately. Log shows `"resolved 0 directories"`. |
-| Notifications fire on `.tmp.` files | Filter regression | Verify `inotifywait` events show the `.tmp.` files; if so check this script's filter logic. |
-| Some files missed under rapid bursts | inotify queue overflow | Increase `fs.inotify.max_queued_events` sysctl. Default 16384 is usually fine. |
-| Permission denied on a watched dir | Directory perms wrong | `chmod 700 <dir>` and confirm owner. |
-
-## HALT awareness
-
-Checks `~/.config/cross-agent-comms/HALT` on each iteration (each inotifywait
-event fired). If HALT exists, the watcher continues running but **suppresses
-the `notify` call**. The event is still logged, with `(suppressed by HALT)`
-appended:
-
-```
-2026-04-27T04:42:00-05:00 career 20260427T094200Z-from-homelab-test.org (suppressed by HALT)
-```
-
-Logged-but-suppressed events are useful for the operator to see what would
-have fired during the halt window — helpful for diagnosing whatever caused
-the halt.
-
-When HALT clears, suppression stops; subsequent events fire normally. Backlog
-events that arrived during halt are NOT replayed — they get picked up via
-cold-start handling (status CLI, agent startup check, or the next agent
-poll once polling resumes).
-
-If the HALT file exists but is unreadable, fail-closed (suppress) — safer
-than fail-open.
-
-See `cross-agent-halt.md` for the full halt mechanism.
-
-## Examples
-
-```bash
-# Watch all projects, log everything, fire notifications
-cross-agent-watch
-
-# Test against a single project, no notifications, verbose
-cross-agent-watch \
- --projects-glob "$HOME/projects/work/inbox/from-agents/" \
- --no-notify
-
-# Production-style: quiet stdout, log only
-cross-agent-watch --quiet
-```
-
-## See also
-
-- `cross-agent-status` — point-in-time snapshot of pending messages.
-- `cross-agent-send` — counterpart writer.
-- `cross-agent-comms.org` — protocol spec.
diff --git a/.ai/scripts/flashcard-stats.py b/.ai/scripts/flashcard-stats.py
index 1fa5afb..cb580ac 100755
--- a/.ai/scripts/flashcard-stats.py
+++ b/.ai/scripts/flashcard-stats.py
@@ -35,7 +35,12 @@ import re
import sys
from pathlib import Path
-CARD_RE = re.compile(r"^\*\*\s+(.+?)\s+:drill:\s*$")
+# A card is a level-2 heading whose trailing org tag block includes `drill`.
+# Group 1 is the front, group 2 the tag block — so a curated card multi-tagged
+# :fundamental:drill: still counts (it would silently drop under a :drill:$
+# anchor, undercounting the deck). HEADING_RE bounds a card's body.
+CARD_RE = re.compile(r"^\*\*\s+(.+?)\s+(:[A-Za-z0-9_@#%:]+:)\s*$")
+HEADING_RE = re.compile(r"^\*{1,2}\s")
ANSWER_RE = re.compile(r"^\*\*\*\s+Answer\b")
PROP_START_RE = re.compile(r"^\s*:PROPERTIES:\s*$")
PROP_END_RE = re.compile(r"^\s*:END:\s*$")
@@ -177,7 +182,8 @@ def parse_cards(lines: list[str]) -> tuple[list[dict], int]:
n = len(lines)
while i < n:
m = CARD_RE.match(lines[i])
- if not m:
+ tags = [t for t in m.group(2).split(":") if t] if m else []
+ if not (m and "drill" in tags):
i += 1
continue
heading = m.group(1).strip()
@@ -188,7 +194,7 @@ def parse_cards(lines: list[str]) -> tuple[list[dict], int]:
body_lines: list[str] = []
while i < n:
line = lines[i]
- if line.startswith("* ") or CARD_RE.match(line):
+ if HEADING_RE.match(line):
break
if PROP_START_RE.match(line):
prop_count += 1
diff --git a/.ai/scripts/flashcard-to-anki.py b/.ai/scripts/flashcard-to-anki.py
index 7227683..e369fd8 100755
--- a/.ai/scripts/flashcard-to-anki.py
+++ b/.ai/scripts/flashcard-to-anki.py
@@ -10,12 +10,21 @@
Parses org-drill structure:
- Top-level "* Section" headings become tags on every card under them.
- Each "** Card name :drill:" entry becomes a card. Front = heading
- text (sans :drill: tag). Back = entry body with newlines converted
+ text (sans the tag block). Back = entry body with newlines converted
to <br>.
-
-Deck name defaults to the input basename, case preserved. Deck and model
-IDs are derived from the deck name via stable hash so re-importing the
-same deck updates existing cards instead of duplicating them.
+ - A card may carry a second org tag ("** Card :fundamental:drill:").
+ Any heading whose tag block includes `drill` is a card; the other
+ tags ride along as Anki tags next to the section tag, so a curated
+ subset stays grep-able in the source. --tag-filter <tag> emits only
+ the cards carrying that tag, and a subset deck built that way should
+ pass --guid-salt so its notes get their own GUID space (Anki dedupes
+ on GUID, so without it the subset imports empty against the full deck).
+
+Deck name defaults to the org #+TITLE: (so the phone deck reads as the
+curated title), falling back to the input basename when the source has
+no #+TITLE. Deck and model IDs are derived from the deck name via stable
+hash so re-importing the same deck updates existing cards instead of
+duplicating them.
Output defaults to ~/sync/phone/anki/<input-basename>.apkg. The .apkg is
a mobile-Anki artifact the phone picks up from its sync dir, so it lands
@@ -25,6 +34,8 @@ Usage:
flashcard-to-anki.py <input.org>
flashcard-to-anki.py <input.org> --deck "My Deck Name"
flashcard-to-anki.py <input.org> --output /path/to/deck.apkg
+ flashcard-to-anki.py <input.org> --tag-filter fundamental \
+ --deck "DeepSat Fundamentals" --guid-salt fundamentals
Requires genanki, which uv resolves automatically via the PEP 723
script metadata above. No venv or system install needed.
@@ -45,6 +56,15 @@ import genanki
ID_BASE = 1_500_000_000
ID_RANGE = 500_000_000
+# A card is any level-2 heading whose trailing org tag block includes `drill`.
+# Group 1 is the front text, group 2 the colon-delimited tag block (e.g.
+# ":fundamental:drill:") — so a curated subset can carry a second org tag
+# (:fundamental:) and stay grep-able in the source without dropping from the
+# full deck. HEADING_RE bounds a card's body at the next L1/L2 heading.
+CARD_RE = re.compile(r"^\*\*\s+(.+?)\s+(:[A-Za-z0-9_@#%:]+:)\s*$")
+HEADING_RE = re.compile(r"^\*{1,2}\s")
+SECTION_RE = re.compile(r"^\*\s+(.+?)\s*$")
+
def stable_id(name: str, salt: str) -> int:
"""Derive a deterministic 32-bit id from `name` and a `salt`.
@@ -118,33 +138,40 @@ def strip_org_metadata(body_lines: list[str]) -> list[str]:
return cleaned
-def parse(org_text: str) -> list[tuple[str, str, str]]:
- """Return [(front, back_html, tag), ...] for every :drill: card."""
- cards: list[tuple[str, str, str]] = []
- current_section: str | None = None
+def parse(
+ org_text: str, tag_filter: str | None = None
+) -> list[tuple[str, str, list[str]]]:
+ """Return [(front, back_html, anki_tags), ...] for every :drill: card.
- section_re = re.compile(r"^\*\s+(.+?)\s*$")
- card_re = re.compile(r"^\*\*\s+(.+?)\s+:drill:\s*$")
+ A card is any level-2 heading whose trailing org tag block includes
+ `drill`. Non-drill org tags on the heading (e.g. :fundamental:) ride
+ along as Anki tags next to the section tag, so a curated subset stays
+ grep-able in the source. When `tag_filter` is set, only cards carrying
+ that org tag are returned (the subset-deck path).
+ """
+ cards: list[tuple[str, str, list[str]]] = []
+ current_section: str | None = None
lines = org_text.splitlines()
i = 0
while i < len(lines):
line = lines[i]
- sec = section_re.match(line)
+ sec = SECTION_RE.match(line)
if sec:
current_section = sec.group(1).strip()
i += 1
continue
- card = card_re.match(line)
- if card:
- front = card.group(1).strip()
+ m = CARD_RE.match(line)
+ tags = [t for t in m.group(2).split(":") if t] if m else []
+ if m and "drill" in tags:
+ front = m.group(1).strip()
body_lines: list[str] = []
i += 1
while i < len(lines):
nxt = lines[i]
- if nxt.startswith("* ") or card_re.match(nxt):
+ if HEADING_RE.match(nxt):
break
body_lines.append(nxt)
i += 1
@@ -154,8 +181,17 @@ def parse(org_text: str) -> list[tuple[str, str, str]]:
while body_lines and not body_lines[-1].strip():
body_lines.pop()
back_html = "<br>".join(escape_html(ln) for ln in body_lines)
- tag = section_to_tag(current_section) if current_section else "drill"
- cards.append((front, back_html, tag))
+
+ org_tags = [t for t in tags if t != "drill"]
+ if tag_filter and tag_filter not in org_tags:
+ continue
+ anki_tags: list[str] = []
+ if current_section:
+ anki_tags.append(section_to_tag(current_section))
+ anki_tags.extend(org_tags)
+ if not anki_tags:
+ anki_tags = ["drill"]
+ cards.append((front, back_html, anki_tags))
continue
i += 1
@@ -163,21 +199,46 @@ def parse(org_text: str) -> list[tuple[str, str, str]]:
return cards
-def build(cards: list[tuple[str, str, str]], deck_name: str) -> genanki.Deck:
+def card_guid(front: str, guid_salt: str | None) -> str:
+ """GUID for a card's front. A salt gives a derived subset deck its own
+ GUID space so its notes don't collide with the full deck's (Anki dedupes
+ on GUID, which would otherwise import the subset empty). No salt is the
+ original behavior, so an unsalted deck's GUIDs and SRS state are untouched.
+ """
+ return genanki.guid_for(guid_salt, front) if guid_salt else genanki.guid_for(front)
+
+
+def build(
+ cards: list[tuple[str, str, list[str]]],
+ deck_name: str,
+ guid_salt: str | None = None,
+) -> genanki.Deck:
deck = genanki.Deck(stable_id(deck_name, "deck"), deck_name)
model = make_model(deck_name)
- for front, back, tag in cards:
+ for front, back, tags in cards:
note = genanki.Note(
model=model,
fields=[front, back],
- tags=[tag],
- guid=genanki.guid_for(front),
+ tags=tags,
+ guid=card_guid(front, guid_salt),
)
deck.add_note(note)
return deck
-def default_deck_name(input_path: Path) -> str:
+def default_deck_name(input_path: Path, org_text: str) -> str:
+ """Deck name defaults to the org #+TITLE:, falling back to the basename.
+
+ The #+TITLE drives both the org-drill display in Emacs and the Anki
+ deck name on the phone, so the consumed deck reads as the curated
+ title ("Refutations") rather than the filename slug
+ ("refutation-drill"). Falls back to the input basename (case
+ preserved) when the source has no non-empty #+TITLE line.
+ """
+ for line in org_text.splitlines():
+ m = re.match(r"^#\+TITLE:\s*(.*\S)\s*$", line, re.IGNORECASE)
+ if m:
+ return m.group(1).strip()
return input_path.stem
@@ -197,7 +258,7 @@ def main() -> int:
)
parser.add_argument(
"--deck",
- help="Deck name. Defaults to the input basename.",
+ help="Deck name. Defaults to the org #+TITLE, or the input basename.",
)
parser.add_argument(
"--output",
@@ -205,6 +266,16 @@ def main() -> int:
help="Output .apkg path. Defaults to "
"~/sync/phone/anki/<input-basename>.apkg.",
)
+ parser.add_argument(
+ "--tag-filter",
+ help="Emit only cards carrying this org tag (e.g. --tag-filter "
+ "fundamental for a curated subset deck).",
+ )
+ parser.add_argument(
+ "--guid-salt",
+ help="Salt note GUIDs so a subset deck gets its own GUID space and "
+ "imports non-empty without disturbing the full deck's SRS state.",
+ )
args = parser.parse_args()
input_path: Path = args.input.expanduser().resolve()
@@ -213,16 +284,22 @@ def main() -> int:
return 1
org_text = input_path.read_text(encoding="utf-8")
- deck_name = args.deck or default_deck_name(input_path)
+ deck_name = args.deck or default_deck_name(input_path, org_text)
output_path: Path = (args.output or default_output_path(input_path)).expanduser().resolve()
output_path.parent.mkdir(parents=True, exist_ok=True)
- cards = parse(org_text)
+ cards = parse(org_text, tag_filter=args.tag_filter)
if not cards:
- print(f"error: no :drill: cards found in {input_path}", file=sys.stderr)
+ if args.tag_filter:
+ print(
+ f"error: no :drill: cards tagged :{args.tag_filter}: in {input_path}",
+ file=sys.stderr,
+ )
+ else:
+ print(f"error: no :drill: cards found in {input_path}", file=sys.stderr)
return 1
- deck = build(cards, deck_name)
+ deck = build(cards, deck_name, guid_salt=args.guid_salt)
genanki.Package(deck).write_to_file(str(output_path))
print(f"wrote {output_path} ({len(cards)} cards, deck '{deck_name}')")
return 0
diff --git a/.ai/scripts/inbox-send.py b/.ai/scripts/inbox-send.py
index 5373bd4..663efcb 100755
--- a/.ai/scripts/inbox-send.py
+++ b/.ai/scripts/inbox-send.py
@@ -31,6 +31,7 @@ import os
import re
import shutil
import sys
+import tempfile
from datetime import datetime
from pathlib import Path
@@ -48,7 +49,7 @@ def resolve_roots() -> list[Path]:
config = Path.home() / ".claude" / "inbox-roots.txt"
if config.is_file():
paths: list[Path] = []
- for line in config.read_text().splitlines():
+ for line in config.read_text(encoding="utf-8").splitlines():
line = line.strip()
if line and not line.startswith("#"):
paths.append(Path(line).expanduser())
@@ -69,17 +70,28 @@ def discover_projects(roots: list[Path]) -> list[Path]:
a specific project root (included directly if it qualifies).
"""
projects: list[Path] = []
+ seen: set[Path] = set()
+
+ def _add(p: Path) -> None:
+ # Dedupe on the resolved path: a roots config naming both a parent and
+ # one of its children would otherwise list the child project twice, at
+ # two different indices.
+ key = p.resolve()
+ if key not in seen:
+ seen.add(key)
+ projects.append(p)
+
for root in roots:
if not root.is_dir():
continue
if _is_project(root):
- projects.append(root)
+ _add(root)
continue
for child in sorted(root.iterdir()):
if not child.is_dir():
continue
if _is_project(child):
- projects.append(child)
+ _add(child)
return projects
@@ -136,8 +148,21 @@ def slugify_filename(stem: str, max_length: int = MAX_SLUG_LENGTH) -> str:
return truncated.strip("-._")
+def display_name(path: Path) -> str:
+ """The name a project is referred to by — its basename with dots stripped.
+
+ Dotted directories (`.emacs.d`, `.dotfiles`) are awkward to name in
+ conversation, so they're addressed dot-stripped: `emacsd`, `dotfiles`.
+ """
+ return path.name.replace(".", "")
+
+
def find_target(target_name: str, projects: list[Path]) -> Path | None:
- """Resolve `target_name` against the project list (basename or numeric index)."""
+ """Resolve `target_name` against the project list (basename or numeric index).
+
+ An exact basename match wins. Failing that, a dot-stripped alias matches —
+ so `emacsd` resolves `.emacs.d` and `dotfiles` resolves `.dotfiles`.
+ """
if target_name.isdigit():
idx = int(target_name) - 1
if 0 <= idx < len(projects):
@@ -146,6 +171,10 @@ def find_target(target_name: str, projects: list[Path]) -> Path | None:
for p in projects:
if p.name == target_name:
return p
+ norm = target_name.replace(".", "")
+ for p in projects:
+ if display_name(p) == norm:
+ return p
return None
@@ -160,6 +189,56 @@ def build_text_org(message: str, source_name: str, timestamp: str) -> str:
)
+def uniquify(dest: Path) -> Path:
+ """Return dest, or dest with a -2/-3/... stem suffix when it already exists.
+
+ Two sends in the same minute whose text starts with the same phrase
+ derive identical filenames, and the second silently overwrote the
+ first (a message was lost this way, 2026-07-02). Never overwrite.
+ """
+ if not dest.exists():
+ return dest
+ n = 2
+ while True:
+ candidate = dest.with_name(f"{dest.stem}-{n}{dest.suffix}")
+ if not candidate.exists():
+ return candidate
+ n += 1
+
+
+def _atomic_write(dest: Path, writer) -> None:
+ """Write to a temp file in dest's directory, then rename it into place.
+
+ dest is another project's inbox/, and a direct write truncates the target
+ on open, so any mid-write failure (a full disk, an encoding error, an
+ interrupted process) leaves a zero-byte .org there. inbox-status counts
+ that phantom as a pending handoff and blocks a turn in the receiving
+ project over a file with no content and no sender (2026-07-23). Writing to
+ a temp sibling and os.replace-ing means the inbox only ever sees a complete
+ file. os.replace is atomic within one filesystem, and the temp sits in the
+ same directory as dest, so it is.
+
+ `writer` receives the open temp path and fills it. On any failure the temp
+ is removed and the error re-raised, so a caught error never leaves debris.
+ """
+ fd, tmp = tempfile.mkstemp(
+ dir=dest.parent, prefix=".inbox-send-", suffix=dest.suffix
+ )
+ os.close(fd)
+ tmp_path = Path(tmp)
+ # mkstemp creates the temp 0600; give the delivered file the umask-default
+ # mode the old direct write produced, so inbox files stay readable as before.
+ umask = os.umask(0)
+ os.umask(umask)
+ os.chmod(tmp_path, 0o666 & ~umask)
+ try:
+ writer(tmp_path)
+ os.replace(tmp_path, dest)
+ except BaseException:
+ tmp_path.unlink(missing_ok=True)
+ raise
+
+
def send_text(
target_inbox: Path,
message: str,
@@ -174,8 +253,9 @@ def send_text(
if not slug:
raise ValueError(f"could not derive a slug from text: {message!r}")
filename = f"{now.strftime(TS_FILENAME_FMT)}-from-{source_name}-{slug}.org"
- dest = target_inbox / filename
- dest.write_text(build_text_org(message, source_name, now.strftime(TS_DOC_FMT)))
+ dest = uniquify(target_inbox / filename)
+ body = build_text_org(message, source_name, now.strftime(TS_DOC_FMT))
+ _atomic_write(dest, lambda p: p.write_text(body, encoding="utf-8"))
return dest
@@ -194,8 +274,8 @@ def send_file(
raise ValueError(f"could not derive a slug from file: {src_path}")
ext = src_path.suffix
filename = f"{now.strftime(TS_FILENAME_FMT)}-from-{source_name}-{slug}{ext}"
- dest = target_inbox / filename
- shutil.copy2(src_path, dest)
+ dest = uniquify(target_inbox / filename)
+ _atomic_write(dest, lambda p: shutil.copyfile(src_path, p))
return dest
@@ -206,9 +286,9 @@ def print_project_list(projects: list[Path], current: Path | None) -> None:
print("No projects (.ai/ + inbox/) found under the configured roots.")
return
print(f"Available .ai projects ({len(others)}):")
- width = max(len(p.name) for p in others)
+ width = max(len(display_name(p)) for p in others)
for i, p in enumerate(others, 1):
- print(f" {i}. {p.name:<{width}} {p}")
+ print(f" {i}. {display_name(p):<{width}} {p}")
def main() -> int:
@@ -276,7 +356,10 @@ def main() -> int:
else:
assert args.file is not None
dest = send_file(target_inbox, args.file, source_name, args.name, now)
- except (ValueError, FileNotFoundError) as exc:
+ except (ValueError, OSError) as exc:
+ # OSError covers FileNotFoundError (missing source), PermissionError
+ # (unreadable source), and any atomic-write failure — all should
+ # surface as the clean "inbox-send: <message>" error, never a traceback.
print(f"inbox-send: {exc}", file=sys.stderr)
return 1
diff --git a/.ai/scripts/inbox-status b/.ai/scripts/inbox-status
index b917144..17031af 100755
--- a/.ai/scripts/inbox-status
+++ b/.ai/scripts/inbox-status
@@ -35,6 +35,7 @@ mapfile -t pending < <(find inbox -maxdepth 1 -type f \
! -name '.gitkeep' \
! -name 'lint-followups.org' \
! -name 'PROCESSED-*' \
+ ! -name '.inbox-send-*' \
-printf '%f\n' 2>/dev/null | sort)
n=${#pending[@]}
diff --git a/.ai/scripts/lint-org.el b/.ai/scripts/lint-org.el
index 8f55cc6..33dc52f 100644
--- a/.ai/scripts/lint-org.el
+++ b/.ai/scripts/lint-org.el
@@ -2,16 +2,19 @@
;;
;; Usage:
;; emacs --batch -q -l lint-org.el FILE.org [FILE.org ...]
+;; report only (the default) — categorize without modifying the file.
+;; A linter reports, it doesn't write; mutation requires --fix.
+;; --check is accepted as an explicit alias of this default.
+;;
+;; emacs --batch -q -l lint-org.el --fix FILE.org [FILE.org ...]
;; apply mechanical fixes in place, emit judgment items on stdout for the
;; command layer to walk
;;
-;; emacs --batch -q -l lint-org.el --check FILE.org [FILE.org ...]
-;; report only — categorize without modifying the file
-;;
-;; emacs --batch -q -l lint-org.el --followups-file=PATH FILE.org
+;; emacs --batch -q -l lint-org.el --fix --followups-file=PATH FILE.org
;; apply mechanical fixes; if any judgment items remain, append them to
;; PATH as an org section dated today. Used by wrap-it-up to defer the
;; judgment walk to the next morning's review without blocking the wrap.
+;; (--followups-file only writes in --fix mode.)
;;
;; Mechanical categories (auto-fixed):
;; item-number add [@N] directive to drifted bullets
@@ -29,6 +32,15 @@
;; link-to-local-file broken file: links
;; invalid-fuzzy-link broken *Heading refs
;; suspicious-language-in-src-block unknown source-block language
+;; org-table-standard table wider than budget / missing rules
+;; level-2-dated-header ** dated header instead of a keyword
+;; indented-heading whitespace before stars (demoted to body)
+;; empty-heading bare stars with no title
+;; malformed-priority-cookie [#x]-shaped token org rejected
+;; level2-done-without-closed completed level-2 task with no CLOSED
+;; task-missing-last-reviewed open level-2 task with no :LAST_REVIEWED:
+;; subtask-done-not-dated level-3+ done sub-task still a DONE keyword
+;; dated-log-heading-active-timestamp dated-log heading with a live SCHEDULED/DEADLINE
;; (anything else) surfaced as judgment with checker name
;;
;; Output format on stdout:
@@ -59,9 +71,23 @@
Each plist has :kind (mechanical-fixed | judgment), :line, :checker, :msg.
Mechanical entries from --check mode also carry :preview t.")
(defvar lo-check-only nil
- "Non-nil means run in report-only mode — no buffer writes.")
+ "Non-nil means run in report-only mode — no buffer writes.
+The CLI defaults this to t (a linter reports, it doesn't write);
+`--fix' is what enables writes on a command-line run.")
(defvar lo-current-file nil
"Path of the file currently being processed.")
+
+(defun lo--spec-file-p ()
+ "Non-nil when the current file lives under a docs/specs/ directory.
+The four todo-format-family checkers encode todo.org completion conventions
+and misfire on a spec: a spec's Decisions section legitimately carries a
+level-2 DONE with no CLOSED cookie, and its review-history section carries
+level-2 dated headings. docs/specs/ is the canonical spec home per the
+docs-lifecycle rule, so a path segment match is the scope test. Link,
+table, and structural checks still run on specs — only the todo-format
+family is scoped out."
+ (and lo-current-file
+ (string-match-p "/docs/specs/" (expand-file-name lo-current-file))))
(defvar lo-followups-file nil
"When non-nil, after a non-check run any judgment items are appended to this
path as an org section dated today. The file is created if missing.")
@@ -280,6 +306,52 @@ Craig-specific annotation marker rather than Babel src-block syntax."
(lo--goto-line line)
(looking-at-p "^[ \t]*#\\+begin_src[ \t]+cj:")))
+(defvar-local lo--matched-blocks-cache nil
+ "Cons of (TICK . REGIONS) memoizing `lo--matched-block-regions'.
+TICK is the `buffer-chars-modified-tick' the regions were computed at, so a
+fix applied mid-pass invalidates them.")
+
+(defun lo--matched-block-regions ()
+ "Return ((BEGIN-LINE . END-LINE) ...) for every correctly paired block.
+Scans lines directly rather than asking org, because org's own parser is what
+mis-reads these blocks: a heading-shaped line inside a verbatim body reads as a
+structural break and loses the open block. The scan applies org's real rule —
+once a block is open, only its own `#+end_TYPE' closes it, so a nested
+`#+begin_' or a foreign `#+end_' in the body is just text."
+ (let ((tick (buffer-chars-modified-tick)))
+ (if (eql (car lo--matched-blocks-cache) tick)
+ (cdr lo--matched-blocks-cache)
+ (let ((case-fold-search t)
+ (regions nil) (open-type nil) (open-line nil) (line 0))
+ (save-excursion
+ (goto-char (point-min))
+ (while (not (eobp))
+ (setq line (1+ line))
+ (let ((text (buffer-substring-no-properties
+ (line-beginning-position) (line-end-position))))
+ (cond
+ (open-type
+ (when (string-match
+ (format "\\`[ \t]*#\\+end_%s[ \t]*\\'"
+ (regexp-quote open-type))
+ text)
+ (push (cons open-line line) regions)
+ (setq open-type nil open-line nil)))
+ ((string-match "\\`[ \t]*#\\+begin_\\([^ \t\n]+\\)" text)
+ (setq open-type (match-string 1 text)
+ open-line line))))
+ (forward-line 1)))
+ (setq lo--matched-blocks-cache (cons tick (nreverse regions)))
+ (cdr lo--matched-blocks-cache)))))
+
+(defun lo--in-matched-block-p (line)
+ "Non-nil when LINE sits within a correctly paired block, delimiters included.
+org-lint reports `invalid-block' at the delimiter lines themselves, so the
+range has to be inclusive for the suppression to reach them."
+ (cl-some (lambda (region)
+ (and (>= line (car region)) (<= line (cdr region))))
+ (lo--matched-block-regions)))
+
(defun lo--handle-item (item)
(let ((name (lo--checker-name item))
(line (lo--line item))
@@ -292,6 +364,13 @@ Craig-specific annotation marker rather than Babel src-block syntax."
wrong-header-argument))
(lo--cj-comment-block-opener-p line))
nil)
+ ;; `invalid-block' on a block that is in fact correctly paired — the
+ ;; checker is org-lint's own, so this filters its output rather than
+ ;; fixing a local checker. A genuinely unterminated block isn't in any
+ ;; matched region, so it still reports.
+ ((and (eq name 'invalid-block)
+ (lo--in-matched-block-p line))
+ nil)
((eq name 'item-number)
(lo--apply-or-preview name line msg #'lo-fix-item-number))
((eq name 'missing-language-in-src-block)
@@ -348,24 +427,266 @@ logical row, matching wrap-org-table.el's grouping."
(defun lo--check-tables ()
"Scan the current buffer for org tables violating the table standard.
-Emits one judgment item per violating table."
+Emits one judgment item per violating table. Pipe-led lines inside
+#+begin_/#+end_ blocks are content (ASCII art, shell pipes), not tables,
+and are skipped — the same block rule `wot-process-file' applies."
(save-excursion
(goto-char (point-min))
- (while (re-search-forward "^[ \t]*|" nil t)
- (let ((start-line (line-number-at-pos))
- (lines nil))
- (beginning-of-line)
- (while (and (not (eobp)) (looking-at "[ \t]*|"))
- (push (buffer-substring-no-properties (line-beginning-position)
- (line-end-position))
- lines)
+ (let ((in-block nil)) ; the open block's type, e.g. "example" — nil outside
+ (while (not (eobp))
+ (cond
+ ;; Type-matched close only: literal #+end_src quoted inside an
+ ;; example block must not clear the flag (see wot-process-file).
+ ((and (not in-block)
+ (looking-at "^[ \t]*#\\+begin_\\([^ \t\n]+\\)"))
+ (setq in-block (downcase (match-string 1)))
+ (forward-line 1))
+ ((and in-block
+ (looking-at-p (format "^[ \t]*#\\+end_%s\\([ \t]\\|$\\)"
+ (regexp-quote in-block))))
+ (setq in-block nil)
(forward-line 1))
- (let ((violations (lo--table-violations (nreverse lines))))
- (when violations
- (lo--emit-judgment
- 'org-table-standard start-line
- (format "table violates the org-table standard: %s — wrap-org-table.el reflows it"
- (string-join violations "; ")))))))))
+ ((and (not in-block) (looking-at-p "^[ \t]*|"))
+ (let ((start-line (line-number-at-pos))
+ (lines nil))
+ (while (and (not (eobp)) (looking-at "[ \t]*|"))
+ (push (buffer-substring-no-properties (line-beginning-position)
+ (line-end-position))
+ lines)
+ (forward-line 1))
+ (let ((violations (lo--table-violations (nreverse lines))))
+ (when violations
+ (lo--emit-judgment
+ 'org-table-standard start-line
+ (format "table violates the org-table standard: %s — wrap-org-table.el reflows it"
+ (string-join violations "; ")))))))
+ (t (forward-line 1)))))))
+
+;;; ---------------------------------------------------------------------------
+;;; level-2 dated-header check (claude-rules/todo-format.md)
+;;
+;; A completed task or resolved VERIFY at level 2 must carry a terminal
+;; keyword (DONE/CANCELLED + CLOSED:), never a dated heading. A `** <date>'
+;; header has no keyword, so todo-cleanup's --archive-done can never archive
+;; it (it accumulates in Open Work forever) and task-review drops it from
+;; selection. Judgment-only, never auto-fixed: the repair needs a
+;; DONE-vs-CANCELLED call and the original heading text, which is a judgment
+;; the sweep can't make. Targets todo/task files; a dated-log-format org
+;; file using `** <date>' headings intentionally will false-positive here, in
+;; which case the human dismisses the judgment item.
+
+(defun lo--check-level2-dated-headers ()
+ "Flag level-2 headings whose text begins with a YYYY-MM-DD date.
+Emits one judgment item per offending heading (checker
+`level-2-dated-header')."
+ (save-excursion
+ (goto-char (point-min))
+ (while (re-search-forward
+ "^\\*\\* \\([0-9]\\{4\\}-[0-9]\\{2\\}-[0-9]\\{2\\}\\)" nil t)
+ (lo--emit-judgment
+ 'level-2-dated-header (line-number-at-pos)
+ "level-2 dated header is a completion defect (todo-format.md): a ** task or VERIFY closes with DONE/CANCELLED + CLOSED:, not a dated heading — convert it so --archive-done can archive it"))))
+
+;;; ---------------------------------------------------------------------------
+;;; structural heading checks (mistakes org-lint does not cover)
+;;
+;; org-lint validates links, drawers, blocks, and babel — but not heading
+;; well-formedness. These four catch hand-edit defects it misses, all
+;; judgment-only (each repair is a human call) and regex-based (no dependence on
+;; which TODO keywords the batch Emacs happens to recognize):
+;;
+;; indented-heading leading whitespace before two-or-more stars; org
+;; demotes it to body text, so the task vanishes from
+;; the agenda and never archives. The worst case — an
+;; invisible task — and silent. Single `*' is left
+;; alone (a valid indented plain-list bullet).
+;; empty-heading a line of bare stars with no title.
+;; malformed-priority-cookie a `[#x]'-shaped token org rejected (lowercase,
+;; multi-char, non-letter) sitting where a cookie
+;; would be.
+;; level2-done-without-closed a level-2 DONE/CANCELLED with no CLOSED line —
+;; directly relevant to todo-cleanup's aging step,
+;; which archives an undated completed task at once.
+
+(defconst lo-done-keywords '("DONE" "CANCELLED")
+ "Heading keywords treated as completed for `lo--check-level2-done-without-closed'.")
+
+(defun lo--check-indented-headings ()
+ "Flag lines that are whitespace + two-or-more stars + space outside any block.
+Org parses a heading only at column 0, so leading whitespace silently demotes a
+would-be heading to body text. Two-or-more stars is required: an indented
+single `*' is a valid plain-list bullet, not a lost heading, so flagging it
+false-positives on legitimate lists; `**'+ is never a bullet, so an indented one
+is unambiguously a demoted level-2+ heading turned invisible. Lines inside
+`#+begin_/#+end_' blocks are skipped — indented asterisks there are legitimate
+content."
+ (save-excursion
+ (goto-char (point-min))
+ (let ((in-block nil))
+ (while (not (eobp))
+ (cond
+ ((looking-at-p "^[ \t]*#\\+begin_") (setq in-block t))
+ ((looking-at-p "^[ \t]*#\\+end_") (setq in-block nil))
+ ((and (not in-block) (looking-at-p "^[ \t]+\\*\\*+[ \t]"))
+ (lo--emit-judgment
+ 'indented-heading (line-number-at-pos)
+ "indented heading: leading whitespace before the stars demotes this to body text — org won't treat it as a heading (it vanishes from the agenda and never archives); dedent to column 0")))
+ (forward-line 1)))))
+
+(defun lo--check-empty-headings ()
+ "Flag headings that are bare stars with no title text.
+A line of nothing but stars is an empty heading — a stray heading-star carrying
+no content."
+ (save-excursion
+ (goto-char (point-min))
+ (while (re-search-forward "^\\*+[ \t]*$" nil t)
+ (lo--emit-judgment
+ 'empty-heading (line-number-at-pos)
+ "empty heading: a line of stars with no title — delete it or give it a title"))))
+
+(defun lo--check-malformed-priority-cookies ()
+ "Flag a heading whose first cookie-shaped token is not a valid priority.
+A valid cookie is a single uppercase letter in `[#A]' form. Verbatim-wrapped
+cookies (`=[#D]=' quoted in a dated-log title) are skipped. Only the first
+token on the line is checked, so a real cookie earlier on the line means a
+later `[#x]' in the title is left alone."
+ (save-excursion
+ (goto-char (point-min))
+ ;; Case-sensitive: a cookie is uppercase only, and case-fold-search defaults
+ ;; to t (which would accept [#a] as valid).
+ (let ((case-fold-search nil))
+ (while (re-search-forward "^\\*+ " nil t)
+ (let ((eol (line-end-position)) (hline (line-number-at-pos)))
+ (when (re-search-forward "\\[#\\([^]]*\\)\\]" eol t)
+ (let ((inner (match-string 1))
+ (before (char-before (match-beginning 0)))
+ (after (char-after (match-end 0))))
+ (unless (or (eq before ?=) (eq after ?=)
+ (string-match-p "\\`[A-Z]\\'" inner))
+ (lo--emit-judgment
+ 'malformed-priority-cookie hline
+ (format "malformed priority cookie [#%s] — a cookie is a single uppercase letter ([#A]) right after the keyword; fix or remove it"
+ inner)))))
+ (goto-char eol))))))
+
+(defun lo--check-level2-done-without-closed ()
+ "Flag a level-2 DONE/CANCELLED heading with no CLOSED line in its own entry.
+todo-cleanup's `--archive-done' aging step archives a completed task with no
+parseable CLOSED date immediately, so an undated completed task silently leaves
+the live file on the next `task-sorted'."
+ (save-excursion
+ (goto-char (point-min))
+ ;; Case-sensitive: DONE/CANCELLED are uppercase keywords, not the words
+ ;; "done"/"cancelled" in a heading title (case-fold-search defaults to t).
+ (let ((case-fold-search nil)
+ (re (format "^\\*\\* \\(%s\\) "
+ (mapconcat #'regexp-quote lo-done-keywords "\\|"))))
+ (while (re-search-forward re nil t)
+ (let ((hline (line-number-at-pos))
+ (entry-end (save-excursion (outline-next-heading) (point))))
+ (save-excursion
+ (forward-line 1)
+ (unless (re-search-forward "^[ \t]*CLOSED:[ \t]*\\[" entry-end t)
+ (lo--emit-judgment
+ 'level2-done-without-closed hline
+ "level-2 DONE/CANCELLED has no CLOSED date — add CLOSED: [YYYY-MM-DD Day]; task-sorted's aging step archives an undated completed task immediately"))))))))
+
+;;; ---------------------------------------------------------------------------
+;;; task-missing-last-reviewed check (claude-rules/todo-format.md)
+;;
+;; A task is stamped `:LAST_REVIEWED:' when it is *created*, not a review cycle
+;; later. An agent filing a task has just written its body and graded its
+;; priority, which is a review by any honest reading — so a fresh task that
+;; carries no stamp reads as "never reviewed" and lands at the top of the next
+;; staleness batch, where re-reviewing it is pure ceremony. Every task filed
+;; during the 2026-07-23 sweep hit exactly that, which is what prompted the rule.
+;;
+;; Judgment-only, deliberately. The stamp's whole value is that its date is
+;; true, and nothing here can know when an unstamped task was actually last
+;; looked at. Auto-stamping today's date would convert a "nobody has reviewed
+;; this" signal into a false "reviewed today" one — worse than the gap it
+;; closes. Flag it; a human or the filing workflow supplies the honest date.
+;;
+;; Scope matches `task-review-staleness.sh' exactly (level-2, open keyword,
+;; priority cookie), so the checker and the staleness count never disagree
+;; about which headings are in the review pool.
+
+(defun lo--check-task-missing-last-reviewed ()
+ "Flag an open level-2 task with a priority cookie and no `:LAST_REVIEWED:'."
+ (save-excursion
+ (goto-char (point-min))
+ (let ((case-fold-search nil))
+ (while (re-search-forward "^\\*\\* \\(TODO\\|DOING\\|VERIFY\\) \\[#[A-D]\\]" nil t)
+ (let ((hline (line-number-at-pos))
+ (entry-end (save-excursion (outline-next-heading) (point))))
+ (save-excursion
+ (forward-line 1)
+ (unless (re-search-forward "^[ \t]*:LAST_REVIEWED:[ \t]*[[0-9]"
+ entry-end t)
+ (lo--emit-judgment
+ 'task-missing-last-reviewed hline
+ "task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed"))))))))
+
+;;; ---------------------------------------------------------------------------
+;;; level-3+ dated-header check (claude-rules/todo-format.md)
+;;
+;; The inverse of the level-2 check above. A completed sub-task — a heading at
+;; level 3 or deeper, under a parent task — becomes a dated event-log entry, not
+;; a DONE keyword, so the parent's subtree grows a chronological history instead
+;; of a long tail of nested DONE lines. An interactive org close
+;; (`org-log-done' → DONE + CLOSED) leaves the keyword in place, and
+;; `--archive-done' only touches level 2, so these accumulate. Flag them for
+;; conversion. Judgment-only and regex-based (independent of which TODO keywords
+;; the batch Emacs recognizes); todo-cleanup.el --convert-subtasks does the fix.
+
+(defun lo--check-subtask-done-not-dated ()
+ "Flag level-3+ headings carrying a done keyword (DONE/CANCELLED/FAILED).
+Emits one judgment item per offending heading (checker
+`subtask-done-not-dated')."
+ (save-excursion
+ (goto-char (point-min))
+ ;; Case-sensitive: the keywords are uppercase, not the words in a title.
+ (let ((case-fold-search nil))
+ (while (re-search-forward
+ "^\\*\\{3,\\} \\(DONE\\|CANCELLED\\|FAILED\\) " nil t)
+ (lo--emit-judgment
+ 'subtask-done-not-dated (line-number-at-pos)
+ "level-3+ done sub-task should be a dated event-log entry (todo-format.md): run todo-cleanup.el --convert-subtasks to rewrite it")))))
+
+;;; ---------------------------------------------------------------------------
+;;; dated-log heading with a stale active planning timestamp (todo-format.md)
+;;
+;; The mechanical backstop for the planning-line-strip rule. A dated event-log
+;; heading (`<stars> YYYY-MM-DD Day @ ...', no TODO keyword) records completed
+;; work — its date lives in the heading. An active `<...>' SCHEDULED or DEADLINE
+;; left on it pins the entry to the agenda forever: org renders any headline with
+;; an active planning timestamp, keyword or not, so a stale SCHEDULED shows as
+;; weeks-overdue long after the work is done. Invisible to a keyword scan (no
+;; TODO) and it survives --archive-done, so nothing else catches it. The
+;; completion rewrite and todo-cleanup --convert-subtasks now strip the planning
+;; line; this flags any that slipped through before that landed, the same way
+;; subtask-done-not-dated backstops the depth rule. Judgment-only.
+
+(defun lo--check-dated-log-active-timestamp ()
+ "Flag a dated event-log heading that still carries an active SCHEDULED/DEADLINE.
+The heading matches `<stars> YYYY-MM-DD Day @ ...' with no TODO keyword; an
+active `<...>' planning timestamp in its entry is the defect. An inactive
+`[...]' timestamp is ignored (org doesn't render it on the agenda). Emits one
+judgment item per offending heading (checker `dated-log-heading-active-timestamp')."
+ (save-excursion
+ (goto-char (point-min))
+ (let ((case-fold-search nil))
+ (while (re-search-forward
+ "^\\*+ [0-9]\\{4\\}-[0-9]\\{2\\}-[0-9]\\{2\\} [A-Za-z]+ @ " nil t)
+ (let ((hline (line-number-at-pos))
+ (entry-end (save-excursion (outline-next-heading) (point))))
+ (save-excursion
+ (forward-line 1)
+ (when (re-search-forward
+ "^[ \t]*\\(?:SCHEDULED\\|DEADLINE\\):[ \t]*<" entry-end t)
+ (lo--emit-judgment
+ 'dated-log-heading-active-timestamp hline
+ "dated-log heading carries an active SCHEDULED/DEADLINE — org renders any active planning timestamp (keyword or not), so it stays on the agenda as weeks-overdue; delete the planning line (todo-format.md)"))))))))
;;; ---------------------------------------------------------------------------
;;; File processing
@@ -401,6 +722,22 @@ left unmodified and mechanical entries are recorded with :preview t."
;; After org-lint items: the custom table-standard scan. Runs on the
;; post-fix buffer; judgment-only, so order doesn't perturb fixes.
(lo--check-tables)
+ ;; Structural heading defects org-lint doesn't cover. These run on
+ ;; every org file, specs included.
+ (lo--check-indented-headings)
+ (lo--check-empty-headings)
+ (lo--check-malformed-priority-cookies)
+ ;; The todo-format family encodes todo.org completion conventions and
+ ;; misfires on a spec (a Decisions section's undated DONE, a
+ ;; review-history dated heading, a phases task with no LAST_REVIEWED).
+ ;; Scope them out of docs/specs/; link, table, and structural checks
+ ;; above still run there.
+ (unless (lo--spec-file-p)
+ (lo--check-level2-dated-headers)
+ (lo--check-level2-done-without-closed)
+ (lo--check-task-missing-last-reviewed)
+ (lo--check-subtask-done-not-dated)
+ (lo--check-dated-log-active-timestamp))
(when (and (not lo-check-only) (buffer-modified-p))
(save-buffer)))
(with-current-buffer buf (set-buffer-modified-p nil))
@@ -507,6 +844,13 @@ After printing, also append judgments to `lo-followups-file' when set."
;;; CLI
(defun lo-main ()
+ ;; Report-only is the CLI default; --fix is the only way a command-line run
+ ;; writes to disk. The old mutate-by-default reformatted five files in one
+ ;; pass before anyone confirmed anything (work project, 2026-07-09).
+ (setq lo-check-only t)
+ (when (member "--fix" command-line-args-left)
+ (setq lo-check-only nil)
+ (setq command-line-args-left (delete "--fix" command-line-args-left)))
(when (member "--check" command-line-args-left)
(setq lo-check-only t)
(setq command-line-args-left (delete "--check" command-line-args-left)))
@@ -518,7 +862,7 @@ After printing, also append judgments to `lo-followups-file' when set."
(setq command-line-args-left (delete followups command-line-args-left))))
(if (null command-line-args-left)
(progn
- (princ "Usage: emacs --batch -q -l lint-org.el [--check] [--followups-file=PATH] FILE.org ...\n")
+ (princ "Usage: emacs --batch -q -l lint-org.el [--fix] [--check] [--followups-file=PATH] FILE.org ...\n")
(kill-emacs 1))
(let ((files command-line-args-left))
(setq command-line-args-left nil)
@@ -537,7 +881,7 @@ this file without firing the CLI dispatch — under `ert-run-tests-batch-and-exi
the trailing args are things like `-f ert-run-tests-batch-and-exit'."
(and command-line-args-left
(cl-every (lambda (a)
- (cond ((member a '("--check")) t)
+ (cond ((member a '("--check" "--fix")) t)
((string-prefix-p "--followups-file=" a) t)
((string-prefix-p "-" a) nil)
(t (file-readable-p a))))
diff --git a/.ai/scripts/route-batch b/.ai/scripts/route-batch
new file mode 100755
index 0000000..8f27d19
--- /dev/null
+++ b/.ai/scripts/route-batch
@@ -0,0 +1,175 @@
+#!/usr/bin/env python3
+"""route-batch — the wrap-up router's mechanical go path.
+
+The wrap-up cross-project router (wrap-it-up.org Step 3; wrapup-routing spec
+D7/D8/D9) surfaces the local tasks that inbox process mode stamped with
+:ROUTE_CANDIDATE: <destination> at file time, and on "go" delivers each to its
+destination project's inbox. This script does the mechanical half so the
+subtree surgery is deterministic:
+
+ route-batch --list [--todo todo.org]
+ One "<destination>\t<heading>" line per :ROUTE_CANDIDATE:-tagged task.
+ Silent with exit 0 when there are no candidates (the workflow's
+ empty-set-equals-zero-interaction rule). Read-only.
+
+ route-batch --go [--todo todo.org]
+ For each candidate, bottom-up: extract the task's whole subtree
+ (children ride along), drop the :ROUTE_CANDIDATE: line (and the
+ property drawer if that leaves it empty), promote the subtree so its
+ top heading is level 1, write it to a temp file, and deliver it via
+ the sibling inbox-send.py to the destination's inbox/ (one file per
+ task, from-<source> provenance stamped by inbox-send). Only after a
+ successful send is the subtree removed from the local todo.org — a
+ failed send leaves that task in place, is reported, and the run exits
+ non-zero after attempting the rest.
+
+The candidate set is exactly the tagged tasks — never the standing backlog.
+Discovery, roots, and the source-project name all come from inbox-send.py
+(INBOX_SEND_ROOTS sandboxes it in tests). The reject-from-another-project
+flow in inbox process mode is the mis-route recovery; that path is why
+removing the local source after a successful send is safe.
+"""
+
+import argparse
+import os
+import re
+import subprocess
+import sys
+import tempfile
+from pathlib import Path
+
+HEADING_RE = re.compile(r"^(\*+)\s+(.*)$")
+MARKER_RE = re.compile(r"^\s*:ROUTE_CANDIDATE:\s+(\S+)\s*$")
+
+
+def find_candidates(lines):
+ """[(heading_idx, end_idx, marker_idx, destination, heading_text)] —
+ end_idx is one past the subtree's last line."""
+ candidates = []
+ for i, line in enumerate(lines):
+ m = MARKER_RE.match(line)
+ if not m:
+ continue
+ head_idx = None
+ for j in range(i, -1, -1):
+ hm = HEADING_RE.match(lines[j])
+ if hm:
+ head_idx = j
+ level = len(hm.group(1))
+ heading = hm.group(2)
+ break
+ if head_idx is None:
+ continue
+ end = len(lines)
+ for k in range(head_idx + 1, len(lines)):
+ km = HEADING_RE.match(lines[k])
+ if km and len(km.group(1)) <= level:
+ end = k
+ break
+ candidates.append((head_idx, end, i, m.group(1), heading))
+ return candidates
+
+
+def extract_handoff(lines, head_idx, end):
+ """The subtree as handoff text: every :ROUTE_CANDIDATE: line dropped
+ (a marker is meaningless at the destination), empty drawers pruned,
+ headings promoted so the task is level 1."""
+ sub = [l for l in lines[head_idx:end] if not MARKER_RE.match(l)]
+
+ pruned = []
+ i = 0
+ while i < len(sub):
+ if sub[i].strip() == ":PROPERTIES:" and i + 1 < len(sub) and sub[i + 1].strip() == ":END:":
+ i += 2
+ continue
+ pruned.append(sub[i])
+ i += 1
+
+ shift = len(HEADING_RE.match(pruned[0]).group(1)) - 1
+ if shift > 0:
+ pruned = [l[shift:] if HEADING_RE.match(l) else l for l in pruned]
+ return "\n".join(pruned).rstrip() + "\n"
+
+
+def send(destination, handoff_text, slug):
+ inbox_send = Path(__file__).with_name("inbox-send.py")
+ with tempfile.NamedTemporaryFile(
+ "w", suffix=".org", prefix=f"route-{slug}-", delete=False, encoding="utf-8"
+ ) as tf:
+ tf.write(handoff_text)
+ tmp = tf.name
+ try:
+ result = subprocess.run(
+ [sys.executable, str(inbox_send), destination, "--file", tmp],
+ capture_output=True, text=True,
+ )
+ return result.returncode == 0, (result.stderr or result.stdout).strip()
+ finally:
+ os.unlink(tmp)
+
+
+def main():
+ ap = argparse.ArgumentParser(prog="route-batch")
+ mode = ap.add_mutually_exclusive_group(required=True)
+ mode.add_argument("--list", action="store_true", dest="list_mode")
+ mode.add_argument("--go", action="store_true")
+ ap.add_argument("--todo", default="todo.org")
+ args = ap.parse_args()
+
+ todo_path = Path(args.todo)
+ if not todo_path.is_file():
+ return 0 # no todo file, no candidates
+ lines = todo_path.read_text(encoding="utf-8").splitlines()
+ candidates = find_candidates(lines)
+
+ # Two markers in one task's drawer are one candidate, not two: same span +
+ # same destination dedupes. Everything else that overlaps — a tagged child
+ # inside a tagged parent, one task tagged for two destinations — is a
+ # conflict: routing either span would silently take the other (or, with a
+ # stale end index, a bystander task) along. Conflicts are left in place
+ # and reported; the human untangles which project the pieces belong to.
+ deduped = []
+ for cand in candidates:
+ if not any(c[0] == cand[0] and c[1] == cand[1] and c[3] == cand[3] for c in deduped):
+ deduped.append(cand)
+ conflicted = set()
+ for a in deduped:
+ for b in deduped:
+ if a is not b and a[0] <= b[0] and b[1] <= a[1]:
+ conflicted.add(a)
+ conflicted.add(b)
+ routable = [c for c in deduped if c not in conflicted]
+
+ if not deduped:
+ return 0
+
+ if args.list_mode:
+ for _h, _e, _m, dest, heading in deduped:
+ flag = "\tCONFLICT (overlapping candidates — resolve by hand)" if (_h, _e, _m, dest, heading) in conflicted else ""
+ print(f"{dest}\t{heading}{flag}")
+ return 0
+
+ failures = 0
+ for _h, _e, _m, dest, heading in sorted(conflicted):
+ failures += 1
+ print(f"CONFLICT: {dest}\t{heading}\t(overlapping candidate subtrees — left in place, resolve by hand)")
+
+ # Bottom-up so earlier indices stay valid as subtrees are removed; the
+ # file is rewritten after every successful send so a crash mid-run never
+ # leaves an already-sent task still present locally.
+ for head_idx, end, _marker_idx, dest, heading in sorted(routable, reverse=True):
+ handoff = extract_handoff(lines, head_idx, end)
+ slug = re.sub(r"[^a-z0-9]+", "-", heading.lower()).strip("-")[:40] or "task"
+ ok, detail = send(dest, handoff, slug)
+ if ok:
+ del lines[head_idx:end]
+ todo_path.write_text("\n".join(lines).rstrip("\n") + "\n", encoding="utf-8")
+ print(f"routed: {dest}\t{heading}")
+ else:
+ failures += 1
+ print(f"FAILED: {dest}\t{heading}\t({detail})")
+ return 1 if failures else 0
+
+
+if __name__ == "__main__":
+ sys.exit(main())
diff --git a/.ai/scripts/route_recommend.py b/.ai/scripts/route_recommend.py
new file mode 100644
index 0000000..12ab132
--- /dev/null
+++ b/.ai/scripts/route_recommend.py
@@ -0,0 +1,145 @@
+#!/usr/bin/env python3
+"""Wrap-up routing recommendation engine.
+
+Given an inbox keeper's text and a list of candidate project names, infer which
+project the item belongs to, with a confidence tier:
+
+ strong a project's name (or its dot-stripped form, or a path containing it)
+ appears literally in the item
+ weak a distinctive name token overlaps, but the full name doesn't
+ none no overlap; the item stays put
+
+A multi-way tie at the top tier is ambiguous, so it downgrades to weak with a
+deterministic pick (most token overlap, then alphabetical). An empty candidate
+list yields none.
+
+The pure core is `recommend(item, projects) -> (destination, confidence)` — the
+shape the wrap-up router (Phase 4) and the process-inbox marker (Phase 2) both
+call. The CLI wires it to inbox-send.py's `discover_projects` so the candidate
+set is the same project universe inbox-send already knows.
+
+CLI:
+ route_recommend.py --item "<text>" [--exclude <current-project>]
+prints "<destination>\\t<confidence>" on a match, or "none".
+"""
+
+import argparse
+import importlib.util
+import re
+import sys
+from pathlib import Path
+
+# A distinctive-enough token for weak matching; shorter tokens (of, to, id) are
+# too noisy to route on.
+MIN_WEAK_TOKEN = 4
+
+_TOKEN_RE = re.compile(r"[a-z0-9]+")
+
+
+def _tokens(text: str) -> set[str]:
+ return set(_TOKEN_RE.findall(text.lower()))
+
+
+def _name_variants(name: str) -> set[str]:
+ """A project name and its dot-stripped alias (.emacs.d -> emacsd)."""
+ return {v for v in (name.lower(), name.replace(".", "").lower()) if v}
+
+
+def _literal_present(name: str, item_lower: str) -> bool:
+ """True if a name variant appears in the item on word-ish boundaries.
+
+ Boundaries keep 'home' from matching inside 'homeowner' while still
+ matching it inside a path ('~/code/home/...') or a hyphenated name.
+ """
+ for variant in _name_variants(name):
+ if re.search(r"(?<![a-z0-9])" + re.escape(variant) + r"(?![a-z0-9])", item_lower):
+ return True
+ return False
+
+
+def _tiebreak(candidates: list[str], item_tokens: set[str]) -> str:
+ """Most token overlap first, then alphabetical — deterministic."""
+ return sorted(candidates, key=lambda p: (-len(_tokens(p) & item_tokens), p))[0]
+
+
+def recommend(item: str, projects: list[str]) -> tuple[str | None, str]:
+ """Infer the destination project for `item` from `projects`.
+
+ Returns (destination, confidence). confidence is "strong" / "weak" / "none";
+ destination is None exactly when confidence is "none".
+ """
+ if not projects:
+ return (None, "none")
+
+ # Collapse identical names first. Projects are addressed by bare basename, so
+ # two projects sharing one across roots (~/code/notes, ~/projects/notes) arrive
+ # twice; both literal-match, and the tie test below then read that as ambiguity
+ # and downgraded a correct strong match to weak. Deduping here rather than in
+ # discover_destination_names protects every caller of the pure core, not just
+ # the CLI path. Order-preserving, and it collapses only identical names — two
+ # *different* projects matching is real ambiguity and still downgrades.
+ projects = list(dict.fromkeys(projects))
+
+ item_lower = item.lower()
+ item_tokens = _tokens(item)
+
+ strong: list[str] = []
+ weak: list[str] = []
+ for project in projects:
+ if _literal_present(project, item_lower):
+ strong.append(project)
+ continue
+ name_tokens = {t for t in _tokens(project) if len(t) >= MIN_WEAK_TOKEN}
+ if name_tokens & item_tokens:
+ weak.append(project)
+
+ if len(strong) == 1:
+ return (strong[0], "strong")
+ if len(strong) > 1:
+ return (_tiebreak(strong, item_tokens), "weak")
+ if len(weak) == 1:
+ return (weak[0], "weak")
+ if len(weak) > 1:
+ return (_tiebreak(weak, item_tokens), "weak")
+ return (None, "none")
+
+
+def _load_inbox_send():
+ """Load the sibling kebab-named inbox-send.py as a module for its discovery."""
+ path = Path(__file__).with_name("inbox-send.py")
+ spec = importlib.util.spec_from_file_location("inbox_send", path)
+ if spec is None or spec.loader is None:
+ raise ImportError(f"cannot load {path}")
+ module = importlib.util.module_from_spec(spec)
+ spec.loader.exec_module(module)
+ return module
+
+
+def discover_destination_names(exclude: str | None = None) -> list[str]:
+ """The candidate project names, reusing inbox-send's discovery.
+
+ `exclude` drops the current project (matched by exact name or dot-stripped
+ alias) so the engine never recommends routing an item to where it already is.
+ """
+ mod = _load_inbox_send()
+ names = [p.name for p in mod.discover_projects(mod.resolve_roots())]
+ if exclude:
+ drop = _name_variants(exclude)
+ names = [n for n in names if not (_name_variants(n) & drop)]
+ return names
+
+
+def main() -> int:
+ parser = argparse.ArgumentParser(description="Recommend a routing destination for an inbox keeper.")
+ parser.add_argument("--item", required=True, help="the keeper's text")
+ parser.add_argument("--exclude", help="current project to exclude from candidates")
+ args = parser.parse_args()
+
+ projects = discover_destination_names(exclude=args.exclude)
+ destination, confidence = recommend(args.item, projects)
+ print("none" if destination is None else f"{destination}\t{confidence}")
+ return 0
+
+
+if __name__ == "__main__":
+ sys.exit(main())
diff --git a/.ai/scripts/self-inject.sh b/.ai/scripts/self-inject.sh
new file mode 100755
index 0000000..e7340c1
--- /dev/null
+++ b/.ai/scripts/self-inject.sh
@@ -0,0 +1,68 @@
+#!/bin/sh
+# self-inject.sh — type text into the tmux pane running this agent session.
+#
+# The building block for AUTO-FLUSH: an agent checkpoints its session-context,
+# then has tmux type "/clear" and a resume prompt at its own idle prompt, so a
+# session flushes with no human at the keyboard.
+#
+# Usage:
+# self-inject.sh -t %PANE <delay> <text> [<delay2> <text2> ...]
+# self-inject.sh <delay> <text> [...] # derive pane from ancestry
+# self-inject.sh [-t %PANE] # no pairs: report the pane
+#
+# Each pair: sleep <delay> seconds, then type <text> literally and press Enter.
+#
+# TWO HARD-WON GOTCHAS (2026-07-02, archsetup session):
+# 1. A detached child (setsid/nohup/&) of an agent tool call DIES when the
+# tool call ends — the harness cleans up the process group. The arm step
+# must run under the tmux SERVER instead:
+# tmux run-shell -b "self-inject.sh -t %1 25 '/clear' 15 'go — resume...'"
+# 2. Under tmux run-shell the process is a child of the tmux server, so
+# ancestry-based pane detection CANNOT work there. Derive the pane FIRST,
+# synchronously from the agent's own shell (no -t), then pass it
+# explicitly with -t when arming.
+#
+# Collision hazard: if the user happens to be typing when the send fires, the
+# injected text merges into their input line (a real /clear became "/clearto"
+# mid-word). Auto-flush is for sessions running unattended; warn the user to
+# keep hands off for the armed window if they're present.
+
+PANE=""
+if [ "$1" = "-t" ]; then
+ PANE=$2; shift 2
+fi
+
+ppid_of() {
+ # /proc/<pid>/stat: pid (comm) state ppid ... — comm may contain spaces,
+ # so take the 2nd field after the LAST ')'.
+ stat=$(cat "/proc/$1/stat" 2>/dev/null) || return 1
+ # shellcheck disable=SC2086 # word-splitting the stat tail is the point
+ set -- ${stat##*) }
+ echo "$2"
+}
+
+find_pane() {
+ anc=" "
+ pid=$$
+ while [ -n "$pid" ] && [ "$pid" -gt 1 ] 2>/dev/null; do
+ anc="$anc$pid "
+ pid=$(ppid_of "$pid") || break
+ done
+ tmux list-panes -a -F "#{pane_pid} #{pane_id}" 2>/dev/null | \
+ while read -r ppid pane; do
+ case "$anc" in *" $ppid "*) echo "$pane"; break;; esac
+ done
+}
+
+[ -n "$PANE" ] || PANE=$(find_pane)
+[ -n "$PANE" ] || { echo "self-inject: no owning pane found (pass -t %PANE)" >&2; exit 1; }
+
+# With no delay/text pairs, just report the pane (the derive-first step).
+[ $# -ge 2 ] || { echo "$PANE"; exit 0; }
+
+while [ $# -ge 2 ]; do
+ sleep "$1"
+ tmux send-keys -t "$PANE" -l "$2"
+ tmux send-keys -t "$PANE" Enter
+ shift 2
+done
diff --git a/.ai/scripts/session-context-path b/.ai/scripts/session-context-path
index 8cc56f6..670a610 100755
--- a/.ai/scripts/session-context-path
+++ b/.ai/scripts/session-context-path
@@ -10,6 +10,14 @@
# instead of clobbering the singleton. The id is sanitized to filename-safe
# characters so a stray value can't escape the .d/ directory.
#
+# The id must be unique per run; the spawner appends an epoch on the tail
+# (recommended shape host.project.runtime.<epoch>) so a re-run of the same
+# logical agent gets a fresh anchor instead of resolving to a prior run's
+# leftover. The epoch is never minted here: this resolver is called many times
+# per session and must return the same path each call, so it can't generate a
+# new value. See protocols.org "Agent-scoped path". A bare, reused id (just
+# "codex") is the bug that motivated this note.
+#
# Workflows call this to resolve the path; both startup (existence check) and
# wrap-up (rename source) read/write through it. Callers should fall back to
# .ai/session-context.org if this script isn't present yet (older checkouts
diff --git a/.ai/scripts/spec-sort b/.ai/scripts/spec-sort
new file mode 100755
index 0000000..ebfef82
--- /dev/null
+++ b/.ai/scripts/spec-sort
@@ -0,0 +1,715 @@
+#!/usr/bin/env python3
+"""spec-sort — one-time docs-pile retrofit for the docs-lifecycle convention.
+
+Classifies every docs/**/*.org outside docs/specs/ by one predicate: a doc
+carrying BOTH a "Decisions" heading AND an "Implementation phases" heading is
+a spec candidate; everything else is a note. For each candidate it shows an
+evidence panel (Status field, decision/finding cookies, the linking todo.org
+task, recent dated history, cheap existence checks on phase-named artifacts)
+and proposes a lifecycle keyword the evidence supports — conservative
+non-terminal (DRAFT) when inconclusive. The helper proposes; a human confirms
+every move.
+
+Dry-run report is the default. --apply executes under the fail-safe contract:
+
+ - Clean-worktree preflight: refuses on a dirty git tree (exit 2) unless
+ --allow-dirty, which prints exactly what recovery loses.
+ - Every candidate must be addressed with --confirm REL=KEYWORD or
+ --skip REL; terminal keywords (IMPLEMENTED SUPERSEDED CANCELLED) also
+ need --reason REL=TEXT, recorded in the status-history line.
+ - The full move + relink plan is computed and validated first (every
+ destination free, every link resolvable), written to a plan file, and
+ only then executed from that recorded plan.
+ - Bare-path mentions of a moving doc inside the rewritten roots are
+ reported, never rewritten; they block --apply until --acknowledge-bare
+ explicitly waives them.
+ - Mid-apply failure stops the run, names what was and wasn't applied, and
+ prints the git-restore recovery recipe (plus deletion of newly created
+ destination copies, which git restore can't remove).
+ - After a successful apply, a residue scan across the rewritten roots must
+ find no link still resolving to an old path, or spec-sort exits non-zero
+ naming the residue.
+
+Per move: rename to carry the -spec.org suffix, prepend the status heading
+(:ID: UUID + dated history line), rewrite the keyword header to the
+two-sequence form, mirror the keyword into the Metadata Status field, and
+recompute every affected file: link (inbound links to the moved doc AND the
+moved doc's own outbound relative links). Rewritten roots: todo.org,
+.ai/notes.org, docs/**, .ai/project-workflows/, .ai/project-scripts/.
+Reported-never-rewritten: .ai/sessions/ (frozen history) and synced template
+paths (.ai/workflows/, .ai/scripts/, .ai/protocols.org — the report names
+the canonical claude-templates file instead).
+
+Finally stamps :LAST_SPEC_SORT: YYYY-MM-DD in .ai/notes.org's
+* Workflow State section (created idempotently), which permanently clears
+the startup nudge. A run with zero candidates still stamps.
+
+Exit codes: 0 done (or clean report), 1 blocked (confirm gate, validation,
+bare mentions, residue, mid-apply failure), 2 usage / preflight refusal.
+
+Test hook: SPEC_SORT_INJECT_FAIL_AFTER=N aborts the apply after N write
+operations, exercising the recovery path in the bats suite.
+"""
+
+import argparse
+import json
+import os
+import re
+import subprocess
+import sys
+import tempfile
+import uuid
+from datetime import datetime
+
+LIFECYCLE = ("DRAFT", "READY", "DOING", "IMPLEMENTED", "SUPERSEDED", "CANCELLED")
+TERMINAL = {"IMPLEMENTED", "SUPERSEDED", "CANCELLED"}
+TODO_HEADER = [
+ "#+TODO: TODO | DONE",
+ "#+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED",
+]
+
+# Project-owned surfaces whose file: links get rewritten.
+REWRITE_ROOTS = ("todo.org", ".ai/notes.org", "docs", ".ai/project-workflows", ".ai/project-scripts")
+# Frozen or synced surfaces: occurrences are reported, never rewritten.
+REPORT_ROOTS = (".ai/sessions", ".ai/workflows", ".ai/scripts", ".ai/protocols.org")
+# Synced template paths map to their canonical rulesets file for the report.
+SYNCED_PREFIX = (".ai/workflows", ".ai/scripts", ".ai/protocols.org")
+
+LINK_RE = re.compile(r"\[\[file:([^\]\[]+)\](?:\[([^\]\[]*)\])?\]")
+HEADING_RE = re.compile(r"^(\*+)\s+(.*)$")
+COOKIE_RE = re.compile(r"\[\d+/\d+\]")
+DATED_RE = re.compile(r"\b\d{4}-\d{2}-\d{2}\b")
+
+
+def read_text(path):
+ try:
+ with open(path, encoding="utf-8") as f:
+ return f.read()
+ except (UnicodeDecodeError, OSError):
+ return None
+
+
+def heading_text(line):
+ """Heading text with the org keyword and priority cookie stripped."""
+ m = HEADING_RE.match(line)
+ if not m:
+ return None
+ text = re.sub(r"^[A-Z]+\s+", "", m.group(2))
+ text = re.sub(r"^\[#[A-Z]\]\s+", "", text)
+ return text.strip()
+
+
+def has_spine(content):
+ """The classification predicate: Decisions AND Implementation phases."""
+ dec = imp = False
+ for line in content.splitlines():
+ t = heading_text(line)
+ if t is None:
+ continue
+ tl = t.lower()
+ if tl.startswith("decisions"):
+ dec = True
+ elif tl.startswith("implementation phases"):
+ imp = True
+ return dec and imp
+
+
+def walk_files(root, rel_base):
+ """Yield project-relative paths of files under rel_base (file or dir)."""
+ abs_base = os.path.join(root, rel_base)
+ if os.path.isfile(abs_base):
+ yield rel_base
+ return
+ for dirpath, dirs, files in os.walk(abs_base):
+ dirs.sort()
+ for name in sorted(files):
+ yield os.path.relpath(os.path.join(dirpath, name), root)
+
+
+def classify(root):
+ """Split docs/**/*.org outside docs/specs/ into candidates / anomalies / notes."""
+ candidates, anomalies, notes = [], [], []
+ docs = os.path.join(root, "docs")
+ if not os.path.isdir(docs):
+ return candidates, anomalies, notes
+ for rel in walk_files(root, "docs"):
+ if not rel.endswith(".org"):
+ continue
+ parts = rel.split(os.sep)
+ if len(parts) > 1 and parts[1] == "specs":
+ continue
+ content = read_text(os.path.join(root, rel))
+ if content is None:
+ continue
+ if has_spine(content):
+ candidates.append(rel)
+ elif os.path.basename(rel).endswith("-spec.org"):
+ anomalies.append(rel)
+ else:
+ notes.append(rel)
+ return candidates, anomalies, notes
+
+
+def dest_for(rel):
+ base = os.path.basename(rel)
+ if not base.endswith("-spec.org"):
+ base = base[: -len(".org")] + "-spec.org"
+ return os.path.join("docs", "specs", base)
+
+
+# ---- Evidence panel ---------------------------------------------------
+
+
+def todo_task_for(root, rel):
+ """Heading of the first todo.org task whose subtree mentions the doc."""
+ content = read_text(os.path.join(root, "todo.org"))
+ if content is None:
+ return None
+ lines = content.splitlines()
+ basename = os.path.basename(rel)
+ for i, line in enumerate(lines):
+ if basename in line or rel in line:
+ for j in range(i, -1, -1):
+ if HEADING_RE.match(lines[j]):
+ return lines[j].lstrip("* ").strip()
+ return None
+ return None
+
+
+def gather_evidence(root, rel, content):
+ ev = {}
+ m = re.search(r"^\|\s*Status\s*\|\s*([^|]*)\|", content, re.MULTILINE | re.IGNORECASE)
+ ev["status"] = m.group(1).strip() if m else None
+
+ cookies = []
+ for line in content.splitlines():
+ t = heading_text(line)
+ if t and COOKIE_RE.search(t) and (
+ t.lower().startswith("decisions") or t.lower().startswith("review findings")
+ ):
+ cookies.append(t)
+ ev["cookies"] = cookies
+
+ ev["todo"] = todo_task_for(root, rel)
+ kw = None
+ if ev["todo"]:
+ m = re.match(r"([A-Z]+)\s", ev["todo"])
+ kw = m.group(1) if m else None
+ ev["todo_keyword"] = kw
+
+ dated = [ln.strip() for ln in content.splitlines() if DATED_RE.search(ln)]
+ ev["history"] = dated[-1][:100] if dated else None
+
+ # Cheap artifact check: =path= tokens inside the Implementation phases section.
+ artifacts, exists = [], 0
+ section = re.split(r"^\*+\s+.*implementation phases.*$", content, maxsplit=1, flags=re.MULTILINE | re.IGNORECASE)
+ if len(section) > 1:
+ for tok in re.findall(r"=([^=\s]+)=", section[1]):
+ if "/" in tok:
+ artifacts.append(tok)
+ if os.path.exists(os.path.join(root, tok)):
+ exists += 1
+ ev["artifacts"] = (exists, artifacts)
+ return ev
+
+
+def propose_keyword(ev):
+ s = (ev["status"] or "").lower()
+ words = set(re.findall(r"[a-z]+", s))
+ if words & {"implemented", "shipped", "complete", "completed", "done"}:
+ return "IMPLEMENTED"
+ if words & {"superseded"}:
+ return "SUPERSEDED"
+ if words & {"cancelled", "canceled", "dead", "abandoned"}:
+ return "CANCELLED"
+ if words & {"doing", "implementing"} or "in progress" in s or "in-progress" in s:
+ return "DOING"
+ if ev["todo_keyword"] == "DOING":
+ return "DOING"
+ if words & {"ready", "approved", "accepted"}:
+ return "READY"
+ return "DRAFT" # conservative non-terminal default
+
+
+# ---- Link scanning ----------------------------------------------------
+
+
+def rewrite_files(root):
+ """Project-relative *.org files under the rewritten roots."""
+ seen = []
+ for base in REWRITE_ROOTS:
+ if not os.path.exists(os.path.join(root, base)):
+ continue
+ for rel in walk_files(root, base):
+ if rel.endswith(".org") and rel not in seen:
+ seen.append(rel)
+ return seen
+
+
+def resolve_target(root, linker_rel, raw_target, moved):
+ """Resolve a file: link target to a project-relative path (org semantics
+ first — relative to the linking file's directory — then project-root
+ anchoring as a fallback for root-anchored links)."""
+ if raw_target.startswith(("/", "~", "http:", "https:")):
+ return None
+ rel_a = os.path.normpath(os.path.join(os.path.dirname(linker_rel), raw_target))
+ if rel_a in moved or os.path.exists(os.path.join(root, rel_a)):
+ return rel_a
+ rel_b = os.path.normpath(raw_target)
+ if rel_b in moved or os.path.exists(os.path.join(root, rel_b)):
+ return rel_b
+ return rel_a
+
+
+def plan_link_edits(root, moved):
+ """Compute every link rewrite: inbound links to moved docs and moved
+ docs' own outbound relative links. Returns ({linker_rel: [(old, new)]},
+ [ambiguity descriptions]) — a link whose file-relative and root-anchored
+ readings are both live and disagree about a moving doc blocks validation
+ rather than being rewritten against a guess."""
+ edits = {}
+ ambiguous = []
+ for linker in rewrite_files(root):
+ content = read_text(os.path.join(root, linker))
+ if content is None:
+ continue
+ linker_post = moved.get(linker, linker)
+ for m in LINK_RE.finditer(content):
+ raw = m.group(1)
+ desc = m.group(2)
+ target_path, sep, anchor = raw.partition("::")
+ target = resolve_target(root, linker, target_path, moved)
+ if target is None:
+ continue
+ rel_a = os.path.normpath(os.path.join(os.path.dirname(linker), target_path))
+ rel_b = os.path.normpath(target_path)
+ if rel_a != rel_b:
+ live_a = rel_a in moved or os.path.exists(os.path.join(root, rel_a))
+ live_b = rel_b in moved or os.path.exists(os.path.join(root, rel_b))
+ if live_a and live_b and (rel_a in moved or rel_b in moved):
+ ambiguous.append(
+ "%s: [[file:%s]] reads as %s (file-relative) or %s (root-anchored) "
+ "and a moving doc is involved — resolve the link by hand" % (linker, raw, rel_a, rel_b))
+ continue
+ if target not in moved and linker not in moved:
+ continue
+ if target not in moved and not os.path.exists(os.path.join(root, target)):
+ continue # already broken before this run; not ours to guess
+ target_post = moved.get(target, target)
+ new_path = os.path.relpath(target_post, os.path.dirname(linker_post) or ".")
+ new_raw = new_path + (sep + anchor if sep else "")
+ if new_raw == raw:
+ continue
+ new_link = "[[file:%s]%s]" % (new_raw, "[%s]" % desc if desc is not None else "")
+ if m.group(0) != new_link:
+ edits.setdefault(linker, []).append((m.group(0), new_link))
+ return edits, ambiguous
+
+
+def scan_bare_mentions(root, moved):
+ """Bare-path mentions of moving docs in the rewritten roots — text
+ occurrences outside any [[...]] link. Reported, never rewritten."""
+ found = []
+ for base in REWRITE_ROOTS:
+ if not os.path.exists(os.path.join(root, base)):
+ continue
+ for rel in walk_files(root, base):
+ content = read_text(os.path.join(root, rel))
+ if content is None:
+ continue
+ for i, line in enumerate(content.splitlines(), 1):
+ stripped = re.sub(r"\[\[[^\]]*\](?:\[[^\]]*\])?\]", "", line)
+ for src in moved:
+ if src in stripped:
+ found.append((rel, i, src))
+ return found
+
+
+def scan_report_only(root, moved):
+ """Occurrences of moving docs in frozen/synced surfaces."""
+ reports = []
+ for base in REPORT_ROOTS:
+ if not os.path.exists(os.path.join(root, base)):
+ continue
+ for rel in walk_files(root, base):
+ content = read_text(os.path.join(root, rel))
+ if content is None:
+ continue
+ for src in moved:
+ if src in content:
+ if rel.startswith(SYNCED_PREFIX):
+ note = ("synced template, not rewritten — a local edit is reverted by the "
+ "next sync; edit the canonical claude-templates/%s instead" % rel)
+ else:
+ note = "frozen history; not rewritten"
+ reports.append((rel, src, note))
+ return reports
+
+
+# ---- Content transforms -----------------------------------------------
+
+
+def transform_spec(content, keyword, reason, title, doc_id, link_edits):
+ """Apply the retrofit rewrite to a moving spec's content: two-sequence
+ keyword header, prepended status heading, Status-field mirror, and the
+ doc's own link edits."""
+ for old, new in link_edits:
+ content = content.replace(old, new)
+ lines = content.splitlines()
+
+ todo_idx = None
+ kept = []
+ for line in lines:
+ if line.startswith("#+TODO:"):
+ if todo_idx is None:
+ todo_idx = len(kept)
+ continue
+ kept.append(line)
+ lines = kept
+ if todo_idx is None:
+ todo_idx = 0
+ while todo_idx < len(lines) and lines[todo_idx].startswith("#+"):
+ todo_idx += 1
+ lines[todo_idx:todo_idx] = TODO_HEADER
+
+ head_end = 0
+ while head_end < len(lines) and (lines[head_end].startswith("#+") or not lines[head_end].strip()):
+ head_end += 1
+ ts = datetime.now().astimezone().strftime("%Y-%m-%d %a @ %H:%M:%S %z")
+ provenance = "reason: %s" % reason if reason else "evidence-based, human-confirmed"
+ block = [
+ "* %s %s" % (keyword, title),
+ ":PROPERTIES:",
+ ":ID: %s" % doc_id,
+ ":END:",
+ "- %s — retrofitted by spec-sort; status set to %s (%s)" % (ts, keyword, provenance),
+ "",
+ ]
+ lines[head_end:head_end] = block
+
+ out = []
+ mirrored = False
+ for line in lines:
+ m = re.match(r"^(\|\s*Status\s*\|)([^|]*)(\|.*)$", line, re.IGNORECASE)
+ if m and not mirrored:
+ value = " %s" % keyword.lower()
+ width = len(m.group(2))
+ line = m.group(1) + (value.ljust(width) if len(value) <= width else value + " ") + m.group(3)
+ mirrored = True
+ out.append(line)
+ return "\n".join(out) + "\n"
+
+
+def title_for(content, rel):
+ m = re.search(r"^#\+TITLE:\s*(.+)$", content, re.MULTILINE | re.IGNORECASE)
+ if m:
+ return m.group(1).strip()
+ base = os.path.basename(rel)[: -len(".org")]
+ return base[: -len("-spec")] if base.endswith("-spec") else base
+
+
+# ---- Marker ------------------------------------------------------------
+
+
+def stamp_marker(root, date):
+ path = os.path.join(root, ".ai", "notes.org")
+ os.makedirs(os.path.dirname(path), exist_ok=True)
+ content = read_text(path) or ""
+ line = ":LAST_SPEC_SORT: %s" % date
+ if ":LAST_SPEC_SORT:" in content:
+ content = re.sub(r":LAST_SPEC_SORT:.*", line, content, count=1)
+ elif re.search(r"^\* Workflow State\s*$", content, re.MULTILINE):
+ content = re.sub(r"(^\* Workflow State\s*$)", r"\1\n" + line, content, count=1, flags=re.MULTILINE)
+ else:
+ if content and not content.endswith("\n"):
+ content += "\n"
+ content += "\n* Workflow State\n\n%s\n" % line
+ with open(path, "w", encoding="utf-8") as f:
+ f.write(content)
+
+
+# ---- Apply -------------------------------------------------------------
+
+
+class ApplyFailure(Exception):
+ """Mid-apply failure: args are (applied_labels, remaining_ops, cause)."""
+
+
+def apply_plan(root, plan, fail_after):
+ """Execute the recorded plan. Returns the applied-op labels; raises
+ ApplyFailure mid-way on a write error or when the test hook fires."""
+ ops = []
+ for mv in plan["moves"]:
+ ops.append(("move", mv))
+ for linker, edits in plan["link_edits"].items():
+ if linker in {mv["src"] for mv in plan["moves"]}:
+ continue # a moving doc's own edits ride along in its transform
+ ops.append(("relink", (linker, edits)))
+
+ applied = []
+ specs_dir = os.path.join(root, "docs", "specs")
+ if plan["moves"] and not os.path.isdir(specs_dir):
+ os.makedirs(specs_dir)
+ plan["created_dirs"].append(os.path.join("docs", "specs"))
+
+ for n, (kind, payload) in enumerate(ops, 1):
+ if fail_after and n > fail_after:
+ raise ApplyFailure(applied, ops[n - 1:], "injected test failure")
+ try:
+ if kind == "move":
+ mv = payload
+ content = read_text(os.path.join(root, mv["src"]))
+ new = transform_spec(content, mv["keyword"], mv["reason"], mv["title"], mv["id"],
+ plan["link_edits"].get(mv["src"], []))
+ with open(os.path.join(root, mv["dest"]), "w", encoding="utf-8") as f:
+ f.write(new)
+ os.remove(os.path.join(root, mv["src"]))
+ applied.append("move %s -> %s" % (mv["src"], mv["dest"]))
+ else:
+ linker, edits = payload
+ path = os.path.join(root, linker)
+ content = read_text(path)
+ for old, new in edits:
+ content = content.replace(old, new)
+ with open(path, "w", encoding="utf-8") as f:
+ f.write(content)
+ applied.append("relink %s (%d link%s)" % (linker, len(edits), "s" if len(edits) != 1 else ""))
+ except OSError as exc:
+ raise ApplyFailure(applied, ops[n - 1:], str(exc))
+ return applied
+
+
+def residue_check(root, plan):
+ """Post-apply: no link in the rewritten roots may still resolve to an
+ old path; bare mentions beyond the acknowledged set fail too."""
+ moved = {mv["src"]: mv["dest"] for mv in plan["moves"]}
+ residue = []
+ for linker in rewrite_files(root):
+ content = read_text(os.path.join(root, linker))
+ if content is None:
+ continue
+ for m in LINK_RE.finditer(content):
+ target_path = m.group(1).partition("::")[0]
+ target = resolve_target(root, linker, target_path, {})
+ if target in moved:
+ residue.append("%s: link still resolves to %s" % (linker, target))
+ # Acknowledged mentions were recorded pre-apply; a mention inside a moved
+ # doc now lives at the doc's destination, so map the file side through the
+ # moves before comparing.
+ acknowledged = {(moved.get(f, f), src) for f, _ln, src in plan["bare"]}
+ for f, ln, src in scan_bare_mentions(root, moved):
+ if (f, src) not in acknowledged:
+ residue.append("%s:%d: bare mention of %s" % (f, ln, src))
+ return residue
+
+
+def print_recovery(plan, applied, not_applied):
+ print("FAILURE — the apply did not complete.")
+ print(" applied:")
+ for a in applied or ["(nothing)"]:
+ print(" %s" % a)
+ print(" not applied:")
+ for kind, payload in not_applied:
+ if kind == "move":
+ print(" move %s -> %s" % (payload["src"], payload["dest"]))
+ else:
+ print(" relink %s" % payload[0])
+ print("RECOVERY — restore the pre-run state (safe: preflight required a clean tree):")
+ touched = [mv["src"] for mv in plan["moves"]] + [l for l in plan["link_edits"] if l not in {mv["src"] for mv in plan["moves"]}]
+ print(" git restore -- %s" % " ".join(touched))
+ created = [mv["dest"] for mv in plan["moves"]]
+ print(" rm -f -- %s # git restore can't remove the created copies" % " ".join(created))
+ for d in plan.get("created_dirs", []):
+ print(" rmdir --ignore-fail-on-non-empty -- %s" % d)
+
+
+# ---- Main ---------------------------------------------------------------
+
+
+def parse_kv(pairs, label):
+ out = {}
+ for item in pairs or []:
+ if "=" not in item:
+ sys.exit("spec-sort: %s expects REL=VALUE, got %r" % (label, item))
+ k, v = item.split("=", 1)
+ out[os.path.normpath(k)] = v
+ return out
+
+
+def main():
+ ap = argparse.ArgumentParser(prog="spec-sort", add_help=True)
+ ap.add_argument("--project-root", default=".")
+ ap.add_argument("--apply", action="store_true")
+ ap.add_argument("--allow-dirty", action="store_true")
+ ap.add_argument("--acknowledge-bare", action="store_true")
+ ap.add_argument("--confirm", action="append", metavar="REL=KEYWORD")
+ ap.add_argument("--reason", action="append", metavar="REL=TEXT")
+ ap.add_argument("--skip", action="append", metavar="REL")
+ ap.add_argument("--plan-file")
+ args = ap.parse_args()
+
+ root = os.path.abspath(args.project_root)
+ confirms = parse_kv(args.confirm, "--confirm")
+ reasons = parse_kv(args.reason, "--reason")
+ skips = {os.path.normpath(s) for s in (args.skip or [])}
+
+ candidates, anomalies, notes = classify(root)
+ if not candidates and not anomalies and not notes and not os.path.isdir(os.path.join(root, "docs")):
+ return 0 # no docs pile at all — silent no-op
+
+ for named in list(confirms) + list(skips) + list(reasons):
+ if named not in candidates:
+ print("spec-sort: %s is not a spec candidate" % named)
+ return 1
+ for rel, kw in confirms.items():
+ if kw not in LIFECYCLE:
+ print("spec-sort: %r is not a lifecycle keyword (%s)" % (kw, " ".join(LIFECYCLE)))
+ return 1
+
+ # ---- Build the plan (shared by report and apply) ----
+ moves = []
+ for rel in candidates:
+ if rel in skips:
+ continue
+ if args.apply and rel not in confirms:
+ continue # gate failure reported below
+ content = read_text(os.path.join(root, rel))
+ moves.append({
+ "src": rel,
+ "dest": dest_for(rel),
+ "keyword": confirms.get(rel, None),
+ "reason": reasons.get(rel),
+ "title": title_for(content, rel),
+ "id": str(uuid.uuid4()),
+ })
+ moved_map = {mv["src"]: mv["dest"] for mv in moves}
+ link_edits, ambiguous = plan_link_edits(root, moved_map)
+ bare = scan_bare_mentions(root, moved_map)
+ reports = scan_report_only(root, moved_map)
+
+ # ---- Report ----
+ for rel in candidates:
+ content = read_text(os.path.join(root, rel))
+ ev = gather_evidence(root, rel, content)
+ proposed = propose_keyword(ev)
+ print("CANDIDATE %s -> %s" % (rel, dest_for(rel)))
+ suffix = " (terminal — requires --reason to apply)" if proposed in TERMINAL else ""
+ print(" proposed keyword: %s%s" % (proposed, suffix))
+ print(" evidence:")
+ print(" status field: %s" % (ev["status"] or "(none)"))
+ print(" cookies: %s" % ("; ".join(ev["cookies"]) or "(none)"))
+ print(" todo.org: %s" % (ev["todo"] or "(no linking task)"))
+ print(" history: %s" % (ev["history"] or "(none)"))
+ n_exist, artifacts = ev["artifacts"]
+ if artifacts:
+ print(" artifacts: %d/%d named paths exist (%s)" % (n_exist, len(artifacts), ", ".join(artifacts)))
+ else:
+ print(" artifacts: (none named)")
+ for rel in anomalies:
+ print("ANOMALY %s: named -spec.org but lacks the spec spine (Decisions + Implementation phases); surfaced, not moved" % rel)
+ for rel in notes:
+ print("NOTE %s" % rel)
+ for linker, edits in sorted(link_edits.items()):
+ for old, new in edits:
+ print("RELINK %s: %s -> %s" % (linker, old, new))
+ for a in ambiguous:
+ print("AMBIGUOUS %s" % a)
+ for f, ln, src in bare:
+ print("BARE-PATH %s:%d: %s (reported for manual handling, never rewritten)" % (f, ln, src))
+ for rel, src, note in reports:
+ print("REPORT %s: reference to %s (%s)" % (rel, src, note))
+
+ if not args.apply:
+ if candidates or anomalies or notes:
+ print("DRY RUN — no changes written. Pass --apply with per-candidate --confirm/--skip to execute.")
+ return 0
+
+ # ---- Apply: preflight ----
+ try:
+ porcelain = subprocess.run(
+ ["git", "status", "--porcelain"], cwd=root,
+ capture_output=True, text=True, check=True,
+ ).stdout
+ except (subprocess.CalledProcessError, FileNotFoundError):
+ print("spec-sort: --apply needs a git worktree (recovery depends on git restore)")
+ return 2
+ if porcelain.strip():
+ dirty = [ln[3:] for ln in porcelain.splitlines()]
+ if not args.allow_dirty:
+ print("spec-sort: refusing --apply on a dirty worktree (%d path%s). Commit or stash first, or pass --allow-dirty."
+ % (len(dirty), "s" if len(dirty) != 1 else ""))
+ return 2
+ print("WARNING --allow-dirty: recovery via git restore would also revert your pre-existing uncommitted changes:")
+ for p in dirty:
+ print(" %s" % p)
+
+ # ---- Apply: confirm gate ----
+ unaddressed = [rel for rel in candidates if rel not in confirms and rel not in skips]
+ if unaddressed:
+ print("spec-sort: unconfirmed candidate(s) — pass --confirm REL=KEYWORD or --skip REL for each:")
+ for rel in unaddressed:
+ print(" %s" % rel)
+ return 1
+ for mv in moves:
+ if mv["keyword"] in TERMINAL and not mv["reason"]:
+ print("spec-sort: %s -> %s is a terminal state and requires an explicit --reason %s=TEXT"
+ % (mv["src"], mv["keyword"], mv["src"]))
+ return 1
+
+ # ---- Apply: validation ----
+ problems = []
+ dests = {}
+ for mv in moves:
+ if os.path.exists(os.path.join(root, mv["dest"])):
+ problems.append("%s: destination exists (%s)" % (mv["src"], mv["dest"]))
+ if mv["dest"] in dests:
+ problems.append("%s and %s: destination exists twice (%s)" % (mv["src"], dests[mv["dest"]], mv["dest"]))
+ dests[mv["dest"]] = mv["src"]
+ for a in ambiguous:
+ problems.append("ambiguous link: %s" % a)
+ if bare and not args.acknowledge_bare:
+ problems.append("bare-path mention(s) listed above need manual handling — re-run with --acknowledge-bare to proceed without rewriting them")
+ if problems:
+ print("spec-sort: validation blocked — nothing written:")
+ for p in problems:
+ print(" %s" % p)
+ return 1
+
+ # ---- Apply: record the plan, then execute from it ----
+ today = datetime.now().astimezone().strftime("%Y-%m-%d")
+ plan = {
+ "root": root, "date": today, "moves": moves,
+ "link_edits": link_edits, "bare": bare,
+ "reports": [list(r) for r in reports], "created_dirs": [],
+ }
+ plan_path = args.plan_file or os.path.join(
+ tempfile.gettempdir(), "spec-sort-plan-%s.json" % os.path.basename(root))
+ with open(plan_path, "w", encoding="utf-8") as f:
+ json.dump(plan, f, indent=2)
+ print("plan written: %s" % plan_path)
+
+ fail_after = int(os.environ.get("SPEC_SORT_INJECT_FAIL_AFTER", "0") or 0)
+ try:
+ applied = apply_plan(root, plan, fail_after)
+ except ApplyFailure as exc:
+ print("write failed: %s" % exc.args[2])
+ print_recovery(plan, exc.args[0], exc.args[1])
+ return 1
+
+ residue = residue_check(root, plan)
+ if residue:
+ print("spec-sort: residue after apply — old paths still referenced:")
+ for r in residue:
+ print(" %s" % r)
+ print_recovery(plan, applied, [])
+ return 1
+
+ stamp_marker(root, today)
+ for a in applied:
+ print("applied: %s" % a)
+ print("spec-sort: done — %d spec(s) sorted, :LAST_SPEC_SORT: %s stamped" % (len(moves), today))
+ return 0
+
+
+if __name__ == "__main__":
+ sys.exit(main())
diff --git a/.ai/scripts/task-review-staleness.sh b/.ai/scripts/task-review-staleness.sh
index ed43712..50e0257 100755
--- a/.ai/scripts/task-review-staleness.sh
+++ b/.ai/scripts/task-review-staleness.sh
@@ -19,9 +19,14 @@
# deeper headings, and cookie-less headings are not review units.
#
# A task is stale (count mode), and sorts oldest (list mode), when its
-# :LAST_REVIEWED: property is missing or unparseable (NIL sorts first), or
-# when its age strictly exceeds the threshold (age > N days; age == N is
-# still fresh).
+# :LAST_REVIEWED: property is missing (NIL sorts first) or when its age
+# strictly exceeds the threshold (age > N days; age == N is still fresh).
+#
+# :LAST_REVIEWED: accepts a bare date (2026-07-09) or an org-native
+# timestamp ([2026-07-09 Thu] or <2026-07-09 Thu ...>); both normalize to
+# the ISO date. A value that is present but parses to neither is a data
+# error: the script warns loudly to stderr (file:line:value) and leaves it
+# out of the stale count rather than silently treating it as never-reviewed.
set -euo pipefail
@@ -46,7 +51,7 @@ num="$2"
# only the property drawer between a qualifying heading and the next heading
# is scanned.
extract_tasks() {
- awk '
+ awk -v fname="$todo_file" '
function flush() {
if (in_task) printf "%s\t%s\t%s\n", hline, (have_lr ? lr : "NONE"), heading
}
@@ -65,7 +70,21 @@ extract_tasks() {
v = $0
sub(/^[ \t]*:LAST_REVIEWED:[ \t]*/, "", v)
sub(/[ \t]*$/, "", v)
- lr = v; have_lr = 1
+ raw = v
+ # Accept org-native stamps: strip a leading [ or < (inactive/active
+ # timestamp bracket) and take the leading ISO date, so [2026-07-09 Thu]
+ # and 2026-07-09 both normalize to 2026-07-09.
+ sub(/^[[<]/, "", v)
+ if (match(v, /^[0-9][0-9][0-9][0-9]-[0-9][0-9]-[0-9][0-9]/)) {
+ lr = substr(v, RSTART, RLENGTH); have_lr = 1
+ } else {
+ # Present but unparseable — a data error, not "never reviewed". Warn
+ # loudly (file:line:value) instead of silently folding it into the
+ # stale count, where a re-review in the same bad format would never
+ # drop the number and nothing would explain why.
+ printf "task-review-staleness: %s:%d: unparseable :LAST_REVIEWED: %s (expected YYYY-MM-DD or [YYYY-MM-DD Day])\n", fname, NR, raw > "/dev/stderr"
+ lr = "INVALID"; have_lr = 1
+ }
next
}
END { flush() }
@@ -98,9 +117,16 @@ while IFS=$'\t' read -r hline value heading; do
continue
fi
- # Unparseable date → treat as NIL (stale).
+ # Malformed stamp: extract_tasks already warned. A data error is not
+ # "never reviewed", so it stays out of the stale count.
+ if [ "$value" = "INVALID" ]; then
+ continue
+ fi
+
+ # A shape-valid but impossible date (e.g. 2026-13-45) slips past the awk
+ # regex; warn and skip rather than silently counting it.
if ! rev_epoch=$(date -d "$value" +%s 2>/dev/null); then
- count=$((count + 1))
+ echo "task-review-staleness: $todo_file: unparseable :LAST_REVIEWED: $value" >&2
continue
fi
diff --git a/.ai/scripts/tests/agent-lock.bats b/.ai/scripts/tests/agent-lock.bats
new file mode 100644
index 0000000..dbcffe1
--- /dev/null
+++ b/.ai/scripts/tests/agent-lock.bats
@@ -0,0 +1,214 @@
+#!/usr/bin/env bats
+#
+# Tests for claude-templates/.ai/scripts/agent-lock — a mkdir-atomic advisory
+# lock helper for agent workflows (sentry's single-runner and roam-write
+# locks). flock can't span an agent's tool calls: every Bash call is its own
+# short-lived shell, so a flock dies with the call that took it. This helper
+# persists the lock on disk between calls and self-clears after a crash via
+# age-based staleness reclaim.
+#
+# Contract under test:
+# agent-lock acquire <name> [--ttl=SECONDS] [--wait[=SECONDS]]
+# exit 0 → acquired (fresh, or reclaimed from a stale prior holder).
+# exit 1 → busy: a live lock holds <name>; deferred (note on stderr).
+# exit 2 → usage error (bad/absent name, unknown subcommand).
+# agent-lock refresh <name> → re-touch a held lock (heartbeat); exit 1 if absent.
+# agent-lock release <name> → remove the lock; idempotent (exit 0 if already free).
+# agent-lock status <name> → print free|held|stale + metadata; exit 0 (query).
+# agent-lock path <name> → print the resolved lock dir path; does not create it.
+#
+# Staleness is age-based on the metadata file's mtime versus the lock's own
+# recorded TTL, so a crashed holder's lock expires instead of wedging every
+# later acquire. Heartbeat (refresh) re-touches the mtime, keeping a live
+# holder's lock young. Every reclaim surfaces a note (never silent).
+#
+# Lock home: /run/user/<uid>/agent-locks/<name>/ (tmpfs: host-local, out of
+# every repo, cleared on reboot), with ~/.cache/agent-locks/ as the fallback
+# where no runtime dir exists. AGENT_LOCK_DIR overrides the base for tests and
+# advanced callers; the helper otherwise owns the path scheme and callers pass
+# only names.
+#
+# Strategy: AGENT_LOCK_DIR points every lock at a temp base, so tests never
+# touch a real runtime dir. Staleness is exercised by aging the metadata
+# file's mtime with `touch` rather than sleeping.
+
+SCRIPT="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/agent-lock"
+BASH_BIN="$(command -v bash)"
+
+setup() {
+ TEST_DIR="$(mktemp -d -t agent-lock-bats.XXXXXX)"
+ LOCK_BASE="$TEST_DIR/locks"
+}
+
+teardown() {
+ rm -rf "$TEST_DIR"
+}
+
+lock() {
+ run env AGENT_LOCK_DIR="$LOCK_BASE" "$BASH_BIN" "$SCRIPT" "$@"
+}
+
+# meta-file path for a lock name, for direct inspection / aging.
+meta_of() {
+ printf '%s/%s/meta\n' "$LOCK_BASE" "$1"
+}
+
+# ---- acquire: fresh win + metadata --------------------------------------
+
+@test "acquire: fresh name wins (exit 0) and writes pid/host/timestamp/ttl" {
+ lock acquire job
+ [ "$status" -eq 0 ]
+ local meta; meta="$(meta_of job)"
+ [ -f "$meta" ]
+ grep -q "^pid=$$\|^pid=[0-9][0-9]*$" "$meta"
+ grep -q "^host=$(uname -n)$" "$meta"
+ grep -qE "^acquired=[0-9]{4}-[0-9]{2}-[0-9]{2}T" "$meta"
+ grep -qE "^ttl=[0-9]+$" "$meta"
+}
+
+@test "acquire: honors an explicit --ttl in the metadata" {
+ lock acquire job --ttl=45
+ [ "$status" -eq 0 ]
+ grep -q "^ttl=45$" "$(meta_of job)"
+}
+
+# ---- acquire: contention (one winner) -----------------------------------
+
+@test "acquire: a second acquire of a live lock defers (exit 1, note)" {
+ lock acquire job
+ [ "$status" -eq 0 ]
+ lock acquire job
+ [ "$status" -eq 1 ]
+ [[ "$output" == *job* ]]
+}
+
+@test "acquire: two racing acquires yield exactly one winner" {
+ # Fire both without releasing; exactly one mkdir wins.
+ env AGENT_LOCK_DIR="$LOCK_BASE" "$BASH_BIN" "$SCRIPT" acquire race & p1=$!
+ env AGENT_LOCK_DIR="$LOCK_BASE" "$BASH_BIN" "$SCRIPT" acquire race & p2=$!
+ local r1=0 r2=0
+ wait $p1 || r1=$?
+ wait $p2 || r2=$?
+ # One exits 0 (won), one exits 1 (deferred).
+ [ "$((r1 + r2))" -eq 1 ]
+}
+
+# ---- release: frees the lock --------------------------------------------
+
+@test "release: frees a held lock so the next acquire wins" {
+ lock acquire job
+ [ "$status" -eq 0 ]
+ lock release job
+ [ "$status" -eq 0 ]
+ [ ! -d "$LOCK_BASE/job" ]
+ lock acquire job
+ [ "$status" -eq 0 ]
+}
+
+@test "release: is idempotent on an already-free lock (exit 0)" {
+ lock release never-held
+ [ "$status" -eq 0 ]
+}
+
+# ---- staleness reclaim (surfaced, never silent) -------------------------
+
+@test "acquire: reclaims a stale lock and surfaces the reclaim note" {
+ lock acquire job --ttl=1
+ [ "$status" -eq 0 ]
+ # Age the metadata mtime well past the 1s TTL.
+ touch -d '1 hour ago' "$(meta_of job)"
+ lock acquire job --ttl=1
+ [ "$status" -eq 0 ]
+ [[ "$output" == *reclaim* ]]
+ [[ "$output" == *job* ]]
+ # The reclaim installed fresh metadata (young again), not the aged holder's.
+ lock status job
+ [[ "$output" == *held* ]]
+ [[ "$output" != *stale* ]]
+}
+
+@test "acquire: a lock inside its TTL is not stale (stays deferred)" {
+ lock acquire job --ttl=3600
+ [ "$status" -eq 0 ]
+ lock acquire job --ttl=3600
+ [ "$status" -eq 1 ]
+}
+
+# ---- heartbeat (refresh keeps a live lock young) ------------------------
+
+@test "refresh: re-touches a held lock so it is no longer stale" {
+ lock acquire job --ttl=1
+ [ "$status" -eq 0 ]
+ touch -d '1 hour ago' "$(meta_of job)"
+ lock status job
+ [[ "$output" == *stale* ]]
+ lock refresh job
+ [ "$status" -eq 0 ]
+ lock status job
+ [[ "$output" == *held* ]]
+ [[ "$output" != *stale* ]]
+}
+
+@test "refresh: an absent lock cannot be refreshed (exit 1)" {
+ lock refresh nothing
+ [ "$status" -eq 1 ]
+}
+
+# ---- status query -------------------------------------------------------
+
+@test "status: reports free for an unheld lock (exit 0)" {
+ lock status job
+ [ "$status" -eq 0 ]
+ [[ "$output" == *free* ]]
+}
+
+@test "status: reports held with metadata for a live lock" {
+ lock acquire job --ttl=3600
+ lock status job
+ [ "$status" -eq 0 ]
+ [[ "$output" == *held* ]]
+ [[ "$output" == *"host=$(uname -n)"* ]]
+}
+
+# ---- path resolution: runtime dir home with cache fallback --------------
+
+@test "path: resolves under AGENT_LOCK_DIR when set" {
+ lock path job
+ [ "$status" -eq 0 ]
+ [ "$output" = "$LOCK_BASE/job" ]
+ [ ! -d "$LOCK_BASE/job" ] # path does not create the lock
+}
+
+@test "path: prefers the runtime dir home when no override is set" {
+ local rt="$TEST_DIR/run"
+ mkdir -p "$rt"
+ run env -u AGENT_LOCK_DIR XDG_RUNTIME_DIR="$rt" "$BASH_BIN" "$SCRIPT" path job
+ [ "$status" -eq 0 ]
+ [ "$output" = "$rt/agent-locks/job" ]
+}
+
+@test "path: falls back to the cache home when no runtime dir exists" {
+ local home="$TEST_DIR/home"
+ mkdir -p "$home"
+ run env -u AGENT_LOCK_DIR -u XDG_RUNTIME_DIR -u XDG_CACHE_HOME \
+ HOME="$home" "$BASH_BIN" "$SCRIPT" path job
+ [ "$status" -eq 0 ]
+ [ "$output" = "$home/.cache/agent-locks/job" ]
+}
+
+# ---- usage errors -------------------------------------------------------
+
+@test "usage: a missing name is a usage error (exit 2)" {
+ lock acquire
+ [ "$status" -eq 2 ]
+}
+
+@test "usage: a name with a slash is rejected (exit 2)" {
+ lock acquire bad/name
+ [ "$status" -eq 2 ]
+}
+
+@test "usage: an unknown subcommand is a usage error (exit 2)" {
+ lock frobnicate job
+ [ "$status" -eq 2 ]
+}
diff --git a/.ai/scripts/tests/agent-roster.bats b/.ai/scripts/tests/agent-roster.bats
new file mode 100644
index 0000000..939a7df
--- /dev/null
+++ b/.ai/scripts/tests/agent-roster.bats
@@ -0,0 +1,141 @@
+#!/usr/bin/env bats
+# Tests for agent-roster: report other live Claude agents in a project.
+#
+# pgrep and /proc are the system boundary, so the test injects both and runs
+# the real include/exclude logic against fixtures — no Claude processes are
+# spawned. Injection points:
+# ROSTER_PGREP command standing in for pgrep (a stub printing $FAKE_PIDS)
+# ROSTER_PROC proc dir (a fixture of <pid>/cwd symlinks + <pid>/status)
+# ROSTER_SELF_PID the scanner's own pid, so the ancestry walk is testable
+#
+# Fixture process tree: pid 1000 (the scanner) is a child of 999 (the current
+# session's claude), which is a child of init (1). So 999 must always be
+# excluded as scanner ancestry; other claude pids are judged by cwd.
+
+setup() {
+ SCRIPT_DIR="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)"
+ ROSTER="$SCRIPT_DIR/agent-roster"
+
+ PROC="$BATS_TEST_TMPDIR/proc"
+ ROOT="$BATS_TEST_TMPDIR/project"
+ mkdir -p "$PROC" "$ROOT/sub" "$BATS_TEST_TMPDIR/elsewhere"
+
+ PGREP_STUB="$BATS_TEST_TMPDIR/pgrep"
+ cat >"$PGREP_STUB" <<'EOF'
+#!/usr/bin/env bash
+printf '%s\n' $FAKE_PIDS
+EOF
+ chmod +x "$PGREP_STUB"
+
+ SELF=1000
+ # scanner (1000) <- session claude (999) <- init (1)
+ mkproc 1000 "$ROOT" 999
+ mkproc 999 "$ROOT" 1
+}
+
+# mkproc PID CWD PPID — register a fake process in the fixture proc dir.
+mkproc() {
+ mkdir -p "$PROC/$1"
+ ln -sf "$2" "$PROC/$1/cwd"
+ printf 'PPid:\t%s\n' "$3" >"$PROC/$1/status"
+}
+
+run_roster() {
+ ROSTER_PGREP="$PGREP_STUB" ROSTER_PROC="$PROC" ROSTER_SELF_PID="$SELF" \
+ FAKE_PIDS="$FAKE_PIDS" run "$ROSTER" "$ROOT"
+}
+
+@test "agent-roster: alone (only the session's own claude) exits 0, no output" {
+ FAKE_PIDS="999"
+ run_roster
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "agent-roster: one other agent in-project is printed, exit 1" {
+ mkproc 2000 "$ROOT" 1
+ FAKE_PIDS="999 2000"
+ run_roster
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"2000"* ]]
+ [[ "$output" == *"$ROOT"* ]]
+}
+
+@test "agent-roster: two other agents both printed, exit 1" {
+ mkproc 2000 "$ROOT" 1
+ mkproc 2001 "$ROOT/sub" 1
+ FAKE_PIDS="999 2000 2001"
+ run_roster
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"2000"* ]]
+ [[ "$output" == *"2001"* ]]
+ [ "${#lines[@]}" -eq 2 ]
+}
+
+@test "agent-roster: the scanner's session-claude ancestor is excluded even with matching cwd" {
+ # 999 has cwd == ROOT but is scanner ancestry; must not appear.
+ FAKE_PIDS="999"
+ run_roster
+ [ "$status" -eq 0 ]
+ [[ "$output" != *"999"* ]]
+}
+
+@test "agent-roster: cwd outside the project root is excluded" {
+ mkproc 3000 "$BATS_TEST_TMPDIR/elsewhere" 1
+ FAKE_PIDS="999 3000"
+ run_roster
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "agent-roster: cwd in a subdirectory of root is included" {
+ mkproc 2002 "$ROOT/sub" 1
+ FAKE_PIDS="999 2002"
+ run_roster
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"2002"* ]]
+}
+
+@test "agent-roster: a sibling path sharing a prefix is not a false match" {
+ # ROOT is .../project; .../project-other must not count as inside it.
+ mkdir -p "$BATS_TEST_TMPDIR/project-other"
+ mkproc 3100 "$BATS_TEST_TMPDIR/project-other" 1
+ FAKE_PIDS="999 3100"
+ run_roster
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "agent-roster: a pid that vanished between pgrep and the proc read is skipped" {
+ # 4000 has no fixture dir, simulating a process gone by readlink time.
+ FAKE_PIDS="999 4000"
+ run_roster
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "agent-roster: missing proc reports unavailable on stderr, exit 2, never silent-alone" {
+ ROSTER_PGREP="$PGREP_STUB" ROSTER_PROC="$BATS_TEST_TMPDIR/nonexistent" \
+ ROSTER_SELF_PID="$SELF" FAKE_PIDS="999 2000" run "$ROSTER" "$ROOT"
+ [ "$status" -eq 2 ]
+ [[ "$output" == *"roster unavailable"* ]]
+}
+
+@test "agent-roster: a missing pgrep reports unavailable, exit 2, never silent-alone" {
+ # If pgrep itself is absent, the scan can't run; reporting "alone" would be a
+ # false negative the "never silent-alone" invariant forbids.
+ mkproc 2000 "$ROOT" 1
+ ROSTER_PGREP="$BATS_TEST_TMPDIR/no-such-pgrep" ROSTER_PROC="$PROC" \
+ ROSTER_SELF_PID="$SELF" FAKE_PIDS="999 2000" run "$ROSTER" "$ROOT"
+ [ "$status" -eq 2 ]
+ [[ "$output" == *"roster unavailable"* ]]
+}
+
+@test "agent-roster: defaults project root to PWD when no argument is given" {
+ mkproc 2000 "$ROOT" 1
+ FAKE_PIDS="999 2000"
+ ROSTER_PGREP="$PGREP_STUB" ROSTER_PROC="$PROC" ROSTER_SELF_PID="$SELF" \
+ FAKE_PIDS="$FAKE_PIDS" run env -C "$ROOT" "$ROSTER"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"2000"* ]]
+}
diff --git a/.ai/scripts/tests/capture-guard.bats b/.ai/scripts/tests/capture-guard.bats
new file mode 100644
index 0000000..31632a4
--- /dev/null
+++ b/.ai/scripts/tests/capture-guard.bats
@@ -0,0 +1,130 @@
+#!/usr/bin/env bats
+#
+# Tests for claude-templates/.ai/scripts/capture-guard — detects live
+# org-capture buffers visiting a target file before a workflow edits that
+# file on disk (the roam inbox, in inbox.org roam mode Phase D). Editing the file
+# underneath an indirect org-capture buffer wedges the capture (see emacs.md).
+#
+# Contract under test:
+# capture-guard [TARGET_FILE] (default TARGET_FILE = ~/org/roam/inbox.org)
+# exit 0 → safe to edit: emacsclient absent, daemon unreachable, or no
+# capture buffer visits TARGET_FILE.
+# exit 1 → a live capture buffer visits TARGET_FILE; its name(s) printed.
+#
+# Strategy: the emacsclient boundary is mocked with a PATH stub. The stub
+# answers the reachability probe (`-e t`) per STUB_REACHABLE and returns a
+# canned, real-emacsclient-shaped result (quoted string) for the buffer query
+# per STUB_BUFS. The script's own quote-stripping and exit logic is the code
+# under test; the file-equal-p precision is real-Emacs behavior we trust.
+
+SCRIPT="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/capture-guard"
+BASH_BIN="$(command -v bash)"
+
+setup() {
+ TEST_DIR="$(mktemp -d -t capture-guard-bats.XXXXXX)"
+ STUB_DIR="$TEST_DIR/bin"
+ mkdir -p "$STUB_DIR"
+
+ cat > "$STUB_DIR/emacsclient" <<'STUB'
+#!/usr/bin/env bash
+# Mock emacsclient. `-e t` is the reachability probe; anything else is the
+# buffer query, answered with the real-emacsclient-shaped quoted string.
+expr="$2"
+if [ "$expr" = "t" ]; then
+ [ "${STUB_REACHABLE:-1}" = "1" ] && { echo t; exit 0; }
+ exit 1
+fi
+printf '%s\n' "${STUB_BUFS:-\"\"}"
+exit 0
+STUB
+ chmod +x "$STUB_DIR/emacsclient"
+
+ EMPTY_DIR="$TEST_DIR/empty"
+ mkdir -p "$EMPTY_DIR"
+}
+
+teardown() {
+ rm -rf "$TEST_DIR"
+}
+
+# ---- Safe-to-edit (exit 0) cases ------------------------------------
+
+@test "capture-guard: emacsclient absent is safe (exit 0, no output)" {
+ run env PATH="$EMPTY_DIR" "$BASH_BIN" "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "capture-guard: daemon unreachable is safe (exit 0)" {
+ run env PATH="$STUB_DIR:$PATH" STUB_REACHABLE=0 "$BASH_BIN" "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "capture-guard: reachable with no capture buffers is safe (exit 0)" {
+ run env PATH="$STUB_DIR:$PATH" STUB_REACHABLE=1 STUB_BUFS='""' "$BASH_BIN" "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+# ---- Blocked (exit 1) cases -----------------------------------------
+
+@test "capture-guard: one live capture buffer blocks (exit 1, name printed)" {
+ run env PATH="$STUB_DIR:$PATH" STUB_REACHABLE=1 STUB_BUFS='"CAPTURE-inbox.org"' \
+ "$BASH_BIN" "$SCRIPT"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"CAPTURE-inbox.org"* ]]
+}
+
+@test "capture-guard: multiple live capture buffers all reported (exit 1)" {
+ run env PATH="$STUB_DIR:$PATH" STUB_REACHABLE=1 \
+ STUB_BUFS='"CAPTURE-inbox.org,CAPTURE-2-inbox.org"' \
+ "$BASH_BIN" "$SCRIPT"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"CAPTURE-inbox.org"* ]]
+ [[ "$output" == *"CAPTURE-2-inbox.org"* ]]
+}
+
+@test "capture-guard: blocked output does not contain stray surrounding quotes" {
+ run env PATH="$STUB_DIR:$PATH" STUB_REACHABLE=1 STUB_BUFS='"CAPTURE-inbox.org"' \
+ "$BASH_BIN" "$SCRIPT"
+ [ "$status" -eq 1 ]
+ [[ "$output" != \"* ]]
+ [[ "$output" != *\" ]]
+}
+
+# ---- Argument handling ----------------------------------------------
+
+@test "capture-guard: accepts an explicit target-file argument" {
+ run env PATH="$STUB_DIR:$PATH" STUB_REACHABLE=1 STUB_BUFS='""' \
+ "$BASH_BIN" "$SCRIPT" "$TEST_DIR/some-other-inbox.org"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+# ---- --wait poll mode -----------------------------------------------
+
+@test "capture-guard --wait: returns 0 instantly when already safe (no sleep)" {
+ SECONDS=0
+ run env PATH="$STUB_DIR:$PATH" STUB_REACHABLE=1 STUB_BUFS='""' \
+ "$BASH_BIN" "$SCRIPT" --wait
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+ [ "$SECONDS" -lt 2 ] # didn't poll-sleep
+}
+
+@test "capture-guard --wait=1: times out to exit 1 when persistently blocked" {
+ # Stub always reports the buffer, so it never clears — the short budget
+ # forces a timeout. Capped sleep keeps this near 1s.
+ run env PATH="$STUB_DIR:$PATH" STUB_REACHABLE=1 STUB_BUFS='"CAPTURE-inbox.org"' \
+ "$BASH_BIN" "$SCRIPT" --wait=1
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"CAPTURE-inbox.org"* ]]
+}
+
+@test "capture-guard --wait=N accepts a target after the flag" {
+ run env PATH="$STUB_DIR:$PATH" STUB_REACHABLE=1 STUB_BUFS='""' \
+ "$BASH_BIN" "$SCRIPT" --wait=1 "$TEST_DIR/some-other-inbox.org"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
diff --git a/.ai/scripts/tests/flashcard-sync.bats b/.ai/scripts/tests/flashcard-sync.bats
index 608a280..e6ffc21 100644
--- a/.ai/scripts/tests/flashcard-sync.bats
+++ b/.ai/scripts/tests/flashcard-sync.bats
@@ -6,6 +6,7 @@
setup() {
SCRIPT_DIR="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)"
SYNC="$SCRIPT_DIR/flashcard-sync"
+ STATS="$SCRIPT_DIR/flashcard-stats.py"
TMP="$(mktemp -d)"
}
@@ -36,3 +37,27 @@ EOF
[ "$status" -eq 1 ]
[ ! -f "$HOME/sync/phone/anki/dirty.apkg" ]
}
+
+@test "flashcard-stats: a multi-tagged :fundamental:drill: card still counts" {
+ # Regression guard: a curated card carrying a second org tag must not drop
+ # from the count. A :drill:$ anchor would have counted only one card here.
+ cat > "$TMP/multitag.org" <<'EOF'
+#+TITLE: Multitag Test
+
+* Orbital Regimes
+** What is LEO? :fundamental:drill:
+:PROPERTIES:
+:ID: c1
+:END:
+Low Earth Orbit is the region below about 2000 kilometers.
+** What is GEO? :drill:
+:PROPERTIES:
+:ID: c2
+:END:
+Geostationary orbit sits at roughly 35786 kilometers of altitude.
+EOF
+ run python3 "$STATS" "$TMP/multitag.org"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"Cards: 2"* ]]
+ [[ "$output" == *clean* ]]
+}
diff --git a/.ai/scripts/tests/inbox-status.bats b/.ai/scripts/tests/inbox-status.bats
index bc8a734..27a497e 100644
--- a/.ai/scripts/tests/inbox-status.bats
+++ b/.ai/scripts/tests/inbox-status.bats
@@ -45,6 +45,18 @@ teardown() {
[[ "$output" == *"0 pending"* ]]
}
+@test "inbox-status: ignores an in-flight .inbox-send-* temp file" {
+ mkdir "$TMP/inbox"
+ # inbox-send writes to a .inbox-send-* temp then renames it into place;
+ # during that window the temp must not read as a pending handoff, or a
+ # concurrent boundary check blocks on a file that's about to become real.
+ touch "$TMP/inbox/.inbox-send-abc123.org"
+ cd "$TMP"
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"0 pending"* ]]
+}
+
@test "inbox-status: -q suppresses the per-item lines" {
mkdir "$TMP/inbox"
echo body > "$TMP/inbox/handoff.org"
diff --git a/.ai/scripts/tests/lint-org-cli.bats b/.ai/scripts/tests/lint-org-cli.bats
index d457696..b9faef6 100644
--- a/.ai/scripts/tests/lint-org-cli.bats
+++ b/.ai/scripts/tests/lint-org-cli.bats
@@ -20,6 +20,24 @@ teardown() {
[[ "$output" == *"lint-org: file="* ]]
}
+@test "lint-org.el default invocation is report-only — file untouched" {
+ # bare #+begin_src is a mechanical fix (→ #+begin_example) that the old
+ # default applied on disk; a linter reports, it doesn't write
+ printf '* H\n\n#+begin_src\nx\n#+end_src\n' > "$TMPFILE"
+ before="$(cat "$TMPFILE")"
+ run emacs --batch -q -l "$SCRIPTS_DIR/lint-org.el" "$TMPFILE"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"would-fix"* ]]
+ [ "$(cat "$TMPFILE")" = "$before" ]
+}
+
+@test "lint-org.el --fix applies mechanical fixes on disk" {
+ printf '* H\n\n#+begin_src\nx\n#+end_src\n' > "$TMPFILE"
+ run emacs --batch -q -l "$SCRIPTS_DIR/lint-org.el" --fix "$TMPFILE"
+ [ "$status" -eq 0 ]
+ grep -q '#+begin_example' "$TMPFILE"
+}
+
@test "wrap-org-table.el loads and runs without -L on the load path" {
run emacs --batch -q -l "$SCRIPTS_DIR/wrap-org-table.el" --width=120 "$TMPFILE"
[ "$status" -eq 0 ]
diff --git a/.ai/scripts/tests/route-batch.bats b/.ai/scripts/tests/route-batch.bats
new file mode 100644
index 0000000..84ded5f
--- /dev/null
+++ b/.ai/scripts/tests/route-batch.bats
@@ -0,0 +1,202 @@
+#!/usr/bin/env bats
+#
+# Tests for claude-templates/.ai/scripts/route-batch — the wrap-up router's
+# mechanical go path (wrapup-routing spec, Phase 4 / D7 / D9).
+#
+# Contract under test:
+# route-batch --list one "<destination>\t<heading>" line per task
+# carrying :ROUTE_CANDIDATE:; silent when none;
+# never modifies anything
+# route-batch --go per candidate: write the subtree (minus the
+# :ROUTE_CANDIDATE: line) as a one-task handoff,
+# deliver via inbox-send to the destination's
+# inbox/, then remove the subtree from the local
+# todo.org. Send failure leaves the task in
+# place and exits non-zero. Empty set: no-op.
+#
+# Strategy: fixture roots under $TEST_DIR hold a source project and two
+# destination projects; INBOX_SEND_ROOTS sandboxes inbox-send's discovery to
+# them (the same hook inbox-send's own tests use).
+
+SCRIPT="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/route-batch"
+
+setup() {
+ TEST_DIR="$(mktemp -d -t route-batch-bats.XXXXXX)"
+ ROOTS="$TEST_DIR/roots"
+ SRC="$ROOTS/srcproj"
+ mkdir -p "$SRC/.ai" "$SRC/inbox" \
+ "$ROOTS/alpha/.ai" "$ROOTS/alpha/inbox" \
+ "$ROOTS/beta/.ai" "$ROOTS/beta/inbox"
+ touch "$ROOTS/alpha/todo.org" # alpha has a todo.org; beta deliberately not
+
+ cat > "$SRC/todo.org" <<'EOF'
+* Srcproj Open Work
+** TODO [#B] Alpha-bound task :feature:
+:PROPERTIES:
+:ROUTE_CANDIDATE: alpha
+:END:
+Body line about the alpha work.
+*** TODO Sub-task that rides along
+** TODO [#C] Purely local task
+Local body stays put.
+** TODO [#C] Beta-bound task :quick:
+:PROPERTIES:
+:CREATED: [2026-07-01 Tue]
+:ROUTE_CANDIDATE: beta
+:END:
+Beta body.
+EOF
+
+ export INBOX_SEND_ROOTS="$ROOTS"
+ cd "$SRC"
+}
+
+teardown() {
+ rm -rf "$TEST_DIR"
+}
+
+# ---- --list ------------------------------------------------------------
+
+@test "route-batch --list: one destination+heading line per candidate, backlog excluded" {
+ run "$SCRIPT" --list
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"alpha"*"Alpha-bound task"* ]]
+ [[ "$output" == *"beta"*"Beta-bound task"* ]]
+ [[ "$output" != *"Purely local task"* ]]
+}
+
+@test "route-batch --list: empty candidate set is silent (exit 0)" {
+ sed -i '/:ROUTE_CANDIDATE:/d' todo.org
+ run "$SCRIPT" --list
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "route-batch --list: modifies nothing (skip leaves all in place)" {
+ before="$(cat todo.org)"
+ run "$SCRIPT" --list
+ [ "$status" -eq 0 ]
+ [ "$(cat todo.org)" = "$before" ]
+ [ -z "$(ls "$ROOTS/alpha/inbox" "$ROOTS/beta/inbox" 2>/dev/null | grep -v ':')" ]
+}
+
+# ---- --go --------------------------------------------------------------
+
+@test "route-batch --go: delivers each candidate to its destination inbox with provenance" {
+ run "$SCRIPT" --go
+ [ "$status" -eq 0 ]
+ alpha_file=$(find "$ROOTS/alpha/inbox" -name '*from-srcproj*' -type f)
+ beta_file=$(find "$ROOTS/beta/inbox" -name '*from-srcproj*' -type f)
+ [ -n "$alpha_file" ]
+ [ -n "$beta_file" ]
+ grep -q 'Alpha-bound task' "$alpha_file"
+ grep -q 'Sub-task that rides along' "$alpha_file" # children ride along
+ grep -q 'Beta-bound task' "$beta_file"
+ ! grep -q ':ROUTE_CANDIDATE:' "$alpha_file"
+ ! grep -q ':ROUTE_CANDIDATE:' "$beta_file"
+}
+
+@test "route-batch --go: removes routed subtrees from todo.org, leaves local tasks" {
+ run "$SCRIPT" --go
+ [ "$status" -eq 0 ]
+ ! grep -q 'Alpha-bound task' todo.org
+ ! grep -q 'Sub-task that rides along' todo.org
+ ! grep -q 'Beta-bound task' todo.org
+ grep -q 'Purely local task' todo.org
+ grep -q 'Local body stays put' todo.org
+}
+
+@test "route-batch --go: a kept property drawer survives minus the marker" {
+ run "$SCRIPT" --go
+ [ "$status" -eq 0 ]
+ beta_file=$(find "$ROOTS/beta/inbox" -name '*from-srcproj*' -type f)
+ grep -q ':CREATED: \[2026-07-01 Tue\]' "$beta_file"
+}
+
+@test "route-batch --go: destination with inbox/ but no todo.org still delivers" {
+ run "$SCRIPT" --go
+ [ "$status" -eq 0 ]
+ [ ! -f "$ROOTS/beta/todo.org" ]
+ [ -n "$(find "$ROOTS/beta/inbox" -name '*from-srcproj*' -type f)" ]
+}
+
+@test "route-batch --go: empty candidate set is a silent no-op (exit 0)" {
+ sed -i '/:ROUTE_CANDIDATE:/d' todo.org
+ before="$(cat todo.org)"
+ run "$SCRIPT" --go
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+ [ "$(cat todo.org)" = "$before" ]
+}
+
+@test "route-batch --go: a failed send leaves that task in place, marker intact, and exits non-zero" {
+ sed -i 's/:ROUTE_CANDIDATE: beta/:ROUTE_CANDIDATE: ghost/' todo.org
+ run "$SCRIPT" --go
+ [ "$status" -ne 0 ]
+ grep -q 'Beta-bound task' todo.org # failed route stays local
+ grep -q ':ROUTE_CANDIDATE: ghost' todo.org # marker survives so it resurfaces next wrap
+ ! grep -q 'Alpha-bound task' todo.org # the good route still landed
+ [ -n "$(find "$ROOTS/alpha/inbox" -name '*from-srcproj*' -type f)" ]
+}
+
+@test "route-batch --go: handoff headings are promoted to top level" {
+ run "$SCRIPT" --go
+ [ "$status" -eq 0 ]
+ alpha_file=$(find "$ROOTS/alpha/inbox" -name '*from-srcproj*' -type f)
+ grep -q '^\* TODO \[#B\] Alpha-bound task' "$alpha_file"
+ grep -q '^\*\* TODO Sub-task that rides along' "$alpha_file"
+}
+
+@test "route-batch --go: a drawer emptied by the marker strip is pruned from the handoff" {
+ run "$SCRIPT" --go
+ [ "$status" -eq 0 ]
+ alpha_file=$(find "$ROOTS/alpha/inbox" -name '*from-srcproj*' -type f)
+ ! grep -q ':PROPERTIES:' "$alpha_file"
+}
+
+# ---- Overlapping candidates (nested marker data-loss regression) --------
+
+@test "route-batch --go: nested candidates conflict — both stay, bystander survives, exit non-zero" {
+ cat > todo.org <<'EOF'
+* Srcproj Open Work
+** TODO [#B] Parent bound for alpha
+:PROPERTIES:
+:ROUTE_CANDIDATE: alpha
+:END:
+Parent body.
+*** TODO Child bound for beta
+:PROPERTIES:
+:ROUTE_CANDIDATE: beta
+:END:
+Child body.
+** TODO [#C] Innocent bystander task
+Bystander body.
+EOF
+ run "$SCRIPT" --go
+ [ "$status" -ne 0 ]
+ [[ "$output" == *"CONFLICT"* ]]
+ grep -q 'Parent bound for alpha' todo.org
+ grep -q 'Child bound for beta' todo.org
+ grep -q 'Innocent bystander task' todo.org
+ grep -q 'Bystander body' todo.org
+ [ -z "$(find "$ROOTS/alpha/inbox" "$ROOTS/beta/inbox" -name '*from-srcproj*' -type f)" ]
+}
+
+@test "route-batch: duplicate identical markers in one drawer dedupe to a single route" {
+ cat > todo.org <<'EOF'
+* Srcproj Open Work
+** TODO [#B] Double-tagged for alpha
+:PROPERTIES:
+:ROUTE_CANDIDATE: alpha
+:ROUTE_CANDIDATE: alpha
+:END:
+Body.
+EOF
+ run "$SCRIPT" --list
+ [ "$status" -eq 0 ]
+ [ "$(echo "$output" | grep -c 'Double-tagged')" -eq 1 ]
+ [[ "$output" != *"CONFLICT"* ]]
+ run "$SCRIPT" --go
+ [ "$status" -eq 0 ]
+ [ "$(find "$ROOTS/alpha/inbox" -name '*from-srcproj*' -type f | wc -l)" -eq 1 ]
+}
diff --git a/.ai/scripts/tests/self-inject.bats b/.ai/scripts/tests/self-inject.bats
new file mode 100644
index 0000000..482f61d
--- /dev/null
+++ b/.ai/scripts/tests/self-inject.bats
@@ -0,0 +1,78 @@
+#!/usr/bin/env bats
+# Tests for self-inject.sh — tmux is the external boundary, stubbed with a
+# recording fake so no real server is needed.
+
+setup() {
+ SCRIPT="$BATS_TEST_DIRNAME/../self-inject.sh"
+ STUB_DIR="$BATS_TEST_TMPDIR/bin"
+ LOG="$BATS_TEST_TMPDIR/tmux.log"
+ mkdir -p "$STUB_DIR"
+}
+
+# A tmux stub that records every invocation and answers list-panes from
+# $STUB_PANES (empty by default, so pane derivation fails unless a test
+# provides ancestry-matching output).
+make_stub() {
+ cat > "$STUB_DIR/tmux" <<'EOF'
+#!/bin/sh
+echo "$@" >> "$LOG"
+case "$1" in
+ list-panes) printf '%s\n' "$STUB_PANES" ;;
+esac
+EOF
+ chmod +x "$STUB_DIR/tmux"
+}
+
+@test "self-inject: -t pane with no pairs echoes the pane and exits 0" {
+ make_stub
+ run env PATH="$STUB_DIR:$PATH" LOG="$LOG" STUB_PANES="" sh "$SCRIPT" -t %42
+ [ "$status" -eq 0 ]
+ [ "$output" = "%42" ]
+ # Pane was supplied, nothing sent: tmux must not have been called.
+ [ ! -e "$LOG" ]
+}
+
+@test "self-inject: no pane derivable and no -t exits 1 with an error" {
+ make_stub
+ run env PATH="$STUB_DIR:$PATH" LOG="$LOG" STUB_PANES="" sh "$SCRIPT" 0 "hello"
+ [ "$status" -eq 1 ]
+ case "$output" in *"no owning pane"*) : ;; *) false ;; esac
+}
+
+@test "self-inject: derives the pane from process ancestry via list-panes" {
+ make_stub
+ # The stub reports the bats test process itself as a pane's pane_pid;
+ # the script runs as our child, so that pid is in its ancestry.
+ run env PATH="$STUB_DIR:$PATH" LOG="$LOG" STUB_PANES="$$ %7" sh "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [ "$output" = "%7" ]
+}
+
+@test "self-inject: one delay/text pair sends literal text then Enter" {
+ make_stub
+ run env PATH="$STUB_DIR:$PATH" LOG="$LOG" STUB_PANES="" sh "$SCRIPT" -t %3 0 "/clear"
+ [ "$status" -eq 0 ]
+ run cat "$LOG"
+ [ "${lines[0]}" = "send-keys -t %3 -l /clear" ]
+ [ "${lines[1]}" = "send-keys -t %3 Enter" ]
+}
+
+@test "self-inject: multiple pairs send in order" {
+ make_stub
+ run env PATH="$STUB_DIR:$PATH" LOG="$LOG" STUB_PANES="" \
+ sh "$SCRIPT" -t %3 0 "/clear" 0 "go — resume"
+ [ "$status" -eq 0 ]
+ run cat "$LOG"
+ [ "${lines[0]}" = "send-keys -t %3 -l /clear" ]
+ [ "${lines[1]}" = "send-keys -t %3 Enter" ]
+ [ "${lines[2]}" = "send-keys -t %3 -l go — resume" ]
+ [ "${lines[3]}" = "send-keys -t %3 Enter" ]
+}
+
+@test "self-inject: dangling odd argument after pairs is ignored" {
+ make_stub
+ run env PATH="$STUB_DIR:$PATH" LOG="$LOG" STUB_PANES="" sh "$SCRIPT" -t %3 0 "one" 99
+ [ "$status" -eq 0 ]
+ run cat "$LOG"
+ [ "${#lines[@]}" -eq 2 ]
+}
diff --git a/.ai/scripts/tests/spec-sort.bats b/.ai/scripts/tests/spec-sort.bats
new file mode 100644
index 0000000..583e458
--- /dev/null
+++ b/.ai/scripts/tests/spec-sort.bats
@@ -0,0 +1,453 @@
+#!/usr/bin/env bats
+#
+# Tests for claude-templates/.ai/scripts/spec-sort — the one-time docs-pile
+# retrofit from the docs-lifecycle spec: classify docs/**/*.org outside
+# docs/specs/ (spec candidate iff it carries BOTH a Decisions heading AND an
+# Implementation phases heading), show an evidence panel, and on --apply
+# move + rename confirmed candidates to docs/specs/*-spec.org, prepend the
+# status heading (:ID:, dated history line), rewrite the keyword header to
+# the two-sequence form, relink file: links across the rewritten roots,
+# stamp :LAST_SPEC_SORT: in .ai/notes.org.
+#
+# Contract under test (docs/specs/2026-07-01-docs-lifecycle-spec.org,
+# "The retrofit"):
+# - dry-run report is the default; --apply writes
+# - --apply refuses on a dirty worktree (exit 2) unless --allow-dirty
+# - every candidate needs --confirm REL=KEYWORD or --skip REL (exit 1
+# otherwise); terminal keywords need --reason REL=TEXT
+# - plan validated before the first write; destination collisions block
+# - bare-path mentions in rewritten roots block --apply until
+# --acknowledge-bare waives them (reported, never rewritten)
+# - mid-apply failure names applied/not-applied + git restore recovery
+# - idempotent: a sorted project yields no candidates, no changes
+#
+# Strategy: each test builds a throwaway git project fixture and runs the
+# real script against it. Mid-apply failure is forced via the test-only
+# SPEC_SORT_INJECT_FAIL_AFTER env hook.
+
+SCRIPT="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/spec-sort"
+
+setup() {
+ TEST_DIR="$(mktemp -d -t spec-sort-bats.XXXXXX)"
+ PROJ="$TEST_DIR/proj"
+ mkdir -p "$PROJ"
+}
+
+teardown() {
+ rm -rf "$TEST_DIR"
+}
+
+# Standard fixture: one spec candidate, one note, a stray root spec with a
+# spine, an anomaly (-spec.org name, no spine), inbound links from todo.org,
+# a sibling note, a session archive (report-only surface), and .ai/notes.org
+# with a Workflow State section.
+make_project() {
+ cd "$PROJ"
+ git init -q
+ git config user.email test@test
+ git config user.name test
+ mkdir -p docs/design .ai/sessions
+
+ cat > docs/design/widget.org <<'EOF'
+#+TITLE: Widget Feature
+#+DATE: 2026-05-01
+#+TODO: DRAFT REVIEW | SHIPPED
+
+* Metadata
+| Status | draft |
+| Owner | Craig |
+
+* Summary
+The widget feature. See [[file:scratch-note.org][the note]].
+
+* Decisions [1/2]
+** DONE Pick the widget shape
+** TODO Pick the color
+
+* Implementation phases
+** Phase 1 — build =src/widget.py=
+EOF
+
+ cat > docs/design/scratch-note.org <<'EOF'
+#+TITLE: Scratch Note
+
+* Metadata
+| Status | n/a |
+
+* Thoughts
+See [[file:widget.org][the widget spec]].
+EOF
+
+ cat > docs/rooty-spec.org <<'EOF'
+#+TITLE: Rooty
+
+* Decisions
+** DONE Only decision
+
+* Implementation phases
+** Phase 1 — nothing
+EOF
+
+ cat > docs/lonely-spec.org <<'EOF'
+#+TITLE: Lonely
+Just prose, no spine.
+EOF
+
+ cat > todo.org <<'EOF'
+* Open Work
+** DOING [#B] Widget feature
+Spec: [[file:docs/design/widget.org][widget spec]].
+Summary anchor: [[file:docs/design/widget.org::*Summary][the summary]].
+EOF
+
+ cat > .ai/notes.org <<'EOF'
+* Active Reminders
+
+* Workflow State
+:LAST_AUDIT: 2026-06-28
+EOF
+
+ cat > .ai/sessions/2026-06-01-old.org <<'EOF'
+Old log: [[file:../../docs/design/widget.org][widget]]
+EOF
+
+ git add -A
+ git commit -qm init
+}
+
+# Confirm flags that satisfy the gate for the standard fixture's candidates.
+CONFIRM_ALL=(--confirm docs/design/widget.org=DRAFT --confirm docs/rooty-spec.org=DRAFT)
+
+# ---- Classification (dry-run) ----------------------------------------
+
+@test "spec-sort: dry-run classifies the spine-carrying doc as a candidate" {
+ make_project
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"CANDIDATE docs/design/widget.org -> docs/specs/widget-spec.org"* ]]
+}
+
+@test "spec-sort: a Metadata table alone does not qualify — note stays a note" {
+ make_project
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"NOTE docs/design/scratch-note.org"* ]]
+ [[ "$output" != *"CANDIDATE docs/design/scratch-note.org"* ]]
+}
+
+@test "spec-sort: stray root spec with a spine is a candidate, suffix not doubled" {
+ make_project
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"CANDIDATE docs/rooty-spec.org -> docs/specs/rooty-spec.org"* ]]
+ [[ "$output" != *"rooty-spec-spec.org"* ]]
+}
+
+@test "spec-sort: -spec.org name without a spine is an anomaly, never auto-moved" {
+ make_project
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"ANOMALY docs/lonely-spec.org"* ]]
+ [[ "$output" != *"CANDIDATE docs/lonely-spec.org"* ]]
+}
+
+@test "spec-sort: docs/specs/ contents are excluded from classification" {
+ make_project
+ mkdir -p docs/specs
+ cp docs/design/widget.org docs/specs/sorted-spec.org
+ git add -A && git commit -qm more
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" != *"CANDIDATE docs/specs/sorted-spec.org"* ]]
+}
+
+@test "spec-sort: no docs/ directory is a silent no-op" {
+ cd "$PROJ"
+ git init -q
+ git config user.email test@test
+ git config user.name test
+ echo x > README.md
+ git add -A && git commit -qm init
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+# ---- Evidence panel ---------------------------------------------------
+
+@test "spec-sort: evidence panel shows status field, cookies, and todo.org task" {
+ make_project
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"status field: draft"* ]]
+ [[ "$output" == *"Decisions [1/2]"* ]]
+ [[ "$output" == *"todo.org:"*"DOING"*"Widget feature"* ]]
+}
+
+@test "spec-sort: keyword proposal follows the evidence — DOING from the linked DOING task" {
+ make_project
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ # status field says draft, but the linking todo.org task is DOING — the
+ # panel proposes the state the strongest evidence supports
+ [[ "$output" == *"proposed keyword: DOING"* ]]
+}
+
+@test "spec-sort: an 'incomplete' status field never proposes the terminal IMPLEMENTED" {
+ make_project
+ sed -i 's/| Status | draft |/| Status | incomplete |/' docs/design/widget.org
+ git add -A && git commit -qm status
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" != *"proposed keyword: IMPLEMENTED"* ]]
+}
+
+# ---- Confirm gate -----------------------------------------------------
+
+@test "spec-sort --apply: refuses when a candidate is neither confirmed nor skipped" {
+ make_project
+ run "$SCRIPT" --apply --confirm docs/design/widget.org=DRAFT
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"unconfirmed"* ]]
+ [[ "$output" == *"docs/rooty-spec.org"* ]]
+ [ -f docs/design/widget.org ] # nothing moved
+}
+
+@test "spec-sort --apply: a terminal keyword without --reason refuses" {
+ make_project
+ run "$SCRIPT" --apply --confirm docs/design/widget.org=IMPLEMENTED --skip docs/rooty-spec.org
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"--reason"* ]]
+ [ -f docs/design/widget.org ]
+}
+
+@test "spec-sort --apply: a terminal keyword with --reason records it in the history line" {
+ make_project
+ run "$SCRIPT" --apply --confirm docs/design/widget.org=IMPLEMENTED \
+ --reason "docs/design/widget.org=shipped in v2, confirmed against src" \
+ --skip docs/rooty-spec.org
+ [ "$status" -eq 0 ]
+ grep -q '^\* IMPLEMENTED Widget Feature' docs/specs/widget-spec.org
+ grep -q 'shipped in v2, confirmed against src' docs/specs/widget-spec.org
+}
+
+@test "spec-sort --apply: --skip leaves the candidate in place and still stamps the marker" {
+ make_project
+ run "$SCRIPT" --apply --skip docs/design/widget.org --skip docs/rooty-spec.org
+ [ "$status" -eq 0 ]
+ [ -f docs/design/widget.org ]
+ grep -q ':LAST_SPEC_SORT:' .ai/notes.org
+}
+
+# ---- Preflight --------------------------------------------------------
+
+@test "spec-sort --apply: refuses on a dirty worktree (exit 2)" {
+ make_project
+ echo "drift" >> todo.org
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 2 ]
+ [[ "$output" == *"dirty"* ]]
+ [ -f docs/design/widget.org ]
+}
+
+@test "spec-sort --apply --allow-dirty: proceeds and names what recovery loses" {
+ make_project
+ echo "drift" >> todo.org
+ git add todo.org && git commit -qm drift # keep the link intact; dirty a different file
+ echo "scratch" > untracked-note.txt
+ echo "local edit" >> .ai/notes.org
+ run "$SCRIPT" --apply --allow-dirty "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"pre-existing"* ]]
+ [[ "$output" == *".ai/notes.org"* ]]
+ [ -f docs/specs/widget-spec.org ]
+}
+
+# ---- Move + rename + rewrite ------------------------------------------
+
+@test "spec-sort --apply: moves, renames to -spec.org, prepends status heading with :ID: and history" {
+ make_project
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ [ -f docs/specs/widget-spec.org ]
+ [ ! -f docs/design/widget.org ]
+ grep -q '^\* DRAFT Widget Feature' docs/specs/widget-spec.org
+ grep -q ':ID:' docs/specs/widget-spec.org
+ grep -q 'retrofitted by spec-sort' docs/specs/widget-spec.org
+}
+
+@test "spec-sort --apply: keyword header rewritten to the two-sequence form" {
+ make_project
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ grep -q '^#+TODO: TODO | DONE$' docs/specs/widget-spec.org
+ grep -q '^#+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED$' docs/specs/widget-spec.org
+ ! grep -q 'DRAFT REVIEW | SHIPPED' docs/specs/widget-spec.org
+}
+
+@test "spec-sort --apply: Metadata Status field mirrors the confirmed keyword in lowercase" {
+ make_project
+ run "$SCRIPT" --apply --confirm docs/design/widget.org=READY --skip docs/rooty-spec.org
+ [ "$status" -eq 0 ]
+ grep -q '^\* READY Widget Feature' docs/specs/widget-spec.org
+ grep -Eq '^\| Status[[:space:]]*\|[[:space:]]*ready' docs/specs/widget-spec.org
+}
+
+# ---- Relink -----------------------------------------------------------
+
+@test "spec-sort --apply: rewrites the todo.org link, preserving the description" {
+ make_project
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ grep -q '\[\[file:docs/specs/widget-spec.org\]\[widget spec\]\]' todo.org
+ ! grep -q 'docs/design/widget.org' todo.org
+}
+
+@test "spec-sort --apply: preserves a ::anchor suffix through the rewrite" {
+ make_project
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ grep -q '\[\[file:docs/specs/widget-spec.org::\*Summary\]\[the summary\]\]' todo.org
+}
+
+@test "spec-sort --apply: recomputes a sibling note's relative link to the moved spec" {
+ make_project
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ grep -q '\[\[file:../specs/widget-spec.org\]\[the widget spec\]\]' docs/design/scratch-note.org
+}
+
+@test "spec-sort --apply: recomputes the moved spec's own outbound link to an unmoved note" {
+ make_project
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ grep -q '\[\[file:../design/scratch-note.org\]\[the note\]\]' docs/specs/widget-spec.org
+}
+
+@test "spec-sort: session archives are reported, never rewritten" {
+ make_project
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"REPORT .ai/sessions/2026-06-01-old.org"* ]]
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ grep -q 'docs/design/widget.org' .ai/sessions/2026-06-01-old.org
+}
+
+@test "spec-sort: a synced template path report names the canonical rulesets file" {
+ make_project
+ mkdir -p .ai/workflows
+ echo 'See [[file:../../docs/design/widget.org][widget]]' > .ai/workflows/startup.org
+ git add -A && git commit -qm wf
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"REPORT .ai/workflows/startup.org"* ]]
+ [[ "$output" == *"claude-templates/.ai/workflows/startup.org"* ]]
+}
+
+# ---- Bare-path mentions -----------------------------------------------
+
+@test "spec-sort --apply: a bare-path mention in a rewritten root blocks until acknowledged" {
+ make_project
+ echo "raw mention: docs/design/widget.org needs review" >> todo.org
+ git add -A && git commit -qm bare
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"BARE"* ]]
+ [ -f docs/design/widget.org ] # nothing moved
+ run "$SCRIPT" --apply --acknowledge-bare "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ grep -q 'raw mention: docs/design/widget.org' todo.org # reported, never rewritten
+}
+
+@test "spec-sort --apply: a moving doc's bare mention of its own old path is acknowledgeable, not post-apply residue" {
+ make_project
+ echo "History: docs/design/widget.org was drafted in May." >> docs/design/widget.org
+ git add -A && git commit -qm selfmention
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"BARE"* ]]
+ run "$SCRIPT" --apply --acknowledge-bare "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ] # the acknowledged mention rides along to docs/specs/; not residue
+ grep -q ':LAST_SPEC_SORT:' .ai/notes.org
+}
+
+# ---- Plan validation ---------------------------------------------------
+
+@test "spec-sort --apply: a destination collision blocks validation, nothing moved" {
+ make_project
+ mkdir -p docs/specs
+ echo "occupied" > docs/specs/widget-spec.org
+ git add -A && git commit -qm occupy
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"destination exists"* ]]
+ [ -f docs/design/widget.org ]
+ [ "$(cat docs/specs/widget-spec.org)" = "occupied" ]
+}
+
+@test "spec-sort --apply: writes the plan file before executing" {
+ make_project
+ run "$SCRIPT" --apply --plan-file "$TEST_DIR/plan.json" "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ [ -f "$TEST_DIR/plan.json" ]
+ grep -q 'widget-spec.org' "$TEST_DIR/plan.json"
+}
+
+# ---- Mid-apply failure recovery ----------------------------------------
+
+@test "spec-sort --apply: forced mid-apply failure yields named recovery, not a half-migrated shrug" {
+ make_project
+ run env SPEC_SORT_INJECT_FAIL_AFTER=1 "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"RECOVERY"* ]]
+ [[ "$output" == *"git restore"* ]]
+ [[ "$output" == *"applied"* ]]
+ [[ "$output" == *"not applied"* ]]
+ ! grep -q ':LAST_SPEC_SORT:' .ai/notes.org # no stamp on a failed apply
+}
+
+# ---- Idempotence + marker ----------------------------------------------
+
+@test "spec-sort --apply: stamps :LAST_SPEC_SORT: in the Workflow State section" {
+ make_project
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ grep -q ':LAST_SPEC_SORT: ' .ai/notes.org
+ # lands inside the Workflow State section, alongside the existing marker
+ awk '/^\* Workflow State/{ws=1} ws && /:LAST_SPEC_SORT:/{found=1} END{exit !found}' .ai/notes.org
+}
+
+@test "spec-sort --apply: creates the Workflow State section when notes.org lacks it" {
+ make_project
+ printf '* Active Reminders\n' > .ai/notes.org
+ git add -A && git commit -qm notes
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ grep -q '^\* Workflow State' .ai/notes.org
+ grep -q ':LAST_SPEC_SORT: ' .ai/notes.org
+}
+
+@test "spec-sort --apply: zero candidates still stamps the marker (clears the nudge)" {
+ make_project
+ rm docs/design/widget.org docs/rooty-spec.org docs/lonely-spec.org
+ git add -A && git commit -qm notes-only
+ run "$SCRIPT" --apply
+ [ "$status" -eq 0 ]
+ grep -q ':LAST_SPEC_SORT:' .ai/notes.org
+}
+
+@test "spec-sort: a second run after a successful apply finds nothing to do" {
+ make_project
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ git add -A && git commit -qm sorted
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" != *"CANDIDATE"* ]]
+ run "$SCRIPT" --apply
+ [ "$status" -eq 0 ]
+ run git status --porcelain
+ # only the re-stamped marker (same date) may differ — tree stays clean
+ [ -z "$(git status --porcelain -- docs todo.org)" ]
+}
diff --git a/.ai/scripts/tests/task-review-staleness.bats b/.ai/scripts/tests/task-review-staleness.bats
index 488b023..79aad79 100644
--- a/.ai/scripts/tests/task-review-staleness.bats
+++ b/.ai/scripts/tests/task-review-staleness.bats
@@ -49,6 +49,16 @@ task_unreviewed() {
printf '** %s [#%s] %s\nBody.\n\n' "$keyword" "$prio" "$title" >> "$TODO"
}
+# Emit a qualifying task whose LAST_REVIEWED is an org-native inactive
+# timestamp — [YYYY-MM-DD Day] — matching the CREATED:/CLOSED: cookies that
+# sit in the same drawer. The date is derived from an ISO date via `date`.
+task_reviewed_org() {
+ local keyword="$1" prio="$2" title="$3" isodate="$4"
+ local org="[$(date -d "$isodate" '+%F %a')]"
+ printf '** %s [#%s] %s\n:PROPERTIES:\n:LAST_REVIEWED: %s\n:END:\nBody.\n\n' \
+ "$keyword" "$prio" "$title" "$org" >> "$TODO"
+}
+
# ---- Normal cases ----------------------------------------------------
@test "staleness: empty file reports zero" {
@@ -85,6 +95,20 @@ task_unreviewed() {
[ "$output" = "2" ]
}
+@test "staleness: org-native bracketed LAST_REVIEWED parses — recent is fresh" {
+ task_reviewed_org TODO A "Reviewed five days ago, org stamp" "$D5"
+ run bash "$SCRIPT" "$TODO" 30
+ [ "$status" -eq 0 ]
+ [ "$output" = "0" ]
+}
+
+@test "staleness: org-native bracketed LAST_REVIEWED parses — old is stale" {
+ task_reviewed_org TODO A "Reviewed forty days ago, org stamp" "$D40"
+ run bash "$SCRIPT" "$TODO" 30
+ [ "$status" -eq 0 ]
+ [ "$output" = "1" ]
+}
+
# ---- Boundary cases --------------------------------------------------
@test "staleness: age exactly equal to threshold is fresh" {
@@ -136,9 +160,23 @@ task_unreviewed() {
[ "$output" = "0" ]
}
-@test "staleness: malformed LAST_REVIEWED is treated as stale" {
+@test "staleness: malformed LAST_REVIEWED warns to stderr and is not counted" {
task_reviewed TODO A "Bad date" "not-a-date"
- run bash "$SCRIPT" "$TODO" 30
+ # stdout carries only the count — the malformed stamp is not folded in.
+ run bash -c "bash '$SCRIPT' '$TODO' 30 2>/dev/null"
+ [ "$status" -eq 0 ]
+ [ "$output" = "0" ]
+ # stderr carries the loud warning naming the offending value.
+ run bash -c "bash '$SCRIPT' '$TODO' 30 2>&1 1>/dev/null"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"not-a-date"* ]]
+ [[ "$output" == *"LAST_REVIEWED"* ]]
+}
+
+@test "staleness: malformed stamp is excluded while real stale tasks still count" {
+ task_reviewed TODO A "Real stale" "$D40"
+ task_reviewed TODO B "Broken stamp" "garbage"
+ run bash -c "bash '$SCRIPT' '$TODO' 30 2>/dev/null"
[ "$status" -eq 0 ]
[ "$output" = "1" ]
}
@@ -161,6 +199,17 @@ task_unreviewed() {
[[ "${lines[2]}" == *"Reviewed recently"* ]]
}
+@test "staleness --list: org-native bracketed stamp sorts by its real date" {
+ task_reviewed TODO A "Bare recent" "$D5"
+ task_reviewed_org TODO B "Org-stamped old" "$D40"
+ run bash "$SCRIPT" --list "$TODO" 10
+ [ "$status" -eq 0 ]
+ # The org-bracketed old stamp must sort ahead of the bare recent one —
+ # proof it parsed to a real date rather than falling to 0000-00-00.
+ [[ "${lines[0]}" == *"Org-stamped old"* ]]
+ [[ "${lines[1]}" == *"Bare recent"* ]]
+}
+
@test "staleness --list: takes only the requested count" {
task_unreviewed TODO A "First"
task_reviewed TODO B "Second" "$D40"
diff --git a/.ai/scripts/tests/test-lint-org.el b/.ai/scripts/tests/test-lint-org.el
index 3a83602..ceee209 100644
--- a/.ai/scripts/tests/test-lint-org.el
+++ b/.ai/scripts/tests/test-lint-org.el
@@ -193,6 +193,65 @@ real suspicious-language warning here
#+end_src
")
+;; invalid-block, false-positive case — a correctly paired example block whose
+;; body holds a heading-shaped line. org's parser reads the `** ' inside the
+;; verbatim body as a structural break, loses the open block, and flags BOTH
+;; delimiters as "Possible incomplete block".
+(defconst lo-test--verbatim-heading-block "\
+* Heading
+
+#+begin_example
+** Feature Name or Topic
+Body line.
+#+end_example
+
+Trailing prose.
+")
+
+;; invalid-block, literal-delimiter case — a paired src block whose body holds
+;; a literal `#+end_example' plus a heading-shaped line. Only `#+end_src'
+;; closes a src block, so all three findings here are false.
+(defconst lo-test--literal-end-in-src "\
+* Heading
+
+#+begin_src text
+#+end_example
+** heading shaped
+#+end_src
+")
+
+;; invalid-block, uppercase-delimiter case — org accepts #+BEGIN_/#+END_ in
+;; either case, and the pre-fix script flagged both delimiters here too.
+(defconst lo-test--uppercase-verbatim-block "\
+* Heading
+
+#+BEGIN_EXAMPLE
+** heading shaped
+#+END_EXAMPLE
+")
+
+;; invalid-block, genuine case — a block that really is never closed. The
+;; suppression must not reach this one.
+(defconst lo-test--unterminated-block "\
+* Heading
+
+#+begin_example
+truly unterminated block body
+")
+
+;; A genuinely unterminated block *after* a correctly paired one — verifies the
+;; suppression is scoped per block rather than per file.
+(defconst lo-test--paired-then-unterminated "\
+* Heading
+
+#+begin_example
+** heading shaped
+#+end_example
+
+#+begin_example
+never closed
+")
+
;; Mixed fixture — each category once.
(defconst lo-test--mixed "\
* Mixed
@@ -392,6 +451,55 @@ suspicious-language judgment."
(should (= 1 suspicious))))
;;; ---------------------------------------------------------------------------
+;;; invalid-block — false positives on correctly paired verbatim blocks
+
+(ert-deftest lo-verbatim-heading-block-emits-no-invalid-block ()
+ "Normal: a paired example block containing a heading-shaped body line emits
+no invalid-block judgment. Both delimiters are flagged by org-lint because the
+parser treats the `** ' inside the verbatim body as a structural break."
+ (let* ((out (lo-test--run lo-test--verbatim-heading-block))
+ (res (plist-get out :result))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ ;; File untouched, no fixes applied — suppression only, never a rewrite.
+ (should (equal lo-test--verbatim-heading-block res))
+ (should (= 0 (plist-get out :fixes)))
+ (should-not (member 'invalid-block (lo-test--checkers judgments)))))
+
+(ert-deftest lo-literal-end-delimiter-in-src-emits-no-invalid-block ()
+ "Boundary: a paired src block whose body holds a literal `#+end_example' and
+a heading-shaped line emits no invalid-block judgment. Only `#+end_src' closes
+a src block, so the interior delimiter is body text."
+ (let* ((out (lo-test--run lo-test--literal-end-in-src))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'invalid-block (lo-test--checkers judgments)))))
+
+(ert-deftest lo-uppercase-verbatim-block-emits-no-invalid-block ()
+ "Boundary: block delimiters are case-insensitive in org, so an uppercase
+`#+BEGIN_EXAMPLE' pair is suppressed the same as a lowercase one."
+ (let* ((out (lo-test--run lo-test--uppercase-verbatim-block))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'invalid-block (lo-test--checkers judgments)))))
+
+(ert-deftest lo-unterminated-block-still-emits-invalid-block ()
+ "Error: a block that is never closed still emits its invalid-block judgment.
+This is the finding the checker exists for — the suppression must not mask it."
+ (let* ((out (lo-test--run lo-test--unterminated-block))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should (member 'invalid-block (lo-test--checkers judgments)))))
+
+(ert-deftest lo-invalid-block-suppression-is-scoped-per-block ()
+ "Boundary: a paired block and an unterminated block in the same file — the
+paired one is suppressed and the unterminated one still reports. Exactly one
+invalid-block judgment, and it points at the unterminated opener (line 7)."
+ (let* ((out (lo-test--run lo-test--paired-then-unterminated))
+ (judgments (lo-test--judgments (plist-get out :issues)))
+ (invalid (cl-remove-if-not
+ (lambda (i) (eq (plist-get i :checker) 'invalid-block))
+ judgments)))
+ (should (= 1 (length invalid)))
+ (should (= 7 (plist-get (car invalid) :line)))))
+
+;;; ---------------------------------------------------------------------------
;;; --check mode
(ert-deftest lo-check-mode-does-not-modify-file ()
@@ -620,6 +728,29 @@ followups file on the next run."
;;; ---------------------------------------------------------------------------
;;; org-table-standard check (width budget + rules between rows)
+(ert-deftest lo-table-inside-example-block-not-flagged ()
+ "Pipe-led ASCII art inside an example block is not a table; no judgment."
+ (let* ((run (lo-test--run
+ "* H\n\n#+begin_example\n| client |----->| server |\n| box | | box |\n#+end_example\n"
+ 1 t))
+ (judgments (lo-test--judgments (plist-get run :issues))))
+ (should-not (memq 'org-table-standard (lo-test--checkers judgments)))))
+
+(ert-deftest lo-table-inside-src-block-not-flagged ()
+ "Shell pipes inside a src block are not a table; no judgment."
+ (let* ((run (lo-test--run
+ "* H\n\n#+begin_src sh\n| sort\n| uniq -c\n#+end_src\n" 1 t))
+ (judgments (lo-test--judgments (plist-get run :issues))))
+ (should-not (memq 'org-table-standard (lo-test--checkers judgments)))))
+
+(ert-deftest lo-real-table-after-block-still-flagged ()
+ "Block safety must not mask a genuine violation later in the file."
+ (let* ((run (lo-test--run
+ "* H\n\n#+begin_example\n| art |\n#+end_example\n\n| a | b |\n| 1 | 2 |\n"
+ 1 t))
+ (judgments (lo-test--judgments (plist-get run :issues))))
+ (should (memq 'org-table-standard (lo-test--checkers judgments)))))
+
(ert-deftest lo-table-over-budget-emits-judgment ()
"A table line rendering wider than 120 surfaces as an org-table-standard judgment."
(let* ((wide (make-string 130 ?x))
@@ -659,5 +790,311 @@ missing-rules violation."
(judgments (lo-test--judgments (plist-get run :issues))))
(should-not (memq 'org-table-standard (lo-test--checkers judgments)))))
+;;; ---------------------------------------------------------------------------
+;;; level-2 dated-header check (claude-rules/todo-format.md)
+
+(ert-deftest lo-level2-dated-header-is-judgment ()
+ "A level-2 heading beginning with a YYYY-MM-DD date is flagged."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** 2026-06-20 Sat @ 10:00:00 -0500 Something resolved\nBody.\n"))
+ (res (plist-get out :result))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should (= 0 (plist-get out :fixes))) ; judgment-only, never auto-fixed
+ (should (member 'level-2-dated-header (lo-test--checkers judgments)))))
+
+(ert-deftest lo-level2-done-task-not-flagged ()
+ "A level-2 task closed with a terminal keyword + CLOSED: is fine."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** DONE [#B] Something resolved\nCLOSED: [2026-06-20 Sat]\nBody.\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'level-2-dated-header (lo-test--checkers judgments)))))
+
+(ert-deftest lo-level3-dated-entry-not-flagged ()
+ "A dated event-log entry at level 3 is the correct sub-task shape, not a defect."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** TODO [#B] Parent task\n*** 2026-06-20 Sat @ 10:00:00 -0500 sub-entry landed\nBody.\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'level-2-dated-header (lo-test--checkers judgments)))))
+
+;;; subtask-done-not-dated check (the inverse: level-3+ done keyword)
+
+(ert-deftest lo-subtask-done-not-dated-flags-level3 ()
+ "A level-3 DONE sub-task still carrying the keyword is flagged for conversion."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** TODO [#B] Parent\n*** DONE [#C] Sub-task done\nCLOSED: [2026-06-20 Sat 10:00]\nBody.\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should (= 0 (plist-get out :fixes))) ; judgment-only, never auto-fixed
+ (should (member 'subtask-done-not-dated (lo-test--checkers judgments)))))
+
+(ert-deftest lo-subtask-done-not-dated-flags-level4-cancelled ()
+ "A level-4 CANCELLED sub-task is flagged too."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** PROJECT [#B] Parent\n*** TODO Mid\n**** CANCELLED Deep abandoned\nCLOSED: [2026-06-20 Sat]\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should (member 'subtask-done-not-dated (lo-test--checkers judgments)))))
+
+(ert-deftest lo-subtask-done-not-dated-ignores-level2 ()
+ "A level-2 DONE task is a top-level task, not a sub-task — this checker skips it."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** DONE [#B] Top-level\nCLOSED: [2026-06-20 Sat]\nBody.\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'subtask-done-not-dated (lo-test--checkers judgments)))))
+
+(ert-deftest lo-subtask-done-not-dated-ignores-dated-and-lowercase ()
+ "An already-dated level-3 entry, and the word done in a title, are not flagged."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0400 landed\n*** TODO wrap the done cleanup\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'subtask-done-not-dated (lo-test--checkers judgments)))))
+
+;;; dated-log-heading-active-timestamp check (stale SCHEDULED/DEADLINE on a
+;;; completed dated-log entry — the home 2026-07-17 agenda-pollution bug)
+
+(ert-deftest lo-dated-log-active-scheduled-is-flagged ()
+ "A dated-log entry still carrying an active SCHEDULED is flagged: org renders
+it on the agenda forever despite the missing keyword."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 trip booked\nSCHEDULED: <2026-06-18 Thu>\nBody.\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should (= 0 (plist-get out :fixes))) ; judgment-only, never auto-fixed
+ (should (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments)))))
+
+(ert-deftest lo-dated-log-active-deadline-is-flagged ()
+ "An active DEADLINE on a dated-log entry is flagged too."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 shipped\nDEADLINE: <2026-06-25 Thu>\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments)))))
+
+(ert-deftest lo-dated-log-clean-entry-not-flagged ()
+ "A dated-log entry with no active planning timestamp is correct — not flagged."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 done cleanly\nBody only.\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments)))))
+
+(ert-deftest lo-dated-log-inactive-timestamp-not-flagged ()
+ "An inactive [..] timestamp doesn't render on the agenda, so it isn't flagged —
+only active <..> planning timestamps are the defect."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 recorded\nSCHEDULED: [2026-06-18 Thu]\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments)))))
+
+(ert-deftest lo-dated-log-active-scheduled-on-live-todo-not-flagged ()
+ "A live TODO (keyword present) that legitimately carries an active SCHEDULED is
+not a dated-log heading, so this checker leaves it alone."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** TODO [#B] Parent\n*** TODO [#C] real upcoming task\nSCHEDULED: <2026-06-18 Thu>\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments)))))
+
+;;; ---------------------------------------------------------------------------
+;;; structural heading checks (org-lint gaps)
+
+(defun lo-test--checker-lines (issues checker)
+ "Lines of judgment ISSUES whose :checker is CHECKER, document order."
+ (mapcar (lambda (i) (plist-get i :line))
+ (cl-remove-if-not
+ (lambda (i) (and (eq (plist-get i :kind) 'judgment)
+ (eq (plist-get i :checker) checker)))
+ (reverse issues))))
+
+(ert-deftest lo-indented-heading-flags-leading-whitespace ()
+ "Error: a heading indented off column 0 is flagged (org demotes it to body)."
+ (let* ((out (lo-test--run "* Open\n ** TODO indented and lost\n** TODO fine\n"))
+ (j (lo-test--judgments (plist-get out :issues))))
+ (should (member 'indented-heading (lo-test--checkers j)))
+ (should (= 1 (length (lo-test--checker-lines (plist-get out :issues)
+ 'indented-heading))))))
+
+(ert-deftest lo-indented-heading-skips-stars-inside-blocks ()
+ "Boundary: indented stars inside a #+begin_/#+end_ block are legitimate content."
+ (let* ((out (lo-test--run "* Open\n#+begin_example\n ** not a heading\n#+end_example\n"))
+ (j (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'indented-heading (lo-test--checkers j)))))
+
+(ert-deftest lo-indented-heading-skips-single-star-list-bullets ()
+ "Normal: an indented single `*' is a valid plain-list bullet, not a demoted
+heading, so it is not flagged — only two-or-more indented stars are."
+ (let* ((out (lo-test--run "* Open\nintro line\n * first bullet\n * second bullet\n * nested bullet\n"))
+ (j (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'indented-heading (lo-test--checkers j)))))
+
+(ert-deftest lo-empty-heading-flags-bare-stars ()
+ "Error: a line of bare stars with no title is flagged."
+ (let* ((out (lo-test--run "* Open\n** \n** TODO real\n"))
+ (j (lo-test--judgments (plist-get out :issues))))
+ (should (member 'empty-heading (lo-test--checkers j)))))
+
+(ert-deftest lo-malformed-priority-flags-lowercase-and-skips-valid ()
+ "Error + Normal: a lowercase/oversized cookie flags; a valid [#B] stays silent."
+ (let* ((bad (lo-test--run "* Open\n** TODO [#a] lowercase cookie\n** TODO [#BB] oversized\n"))
+ (ok (lo-test--run "* Open\n** TODO [#B] valid cookie\n"))
+ (jo (lo-test--judgments (plist-get ok :issues))))
+ (should (= 2 (length (lo-test--checker-lines (plist-get bad :issues)
+ 'malformed-priority-cookie))))
+ (should-not (member 'malformed-priority-cookie (lo-test--checkers jo)))))
+
+(ert-deftest lo-malformed-priority-skips-verbatim-cookie-in-title ()
+ "Boundary: a dated-log title quoting =[#D]= verbatim is not a real cookie."
+ (let* ((out (lo-test--run "* Open\n** TODO [#B] parent\n*** 2026-05-14 reprioritized =[#D]= -> =[#B]=\n"))
+ (j (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'malformed-priority-cookie (lo-test--checkers j)))))
+
+(ert-deftest lo-done-without-closed-flags-undated-level2 ()
+ "Error: a level-2 DONE with no CLOSED line is flagged; a dated one is not."
+ (let* ((bad (lo-test--run "* Resolved\n** DONE undated finished\nbody\n"))
+ (jb (lo-test--judgments (plist-get bad :issues)))
+ (ok (lo-test--run "* Resolved\n** DONE dated\nCLOSED: [2026-06-29 Mon]\n"))
+ (jo (lo-test--judgments (plist-get ok :issues))))
+ (should (member 'level2-done-without-closed (lo-test--checkers jb)))
+ (should-not (member 'level2-done-without-closed (lo-test--checkers jo)))))
+
+(ert-deftest lo-done-without-closed-ignores-deeper-levels ()
+ "Boundary: a level-3 DONE (a dated-log sub-entry) need not carry CLOSED."
+ (let* ((out (lo-test--run "* Resolved\n** DONE parent\nCLOSED: [2026-06-29 Mon]\n*** DONE nested no-closed\n"))
+ (j (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'level2-done-without-closed (lo-test--checkers j)))))
+
+(ert-deftest lo-structural-checks-silent-on-clean-file ()
+ "Normal: a well-formed file trips none of the four structural checkers."
+ (let* ((out (lo-test--run "* Open Work\n** TODO [#A] a task :tag:\n** DOING [#B] another\n* Resolved\n** DONE [#C] done\nCLOSED: [2026-06-29 Mon]\n"))
+ (checkers (lo-test--checkers (lo-test--judgments (plist-get out :issues)))))
+ (dolist (c '(indented-heading empty-heading malformed-priority-cookie
+ level2-done-without-closed))
+ (should-not (member c checkers)))))
+
(provide 'test-lint-org)
;;; test-lint-org.el ends here
+
+;;; ---------------------------------------------------------------------------
+;;; task-missing-last-reviewed (claude-rules/todo-format.md)
+
+(ert-deftest lo-task-without-last-reviewed-is-judgment ()
+ "An open level-2 task with no :LAST_REVIEWED: is flagged."
+ (let* ((out (lo-test--run "* Open Work\n** TODO [#B] A task :feature:\nBody.\n"))
+ (js (lo-test--judgments (plist-get out :issues))))
+ (should (memq 'task-missing-last-reviewed (lo-test--checkers js)))))
+
+(ert-deftest lo-task-with-last-reviewed-is-clean ()
+ "A task carrying the property is not flagged."
+ (let* ((out (lo-test--run (concat "* Open Work\n** TODO [#B] A task :feature:\n"
+ ":PROPERTIES:\n:LAST_REVIEWED: 2026-07-23\n:END:\n"
+ "Body.\n")))
+ (js (lo-test--judgments (plist-get out :issues))))
+ (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js)))))
+
+(ert-deftest lo-task-last-reviewed-accepts-org-timestamp ()
+ "The org-native [YYYY-MM-DD Day] form counts, matching the staleness script."
+ (let* ((out (lo-test--run (concat "* Open Work\n** TODO [#B] A task :feature:\n"
+ ":PROPERTIES:\n:LAST_REVIEWED: [2026-07-23 Thu]\n:END:\n")))
+ (js (lo-test--judgments (plist-get out :issues))))
+ (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js)))))
+
+(ert-deftest lo-done-task-without-last-reviewed-is-clean ()
+ "Completed tasks leave the review pool, so they are never flagged."
+ (let* ((out (lo-test--run (concat "* Open Work\n** DONE [#B] A task :feature:\n"
+ "CLOSED: [2026-07-23 Thu]\n")))
+ (js (lo-test--judgments (plist-get out :issues))))
+ (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js)))))
+
+(ert-deftest lo-subtask-without-last-reviewed-is-clean ()
+ "Only level-2 tasks are in the review pool; deeper headings are not."
+ (let* ((out (lo-test--run (concat "* Open Work\n** TODO [#B] Parent :feature:\n"
+ ":PROPERTIES:\n:LAST_REVIEWED: 2026-07-23\n:END:\n"
+ "*** TODO A sub-task\n")))
+ (js (lo-test--judgments (plist-get out :issues))))
+ (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js)))))
+
+(ert-deftest lo-cookieless-task-without-last-reviewed-is-clean ()
+ "The staleness script selects on a priority cookie, so match that scope."
+ (let* ((out (lo-test--run "* Open Work\n** TODO Manual testing and validation\n"))
+ (js (lo-test--judgments (plist-get out :issues))))
+ (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js)))))
+
+(ert-deftest lo-verify-task-without-last-reviewed-is-judgment ()
+ "VERIFY is in the review pool too."
+ (let* ((out (lo-test--run "* Open Work\n** VERIFY [#B] Waiting on Craig\n"))
+ (js (lo-test--judgments (plist-get out :issues))))
+ (should (memq 'task-missing-last-reviewed (lo-test--checkers js)))))
+
+;;; ---------------------------------------------------------------------------
+;;; todo-format checkers skip docs/specs/ files (claude-rules/todo-format.md)
+;;
+;; The four todo-format-family checkers encode todo.org completion conventions.
+;; A spec legitimately uses ** DONE <decision> with no CLOSED cookie and
+;; ** <dated> — <who> review-history headings, so those checkers misfire on
+;; every spec. They must skip any file under a docs/specs/ path segment.
+
+(defun lo-test--run-at (relpath content)
+ "Write CONTENT to <tmpdir>/RELPATH, run lint on it, return :issues.
+RELPATH is a relative path (may contain slashes) so a docs/specs/ segment
+can be exercised — the checkers key on the file's path, not just its name."
+ (let* ((root (make-temp-file "lo-test-root-" t))
+ (file (expand-file-name relpath root)))
+ (make-directory (file-name-directory file) t)
+ (unwind-protect
+ (progn
+ (with-temp-file file (insert content))
+ (lo-test--reset)
+ (lo-process-file file)
+ (prog1 (list :issues lo-issues)
+ (lo-test--drop-buffer file)))
+ (delete-directory root t))))
+
+(defconst lo-test--spec-decisions
+ "* Decisions [1/1]\n** DONE Some decision\n- Context: x\n"
+ "A spec Decisions section: a level-2 DONE with no CLOSED cookie.")
+
+(defconst lo-test--spec-history
+ "* Review history\n** 2026-07-14 Tue @ 02:03:28 -0500 — Claude — responder\n- What: x\n"
+ "A spec review-history section: a level-2 dated header.")
+
+(ert-deftest lo-todo-checkers-fire-on-a-normal-org-file ()
+ "Baseline: the checkers DO fire on a non-spec path (the bug is scope, not silence)."
+ (let* ((out (lo-test--run-at "todo.org" lo-test--spec-decisions))
+ (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues)))))
+ (should (memq 'level2-done-without-closed cs))))
+
+(ert-deftest lo-level2-done-without-closed-skips-specs ()
+ (let* ((out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" lo-test--spec-decisions))
+ (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues)))))
+ (should-not (memq 'level2-done-without-closed cs))))
+
+(ert-deftest lo-level2-dated-header-skips-specs ()
+ (let* ((out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" lo-test--spec-history))
+ (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues)))))
+ (should-not (memq 'level-2-dated-header cs))))
+
+(ert-deftest lo-dated-log-active-timestamp-skips-specs ()
+ (let* ((c "* History\n** 2026-07-14 Tue @ 02:03:28 -0500 — did a thing\nSCHEDULED: <2026-07-20 Mon>\n")
+ (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c))
+ (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues)))))
+ (should-not (memq 'dated-log-heading-active-timestamp cs))))
+
+(ert-deftest lo-subtask-done-not-dated-skips-specs ()
+ (let* ((c "* Work\n** TODO Parent\n*** DONE A sub-decision\n")
+ (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c))
+ (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues)))))
+ (should-not (memq 'subtask-done-not-dated cs))))
+
+(ert-deftest lo-link-checks-still-fire-on-specs ()
+ "Only the todo-format family is scoped out; a broken link in a spec still flags."
+ (let* ((c "* X\n[[file:does-not-exist-xyz.org][link]]\n")
+ (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c))
+ (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues)))))
+ (should (memq 'link-to-local-file cs))))
+
+(ert-deftest lo-task-missing-last-reviewed-skips-specs ()
+ "The fifth todo-format checker (added 2026-07-23) skips specs too — a spec's
+phases section may carry ** TODO [#x] items that aren't backlog tasks."
+ (let* ((c "* Implementation phases\n** TODO [#B] Phase one\nBody.\n")
+ (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c))
+ (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues)))))
+ (should-not (memq 'task-missing-last-reviewed cs)))
+ ;; And still fires on a normal file.
+ (let* ((c "* Work\n** TODO [#B] Real backlog task\nBody.\n")
+ (out (lo-test--run-at "todo.org" c))
+ (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues)))))
+ (should (memq 'task-missing-last-reviewed cs))))
diff --git a/.ai/scripts/tests/test-todo-cleanup.el b/.ai/scripts/tests/test-todo-cleanup.el
index ad9260b..1e964b3 100644
--- a/.ai/scripts/tests/test-todo-cleanup.el
+++ b/.ai/scripts/tests/test-todo-cleanup.el
@@ -30,16 +30,22 @@
;;; Harness
(defun tc-test--reset (&optional check)
- (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-issues nil
+ (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-archived-to-file 0 tc-issues nil
+ tc-sealed 0 tc-seal nil tc-convert-subtasks nil
tc-check-only (and check t)
tc-archive-done t tc-sync-child-priority nil
- tc-current-file nil))
+ tc-current-file nil
+ ;; Aging step OFF by default so the in-file-move tests are unaffected by
+ ;; the wall clock; the aging harness re-enables it with fixed params.
+ tc-archive-retain-days nil tc-archive-reference-date nil tc-archive-file nil))
(defun tc-test--reset-sync (&optional check)
- (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-issues nil
+ (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-archived-to-file 0 tc-issues nil
+ tc-sealed 0 tc-seal nil
tc-check-only (and check t)
tc-archive-done nil tc-sync-child-priority t
- tc-current-file nil))
+ tc-current-file nil
+ tc-archive-retain-days nil tc-archive-reference-date nil tc-archive-file nil))
(defun tc-test--drop-buffer (file)
(let ((buf (find-buffer-visiting file)))
@@ -355,6 +361,207 @@ from the heading line through (not including) the next level-1 heading or EOF."
(should (tc-test--has (plist-get out :report) "skipped"))))
;;; ---------------------------------------------------------------------------
+;;; --archive-done file-aging: keep last week in-file, move older to task-archive
+
+(defun tc-test--age (content &optional opts)
+ "Run `--archive-done' with the file-aging step enabled.
+OPTS is a plist: :retain (days; default 7, may be nil to disable), :ref
+\(YEAR MONTH DAY reference date), :runs (default 1), :check. Writes CONTENT to a
+temp todo file and points `tc-archive-file' at a not-yet-existing temp archive.
+Returns a plist: :result (todo contents), :archive (archive-file contents or
+nil), :archived (in-file move count), :to-file (aged count), :issues — all from
+the last run."
+ (let* ((retain (if (plist-member opts :retain) (plist-get opts :retain) 7))
+ (ref (plist-get opts :ref))
+ (runs (or (plist-get opts :runs) 1))
+ (check (plist-get opts :check))
+ (todo (make-temp-file "tc-age-todo-" nil ".org"))
+ (adir (make-temp-file "tc-age-arch-" t))
+ (afile (expand-file-name "task-archive.org" adir))
+ last)
+ (unwind-protect
+ (progn
+ (with-temp-file todo (insert content))
+ (dotimes (_ runs)
+ (tc-test--reset check)
+ (setq tc-archive-retain-days retain
+ tc-archive-reference-date ref
+ tc-archive-file afile)
+ (tc-process-file todo)
+ (setq last (list :archived tc-archived :to-file tc-archived-to-file
+ :issues tc-issues))
+ (tc-test--drop-buffer todo))
+ (append
+ last
+ (list :result (with-temp-buffer (insert-file-contents todo) (buffer-string))
+ :archive (and (file-readable-p afile)
+ (with-temp-buffer (insert-file-contents afile)
+ (buffer-string))))))
+ (tc-test--drop-buffer todo)
+ (delete-file todo)
+ (delete-directory adir t))))
+
+;; Reference "today" for these fixtures is 2026-06-29; with retain 7 the cutoff
+;; is 2026-06-22, so a task closed on or after 2026-06-22 stays in-file.
+(defconst tc-test--age-resolved "\
+* Age Open Work
+** TODO [#A] still open
+* Age Resolved
+** DONE [#B] recent within window
+CLOSED: [2026-06-25 Thu]
+recent body
+** DONE [#C] old beyond window
+CLOSED: [2026-05-01 Fri]
+old body line
+** CANCELLED [#C] old cancelled too
+CLOSED: [2026-04-15 Wed]
+** DONE [#B] exactly at cutoff stays
+CLOSED: [2026-06-22 Sun]
+** DONE [#C] undated no-date archived
+no closed date in this body
+")
+
+(defconst tc-test--age-straggler "\
+* Age Open Work
+** TODO [#A] still open
+** DONE [#C] old straggler
+CLOSED: [2026-03-01 Sun]
+straggler body
+* Age Resolved
+** DONE [#B] recent stays
+CLOSED: [2026-06-26 Fri]
+")
+
+(ert-deftest tc-age-moves-old-and-undated-resolved ()
+ "Normal: closed-beyond-window AND undated subtrees leave the file; only those
+closed within the window (cutoff inclusive) stay."
+ (let* ((out (tc-test--age tc-test--age-resolved '(:ref (2026 6 29))))
+ (resolved (tc-test--section (plist-get out :result) "Age Resolved"))
+ (arch (plist-get out :archive)))
+ (should (= 3 (plist-get out :to-file)))
+ (should-not (tc-test--has resolved "old beyond window"))
+ (should-not (tc-test--has resolved "old cancelled too"))
+ (should-not (tc-test--has resolved "undated no-date archived"))
+ (should (tc-test--has resolved "recent within window"))
+ (should (tc-test--has resolved "exactly at cutoff stays"))
+ (should arch)
+ (should (tc-test--has arch "Resolved (archived)"))
+ (should (tc-test--has arch "old beyond window"))
+ (should (tc-test--has arch "old body line"))
+ (should (tc-test--has arch "old cancelled too"))
+ (should (tc-test--has arch "undated no-date archived"))
+ (should-not (tc-test--has arch "recent within window"))))
+
+(ert-deftest tc-age-disabled-when-retain-nil ()
+ "Boundary: nil retain disables the aging step entirely (legacy behavior)."
+ (let ((out (tc-test--age tc-test--age-resolved '(:retain nil :ref (2026 6 29)))))
+ (should (= 0 (plist-get out :to-file)))
+ (should (equal tc-test--age-resolved (plist-get out :result)))
+ (should-not (plist-get out :archive))))
+
+(ert-deftest tc-age-is-idempotent ()
+ "Boundary: a second run finds nothing new to age; the todo file is stable."
+ (let ((once (tc-test--age tc-test--age-resolved '(:ref (2026 6 29) :runs 1)))
+ (twice (tc-test--age tc-test--age-resolved '(:ref (2026 6 29) :runs 2))))
+ (should (equal (plist-get once :result) (plist-get twice :result)))
+ (should (= 0 (plist-get twice :to-file)))))
+
+(ert-deftest tc-age-check-mode-previews-without-writing ()
+ "Boundary: --check reports the aged count but writes neither file."
+ (let ((out (tc-test--age tc-test--age-resolved '(:ref (2026 6 29) :check t))))
+ (should (= 3 (plist-get out :to-file)))
+ (should (equal tc-test--age-resolved (plist-get out :result)))
+ (should-not (plist-get out :archive))))
+
+(ert-deftest tc-age-straggler-moves-through-to-archive ()
+ "Normal: an old-dated DONE in Open Work moves to Resolved then ages out in one run."
+ (let* ((out (tc-test--age tc-test--age-straggler '(:ref (2026 6 29))))
+ (open (tc-test--section (plist-get out :result) "Age Open Work"))
+ (resolved (tc-test--section (plist-get out :result) "Age Resolved"))
+ (arch (plist-get out :archive)))
+ (should-not (tc-test--has open "old straggler"))
+ (should-not (tc-test--has resolved "old straggler"))
+ (should (tc-test--has arch "old straggler"))
+ (should (tc-test--has arch "straggler body"))
+ (should (tc-test--has resolved "recent stays"))
+ (should (= 1 (plist-get out :archived)))
+ (should (= 1 (plist-get out :to-file)))))
+
+(ert-deftest tc-age-append-preserves-existing-archive ()
+ "Error/edge: appending to a populated archive keeps prior entries and one scaffold."
+ (let* ((adir (make-temp-file "tc-arch-" t))
+ (afile (expand-file-name "task-archive.org" adir)))
+ (unwind-protect
+ (progn
+ (tc--append-subtrees-to-archive-file afile (list "** DONE one\n"))
+ (tc--append-subtrees-to-archive-file afile (list "** DONE two\n"))
+ (let ((content (with-temp-buffer (insert-file-contents afile)
+ (buffer-string)))
+ (n 0) (start 0))
+ (should (tc-test--has content "** DONE one"))
+ (should (tc-test--has content "** DONE two"))
+ (should (tc-test--before-p content "** DONE one" "** DONE two"))
+ (while (string-match "\\* Resolved (archived)" content start)
+ (setq n (1+ n) start (match-end 0)))
+ (should (= 1 n))))
+ (delete-directory adir t))))
+
+;;; ---------------------------------------------------------------------------
+;;; --archive-done aging: the archive follows the todo file's gitignore status
+
+(defun tc-test--age-in-git-repo (gitignore-todo)
+ "Init a temp git repo, write todo.org with an old Resolved entry, optionally
+gitignore todo.org, then run `--archive-done' aging with the DEFAULT archive path
+(archive/task-archive.org beside the todo file). Return a plist: :gitignore (final
+.gitignore contents or nil), :archive-ignored (whether git ignores the archive),
+:archive-exists."
+ (let* ((root (make-temp-file "tc-git-" t))
+ ;; Private backup dir: this helper writes a file literally named
+ ;; todo.org and runs a real (non-check) pass, so without this its
+ ;; backup lands in the shared temp dir under the exact production
+ ;; name and is indistinguishable from a real one.
+ (temporary-file-directory
+ (file-name-as-directory (make-temp-file "tc-git-bk-" t)))
+ (todo (expand-file-name "todo.org" root))
+ (archive (expand-file-name "archive/task-archive.org" root))
+ (gi (expand-file-name ".gitignore" root)))
+ (unwind-protect
+ (let ((default-directory root))
+ (call-process "git" nil nil nil "init" "-q")
+ (with-temp-file todo (insert tc-test--age-resolved))
+ (when gitignore-todo (with-temp-file gi (insert "/todo.org\n")))
+ (tc-test--reset nil)
+ (setq tc-archive-retain-days 7
+ tc-archive-reference-date '(2026 6 29)
+ tc-archive-file nil) ; default path, beside the todo file
+ (tc-process-file todo)
+ (tc-test--drop-buffer todo)
+ (list :gitignore (and (file-readable-p gi)
+ (with-temp-buffer (insert-file-contents gi)
+ (buffer-string)))
+ :archive-ignored
+ (eq 0 (call-process "git" nil nil nil "check-ignore" "-q" archive))
+ :archive-exists (file-readable-p archive)))
+ (delete-directory root t)
+ (delete-directory temporary-file-directory t))))
+
+(ert-deftest tc-age-self-protect-gitignores-archive-when-todo-ignored ()
+ "When the todo file is gitignored, the aged-out archive is added to .gitignore
+so it inherits the same privacy."
+ (let ((out (tc-test--age-in-git-repo t)))
+ (should (plist-get out :archive-exists))
+ (should (string-match-p "task-archive" (or (plist-get out :gitignore) "")))
+ (should (plist-get out :archive-ignored))))
+
+(ert-deftest tc-age-self-protect-leaves-tracked-todo-archive-tracked ()
+ "When the todo file is tracked, the archive is not gitignored — no .gitignore
+entry is added for it."
+ (let ((out (tc-test--age-in-git-repo nil)))
+ (should (plist-get out :archive-exists))
+ (should-not (plist-get out :archive-ignored))
+ (should-not (string-match-p "task-archive" (or (plist-get out :gitignore) "")))))
+
+;;; ---------------------------------------------------------------------------
;;; Realistic synthetic sample (committed under fixtures/)
(defun tc-test--sample-file ()
@@ -380,6 +587,95 @@ from the heading line through (not including) the next level-1 heading or EOF."
(should (> (plist-get out :archived) 0)))))
;;; ---------------------------------------------------------------------------
+;;; --archive-done retention default
+
+(ert-deftest tc-archive-retain-default-is-one-month ()
+ "The shipped retention default is one month (31 days), not the legacy 7.
+The defvar initializes from this defconst; the live var itself is mutated by
+other tests, so the immutable defconst is the stable contract to pin."
+ (should (= 31 tc-archive-retain-days-default)))
+
+;;; ---------------------------------------------------------------------------
+;;; --seal: rename the working archive to resolved-YYYY-MM-DD.org
+
+(defun tc-test--seal (&optional opts)
+ "Run `--seal' against a temp todo file with a temp archive dir.
+OPTS is a plist: :archive-content (seed task-archive.org with this; nil = no
+working archive), :ref (YEAR MONTH DAY seal date; default (2026 7 18)),
+:check, :presealed (also create resolved-<ref>.org first, to test collision).
+Returns a plist: :sealed count, :issues, :working-exists, :sealed-exists,
+:sealed-name, :report."
+ (let* ((ref (or (plist-get opts :ref) '(2026 7 18)))
+ (check (plist-get opts :check))
+ (archive-content (plist-get opts :archive-content))
+ (todo (make-temp-file "tc-seal-todo-" nil ".org"))
+ (adir (make-temp-file "tc-seal-arch-" t))
+ (afile (expand-file-name "task-archive.org" adir))
+ (sealed-name (format "resolved-%04d-%02d-%02d.org"
+ (nth 0 ref) (nth 1 ref) (nth 2 ref)))
+ (sealed (expand-file-name sealed-name adir)))
+ (unwind-protect
+ (progn
+ (with-temp-file todo (insert "* Open Work\n** TODO [#A] live\n"))
+ (when archive-content (with-temp-file afile (insert archive-content)))
+ (when (plist-get opts :presealed)
+ (with-temp-file sealed (insert "pre-existing seal\n")))
+ (tc-test--reset check)
+ ;; Set every mode flag explicitly: tc-test--reset leaves
+ ;; tc-convert-subtasks untouched, so a convert test running earlier in
+ ;; the suite would otherwise still own the dispatch and run convert.
+ (setq tc-archive-done nil tc-sync-child-priority nil
+ tc-convert-subtasks nil tc-seal t tc-sealed 0
+ tc-archive-reference-date ref
+ tc-archive-file afile)
+ (let ((report (with-output-to-string (tc-process-file todo) (tc-emit-report))))
+ (tc-test--drop-buffer todo)
+ (list :sealed tc-sealed
+ :issues tc-issues
+ :working-exists (file-readable-p afile)
+ :sealed-exists (file-readable-p sealed)
+ :sealed-name sealed-name
+ :report report)))
+ (tc-test--drop-buffer todo)
+ (delete-file todo)
+ (delete-directory adir t))))
+
+(ert-deftest tc-seal-renames-working-archive-to-dated-file ()
+ "Normal: --seal renames task-archive.org to resolved-<seal-date>.org."
+ (let ((out (tc-test--seal '(:archive-content "* Resolved (archived)\n** DONE old\n"
+ :ref (2026 7 18)))))
+ (should (= 1 (plist-get out :sealed)))
+ (should-not (plist-get out :working-exists))
+ (should (plist-get out :sealed-exists))
+ (should (equal "resolved-2026-07-18.org" (plist-get out :sealed-name)))
+ (should (tc-test--has (plist-get out :report) "sealed task-archive.org → resolved-2026-07-18.org"))))
+
+(ert-deftest tc-seal-nothing-to-seal-is-a-reported-noop ()
+ "Boundary: no working archive present — reported no-op, nothing created."
+ (let ((out (tc-test--seal '(:ref (2026 7 18)))))
+ (should (= 0 (plist-get out :sealed)))
+ (should-not (plist-get out :sealed-exists))
+ (should (tc-test--has (plist-get out :report) "no working archive to seal"))))
+
+(ert-deftest tc-seal-check-mode-previews-without-renaming ()
+ "Boundary: --check reports the seal but leaves the working archive in place."
+ (let ((out (tc-test--seal '(:archive-content "* Resolved (archived)\n"
+ :ref (2026 7 18) :check t))))
+ (should (= 1 (plist-get out :sealed)))
+ (should (plist-get out :working-exists))
+ (should-not (plist-get out :sealed-exists))
+ (should (tc-test--has (plist-get out :report) "would seal"))))
+
+(ert-deftest tc-seal-refuses-to-clobber-existing-sealed-file ()
+ "Error: resolved-<today>.org already exists — refuse, leave both files intact."
+ (let ((out (tc-test--seal '(:archive-content "* Resolved (archived)\n"
+ :ref (2026 7 18) :presealed t))))
+ (should (= 0 (plist-get out :sealed)))
+ (should (plist-get out :working-exists))
+ (should (plist-get out :sealed-exists))
+ (should (tc-test--has (plist-get out :report) "already exists"))))
+
+;;; ---------------------------------------------------------------------------
;;; Sync-child-priority harness + fixtures
(defun tc-test--sync (content &optional runs check)
@@ -570,5 +866,311 @@ in ISSUES, in document order."
(should (= 2 (plist-get once :bumped)))
(should (= 2 (plist-get twice :bumped)))))
+;;; ---------------------------------------------------------------------------
+;;; --convert-subtasks harness + tests
+
+(defun tc-test--reset-convert (&optional check)
+ (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-converted 0 tc-archived-to-file 0
+ tc-issues nil tc-sealed 0 tc-seal nil
+ tc-check-only (and check t)
+ tc-archive-done nil tc-sync-child-priority nil tc-convert-subtasks t
+ tc-current-file nil
+ tc-archive-retain-days nil tc-archive-reference-date nil tc-archive-file nil))
+
+(defun tc-test--convert (content &optional runs check)
+ "Write CONTENT to a temp .org file, run `--convert-subtasks' RUNS times (default 1).
+Return a plist: :result final file contents, :converted count from the last run,
+:issues from the last run. CHECK non-nil ⇒ --check (preview, no writes)."
+ (let ((file (make-temp-file "tc-test-" nil ".org"))
+ last-converted last-issues)
+ (unwind-protect
+ (progn
+ (with-temp-file file (insert content))
+ (dotimes (_ (or runs 1))
+ (tc-test--reset-convert check)
+ (tc-process-file file)
+ (setq last-converted tc-converted last-issues tc-issues)
+ (tc-test--drop-buffer file))
+ (list :result (with-temp-buffer (insert-file-contents file)
+ (buffer-string))
+ :converted last-converted
+ :issues last-issues))
+ (tc-test--drop-buffer file)
+ (delete-file file))))
+
+;; The UTC offset in a converted header is the test machine's local offset for
+;; that date, so assertions match it as `[-+]NNNN' rather than a fixed value —
+;; the mode's job is to emit a well-formed offset, not to run in one timezone.
+
+(defconst tc-test--convert-timed
+ "* Project Open Work
+** TODO [#B] Parent task
+*** DONE [#C] F12 opens the terminal :feature:quick:
+CLOSED: [2026-06-27 Sat 12:50]
+Verified live: docks, toggles, colors clean.
+")
+
+(ert-deftest tc-convert-timed-subtask-normal ()
+ "Normal: a timed CLOSED close becomes a dated header, keyword/priority/tags/CLOSED gone."
+ (let* ((out (tc-test--convert tc-test--convert-timed))
+ (res (plist-get out :result)))
+ (should (= 1 (plist-get out :converted)))
+ (should (string-match-p
+ "^\\*\\*\\* 2026-06-27 Sat @ 12:50:00 [-+][0-9]\\{4\\} F12 opens the terminal$"
+ res))
+ (should-not (string-match-p "CLOSED:" res))
+ (should-not (string-match-p "DONE" res))
+ (should (string-match-p "Verified live: docks, toggles, colors clean\\." res))
+ (should (string-match-p "^\\*\\* TODO \\[#B\\] Parent task$" res))))
+
+(defconst tc-test--convert-dateonly
+ "* Project Open Work
+** PROJECT [#B] Parent
+**** DONE [#B] Write full spec :refactor:
+CLOSED: [2026-05-04 Mon]
+Body.
+")
+
+(ert-deftest tc-convert-dateonly-boundary-midnight ()
+ "Boundary: a date-only CLOSED (no time) yields 00:00:00, at level 4."
+ (let ((res (plist-get (tc-test--convert tc-test--convert-dateonly) :result)))
+ (should (string-match-p
+ "^\\*\\*\\*\\* 2026-05-04 Mon @ 00:00:00 [-+][0-9]\\{4\\} Write full spec$"
+ res))
+ (should-not (string-match-p "CLOSED:" res))))
+
+(defconst tc-test--convert-level2
+ "* Project Open Work
+** DONE [#B] Top-level task
+CLOSED: [2026-06-01 Mon 09:00]
+Body.
+")
+
+(ert-deftest tc-convert-leaves-level-2-alone-boundary ()
+ "Boundary: a level-2 DONE task is a top-level task, not a sub-task — untouched."
+ (let ((out (tc-test--convert tc-test--convert-level2)))
+ (should (= 0 (plist-get out :converted)))
+ (should (equal tc-test--convert-level2 (plist-get out :result)))))
+
+(ert-deftest tc-convert-idempotent-boundary ()
+ "Boundary: a second run over an already-dated entry converts nothing new."
+ (let ((once (tc-test--convert tc-test--convert-timed 1))
+ (twice (tc-test--convert tc-test--convert-timed 2)))
+ (should (equal (plist-get once :result) (plist-get twice :result)))
+ (should (= 0 (plist-get twice :converted)))))
+
+(defconst tc-test--convert-nested
+ "* Project Open Work
+** TODO [#B] Parent
+*** DONE Outer sub :feature:
+CLOSED: [2026-06-10 Wed 08:15]
+**** DONE Inner sub
+CLOSED: [2026-06-09 Tue 07:00]
+Inner body.
+")
+
+(ert-deftest tc-convert-nested-done-subtasks-boundary ()
+ "Boundary: a done sub-task nested under a done sub-task — both convert."
+ (let* ((out (tc-test--convert tc-test--convert-nested))
+ (res (plist-get out :result)))
+ (should (= 2 (plist-get out :converted)))
+ (should (string-match-p
+ "^\\*\\*\\* 2026-06-10 Wed @ 08:15:00 [-+][0-9]\\{4\\} Outer sub$" res))
+ (should (string-match-p
+ "^\\*\\*\\*\\* 2026-06-09 Tue @ 07:00:00 [-+][0-9]\\{4\\} Inner sub$" res))
+ (should-not (string-match-p "CLOSED:" res))))
+
+(defconst tc-test--convert-cancelled
+ "* Project Open Work
+** TODO [#B] Parent
+*** CANCELLED [#C] Abandoned idea :feature:
+CLOSED: [2026-06-15 Mon 10:00]
+")
+
+(ert-deftest tc-convert-cancelled-subtask-boundary ()
+ "Boundary: a CANCELLED sub-task converts too (terminal state)."
+ (let ((res (plist-get (tc-test--convert tc-test--convert-cancelled) :result)))
+ (should (string-match-p
+ "^\\*\\*\\* 2026-06-15 Mon @ 10:00:00 [-+][0-9]\\{4\\} Abandoned idea$" res))
+ (should-not (string-match-p "CANCELLED" res))))
+
+(defconst tc-test--convert-noclosed
+ "* Project Open Work
+** TODO [#B] Parent
+*** DONE Orphan with no closed date
+Body only.
+")
+
+(ert-deftest tc-convert-skips-subtask-without-closed-error ()
+ "Error: a done sub-task with no parseable CLOSED is flagged and left unchanged."
+ (let ((out (tc-test--convert tc-test--convert-noclosed)))
+ (should (= 0 (plist-get out :converted)))
+ (should (equal tc-test--convert-noclosed (plist-get out :result)))
+ (should (cl-some (lambda (i) (eq (plist-get i :kind) 'convert-skip))
+ (plist-get out :issues)))))
+
+(ert-deftest tc-convert-check-mode-previews-without-writing ()
+ "Check mode reports the conversion but writes nothing."
+ (let ((out (tc-test--convert tc-test--convert-timed 1 t)))
+ (should (= 1 (plist-get out :converted)))
+ (should (equal tc-test--convert-timed (plist-get out :result)))
+ (should (cl-some (lambda (i) (eq (plist-get i :kind) 'convert-would))
+ (plist-get out :issues)))))
+
+(defconst tc-test--convert-closed-with-deadline
+ "* Project Open Work
+** TODO [#B] Parent task
+*** DONE [#C] Ship the panel :feature:
+CLOSED: [2026-06-27 Sat 12:50] DEADLINE: <2026-06-30 Tue>
+Body line.
+")
+
+(ert-deftest tc-convert-strips-deadline-sharing-the-planning-line-boundary ()
+ "Boundary: a DEADLINE sharing the CLOSED planning line goes too — a dated-log
+entry carries no active planning timestamp (todo-format.md). Body survives."
+ (let* ((out (tc-test--convert tc-test--convert-closed-with-deadline))
+ (res (plist-get out :result)))
+ (should (= 1 (plist-get out :converted)))
+ (should (string-match-p
+ "^\\*\\*\\* 2026-06-27 Sat @ 12:50:00 [-+][0-9]\\{4\\} Ship the panel$"
+ res))
+ (should-not (string-match-p "CLOSED:" res))
+ (should-not (string-match-p "DEADLINE:" res))
+ (should (string-match-p "^Body line\\.$" res))))
+
+(defconst tc-test--convert-closed-and-scheduled-separate-lines
+ "* Project Open Work
+** TODO [#B] Parent task
+*** DONE [#C] Book the venue :feature:
+CLOSED: [2026-06-27 Sat 12:50]
+SCHEDULED: <2026-06-20 Sat>
+Body line.
+")
+
+(ert-deftest tc-convert-strips-scheduled-on-its-own-line ()
+ "Normal (the home bug): a SCHEDULED planning line on its own — the completion
+rewrite dropped keyword/priority/tags but left the SCHEDULED, pinning the dated
+entry to the agenda as weeks-overdue. Both planning lines go; body survives."
+ (let* ((out (tc-test--convert tc-test--convert-closed-and-scheduled-separate-lines))
+ (res (plist-get out :result)))
+ (should (= 1 (plist-get out :converted)))
+ (should (string-match-p
+ "^\\*\\*\\* 2026-06-27 Sat @ 12:50:00 [-+][0-9]\\{4\\} Book the venue$"
+ res))
+ (should-not (string-match-p "CLOSED:" res))
+ (should-not (string-match-p "SCHEDULED:" res))
+ (should (string-match-p "^Body line\\.$" res))))
+
+(defconst tc-test--convert-scheduled-in-body-prose
+ "* Project Open Work
+** TODO [#B] Parent task
+*** DONE [#C] Note the mechanism :feature:
+CLOSED: [2026-06-27 Sat 12:50]
+An active SCHEDULED: <2026-06-20 Sat> in prose must survive.
+")
+
+(ert-deftest tc-convert-leaves-planning-shaped-body-prose-alone ()
+ "Boundary: a planning-shaped token inside body prose (not a canonical planning
+line) is left untouched — the strip stops at the first non-planning line."
+ (let* ((out (tc-test--convert tc-test--convert-scheduled-in-body-prose))
+ (res (plist-get out :result)))
+ (should (= 1 (plist-get out :converted)))
+ (should-not (string-match-p "CLOSED:" res))
+ (should (string-match-p "An active SCHEDULED: <2026-06-20 Sat> in prose must survive\\." res))))
+
(provide 'test-todo-cleanup)
;;; test-todo-cleanup.el ends here
+
+;;; ---------------------------------------------------------------------------
+;;; Backup before mutating (parity with lint-org.el / wrap-org-table.el)
+;;
+;; todo-cleanup rewrites todo.org in place and left no copy behind, while both
+;; sibling org-mutators back up to /tmp first. It is also the one that runs most
+;; often (every wrap, every sentry cycle). Emacs's own backup does not fire under
+;; --batch -q, so there was genuinely no undo short of git.
+
+(ert-deftest tc-backup-written-before-a-real-mutation ()
+ "A real (non-check) run leaves a copy holding the pre-edit content.
+
+`temporary-file-directory' is rebound to a private dir for the duration: the
+backup name derives from the *file's* basename, and the real todo.org shares
+that basename, so a live sentry run writing /tmp/todo.org.before-todo-cleanup.*
+would otherwise be indistinguishable from this test's own artifact. The first
+version of this test globbed the shared /tmp and passed only until a real run
+created one (2026-07-24)."
+ (let* ((dir (make-temp-file "tc-backup-" t))
+ (bdir (file-name-as-directory (make-temp-file "tc-bk-" t)))
+ (file (expand-file-name "todo.org" dir))
+ (before "* P Open Work\n** TODO [#B] parent\n*** DONE a subtask\nCLOSED: [2026-07-01 Tue]\n"))
+ (unwind-protect
+ (progn
+ (with-temp-file file (insert before))
+ (let ((tc-check-only nil)
+ (tc-convert-subtasks t)
+ (temporary-file-directory bdir))
+ (tc-process-file file))
+ (let ((backups (file-expand-wildcards
+ (concat bdir "todo.org.before-todo-cleanup.*"))))
+ (should backups)
+ (should (string-match-p
+ "a subtask"
+ (with-temp-buffer (insert-file-contents (car backups))
+ (buffer-string))))))
+ (delete-directory dir t)
+ (delete-directory bdir t))))
+
+(ert-deftest tc-no-backup-in-check-mode ()
+ "--check writes nothing, so it must not leave a backup either.
+Uses a private `temporary-file-directory' for the same isolation reason."
+ (let* ((dir (make-temp-file "tc-backup-" t))
+ (bdir (file-name-as-directory (make-temp-file "tc-bk-" t)))
+ (file (expand-file-name "todo.org" dir)))
+ (unwind-protect
+ (progn
+ (with-temp-file file
+ (insert "* P Open Work\n** TODO [#B] parent\n*** DONE sub\nCLOSED: [2026-07-01 Tue]\n"))
+ (let ((tc-check-only t)
+ (tc-convert-subtasks t)
+ (temporary-file-directory bdir))
+ (tc-process-file file))
+ (should-not (file-expand-wildcards
+ (concat bdir "todo.org.before-todo-cleanup.*"))))
+ (delete-directory dir t)
+ (delete-directory bdir t))))
+
+(ert-deftest tc-backup-never-overwrites-an-earlier-one ()
+ "Two invocations in the same second must not collapse to one backup.
+
+open-tasks.org runs --convert-subtasks then --archive-done back to back, each
+a sub-second batch run. With a second-resolution stamp and copy-file's
+OK-IF-ALREADY-EXISTS, the second invocation overwrote the first's backup with
+already-mutated content, so the true pre-session original was unrecoverable —
+the exact state the backup exists to preserve (found 2026-07-24 in review)."
+ (let* ((dir (make-temp-file "tc-collide-" t))
+ (bdir (file-name-as-directory (make-temp-file "tc-cbk-" t)))
+ (file (expand-file-name "todo.org" dir))
+ (original (concat "* P Open Work\n** TODO [#B] parent\n*** DONE sub\n"
+ "CLOSED: [2026-07-01 Tue]\n"
+ "* P Resolved\n** DONE [#C] old\nCLOSED: [2025-01-01 Wed]\n")))
+ (unwind-protect
+ (progn
+ (with-temp-file file (insert original))
+ ;; Two back-to-back invocations, as the shipped workflow does.
+ (let ((temporary-file-directory bdir))
+ (let ((tc-check-only nil) (tc-convert-subtasks t))
+ (tc-process-file file))
+ (let ((tc-check-only nil) (tc-convert-subtasks nil) (tc-archive-done t)
+ (tc-archive-retain-days nil))
+ (tc-process-file file)))
+ (let ((backups (file-expand-wildcards
+ (concat bdir "todo.org.before-todo-cleanup.*"))))
+ ;; Both invocations kept their own backup.
+ (should (= (length backups) 2))
+ ;; And one of them still holds the true original.
+ (should (cl-some (lambda (b)
+ (string= original
+ (with-temp-buffer (insert-file-contents b)
+ (buffer-string))))
+ backups))))
+ (delete-directory dir t)
+ (delete-directory bdir t))))
diff --git a/.ai/scripts/tests/test-wrap-org-table.el b/.ai/scripts/tests/test-wrap-org-table.el
index 8d1ecb6..0b3b375 100644
--- a/.ai/scripts/tests/test-wrap-org-table.el
+++ b/.ai/scripts/tests/test-wrap-org-table.el
@@ -186,3 +186,45 @@
(should (string-match-p "Prose before\\." content))
(should (string-match-p "Prose after\\." content))))
(delete-file file))))
+
+;;; ---------------------------------------------------------------------------
+;;; block safety — pipe lines inside #+begin_/#+end_ blocks are never tables
+
+(defconst wot-test--block-content
+ "#+begin_example
+| client |----->| server |
+| box | | box |
+#+end_example
+"
+ "An example block whose ASCII-art lines start with pipes.")
+
+(defun wot-test--process-content (content budget)
+ "Write CONTENT to a temp file, run `wot-process-file' at BUDGET, return result."
+ (let ((file (make-temp-file "wot-test" nil ".org")))
+ (unwind-protect
+ (progn
+ (with-temp-file file (insert content))
+ (wot-process-file file budget)
+ (with-temp-buffer (insert-file-contents file) (buffer-string)))
+ (delete-file file))))
+
+(ert-deftest wot-process-file-leaves-example-block-byte-identical ()
+ (let ((content (concat "* Diagram\n\n" wot-test--block-content)))
+ (should (equal (wot-test--process-content content 120) content))))
+
+(ert-deftest wot-process-file-reformats-table-but-not-block ()
+ (let* ((content (concat "* Doc\n\n" wot-test--block-content "\n"
+ wot-test--wide-input))
+ (result (wot-test--process-content content 40)))
+ (should (string-match-p (regexp-quote wot-test--block-content) result))
+ (should (string-match-p (regexp-quote wot-test--wide-expected) result))))
+
+(ert-deftest wot-process-file-skips-pipes-in-src-block ()
+ (let ((content "* Pipeline\n\n#+begin_src sh\n| sort\n| uniq -c\n#+end_src\n"))
+ (should (equal (wot-test--process-content content 120) content))))
+
+(ert-deftest wot-process-file-literal-inner-end-marker-stays-in-block ()
+ "A literal #+end_src quoted inside an example block must not close it."
+ (let ((content (concat "* Doc\n\n#+begin_example\n#+begin_src sh\nx\n"
+ "#+end_src\n| art |----| art |\n#+end_example\n")))
+ (should (equal (wot-test--process-content content 120) content))))
diff --git a/.ai/scripts/tests/test_apkg_to_orgdrill.py b/.ai/scripts/tests/test_apkg_to_orgdrill.py
new file mode 100644
index 0000000..6a95ea4
--- /dev/null
+++ b/.ai/scripts/tests/test_apkg_to_orgdrill.py
@@ -0,0 +1,301 @@
+"""Tests for apkg-to-orgdrill.py — the inverse of flashcard-to-anki.py.
+
+The converter reads an Anki .apkg (a zip holding collection.anki2 / .anki21
+sqlite) and emits an org-drill .org in the house canonical shape. It is
+stdlib-only (zipfile + sqlite3), so it imports directly — no genanki stub.
+
+The apkg schema these tests build by hand mirrors what genanki actually
+writes, confirmed against a real apkg generated from flashcard-to-anki.py:
+ - col.decks : JSON {did: {"name": ...}}, always including id-1 "Default"
+ - col.models : JSON {mid: {"name": ..., "flds": [{"name": "Front"}, ...]}}
+ - notes.flds : fields joined by \x1f; tags space-padded (" tag ")
+ - cards : nid -> did (the Default deck carries no cards)
+
+The round-trip test closes the loop through flashcard-to-anki.py's own
+parse(): original org -> forward parse tuples -> apkg fixture -> converter
+-> recovered org -> forward parse -> assert the (front, back, tag) tuples
+match. Only the apkg materialization is hand-built (the genanki boundary);
+everything else is the real code on both sides.
+"""
+from __future__ import annotations
+
+import importlib.util
+import json
+import sqlite3
+import sys
+import types
+import zipfile
+from pathlib import Path
+
+import pytest
+
+SCRIPTS = Path(__file__).resolve().parents[1]
+CONVERTER = SCRIPTS / "apkg-to-orgdrill.py"
+FORWARD = SCRIPTS / "flashcard-to-anki.py"
+
+
+def _load(path: Path, name: str, stub_genanki: bool = False):
+ if stub_genanki:
+ sys.modules.setdefault("genanki", types.ModuleType("genanki"))
+ spec = importlib.util.spec_from_file_location(name, path)
+ assert spec and spec.loader
+ module = importlib.util.module_from_spec(spec)
+ # Register before exec: @dataclass resolves cls.__module__ via sys.modules
+ # (Python 3.14), which is None for an unregistered importlib module.
+ sys.modules[name] = module
+ spec.loader.exec_module(module)
+ return module
+
+
+@pytest.fixture(scope="module")
+def conv():
+ return _load(CONVERTER, "apkg_to_orgdrill")
+
+
+@pytest.fixture(scope="module")
+def forward():
+ return _load(FORWARD, "flashcard_to_anki", stub_genanki=True)
+
+
+# --- fixture builder: write a genanki-shaped apkg by hand ------------------
+
+def _make_apkg(
+ path: Path,
+ decks: dict[int, str],
+ models: dict[int, list[str]],
+ notes: list[tuple[int, int, list[str], str]], # (nid, mid, fields, tag)
+ cards: list[tuple[int, int]], # (nid, did)
+ *,
+ media: str = "{}",
+) -> None:
+ """Materialize a minimal apkg matching genanki's collection.anki2 shape."""
+ col_dir = path.parent / f"{path.stem}-build"
+ col_dir.mkdir(parents=True, exist_ok=True)
+ db = col_dir / "collection.anki2"
+ if db.exists():
+ db.unlink()
+ con = sqlite3.connect(db)
+ con.execute("CREATE TABLE col (id INTEGER, decks TEXT, models TEXT)")
+ decks_json = {"1": {"name": "Default"}}
+ decks_json.update({str(did): {"name": name} for did, name in decks.items()})
+ models_json = {
+ str(mid): {"name": f"{decks.get(list(decks)[0], 'M')} model",
+ "flds": [{"name": n, "ord": i} for i, n in enumerate(flds)]}
+ for mid, flds in models.items()
+ }
+ con.execute("INSERT INTO col (id, decks, models) VALUES (1, ?, ?)",
+ (json.dumps(decks_json), json.dumps(models_json)))
+ con.execute("CREATE TABLE notes (id INTEGER, mid INTEGER, flds TEXT, tags TEXT)")
+ for nid, mid, fields, tag in notes:
+ con.execute("INSERT INTO notes (id, mid, flds, tags) VALUES (?, ?, ?, ?)",
+ (nid, mid, "\x1f".join(fields), f" {tag} " if tag else " "))
+ con.execute("CREATE TABLE cards (id INTEGER, nid INTEGER, did INTEGER)")
+ for i, (nid, did) in enumerate(cards):
+ con.execute("INSERT INTO cards (id, nid, did) VALUES (?, ?, ?)", (1000 + i, nid, did))
+ con.commit()
+ con.close()
+ with zipfile.ZipFile(path, "w") as z:
+ z.write(db, "collection.anki2")
+ z.writestr("media", media)
+
+
+# --- html_to_org_body ------------------------------------------------------
+
+def test_html_to_org_splits_br_into_lines(conv):
+ assert conv.html_to_org_body("one<br>two<br>three") == ["one", "two", "three"]
+
+
+def test_html_to_org_handles_br_variants(conv):
+ assert conv.html_to_org_body("a<br/>b<br />c<BR>d") == ["a", "b", "c", "d"]
+
+
+def test_html_to_org_unescapes_entities_amp_last(conv):
+ # Inverts escape_html (which escapes & first): &lt; &gt; &amp; -> < > &.
+ assert conv.html_to_org_body("x &lt;tag&gt; &amp; y") == ["x <tag> & y"]
+
+
+def test_html_to_org_preserves_a_literal_escaped_entity(conv):
+ # Forward-escaping the literal "&lt;" yields "&amp;lt;"; the inverse must
+ # recover "&lt;", not "<".
+ assert conv.html_to_org_body("&amp;lt;") == ["&lt;"]
+
+
+def test_html_to_org_strips_answer_hr(conv):
+ assert conv.html_to_org_body('front<hr id="answer">back') == ["front", "back"]
+
+
+def test_html_to_org_empty_back_is_empty(conv):
+ assert conv.html_to_org_body("") == []
+
+
+# --- read_apkg -------------------------------------------------------------
+
+def test_read_apkg_single_deck_recovers_front_back_tag_deck(conv, tmp_path):
+ apkg = tmp_path / "d.apkg"
+ _make_apkg(
+ apkg,
+ decks={20: "My Deck"},
+ models={9: ["Front", "Back"]},
+ notes=[(100, 9, ["Q1?", "A1.<br>line2"], "sec-one")],
+ cards=[(100, 20)],
+ )
+ recovered = conv.read_apkg(apkg)
+ assert len(recovered) == 1
+ note = recovered[0]
+ assert note.deck == "My Deck"
+ assert note.front == "Q1?"
+ assert note.back_html == "A1.<br>line2"
+ assert note.tag == "sec-one"
+
+
+def test_read_apkg_multiple_decks_grouped(conv, tmp_path):
+ apkg = tmp_path / "multi.apkg"
+ _make_apkg(
+ apkg,
+ decks={20: "Deck A", 21: "Deck B"},
+ models={9: ["Front", "Back"]},
+ notes=[(100, 9, ["QA?", "AA"], "ta"), (101, 9, ["QB?", "AB"], "tb")],
+ cards=[(100, 20), (101, 21)],
+ )
+ decks = {n.deck for n in conv.read_apkg(apkg)}
+ assert decks == {"Deck A", "Deck B"}
+
+
+def test_read_apkg_skips_default_deck_without_cards(conv, tmp_path):
+ apkg = tmp_path / "def.apkg"
+ _make_apkg(
+ apkg,
+ decks={20: "Real Deck"},
+ models={9: ["Front", "Back"]},
+ notes=[(100, 9, ["Q?", "A"], "t")],
+ cards=[(100, 20)],
+ )
+ assert {n.deck for n in conv.read_apkg(apkg)} == {"Real Deck"}
+
+
+def test_read_apkg_warns_and_skips_non_basic_model(conv, tmp_path, capsys):
+ apkg = tmp_path / "cloze.apkg"
+ _make_apkg(
+ apkg,
+ decks={20: "Cloze Deck"},
+ models={9: ["Text", "Extra"]}, # not Front/Back
+ notes=[(100, 9, ["some {{c1::text}}", "extra"], "t")],
+ cards=[(100, 20)],
+ )
+ recovered = conv.read_apkg(apkg)
+ assert recovered == []
+ assert "skip" in capsys.readouterr().err.lower()
+
+
+def test_read_apkg_reads_anki21_collection_name(conv, tmp_path):
+ # A .anki21 collection filename must be read the same as .anki2.
+ apkg = tmp_path / "new.apkg"
+ _make_apkg(
+ apkg,
+ decks={20: "Deck"},
+ models={9: ["Front", "Back"]},
+ notes=[(100, 9, ["Q?", "A"], "t")],
+ cards=[(100, 20)],
+ )
+ # Rewrite the zip renaming the collection member to .anki21.
+ with zipfile.ZipFile(apkg) as z:
+ data = z.read("collection.anki2")
+ media = z.read("media")
+ with zipfile.ZipFile(apkg, "w") as z:
+ z.writestr("collection.anki21", data)
+ z.writestr("media", media)
+ assert conv.read_apkg(apkg)[0].front == "Q?"
+
+
+def test_read_apkg_flags_media_reference(conv, tmp_path, capsys):
+ apkg = tmp_path / "media.apkg"
+ _make_apkg(
+ apkg,
+ decks={20: "Deck"},
+ models={9: ["Front", "Back"]},
+ notes=[(100, 9, ["Q?", 'see <img src="x.png">'], "t")],
+ cards=[(100, 20)],
+ )
+ conv.read_apkg(apkg)
+ assert "media" in capsys.readouterr().err.lower()
+
+
+# --- notes_to_org ----------------------------------------------------------
+
+def test_notes_to_org_emits_canonical_shape(conv):
+ Note = conv.Note
+ notes = [
+ Note(deck="My Deck", front="Q1?", back_html="A1.", tag="alpha"),
+ Note(deck="My Deck", front="Q2?", back_html="A2.", tag="alpha"),
+ ]
+ ids = iter(["id-1", "id-2"])
+ org = conv.notes_to_org(notes, "My Deck", new_id=lambda: next(ids))
+ assert "#+TITLE: My Deck" in org
+ assert "* alpha" in org
+ assert "** Q1? :drill:" in org
+ assert ":ID: id-1" in org
+ assert ":ID: id-2" in org
+ assert org.count("* alpha") == 1 # both cards share one section
+
+
+def test_notes_to_org_distinct_tags_get_distinct_sections(conv):
+ Note = conv.Note
+ notes = [
+ Note(deck="D", front="Qa?", back_html="a", tag="alpha"),
+ Note(deck="D", front="Qb?", back_html="b", tag="beta"),
+ ]
+ org = conv.notes_to_org(notes, "D", new_id=lambda: "x")
+ assert "* alpha" in org and "* beta" in org
+
+
+# --- round-trip through the real forward parse() ---------------------------
+
+def test_round_trip_matches_forward_parse_tuples(conv, forward, tmp_path):
+ original = (
+ "#+TITLE: RT Deck\n"
+ "\n"
+ "* First Section\n"
+ "** What is 2+2? :drill:\n"
+ ":PROPERTIES:\n:ID: aaaa\n:END:\n"
+ "Four.\n"
+ "Second line with <angle> & amp.\n"
+ "\n"
+ "* Second Section\n"
+ "** Capital of France? :drill:\n"
+ "Paris.\n"
+ )
+ tuples = forward.parse(original) # [(front, back_html, anki_tags), ...]
+ assert len(tuples) == 2
+
+ apkg = tmp_path / "rt.apkg"
+ _make_apkg(
+ apkg,
+ decks={20: "RT Deck"},
+ models={9: ["Front", "Back"]},
+ # anki_tags is a list; the apkg tags field is space-joined.
+ notes=[(100 + i, 9, [f, b], " ".join(tags))
+ for i, (f, b, tags) in enumerate(tuples)],
+ cards=[(100 + i, 20) for i in range(len(tuples))],
+ )
+
+ by_deck = conv.convert(apkg)
+ assert set(by_deck) == {"RT Deck"}
+ recovered_tuples = forward.parse(by_deck["RT Deck"])
+ assert recovered_tuples == tuples
+
+
+# --- errors ----------------------------------------------------------------
+
+def test_read_apkg_missing_collection_errors(conv, tmp_path):
+ bad = tmp_path / "bad.apkg"
+ with zipfile.ZipFile(bad, "w") as z:
+ z.writestr("media", "{}")
+ with pytest.raises(Exception):
+ conv.read_apkg(bad)
+
+
+def test_read_apkg_not_a_zip_errors(conv, tmp_path):
+ notzip = tmp_path / "plain.apkg"
+ notzip.write_text("not a zip")
+ with pytest.raises(Exception):
+ conv.read_apkg(notzip)
diff --git a/.ai/scripts/tests/test_cj_remove_block.py b/.ai/scripts/tests/test_cj_remove_block.py
index 2c8dade..3cdee46 100644
--- a/.ai/scripts/tests/test_cj_remove_block.py
+++ b/.ai/scripts/tests/test_cj_remove_block.py
@@ -14,6 +14,34 @@ import pytest
SCRIPT = Path(__file__).parent.parent / "cj-remove-block.py"
+@pytest.fixture(autouse=True)
+def isolated_tmpdir(tmp_path, monkeypatch):
+ """Give every test in this module a private TMPDIR.
+
+ The script backs up to the system temp dir under a name derived from the
+ edited file's BASENAME. The real todo.org shares that basename, so any test
+ operating on a fixture named todo.org writes something indistinguishable
+ from a production backup — and an earlier version of this file globbed the
+ shared /tmp and unlinked every match, so a routine `make test` destroyed
+ Craig's real backups (found in review, 2026-07-24).
+
+ Isolating at module scope rather than per-test is deliberate: the same bug
+ was fixed once in the elisp sibling and left here, so relying on each new
+ test to remember is exactly how it recurred. Autouse makes it structural.
+ """
+ d = tmp_path / "_tmpdir"
+ d.mkdir()
+ # TMPDIR covers subprocess invocations of the script.
+ monkeypatch.setenv("TMPDIR", str(d))
+ # tempfile.gettempdir() caches its answer on first call, so a test that
+ # loads the module in-process would keep writing to the real /tmp no matter
+ # what TMPDIR says. Override the cache too — this is the gap that made the
+ # env-var-only version still leak one backup per suite run.
+ import tempfile as _tempfile
+ monkeypatch.setattr(_tempfile, "tempdir", str(d))
+ return d
+
+
@pytest.fixture
def run_remove(tmp_path):
"""Write content to a temp org file, run cj-remove-block, return new contents."""
@@ -155,3 +183,142 @@ class TestCjRemoveBlockSafety:
err, post_content = run_remove_expecting_failure(original, start=4, end=2)
assert err.returncode != 0
assert post_content == original
+
+
+class TestMultiBlockRangeRefused:
+ """The validation exists to catch a drifted range, but it only checked the
+ first and last lines of that range. A span from one block's opening fence to
+ a LATER block's closing fence passed, and the removal silently deleted every
+ line between — real prose, headings, whole tasks — with a zero exit. Drift is
+ the skill's normal operating mode (respond-to-cj-comments edits the file as it
+ processes, and a file under cj review usually holds several blocks), so this
+ is the exact scenario the check was written for. Reproduced 2026-07-24."""
+
+ TWO_BLOCKS = (
+ "* Alpha\n"
+ "#+begin_src cj:\n"
+ "note A\n"
+ "#+end_src\n"
+ "KEEP THIS LINE\n"
+ "* Beta\n"
+ "#+begin_src cj:\n"
+ "note B\n"
+ "#+end_src\n"
+ )
+
+ def test_range_spanning_two_blocks_is_refused(self, run_remove_expecting_failure):
+ # Lines 2..9: block one's opener through block two's closer.
+ err, content = run_remove_expecting_failure(self.TWO_BLOCKS, 2, 9)
+ assert err.returncode == 1
+ assert "KEEP THIS LINE" in content, "content between the blocks was destroyed"
+ assert "* Beta" in content, "a heading between the blocks was destroyed"
+
+ def test_refusal_names_the_reason(self, run_remove_expecting_failure):
+ err, _ = run_remove_expecting_failure(self.TWO_BLOCKS, 2, 9)
+ assert "more than one" in err.stderr.decode().lower()
+
+ def test_a_correct_single_block_range_still_removes(self, run_remove):
+ # The fix must not over-tighten: the legitimate range still works.
+ out = run_remove(self.TWO_BLOCKS, 2, 4)
+ assert "note A" not in out
+ assert "KEEP THIS LINE" in out
+ assert "note B" in out, "the second block must be untouched"
+
+ def test_a_nested_end_src_inside_the_range_is_refused(self, run_remove_expecting_failure):
+ # Any #+end_src before the final line means the range covers >1 block.
+ content = (
+ "#+begin_src cj:\n"
+ "a\n"
+ "#+end_src\n"
+ "middle\n"
+ "#+begin_src cj:\n"
+ "b\n"
+ "#+end_src\n"
+ )
+ err, after = run_remove_expecting_failure(content, 1, 7)
+ assert err.returncode == 1
+ assert "middle" in after
+
+
+class TestSafeMutation:
+ """The script rewrites Craig's org files (todo.org, notes.org). It wrote with
+ a bare write_text, which truncates the target on open, and took no backup —
+ so a mid-write failure left the file truncated with no copy to recover from.
+ lint-org.el, the other tool that mutates these files, backs up to a temp dir
+ first. Match that, and make the write atomic.
+
+ Every test here redirects TMPDIR to a private directory. The backup name
+ derives from the file's basename, and the real todo.org shares it, so a test
+ globbing the shared temp dir cannot tell its own artifact from a genuine
+ backup — and an earlier version of this class globbed /tmp and unlinked every
+ match, so a routine `make test` destroyed real backups (found in review,
+ 2026-07-24). Never glob or delete across the shared temp dir."""
+
+ ONE_BLOCK = "* T\n#+begin_src cj:\nnote\n#+end_src\nkeep\n"
+
+ def test_a_backup_is_written_before_mutating(self, tmp_path):
+ import subprocess, glob, os
+ bdir = tmp_path / "bk"
+ bdir.mkdir()
+ f = tmp_path / "todo.org"
+ f.write_text(self.ONE_BLOCK)
+ subprocess.run(
+ ["python3", str(SCRIPT), "--file", str(f), "--start", "2", "--end", "4"],
+ check=True, capture_output=True,
+ env={**os.environ, "TMPDIR": str(bdir)},
+ )
+ backups = glob.glob(str(bdir / "todo.org.before-cj-remove.*"))
+ assert backups, "no backup was written before mutating the org file"
+ assert "note" in Path(max(backups)).read_text()
+
+ def test_no_partial_file_when_the_write_fails(self, tmp_path, monkeypatch):
+ import importlib.util
+ spec = importlib.util.spec_from_file_location("crb", SCRIPT)
+ mod = importlib.util.module_from_spec(spec)
+ spec.loader.exec_module(mod)
+ bdir = tmp_path / "bk"
+ bdir.mkdir()
+ monkeypatch.setenv("TMPDIR", str(bdir))
+ f = tmp_path / "todo.org"
+ f.write_text(self.ONE_BLOCK)
+ def boom(*a, **k):
+ raise OSError("disk full")
+ monkeypatch.setattr(mod.os, "replace", boom)
+ with pytest.raises(OSError):
+ mod.remove_range(f, 2, 4)
+ # The original survives intact — no truncation, no partial.
+ assert f.read_text() == self.ONE_BLOCK
+
+
+class TestBackupNeverOverwrites:
+ """Same defect class as todo-cleanup's, and more reachable here: the
+ respond-to-cj-comments skill removes several annotations in quick
+ succession, so a second-resolution stamp collides and the later backup
+ overwrote the earlier one with already-mutated content."""
+
+ TWO_BLOCKS = (
+ "* A\n#+begin_src cj:\nfirst\n#+end_src\n"
+ "* B\n#+begin_src cj:\nsecond\n#+end_src\n"
+ )
+
+ def test_consecutive_removals_each_keep_a_backup(self, tmp_path, monkeypatch):
+ import subprocess, glob
+ bdir = tmp_path / "bk"
+ bdir.mkdir()
+ monkeypatch.setenv("TMPDIR", str(bdir))
+ f = tmp_path / "todo.org"
+ f.write_text(self.TWO_BLOCKS)
+ original = f.read_text()
+ # Remove the second block, then the first — back to back, same second.
+ subprocess.run(["python3", str(SCRIPT), "--file", str(f),
+ "--start", "6", "--end", "8"],
+ check=True, capture_output=True,
+ env={**__import__("os").environ, "TMPDIR": str(bdir)})
+ subprocess.run(["python3", str(SCRIPT), "--file", str(f),
+ "--start", "2", "--end", "4"],
+ check=True, capture_output=True,
+ env={**__import__("os").environ, "TMPDIR": str(bdir)})
+ backups = glob.glob(str(bdir / "todo.org.before-cj-remove.*"))
+ assert len(backups) == 2, f"expected 2 backups, got {len(backups)}"
+ contents = [Path(b).read_text() for b in backups]
+ assert original in contents, "no backup holds the true original"
diff --git a/.ai/scripts/tests/test_cross_agent_discover.py b/.ai/scripts/tests/test_cross_agent_discover.py
deleted file mode 100644
index f0d2bb7..0000000
--- a/.ai/scripts/tests/test_cross_agent_discover.py
+++ /dev/null
@@ -1,204 +0,0 @@
-"""Tests for cross-agent-discover (TDD: tests written before implementation)."""
-
-from __future__ import annotations
-
-import json
-import os
-import subprocess
-import textwrap
-from pathlib import Path
-
-import pytest
-
-SCRIPT = Path(__file__).resolve().parent.parent / "cross-agent-comms" / "cross-agent-discover"
-
-
-def _run(args: list[str], env: dict | None = None) -> subprocess.CompletedProcess:
- return subprocess.run([str(SCRIPT), *args], capture_output=True, text=True, env=env)
-
-
-@pytest.fixture
-def fake_home(tmp_path, monkeypatch):
- home = tmp_path / "home"
- home.mkdir()
- monkeypatch.setenv("HOME", str(home))
- return home
-
-
-def _make_project(home: Path, name: str) -> Path:
- proj = home / "projects" / name
- (proj / ".ai").mkdir(parents=True)
- return proj
-
-
-def _write_peers_toml(home: Path, content: str) -> Path:
- cfg = home / ".config" / "cross-agent-comms"
- cfg.mkdir(parents=True, exist_ok=True)
- peers = cfg / "peers.toml"
- peers.write_text(content)
- return peers
-
-
-def test_discover_help(fake_home):
- result = _run(["--help"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0
- assert "discover" in result.stdout.lower() or "enumerate" in result.stdout.lower()
-
-
-def test_discover_local_only_no_projects(fake_home):
- """Empty home → reports zero local projects, zero peers."""
- result = _run(["--no-cache"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0
- # No crash; mentions local somehow.
- assert "local" in result.stdout.lower() or "0 project" in result.stdout.lower()
-
-
-def test_discover_lists_local_projects(fake_home):
- _make_project(fake_home, "homelab")
- _make_project(fake_home, "career")
- _make_project(fake_home, "claude-templates")
- result = _run(["--no-cache"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0
- assert "homelab" in result.stdout
- assert "career" in result.stdout
- assert "claude-templates" in result.stdout
-
-
-def test_discover_excludes_dirs_without_ai_subdir(fake_home):
- """Directories under ~/projects/ that lack .ai/ are NOT projects."""
- _make_project(fake_home, "real-project")
- (fake_home / "projects" / "not-a-project").mkdir(parents=True)
- result = _run(["--no-cache"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0
- assert "real-project" in result.stdout
- assert "not-a-project" not in result.stdout
-
-
-def test_discover_no_peers_toml_just_local(fake_home):
- _make_project(fake_home, "homelab")
- result = _run(["--no-cache"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0
- # No peers section since no toml.
- assert "homelab" in result.stdout
-
-
-def test_discover_lists_peers_from_toml(fake_home):
- _write_peers_toml(fake_home, textwrap.dedent("""\
- [peers.velox]
- host = "velox"
- ssh_user = "cjennings"
-
- [peers.bastion]
- host = "bastion.local"
- ssh_user = "cjennings"
- """))
- _make_project(fake_home, "homelab")
- result = _run(["--no-cache"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0
- assert "velox" in result.stdout
- assert "bastion" in result.stdout
-
-
-def test_discover_malformed_peers_toml_errors_clearly(fake_home):
- _write_peers_toml(fake_home, "not valid toml at all = = =")
- result = _run(["--no-cache"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode != 0
- assert "peers.toml" in result.stderr or "TOML" in result.stderr or "parse" in result.stderr.lower()
-
-
-def test_discover_json_output_schema(fake_home):
- _make_project(fake_home, "homelab")
- _make_project(fake_home, "career")
- _write_peers_toml(fake_home, textwrap.dedent("""\
- [peers.velox]
- host = "velox"
- """))
- result = _run(["--json", "--no-cache"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0
- payload = json.loads(result.stdout)
- assert "local" in payload
- assert "peers" in payload
- assert isinstance(payload["local"], list)
- assert isinstance(payload["peers"], list)
- assert "homelab" in payload["local"]
- assert "career" in payload["local"]
- velox = next((p for p in payload["peers"] if p["name"] == "velox"), None)
- assert velox is not None
- # Reachability is a key — value depends on actual SSH state.
- assert "reachable" in velox
-
-
-def test_discover_peer_scope(fake_home):
- _write_peers_toml(fake_home, textwrap.dedent("""\
- [peers.velox]
- host = "velox"
-
- [peers.bastion]
- host = "bastion.local"
- """))
- result = _run(["--peer", "velox", "--no-cache", "--json"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0
- payload = json.loads(result.stdout)
- peer_names = [p["name"] for p in payload["peers"]]
- assert "velox" in peer_names
- assert "bastion" not in peer_names
-
-
-def test_discover_unreachable_peer_marked(fake_home):
- """A peer with a definitely-unreachable host gets reachable=False."""
- _write_peers_toml(fake_home, textwrap.dedent("""\
- [peers.bogus]
- host = "definitely-not-a-real-host.invalid"
- ssh_user = "nobody"
- """))
- result = _run(["--no-cache", "--json"], env={**os.environ, "HOME": str(fake_home)}, )
- assert result.returncode == 0
- payload = json.loads(result.stdout)
- bogus = next((p for p in payload["peers"] if p["name"] == "bogus"), None)
- assert bogus is not None
- assert bogus["reachable"] is False
-
-
-def test_discover_cache_hit_within_window(fake_home):
- """Second invocation within 5 min reads cache (skip the SSH probe)."""
- _make_project(fake_home, "homelab")
- # First call populates cache.
- result1 = _run(["--json"], env={**os.environ, "HOME": str(fake_home)})
- assert result1.returncode == 0
- cache = fake_home / ".cache" / "cross-agent-comms" / "discovery.json"
- assert cache.exists()
- # Tamper with the cache to a marker only the cache path can produce.
- payload = json.loads(cache.read_text())
- payload["_test_marker"] = True
- cache.write_text(json.dumps(payload))
- # Second call (no --no-cache) should return the tampered payload.
- result2 = _run(["--json"], env={**os.environ, "HOME": str(fake_home)})
- assert result2.returncode == 0
- payload2 = json.loads(result2.stdout)
- assert payload2.get("_test_marker") is True
-
-
-def test_discover_no_cache_flag_bypasses(fake_home):
- """--no-cache ignores even a fresh cache."""
- _make_project(fake_home, "homelab")
- cache_dir = fake_home / ".cache" / "cross-agent-comms"
- cache_dir.mkdir(parents=True)
- cache_dir.joinpath("discovery.json").write_text(json.dumps({
- "_test_marker": True, "local": [], "peers": []
- }))
- result = _run(["--no-cache", "--json"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0
- payload = json.loads(result.stdout)
- # Cache marker should NOT appear in fresh result.
- assert payload.get("_test_marker") is None or payload.get("_test_marker") is False
- assert "homelab" in payload["local"]
-
-
-def test_discover_halt_shows_banner(fake_home):
- halt = fake_home / ".config" / "cross-agent-comms" / "HALT"
- halt.parent.mkdir(parents=True)
- halt.write_text("halted")
- _make_project(fake_home, "homelab")
- result = _run(["--no-cache"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0 # discover continues to print under HALT
- assert "HALT" in result.stdout
diff --git a/.ai/scripts/tests/test_cross_agent_halt.py b/.ai/scripts/tests/test_cross_agent_halt.py
deleted file mode 100644
index f8bf0b3..0000000
--- a/.ai/scripts/tests/test_cross_agent_halt.py
+++ /dev/null
@@ -1,204 +0,0 @@
-"""Tests for cross-agent-halt and cross-agent-resume (TDD)."""
-
-from __future__ import annotations
-
-import os
-import subprocess
-import textwrap
-from pathlib import Path
-
-import pytest
-
-HALT_SCRIPT = Path(__file__).resolve().parent.parent / "cross-agent-comms" / "cross-agent-halt"
-RESUME_SCRIPT = Path(__file__).resolve().parent.parent / "cross-agent-comms" / "cross-agent-resume"
-
-
-def _run(script: Path, args: list[str], env: dict | None = None) -> subprocess.CompletedProcess:
- return subprocess.run([str(script), *args], capture_output=True, text=True, env=env)
-
-
-@pytest.fixture
-def isolated_env(tmp_path, monkeypatch):
- """Isolated HOME + a fake systemctl that records calls without acting."""
- fake_home = tmp_path / "home"
- fake_home.mkdir()
- fake_bin = tmp_path / "bin"
- fake_bin.mkdir()
- # Fake systemctl: no-op, exit 0.
- fake_systemctl = fake_bin / "systemctl"
- fake_systemctl.write_text("#!/usr/bin/env bash\nexit 0\n")
- fake_systemctl.chmod(0o755)
- # Fake ssh: succeed only for known-good host.
- fake_ssh = fake_bin / "ssh"
- fake_ssh.write_text(textwrap.dedent("""\
- #!/usr/bin/env bash
- # Find the destination arg (skip flags).
- target=""
- for arg in "$@"; do
- case "$arg" in
- -*|*=*) ;;
- *@*|localhost|*.local|*.invalid) target="$arg"; break ;;
- *) target="$arg"; break ;;
- esac
- done
- case "$target" in
- *invalid*|*unreachable*) exit 255 ;;
- *) exit 0 ;;
- esac
- """))
- fake_ssh.chmod(0o755)
-
- monkeypatch.setenv("HOME", str(fake_home))
- # Prepend our fake bin so systemctl + ssh are intercepted, but keep real /bin etc.
- monkeypatch.setenv("PATH", f"{fake_bin}:{os.environ.get('PATH', '')}")
- return fake_home
-
-
-# ---- cross-agent-halt ----
-
-
-def test_halt_help(isolated_env):
- result = _run(HALT_SCRIPT, ["--help"], env={**os.environ, "HOME": str(isolated_env),
- "PATH": os.environ["PATH"]})
- assert result.returncode == 0
- assert "halt" in result.stdout.lower()
-
-
-def test_halt_creates_halt_file(isolated_env):
- halt_file = isolated_env / ".config" / "cross-agent-comms" / "HALT"
- assert not halt_file.exists()
- result = _run(HALT_SCRIPT, [], env={**os.environ, "HOME": str(isolated_env),
- "PATH": os.environ["PATH"]})
- assert result.returncode == 0
- assert halt_file.exists()
-
-
-def test_halt_with_reason_writes_body(isolated_env):
- result = _run(HALT_SCRIPT, ["pausing for incident review"],
- env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert result.returncode == 0
- halt_file = isolated_env / ".config" / "cross-agent-comms" / "HALT"
- assert halt_file.exists()
- assert "pausing for incident review" in halt_file.read_text()
-
-
-def test_halt_idempotent(isolated_env):
- """Running halt twice doesn't error."""
- halt_file = isolated_env / ".config" / "cross-agent-comms" / "HALT"
- r1 = _run(HALT_SCRIPT, [], env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert r1.returncode == 0
- assert halt_file.exists()
- r2 = _run(HALT_SCRIPT, [], env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert r2.returncode == 0
- assert halt_file.exists()
-
-
-def test_halt_does_not_pkill(isolated_env):
- """Per design: halt does NOT call pkill. Verify by checking no pkill process gets launched."""
- # Replace pkill in PATH with something that fails loudly so we'd see if halt invoked it.
- fake_bin = isolated_env.parent / "bin"
- pkill = fake_bin / "pkill"
- pkill.write_text("#!/usr/bin/env bash\necho 'PKILL CALLED' >&2\nexit 99\n")
- pkill.chmod(0o755)
- result = _run(HALT_SCRIPT, [], env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert result.returncode == 0
- assert "PKILL CALLED" not in result.stderr
-
-
-def test_halt_tailnet_reports_per_peer(isolated_env):
- """--tailnet iterates peers.toml and reports per-peer status."""
- cfg = isolated_env / ".config" / "cross-agent-comms"
- cfg.mkdir(parents=True)
- (cfg / "peers.toml").write_text(textwrap.dedent("""\
- [peers.velox]
- host = "velox"
- ssh_user = "cjennings"
-
- [peers.bogus]
- host = "definitely-unreachable.invalid"
- ssh_user = "cjennings"
- """))
- result = _run(HALT_SCRIPT, ["--tailnet"],
- env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- # Partial halt → exit 1.
- assert result.returncode == 1
- assert "velox" in result.stdout
- assert "bogus" in result.stdout
- # ✓ marker for velox, ✗ for bogus.
- assert "✓" in result.stdout
- assert "✗" in result.stdout
- assert "PARTIAL" in result.stdout or "partial" in result.stdout.lower()
-
-
-def test_halt_tailnet_all_reachable_exits_zero(isolated_env):
- cfg = isolated_env / ".config" / "cross-agent-comms"
- cfg.mkdir(parents=True)
- (cfg / "peers.toml").write_text(textwrap.dedent("""\
- [peers.velox]
- host = "velox"
- ssh_user = "cjennings"
- """))
- result = _run(HALT_SCRIPT, ["--tailnet"],
- env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert result.returncode == 0
- assert "velox" in result.stdout
-
-
-# ---- cross-agent-resume ----
-
-
-def test_resume_help(isolated_env):
- result = _run(RESUME_SCRIPT, ["--help"],
- env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert result.returncode == 0
- assert "resume" in result.stdout.lower()
-
-
-def test_resume_removes_halt_file(isolated_env):
- halt_file = isolated_env / ".config" / "cross-agent-comms" / "HALT"
- halt_file.parent.mkdir(parents=True)
- halt_file.write_text("halted")
- assert halt_file.exists()
- result = _run(RESUME_SCRIPT, [],
- env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert result.returncode == 0
- assert not halt_file.exists()
-
-
-def test_resume_when_no_halt_active_succeeds(isolated_env):
- """No HALT to clear is not an error."""
- result = _run(RESUME_SCRIPT, [],
- env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert result.returncode == 0
-
-
-def test_resume_prints_per_session_instructions(isolated_env):
- """Resume must surface that polling does NOT auto-resume."""
- halt_file = isolated_env / ".config" / "cross-agent-comms" / "HALT"
- halt_file.parent.mkdir(parents=True)
- halt_file.write_text("halted")
- result = _run(RESUME_SCRIPT, [],
- env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert result.returncode == 0
- out = result.stdout.lower()
- assert "polling" in out
- assert "auto" in out or "explicit" in out or "session" in out
-
-
-def test_resume_tailnet_partial_failure_exit_1(isolated_env):
- cfg = isolated_env / ".config" / "cross-agent-comms"
- cfg.mkdir(parents=True)
- (cfg / "peers.toml").write_text(textwrap.dedent("""\
- [peers.velox]
- host = "velox"
-
- [peers.bogus]
- host = "unreachable-host.invalid"
- """))
- halt_file = cfg / "HALT"
- halt_file.write_text("halted")
- result = _run(RESUME_SCRIPT, ["--tailnet"],
- env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert result.returncode == 1
- assert "velox" in result.stdout
- assert "bogus" in result.stdout
diff --git a/.ai/scripts/tests/test_cross_agent_recv.py b/.ai/scripts/tests/test_cross_agent_recv.py
deleted file mode 100644
index 27c53a5..0000000
--- a/.ai/scripts/tests/test_cross_agent_recv.py
+++ /dev/null
@@ -1,176 +0,0 @@
-"""Tests for cross-agent-recv."""
-
-from __future__ import annotations
-
-import json
-import os
-import subprocess
-from pathlib import Path
-
-import pytest
-
-SCRIPT = Path(__file__).resolve().parent.parent / "cross-agent-comms" / "cross-agent-recv"
-
-
-def _make_message(path: Path, *, conv_id: str = "test-conv", seq: int = 1, msg_type: str = "request",
- proto_version: str = "5", title: str = "Test", requires_tools: str | None = None,
- body: str = "Body.\n") -> Path:
- fm_lines = [
- f"#+TITLE: {title}",
- f"#+CONVERSATION_ID: {conv_id}",
- f"#+MESSAGE_TYPE: {msg_type}",
- f"#+SEQUENCE: {seq}",
- "#+TIMESTAMP: 2026-04-27T05:00:00-05:00",
- f"#+PROTOCOL_VERSION: {proto_version}",
- ]
- if requires_tools:
- fm_lines.append(f"#+REQUIRES_TOOLS: {requires_tools}")
- path.write_text("\n".join(fm_lines) + "\n\n" + body)
- return path
-
-
-def _run(args: list[str], env: dict | None = None) -> subprocess.CompletedProcess:
- return subprocess.run([str(SCRIPT), *args], capture_output=True, text=True, env=env)
-
-
-@pytest.fixture
-def isolated_env(tmp_path, monkeypatch):
- fake_home = tmp_path / "home"
- fake_home.mkdir()
- monkeypatch.setenv("HOME", str(fake_home))
- return fake_home
-
-
-def test_recv_help(isolated_env):
- result = _run(["--help"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 0
- assert "Receive and decide" in result.stdout
-
-
-def test_recv_missing_file_rejects(isolated_env, tmp_path):
- result = _run([str(tmp_path / "nope.org")], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 3 # reject
-
-
-def test_recv_malformed_frontmatter_rejects(isolated_env, tmp_path):
- bad = tmp_path / "bad.org"
- bad.write_text("not org-mode at all\n")
- result = _run([str(bad), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 3
- assert "decision: reject" in result.stdout
-
-
-def test_recv_missing_required_field_rejects(isolated_env, tmp_path):
- msg = tmp_path / "msg.org"
- # Missing PROTOCOL_VERSION among others.
- msg.write_text("#+TITLE: x\n#+CONVERSATION_ID: c\n\nBody.\n")
- result = _run([str(msg), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 3
- assert "missing required" in result.stdout
-
-
-def test_recv_protocol_version_mismatch_query(isolated_env, tmp_path):
- msg = _make_message(tmp_path / "msg.org", proto_version="4")
- result = _run([str(msg), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 2 # query
- assert "PROTOCOL_VERSION mismatch" in result.stdout
-
-
-def test_recv_invalid_message_type_rejects(isolated_env, tmp_path):
- msg = _make_message(tmp_path / "msg.org", msg_type="banana")
- result = _run([str(msg), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 3
- assert "invalid MESSAGE_TYPE" in result.stdout
-
-
-def test_recv_missing_signature_rejects(isolated_env, tmp_path):
- """When verify is on, a missing .asc sibling rejects."""
- msg = _make_message(tmp_path / "msg.org")
- # No .asc sidecar.
- result = _run([str(msg)], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 3
- assert "signature file missing" in result.stdout
-
-
-def test_recv_valid_processes(isolated_env, tmp_path):
- """A valid message with --no-verify and no dedup match → process."""
- msg = _make_message(tmp_path / "msg.org")
- result = _run([str(msg), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 0 # process
- assert "decision: process" in result.stdout
- assert "sha256:" in result.stdout
-
-
-def test_recv_dedup_against_identical_existing(isolated_env, tmp_path):
- """Same content + same SEQUENCE in same dir → dedup."""
- inbox = tmp_path / "inbox"
- inbox.mkdir()
- first = _make_message(inbox / "20260427T100000Z-from-x-c.org", conv_id="c", seq=5)
- # Second message with same content — name differs (canonical-style would have different timestamp).
- second = _make_message(inbox / "20260427T100100Z-from-x-c.org", conv_id="c", seq=5)
- # Bodies must be byte-identical for hash equality.
- second.write_bytes(first.read_bytes())
- result = _run([str(second), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 1 # dedup
- assert "decision: dedup" in result.stdout
-
-
-def test_recv_collision_with_different_content_processes(isolated_env, tmp_path):
- """Same SEQUENCE + same CONVERSATION_ID but different content → process both."""
- inbox = tmp_path / "inbox"
- inbox.mkdir()
- _make_message(inbox / "20260427T100000Z-from-x-c.org", conv_id="c", seq=5, body="First body.\n")
- second = _make_message(inbox / "20260427T100100Z-from-x-c.org", conv_id="c", seq=5, body="Different body.\n")
- result = _run([str(second), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 0 # process
- assert "decision: process" in result.stdout
-
-
-def test_recv_requires_tools_missing_query(isolated_env, tmp_path):
- """REQUIRES_TOOLS naming a definitely-missing binary → query."""
- msg = _make_message(tmp_path / "msg.org", requires_tools="definitely-not-installed-xyzzy-9000")
- result = _run([str(msg), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 2 # query
- assert "required tools unavailable" in result.stdout
-
-
-def test_recv_requires_tools_present_processes(isolated_env, tmp_path):
- """REQUIRES_TOOLS naming a real binary → process."""
- msg = _make_message(tmp_path / "msg.org", requires_tools="ls,cat")
- result = _run([str(msg), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 0
- assert "decision: process" in result.stdout
-
-
-def test_recv_json_output(isolated_env, tmp_path):
- msg = _make_message(tmp_path / "msg.org")
- result = _run([str(msg), "--no-verify", "--json"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 0
- payload = json.loads(result.stdout)
- assert payload["decision"] == "process"
- assert payload["message_type"] == "request"
- assert payload["conversation_id"] == "test-conv"
-
-
-def test_recv_halt_blocks(isolated_env, tmp_path):
- halt = isolated_env / ".config" / "cross-agent-comms" / "HALT"
- halt.parent.mkdir(parents=True)
- halt.write_text("halted\n")
- msg = _make_message(tmp_path / "msg.org")
- result = _run([str(msg), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 5
- assert "halt active" in result.stderr.lower()
-
-
-def test_recv_halt_leaves_message_in_place(isolated_env, tmp_path):
- """Per spec: under HALT, recv must NOT move/dedup/reject — leave file in place."""
- halt = isolated_env / ".config" / "cross-agent-comms" / "HALT"
- halt.parent.mkdir(parents=True)
- halt.write_text("halted\n")
- msg = _make_message(tmp_path / "msg.org")
- pre_content = msg.read_text()
- result = _run([str(msg), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 5
- # File still exists with same content.
- assert msg.exists()
- assert msg.read_text() == pre_content
diff --git a/.ai/scripts/tests/test_cross_agent_send.py b/.ai/scripts/tests/test_cross_agent_send.py
deleted file mode 100644
index f716e95..0000000
--- a/.ai/scripts/tests/test_cross_agent_send.py
+++ /dev/null
@@ -1,210 +0,0 @@
-"""Tests for cross-agent-send.
-
-Subprocess-based: treat the script as a black-box CLI and assert on its
-exit codes, stdout, and the files it produces.
-"""
-
-from __future__ import annotations
-
-import os
-import subprocess
-import textwrap
-from pathlib import Path
-
-import pytest
-
-SCRIPT = Path(__file__).resolve().parent.parent / "cross-agent-comms" / "cross-agent-send"
-
-
-def _make_message(tmp_path: Path, conv_id: str = "test-conv", seq: int = 1, msg_type: str = "request",
- proto_version: str = "5") -> Path:
- msg = tmp_path / "msg.org"
- msg.write_text(textwrap.dedent(f"""\
- #+TITLE: Test message
- #+CONVERSATION_ID: {conv_id}
- #+MESSAGE_TYPE: {msg_type}
- #+SEQUENCE: {seq}
- #+TIMESTAMP: 2026-04-27T05:00:00-05:00
- #+PROTOCOL_VERSION: {proto_version}
-
- Body.
- """))
- return msg
-
-
-def _run(args: list[str], env: dict | None = None, cwd: Path | None = None) -> subprocess.CompletedProcess:
- return subprocess.run(
- [str(SCRIPT), *args],
- capture_output=True,
- text=True,
- env=env,
- cwd=cwd,
- )
-
-
-@pytest.fixture
-def isolated_env(tmp_path, monkeypatch):
- """Redirect HOME so peers.toml, HALT, marker files are scoped to the test."""
- fake_home = tmp_path / "home"
- fake_home.mkdir()
- monkeypatch.setenv("HOME", str(fake_home))
- # Pre-create projects/ so derive_sender_project has somewhere to look.
- (fake_home / "projects" / "homelab").mkdir(parents=True)
- return fake_home
-
-
-def test_send_help(isolated_env):
- """--help works without side effects."""
- result = _run(["--help"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 0
- assert "Send a cross-agent message" in result.stdout
-
-
-def test_send_missing_message_file(isolated_env):
- """Nonexistent message file returns general error."""
- import socket
- machine = socket.gethostname().split(".")[0]
- result = _run(
- [f"{machine}.homelab", str(isolated_env / "nonexistent.org")],
- env={**os.environ, "HOME": str(isolated_env)},
- )
- assert result.returncode == 1
- assert "not found" in result.stderr.lower()
-
-
-def test_send_invalid_destination_format(isolated_env, tmp_path):
- """Destination without . returns dest-not-found exit code."""
- msg = _make_message(tmp_path)
- result = _run(
- ["bogus", str(msg)],
- env={**os.environ, "HOME": str(isolated_env)},
- )
- assert result.returncode == 2
- assert "<machine>.<project>" in result.stderr or "destination" in result.stderr.lower()
-
-
-def test_send_dest_not_in_peers(isolated_env, tmp_path):
- """Cross-machine destination with no peers.toml entry exits 2."""
- msg = _make_message(tmp_path)
- result = _run(
- ["unknownmachine.homelab", str(msg)],
- env={**os.environ, "HOME": str(isolated_env)},
- )
- assert result.returncode == 2
- assert "not found in peers" in result.stderr
-
-
-def test_send_frontmatter_missing_required(isolated_env, tmp_path):
- """Message missing required fields exits 4."""
- bad = tmp_path / "bad.org"
- bad.write_text("#+TITLE: nope\n\nBody.\n")
- import socket
- machine = socket.gethostname().split(".")[0]
- result = _run(
- [f"{machine}.homelab", str(bad)],
- env={**os.environ, "HOME": str(isolated_env)},
- )
- assert result.returncode == 4
- assert "missing required fields" in result.stderr
-
-
-def test_send_invalid_message_type(isolated_env, tmp_path):
- """Unknown MESSAGE_TYPE exits 4."""
- msg = _make_message(tmp_path, msg_type="frobnicate")
- import socket
- machine = socket.gethostname().split(".")[0]
- result = _run(
- [f"{machine}.homelab", str(msg)],
- env={**os.environ, "HOME": str(isolated_env)},
- )
- assert result.returncode == 4
- assert "MESSAGE_TYPE" in result.stderr
-
-
-def test_send_halt_blocks(isolated_env, tmp_path):
- """When HALT exists, send refuses with exit 5."""
- halt = isolated_env / ".config" / "cross-agent-comms" / "HALT"
- halt.parent.mkdir(parents=True)
- halt.write_text("test halt\n")
- msg = _make_message(tmp_path)
- import socket
- machine = socket.gethostname().split(".")[0]
- result = _run(
- [f"{machine}.homelab", str(msg)],
- env={**os.environ, "HOME": str(isolated_env)},
- )
- assert result.returncode == 5
- assert "halt active" in result.stderr.lower()
-
-
-def test_send_same_machine_no_sign_delivers(isolated_env, tmp_path):
- """Same-machine delivery with --no-sign produces a canonically named file."""
- msg = _make_message(tmp_path, conv_id="my-conv")
- import socket
- machine = socket.gethostname().split(".")[0]
- # Sender is derived from CWD walking up to ~/projects/<name>/
- cwd = isolated_env / "projects" / "homelab"
- result = _run(
- [f"{machine}.homelab", str(msg), "--no-sign"],
- env={**os.environ, "HOME": str(isolated_env)},
- cwd=cwd,
- )
- assert result.returncode == 0, f"stderr={result.stderr}"
- inbox = isolated_env / "projects" / "homelab" / "inbox" / "from-agents"
- files = list(inbox.glob("*-from-homelab-my-conv.org"))
- assert len(files) == 1
- # No sig file with --no-sign.
- assert not list(inbox.glob("*.asc"))
- # Canonical filename pattern.
- assert files[0].name.startswith("2026") and files[0].name.endswith("-from-homelab-my-conv.org")
-
-
-def test_send_same_machine_signed_writes_asc(isolated_env, tmp_path):
- """Signed delivery writes both .org and .asc."""
- msg = _make_message(tmp_path, conv_id="signed-conv")
- import socket
- machine = socket.gethostname().split(".")[0]
- cwd = isolated_env / "projects" / "homelab"
- # Use the real GPG keyring (not isolating GPG — Craig's existing keys are fine for tests).
- real_env = {**os.environ, "HOME": str(isolated_env), "GNUPGHOME": str(Path.home() / ".gnupg")}
- result = _run(
- [f"{machine}.homelab", str(msg)],
- env=real_env,
- cwd=cwd,
- )
- if result.returncode != 0:
- pytest.skip(f"GPG signing unavailable in this environment: {result.stderr}")
- inbox = isolated_env / "projects" / "homelab" / "inbox" / "from-agents"
- org_files = list(inbox.glob("*-from-homelab-signed-conv.org"))
- asc_files = list(inbox.glob("*-from-homelab-signed-conv.org.asc"))
- assert len(org_files) == 1
- assert len(asc_files) == 1
-
-
-def test_send_filename_ignores_input_basename(isolated_env, tmp_path):
- """User's input filename is ignored; canonical filename is generated."""
- weird = tmp_path / "weird-user-name.org"
- weird.write_text(textwrap.dedent("""\
- #+TITLE: Title
- #+CONVERSATION_ID: ignored-input
- #+MESSAGE_TYPE: request
- #+SEQUENCE: 1
- #+TIMESTAMP: 2026-04-27T05:00:00-05:00
- #+PROTOCOL_VERSION: 5
-
- Body.
- """))
- import socket
- machine = socket.gethostname().split(".")[0]
- cwd = isolated_env / "projects" / "homelab"
- result = _run(
- [f"{machine}.homelab", str(weird), "--no-sign"],
- env={**os.environ, "HOME": str(isolated_env)},
- cwd=cwd,
- )
- assert result.returncode == 0
- inbox = isolated_env / "projects" / "homelab" / "inbox" / "from-agents"
- # No file named after the user's input.
- assert not (inbox / "weird-user-name.org").exists()
- # Canonical naming used.
- assert list(inbox.glob("*-from-homelab-ignored-input.org"))
diff --git a/.ai/scripts/tests/test_cross_agent_status.py b/.ai/scripts/tests/test_cross_agent_status.py
deleted file mode 100644
index bb5b8ba..0000000
--- a/.ai/scripts/tests/test_cross_agent_status.py
+++ /dev/null
@@ -1,165 +0,0 @@
-"""Tests for cross-agent-status (TDD: tests written before implementation)."""
-
-from __future__ import annotations
-
-import json
-import os
-import subprocess
-import textwrap
-from pathlib import Path
-
-import pytest
-
-SCRIPT = Path(__file__).resolve().parent.parent / "cross-agent-comms" / "cross-agent-status"
-
-
-def _make_msg(path: Path, *, conv_id: str, seq: int, msg_type: str = "request",
- proto_version: str = "5", timestamp: str = "2026-04-27T05:00:00-05:00") -> Path:
- path.parent.mkdir(parents=True, exist_ok=True)
- path.write_text(textwrap.dedent(f"""\
- #+TITLE: T
- #+CONVERSATION_ID: {conv_id}
- #+MESSAGE_TYPE: {msg_type}
- #+SEQUENCE: {seq}
- #+TIMESTAMP: {timestamp}
- #+PROTOCOL_VERSION: {proto_version}
-
- Body.
- """))
- return path
-
-
-def _run(args: list[str], env: dict | None = None) -> subprocess.CompletedProcess:
- return subprocess.run([str(SCRIPT), *args], capture_output=True, text=True, env=env)
-
-
-@pytest.fixture
-def fake_projects(tmp_path, monkeypatch):
- """Create a fake ~/projects/<name>/inbox/from-agents/ tree under tmp_path."""
- home = tmp_path / "home"
- home.mkdir()
- monkeypatch.setenv("HOME", str(home))
- return home
-
-
-def test_status_help(fake_projects):
- result = _run(["--help"], env={**os.environ, "HOME": str(fake_projects)})
- assert result.returncode == 0
- assert "snapshot" in result.stdout.lower() or "pending" in result.stdout.lower()
-
-
-def test_status_no_projects_clean_output(fake_projects):
- result = _run([], env={**os.environ, "HOME": str(fake_projects)})
- assert result.returncode == 0
- # Empty machine prints either header-only table or "no projects" — accept either.
- # No crash, no pending claims.
- assert "pending" in result.stdout.lower() or result.stdout.strip() == ""
-
-
-def test_status_one_pending_shows_up(fake_projects):
- inbox = fake_projects / "projects" / "homelab" / "inbox" / "from-agents"
- _make_msg(inbox / "20260427T100000Z-from-career-fixup.org", conv_id="fixup", seq=1)
- result = _run([], env={**os.environ, "HOME": str(fake_projects)})
- assert result.returncode == 0
- assert "homelab" in result.stdout
- assert "1" in result.stdout # pending count
- assert "20260427T100000Z-from-career-fixup.org" in result.stdout
-
-
-def test_status_released_conversation_zero_pending(fake_projects):
- """A conversation with a release message in it counts as 0 pending."""
- inbox = fake_projects / "projects" / "homelab" / "inbox" / "from-agents"
- _make_msg(inbox / "20260427T100000Z-from-career-done.org", conv_id="done", seq=1)
- _make_msg(inbox / "20260427T100100Z-from-homelab-done.org", conv_id="done", seq=2, msg_type="release")
- result = _run([], env={**os.environ, "HOME": str(fake_projects)})
- assert result.returncode == 0
- # Check the homelab row shows 0 pending.
- lines = [ln for ln in result.stdout.splitlines() if "homelab" in ln]
- # At least one homelab line should show 0 pending or "—".
- assert any("0" in ln or "—" in ln for ln in lines)
-
-
-def test_status_partial_release(fake_projects):
- """Conversation with release + a later message → that later message counts as pending."""
- inbox = fake_projects / "projects" / "homelab" / "inbox" / "from-agents"
- _make_msg(inbox / "20260427T100000Z-from-career-x.org", conv_id="x", seq=1,
- timestamp="2026-04-27T05:00:00-05:00")
- _make_msg(inbox / "20260427T100100Z-from-homelab-x.org", conv_id="x", seq=2, msg_type="release",
- timestamp="2026-04-27T05:01:00-05:00")
- # New message AFTER release: starts a fresh thread that's pending.
- _make_msg(inbox / "20260427T200000Z-from-career-x.org", conv_id="x", seq=3,
- timestamp="2026-04-27T15:00:00-05:00")
- result = _run([], env={**os.environ, "HOME": str(fake_projects)})
- assert result.returncode == 0
- homelab_line = next(ln for ln in result.stdout.splitlines() if "homelab" in ln)
- assert "1" in homelab_line # the post-release message is pending
-
-
-def test_status_multiple_projects(fake_projects):
- inbox_a = fake_projects / "projects" / "homelab" / "inbox" / "from-agents"
- inbox_b = fake_projects / "projects" / "career" / "inbox" / "from-agents"
- _make_msg(inbox_a / "20260427T100000Z-from-x-a.org", conv_id="a", seq=1)
- _make_msg(inbox_b / "20260427T100100Z-from-x-b.org", conv_id="b", seq=1)
- _make_msg(inbox_b / "20260427T100200Z-from-x-c.org", conv_id="c", seq=1)
- result = _run([], env={**os.environ, "HOME": str(fake_projects)})
- assert result.returncode == 0
- # career has 2 pending, homelab has 1.
- career_line = next(ln for ln in result.stdout.splitlines() if "career" in ln)
- homelab_line = next(ln for ln in result.stdout.splitlines() if "homelab" in ln)
- assert "2" in career_line
- assert "1" in homelab_line
-
-
-def test_status_json_output(fake_projects):
- inbox = fake_projects / "projects" / "homelab" / "inbox" / "from-agents"
- _make_msg(inbox / "20260427T100000Z-from-career-test.org", conv_id="test", seq=1)
- result = _run(["--json"], env={**os.environ, "HOME": str(fake_projects)})
- assert result.returncode == 0
- payload = json.loads(result.stdout)
- assert "projects" in payload
- assert isinstance(payload["projects"], list)
- homelab = next((p for p in payload["projects"] if p["name"] == "homelab"), None)
- assert homelab is not None
- assert homelab["pending_count"] == 1
-
-
-def test_status_sort_pending_first(fake_projects):
- """Projects with pending messages sort before projects with 0."""
- (fake_projects / "projects" / "alpha" / "inbox" / "from-agents").mkdir(parents=True)
- inbox_zeta = fake_projects / "projects" / "zeta" / "inbox" / "from-agents"
- _make_msg(inbox_zeta / "20260427T100000Z-from-x-z.org", conv_id="z", seq=1)
- result = _run([], env={**os.environ, "HOME": str(fake_projects)})
- assert result.returncode == 0
- lines = result.stdout.splitlines()
- zeta_idx = next(i for i, ln in enumerate(lines) if "zeta" in ln)
- alpha_idx = next(i for i, ln in enumerate(lines) if "alpha" in ln)
- assert zeta_idx < alpha_idx, "pending project should sort before zero-pending project"
-
-
-def test_status_halt_shows_banner(fake_projects):
- halt = fake_projects / ".config" / "cross-agent-comms" / "HALT"
- halt.parent.mkdir(parents=True)
- halt.write_text("halted for test")
- inbox = fake_projects / "projects" / "homelab" / "inbox" / "from-agents"
- _make_msg(inbox / "20260427T100000Z-from-x-x.org", conv_id="x", seq=1)
- result = _run([], env={**os.environ, "HOME": str(fake_projects)})
- assert result.returncode == 0 # status continues to print under HALT
- assert "HALT" in result.stdout
- # Banner should mention the reason.
- assert "halted for test" in result.stdout
-
-
-def test_status_projects_glob_override(fake_projects):
- inbox = fake_projects / "projects" / "homelab" / "inbox" / "from-agents"
- _make_msg(inbox / "20260427T100000Z-from-x-a.org", conv_id="a", seq=1)
- other_inbox = fake_projects / "projects" / "career" / "inbox" / "from-agents"
- _make_msg(other_inbox / "20260427T100100Z-from-x-b.org", conv_id="b", seq=1)
- # Glob limits to homelab only.
- result = _run(
- ["--projects-glob", str(fake_projects / "projects" / "homelab" / "inbox" / "from-agents") + "/"],
- env={**os.environ, "HOME": str(fake_projects)},
- )
- assert result.returncode == 0
- assert "homelab" in result.stdout
- # career not in scope.
- assert "career" not in result.stdout
diff --git a/.ai/scripts/tests/test_cross_agent_watch.py b/.ai/scripts/tests/test_cross_agent_watch.py
deleted file mode 100644
index 417cc19..0000000
--- a/.ai/scripts/tests/test_cross_agent_watch.py
+++ /dev/null
@@ -1,155 +0,0 @@
-"""Tests for cross-agent-watch.
-
-Black-box: spawn the script, drop files into a watched dir, read the log.
-Tests use --no-notify to avoid firing real desktop notifications.
-"""
-
-from __future__ import annotations
-
-import os
-import subprocess
-import time
-from pathlib import Path
-
-import pytest
-
-SCRIPT = Path(__file__).resolve().parent.parent / "cross-agent-comms" / "cross-agent-watch"
-
-
-def _spawn(watched_dir: Path, log_path: Path, env: dict) -> subprocess.Popen:
- return subprocess.Popen(
- [
- str(SCRIPT),
- "--projects-glob", str(watched_dir) + "/",
- "--log", str(log_path),
- "--no-notify",
- "--quiet",
- ],
- stdout=subprocess.DEVNULL,
- stderr=subprocess.PIPE,
- env=env,
- )
-
-
-def _wait_for_log_lines(log_path: Path, expected: int, timeout: float = 5.0) -> list[str]:
- deadline = time.time() + timeout
- while time.time() < deadline:
- if log_path.exists():
- lines = [ln for ln in log_path.read_text().splitlines() if ln]
- if len(lines) >= expected:
- return lines
- time.sleep(0.1)
- if log_path.exists():
- return [ln for ln in log_path.read_text().splitlines() if ln]
- return []
-
-
-@pytest.fixture
-def isolated_env(tmp_path, monkeypatch):
- fake_home = tmp_path / "home"
- fake_home.mkdir()
- monkeypatch.setenv("HOME", str(fake_home))
- return fake_home
-
-
-def test_watch_help(isolated_env):
- result = subprocess.run(
- [str(SCRIPT), "--help"],
- capture_output=True, text=True,
- env={**os.environ, "HOME": str(isolated_env)},
- )
- assert result.returncode == 0
- assert "Usage:" in result.stdout
-
-
-def test_watch_empty_glob_exits_nonzero(isolated_env):
- """Glob resolving to zero dirs should exit non-zero with a clear message."""
- result = subprocess.run(
- [str(SCRIPT), "--projects-glob", "/nonexistent/path/*/foo/", "--no-notify", "--quiet"],
- capture_output=True, text=True,
- env={**os.environ, "HOME": str(isolated_env)},
- timeout=3,
- )
- assert result.returncode != 0
- assert "0 directories" in result.stderr
-
-
-def test_watch_logs_org_file_create(isolated_env, tmp_path):
- watched = tmp_path / "watched"
- watched.mkdir()
- log = tmp_path / "watch.log"
- proc = _spawn(watched, log, {**os.environ, "HOME": str(isolated_env)})
- try:
- # Give inotifywait a moment to attach.
- time.sleep(0.3)
- (watched / "test-msg.org").write_text("hello")
- lines = _wait_for_log_lines(log, expected=1, timeout=3.0)
- assert len(lines) >= 1
- assert "test-msg.org" in lines[-1]
- finally:
- proc.terminate()
- proc.wait(timeout=2)
-
-
-def test_watch_filters_tmp_files(isolated_env, tmp_path):
- """Files starting with .tmp. must NOT trigger log entries."""
- watched = tmp_path / "watched"
- watched.mkdir()
- log = tmp_path / "watch.log"
- proc = _spawn(watched, log, {**os.environ, "HOME": str(isolated_env)})
- try:
- time.sleep(0.3)
- (watched / ".tmp.staging-file.org").write_text("hello")
- # Wait briefly to confirm nothing logs.
- time.sleep(0.5)
- if log.exists():
- content = log.read_text()
- assert ".tmp.staging-file" not in content
- # Then drop a real file to confirm watcher is alive.
- (watched / "real.org").write_text("real")
- lines = _wait_for_log_lines(log, expected=1, timeout=3.0)
- assert any("real.org" in ln for ln in lines)
- finally:
- proc.terminate()
- proc.wait(timeout=2)
-
-
-def test_watch_filters_asc_sidecars(isolated_env, tmp_path):
- """Only .org events fire; .asc sidecars are silent."""
- watched = tmp_path / "watched"
- watched.mkdir()
- log = tmp_path / "watch.log"
- proc = _spawn(watched, log, {**os.environ, "HOME": str(isolated_env)})
- try:
- time.sleep(0.3)
- (watched / "msg.org.asc").write_text("sig")
- time.sleep(0.5)
- if log.exists():
- assert "msg.org.asc" not in log.read_text()
- # .org event still works.
- (watched / "msg.org").write_text("body")
- lines = _wait_for_log_lines(log, expected=1, timeout=3.0)
- assert any(ln.endswith("msg.org") for ln in lines)
- finally:
- proc.terminate()
- proc.wait(timeout=2)
-
-
-def test_watch_halt_suppresses_but_logs(isolated_env, tmp_path):
- """When HALT is set, watcher logs the event with (suppressed by HALT) marker."""
- halt = isolated_env / ".config" / "cross-agent-comms" / "HALT"
- halt.parent.mkdir(parents=True)
- halt.write_text("halted")
- watched = tmp_path / "watched"
- watched.mkdir()
- log = tmp_path / "watch.log"
- proc = _spawn(watched, log, {**os.environ, "HOME": str(isolated_env)})
- try:
- time.sleep(0.3)
- (watched / "halted-event.org").write_text("body")
- lines = _wait_for_log_lines(log, expected=1, timeout=3.0)
- assert len(lines) >= 1
- assert "suppressed by HALT" in lines[-1]
- finally:
- proc.terminate()
- proc.wait(timeout=2)
diff --git a/.ai/scripts/tests/test_flashcard_stats.py b/.ai/scripts/tests/test_flashcard_stats.py
index 606f7c1..46deccc 100644
--- a/.ai/scripts/tests/test_flashcard_stats.py
+++ b/.ai/scripts/tests/test_flashcard_stats.py
@@ -217,6 +217,31 @@ def test_parse_cards_captures_body_without_drawer_planning_or_answer_header(stat
assert c["body"] == "the real answer"
+def test_parse_cards_counts_a_multitag_heading_as_a_card(stats):
+ """A card multi-tagged :fundamental:drill: still counts; the front is clean."""
+ text = "* Sec\n** Q multi? :fundamental:drill:\nthe answer\n"
+ cards, _ = stats.parse_cards(text.splitlines())
+ assert len(cards) == 1
+ assert cards[0]["heading"] == "Q multi?"
+ assert cards[0]["body"] == "the answer"
+
+
+def test_parse_cards_ignores_a_tagged_heading_without_drill(stats):
+ """A tagged heading missing :drill: is not a drill card."""
+ text = "* Sec\n** Just a note :note:\nbody\n"
+ cards, _ = stats.parse_cards(text.splitlines())
+ assert cards == []
+
+
+def test_parse_cards_body_stops_at_next_multitag_card(stats):
+ """The body scan ends at the next L2 card even when it is multi-tagged."""
+ text = "** Q1? :a:drill:\nbody1\n** Q2? :drill:\nbody2\n"
+ cards, _ = stats.parse_cards(text.splitlines())
+ assert len(cards) == 2
+ assert cards[0]["body"] == "body1"
+ assert cards[1]["body"] == "body2"
+
+
def test_find_duplicate_fronts_matches_normalized_headings(stats):
cards = [
{"heading": "What is LEO?"},
diff --git a/.ai/scripts/tests/test_flashcard_to_anki.py b/.ai/scripts/tests/test_flashcard_to_anki.py
index 058b0cd..fa38b64 100644
--- a/.ai/scripts/tests/test_flashcard_to_anki.py
+++ b/.ai/scripts/tests/test_flashcard_to_anki.py
@@ -34,14 +34,33 @@ def test_default_output_path_targets_phone_anki_dir(drill):
assert result == Path.home() / "sync" / "phone" / "anki" / "health-drill.apkg"
-def test_default_deck_name_is_raw_basename(drill):
- """Deck name is the input basename with case preserved; #+TITLE is ignored."""
- assert drill.default_deck_name(Path("/x/deepsat.org")) == "deepsat"
+def test_default_deck_name_uses_org_title(drill):
+ """The #+TITLE drives the Anki deck name, not the filename slug."""
+ org = "#+TITLE: Refutations\n* Section\n** Q? :drill:\na\n"
+ assert drill.default_deck_name(Path("/x/refutation-drill.org"), org) == "Refutations"
-def test_default_deck_name_keeps_hyphens(drill):
- """A hyphenated basename is kept verbatim rather than title-cased."""
- assert drill.default_deck_name(Path("/x/health-drill.org")) == "health-drill"
+def test_default_deck_name_title_is_trimmed(drill):
+ """Surrounding whitespace on the #+TITLE value is stripped."""
+ org = "#+TITLE: DeepSat Flashcards \n"
+ assert drill.default_deck_name(Path("/x/deepsat.org"), org) == "DeepSat Flashcards"
+
+
+def test_default_deck_name_title_match_is_case_insensitive(drill):
+ """A lowercase #+title: keyword is still recognized."""
+ org = "#+title: Health Flashcards\n"
+ assert drill.default_deck_name(Path("/x/health-drill.org"), org) == "Health Flashcards"
+
+
+def test_default_deck_name_falls_back_to_basename_without_title(drill):
+ """No #+TITLE line falls back to the input basename, case preserved."""
+ org = "* Section\n** Q? :drill:\na\n"
+ assert drill.default_deck_name(Path("/x/deepsat.org"), org) == "deepsat"
+
+
+def test_default_deck_name_blank_title_falls_back_to_basename(drill):
+ """An empty #+TITLE value is ignored in favour of the basename."""
+ assert drill.default_deck_name(Path("/x/health-drill.org"), "#+TITLE: \n") == "health-drill"
# --- section_to_tag (pure) ---
@@ -139,17 +158,18 @@ Geostationary Earth Orbit.
def test_parse_returns_front_back_tag_per_card(drill):
cards = drill.parse(SECTIONED)
assert len(cards) == 2
- assert cards[0] == ("What is LEO?", "Low Earth Orbit.", "orbital-regimes")
+ # The section becomes the sole Anki tag (as a one-element list).
+ assert cards[0] == ("What is LEO?", "Low Earth Orbit.", ["orbital-regimes"])
assert cards[1][0] == "What is GEO?"
def test_parse_card_without_a_section_gets_the_drill_tag(drill):
- assert drill.parse("** Lone card? :drill:\nbody\n") == [("Lone card?", "body", "drill")]
+ assert drill.parse("** Lone card? :drill:\nbody\n") == [("Lone card?", "body", ["drill"])]
def test_parse_strips_properties_drawer_from_back(drill):
text = "** Q? :drill:\n:PROPERTIES:\n:ID: abc\n:END:\nThe answer.\n"
- assert drill.parse(text) == [("Q?", "The answer.", "drill")]
+ assert drill.parse(text) == [("Q?", "The answer.", ["drill"])]
def test_parse_trims_leading_and_trailing_blank_body_lines(drill):
@@ -159,7 +179,59 @@ def test_parse_trims_leading_and_trailing_blank_body_lines(drill):
def test_parse_card_with_only_a_drawer_has_empty_back(drill):
text = "** Q? :drill:\n:PROPERTIES:\n:ID: x\n:END:\n"
- assert drill.parse(text) == [("Q?", "", "drill")]
+ assert drill.parse(text) == [("Q?", "", ["drill"])]
+
+
+# --- multi-tag headings, --tag-filter, --guid-salt -------------------------
+
+MULTITAG = """* Fundamentals
+** What is LEO? :fundamental:drill:
+Low Earth Orbit.
+** What is GEO? :drill:
+Geostationary Earth Orbit.
+"""
+
+
+def test_parse_multitag_heading_is_a_card_when_drill_is_present(drill):
+ """A heading with a second org tag still parses when drill is among them."""
+ cards = drill.parse(MULTITAG)
+ assert len(cards) == 2
+ assert cards[0][0] == "What is LEO?"
+
+
+def test_parse_multitag_tags_ride_along_next_to_the_section_tag(drill):
+ """Non-drill org tags become Anki tags alongside the section tag."""
+ cards = drill.parse(MULTITAG)
+ assert cards[0][2] == ["fundamentals", "fundamental"] # section slug + org tag
+ assert cards[1][2] == ["fundamentals"] # drill-only -> section only
+
+
+def test_parse_heading_without_drill_tag_is_not_a_card(drill):
+ """A tagged heading missing :drill: is not a card (e.g. :note:)."""
+ assert drill.parse("* S\n** Just a note :note:\nbody\n") == []
+
+
+def test_parse_tag_filter_returns_only_cards_with_that_org_tag(drill):
+ """--tag-filter narrows to cards carrying the given org tag."""
+ cards = drill.parse(MULTITAG, tag_filter="fundamental")
+ assert len(cards) == 1
+ assert cards[0][0] == "What is LEO?"
+
+
+def test_parse_body_bounded_by_any_l1_or_l2_heading(drill):
+ """A card body stops at the next L1/L2 heading, multi-tagged or not."""
+ text = "** Q1? :a:drill:\nbody1\n** Q2? :drill:\nbody2\n"
+ cards = drill.parse(text)
+ assert cards[0][1] == "body1"
+ assert cards[1][1] == "body2"
+
+
+def test_card_guid_salt_changes_the_guid(drill, monkeypatch):
+ """--guid-salt gives a subset deck its own GUID space; no salt is unchanged."""
+ monkeypatch.setattr(drill.genanki, "guid_for", lambda *a: ":".join(a), raising=False)
+ assert drill.card_guid("front", None) == "front"
+ assert drill.card_guid("front", "fundamentals") == "fundamentals:front"
+ assert drill.card_guid("front", None) != drill.card_guid("front", "fundamentals")
def test_parse_joins_multiline_body_with_br(drill):
diff --git a/.ai/scripts/tests/test_inbox_send.py b/.ai/scripts/tests/test_inbox_send.py
index a0094dc..9b0a8c6 100644
--- a/.ai/scripts/tests/test_inbox_send.py
+++ b/.ai/scripts/tests/test_inbox_send.py
@@ -97,6 +97,52 @@ class TestInboxSendDiscovery:
result = run_script(["--list"], roots=[tmp_path / "does-not-exist"])
assert result.returncode == 0
+ def test_inbox_send_list_displays_dot_stripped_name(self, project_root, run_script, tmp_path):
+ """Dotted project basenames display dot-stripped (.emacs.d → emacsd)."""
+ project_root(".emacs.d")
+ result = run_script(["--list"], roots=[tmp_path / "projects"])
+ assert "emacsd" in result.stdout
+
+
+class TestInboxSendDotAlias:
+ """A dotted project basename resolves both verbatim and dot-stripped."""
+
+ def test_resolves_by_dot_stripped_alias(self, project_root, run_script, tmp_path):
+ """'emacsd' delivers to the .emacs.d project."""
+ project_root(".emacs.d")
+ cwd = project_root("source")
+ run_script(
+ ["emacsd", "--text", "hi"],
+ cwd=cwd, roots=[tmp_path / "projects"],
+ )
+ files = list((tmp_path / "projects" / ".emacs.d" / "inbox").iterdir())
+ assert len(files) == 1
+
+ def test_resolves_by_exact_dotted_name_still(self, project_root, run_script, tmp_path):
+ """Backward-compat: the verbatim '.emacs.d' target still resolves."""
+ project_root(".emacs.d")
+ cwd = project_root("source")
+ run_script(
+ [".emacs.d", "--text", "hi"],
+ cwd=cwd, roots=[tmp_path / "projects"],
+ )
+ files = list((tmp_path / "projects" / ".emacs.d" / "inbox").iterdir())
+ assert len(files) == 1
+
+ def test_exact_match_wins_over_alias(self, project_root, run_script, tmp_path):
+ """An exact basename match is preferred over a dot-stripped collision."""
+ project_root("emacsd") # exact
+ project_root(".emacs.d") # would also normalize to 'emacsd'
+ cwd = project_root("source")
+ run_script(
+ ["emacsd", "--text", "hi"],
+ cwd=cwd, roots=[tmp_path / "projects"],
+ )
+ exact = list((tmp_path / "projects" / "emacsd" / "inbox").iterdir())
+ dotted = list((tmp_path / "projects" / ".emacs.d" / "inbox").iterdir())
+ assert len(exact) == 1
+ assert dotted == []
+
# ----------------------------------------------------------------------
# Slug derivation from text and from filenames
@@ -355,3 +401,192 @@ class TestInboxSendErrors:
assert result.returncode != 0
files = list((tmp_path / "projects" / "target" / "inbox").iterdir())
assert files == []
+
+
+# ----------------------------------------------------------------------
+# Filename collisions (two sends deriving the same name must not overwrite)
+# ----------------------------------------------------------------------
+
+def _load_module():
+ import importlib.util
+ spec = importlib.util.spec_from_file_location("inbox_send", SCRIPT)
+ mod = importlib.util.module_from_spec(spec)
+ spec.loader.exec_module(mod)
+ return mod
+
+
+class TestFilenameCollisions:
+ """Two sends in the same minute with the same leading phrase derived
+ identical filenames and the second silently overwrote the first
+ (a message was lost this way, 2026-07-02)."""
+
+ def test_send_text_same_minute_same_phrase_keeps_both(self, tmp_path):
+ from datetime import datetime
+ mod = _load_module()
+ inbox = tmp_path / "inbox"
+ inbox.mkdir()
+ now = datetime(2026, 7, 2, 5, 42, 0)
+ prefix = "identical leading phrase long enough to fill the whole slug budget entirely"
+ first = mod.send_text(inbox, prefix + " tail one", "archsetup", None, now)
+ second = mod.send_text(inbox, prefix + " tail two", "archsetup", None, now)
+ assert first != second
+ assert first.exists() and second.exists()
+ assert first.name != second.name
+ assert "tail one" in first.read_text()
+ assert "tail two" in second.read_text()
+
+ def test_send_text_collision_suffix_increments(self, tmp_path):
+ from datetime import datetime
+ mod = _load_module()
+ inbox = tmp_path / "inbox"
+ inbox.mkdir()
+ now = datetime(2026, 7, 2, 5, 42, 0)
+ paths = [mod.send_text(inbox, "same lead phrase differs later A", "src", "fixed-slug", now)
+ for _ in range(3)]
+ names = [p.name for p in paths]
+ assert names[0].endswith("fixed-slug.org")
+ assert names[1].endswith("fixed-slug-2.org")
+ assert names[2].endswith("fixed-slug-3.org")
+
+ def test_send_file_collision_preserves_extension(self, tmp_path):
+ from datetime import datetime
+ mod = _load_module()
+ inbox = tmp_path / "inbox"
+ inbox.mkdir()
+ src = tmp_path / "note.org"
+ src.write_text("body one")
+ now = datetime(2026, 7, 2, 5, 42, 0)
+ first = mod.send_file(inbox, src, "src", None, now)
+ src.write_text("body two")
+ second = mod.send_file(inbox, src, "src", None, now)
+ assert second.name.endswith("note-2.org")
+ assert first.read_text() == "body one"
+ assert second.read_text() == "body two"
+
+ def test_cli_two_rapid_sends_lose_nothing(self, project_root, run_script, tmp_path):
+ project_root("sender")
+ target = project_root("receiver")
+ roots = [tmp_path / "projects"]
+ prefix = "identical leading phrase long enough to fill the whole slug budget entirely"
+ run_script(["receiver", "--text", prefix + " message one"],
+ cwd=tmp_path / "projects" / "sender", roots=roots)
+ run_script(["receiver", "--text", prefix + " message two"],
+ cwd=tmp_path / "projects" / "sender", roots=roots)
+ files = list((target / "inbox").iterdir())
+ assert len(files) == 2
+ bodies = "".join(f.read_text() for f in files)
+ assert "message one" in bodies and "message two" in bodies
+
+
+class TestAtomicWrite:
+ """A send wrote straight to the destination path in another project's
+ inbox/, and write_text truncates on open, so any mid-write failure left a
+ zero-byte .org there. inbox-status counts that phantom as a pending
+ handoff, blocking a turn in the receiving project over a file with no
+ content (2026-07-23). The write must be atomic: the inbox sees a complete
+ file or nothing."""
+
+ def test_send_text_writes_utf8(self, tmp_path):
+ from datetime import datetime
+ mod = _load_module()
+ inbox = tmp_path / "inbox"
+ inbox.mkdir()
+ now = datetime(2026, 7, 23, 4, 36, 0)
+ # An em dash and an accented char — both non-ASCII.
+ dest = mod.send_text(inbox, "accent café and dash — here", "src", None, now)
+ # Reading as utf-8 must round-trip; a locale-encoded write would raise
+ # under a C locale, and reading back proves the bytes are utf-8.
+ assert "—" in dest.read_text(encoding="utf-8")
+
+ def test_send_text_no_partial_on_write_failure(self, tmp_path, monkeypatch):
+ from datetime import datetime
+ mod = _load_module()
+ inbox = tmp_path / "inbox"
+ inbox.mkdir()
+ now = datetime(2026, 7, 23, 4, 36, 0)
+ # Force the atomic finalize to fail after the temp file is written.
+ def boom(*a, **k):
+ raise OSError("disk full")
+ monkeypatch.setattr(mod.os, "replace", boom)
+ with pytest.raises(OSError):
+ mod.send_text(inbox, "a message that should never half-land", "src", None, now)
+ # No phantom, no leftover temp: the inbox is empty.
+ assert list(inbox.iterdir()) == []
+
+ def test_send_text_leaves_no_temp_on_success(self, tmp_path):
+ from datetime import datetime
+ mod = _load_module()
+ inbox = tmp_path / "inbox"
+ inbox.mkdir()
+ now = datetime(2026, 7, 23, 4, 36, 0)
+ dest = mod.send_text(inbox, "clean send", "src", None, now)
+ assert list(inbox.iterdir()) == [dest]
+
+ def test_send_file_no_partial_on_write_failure(self, tmp_path, monkeypatch):
+ from datetime import datetime
+ mod = _load_module()
+ inbox = tmp_path / "inbox"
+ inbox.mkdir()
+ src = tmp_path / "note.org"
+ src.write_text("body")
+ now = datetime(2026, 7, 23, 4, 36, 0)
+ def boom(*a, **k):
+ raise OSError("disk full")
+ monkeypatch.setattr(mod.os, "replace", boom)
+ with pytest.raises(OSError):
+ mod.send_file(inbox, src, "src", None, now)
+ assert list(inbox.iterdir()) == []
+
+ def test_send_file_leaves_no_temp_on_success(self, tmp_path):
+ from datetime import datetime
+ mod = _load_module()
+ inbox = tmp_path / "inbox"
+ inbox.mkdir()
+ src = tmp_path / "note.org"
+ src.write_text("payload")
+ now = datetime(2026, 7, 23, 4, 36, 0)
+ dest = mod.send_file(inbox, src, "src", None, now)
+ assert list(inbox.iterdir()) == [dest]
+ assert dest.read_text() == "payload"
+
+
+class TestSmallerDefects:
+ """Two low-severity defects found reading inbox-send during the 2026-07-23
+ sweep: an unreadable source raised an uncaught traceback instead of the
+ clean error every other failure path produces, and a roots config naming
+ both a parent and one of its children listed the same project twice."""
+
+ def test_unreadable_source_gives_clean_error_not_traceback(
+ self, project_root, run_script, tmp_path
+ ):
+ project_root("sender")
+ project_root("receiver")
+ roots = [tmp_path / "projects"]
+ src = tmp_path / "secret.bin"
+ src.write_text("x")
+ src.chmod(0o000)
+ try:
+ result = run_script(
+ ["receiver", "--file", str(src)],
+ cwd=tmp_path / "projects" / "sender",
+ roots=roots,
+ expect_failure=True,
+ )
+ finally:
+ src.chmod(0o644)
+ assert result.returncode == 1
+ # The clean "inbox-send: <message>" shape, not a Python traceback.
+ assert result.stderr.startswith("inbox-send:")
+ assert "Traceback" not in result.stderr
+
+ def test_discover_projects_dedupes_parent_and_child_root(self, tmp_path):
+ mod = _load_module()
+ # A project directory, reachable both as a child of its parent root and
+ # as a root in its own right.
+ parent = tmp_path / "projects"
+ proj = parent / "app"
+ (proj / ".ai").mkdir(parents=True)
+ (proj / "inbox").mkdir()
+ found = mod.discover_projects([parent, proj])
+ resolved = [p.resolve() for p in found]
+ assert resolved.count(proj.resolve()) == 1
diff --git a/.ai/scripts/tests/test_route_recommend.py b/.ai/scripts/tests/test_route_recommend.py
new file mode 100644
index 0000000..2ec900a
--- /dev/null
+++ b/.ai/scripts/tests/test_route_recommend.py
@@ -0,0 +1,152 @@
+"""Tests for route_recommend.py — the wrap-up routing recommendation engine.
+
+The core is a pure function recommend(item, projects) -> (destination, confidence):
+- strong: a project's name (or its dot-stripped form) appears literally in the item
+- weak: a distinctive name token overlaps, but the full name doesn't
+- none: no overlap; the item stays put (destination is None)
+
+A multi-way tie at the top tier downgrades to weak with a deterministic pick.
+An empty project list yields none.
+
+The CLI wires this to inbox-send.py's discover_projects (sandboxed here via the
+INBOX_SEND_ROOTS env var, the same hook inbox-send's own tests use).
+"""
+
+import subprocess
+import sys
+from pathlib import Path
+
+SCRIPTS = Path(__file__).parent.parent
+SCRIPT = SCRIPTS / "route_recommend.py"
+sys.path.insert(0, str(SCRIPTS))
+
+import route_recommend as rr # noqa: E402
+
+
+# --- pure function: the five spec'd cases -----------------------------------
+
+def test_strong_match_named_literally():
+ dest, conf = rr.recommend("fix the rulesets refactor command", ["rulesets", "home", "work"])
+ assert (dest, conf) == ("rulesets", "strong")
+
+
+def test_strong_match_via_dot_stripped_name():
+ # ".emacs.d" addressed as "emacsd" in the item is still a literal hit.
+ dest, conf = rr.recommend("update the emacsd ai-term module", [".emacs.d", "rulesets"])
+ assert (dest, conf) == (".emacs.d", "strong")
+
+
+def test_strong_match_dotted_name_verbatim():
+ dest, conf = rr.recommend("patch .emacs.d startup", [".emacs.d", "rulesets"])
+ assert (dest, conf) == (".emacs.d", "strong")
+
+
+def test_weak_match_topic_token_only():
+ # "wttrin" is a token of "emacs-wttrin" but the full name isn't present.
+ dest, conf = rr.recommend("the wttrin weather bug", ["emacs-wttrin", "rulesets"])
+ assert (dest, conf) == ("emacs-wttrin", "weak")
+
+
+def test_no_match_stays_put():
+ dest, conf = rr.recommend("calibrate the telescope mount", ["rulesets", "deepsat"])
+ assert dest is None
+ assert conf == "none"
+
+
+def test_two_project_strong_tie_downgrades_to_weak():
+ # Both named literally → ambiguous → weak, deterministic tie-break (alphabetical).
+ dest, conf = rr.recommend("sync rulesets and home configs", ["rulesets", "home", "work"])
+ assert conf == "weak"
+ assert dest == "home" # tie-break: most-overlap then alphabetical
+
+
+def test_empty_project_list_is_none():
+ assert rr.recommend("anything at all", []) == (None, "none")
+
+
+# --- boundary / robustness --------------------------------------------------
+
+def test_literal_name_requires_word_boundary():
+ # "home" must not match inside "homeowner".
+ dest, conf = rr.recommend("the homeowner association meeting", ["home", "rulesets"])
+ assert dest is None and conf == "none"
+
+
+def test_path_mention_counts_as_literal():
+ dest, conf = rr.recommend("edit ~/code/rulesets/Makefile", ["rulesets", "home"])
+ assert (dest, conf) == ("rulesets", "strong")
+
+
+def test_strong_beats_weak_when_both_present():
+ # "rulesets" named literally (strong) outranks an emacs-wttrin token hit (weak).
+ dest, conf = rr.recommend("the wttrin fix belongs in rulesets", ["rulesets", "emacs-wttrin"])
+ assert (dest, conf) == ("rulesets", "strong")
+
+
+# --- CLI + discovery reuse (sandboxed roots) --------------------------------
+
+def _run(args, roots, item):
+ import os
+ env = {"PATH": os.environ.get("PATH", ""), "HOME": os.environ.get("HOME", "/tmp"),
+ "INBOX_SEND_ROOTS": ":".join(str(r) for r in roots)}
+ return subprocess.run([sys.executable, str(SCRIPT), "--item", item, *args],
+ capture_output=True, text=True, env=env)
+
+
+def _mk_project(tmp_path, name):
+ proj = tmp_path / "projects" / name
+ (proj / ".ai").mkdir(parents=True, exist_ok=True)
+ (proj / "inbox").mkdir(exist_ok=True)
+ return proj
+
+
+def test_cli_discovers_and_recommends(tmp_path):
+ _mk_project(tmp_path, "foo")
+ _mk_project(tmp_path, "bar")
+ r = _run([], roots=[tmp_path / "projects"], item="fix the foo widget")
+ assert r.returncode == 0
+ assert r.stdout.strip() == "foo\tstrong"
+
+
+def test_cli_no_match_prints_none(tmp_path):
+ _mk_project(tmp_path, "foo")
+ r = _run([], roots=[tmp_path / "projects"], item="unrelated grocery list")
+ assert r.returncode == 0
+ assert r.stdout.strip() == "none"
+
+
+def test_cli_exclude_drops_current_project(tmp_path):
+ _mk_project(tmp_path, "foo")
+ _mk_project(tmp_path, "bar")
+ # Item names foo, but foo is excluded as the current project → no other match.
+ r = _run(["--exclude", "foo"], roots=[tmp_path / "projects"], item="fix the foo widget")
+ assert r.returncode == 0
+ assert r.stdout.strip() == "none"
+
+
+# ----------------------------------------------------------------------
+# Duplicate candidate names
+#
+# Projects are collapsed to bare basenames, so two projects sharing a basename
+# across roots (~/code/notes and ~/projects/notes) appear twice in the candidate
+# list. Both literal-match, recommend read len(strong) > 1 as an ambiguous tie,
+# and a correct strong match was downgraded to weak. Latent when discovered
+# 2026-07-24 (27 projects, 27 distinct basenames) but real.
+# ----------------------------------------------------------------------
+
+def test_duplicate_candidate_name_keeps_strong_confidence():
+ assert rr.recommend("fix the notes thing", ["notes", "other"]) == ("notes", "strong")
+ # The same name twice must not read as a tie.
+ assert rr.recommend("fix the notes thing", ["notes", "notes", "other"]) == ("notes", "strong")
+
+
+def test_genuine_ambiguity_still_downgrades():
+ # Two DIFFERENT projects both matching is a real tie and stays weak — the
+ # dedupe must collapse identical names only, never real ambiguity.
+ dest, conf = rr.recommend("notes and other both", ["notes", "other"])
+ assert conf == "weak"
+
+
+def test_duplicates_do_not_change_the_chosen_destination():
+ dest, _ = rr.recommend("fix the notes thing", ["notes", "notes"])
+ assert dest == "notes"
diff --git a/.ai/scripts/tests/test_upcoming_birthdays.py b/.ai/scripts/tests/test_upcoming_birthdays.py
new file mode 100644
index 0000000..1e15183
--- /dev/null
+++ b/.ai/scripts/tests/test_upcoming_birthdays.py
@@ -0,0 +1,168 @@
+"""Tests for upcoming_birthdays.py — the daily-prep upcoming-birthdays block.
+
+Pure core:
+ parse_birthdays(text) -> [Birthday(name, month, day, year|None), ...]
+ upcoming(birthdays, today, window=30) -> [Upcoming(name, date, days_away, age|None), ...]
+ format_block(items, window, callout_days=7) -> str
+
+Birth year 1900 is the placeholder org-contacts uses when the real year is
+unknown; those entries carry year=None and render date-only (no age).
+"""
+
+import datetime as dt
+import subprocess
+import sys
+from pathlib import Path
+
+SCRIPTS = Path(__file__).parent.parent
+SCRIPT = SCRIPTS / "upcoming_birthdays.py"
+sys.path.insert(0, str(SCRIPTS))
+
+import upcoming_birthdays as ub # noqa: E402
+
+
+# --- parse_birthdays --------------------------------------------------------
+
+def test_parse_reads_name_month_day_year():
+ text = "** Jane Doe\n:PROPERTIES:\n:BIRTHDAY: 1970-08-05\n:END:\n"
+ bdays = ub.parse_birthdays(text)
+ assert bdays == [ub.Birthday("Jane Doe", 8, 5, 1970)]
+
+
+def test_parse_placeholder_year_1900_becomes_none():
+ text = "** John Smith\n:PROPERTIES:\n:BIRTHDAY: 1900-07-14\n:END:\n"
+ bdays = ub.parse_birthdays(text)
+ assert bdays == [ub.Birthday("John Smith", 7, 14, None)]
+
+
+def test_parse_skips_contacts_without_birthday():
+ text = (
+ "** No Birthday\n:PROPERTIES:\n:PHONE: 555\n:END:\n"
+ "** Has Birthday\n:PROPERTIES:\n:BIRTHDAY: 1990-03-02\n:END:\n"
+ )
+ bdays = ub.parse_birthdays(text)
+ assert [b.name for b in bdays] == ["Has Birthday"]
+
+
+def test_parse_strips_heading_stars_and_tags():
+ text = "*** Bob Jones :friend:\n:PROPERTIES:\n:BIRTHDAY: 1980-01-01\n:END:\n"
+ bdays = ub.parse_birthdays(text)
+ assert bdays[0].name == "Bob Jones"
+
+
+def test_parse_ignores_malformed_birthday_lines():
+ text = "** Bad Date\n:PROPERTIES:\n:BIRTHDAY: not-a-date\n:END:\n"
+ assert ub.parse_birthdays(text) == []
+
+
+# --- upcoming ---------------------------------------------------------------
+
+TODAY = dt.date(2026, 7, 18)
+
+
+def test_upcoming_birthday_today_is_zero_days():
+ bdays = [ub.Birthday("Today Person", 7, 18, 1990)]
+ got = ub.upcoming(bdays, TODAY)
+ assert got[0].days_away == 0
+ assert got[0].date == dt.date(2026, 7, 18)
+
+
+def test_upcoming_includes_within_window():
+ bdays = [ub.Birthday("Soon", 7, 23, 1990)]
+ got = ub.upcoming(bdays, TODAY, window=30)
+ assert got[0].days_away == 5
+
+
+def test_upcoming_excludes_beyond_window():
+ bdays = [ub.Birthday("Far", 9, 1, 1990)] # 45 days out
+ assert ub.upcoming(bdays, TODAY, window=30) == []
+
+
+def test_upcoming_boundary_day_30_included_day_31_excluded():
+ on = [ub.Birthday("On", 8, 17, 1990)] # exactly 30 days
+ off = [ub.Birthday("Off", 8, 18, 1990)] # 31 days
+ assert ub.upcoming(on, TODAY, window=30)[0].days_away == 30
+ assert ub.upcoming(off, TODAY, window=30) == []
+
+
+def test_upcoming_uses_next_year_when_this_years_passed():
+ # today is 2026-07-18; a Jan 5 birthday recurs on 2027-01-05
+ today = dt.date(2026, 12, 27)
+ bdays = [ub.Birthday("New Year", 1, 5, 1990)]
+ got = ub.upcoming(bdays, today, window=30)
+ assert got[0].date == dt.date(2027, 1, 5)
+ assert got[0].days_away == 9
+
+
+def test_upcoming_age_is_occurrence_year_minus_birth_year():
+ bdays = [ub.Birthday("Ager", 7, 23, 1970)]
+ got = ub.upcoming(bdays, TODAY)
+ assert got[0].age == 56 # 2026 - 1970
+
+
+def test_upcoming_age_none_for_placeholder():
+ bdays = [ub.Birthday("Placeholder", 7, 23, None)]
+ got = ub.upcoming(bdays, TODAY)
+ assert got[0].age is None
+
+
+def test_upcoming_sorted_by_days_away():
+ bdays = [
+ ub.Birthday("Later", 8, 10, 1990),
+ ub.Birthday("Sooner", 7, 20, 1990),
+ ]
+ got = ub.upcoming(bdays, TODAY)
+ assert [u.name for u in got] == ["Sooner", "Later"]
+
+
+def test_upcoming_leap_day_maps_to_feb_28_in_non_leap_year():
+ today = dt.date(2027, 2, 1) # 2027 is not a leap year
+ bdays = [ub.Birthday("Leapling", 2, 29, 2000)]
+ got = ub.upcoming(bdays, today, window=30)
+ assert got[0].date == dt.date(2027, 2, 28)
+
+
+# --- format_block -----------------------------------------------------------
+
+def test_format_block_empty_reports_none():
+ out = ub.format_block([], window=30)
+ assert "No birthdays" in out
+
+
+def test_format_block_callout_marks_within_seven_days():
+ items = [ub.Upcoming("Soon", dt.date(2026, 7, 22), 4, 40)]
+ out = ub.format_block(items, window=30, callout_days=7)
+ assert "⚠" in out
+ assert "Soon" in out
+ assert "40" in out # age shown
+
+
+def test_format_block_beyond_callout_is_not_flagged():
+ items = [ub.Upcoming("Later", dt.date(2026, 8, 10), 23, 30)]
+ out = ub.format_block(items, window=30, callout_days=7)
+ assert "⚠" not in out
+
+
+def test_format_block_placeholder_shows_date_only_no_age():
+ items = [ub.Upcoming("NoYear", dt.date(2026, 7, 25), 7, None)]
+ out = ub.format_block(items, window=30)
+ assert "NoYear" in out
+ assert "turns" not in out
+
+
+# --- CLI --------------------------------------------------------------------
+
+def test_cli_runs_against_a_fixture_file(tmp_path):
+ contacts = tmp_path / "contacts.org"
+ contacts.write_text(
+ "** Alice\n:PROPERTIES:\n:BIRTHDAY: 1990-07-20\n:END:\n"
+ "** Bob\n:PROPERTIES:\n:BIRTHDAY: 1900-12-01\n:END:\n"
+ )
+ res = subprocess.run(
+ [sys.executable, str(SCRIPT), "--file", str(contacts),
+ "--today", "2026-07-18", "--window", "30"],
+ capture_output=True, text=True,
+ )
+ assert res.returncode == 0
+ assert "Alice" in res.stdout
+ assert "Bob" not in res.stdout # Dec 1 is outside the 30-day window
diff --git a/.ai/scripts/todo-cleanup.el b/.ai/scripts/todo-cleanup.el
index 6b3081a..516e9b1 100644
--- a/.ai/scripts/todo-cleanup.el
+++ b/.ai/scripts/todo-cleanup.el
@@ -5,10 +5,14 @@
;; emacs --batch -q -l todo-cleanup.el --check todo.org # hygiene report only
;; emacs --batch -q -l todo-cleanup.el --archive-done todo.org # archive completed subtrees
;; emacs --batch -q -l todo-cleanup.el --archive-done --check todo.org # preview the archive
+;; emacs --batch -q -l todo-cleanup.el --seal todo.org # seal the working archive to resolved-YYYY-MM-DD.org
+;; emacs --batch -q -l todo-cleanup.el --seal --check todo.org # preview the seal
+;; emacs --batch -q -l todo-cleanup.el --convert-subtasks todo.org # dated-rewrite done level-3+ sub-tasks
+;; emacs --batch -q -l todo-cleanup.el --convert-subtasks --check todo.org # preview the conversion
;; emacs --batch -q -l todo-cleanup.el --sync-child-priority todo.org # bump children whose priority drifted below the parent's
;; emacs --batch -q -l todo-cleanup.el --check-child-priority todo.org # preview the sync (same as --sync-child-priority --check)
;;
-;; Three independent modes:
+;; Four independent modes:
;;
;; * Default (hygiene). Designed for the wrap-it-up workflow: cheap, idempotent,
;; safe to run every session.
@@ -25,14 +29,60 @@
;; line isn't in canonical position. Reports these for manual fix; doesn't
;; auto-rewrite (preserving real state-log history is judgement work).
;;
-;; * --archive-done (opt-in). Moves every level-2 subtree whose TODO state is
-;; DONE or CANCELLED out of the "Open Work" section and into the "Resolved"
-;; section of the same file, subtree intact. The sections are matched by a
-;; unique level-1 heading containing "Open Work" (case-insensitive) and one
-;; containing "Resolved"; if either is missing or ambiguous, the file is
-;; skipped with a message. Only direct level-2 children move — a DONE entry
-;; nested under an open parent stays put. Archiving is consequential, so it's
-;; never run by default; it does *not* also run the hygiene passes.
+;; * --archive-done (opt-in). Two steps, in order:
+;;
+;; 1. Moves every level-2 subtree whose TODO state is DONE or CANCELLED out of
+;; the "Open Work" section and into the "Resolved" section of the same
+;; file, subtree intact. The sections are matched by a unique level-1
+;; heading containing "Open Work" (case-insensitive) and one containing
+;; "Resolved"; if either is missing or ambiguous, the file is skipped with
+;; a message. Only direct level-2 children move — a DONE entry nested under
+;; an open parent stays put.
+;;
+;; 2. Ages the "Resolved" section: a level-2 DONE/CANCELLED subtree is moved
+;; out to `tc-archive-file' (default `archive/task-archive.org' beside the
+;; todo file) when its CLOSED date is older than `tc-archive-retain-days'
+;; (default 31 — one month) OR its CLOSED date can't be parsed. The last
+;; month of closed tasks stays browsable in the file itself; older ones age
+;; out. The unparseable-CLOSED case archives too, deliberately: a
+;; keyword-complete task with no readable close date is cruft, not live
+;; work. Set `tc-archive-retain-days' to nil to disable this step (legacy
+;; in-file-only behavior). The aging date is `tc-archive-reference-date'
+;; when set (tests), otherwise the real current date. The archive inherits
+;; the todo file's gitignore status: when the todo file is gitignored, the
+;; archive path is added to .gitignore before the first write, so private
+;; task history never lands in a tracked path (see
+;; `tc--ensure-archive-gitignored').
+;;
+;; Archiving is consequential, so it's never run by default; it does *not*
+;; also run the hygiene passes.
+;;
+;; * --seal (opt-in). Renames the working archive file (`tc-archive-file',
+;; default `archive/task-archive.org') to `resolved-YYYY-MM-DD.org' beside it,
+;; dated by the seal run, and leaves the next `--archive-done' to recreate a
+;; fresh working file. The dated file means "everything sealed as of that
+;; date" — not a calendar quarter — so a task closed late in a quarter and
+;; archived after the boundary is never mislabeled; cadence (e.g. quarterly)
+;; becomes independent of correctness and any slip is harmless. The sealed
+;; file inherits the todo file's gitignore status the same way the working
+;; archive does. A no-op (reported) when there's no working archive to seal;
+;; refuses to clobber an existing `resolved-<today>.org'. Honors `--check'.
+;; The seal date is `tc-archive-reference-date' when set (tests), otherwise the
+;; real current date.
+;;
+;; * --convert-subtasks (opt-in). Rewrites every level-3-and-deeper heading whose
+;; TODO state is DONE/CANCELLED/FAILED into a dated event-log entry
+;; (`<stars> YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <text>'), dropping the keyword,
+;; priority cookie, and tags, and removing the now-redundant CLOSED line. The
+;; date and time come from that entry's own CLOSED cookie; a date-only close
+;; yields 00:00:00, and the UTC offset is computed DST-aware for that date.
+;; This enforces the todo-format depth rule that interactive closes
+;; (`org-log-done' → DONE + CLOSED) and `--archive-done' (level-2 only) leave
+;; unapplied. The heading text is preserved verbatim — a batch tool can't
+;; past-tense an imperative title reliably. Idempotent (an already-dated
+;; heading has no done keyword); a done sub-task with no parseable CLOSED date
+;; is flagged and left alone, never stamped with a fabricated date. Like
+;; --archive-done it does not also run the hygiene passes.
;;
;; * --sync-child-priority (opt-in). Walks every heading with a priority cookie
;; ([#A]-[#D]) and, for each of its direct child headings whose own priority
@@ -50,18 +100,41 @@
;; --check-child-priority is the report-only alias for --sync-child-priority
;; --check.
+;; Before any modification a backup is copied to
+;; /tmp/<basename>.before-todo-cleanup.<YYYYMMDD-HHMMSS>
+;; matching lint-org.el and wrap-org-table.el. Skipped under --check, which
+;; writes nothing.
+;;
+
(require 'org)
(require 'cl-lib)
+(require 'calendar)
(setq org-todo-keywords
- '((sequence "TODO" "DOING" "WAITING" "NEXT" "|" "DONE" "CANCELLED")))
+ '((sequence "TODO" "DOING" "WAITING" "NEXT" "|" "DONE" "CANCELLED" "FAILED")))
(defconst tc-done-states '("DONE" "CANCELLED")
"TODO keywords that mark an entry as completed for `--archive-done'.")
+(defconst tc--convert-done-states '("DONE" "CANCELLED" "FAILED")
+ "TODO keywords whose level-3-and-deeper entries `--convert-subtasks' rewrites
+to dated event-log entries. Broader than `tc-done-states' because a FAILED
+sub-task is terminal too and belongs in the parent's dated history.")
+
(defconst tc--priority-cookie-regexp "\\[#\\([A-Z]\\)\\]"
"Regexp matching an org priority cookie. Match group 1 is the letter.")
+(defconst tc--planning-cookie-regexp
+ "\\(?:CLOSED\\|DEADLINE\\|SCHEDULED\\):[ \t]*[[<][^]>\n]*[]>]"
+ "One org planning cookie: a CLOSED/DEADLINE/SCHEDULED keyword followed by a
+bracketed (inactive) or angled (active) timestamp.")
+
+(defconst tc--planning-line-regexp
+ (concat "\\`[ \t]*\\(?:" tc--planning-cookie-regexp "[ \t]*\\)+\\'")
+ "A whole org planning line: nothing but planning cookies and whitespace.
+Anchored to a single line's contents so a line mixing a cookie with real body
+text is never matched.")
+
(defconst tc-no-sync-tag "no-sync"
"Org tag that opts a heading and all its descendants out of
`--sync-child-priority'. Inherits down: a tag on an ancestor counts for
@@ -70,11 +143,40 @@ every heading below it.")
(defvar tc-fixes 0)
(defvar tc-archived 0)
(defvar tc-bumped 0)
+(defvar tc-converted 0)
(defvar tc-issues nil)
+(defvar tc-sealed 0)
(defvar tc-check-only nil)
(defvar tc-archive-done nil)
(defvar tc-sync-child-priority nil)
+(defvar tc-convert-subtasks nil)
+(defvar tc-seal nil)
(defvar tc-current-file nil)
+(defvar tc-current-dir nil)
+(defvar tc-archived-to-file 0)
+
+(defconst tc-archive-retain-days-default 31
+ "Default retention window (days) for the `--archive-done' file-aging step —
+one month. A closed Resolved subtree stays in-file for this long before it ages
+out to `tc-archive-file'; the last month of resolved work stays browsable in the
+todo file itself. Named so the \"one month\" contract is explicit and testable.")
+
+(defvar tc-archive-retain-days tc-archive-retain-days-default
+ "Retention window for the `--archive-done' file-aging step. A closed Resolved
+subtree whose CLOSED date is within this many days of the reference date stays
+in the in-file Resolved section; an older one is moved out to `tc-archive-file'.
+A subtree with no parseable CLOSED date is aged out too (a keyword-complete task
+with no readable close date is cruft, not live work). nil disables the aging
+step entirely, leaving the legacy in-file-only behavior. Defaults to
+`tc-archive-retain-days-default' (one month).")
+
+(defvar tc-archive-reference-date nil
+ "(YEAR MONTH DAY) treated as \"today\" when aging Resolved subtrees out to a
+file; nil means the real current date. Set in tests for determinism.")
+
+(defvar tc-archive-file nil
+ "Destination file for aged-out Resolved subtrees; nil means
+`archive/task-archive.org' beside the todo file being processed.")
;;; ---------------------------------------------------------------------------
;;; Hygiene mode
@@ -224,7 +326,8 @@ are reported but not performed."
:line (line-number-at-pos)
:heading (org-get-heading t t t t))
tc-issues)
- (cl-incf tc-archived))))
+ (cl-incf tc-archived)))
+ (tc-archive-old-resolved-to-file))
(t
(catch 'done
(while t
@@ -252,7 +355,216 @@ are reported but not performed."
(cl-incf tc-archived)
(push (list :kind 'archive-moved :file tc-current-file
:line line :heading heading)
- tc-issues)))))))))
+ tc-issues)))))
+ (tc-archive-old-resolved-to-file)))))
+
+;;; ---------------------------------------------------------------------------
+;;; --archive-done: age old Resolved subtrees out to a file
+
+(defconst tc-archive-file-scaffold
+ "#+TITLE: Task Archive\n#+FILETAGS: :archive:\n\n* Resolved (archived)\n"
+ "Initial content written to a fresh `tc-archive-file'. Aged subtrees are
+appended as level-2 children under the level-1 heading.")
+
+(defun tc--reference-absolute ()
+ "Absolute (Gregorian serial) day number of the aging reference date —
+`tc-archive-reference-date' when set, otherwise the real current date."
+ (if tc-archive-reference-date
+ (pcase-let ((`(,y ,m ,d) tc-archive-reference-date))
+ (calendar-absolute-from-gregorian (list m d y)))
+ (pcase-let ((`(,m ,d ,y) (calendar-current-date)))
+ (calendar-absolute-from-gregorian (list m d y)))))
+
+(defun tc--closed-absolute-in-region (beg end)
+ "Absolute day number of the first CLOSED: [YYYY-MM-DD ...] line in BEG..END,
+or nil when the region carries no parseable CLOSED date. The task's own CLOSED
+line sits in canonical position directly under the heading, so the first match
+in the subtree is the task's close."
+ (save-excursion
+ (goto-char beg)
+ (when (re-search-forward
+ "CLOSED:[ \t]*\\[\\([0-9][0-9][0-9][0-9]\\)-\\([0-9][0-9]\\)-\\([0-9][0-9]\\)"
+ end t)
+ (calendar-absolute-from-gregorian
+ (list (string-to-number (match-string 2))
+ (string-to-number (match-string 3))
+ (string-to-number (match-string 1)))))))
+
+(defun tc--archive-file-path ()
+ "Resolve the destination file for aged-out subtrees: `tc-archive-file' if set,
+else `archive/task-archive.org' beside the todo file being processed."
+ (or tc-archive-file
+ (and tc-current-dir
+ (expand-file-name "archive/task-archive.org" tc-current-dir))))
+
+(defun tc--git-ignored-p (path)
+ "Non-nil when PATH is gitignored (git check-ignore exits 0). nil on any git
+error or when git is unavailable."
+ (let ((default-directory (or tc-current-dir default-directory)))
+ (eq 0 (ignore-errors
+ (call-process "git" nil nil nil "check-ignore" "-q"
+ (expand-file-name path))))))
+
+(defun tc--ensure-archive-gitignored (archive-path)
+ "Keep the aged-out archive as private as the todo file it derives from. When the
+todo file being processed is gitignored but ARCHIVE-PATH is not, append a
+root-relative ignore entry for ARCHIVE-PATH to the project's .gitignore. No-op
+when the todo file is tracked, the archive is already ignored, or there is no git
+work tree — so track-mode projects (todo file tracked) leave the archive tracked
+too. This is what makes the aging step safe to ship to gitignore-mode projects,
+where todo.org is private: the archive inherits that privacy instead of leaking
+previously-ignored task history into a tracked path."
+ (when (and tc-current-file tc-current-dir)
+ (let* ((todo (expand-file-name tc-current-file tc-current-dir))
+ (default-directory tc-current-dir)
+ (root (with-temp-buffer
+ (when (eq 0 (ignore-errors
+ (call-process "git" nil (current-buffer) nil
+ "rev-parse" "--show-toplevel")))
+ (string-trim (buffer-string))))))
+ (when (and root (> (length root) 0) (file-directory-p root)
+ (tc--git-ignored-p todo)
+ (not (tc--git-ignored-p archive-path)))
+ (let ((entry (concat "/" (file-relative-name
+ (expand-file-name archive-path) root)))
+ (gi (expand-file-name ".gitignore" root)))
+ (with-temp-buffer
+ (when (file-readable-p gi) (insert-file-contents gi))
+ (unless (save-excursion
+ (goto-char (point-min))
+ (re-search-forward (concat "^" (regexp-quote entry) "$") nil t))
+ (goto-char (point-max))
+ (unless (bolp) (insert "\n"))
+ (insert "\n# Claude Code: task archive (follows todo file privacy)\n"
+ entry "\n")
+ (write-region (point-min) (point-max) gi nil 'silent))))))))
+
+(defun tc--append-subtrees-to-archive-file (path texts)
+ "Append TEXTS (subtree strings) under the level-1 heading in PATH, creating the
+file with `tc-archive-file-scaffold' and the parent directory when absent.
+Ensures the archive inherits the todo file's gitignore status first."
+ (when (and path texts)
+ (tc--ensure-archive-gitignored path)
+ (let ((dir (file-name-directory path)))
+ (when (and dir (not (file-directory-p dir)))
+ (make-directory dir t)))
+ (with-temp-buffer
+ (when (file-readable-p path)
+ (insert-file-contents path))
+ (when (= (point-min) (point-max))
+ (insert tc-archive-file-scaffold))
+ ;; Guarantee a level-1 heading to append under (older files might lack one).
+ (goto-char (point-min))
+ (unless (re-search-forward "^\\* " nil t)
+ (goto-char (point-max))
+ (unless (bolp) (insert "\n"))
+ (insert "* Resolved (archived)\n"))
+ (goto-char (point-max))
+ (unless (bolp) (insert "\n"))
+ (dolist (text texts)
+ (insert text)
+ (unless (bolp) (insert "\n")))
+ (write-region (point-min) (point-max) path nil 'silent))))
+
+(defun tc-archive-old-resolved-to-file ()
+ "Move level-2 DONE/CANCELLED subtrees in the \"Resolved\" section whose CLOSED
+date predates the `tc-archive-retain-days' window out to `tc--archive-file-path'.
+Only subtrees closed within the window stay; older ones, and those with no
+parseable CLOSED date, are moved out. A nil `tc-archive-retain-days' disables the
+step. Honors `tc-check-only' (report only)."
+ (when tc-archive-retain-days
+ (let ((res (tc--find-section "resolved")))
+ (when (integerp res)
+ (let* ((cutoff (- (tc--reference-absolute) tc-archive-retain-days))
+ (moves nil))
+ (dolist (pos (tc--done-level-2-children res))
+ (save-excursion
+ (goto-char pos)
+ (let* ((region (tc--subtree-region))
+ (beg (car region))
+ (end (cdr region))
+ (closed (tc--closed-absolute-in-region beg end)))
+ ;; Archive anything not provably within the window: closed
+ ;; before the cutoff, or with no parseable CLOSED date at all.
+ (when (or (null closed) (< closed cutoff))
+ (push (list :beg beg :end end
+ :heading (org-get-heading t t t t)
+ :line (line-number-at-pos beg))
+ moves)))))
+ (setq moves (nreverse moves)) ; document order
+ (cond
+ ((null moves) nil)
+ (tc-check-only
+ (dolist (m moves)
+ (cl-incf tc-archived-to-file)
+ (push (list :kind 'archive-file-would :file tc-current-file
+ :line (plist-get m :line) :heading (plist-get m :heading))
+ tc-issues)))
+ (t
+ ;; Capture text before any deletion (positions are still valid), then
+ ;; delete bottom-up so earlier subtree positions stay correct.
+ (let ((texts (mapcar
+ (lambda (m)
+ (concat (string-trim-right
+ (buffer-substring-no-properties
+ (plist-get m :beg) (plist-get m :end))
+ "[ \t\n]+")
+ "\n"))
+ moves)))
+ (dolist (m (sort (copy-sequence moves)
+ (lambda (a b) (> (plist-get a :beg) (plist-get b :beg)))))
+ (delete-region (plist-get m :beg) (plist-get m :end)))
+ (tc--append-subtrees-to-archive-file (tc--archive-file-path) texts)
+ (dolist (m moves)
+ (cl-incf tc-archived-to-file)
+ (push (list :kind 'archive-file-moved :file tc-current-file
+ :line (plist-get m :line) :heading (plist-get m :heading))
+ tc-issues))))))))))
+
+;;; ---------------------------------------------------------------------------
+;;; --seal mode: rename the working archive to a dated resolved-YYYY-MM-DD.org
+
+(defun tc--seal-date-string ()
+ "YYYY-MM-DD for the seal — `tc-archive-reference-date' when set (tests),
+otherwise the real current date."
+ (if tc-archive-reference-date
+ (pcase-let ((`(,y ,m ,d) tc-archive-reference-date))
+ (format "%04d-%02d-%02d" y m d))
+ (format-time-string "%Y-%m-%d")))
+
+(defun tc-seal-archive-file ()
+ "Rename the working archive file to `resolved-YYYY-MM-DD.org' beside it.
+The next `--archive-done' run recreates a fresh working file. No-op (reported)
+when there is no working archive to seal; refuses to clobber an existing
+`resolved-<today>.org'. Ensures the sealed file inherits the todo file's
+gitignore status. Honors `tc-check-only'."
+ (let ((path (tc--archive-file-path)))
+ (cond
+ ((or (null path) (not (file-readable-p path)))
+ (push (list :kind 'seal-nothing :file tc-current-file) tc-issues))
+ (t
+ (let* ((dir (file-name-directory path))
+ (sealed (expand-file-name
+ (format "resolved-%s.org" (tc--seal-date-string)) dir)))
+ (cond
+ ((file-exists-p sealed)
+ (push (list :kind 'seal-collision :file tc-current-file
+ :detail (file-name-nondirectory sealed))
+ tc-issues))
+ (tc-check-only
+ (cl-incf tc-sealed)
+ (push (list :kind 'seal-would :file tc-current-file
+ :detail (file-name-nondirectory sealed))
+ tc-issues))
+ (t
+ ;; Ignore the sealed name before the rename so its history stays as
+ ;; private as the working archive it derives from.
+ (tc--ensure-archive-gitignored sealed)
+ (rename-file path sealed)
+ (cl-incf tc-sealed)
+ (push (list :kind 'seal-done :file tc-current-file
+ :detail (file-name-nondirectory sealed))
+ tc-issues))))))))
;;; ---------------------------------------------------------------------------
;;; --sync-child-priority mode
@@ -377,10 +689,186 @@ before their descendants — a [#A] → [#B] → [#D] chain collapses in one pas
(org-map-entries #'tc-sync-child-priority-at-heading nil 'file))
;;; ---------------------------------------------------------------------------
+;;; --convert-subtasks mode
+;;
+;; A sub-task (a heading at level 3 or deeper, i.e. under a parent task) that is
+;; marked DONE/CANCELLED/FAILED should become a dated event-log entry per the
+;; todo-format depth rule: drop the keyword, priority cookie, and tags, and
+;; rewrite the heading to `<stars> YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <text>' so the
+;; parent's subtree grows a chronological history instead of a long tail of
+;; nested DONE lines. Nothing enforced this before: `org-log-done' just flips an
+;; interactive close to DONE + CLOSED, and `--archive-done' only touches level 2.
+;; So level-3+ closes piled up as DONE keywords. This mode converts them
+;; mechanically, pulling the timestamp from each entry's own CLOSED cookie. The
+;; heading text is kept verbatim (a batch tool can't reliably past-tense an
+;; imperative title, and guessing prose in the task file is worse than leaving it
+;; as written). Idempotent: an already-dated heading has no done keyword, so it
+;; is skipped. A done sub-task with no parseable CLOSED cookie can't be dated, so
+;; it is flagged and left alone rather than stamped with a fabricated date.
+;;
+;; The planning line goes entirely. A dated-log entry carries its date in the
+;; heading, so CLOSED is redundant and an active DEADLINE/SCHEDULED is wrong: org
+;; renders any headline with an active planning timestamp — keyword or not — so a
+;; SCHEDULED left on a dated-log heading pins it to the agenda as weeks-overdue
+;; long after the work is done. The conversion deletes the whole planning line,
+;; not just the CLOSED cookie (todo-format.md; lint checker
+;; `dated-log-heading-active-timestamp' backstops any that slip through).
+
+(defun tc--closed-parts-in-entry ()
+ "Return a plist (:year :month :day :dow :hour :minute) from the CLOSED cookie
+of the entry at point, or nil when the entry has no parseable CLOSED line.
+:hour and :minute are nil when the cookie carries only a date. The CLOSED line
+sits in canonical position directly under the heading, so the first match within
+the entry is the task's own close."
+ (save-excursion
+ (org-back-to-heading t)
+ (let ((end (save-excursion
+ (or (outline-next-heading) (goto-char (point-max)))
+ (point))))
+ (when (re-search-forward
+ (concat "CLOSED:[ \t]*\\[\\([0-9]\\{4\\}\\)-\\([0-9]\\{2\\}\\)-\\([0-9]\\{2\\}\\)"
+ "[ \t]+\\([A-Za-z]+\\)"
+ "\\(?:[ \t]+\\([0-9]\\{2\\}\\):\\([0-9]\\{2\\}\\)\\)?\\]")
+ end t)
+ (list :year (match-string 1) :month (match-string 2) :day (match-string 3)
+ :dow (match-string 4)
+ :hour (match-string 5) :minute (match-string 6))))))
+
+(defun tc--tz-offset-string (year month day hour minute)
+ "Return the local UTC offset (e.g. \"-0500\") for the given wall-clock instant.
+DST-aware: `encode-time' with an unknown-DST field lets the system pick the
+correct offset for that date, so a summer close reads -0400 and a winter one
+-0500 without hardcoding either."
+ (format-time-string
+ "%z" (encode-time (list 0 minute hour day month year nil -1 nil))))
+
+(defun tc--dated-header-line (level parts title)
+ "Build the dated event-log heading string from LEVEL, CLOSED PARTS, and TITLE.
+Missing time in PARTS defaults to 00:00:00 (the close logged only a date)."
+ (let* ((year (plist-get parts :year))
+ (month (plist-get parts :month))
+ (day (plist-get parts :day))
+ (dow (plist-get parts :dow))
+ (hh (or (plist-get parts :hour) "00"))
+ (mm (or (plist-get parts :minute) "00"))
+ (tz (tc--tz-offset-string (string-to-number year)
+ (string-to-number month)
+ (string-to-number day)
+ (string-to-number hh)
+ (string-to-number mm))))
+ (format "%s %s-%s-%s %s @ %s:%s:00 %s %s"
+ (make-string level ?*) year month day dow hh mm tz title)))
+
+(defun tc--convert-collect-targets ()
+ "Markers at every heading at level >= 3 whose TODO state is a done state.
+Collected up front so the rewrite loop can edit the buffer without disturbing an
+in-progress `org-map-entries' walk; markers track their headings across edits."
+ (let (targets)
+ (org-map-entries
+ (lambda ()
+ (when (and (>= (org-current-level) 3)
+ (member (org-get-todo-state) tc--convert-done-states))
+ (push (copy-marker (point)) targets)))
+ nil 'file)
+ (nreverse targets)))
+
+(defun tc--strip-planning-lines-in-entry ()
+ "Delete the canonical planning line(s) directly under the heading at point.
+A planning line is one composed solely of CLOSED/DEADLINE/SCHEDULED cookies and
+whitespace. Walks the lines immediately after the heading and stops at the first
+non-planning line, so a planning-shaped line deeper in the body (e.g. in a code
+block) is never touched. Returns the count of lines removed."
+ (save-excursion
+ (org-back-to-heading t)
+ (forward-line 1)
+ (let ((removed 0) (continue t))
+ (while (and continue (not (eobp)))
+ (let ((line (buffer-substring-no-properties
+ (line-beginning-position) (line-end-position))))
+ (if (string-match-p tc--planning-line-regexp line)
+ (progn
+ (delete-region (line-beginning-position)
+ (min (1+ (line-end-position)) (point-max)))
+ (cl-incf removed))
+ (setq continue nil))))
+ removed)))
+
+(defun tc--convert-one-subtask (marker)
+ "Convert the done sub-task heading at MARKER to a dated event-log entry.
+Under `tc-check-only' the conversion is reported but not performed."
+ (goto-char marker)
+ (org-back-to-heading t)
+ (let* ((level (org-current-level))
+ (title (org-get-heading t t t t))
+ (line (line-number-at-pos))
+ (parts (tc--closed-parts-in-entry)))
+ (cond
+ ((null parts)
+ (push (list :kind 'convert-skip :file tc-current-file
+ :line line :heading title
+ :detail "no CLOSED date to derive the timestamp")
+ tc-issues))
+ (t
+ (let ((new (tc--dated-header-line level parts title)))
+ (cl-incf tc-converted)
+ (if tc-check-only
+ (push (list :kind 'convert-would :file tc-current-file
+ :line line :heading title :new new)
+ tc-issues)
+ ;; Replace the heading line, then drop the whole planning line. The
+ ;; date now lives in the header, so CLOSED is redundant and an active
+ ;; DEADLINE/SCHEDULED would wrongly pin this completed entry to the
+ ;; agenda (todo-format.md). Both go, not just the CLOSED cookie.
+ (delete-region (line-beginning-position) (line-end-position))
+ (insert new)
+ (tc--strip-planning-lines-in-entry)
+ (push (list :kind 'convert-done :file tc-current-file
+ :line line :heading title :new new)
+ tc-issues)))))))
+
+(defun tc-convert-subtasks-in-file ()
+ "Rewrite every level-3-and-deeper DONE/CANCELLED/FAILED heading to a dated
+event-log entry, pulling the timestamp from its CLOSED cookie. Honors
+`tc-check-only'."
+ (let ((targets (tc--convert-collect-targets)))
+ (dolist (m targets)
+ (tc--convert-one-subtask m)
+ (set-marker m nil))))
+
+;;; ---------------------------------------------------------------------------
;;; Driver + reporting
+(defun tc--backup (file)
+ "Copy FILE to /tmp before any modification. Skipped in --check mode.
+
+Matches `lint-org.el' and `wrap-org-table.el', the other tools that rewrite
+these org files. todo-cleanup runs the most often of the three (every wrap,
+every sentry cycle), and Emacs's own backup does not fire under --batch -q, so
+without this a mechanical rewrite has no undo short of git — which recovers
+only to the last commit and loses intra-session work."
+ (let* ((base (format "%s%s.before-todo-cleanup.%s"
+ temporary-file-directory
+ (file-name-nondirectory file)
+ (format-time-string "%Y%m%d-%H%M%S")))
+ (backup base)
+ (n 2))
+ ;; Never overwrite an earlier backup. A second-resolution stamp collides
+ ;; when two invocations run back to back, which the shipped workflow does
+ ;; (open-tasks.org runs --convert-subtasks then --archive-done, each a
+ ;; sub-second batch run). Overwriting there replaces the true pre-session
+ ;; original with already-mutated content — losing exactly what the backup
+ ;; exists to preserve. Suffix instead, so every invocation keeps its own.
+ (while (file-exists-p backup)
+ (setq backup (format "%s-%d" base n))
+ (setq n (1+ n)))
+ (copy-file file backup nil)
+ backup))
+
(defun tc-process-file (file)
(setq tc-current-file (file-name-nondirectory file))
+ (setq tc-current-dir (file-name-directory (expand-file-name file)))
+ (unless tc-check-only
+ (tc--backup file))
(with-current-buffer (find-file-noselect file)
(org-mode)
(cond
@@ -388,6 +876,10 @@ before their descendants — a [#A] → [#B] → [#D] chain collapses in one pas
(tc-archive-done-in-file))
(tc-sync-child-priority
(tc-sync-child-priority-in-file))
+ (tc-convert-subtasks
+ (tc-convert-subtasks-in-file))
+ (tc-seal
+ (tc-seal-archive-file))
(t
;; Pass 1: auto-fix bogus state logs (or report under --check).
(org-map-entries #'tc-fix-bogus-state-log-in-entry nil 'file)
@@ -420,6 +912,21 @@ before their descendants — a [#A] → [#B] → [#D] chain collapses in one pas
(plist-get i :file)
(plist-get i :line)
(if tc-check-only "would move" "moved")
+ (plist-get i :heading)))))))
+ ;; Aged-out subtrees: only reported when some moved (or would). Additive to
+ ;; the in-file report above, and absent when the aging step is disabled.
+ (when (> tc-archived-to-file 0)
+ (princ (format "todo-cleanup --archive-done: %d aged subtree(s) %s task-archive.org%s\n"
+ tc-archived-to-file
+ (if tc-check-only "would move to" "moved to")
+ (if tc-check-only " — CHECK MODE (no writes)" "")))
+ (dolist (i (reverse tc-issues))
+ (pcase (plist-get i :kind)
+ ((or 'archive-file-moved 'archive-file-would)
+ (princ (format " %s:%d: %s %s\n"
+ (plist-get i :file)
+ (plist-get i :line)
+ (if tc-check-only "would archive" "archived")
(plist-get i :heading)))))))))
(defun tc--emit-hygiene-report ()
@@ -467,9 +974,50 @@ before their descendants — a [#A] → [#B] → [#D] chain collapses in one pas
(plist-get i :child-heading)
(plist-get i :parent-heading)))))))
+(defun tc--emit-convert-report ()
+ ;; Silent on a real-mode no-op (nothing to convert and nothing skipped), for
+ ;; the same reason as the archive report: the wrap runs cleanup passes more
+ ;; than once, and a vocal \"0 converted\" reads as noise. Check mode always
+ ;; reports (the preview is what the caller asked for), and a skip always
+ ;; reports (a done sub-task with no CLOSED date is a real condition to see).
+ (let ((has-skip (cl-some (lambda (i) (eq (plist-get i :kind) 'convert-skip))
+ tc-issues)))
+ (when (or tc-check-only (> tc-converted 0) has-skip)
+ (princ (format "todo-cleanup --convert-subtasks: %d sub-task(s) %s%s\n"
+ tc-converted
+ (if tc-check-only "would convert" "converted")
+ (if tc-check-only " — CHECK MODE (no writes)" "")))
+ (dolist (i (reverse tc-issues))
+ (pcase (plist-get i :kind)
+ ((or 'convert-done 'convert-would)
+ (princ (format " %s:%d: %s\n → %s\n"
+ (plist-get i :file) (plist-get i :line)
+ (plist-get i :heading) (plist-get i :new))))
+ ('convert-skip
+ (princ (format " skipped %s:%d: %s — %s\n"
+ (plist-get i :file) (plist-get i :line)
+ (plist-get i :heading) (plist-get i :detail)))))))))
+
+(defun tc--emit-seal-report ()
+ (dolist (i (reverse tc-issues))
+ (pcase (plist-get i :kind)
+ ('seal-done
+ (princ (format "todo-cleanup --seal: sealed task-archive.org → %s\n"
+ (plist-get i :detail))))
+ ('seal-would
+ (princ (format "todo-cleanup --seal: would seal task-archive.org → %s — CHECK MODE (no writes)\n"
+ (plist-get i :detail))))
+ ('seal-collision
+ (princ (format "todo-cleanup --seal: %s already exists — not sealing (already sealed today?)\n"
+ (plist-get i :detail))))
+ ('seal-nothing
+ (princ "todo-cleanup --seal: no working archive to seal\n")))))
+
(defun tc-emit-report ()
(cond (tc-archive-done (tc--emit-archive-report))
(tc-sync-child-priority (tc--emit-sync-report))
+ (tc-convert-subtasks (tc--emit-convert-report))
+ (tc-seal (tc--emit-seal-report))
(t (tc--emit-hygiene-report))))
(defun tc-main ()
@@ -484,6 +1032,12 @@ before their descendants — a [#A] → [#B] → [#D] chain collapses in one pas
(when (member "--sync-child-priority" command-line-args-left)
(setq tc-sync-child-priority t)
(setq command-line-args-left (delete "--sync-child-priority" command-line-args-left)))
+ (when (member "--convert-subtasks" command-line-args-left)
+ (setq tc-convert-subtasks t)
+ (setq command-line-args-left (delete "--convert-subtasks" command-line-args-left)))
+ (when (member "--seal" command-line-args-left)
+ (setq tc-seal t)
+ (setq command-line-args-left (delete "--seal" command-line-args-left)))
;; --check-child-priority is the report-only alias for
;; `--sync-child-priority --check'.
(when (member "--check-child-priority" command-line-args-left)
@@ -491,7 +1045,7 @@ before their descendants — a [#A] → [#B] → [#D] chain collapses in one pas
(setq command-line-args-left (delete "--check-child-priority" command-line-args-left)))
(if (null command-line-args-left)
(progn
- (princ "Usage: emacs --batch -q -l todo-cleanup.el [--check] [--archive-done | --sync-child-priority | --check-child-priority] FILE...\n")
+ (princ "Usage: emacs --batch -q -l todo-cleanup.el [--check] [--archive-done | --seal | --convert-subtasks | --sync-child-priority | --check-child-priority] FILE...\n")
(kill-emacs 1))
(let ((files command-line-args-left))
(setq command-line-args-left nil)
@@ -510,6 +1064,8 @@ ert-run-tests-batch-and-exit'."
(cl-every (lambda (a)
(cond ((member a '("--check"
"--archive-done"
+ "--seal"
+ "--convert-subtasks"
"--sync-child-priority"
"--check-child-priority"))
t)
diff --git a/.ai/scripts/upcoming_birthdays.py b/.ai/scripts/upcoming_birthdays.py
new file mode 100755
index 0000000..d3f30c0
--- /dev/null
+++ b/.ai/scripts/upcoming_birthdays.py
@@ -0,0 +1,182 @@
+#!/usr/bin/env python3
+"""Upcoming-birthdays block for daily prep.
+
+Reads an org-contacts file for ``:BIRTHDAY: YYYY-MM-DD`` properties, finds the
+ones whose next occurrence falls within a window (default 30 days) from today,
+and prints a daily-prep block: name, date, days-away, and — when the birth year
+is real — the age the person is turning. Many contacts use ``1900`` as a
+placeholder year when the real one is unknown; those render date-only, no age.
+Anything within the callout window (default 7 days) is flagged so a gift or
+plan gets prompted.
+
+Pure core:
+ parse_birthdays(text) -> [Birthday(name, month, day, year|None), ...]
+ upcoming(birthdays, today, window=30) -> [Upcoming(name, date, days_away, age|None), ...]
+ format_block(items, window, callout_days=7) -> str
+
+CLI:
+ upcoming_birthdays.py [--file PATH] [--today YYYY-MM-DD] [--window N] [--callout N]
+prints the block on stdout. Default --file is ~/sync/org/contacts.org.
+"""
+
+import argparse
+import datetime as dt
+import re
+import sys
+from dataclasses import dataclass
+from pathlib import Path
+
+PLACEHOLDER_YEAR = 1900
+DEFAULT_CONTACTS = Path.home() / "sync" / "org" / "contacts.org"
+
+_HEADING_RE = re.compile(r"^\*+\s+(.*?)\s*$")
+_TAGS_RE = re.compile(r"\s+:[A-Za-z0-9_@#%:]+:$")
+_BIRTHDAY_RE = re.compile(r"^\s*:BIRTHDAY:\s*(\d{4})-(\d{2})-(\d{2})\s*$")
+
+
+@dataclass(frozen=True)
+class Birthday:
+ name: str
+ month: int
+ day: int
+ year: int | None # None when the source used the 1900 placeholder
+
+
+@dataclass(frozen=True)
+class Upcoming:
+ name: str
+ date: dt.date
+ days_away: int
+ age: int | None # None when the birth year is unknown
+
+
+def _clean_name(heading_text: str) -> str:
+ """Strip a trailing org tag cluster from a heading's text."""
+ return _TAGS_RE.sub("", heading_text).strip()
+
+
+def parse_birthdays(text: str) -> list[Birthday]:
+ """Extract (name, month, day, year|None) for every contact with a BIRTHDAY.
+
+ The name is the nearest preceding org heading. A birth year of 1900 is the
+ placeholder org-contacts uses for an unknown year and is returned as None.
+ Malformed birthday lines are ignored.
+ """
+ birthdays: list[Birthday] = []
+ current_name: str | None = None
+ for line in text.splitlines():
+ heading = _HEADING_RE.match(line)
+ if heading:
+ current_name = _clean_name(heading.group(1))
+ continue
+ bday = _BIRTHDAY_RE.match(line)
+ if bday and current_name:
+ year, month, day = (int(g) for g in bday.groups())
+ # Guard against a nonsense month/day that regex width still admits.
+ try:
+ dt.date(2000, month, day)
+ except ValueError:
+ continue
+ birthdays.append(
+ Birthday(
+ current_name,
+ month,
+ day,
+ None if year == PLACEHOLDER_YEAR else year,
+ )
+ )
+ return birthdays
+
+
+def _next_occurrence(month: int, day: int, today: dt.date) -> dt.date:
+ """First date on/after ``today`` landing on this month/day.
+
+ A Feb 29 birthday maps to Feb 28 in a non-leap year.
+ """
+ def on(year: int) -> dt.date:
+ try:
+ return dt.date(year, month, day)
+ except ValueError:
+ # Only Feb 29 can fail here; fall back to Feb 28.
+ return dt.date(year, 2, 28)
+
+ candidate = on(today.year)
+ if candidate < today:
+ candidate = on(today.year + 1)
+ return candidate
+
+
+def upcoming(
+ birthdays: list[Birthday], today: dt.date, window: int = 30
+) -> list[Upcoming]:
+ """Birthdays whose next occurrence is within ``window`` days, soonest first."""
+ items: list[Upcoming] = []
+ for b in birthdays:
+ occ = _next_occurrence(b.month, b.day, today)
+ days = (occ - today).days
+ if 0 <= days <= window:
+ age = None if b.year is None else occ.year - b.year
+ items.append(Upcoming(b.name, occ, days, age))
+ items.sort(key=lambda u: (u.days_away, u.name))
+ return items
+
+
+def _days_phrase(days: int) -> str:
+ if days == 0:
+ return "today"
+ if days == 1:
+ return "tomorrow"
+ return f"in {days} days"
+
+
+def format_block(items: list[Upcoming], window: int, callout_days: int = 7) -> str:
+ """Render the daily-prep block. Callout entries (within ``callout_days``) are
+ flagged with a marker and a plan-a-gift nudge."""
+ if not items:
+ return f"No birthdays in the next {window} days."
+
+ lines = [f"Upcoming birthdays (next {window} days):"]
+ for u in items:
+ callout = u.days_away <= callout_days
+ marker = "⚠" if callout else "·"
+ date_str = u.date.strftime("%a %b %d")
+ piece = f" {marker} {date_str} — {u.name}"
+ if u.age is not None:
+ piece += f" turns {u.age}"
+ piece += f" ({_days_phrase(u.days_away)})"
+ if callout:
+ piece += " — plan a gift/card"
+ lines.append(piece)
+ return "\n".join(lines)
+
+
+def _parse_today(value: str | None) -> dt.date:
+ if value is None:
+ return dt.date.today()
+ return dt.date.fromisoformat(value)
+
+
+def main(argv: list[str] | None = None) -> int:
+ parser = argparse.ArgumentParser(description="Print the upcoming-birthdays daily-prep block.")
+ parser.add_argument("--file", type=Path, default=DEFAULT_CONTACTS,
+ help="org-contacts file (default: ~/sync/org/contacts.org)")
+ parser.add_argument("--today", default=None,
+ help="override today's date (YYYY-MM-DD), for testing")
+ parser.add_argument("--window", type=int, default=30,
+ help="look-ahead window in days (default: 30)")
+ parser.add_argument("--callout", type=int, default=7,
+ help="flag birthdays within this many days (default: 7)")
+ args = parser.parse_args(argv)
+
+ if not args.file.exists():
+ print(f"contacts file not found: {args.file}", file=sys.stderr)
+ return 1
+
+ text = args.file.read_text(encoding="utf-8")
+ items = upcoming(parse_birthdays(text), _parse_today(args.today), args.window)
+ print(format_block(items, args.window, args.callout))
+ return 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/.ai/scripts/wrap-org-table.el b/.ai/scripts/wrap-org-table.el
index ddbea65..173e44d 100644
--- a/.ai/scripts/wrap-org-table.el
+++ b/.ai/scripts/wrap-org-table.el
@@ -228,22 +228,40 @@ continuation lines merge back into their logical row before re-wrapping."
;;; file layer
(defun wot-process-file (file &optional budget)
- "Reformat every org table in FILE in place to BUDGET width."
+ "Reformat every org table in FILE in place to BUDGET width.
+Pipe-led lines inside #+begin_/#+end_ blocks (example, src, quote, …) are
+content, not tables — ASCII art in an example block once got mangled into a
+bordered table — so block regions are skipped verbatim."
(with-temp-buffer
(insert-file-contents file)
(goto-char (point-min))
- (while (re-search-forward "^[ \t]*|" nil t)
- (let ((start (line-beginning-position)))
- (while (and (not (eobp))
- (save-excursion (beginning-of-line)
- (looking-at "[ \t]*|")))
+ (let ((in-block nil)) ; the open block's type, e.g. "example" — nil outside
+ (while (not (eobp))
+ (cond
+ ;; Only the matching #+end_<type> closes a block: an example block
+ ;; often quotes literal #+begin_src/#+end_src lines, and a boolean
+ ;; flag would let that inner literal end-marker re-expose the rest
+ ;; of the block to reformatting.
+ ((and (not in-block)
+ (looking-at "^[ \t]*#\\+begin_\\([^ \t\n]+\\)"))
+ (setq in-block (downcase (match-string 1)))
(forward-line 1))
- (let* ((end (point))
- (table (buffer-substring-no-properties start end))
- (reformatted (wot-reformat-table-string table budget)))
- (delete-region start end)
- (goto-char start)
- (insert reformatted))))
+ ((and in-block
+ (looking-at-p (format "^[ \t]*#\\+end_%s\\([ \t]\\|$\\)"
+ (regexp-quote in-block))))
+ (setq in-block nil)
+ (forward-line 1))
+ ((and (not in-block) (looking-at-p "^[ \t]*|"))
+ (let ((start (point)))
+ (while (and (not (eobp)) (looking-at-p "^[ \t]*|"))
+ (forward-line 1))
+ (let* ((end (point))
+ (table (buffer-substring-no-properties start end))
+ (reformatted (wot-reformat-table-string table budget)))
+ (delete-region start end)
+ (goto-char start)
+ (insert reformatted))))
+ (t (forward-line 1)))))
(write-region (point-min) (point-max) file)))
;;; ---------------------------------------------------------------------------
@@ -289,7 +307,21 @@ so the ERT suite can `require' this file without firing the CLI dispatch."
(t (file-readable-p a))))
command-line-args-left)))
-(when (and noninteractive (wot--cli-invocation-p))
+(defun wot--entry-script-p ()
+ "Non-nil when wrap-org-table.el itself was named on the command line.
+lint-org.el `require's this file, and a load-triggered dispatch would run
+the table reformatter over lint-org's file arguments — that's how a lint
+invocation once reformatted the files it was only supposed to report on.
+Only dispatch when a -l/--load argument names this very file."
+ (and load-file-name
+ (cl-loop for (flag arg) on command-line-args
+ thereis (and (member flag '("-l" "--load"))
+ (stringp arg)
+ (file-exists-p arg)
+ (string= (file-truename (expand-file-name arg))
+ (file-truename load-file-name))))))
+
+(when (and noninteractive (wot--entry-script-p) (wot--cli-invocation-p))
(wot-main))
(provide 'wrap-org-table)
diff --git a/.ai/sessions/2026-06-13-12-11-inbox-zero-build-and-session-title.org b/.ai/sessions/2026-06-13-12-11-inbox-zero-build-and-session-title.org
new file mode 100644
index 0000000..fc149fb
--- /dev/null
+++ b/.ai/sessions/2026-06-13-12-11-inbox-zero-build-and-session-title.org
@@ -0,0 +1,57 @@
+#+TITLE: Session Context
+#+DATE: 2026-06-13
+
+* Summary
+
+** Active Goal
+
+Run startup, triage two .emacs.d inbox handoffs through Skeptical Review, then implement the session-title task that the inbox pass filed.
+
+** Decisions
+
+- validate-el -L fix: accept as-is. Apply the Phase 1 byte-compile load-path fix to the elisp bundle canonical; leave the Phase 2 test runner untouched (the sender's open question), since cross-project test runs already resolve their own siblings.
+- inbox-zero workflow: build the single-destination version now (claim by =<project>:= prefix, file via process-inbox, leave foreign/unowned items), defer the domain-aware empty-it-all registry. Every scan reports total inbox count plus items appearing related to the current project.
+- Exercise inbox-zero on the one live roam item before pushing.
+- Session title format: =host-project= with a hyphen, no space.
+
+** Data Collected / Findings
+
+- Live roam inbox carried one rulesets-prefixed item ("session name should be hostname-project, no space") and later one archsetup-prefixed item (foreign, left untouched).
+- The session-title hook emitted "$host $project" with a space; the task was to join with a hyphen.
+
+** Files Modified
+
+- languages/elisp/claude/hooks/validate-el.sh (Phase 1 -L fix)
+- claude-templates/.ai/workflows/inbox-zero.org (new) + INDEX.org + startup.org (Phase A count, Phase C nudge) + wrap-it-up.org (Step 3 roam sweep); synced canonical→mirror
+- hooks/session-title.sh (hyphen join) + scripts/tests/session-title-hook.bats + todo.org (task filed then DONE)
+- .ai/notes.org (:LAST_INBOX_PROCESS: marker)
+
+** Next Steps
+
+Session complete. All work committed and pushed: 9e4c580, 651b65e, 25bde1f, f537150, bbd07b6 (rulesets), e6e1caa (roam).
+
+* Session Log
+
+** 2026-06-13 Sat @ 12:11 CDT — Startup + inbox triage
+
+Ran startup. Phase A.0 clean (rulesets up to date, project repo up to date, nothing new to link). Phase A synced .ai/ from templates; session-context absent (last session wrapped cleanly); task staleness 0; no language-bundle drift; no cross-agent messages; notes.org has no active reminders or pending decisions.
+
+Inbox carried four files = two handoffs from .emacs.d, both proposals to change rulesets-owned shared assets, so both go through process-inbox's Skeptical Review:
+
+1. validate-el.sh -L fix (note + attachment) — adds two load-path entries to the elisp bundle hook's Phase 1 byte-compile block so cross-project .el edits compile against their own sibling modules. Diff vs canonical is exactly the comment + two `-L "$(dirname "$f")"` / `-L "$(dirname "$f")/.."` lines. Redundant for in-project edits (already on path), only does work for cross-project edits. Canonical: languages/elisp/claude/hooks/validate-el.sh. Sender flagged one open question: Phase 2 (test runner) was not given the same treatment.
+
+2. inbox-zero.org workflow (note + draft) — a new template workflow routing the global roam inbox to owning projects by prefix. The sender's own note lists 5 "coordination open questions for the rulesets canonical to settle" (the big one: domain-aware empty-it-all mode). Spec-shaped, not accept-as-is. Adjacent to the existing DOING task "Wrap-up inbox/transcript routing" but distinct (that routes session-filed keepers; this routes the shared roam inbox).
+
+Item 1 (validate-el): Craig picked accept-as-is. Applied the Phase 1 -L fix to canonical (byte-identical to attachment), bundle bats green (22), manual smoke test exit 0, /review-code Approve. Committed 9e4c580. Replied to .emacs.d (accept + the Phase 2 left-alone rationale). Inbox files removed. Closed. (.emacs.d sent a pure-FYI ack later; deleted, no reply.)
+
+Item 2 (inbox-zero): Craig picked build-the-single-destination-version-now, defer the domain registry. Added a scan-summary requirement: every scan reports total inbox count + items appearing related to this project. Built claude-templates/.ai/workflows/inbox-zero.org (single-destination: claim by <project>: prefix, file via process-inbox discipline, leave foreign/unowned; deferred domain-aware mode written into the file). Wired both callers: startup.org Phase A item 11 count + Phase C nudge bullet; wrap-it-up.org Step 3 roam-sweep sub-step before cleanup. INDEX entry added. Synced canonical→mirror (sync-check exit 0), workflow-integrity OK (44), /review-code Approve (one Minor: heading-count heuristic assumes single top-level section, true today). Committed 651b65e. Replied to .emacs.d (single-destination landed, registry deferred). Inbox files removed.
+
+Live roam inbox has exactly one item, rulesets-prefixed: "[#C] rulesets: claude code session name should be hostname-project (no space)" — a refinement on the 2026-06-11 session-title hook work. Candidate to exercise the new workflow on.
+
+Both commits unpushed (9e4c580, 651b65e) pending push confirmation.
+
+Craig picked: run inbox-zero on the live item, then push. Exercised inbox-zero end to end on the rulesets roam item. Phase A: roam pull (ff-only) brought in a new archsetup-prefixed item (foreign, left untouched — restored a trailing space I'd accidentally trimmed on its line). Phase B: filed "[#C] Session title hostname-project, no space :feature:quick:" into rulesets todo.org Open Work (status-checked: no existing task; current hook hooks/session-title.sh emits "$host $project" with a space). Phase C: removed only the rulesets line from roam inbox.org, committed roam separately (e6e1caa) and pushed. Filed-task commit 25bde1f. Pushed rulesets main to origin (d22e747..25bde1f): all three commits (9e4c580, 651b65e, 25bde1f) live. The workflow ran clean end to end as its own first real exercise.
+
+Uncommitted, riding wrap-up: notes.org (:LAST_INBOX_PROCESS: marker), .ai/session-context.org (untracked anchor).
+
+Craig: implement the filed session-title task. TDD: flipped the 3 session-title-hook.bats expectations to the hyphen format first (confirmed red, 3 failed), changed hooks/session-title.sh join "$host $project" → "$host-$project" plus both header comments (green, 6/6), updated the bats top comment, marked the todo task DONE with a resolution note. Full suite green: make test exit 0 (pytest 415+54+12, ERT 33/28/23/12, all bats ok). /review-code Approve. The hook is symlinked into ~/.claude/hooks, so the new format is live next session. Commits: f537150 (feat, hook+bats+todo), bbd07b6 (chore, notes marker). Pushed to origin (25bde1f..bbd07b6). Only .ai/session-context.org remains (the live anchor).
diff --git a/.ai/sessions/2026-06-13-15-29-codex-inbox-ack.org b/.ai/sessions/2026-06-13-15-29-codex-inbox-ack.org
new file mode 100644
index 0000000..817d280
--- /dev/null
+++ b/.ai/sessions/2026-06-13-15-29-codex-inbox-ack.org
@@ -0,0 +1,30 @@
+#+TITLE: Session Context - codex
+#+DATE: 2026-06-13
+
+* Summary
+
+** Active Goal
+
+Startup as agent =codex=, then clear the inbox.
+
+** Decisions
+
+The single inbox handoff (an .emacs.d acknowledgment that rulesets commit 651b65e landed) explicitly required no local action, so the inbox pass cleared it without further work.
+
+** Data Collected / Findings
+
+No =codex= crash-recovery context existed. One unprocessed inbox handoff from .emacs.d, an ack-only FYI.
+
+** Files Modified
+
+.ai/notes.org (:LAST_INBOX_PROCESS: stamp updated to the afternoon ack-cleared note).
+
+** Next Steps
+
+None. Session complete.
+
+* Session Log
+
+** Startup and inbox pass
+
+Saturday 2026-06-13 15:29 CDT: Started as agent =codex= and followed =.ai/protocols.org=. Phase A.0 found rulesets already current and no new =make install= symlinks. Startup synced =.ai/= from templates, found no =codex= crash-recovery context, surfaced recent session summaries, and found one unprocessed inbox handoff from =.emacs.d=. The handoff was an acknowledgment that rulesets commit =651b65e= landed and explicitly required no local action, so the inbox pass cleared it and updated =.ai/notes.org='s inbox-process stamp.
diff --git a/.ai/sessions/2026-06-15-08-43-czsusp-epoch-helper-instance-slices.org b/.ai/sessions/2026-06-15-08-43-czsusp-epoch-helper-instance-slices.org
new file mode 100644
index 0000000..7d150d5
--- /dev/null
+++ b/.ai/sessions/2026-06-15-08-43-czsusp-epoch-helper-instance-slices.org
@@ -0,0 +1,61 @@
+#+TITLE: Session Context
+#+DATE: 2026-06-14
+
+* Summary
+
+** Active Goal
+
+Fix an accidental C-z that suspended Claude, clear two interrupted 2026-06-13 session anchors, document the helper-agent epoch-id convention, then start the helper-instance support feature (#B) — shipping its first two slices.
+
+** Decisions
+
+- C-z suspend: scope the disable to ai-launched panes only (stty susp undef in the launcher), not globally. Keybindings.json can't touch it — C-z is raw tty SIGTSTP, not a Claude Code keybinding.
+- Helper-agent id uniqueness: epoch lives in AI_AGENT_ID, baked in by the spawner, convention-only. It can't be minted inside session-context-path (the resolver must return the same path across many calls per session).
+- For the rest of the session: commit and push together, don't hold pushes (Craig).
+- Helper-instance build order: agent-roster first (the detection primitive), then helper-mode.org (the contract); both inert. The behavior-changing wiring stays behind the spec's bats→drills→pilot gate.
+- agent-roster exit codes 0=alone / 1=others / 2=unavailable; pgrep-absent hardened to exit 2 (never silent-alone) after a review Minor.
+
+** Data Collected / Findings
+
+- Real terminal is foot → tmux → Claude (not "ghostel"; ghostel is the separate Emacs-native terminal, per the spec).
+- agent-roster live-verified against 4 concurrent real sessions: alone in rulesets (own session excluded via ancestry, 3 out-of-project claudes excluded by cwd), not-alone in ~/.emacs.d (2 agents listed).
+- helper-mode.org orphan-drift wrinkle solved by a triggerless INDEX catalog entry (startup.org's existing pattern) — no integrity-checker or startup.org change.
+
+** Files Modified
+
+- claude-templates/bin/ai — LAUNCH_PREFIX="stty susp undef; " at the three launch sites (C-z fix). Commit dedbca3.
+- .ai/sessions/2026-06-13-* (x2) + notes.org marker — archived yesterday's interrupted anchors. Commit 16e64fb.
+- protocols.org + wrap-it-up.org + session-context-path header — epoch-on-tail convention. Commit e0f914d.
+- .ai/scripts/agent-roster + tests (canonical+mirror) — detection primitive, 11 bats. Commit f8bdf30.
+- .ai/workflows/helper-mode.org (new) + INDEX.org + protocols.org pointer — helper contract. Commit 0b681dc.
+- todo.org — helper-instance DOING + resume note + Craig's open question. Commits 25bde1f-era flip, bef1d66, and wrap.
+
+** Next Steps
+
+Helper-instance is the big item next session. Re-orient on what it is/does/why before building (resume note in the helper-instance task body). Answer Craig's open question first: does helper-instance depend on the generic-agent-runtime work, or is Phase 1.5 genuinely independent of phases 2-6? Then the gated wiring: startup roster branch, wrap-it-up helper branch, ai --helper launcher (+ Emacs surface via .emacs.d handoff), hygiene live-helper gate, todo-cleanup.el /tmp backstop — all behind the drill rig.
+
+KB: promoted 0 / consulted no
+
+* Session Log
+
+** 2026-06-14 Sun @ 17:52 CDT — Startup, then disable C-z suspend in ai-launched Claude panes
+
+New session (2026-06-14). Startup found two un-wrapped 2026-06-13 anchors (the main session-context.org + session-context.d/codex.org) with all work already pushed; offered cleanup but Craig pivoted to a live problem: he accidentally hit C-z during the session, suspending Claude to the shell. Wants C-z's suspend disabled but scoped, not nuked machine-wide.
+
+Diagnosed the layers: foot (terminal) → tmux → Claude Code. tmux has no root C-z binding, so C-z passes through to the pane's tty, which generates SIGTSTP. Confirmed via the keybindings skill that ctrl+z is NOT a Claude Code keybinding action — it's listed under Reserved Shortcuts as Unix SIGTSTP, no action mapped — so keybindings.json can't touch it. Fix has to be at the tty layer (stty susp undef).
+
+Craig picked launcher-scoped over global zshrc. Edited claude-templates/bin/ai: added a commented LAUNCH_PREFIX="stty susp undef; " var after CLAUDE_CMD, prepended it at all three tmux send-keys launch sites (lines ~76, ~321, ~381). Scopes the disable to exactly the ai-launched Claude pane; C-z keeps working everywhere else. Verified: bash -n clean, all three sites carry the prefix, stty susp undef in a real pty leaves susp = <undef>. Live now via the ~/.local/bin/ai symlink, but only for NEW ai-launched sessions — the current running session keeps its old tty setting. /review-code Approve, /voice personal walk, committed dedbca3 (unpushed). Note: with susp undef the 0x1a byte now reaches Claude as input (no binding → expected to be ignored).
+
+Then Craig's task order: (2) commit the C-z fix [done, dedbca3], (3) clean up yesterday's two un-wrapped anchors, (1) return to the epoch-suffix thread. Cleanup: split yesterday's shipped 12:11 session out of this anchor into .ai/sessions/2026-06-13-12-11-inbox-zero-build-and-session-title.org (Summary populated), archived the codex sub-session to .ai/sessions/2026-06-13-15-29-codex-inbox-ack.org and removed session-context.d/codex.org, reset this anchor to today only.
+
+Then the epoch thread (task 1). Lost the original design in the crash, so reconstructed from code and surfaced the determinism constraint (session-context-path is called many times per session and must return the same path, so it can't mint date +%s; the Bash tool also drops env between calls, so the agent can't export once). Craig picked option 1: epoch baked into AI_AGENT_ID by the spawner, convention-only, no resolver change. Documented the convention in three canonical spots (protocols.org Agent-scoped path, wrap-it-up.org recommended-shape note, session-context-path header comment), with the 2026-06-13 codex collision as the worked example. sync-check --fix paired the mirror; session-context-path.bats still 5/5 green (comment-only). /review-code Approve, /voice personal walk, committed e0f914d.
+
+Session commits: dedbca3 (C-z fix), 16e64fb (archive yesterday's two anchors + inbox marker), e0f914d (epoch convention docs). Pushed bbd07b6..e0f914d. Craig then set "commit and push together for the rest of this session."
+
+Then started the helper-instance task (#B, DOING) via start-work, first slice only: the agent-roster detection script. Built claude-templates/.ai/scripts/agent-roster + tests/agent-roster.bats (11 tests), synced to mirror. Algorithm per spec: pgrep -x claude, /proc cwd resolution, cwd-within-root filter, self-ancestry exclusion. Exit 0 alone / 1 others / 2 unavailable. pgrep + /proc + self-pid injectable (ROSTER_PGREP/PROC/SELF_PID) so bats run the real filter against fixtures, no agents spawned. TDD red→green. Live-verified against the real process tree: alone in rulesets (own session excluded via ancestry, 3 out-of-project claudes excluded by cwd, exit 0), not-alone against ~/.emacs.d (2 live agents listed, exit 1). /review-code flagged one Minor (pgrep-absent read as silent-alone, violating the spec's never-silent-alone invariant); hardened it with a command -v guard → exit 2 (TDD red→green, 11th test). make test exit 0 (180 ok). Committed f8bdf30, pushed e0f914d..f8bdf30. todo task stays DOING (epic; slice 1 of ~7 done).
+
+Slice 2: helper-mode.org workflow contract (the spec's "single canonical home" of helper rules). Authored claude-templates/.ai/workflows/helper-mode.org — Overview + subagent boundary, When-to-Use (no trigger, three routing paths), Identity (helper-<rand4>, self-assign, recorded as first line of .d/ file), four read/write tiers, four data-integrity rules + inbox-send slug nit, Light Startup, Helper Wrap-Up (orphaned-helper lifts git ban), Status (wiring gated/not-live). Solved the orphan-drift-check wrinkle via a triggerless INDEX catalog entry (startup.org's pattern) — no integrity-checker or startup.org change. Added protocols.org one-paragraph pointer. Synced mirror. workflow-integrity OK (45 workflows), make test exit 0 (180 ok), /review-code Approve. Committed 0b681dc, pushed f8bdf30..0b681dc.
+
+Remaining slices, all behind the spec's bats→drills→pilot gate (change live session behavior in synced paths): startup roster-detection branch, wrap-it-up helper branch, ai --helper launcher (+ ai-term.el via .emacs.d handoff), hygiene-pass live-helper gate, todo-cleanup.el /tmp backstop.
+
+Stopped here for the session (Craig's call). Durable resume note filed in the helper-instance task body in todo.org — what it is/does/why plus done-vs-next — flagged to re-orient on the feature's purpose before picking it up as the big item next session. The wiring is the gated, drill-rig half. Not a full wrap-up; this anchor stays live.
diff --git a/.ai/sessions/2026-06-16-23-37-cross-agent-comms-removal-and-batch-specs.org b/.ai/sessions/2026-06-16-23-37-cross-agent-comms-removal-and-batch-specs.org
new file mode 100644
index 0000000..1857e58
--- /dev/null
+++ b/.ai/sessions/2026-06-16-23-37-cross-agent-comms-removal-and-batch-specs.org
@@ -0,0 +1,194 @@
+#+TITLE: Session Context
+#+DATE: 2026-06-15
+
+* Summary
+
+** Active Goal
+
+A long overnight session: process inbound, then run an autonomous 30-min work loop, ending with two directed tasks — write specs from Craig's cj instructions, and remove the unused cross-agent-comms subsystem.
+
+** Decisions
+
+- Reconciled the Phase E (inbox-zero) and "fix speedrun" proposals into ONE autonomous-batch-execution spec: a dedicated =work-the-backlog.org= holds the loop, inbox-zero stays routing-only, "fix speedrun" is a thin preset. (Craig's "your call".)
+- Demoted create-documentation + research-writer to [#D] (designed, unbuilt, awaiting a trigger).
+- Shared-asset change proposals park as [#B] VERIFYs in no-approvals mode, never self-apply (Phase E, dotfiles-discovery, archsetup ai-launcher).
+- cross-agent-comms is removable: an unused parallel system; inbox-send is the live handoff mechanism (kept). Removed it; repointed helper-mode escalation to inbox-send.
+- Wrap-up-routing implementation deferred: it moves tasks across projects' todo.org files (data-loss-adjacent), needs a focused /start-work, not a tail-end rush.
+
+** Data Collected / Findings
+
+- The 30-min loop ran ~28 idle cycles (cycles 1-30); most inbound was =emacs:=-prefixed roam items (foreign, left for .emacs.d). No rulesets task was ever tagged :next: / :solo:+:quick:, so the loop never had eligible work.
+- Candidate quick-wins surfaced: shellcheck warnings across 6 scripts (2 latent-bug suspects in bin/ai — SC2088 tilde-in-quotes, SC1083 literal brace), the parked dotfiles one-liner, token-rotation helper.
+- cross-agent-comms footprint removed: 7 scripts + 7 READMEs (canonical + mirror), cross-agent-comms.org workflow, INDEX entry, 6 test files, 3 startup.org wirings + 2 summary mentions, helper-mode escalation ref, 7 legacy ~/.local/bin symlinks. Verified: no residual refs, sync-check clean, make test green.
+
+** Files Modified
+
+- Removed: claude-templates/.ai/scripts/cross-agent-comms/ + mirror, the 6 test_cross_agent_*.py, cross-agent-comms.org (+ mirror).
+- Edited (canonical + sync'd mirror): startup.org (3 wirings + 2 summary lines, renumbered Phase A), INDEX.org (dropped entry), helper-mode.org (escalation → inbox-send).
+- docs/design/: 2026-06-16-autonomous-batch-execution-spec.org (new), 2026-06-16-encourage-kb-contribution-spec.org (new), generic-agent-runtime-spec.org (removal status note), 2026-06-15-fix-speedrun-workflow-proposal.org (filed source).
+- todo.org: many task-state edits (specs filed with review VERIFYs, audit reconciliation, demotions, parked VERIFYs, Craig's reorder).
+- working/: inbox-zero-phase-e/, ai-dotfiles-discovery/ (staged parked changes).
+- ~/.dotfiles: bootstrapped .ai/ in gitignore mode (install-ai).
+
+** Next Steps
+
+- Three parked VERIFYs await Craig: dotfiles-discovery (one line), autonomous-batch spec (6 decisions), KB-contribution spec (5 decisions).
+- Wrap-up-routing implementation — the next focused /start-work build (data-loss checkpoint).
+- Quick-win backlog to tag :solo:+:quick: when wanted: the shellcheck cleanup (with the 2 bin/ai latent-bug suspects flagged for eyes).
+
+KB: promoted 0 / consulted no
+
+* Session Log
+
+** Startup + inbox pass (2026-06-15 23:29 CDT)
+
+Ran startup clean: no crashed anchor, templates synced, no reminders/pending decisions, roam inbox 0, no cross-agent messages. Staleness nudge: 1 top-level task unreviewed >7 days. Two DOING items remain open (wrap-up routing spec ready for spec-review; helper-instance Phase 1.5 awaiting go to build).
+
+Processed 3 inbox items:
+- .emacs.d "fix speedrun" reusable-workflow proposal — ran the Skeptical Review with the cross-project battery. Passes the value gate but carries 4 unresolved design questions (new-workflow-vs-preset, page firing point + mechanism amid the page-signal removal, auto-pull vs explicit list, guardrails against design/data-loss work). Craig chose option 1: file as a task, build via spec-create later. Preserved the proposal at docs/design/2026-06-15-fix-speedrun-workflow-proposal.org and filed a [#C] :feature:spec: task carrying the skeptical-review open questions.
+- Two pearl FYIs (gitignore-tooling applied; memory sweep Phase 1.5 done) — acks of handoffs we sent, nothing asked. Deleted.
+
+Updated :LAST_INBOX_PROCESS: marker to 2026-06-15. Inbox empty.
+
+** Autonomous 30-min loop started (2026-06-15 23:44 CDT)
+
+Craig set up an autonomous loop: every 30 min, inbox-zero (project inbox + roam global), then find a task tagged :next: OR both :solo: and :quick:, evaluate it, write a spec if useful, implement it in no-approvals mode until the commit is pushed, then inbox-zero again. Runs until told to stop.
+
+Committed + pushed the inbox pass (4fe184e → e7be0de..4fe184e on origin/main).
+
+Cycle 1 (23:44): both inboxes already at zero. No eligible task — nothing tagged :next:, nothing tagged both :solo:+:quick: (no open task carries :solo: at all). One task has :quick: alone (Token-rotation helper, line 1031) but not :solo:, so it doesn't qualify. No-op cycle. Scheduled next wake in 30 min.
+
+Guardrails honored each cycle: shared-asset/convention change proposals never self-apply even in no-approvals (defer-and-stage as [#B] VERIFY per process-inbox Skeptical Review); refuse to speedrun tasks needing design decisions or carrying data-loss risk without a checkpoint (the fix-speedrun guardrail).
+
+** Full task audit (2026-06-15, Craig-requested, interrupts the loop)
+
+Ran task-audit.org over all 11 open tasks. Fanned out 3 read-only Explore subagents over batches; reconciled against session summaries + git + repo state; applied edits serially.
+
+STALE → updated autonomously:
+- Helper-instance (line 46): bumped LAST_REVIEWED 2026-06-12 → 2026-06-15 (this morning's session shipped agent-roster f8bdf30 + helper-mode.org 0b681dc; the 2026-06-15 progress note already captures the shipped-vs-wiring split accurately, so only the stamp lagged).
+- Memories-sync (line 87): rewrote the stale preamble (it claimed phases + validation both "remaining"; Phases 0-4 are all shipped) and bumped LAST_REVIEWED → 2026-06-15. Implementation complete; manual-testing child + other-machine roam.git clone (archsetup handoff) remain before DONE.
+
+CURRENT (left as-is): wrap-up routing (line 37, still awaiting spec-review), morning-ops (182), c4-rename (885), token-rotation (1031, :solo: deliberately withheld), spec-storage (1084) + fix-speedrun (1104) both filed today.
+
+NEEDS-USER (flagged, awaiting Craig): create-documentation [#C] vs [#D]; research-writer [#C] vs [#D]; generic-agent-runtime (1058) overlap/dependency with helper-instance (46).
+
+Phase E: stamped :LAST_AUDIT: 2026-06-15 in notes.org. Phase F (task-review chain) pending the NEEDS-USER adjudication.
+
+Craig adjudicated the 3 NEEDS-USER flags (chose accept-all): demoted create-documentation (line 191) and research-writer (line 828) [#C]→[#D]; kept generic-agent-runtime separate from helper-instance (open dependency question stays the recorded blocker). Phase F completed: re-reviewed all remaining open tasks (reconciled CURRENT in the audit) and bumped their LAST_REVIEWED to 2026-06-15. No kills. Committed + pushed the audit (ab9f79a → 4fe184e..ab9f79a on origin/main).
+
+** install-ai on ~/.dotfiles (2026-06-16, Craig-requested, cross-project)
+
+Craig asked to install-ai ~/.dotfiles in gitignore mode and page him. Ran scripts/install-ai.sh --gitignore $HOME/.dotfiles: created .ai/ (protocols, workflows, scripts, notes.org seeded project=.dotfiles) + inbox/, and appended .ai/ .claude/ CLAUDE.md AGENTS.md to ~/.dotfiles/.gitignore. Verified .ai/ is git-ignored (git check-ignore IGNORED; absent from git status). (.claude shows not-ignored via check-ignore only because the dir doesn't exist yet — directory-scoped pattern; will ignore once created.) Left the .gitignore change uncommitted — ~/.dotfiles had 11 unrelated modified WIP files, so the commit decision is Craig's. Paged him via notify success --persist (exit 0).
+
+** Loop cycle 2 (2026-06-16 00:15 CDT)
+
+Inbox had real work this time: 2 .emacs.d handoffs + 1 roam item. No eligible task (no :next:, no :solo:+:quick:).
+
+- .emacs.d inbox-zero Phase E proposal (adds autonomous task execution to the synced inbox-zero.org): shared-asset change in a no-approvals session → defer-and-stage per the Skeptical Review gate. Staged the proposed file + diff + sender note under working/inbox-zero-phase-e/, filed a [#B] VERIFY (recommend spec it, not apply: assumes .emacs.d's commit waiver, hardcodes eligibility tags, undefined do-not-implement set + kill-switch, unresolved seam question; overlaps the fix-speedrun task — reconcile into one spec). Replied to .emacs.d that it's parked.
+- Roam item "rulesets: encourage building a knowledge base..." (:next:, rulesets-prefixed): claimed and filed as [#C] :feature: "Encourage org-roam KB contribution across workflows." Dropped the :next: tag — it touches 4 synced workflows and needs a best-practices curation decision, so it's a design task, not a loop auto-implement (guardrail). Removed from roam inbox, committed + pushed roam (af1e09f).
+- Rulesets commit 26bcae6 (todo.org VERIFY + task, working/ staged). Both inboxes verified empty. Rescheduled next wake +30 min.
+
+** /respond-to-cj-comments on todo.org (2026-06-16, Craig-invoked)
+
+cj-scan found 4 cj comments. Craig had also flipped Helper-instance and memories-sync to VERIFY in his buffer (his edits, folded into the same commit).
+
+- cj on Phase E VERIFY ("write a spec, file a verify subtask") + cj on fix-speedrun ("your call" on workflow-vs-preset + effectiveness measurement): reconciled into ONE autonomous-batch execution spec (docs/design/2026-06-16-autonomous-batch-execution-spec.org). Design call (mine, per "your call"): a dedicated work-the-backlog.org holds the loop; "fix speedrun" is a thin preset; inbox-zero stays routing-only. Spec also designs the effectiveness-measurement trial (per-task JSONL + org-roam synthesis articles). Phase E VERIFY folded to a dated entry; fix-speedrun got a dated answer + a *** VERIFY "Review the autonomous-batch execution spec".
+- cj on KB-encouragement ("write a spec, file a verify subtask"): docs/design/2026-06-16-encourage-kb-contribution-spec.org (4 light workflow prompts + curated best-practices node, sources cited). Filed *** VERIFY "Review the KB-contribution spec".
+- cj on Wrap-up routing ("approved, take through spec-response → implementation"): LEFT IN PLACE / deferred. The build moves tasks between projects' todo.org files = data-loss-adjacent cross-project mutation; per the guardrail it needs a focused /start-work session with a checkpoint, not a tail-end rush after two specs. Craig's approval edit to wrapup-routing-spec.org is unstaged (belongs with that build).
+
+Two spec drafts came from parallel general-purpose subagents (held the design decisions myself; they drafted the docs). Commit 7467d1f pushed. Both review VERIFYs await Craig.
+
+Pending at task boundary: a new inbox handoff arrived mid-task — inbox/2026-06-16-0054-from-archsetup-ai-launcher-dotfiles-discovery.org — not yet processed (next loop cycle or a follow-up).
+
+** Loop cycle 3 (2026-06-16 00:59 CDT)
+
+Processed the archsetup handoff (add =~/.dotfiles= to bin/ai's build_candidates() discovery, completing the dotfiles bootstrap). Synced asset + no-approvals → defer-and-stage. Skeptical review confirmed correct (verified build_candidates only scans .emacs.d/code/projects; maybe_add_candidate guards on .ai/protocols.org; no bin/ai discovery test to also update). Staged the one-line diff under working/ai-dotfiles-discovery/, filed a [#B] VERIFY "Parked: add ~/.dotfiles to the ai launcher's project discovery", replied to archsetup (parked). Commit 4e2db8f pushed. No eligible task (no :next:, no :solo:+:quick:). Both inboxes verified empty. Rescheduled +30 min.
+
+** Loop cycle 4 (2026-06-16 01:32 CDT)
+
+No-op cycle. Both inboxes empty, no eligible task. Craig reordered todo.org in his Emacs buffer (moved the Phase E / wrap-up-routing / KB-contribution blocks down below the memories-sync task) — a pure reorder, byte-identical content, no new cj/approval/tag. That edit plus his earlier wrapup-routing-spec.org approval annotation are uncommitted; left untouched (his in-progress WIP, not the loop's to commit). No commit this cycle. Rescheduled +30 min.
+
+** Loop cycle 5 (2026-06-16 02:04 CDT)
+
+No-op (3rd idle cycle). Inboxes empty, no eligible task, no new cj comments (the wrap-up-routing one is still the only one, still deferred). Craig's todo.org reorder + wrapup-routing-spec.org approval still uncommitted, still left alone. No commit. Rescheduled +30 min.
+
+** Loop cycle 6 (2026-06-16 02:35 CDT)
+
+No-op (4th idle cycle). State unchanged from cycle 5: inboxes empty, no eligible task, one deferred cj (wrap-up routing), Craig's WIP still uncommitted. No commit. Rescheduled +30 min.
+
+** Loop cycle 7 (2026-06-16 03:06 CDT)
+
+No-op (5th idle cycle). State frozen since cycle 4: inboxes empty, no eligible task, wrap-up-routing cj still deferred, Craig's reorder + spec approval still uncommitted. No commit. Rescheduled +30 min.
+
+** Loop cycle 8 (2026-06-16 03:37 CDT)
+No-op (6th idle cycle). State unchanged. No commit. Rescheduled +30 min.
+
+** Loop cycle 9 (2026-06-16 04:08 CDT)
+No-op (7th idle cycle). State unchanged. No commit. Rescheduled +30 min.
+
+** Loop cycle 10 (2026-06-16 04:39 CDT)
+No-op (8th idle cycle). State unchanged. No commit. Rescheduled +30 min.
+
+** Loop cycle 11 (2026-06-16 05:10 CDT)
+No-op for rulesets (9th idle cycle). A new roam inbox item appeared but it's =emacs:=-prefixed (difftastic "d" diff option for the buffer-differs prompt, :next:) — foreign to rulesets, left untouched for .emacs.d's session to claim per the ownership rule. No rulesets-owned roam items, no project inbox, no eligible todo.org task. No commit. Rescheduled +30 min.
+
+** Loop cycle 12 (2026-06-16 05:41 CDT)
+No-op (10th idle cycle). The foreign =emacs:= roam item from cycle 11 is gone (claimed by .emacs.d). Roam + project inboxes empty, no eligible task. No commit. Rescheduled +30 min.
+
+** Loop cycle 13 (2026-06-16 06:12 CDT)
+No-op for rulesets (11th idle cycle). Another =emacs:=-prefixed roam item appeared (studio view-assignment dropdown next/prev buttons, :next:) — foreign, left for .emacs.d. No rulesets-owned items, no project inbox, no eligible task. No commit. Rescheduled +30 min.
+
+** Loop cycle 14 (2026-06-16 06:43 CDT)
+No-op (12th idle cycle). Cycle 13's foreign =emacs:= roam item cleared (.emacs.d claimed it). All inboxes empty, no eligible task. No commit. Rescheduled +30 min.
+
+** Loop cycle 15 (2026-06-16 07:14 CDT)
+No-op (13th idle cycle). All inboxes empty, no eligible task, state unchanged. No commit. Rescheduled +30 min.
+
+** Loop cycle 16 (2026-06-16 07:45 CDT)
+No-op (14th idle cycle). All inboxes empty, no eligible task, state unchanged. No commit. Rescheduled +30 min.
+
+** Loop cycle 17 (2026-06-16 08:16 CDT)
+No-op (15th idle cycle). All inboxes empty, no eligible task, state unchanged. No commit. Rescheduled +30 min.
+
+** Loop cycle 18 (2026-06-16 08:47 CDT)
+No-op for rulesets (16th idle cycle). Two new =emacs:=-prefixed roam items (theme-studio palette-closed default, assignment-view dropdown improvements, both :next:) — foreign, left for .emacs.d. No rulesets-owned items, no project inbox, no eligible task. No commit. Rescheduled +30 min.
+
+** Loop cycle 19 (2026-06-16 09:18 CDT)
+No-op (17th idle cycle). Cycle 18's foreign =emacs:= roam items cleared (.emacs.d claimed them). All inboxes empty, no eligible task. No commit. Rescheduled +30 min.
+
+** Loop cycle 20 (2026-06-16 09:49 CDT)
+No-op (18th idle cycle). All inboxes empty, no eligible task, state unchanged. No commit. Rescheduled +30 min.
+
+** Loop cycle 21 (2026-06-16 10:20 CDT)
+No-op (19th idle cycle). All inboxes empty, no eligible task, state unchanged. No commit. Rescheduled +30 min.
+
+** Loop cycle 22 (2026-06-16 10:51 CDT)
+No-op (20th idle cycle). All inboxes empty, no eligible task, state unchanged. No commit. Rescheduled +30 min.
+
+** Loop cycle 23 (2026-06-16 11:22 CDT)
+No-op (21st idle cycle). All inboxes empty, no eligible task, state unchanged. No commit. Rescheduled +30 min.
+
+** Loop cycle 24 (2026-06-16 11:53 CDT)
+No-op (22nd idle cycle). All inboxes empty, no eligible task, state unchanged. No commit. Rescheduled +30 min.
+
+** Loop cycle 25 (2026-06-16 12:24 CDT)
+No-op (23rd idle cycle). All inboxes empty, no eligible task, state unchanged. No commit. Rescheduled +30 min.
+
+** Loop cycle 26 (2026-06-16 12:55 CDT)
+No-op (24th idle cycle). All inboxes empty, no eligible task. Craig edited .claude/settings.json (removed the pinned =model: opus= line) — his config WIP, left uncommitted with his other edits. No commit. Earlier this turn surfaced candidate work for Craig (shellcheck warnings across 6 scripts incl. 2 latent-bug-suspect ones in bin/ai; the parked dotfiles one-liner; token-rotation helper; spec-storage decision) — awaiting his pick on what to tag/fix. Rescheduled +30 min.
+
+** Loop cycle 27 (2026-06-16 13:27 CDT)
+No-op (25th idle cycle). All inboxes empty, no eligible task, state unchanged (Craig's settings.json + reorder WIP still uncommitted). No commit. Rescheduled +30 min.
+
+** Loop cycle 28 (2026-06-16 13:58 CDT)
+No-op (26th idle cycle). All inboxes empty, no eligible task, state unchanged. No commit. Rescheduled +30 min.
+
+** Loop cycle 29 (2026-06-16 14:29 CDT)
+No-op (27th idle cycle). All inboxes empty, no eligible task, state unchanged. No commit. Rescheduled +30 min.
+
+** Loop cycle 30 (2026-06-16 14:59 CDT)
+No-op (28th idle cycle). All inboxes empty, no eligible task, state unchanged. No commit. Rescheduled +30 min.
+
+** Loop STOPPED + remove cross-agent-comms (2026-06-16 ~15:05 CDT)
+
+Craig stopped the 30-min loop (the queued 15:31 wakeup fired once after; not continued). Then directed removing the unused cross-agent-comms subsystem (option 2: remove now, repoint helper-mode/spec refs to inbox-send). Distinct from inbox-send (the live handoff mechanism, kept). Removing: 7 scripts + 7 READMEs, cross-agent-comms.org workflow, INDEX entry, 5 test files, 3 startup.org wirings (Phase A step 6, Phase B step 4, Phase C bullet) + 2 summary mentions, helper-mode.org escalation ref (repoint to inbox-send / report-to-Craig), 7 legacy ~/.local/bin/cross-agent-* symlinks. Editing canonical (claude-templates/.ai/) then sync-check --fix to mirror. generic-agent-runtime-spec.org gets a removal note (not a full rewrite of its 9 historical refs).
diff --git a/.ai/sessions/2026-06-21-02-44-launcher-fix-kb-feature-wrapup-routing.org b/.ai/sessions/2026-06-21-02-44-launcher-fix-kb-feature-wrapup-routing.org
new file mode 100644
index 0000000..935e61d
--- /dev/null
+++ b/.ai/sessions/2026-06-21-02-44-launcher-fix-kb-feature-wrapup-routing.org
@@ -0,0 +1,110 @@
+#+TITLE: Session Context
+#+DATE: 2026-06-20
+
+* Summary
+
+** Active Goal
+
+Started as "~/.dotfiles not showing in the ai launcher," then ran long across tooling and convention work: the launcher fix, a level-2 VERIFY completion rule change, the KB-contribution feature + a new lint checker, an eight-project conversion broadcast, and the wrap-up routing spec taken through review + response to Ready with a task breakdown.
+
+** Decisions
+
+- ~/.dotfiles → track mode for its .ai/ (private cjennings.net remote, so no leak concern) — the durable fix so the scaffold travels to every machine instead of needing per-machine bootstrap.
+- Level-2 (=**=) VERIFY completes task-shaped (DONE/CANCELLED + CLOSED:), never a dated header; dated rewrites are =***=+ only (from .emacs.d; applied across 4 producers + a lint checker).
+- KB-contribution spec ratified, plus a new decision D6 (read-side startup consult-nudge — the counterpart to the write-side encouragement the spec lacked).
+- Wrap-up routing redesigned (Craig's challenge): deliver routable keepers via =inbox-send= to the destination's inbox, not a direct cross-repo =todo.org= move. Spec now Ready, [9/9] decisions.
+- ratio's diverged rulesets commit f118905 (stty-susp "fix") discarded — its premise (LAUNCH_PREFIX breaks launch) was not reproducible; backed up to a patch first.
+
+** Data Collected / Findings
+
+- A gitignore-mode .ai/ scaffold is per-machine — it doesn't travel via git, so a project bootstrapped on one machine fails the launcher's marker check on another. (Promoted to KB.)
+- inbox-send's resolve_roots REPLACES the built-in roots when ~/.claude/inbox-roots.txt exists, so parent roots must be re-listed alongside single-$HOME project dirs.
+- Several ~/code projects lacked an inbox/ dir entirely (little-elisper, rsyncshot, winvm, chime, yt-sync) — created during the conversion broadcast.
+- chime and yt-sync have inbox/ but no todo.org — which clinched the wrap-up routing redesign (direct-move would silently drop keepers headed there; inbox-route delivers).
+- stty susp undef exits 0 in a pty on both machines; the exact tmux send-keys launch path runs clean — no reproducible launch bug.
+
+** Files Modified
+
+- rulesets (committed + pushed): claude-templates/bin/ai (launcher line, f5609ec); todo-format.md + respond-to-cj-comments.md + process-inbox.org + 2 dated-header repairs (8a50088); KB-contribution spec + startup/triage-intake/inbox-zero/wrap-it-up prompts (76e5559); lint-org.el level-2-dated-header + tests + lint-org.md (f6dde4e); 3 filed-handoff tasks + design bundles (d9d3be9); wrap-up routing spec-review (23fac08), spec-response redesign (af15fae), task breakdown (df1555d).
+- ~/.dotfiles (e4a7cee): .ai/ converted to track mode. ratio: dotfiles converged, rulesets reset to origin.
+- ~/.claude/inbox-roots.txt (velox-local): added ~/.emacs.d + ~/.dotfiles so inbox-send reaches them.
+- org-roam KB: best-practices node + 2 lesson nodes (gitignored-.ai-per-machine, inbox-over-foreign-edit).
+- Handoffs sent: home + work (3 filed tasks), .emacs.d (VERIFY rule applied), and conversion handoffs to 8 ~/code projects.
+
+** Next Steps
+
+- RECOMMENDED NEXT SESSION: build the wrap-up routing feature. Spec is Ready ([[file:../docs/design/wrapup-routing-spec.org]]); 4 =:solo:= tasks under the parent at todo.org (recommendation engine + discovery, =:ROUTE_CANDIDATE:= marker in process-inbox, the wrap-it-up router sub-step, the test surface) plus 1 manual end-to-end validation. A clean fresh-session build.
+- Pending Craig review: the home "spec-response readiness-gate" proposal sitting in rulesets/inbox/ (deferred from this session — it's a proposal needing review).
+- The 8 ~/code conversion handoffs apply on each project's own next session.
+- Backlog filed this session: ntfy agent-comms, flashcard reconcile, triage-intake phone-push (all [#C], the latter two depend on decisions).
+
+KB: promoted 2 / consulted yes
+
+* Session Log
+
+** Dotfiles launcher discovery — full resolution
+
+- Root cause was two-part (launcher line missing + .ai/ gitignored so the bootstrap didn't travel). velox: applied launcher line (f5609ec), bootstrapped .ai/, then converted ~/.dotfiles to track mode (.ai/ now tracked, dotfiles commit e4a7cee) so the scaffold travels via git. velox fully working.
+- ratio: dotfiles converged to tracked .ai/ (ff-pull after removing a blocking untracked inbox/.gitkeep). But ratio's ~/code/rulesets is DIVERGED: local commit f118905 (1 ahead) vs my f5609ec (1 behind). f118905 = ratio's local fix for "stty susp undef; LAUNCH_PREFIX prevents ai launch" — but it's broken (printf split across two lines = bash syntax error), reverts C-z protection wholesale, and includes experimental "Mr. Moto" agent-id content. NOT pushable as-is. Left untouched pending Craig's decision.
+- Possible real bug flagged: LAUNCH_PREFIX="stty susp undef; " may actually break `ai` launch (ratio symptom). Needs proper verify/fix on velox canonical, separate from ratio's broken commit.
+- RESOLVED (option 1): launch bug not reproducible. Tested stty susp undef in a pty (velox + ratio, zsh + bash, coreutils 9.11) → exit 0; tested the exact tmux send-keys launch path on both → command runs (LAUNCHED-OK). f118905 misdiagnosed it (or hit a transient half-synced bin/ai on Jun 17), bundled with a broken printf split + experimental "Mr. Moto" build_instructions. Canonical needs no change; LAUNCH_PREFIX/C-z protection stays.
+- ratio fixed: backed up f118905 to ~/0001-fixing-the-stty-susp-undef-*.patch (recoverable), git reset --hard origin/main → HEAD f5609ec, launcher line present, 0/0 with origin, build_candidates lists ~/.dotfiles. Untracked ratio inbox handoffs left in place (ratio's own session concern).
+- Both machines now: ~/.dotfiles shows in the launcher picker. Done.
+
+** Processed ratio's stranded rulesets inbox (11 files, same project)
+
+Copied ratio's untracked ~/code/rulesets/inbox/ handoffs to velox, examined, none done in canonical. Three batches, all filed + preserved + originals deleted on both machines (commit d9d3be9):
+- B1 ntfy agent-comms proposal (from home, 2026-06-17): [#C] task + docs/design/2026-06-17-ntfy-agent-comms-proposal.org.
+- B2 flashcard multi-tag tooling (from work, 2026-06-17): [#C] reconcile task + docs/design/2026-06-17-flashcard-multitag-{note.md,to-anki.py,stats.py} (kept 0953 over superseded 0924).
+- B3 triage-intake auto-mode phone push (from work, 2026-06-18): [#C] task (depends on B1) + docs/design/2026-06-18-triage-intake-phone-push-{note,workflow}.org (kept 1515; 1512 identical).
+Replied to home + work confirming filed. ratio gets the todo/docs on its next launch (same project, pulled).
+
+** Processed: .emacs.d level-2 VERIFY completion rule change (commit 8a50088)
+
+.emacs.d directive: a level-2 (**) VERIFY must complete task-shaped (DONE/CANCELLED + CLOSED:), never a dated header; dated rewrites are ***+ only. Reason: a ** dated header has no keyword, so todo-cleanup --archive-done can't archive it and task-review drops it. Skeptical-reviewed → sound, agreed.
+Applied all four producer locations: claude-rules/todo-format.md (dropped the attached edited version; diff confirmed it touched only the 3 VERIFY-completion passages), .claude/commands/respond-to-cj-comments.md (3 edits + the "resolved body" nicety), claude-templates/.ai/workflows/process-inbox.org Park step (+ mirror via sync-check). Also repaired two pre-existing ** dated headers in rulesets todo.org (line 179 Phase E, line ~2698 new-personal-projects) → DONE + CLOSED. And fixed the dotfiles VERIFY I dated earlier this session (todo.org:37) → DONE [#B] + CLOSED.
+Verified: make test exit 0, sync-check clean. Replied to .emacs.d by hand-writing into ~/.emacs.d/inbox/ (inbox-send can't reach .emacs.d — no ~/.claude/inbox-roots.txt on velox). Inbox files deleted.
+
+** Fixed: inbox-send can't reach .emacs.d / .dotfiles
+
+Created ~/.claude/inbox-roots.txt (velox-local) listing ~/projects, ~/code, ~/.emacs.d, ~/.dotfiles. inbox-send's resolve_roots REPLACES defaults when the file exists, so the parent roots had to be re-listed alongside the single-$HOME project dirs. Verified: inbox-send --list now shows 10 projects incl .emacs.d + .dotfiles. Machine-local — ratio needs its own if Craig sends from there.
+
+** Investigated: KB lessons-learned strength (wrap-up write + startup read)
+
+Findings for Craig's recommendation ask: wrap-up HAS a Step-1 KB-promotion check + mandatory "KB: promoted N / consulted yes-no" receipt, but it's NOT effective — last ~5 session receipts all "promoted 0 / consulted no" (one "promoted 1"); KB's newest node is 2026-06-15 and most growth came from explicit memory-sweeps, not the wrap habit. Startup has NO KB-read step at all (only the roam *task* inbox nudge; the agent KB at ~/org/roam/agents/ is never surfaced). Existing spec docs/design/2026-06-16-encourage-kb-contribution-spec.org + DOING task todo.org:198 address the write side (4 prompts + best-practices node) but its 5 decisions are unratified AND it omits Craig's read-side ask (surfacing relevant lessons at startup). KB has 30 nodes.
+
+** Built: KB-contribution feature + read-side + lint checker (all 3 of Craig's asks)
+
+Craig approved all six decisions (5 spec + D6 new). Shipped:
+- inbox-send fix: ~/.claude/inbox-roots.txt (covered above).
+- Spec amended: docs/design/2026-06-16-encourage-kb-contribution-spec.org — 5 decisions ratified, D6 (read-side consult nudge) added + threaded through Design/Phase 2/acceptance/history, status draft→approved.
+- Best-practices KB node: ~/org/roam/agents/20260620232112-agent-kb-best-practices.org (8 sections, links how-kb-works node, defers boundary to knowledge-base.md). Committed+pushed to roam (48e50f1), findable by the documented rg.
+- 4 workflow prompts (canonical+mirror): startup.org Phase A KB-surface step + 2 Phase C nudges (D1 contribute link, D6 consult titles, both roam-gated); triage-intake + inbox-zero conditional end-of-flow capture reminders; wrap-it-up early reflection at top of Step 1 feeding the existing receipt. Commit 76e5559.
+- Lint checker: lint-org.el level-2-dated-header judgment check (custom, like org-table-standard) + 3 ERT tests + lint-org.md doc + mirrors. Commit f6dde4e.
+- Closed DONE: todo.org:200 "Encourage org-roam KB contribution" (parent DONE+CLOSED, sub-VERIFY → dated entry). Not yet committed (rides wrap-up).
+Verified: make test exit 0 (350+54+12 pytest, all bats, 36 ERT), sync-check clean. All pushed.
+
+** ~/code AI-project conversion broadcast + worktree cleanup + page
+
+Surveyed all ~/code AI projects for project-owned tooling currency (todo.org/Priority Scheme, .claude/CLAUDE.md, .ai layout). Sent tailored conversion handoffs to 7 behind projects: emacs-wttrin, archangel, website (via inbox-send); little-elisper, rsyncshot, winvm, chime (direct write — they had NO inbox/ dir, so created inbox/ + .gitkeep first, which is itself a gap fixed). Each handoff lists only that project's actual gaps. Fully-current already: auto-dim-other-buffers.el, archsetup, pearl.
+- emacs-wttrin's first send (23:37) vanished from its inbox (dir mtime 23:40, likely a live emacs-wttrin session swept it); re-delivered at 23:44, verified persists.
+- yt-sync: Craig confirmed convert it (reversing the 2026-06-12 no-todo call). Delivered the full conversion handoff (created inbox/ first; no todo.org/.claude/CLAUDE.md, old layout). 8 of 8 behind projects now handed off.
+Worktree cleanup (Craig's request, so downstream can pull rulesets): committed todo.org KB-task closure (9bcc8d1), pushed, origin 0/0, tracked tree clean. Left untracked: .ai/session-context.org (live anchor, not wrapping up) + a new home handoff (2026-06-20-2339 spec-response-readiness-gate-proposal — a proposal, left for next session's review). Paged Craig via notify --persist (23:46).
+
+** Running spec-review on wrapup-routing-spec (Craig's inbox-route challenge)
+
+Craig picked option 2 (spec-review) and raised a strong design challenge: why does the router move keepers directly into the destination's todo.org (spec D2 = atomic cross-project move helper) rather than inbox-send them to the destination's inbox/ (file-per-task or one file), letting the destination's own process-inbox file them? The inbox route reuses the sanctioned cross-project path (cross-project.md), avoids cross-repo todo.org writes (the data-loss-adjacent risk), and makes a wrong-confidence destination recoverable (receiving session rejects) instead of corrupting another tracker. Dispatched a fresh-context adversarial reviewer to evaluate this as the headline finding and write docs/design/wrapup-routing-spec-review.org. Note also surfaced this session: several ~/code projects lacked inbox/ entirely (created during the conversion broadcast), which bears on the inbox-route's "destination must have an inbox" precondition.
+
+** spec-response: wrap-up routing redesigned to inbox-send, now Ready (af15fae)
+
+Ran spec-response on wrapup-routing-spec. Accepted the review's H1 (inbox-route) + H2 (tag-at-file-time); one modify (D9: local source removal + reject-flow undo, closing the review's vague source-handling). Superseded D2/D3; added D7/D8/D9; rewrote Summary/Goals/mechanics/Phases/Acceptance for inbox-send delivery. Decisions [9/9], Status Ready. Deleted the review file. Updated todo.org:187 task body (removed its cj comment — instruction satisfied; added dated sub-entry). cj comment in the spec also removed (processed). Pending: spec-response Phase 6 (implementation-task breakdown into todo.org) on Craig's go — offered, deferrable given it's ~2am.
+
+** Session-scoped instruction
+
+Craig: always commit and push this session — don't ask for push confirmation. (2026-06-20)
+
+* Session Log
+
+** Diagnosing dotfiles-not-in-launcher
+
+Startup ran clean (no crash anchor, empty inbox). Craig reported ~/.dotfiles missing from the ai launcher list. Traced to two independent causes documented above. Craig chose to fix both.
diff --git a/.ai/sessions/2026-06-22-01-33-spec-review-fold-coverage-fix-inbox-triage.org b/.ai/sessions/2026-06-22-01-33-spec-review-fold-coverage-fix-inbox-triage.org
new file mode 100644
index 0000000..98e745e
--- /dev/null
+++ b/.ai/sessions/2026-06-22-01-33-spec-review-fold-coverage-fix-inbox-triage.org
@@ -0,0 +1,57 @@
+#+TITLE: Session Context
+#+DATE: 2026-06-21
+
+* Summary
+
+** Active Goal
+
+Process the rulesets inbox — 10 logical proposals across three groups (spec-workflow tightening, emacs-wttrin coverage/rules, home flashcard family) plus two archsetup handoffs. Built Groups A and B; filed Group C + archsetup as backlog tasks; wrapped.
+
+** Decisions
+
+- Spec review/response: incorporate the review INTO the spec. Findings are now =* Review findings= TODO tasks with a =[/]= cookie (mirroring =* Decisions=), =:blocking:= marks high-priority; the responder completes each in place. This supersedes the proposal's keep-vs-delete-review-file fork (no file at all). A1 (rerun readiness rubric on scope-expanding response) folded into the spec-response Phase 4 gate; A3 (roles explicit) + A4 (source external-dep checks) added as spec-review principles.
+- coverage-summary.el: applied the generated-package-file exclusion bugfix (=-autoloads.el=/=-pkg.el=); did NOT adopt emacs-wttrin's header rewrite (=.claude/scripts/= → tracked =scripts/=) — flagged as a separate install-location question.
+- no-attribution tightening: option 1 (documented scan discipline + exemptions, no hook). Accepted proposal changes 1-3; modified change 4 (rigid token-grep → documented Before-Committing scan) because a blanket grep false-positives on legit subject mentions, file-is-the-change commits, and private repos. Both new rules carry file-is-the-change + private-single-user-repo exemptions, framed around public/shared-remote repos.
+- Group C (flashcard) + archsetup: filed as detailed, prioritized backlog tasks rather than built (Craig paused).
+
+** Data Collected / Findings
+
+- Pre-existing committed drift in inbox-zero.org (canonical = new two-inbox version, mirror = old) — fixed by mirror sync (98ebb2f). This also satisfied the roam "inbox zero should check two places" item (already gone from roam by session end).
+- review-code (Step 1) caught that the spec refold left spec-create.org + INDEX.org still describing the old review-file convention — fixed in the same commit.
+- coverage-summary ERT tests already shipped (b46619c, 12 cs-* tests), so the "add tests" proposal was mostly pre-satisfied. Added 3 (autoloads-exclusion + =--source-files= non-recursive + =--under-dir= filter/rekey); 15 cs-* tests now.
+- Open: coverage-summary.el installs to =.claude/scripts/= (gitignored in code projects) so CI can't run =make coverage-summary= — filed [#C].
+- Roam inbox evolved mid-session: new "rulesets: multiple agent source improvements" item (naming the agent, Codex-friendly workflow wording, multi-LLM ai-term) — left for a future session.
+
+** Files Modified
+
+Committed + pushed (origin 0/0): 98ebb2f (chore: inbox-zero mirror sync), ed27e3c (refactor: spec review fold — spec-review/response/create + INDEX, canonical+mirror), fb86736 (fix: coverage-summary autoloads exclusion), 0751b3c (test: coverage-summary source-files + under-dir), 91217d9 (docs: no-attribution tooling-path tightening — commits.md + protocols.org).
+
+Wrap-up commit (this session's tail): todo.org (6 new tasks), 6 docs/design files (anki-titlefix bundle, apkg buildreq, refutation proposal, host-identity proposal), session archive.
+
+Handoffs sent: home (spec-workflow reconcile; flashcard items filed), emacs-wttrin (coverage cluster; no-attribution tightening), archsetup (host-identity filed).
+
+** Next Steps
+
+- Flashcard cluster coordination: "Anki deck name from #+TITLE" [#B, ready code], "Reconcile flashcard multi-tag tooling" (:315), and "flashcard-stats refutation mode" [#C] all edit flashcard-to-anki.py / flashcard-stats.py — build together to avoid conflicting edits.
+- apkg → org-drill converter [#C], host-identity guard [#C], coverage-summary install location [#C], warn-only enum hook [#D] — all filed.
+- New roam item "rulesets: multiple agent source improvements" awaits a future session.
+
+KB: promoted 0 / consulted yes (session was process/workflow edits — no durable cross-project facts to promote).
+
+* Session Log
+
+** Startup + inbox triage
+
+Clean startup (no crash anchor). Inbox held 12 files / 10 logical proposals + (mid-session) 2 archsetup handoffs. Grouped into A (spec-workflow), B (emacs-wttrin), C (flashcard). Walked group by group per Craig.
+
+** Group A — spec review folded into the spec
+
+Read both canonical workflows + diffed home's edited spec-review.org. Four candidate changes (A1 readiness rerun, A2 review-file retention, A3 roles-explicit, A4 source-checks). Craig chose to incorporate the review into the spec (option 1), reusing the decisions-as-tasks machinery. Applied all edits to spec-review.org + spec-response.org; review-code caught spec-create.org + INDEX.org stragglers (fixed). make test green ×2. Commits 98ebb2f + ed27e3c, pushed. Replied to home; deleted 3 inbox files.
+
+** Group B — coverage-summary cluster + no-attribution tightening
+
+Found canonical coverage-summary.el is the elisp bundle and tests already existed. Applied the autoloads bugfix via TDD (red→green, fb86736) + 3 tests (0751b3c). Did not adopt the header relocation — flagged install-location as a follow-up. No-attribution tightening: option 1, edited commits.md (2 spots) + protocols.org item 3 (91217d9). Replied to emacs-wttrin twice; deleted 3 inbox files.
+
+** Pause + inbox zero
+
+Filed Group C (3 flashcard) + archsetup host-identity as 4 backlog tasks, plus 2 session follow-ups (coverage-summary location, enum hook) — 6 tasks total under Rulesets Open Work, properly prioritized. Preserved the anki edited code + all proposals in docs/design/. Replied to home + archsetup; project inbox empty. Then wrapped.
diff --git a/.ai/sessions/2026-06-23-22-36-inbox-guard-bash-bundle-consolidation-spec.org b/.ai/sessions/2026-06-23-22-36-inbox-guard-bash-bundle-consolidation-spec.org
new file mode 100644
index 0000000..19e9a2c
--- /dev/null
+++ b/.ai/sessions/2026-06-23-22-36-inbox-guard-bash-bundle-consolidation-spec.org
@@ -0,0 +1,164 @@
+#+TITLE: Session — inbox guard, bash bundle, agent-neutral rules, inbox-consolidation spec
+
+* Summary
+
+** Active Goal
+
+Started by processing two inbox handoffs, then ran long across a queue of work:
+two handoff fixes, a new bash bundle, item 1-2-3 from the roam inbox, and a spec
+for consolidating the inbox workflows. Five commits landed; the inbox-consolidation
+spec reached Ready for a next-session build.
+
+** Decisions
+
+- Item 1: build the guard as a shared, testable script (.ai/scripts/), not
+ inline in the workflow. Fix the handoff's malformed lisp (#(quote …) →
+ #'buffer-name). Add an emacs.md cross-reference.
+- Item 2 (Craig's design answers, all recommendations): (Q1) language-neutral
+ default *fallback* template install-lang seeds when the bundle ships none —
+ per-bundle templates still win where present; (Q2) the default names *no*
+ language — Craig fills it in; (Q3) inbox-zero's wrap-up roam sub-step, on a
+ live-capture collision, *skips that sub-step non-blocking* and finishes the
+ rest of wrap-up.
+- Bash/shell language bundle: built this session (shellcheck the gate, shfmt left
+ advisory since shell has no canonical style). Confirmed the five bundles cover
+ Craig's ecosystem — no further bundles warranted.
+- Item 1 (inbox-zero empty sweep): built. Item 2 (agent-source): claude-rules/
+ neutralized + .emacs.d note sent; the workflow sweep parked behind consolidation
+ (don't neutralize files about to be merged). Item 3 (wrap-teardown): filed, not
+ built (open decisions).
+- Inbox consolidation: spec it before building (load-bearing 3→1 merge of synced
+ workflows). Engine shape = Option A (one inbox.org, process/monitor/roam modes).
+ "auto inbox zero" = interactive /loop only in v1; fully-unattended /schedule
+ cron pass deferred to vNext (Codex finding, narrowed).
+
+** Data Collected / Findings
+
+- Verified handoff #2 against the tree: claim "only elisp ships CLAUDE.md" is
+ stale — go ships one too; python + typescript ship none; install-lang guards
+ on [ -f "$SRC/CLAUDE.md" ]. Core gap holds.
+- Only inbox-zero.org Phase D writes the roam inbox on disk (startup reads,
+ wrap-up delegates to inbox-zero). So one guard call covers it.
+- Canonical/mirror: claude-templates/.ai/ is canonical; .ai/ root is the mirror
+ kept in sync by scripts/sync-check.sh (pre-commit enforced). Edit canonical,
+ then sync --fix.
+
+** Files Modified
+
+Item 1 (capture guard): NEW claude-templates/.ai/scripts/capture-guard + its
+bats test; edited claude-templates/.ai/workflows/inbox-zero.org (Phase D guard
+step, before the pull); edited claude-rules/emacs.md (new "don't edit on disk a
+file the daemon is capturing into" section). Mirror synced to .ai/.
+
+Item 2 (CLAUDE.md fallback): NEW languages/default-CLAUDE.md (neutral, names no
+language); edited scripts/install-lang.sh (fall back to default when bundle
+ships none); Makefile LANGUAGES glob hardened to dirs-only; scripts/lint.sh
+lints the default; scripts/tests/install-lang.bats +3 tests.
+
+Housekeeping: bash bundle filed todo.org [#C]; both handoffs preserved to
+docs/design/ (2026-06-22-inbox-zero-capture-hardening.org,
+2026-06-23-install-lang-claude-md-gap.org); TODO link repointed; replies sent to
+home + archangel; notes.org :LAST_INBOX_PROCESS: stamped 2026-06-23.
+
+Second half: NEW languages/bash/ bundle (rules, validate-bash.sh shellcheck hook
++ 8 bats tests, pre-commit githook, settings, CLAUDE.md), Makefile glob for
+languages/*/tests/*.bats, README + notes bundle-list fixes. inbox-zero.org Phase
+B/D empty-sweep. claude-rules/ (interaction, cross-project, working-files,
+triggers) agent-neutralized. NEW docs/inbox-workflow-consolidation-spec.org
+(Ready). todo.org: bash bundle DONE, + filed wrap-teardown [#B], inbox-empty
+[#C], agent-source [#C], consolidation [#B], unattended-cron [#D].
+
+Commits this session: 603abc4 (capture-guard), 71db71b (install-lang fallback),
+3626285 (bash bundle), 3da2725 (inbox empty sweep), 6ad0442 (agent-neutral rules).
+Roam repo: 436d646 (route rulesets tasks). All un-pushed until this wrap.
+
+Verification: full make test exit 0 (200 ok, 0 not-ok); lint clean except a
+pre-existing remove.sh chmod warning (untouched); capture-guard lisp eval'd in the
+live daemon; bash hook + githook dogfood shellcheck-clean.
+
+** Next Steps
+
+Build the inbox consolidation from the Ready spec (docs/inbox-workflow-consolidation-spec.org):
+Phase 1 author inbox.org, Phase 2 reconcile callers + retire the 3 old files +
+stale-ref grep, Phase 3 auto-inbox-zero, Phase 4 verify. Then the parked agent-
+neutrality workflow sweep (over the consolidated file), then item 3 (wrap-teardown,
+needs Craig's 3 open decisions). Queue tasks #5, #6, #7 carry these.
+
+KB: promoted 0 / consulted no
+
+* Session Log
+
+** Committed both fixes (603abc4, 71db71b)
+
+Two feat commits landed as Craig: capture-guard (603abc4) and install-lang
+neutral CLAUDE.md (71db71b). Author verified c@cjennings.net. Push deferred to
+wrap-up. notes.org marker + session-context.org stay for the wrap commit.
+
+Built the bash/shell bundle (languages/bash/): rules (bash.md, bash-testing.md),
+shellcheck validate hook (8 bats tests, handles extensionless shell via shebang,
+caught a real path-dot bug during TDD), shellcheck pre-commit githook,
+settings.json, gitignore-add.txt, "Bash/shell project" CLAUDE.md. Makefile test
+target extended to discover languages/*/tests/*.bats. README bundle table + notes
+bundle list corrected (were stale at elisp-only / elisp+python). bash-bundle TODO
+closed DONE. Full make test exit 0 (200 ok), lint clean, hook+githook dogfood
+shellcheck-clean, installs + fingerprint-detects correctly.
+
+Committed bash bundle (3626285). Awaiting Craig's call on further bundles (I
+recommended none — the five cover his visible ecosystem).
+
+** Inbox zero (local + roam)
+
+Local inbox: a new home handoff arrived mid-session — wrap-it-up teardown +
+"wrap it up and shutdown" (Craig's own design). Filed [#B] :feature:, preserved
+to docs/design/2026-06-23-wrap-teardown-shutdown-proposal.org, replied to home,
+NOT built (shared-asset + open decisions). Companion cj/ai-term-quit routes to
+.emacs.d when built.
+
+Roam inbox (11 items, 2 rulesets-claimed): imported both to rulesets todo.org —
+"inbox-zero: delete empty roam entries on triage" [#C], and "Multiple
+agent-source improvements" [#C] :spec: (naming the agent, agent-neutral rule
+source / Codex review, multi-LLM ai-term note to .emacs.d). Removed both from the
+roam inbox via the new capture-guard (exit 0, safe) + pull, committed the roam
+repo (436d646). 9 foreign items left untouched (emacs, pocketbook, archsetup x4,
+chime, emacs-wttrin, pearl). No empty entries existed to delete this pass.
+
+Pending push: rulesets 4 commits ahead (603abc4, 71db71b, 3626285 + the inbox
+filing), roam 1 commit (436d646, roam-sync timer will push). Hold for wrap-up.
+
+** Startup + inbox triage
+
+Ran startup. Clean prior wrap-up (no session-context). 16 stale tasks, 11 roam
+inbox items (2 rulesets-related). Two pending inbox handoffs, both shared-asset
+change proposals → skeptical review, surfaced to Craig. Craig chose to implement
+both now (bash bundle filed separately) and answered the three design questions
+with all recommendations.
+
+** Items 1-2 + the consolidation pivot to a spec
+
+Craig set up a visible task queue (TaskCreate #1-#7) and worked items 1, 2, 3 in
+order. Item 1 (inbox-zero empty-entry sweep): added a Phase B "empty" bucket +
+Phase D removal so aborted/blank roam headings get swept every triage; committed
+3da2725. Item 2 (agent-source): thread 3 (.emacs.d multi-LLM ai-term note) sent;
+claude-rules/ neutralized (7 agent-as-actor "Claude" → "the agent" edits, commits.md
+correctly untouched) committed 6ad0442; the 45-file workflow-neutralization sweep
+PARKED.
+
+Mid-flow Craig dropped a roam item — "consolidate inbox workflows, there's too
+many" + how to schedule inbox checks. Routing a 45-file neutrality sweep over
+workflows about to be consolidated would be wasted, so parked #5 behind a new
+consolidation task #7, routed the roam item to todo [#B], and mapped the inbox
+landscape (3 overloaded surfaces: local inbox/ dir → process-inbox+monitor-inbox;
+roam → inbox-zero; external → triage-intake). Proposed 3→1 engine; Craig chose to
+SPEC it.
+
+Ran the spec trio: spec-create authored docs/inbox-workflow-consolidation-spec.org
+(Option A: one inbox.org engine, process/monitor/roam modes, shared core);
+Craig resolved all 4 decisions and specified "auto inbox zero" (interactive /loop,
+ask-interval, acknowledge-only-on-empty, find → summarize → file → displayed queue
+→ ask-to-execute, cross-cycle dedup). Codex reviewed (2 findings, 1 blocking).
+spec-response folded both: finding 1 (unattended behavior unspecified) accepted via
+narrow — v1 = interactive auto inbox zero only, fully-unattended /schedule cron pass
+deferred to vNext [#D] with its open contract questions; finding 2 (checker doesn't
+validate workflow links) accepted — added an explicit stale-reference grep as an
+acceptance item, dropped the over-claim. Spec Status → Ready, Decisions [4/4],
+Findings [2/2]. The build (Phase 1 author inbox.org) is the next session's work.
diff --git a/.ai/sessions/2026-06-24-00-14-inbox-consolidation-wrap-teardown-roam-fix.org b/.ai/sessions/2026-06-24-00-14-inbox-consolidation-wrap-teardown-roam-fix.org
new file mode 100644
index 0000000..084a233
--- /dev/null
+++ b/.ai/sessions/2026-06-24-00-14-inbox-consolidation-wrap-teardown-roam-fix.org
@@ -0,0 +1,85 @@
+#+TITLE: Session — inbox consolidation, chime fix, wrap-teardown, roam-sync fix
+
+* Summary
+
+** Active Goal
+
+Continuation past an earlier wrap (Craig chose to keep going). Ran a "1 then 2
+then 3" sequence, then a follow-on fix and a recurring-loop setup. All shipped
+and pushed; ended on a clean wrap. No open work item carried forward.
+
+** What shipped (all pushed to origin/main)
+
+1. *Inbox consolidation* (24ca58d). Merged process-inbox + monitor-inbox +
+ inbox-zero into one =inbox.org= engine: shared core (value gate, skeptical
+ review, disposition ladder, reply discipline, capture-guard, priority-scheme)
+ + process/monitor/roam/auto modes. Repointed every caller (INDEX, protocols,
+ startup, wrap-up, triage-intake, broadcast, two script comments, two
+ claude-rules files), deleted the three old files. Built from the Ready spec
+ (all 4 phases). Closed todo.org [#B] consolidation + [#C] empty-sweep; the
+ fully-unattended /schedule pass stays the [#D] vNext task.
+2. *Chime validate-el.sh fix* (e5aab19). Added the one-line =(cd tests/)= before
+ the Phase 2 ERT load in the canonical elisp hook — restores regression b2e9038
+ lost when .claude refreshed to canonical. Verified identical to chime's diff +
+ shellcheck-clean; replied to chime.
+3. *Wrap-teardown rulesets side* (f87f59c) + *Stop-hook wiring* (96cd34f).
+ Craig's decisions: both summary qualifiers ("with summary" / "and summarize"),
+ Emacs-timer countdown, cj/ai-term-live-count gate. Built hooks/ai-wrap-
+ teardown.sh (Stop hook, sentinel-gated, 8 bats green), settings-snippet +
+ live .claude/settings.json Stop block, wrap-it-up Teardown-mode section +
+ Step 6 + checklist, INDEX. Companion spec (cj/ai-term-quit, -live-count,
+ -shutdown-countdown) routed to .emacs.d; it confirmed receipt + filed it.
+4. *Roam-sync fix* (f83d4bb). Roam mode no longer git-pulls the chronically-dirty
+ roam repo — the scan reads the working tree, the rare write edits + triggers
+ roam-sync (which commits-first-then-rebases). Fixes the loop failing every
+ cycle on a dirty tree.
+
+** Decisions
+
+- Wrap-teardown: both non-destructive qualifiers accepted; Emacs run-at-time
+ countdown; cj/ai-term-live-count safety gate (Craig, 2026-06-23).
+- Roam triage hands git to roam-sync rather than pulling (Craig picked option 1,
+ 2026-06-24). Trade-off accepted: generic roam-sync commit message; provenance
+ lives in todo.org + session log.
+- This wrap is a NORMAL wrap, not the new teardown — that feature isn't
+ operational until the Stop hook activates next session and the .emacs.d
+ companion lands.
+
+** Open / carryover
+
+- *wrap-teardown task is DOING*, blocked on: (c) .emacs.d lands the three
+ companion functions (handoff in its inbox, confirmed received); (d) the manual-
+ validation checklist under the task in todo.org (needs Craig's live Emacs/tmux).
+ Install steps (a) hook symlink + (b) settings Stop block are done.
+- *Deferred dirty (Craig's call):* docs/design/2026-06-15-fix-speedrun-workflow-
+ proposal.org — his in-flight "fix speedrun" → "speedrun" rename, only in that
+ file; ~20 references remain elsewhere. Left untouched again.
+- *12 top-level tasks unreviewed >30 days* — the daily task-review keeps slipping;
+ a task-review cycle is overdue.
+- The one :quick: task (token-rotation helper) is held by decision until a real
+ token rotation is imminent (can't verify without a live browser re-grant).
+
+KB: promoted 0 / consulted no. The durable lessons this session (don't pull the
+roam repo during triage; Stop-hook sentinel decoupling for teardown) landed in
+the canonical workflows + hook themselves, so a KB node would only duplicate what
+the repo now records.
+
+* Session Log
+
+** 2026-06-23 Tue @ 22:52:33 -0400 — flushed
+Clean boundary after the earlier wrap/push. Resume = build the inbox consolidation from the Ready spec, then wrap-teardown.
+
+** 2026-06-23 Tue @ 23:07:23 -0400 — inbox consolidation built
+Built the full inbox.org engine from the Ready spec (all four phases): shared core + process/monitor/roam/auto modes, absorbing the three old workflows. Repointed every live caller, deleted the old files, synced the mirror. Verified: workflow-integrity OK both trees (42 workflows), stale-ref grep clean, sync-check clean, make test green. /review-code → Approve, /voice personal on the message. Pushed 24ca58d. Closed [#B] consolidation + [#C] empty-sweep; [#D] /schedule stays vNext.
+
+** 2026-06-23 Tue @ ~23:15 -0400 — chime validate-el fix
+Processed the chime handoff (inbox.org process mode): applied the one-line Phase 2 cd-to-tests fix to the canonical elisp hook, verified identical to chime's diff + shellcheck-clean. Pushed e5aab19, replied to chime, cleared the inbox.
+
+** 2026-06-23 Tue @ ~23:30 -0400 — wrap-teardown rulesets side
+Craig decided the three open questions. Built the Stop hook (8 bats green, shellcheck clean), settings wiring, wrap-it-up Step 6 + Teardown-mode section + checklist, INDEX. Pushed f87f59c. Routed the cj/ai-term-* companion spec to .emacs.d. On Craig's "why can't you do install steps" push: wired the live .claude/settings.json Stop block (it's a tracked repo file) + ran make install-hooks — pushed 96cd34f. .emacs.d confirmed receipt + filed the companion.
+
+** 2026-06-24 Wed @ ~00:00 -0400 — auto inbox zero loop + roam-sync fix
+Set up the auto inbox zero /loop (cron, every 10 min). First two cycles found only a .emacs.d FYI (companion received) + nothing for rulesets in roam. The roam pull failed on a dirty tree; root-caused it (constant captures + 15-min roam-sync timer = chronically dirty) and Craig picked the full fix (option 1): roam mode never pulls — read-only scan + edit-then-trigger-roam-sync. Pushed f83d4bb, replaced the loop prompt (job a37f53bc), then stopped the loop on Craig's go.
+
+** 2026-06-24 Wed @ 00:14:02 -0400 — wrap
+Stopped the loop. Checked todo.org: nothing speedrunnable (the one :quick: task is held by decision; the rest are substantive specs/features or blocked DOING). Ran a normal wrap (teardown feature not operational this session). todo-cleanup archived 2 done subtrees, lint reformatted one table, inbox clean, roam sweep a no-op.
diff --git a/.ai/sessions/2026-06-24-09-27-task-audit-blocked-deps-anki-wrap-teardown.org b/.ai/sessions/2026-06-24-09-27-task-audit-blocked-deps-anki-wrap-teardown.org
new file mode 100644
index 0000000..b0a4994
--- /dev/null
+++ b/.ai/sessions/2026-06-24-09-27-task-audit-blocked-deps-anki-wrap-teardown.org
@@ -0,0 +1,75 @@
+#+TITLE: Session — task audit, blocked/blocker deps, Anki fix, wrap-teardown unblocked
+
+* Summary
+
+** Active Goal
+
+Long post-wrap continuation (past the 00:14 wrap). Ran a task audit, built two
+task-workflow features (cross-project dependency tags + audit consolidation),
+shipped the Anki #+TITLE fix, and unblocked the wrap-teardown feature when the
+.emacs.d companion landed. Ended on a wrap + a live test of the wrap-teardown
+workflow. 15 rulesets commits pushed (5cdbf13..9709638) plus a roam sync.
+
+** What shipped (all pushed to origin/main)
+
+- *Roam sync* — committed + pushed the dirty roam tree via roam-sync (65514c2).
+- *Task audit, Phases A-F* (5cdbf13): reconciled 23 open tasks (19 current, 2
+ stale fixed, 2 VERIFY flags). Resolved the helper-instance dependency question
+ to a buildable TODO (cc93fa8). Memory-sync VERIFY parked. Added the
+ =daily-drivers.md= rule (ratio/velox machine-sync, 03ad150). Chained a
+ task-review pass: stamped 12 never-reviewed tasks, tagged two :quick:solo:
+ (558624e).
+- *"session wrapped." signoff* (d5cc37c): wrap-it-up valediction now ends with
+ =session wrapped.= on its own line. (Your roam-inbox request.)
+- *capture-guard --wait* (1eaec82): poll mode so a transient org-capture clears
+ itself instead of bouncing a roam edit; roam mode + auto-loop fall back only
+ after the wait. 3 new bats cases.
+- *Cross-project dependency tags* (4d2f83d, 0d87c80, then 06b6cbc + 9709638 for
+ the bidirectional + tag-form revision): =:blocked:= on the waiting task,
+ =:blocker:= on the task that owes the work, detail in the body (no property).
+ Setting :blocked: requires a reciprocal inbox-send so the blocker learns;
+ open-tasks surfaces :blocker: first and pulls :blocked: out of the cascade.
+ Global (todo-format.md + open-tasks.org + inbox.org). The two filed
+ task-mgmt ideas (6de1712) that spawned this are DONE.
+- *Task-audit consolidation* (bcfce0e): Phase C.5 proposes merge-or-parent for
+ related-task clusters.
+- *Anki #+TITLE fix* (060a938, closed 3b48416): default_deck_name reads the
+ org #+TITLE, not the filename. TDD red->green, 29 pass. Coordination note left
+ on the flashcard multi-tag task (its preserved file now predates this fix).
+- *wrap-teardown unblocked* (0127889): .emacs.d landed the three ai-term
+ companion functions (double-checked the bodies — they match + exceed the
+ contract: TOCTOU re-check, configurable shutdown command). Dropped :blocked:.
+
+** State of the wrap-teardown feature
+
+Code-complete on both sides (rulesets Stop hook + wrap-it-up Step 6; .emacs.d
+companion + 13 ERT tests). The feature is ARMED: a bare "wrap it up" now tears
+the session down, "wrap it up and shutdown" powers off after the gate. The only
+remaining item is the manual end-to-end validation (the checklist under the
+task). IMPORTANT: the Stop hook was wired mid-session, and the harness loads
+hooks at session start — so the teardown fires reliably from the NEXT session,
+or this session only after =/hooks= is opened once. Don't drop a teardown
+sentinel blind, or it misfires on the next session's first stop.
+
+** Open / carryover
+
+- wrap-teardown: DOING, manual validation pending (your env). Feature armed.
+- Wrap-up inbox/transcript routing: DOING, spec Ready, 5 sub-tasks; the
+ recommendation-engine sub-task (:solo:) is the clean entry point. Craig may
+ pick this or another up next session.
+- fix-speedrun proposal: still the deferred dirty file (docs/design/2026-06-15),
+ untouched.
+- flashcard multi-tag task: re-derive against the post-Anki-fix canonical.
+
+KB: promoted 0 / consulted no. Durable lessons (the blocked/blocker convention,
+the roam-no-pull rule, capture-guard --wait) all landed in the synced rules +
+workflows themselves, so a KB node would duplicate the repo.
+
+* Session Log
+
+** 2026-06-24 Wed @ 09:27 — wrap + wrap-teardown live test
+Closed out a long continuation: task audit, the blocked/blocker dependency
+feature (built property-based, then refactored to plain tags on Craig's call),
+the Anki #+TITLE fix, and the wrap-teardown unblock. todo-cleanup archived 2
+done subtrees (Anki, Morning-ops cancel); lint reflowed a table; roam sweep +
+inbox both clean. Wrapping to test the now-armed wrap-teardown workflow live.
diff --git a/.ai/sessions/2026-06-28-15-57-inbox-proposals-shipped-and-task-audit.org b/.ai/sessions/2026-06-28-15-57-inbox-proposals-shipped-and-task-audit.org
new file mode 100644
index 0000000..bf67b8e
--- /dev/null
+++ b/.ai/sessions/2026-06-28-15-57-inbox-proposals-shipped-and-task-audit.org
@@ -0,0 +1,278 @@
+#+TITLE: Session Context
+#+DATE: 2026-06-28
+
+* Summary
+
+** Active Goal
+
+Startup → processed all 11 inbox handoffs (7 shared-asset proposals, skeptical-
+reviewed via parallel subagents, walked A–G with Craig), then two follow-ups
+(code-quality umbrella workflow + a task review), then a full task audit, then
+wrap. Everything shipped is committed + pushed to origin/main.
+
+** Decisions
+
+- A simplification mode added to /refactor, and Craig chose to include it in the
+ default full scan (not own-mode-only).
+- locating-craig.md is a standalone rule (not folded into daily-drivers.md).
+- suspend.org built standalone-but-lean (not folded into flush), always-commit
+ step dropped.
+- Commit gate: prose tightening AND a hard PreToolUse deny on bundled
+ test+commit (Craig picked the hook backstop over prose-only).
+- Bug-priority matrix is BINDING for any project with a codebase (not opt-in);
+ mapping P1→[#A], P2→[#B], P3→[#C], P4→[#D]; home + work notified and work has
+ adopted it.
+- Dot-stripped project names: alias approach (exact match still wins).
+- readability-audit kept separate from /refactor (it feeds /refactor by filing
+ :refactor: tasks).
+- Wrap done as NORMAL wrap, not teardown — the teardown feature is unvalidated
+ (its manual test was deferred this session), so no teardown sentinel dropped.
+
+** Data Collected / Findings
+
+- Task audit verdict: contrary to "many shipped," NONE of the 21 open tasks are
+ fully-done-but-open. The two closest (wrap-teardown, memories-sync) are
+ code-complete, gated only on Craig's manual validation.
+- The git-commit-confirm hook's deny path: `** TODO` is a substring of
+ `*** TODO`, so Edit old_strings on demoted headings need a leading-newline
+ boundary to disambiguate.
+- route_recommend matching: word-boundary literal match avoids home/homeowner
+ false positives; weak matching on common-word names (home, work) can
+ over-route — accepted v1 risk (labeled weak, reject-flow recovers).
+
+** Files Modified (all committed + pushed)
+
+- b621914 .claude/commands/refactor.md — simplification mode (later folded into full scan, 96dfa63 era edits)
+- d4e9d7d claude-rules/locating-craig.md
+- 797c426 suspend.org + readability-audit.org (+ INDEX, protocols)
+- 92dfc35 verification.md + commits.md + hooks/{git-commit-confirm,_common}.py + tests (bundled-test deny)
+- 9753d03 triggers.md + inbox-send.py (+ mirror, tests) — dot-stripped names
+- 798ef02 claude-rules/todo-format.md + docs/design bug-priority bundle — binding matrix
+- 6fb6797 notes.org inbox marker
+- 96dfa63 code-quality.org umbrella workflow (+ INDEX)
+- 5263cd6 todo.org — task review (3 restamped, generic-runtime → [#D])
+- 6be62ae .ai/scripts/route_recommend.py (+ mirror, tests) — wrap-up routing recommendation engine (spec Phases 1+3)
+- 749566c todo.org — task audit reconciliation + flashcard tooling cluster
+
+** Next Steps — PICK UP HERE
+
+Phase D of the task audit was started but only 1 of 5 items handled. Remaining,
+in suggested order:
+
+1. *wrap-teardown (task 42) manual validation — DEFERRED today.* Feature shipped
+ + pushed both sides; only the 5-test checklist remains (in the task body).
+ HAZARDS when running: test 1 tears down whatever session runs it (use a
+ SCRATCH ai-term session, never the live one); test 4 powers off (stub
+ `sudo shutdown now` → echo first); each "wrap" test ends its session so they
+ don't chain. On all-pass close task 42 DONE + CLOSED:; a failure becomes a bug.
+2. *memories-sync VERIFY (task 214).* Implementation fully shipped; can't close
+ until (a) its manual-validation child runs and (b) ratio gets the one-time
+ roam.git clone + roam-sync timer (velox confirmed; ratio still outstanding,
+ can't verify from velox). See daily-drivers.md.
+3. *Spec storage location + lifecycle convention (task 362).* Stalled on one
+ decision: filename-suffix vs org-keyword for lifecycle status. Needs Craig's call.
+4. *"fix speedrun" autonomous-batch (task 383, DOING).* Stalled at spec-review;
+ needs Craig to ratify (or re-park) the 6 open spec decisions before building
+ work-the-backlog.org.
+5. *Tooling-path warn-hook (task 434, [#D]).* Craig chose docs-only before;
+ greenlight building the warn-only hook, or leave [#D].
+
+Also queued (not Phase D):
+- Wrap-up routing feature (task 133, DOING): engine landed (6be62ae). Next
+ sub-tasks under it: :ROUTE_CANDIDATE: marker in inbox process mode, the
+ wrap-it-up router sub-step, the test surface, then manual e2e validation. All
+ call route_recommend.py. Spec: docs/design/wrapup-routing-spec.org.
+- Flashcard tooling cluster (task parent created this session): apkg converter,
+ refutation (generic header-exemption per cj), multi-tag reconcile — build
+ together, same scripts; re-derive against the post-#+TITLE-fix canonical.
+
+KB: promoted 0 / consulted no
+
+* Session Log
+
+** 2026-06-28 Sun — Startup + inbox triage
+
+Ran startup (Phase A.0/A/B). Clean prior wrap (no session-context.org). .ai/
+synced from templates fine. Findings: 11 pending inbox handoffs, 3 top-level
+tasks unreviewed >7 days, roam inbox 2 items, KB 51 nodes (none relevant).
+
+The 11 inbox items resolve to ~7 substantive proposals (some carry a cover
+note):
+- A: Simplification lens for the refactor skill (.emacs.d)
+- B: new rule locating-craig.md (home) + cover note
+- C: new workflow suspend.org (.emacs.d) + cover note
+- D: bug-priority severity×frequency matrix (wttrin) + cover note
+- E: harden commit gate to require green full suite (wttrin)
+- F: generalize readability-audit.org into a template workflow (.emacs.d) + cover
+- G: strip dots from project names .emacs.d→emacsd (.emacs.d roam item)
+
+All are shared-asset/convention changes → skeptical review + Craig approval, none
+self-applies. Dispatching parallel read-only skeptical-review subagents for A-F;
+G is a lightweight file-as-task. Surfacing dispositions to Craig next.
+
+Six skeptical reviews returned (all "do it with named changes"):
+- A: keep only new lenses 1/2/3/4/7; cross-ref 5(dead-code)/6(duplication)/8(dead-code)
+ instead of re-specifying; own mode, NOT in default full scan; state /refactor↔/simplify
+ boundary. Lands in .claude/commands/refactor.md (it's a slash command, no refactor/ dir).
+- B: standalone claude-rules/locating-craig.md. Changes: add whereami-fails fallback
+ (never fabricate), positive gate (velox→run else ask), drop -lts secondary tell,
+ add public-artifact-scope line (geolocated address is personal). whereami confirmed
+ real at ~/.local/bin/whereami; host is velox.
+- C: REAL DESIGN FORK — substantial overlap with the flush skill (suspend ≈ flush's
+ pre-clear checkpoint minus /clear, plus a commit, reframed for departure). Reviewer
+ prefers folding into flush as a "checkpoint-and-stop" variant over a standalone
+ workflow. Also: drop the always-commit-working-files step (.emacs.d-ism; blind commits
+ fight wrap-it-up's leftover discipline). Startup's resume path already exists.
+ Companion: INDEX + protocols triggers if kept standalone.
+- D: fold into todo-format.md as an OPTIONAL opt-in scheme; drop the global
+ letter-mapping mandate (present as template); keep security/privacy carve-out; mark
+ release-vehicle column illustrative. Defer the non-coding companion matrix as a
+ separate task.
+- E: only the new mechanic is load-bearing (run suite as own command, gate on
+ failures==0, never bundle `make test; git commit`). Two-line edits to verification.md
+ #1 and commits.md #5; exact wording provided. Asks #2/#4 already covered.
+- F: generalize Elisp specifics; drop concrete ;;; markers (keep principle); add
+ graceful degradation; add cross-links to /refactor + /simplify + A. Coherent division
+ of labor confirmed (readability = prose/naming clarity; /refactor = structural metrics;
+ /simplify + A = behavior-preserving logic reduction). Lands as a template workflow.
+- G: file-as-task (strip dots from project names; touches triggers.md + inbox-send).
+
+*** Item A — APPLIED (not yet committed)
+Craig approved option 1, then changed his mind: simplification IS part of the
+default full scan. Added Mode:
+Simplification to .claude/commands/refactor.md (frontmatter Six→Seven modes,
+argument-table row, new section after Mode: Dead Code) with the 5 new lenses,
+cross-refs to Dead Code (twin branches, plain deletion) + Duplication (repeated
+literals), verify-all-call-sites rule, and a /simplify boundary note. Harness
+picked up the skill change live mid-session. Confirmed for Craig: Rename mode
+executes renames but doesn't flag bad names; no organization scan exists — both
+gaps are proposal F's territory.
+
+*** Item B — APPLIED (not yet committed)
+Craig approved. Wrote claude-rules/locating-craig.md (standalone) with the four
+review changes: whereami-fails fallback (never fabricate), positive host gate
+(velox→run, any other host→ask), dropped the -lts secondary tell, added a
+keep-out-of-shared-artifacts section. make install linked it into
+~/.claude/rules/locating-craig.md.
+
+*** Item C — APPLIED (not yet committed)
+Craig picked option 1 (standalone, lean). Wrote
+claude-templates/.ai/workflows/suspend.org with the review changes: drop the
+always-commit step (note uncommitted work, leave tree as-is; project-opt-in
+always-commit set only), cross-refs to flush + wrap-it-up, states it's the
+capture half (startup is the resume half), flags "I need to go" breadth.
+Registered in INDEX.org (Session lifecycle, after wrap-it-up) + protocols.org
+trigger section. sync-check --fix synced canonical→mirror; re-verified exit 0,
+suspend.org mirror matches.
+
+*** Item D — REVERTED then RE-APPLIED (binding), not yet committed
+First applied an opt-in version; Craig reverted it ("do it differently"). His
+intent: the matrix is BINDING, not opt-in — any project with a codebase (incl.
+home + work, which have one despite being non-code) must prioritize its codebase
+bugs by the matrix. Re-applied to claude-rules/todo-format.md as a mandatory
+subsection. Mapping per Craig (2a): P1→[#A], P2→[#B], P3→[#C], P4→[#D] (fixed,
+not a per-project knob). Bands defined per codebase; matrix structure + mapping
+fixed. Severity-alone carve-out kept. Sent adoption handoffs to home + work
+(inbox-send, 2026-06-28-1212). Non-coding companion matrix dropped — scope is
+codebase bugs (home/work codebases covered).
+
+*** Item E — APPLIED (not yet committed)
+Craig picked option 1 (prose tightening + bundling-detection hard gate in the
+PreToolUse hook); asked first whether a hard gate existed — it didn't (githooks
+pre-commit only runs sync-check; git-commit-confirm.py only scanned attribution).
+Applied: verification.md "Before Committing" #1 and commits.md #5 rewritten to
+"run the full suite as its own command, gate on zero failures, never bundle the
+run with the commit." Added detect_bundled_test_run() + respond_deny() to the
+hook (hooks/git-commit-confirm.py + hooks/_common.py): denies a test runner
+chained into git commit via any ungated connector (;, &, |, ||, newline, or a
+pipe that masks exit), allows the gated && form, matches the runner only in the
+prefix before git commit so a runner name in the message doesn't trip it. TDD:
+13 new tests red→green; full make test exit 0; end-to-end smoke test confirms
+deny on bundled / pass on gated+plain.
+
+*** Item F — APPLIED (not yet committed)
+Craig asked "should this be part of refactoring?" — concluded separate-but-linked
+(it's a multi-phase workflow that FILES structural work as :refactor: tasks, i.e.
+feeds /refactor rather than being a mode of it; /refactor is structure-only
+scan-and-apply). Craig approved option 1. Wrote
+claude-templates/.ai/workflows/readability-audit.org (generalized from .emacs.d's
+Elisp draft: header convention / public-private naming / doc-linters all
+"the project's X if it has one"; dropped concrete ;;; markers, kept the
+mechanical-applier principle; added graceful degradation for no-suite/no-header/
+no-linter; added the pipeline cross-links to /refactor + /simplify). INDEX entry
+under new "Code quality" section. sync-check exit 0, mirror matches.
+Told Craig the run sequence: /refactor (incl. simplification) + readability-audit
+= existing-code sweep; /simplify = in-flight-diff cleanup. Offered (not built) a
+code-quality umbrella workflow to chain them.
+
+*** Item G — APPLIED (not yet committed)
+Craig picked option 2 (do it now) + alias approach. Implemented dot-stripped
+project-name resolution: inbox-send.py gained display_name() (basename with dots
+stripped), find_target() falls back to a dot-stripped alias after exact match
+(exact wins), print_project_list shows the stripped name. triggers.md launch
+resolution gained the dot-stripped match rule. TDD: 3 new alias tests red→green
+(incl. exact-wins-over-alias), 26 inbox-send tests pass; sync-check exit 0; full
+make test exit 0. .emacs.d→emacsd, .dotfiles→dotfiles now resolve in both ai
+launch and inbox-send.
+
+** Walk complete; inbox close-out done — commits pending
+A,B,C,D,E,F,G all applied + verified, uncommitted. Inbox cleared (0 pending):
+bug-priority proposal + cover preserved to docs/design/2026-06-27-*; 9 other
+handoffs deleted (content in canonical files). :LAST_INBOX_PROCESS: stamped
+2026-06-28. Replies sent: emacsd (4 items, via the new alias), home (locating-craig),
+work (bug-matrix FYI); plus binding-adoption handoffs to home + work.
+Commits DONE — 6 landed on main (publish flow: /review-code over staged diff =
+Approve; /voice personal over all 6 messages; Craig approved all):
+ b621914 feat(refactor): add simplification scan mode (A)
+ d4e9d7d feat(rules): add locating-craig rule (B)
+ 797c426 feat(workflows): add suspend and readability-audit workflows (C+F)
+ 92dfc35 feat(hooks): block bundled test+commit, require full suite before commit (E)
+ 9753d03 feat(inbox-send): resolve dot-stripped project names (G)
+ 798ef02 feat(todo-format): make the bug-priority matrix binding for codebases (D)
+PUSHED to origin/main (ecd33e0..798ef02); in sync (0/0). Pre-push reconcile
+confirmed ahead-only. Working tree: only .ai/notes.org marker +
+.ai/session-context.org (both for wrap-up).
+
+** Post-push follow-ups (Craig: "do both")
+- Task review: 3 stale tasks (reviewed 2026-06-15) re-stamped 2026-06-28; generic
+ agent-runtime spec re-graded [#C]→[#D] (speculative large arc, not committed);
+ memories-sync VERIFY + token-rotation helper kept. Staleness now 0. todo.org
+ uncommitted.
+- Umbrella workflow: created claude-templates/.ai/workflows/code-quality.org — one
+ trigger sequencing /refactor → readability-audit over a scope, surfaces the
+ filed :refactor: backlog, documents the /simplify boundary. INDEX entry under
+ Code quality; sync-check exit 0. Uncommitted.
+Both committed + pushed (publish flow: /review-code Approve, /voice personal on
+the workflow body, Craig approved):
+ 96dfa63 feat(workflows): add code-quality sweep workflow
+ 5263cd6 chore(todo): task review — restamp stale tasks, downgrade generic-runtime to [#D]
+origin/main in sync (0/0). Staleness nudge cleared (0).
+Then committed the notes.org inbox marker (6fb6797 chore) to clean the tree;
+working tree now only .ai/session-context.org (live anchor). 1 ahead of origin
+(the marker commit, unpushed). [Pushed 6fb6797.]
+
+** Next work: wrap-up routing feature (Craig: "1 then 2")
+*** 1 — Recommendation engine (spec Phase 1+3) — BUILT, tested
+Added .ai/scripts/route_recommend.py (canonical+mirror): pure recommend(item,
+projects)→(destination, confidence) — strong/weak/none, word-boundary literal
+match, dot-stripped alias aware, top-tier tie→weak deterministic, empty→none.
+CLI (--item/--exclude) reuses inbox-send discover_projects via importlib. 13
+tests green, full make test exit 0, mirror synced. Sub-task in todo.org rewritten
+to dated entry. UNCOMMITTED — committing via publish flow next.
+*** 2 — wrap-teardown manual validation — DEFERRED (Craig redirected)
+Engine committed + pushed (6be62ae). Then Craig redirected: "many tasks in
+todo.org were shipped — let's do a full task audit." Pivoted to task-audit.
+
+** TASK AUDIT (Craig: many tasks shipped)
+Phase A: 21 open tasks (lines 42-1134 in Open Work). Phase B: dispatched 4
+parallel read-only reconciliation subagents over batches, each checking tasks
+vs git log + repo tree + sessions, returning CURRENT/DONE/STALE/NEEDS-USER.
+Verdict: contrary to "many shipped," NONE are fully-done-but-open. Most CURRENT
+(backlog). The 2 closest (wrap-teardown 42, memories-sync VERIFY 214) are
+code-complete, gated only on Craig's manual validation.
+Phase C autonomous updates applied: task 186 (folded cj generic-header redirect,
+superseded 2-option fix), task 203 (folded cj "document as local-only"; :bug:→
+:chore:, reframed as docs task), task 428 (precondition-landed note + LAST_REVIEWED).
+Phase E: :LAST_AUDIT: stamped 2026-06-28. Phase F: skip task-review chain (ran
+today). NEEDS-USER + clusters surfaced to Craig next. todo.org + notes.org
+uncommitted (audit edits).
diff --git a/.ai/sessions/2026-06-29-03-56-spec-lifecycle-decision-and-speedrun-ratified.org b/.ai/sessions/2026-06-29-03-56-spec-lifecycle-decision-and-speedrun-ratified.org
new file mode 100644
index 0000000..2a61f75
--- /dev/null
+++ b/.ai/sessions/2026-06-29-03-56-spec-lifecycle-decision-and-speedrun-ratified.org
@@ -0,0 +1,107 @@
+#+TITLE: Session Context
+#+DATE: 2026-06-28
+
+* Summary
+
+** Active Goal
+Handle todo.org items 4 (spec storage location + lifecycle convention) and 5
+(speedrun / autonomous-batch) — both decision-gated — then wrap.
+
+** Decisions
+- Item 4 status mechanism: org-keyword authoritative + Status field in Metadata,
+ drop the filename suffix (Craig chose option 1 over his earlier filename-suffix
+ lean, 2026-06-28).
+- Item 4 scope addition: retrofit existing docs across ALL projects, not just
+ document the convention going forward (Craig, 2026-06-28).
+- Speedrun naming: the workflow is "speedrun" / "no approvals speedrun" (not
+ "fix speedrun"); threaded through task heading, body, and the spec prose.
+- Item 5 criteria recast (Craig found them too soft): removed the task-size gate
+ entirely (large tasks decompose into per-commit chunks; size gating defeated the
+ away-from-desk use case); replaced act-vs-file adjectives with a crisp 4-item
+ defer checklist keyed on test-writability; eligibility simplified to status TODO
+ AND :solo:.
+- :solo: / :quick: get hard definitions in todo-format.md, applied at creation and
+ enforced as a mandatory step in task-review + task-audit.
+- Added the speedrun pre-flight decision-gathering step: batch all quick decisions
+ up front, "skip this" drops a task, then run hands-off. Unattended loop has no
+ kickoff human, so it still defers decision-needing tasks.
+- Craig ratified all 8 revised decisions; spec Status → ready.
+
+** Data Collected / Findings
+- No abandoned work from any shutdown: clean wrap last session (no crash anchor,
+ clean tree, last commit was the wrap archive at 15:59). Craig's "machine shut
+ down" recollection didn't match the record; deferred work (wrap-teardown
+ validation) was the closest match.
+- The autonomous-batch spec already existed and reconciled the old fix-speedrun +
+ inbox-zero Phase E proposals; it had 6 drafted decisions awaiting ratification.
+ The revision grew it to 8 (added tag-definitions/enforcement + pre-flight Q&A).
+
+** Files Modified
+- docs/design/2026-06-16-autonomous-batch-execution-spec.org — major revision
+ (size gate removed, defer checklist, tag definitions, pre-flight Q&A, naming),
+ then ratified: Status ready, cookie [8/8], all 8 decisions DONE, history entries.
+- todo.org — item 4 (:373) decision + retrofit requirement recorded; item 5 (:394)
+ heading/body renamed to "No-approvals speedrun"; the spec-review VERIFY rewritten
+ to a dated event-log entry.
+
+** Next Steps
+- Item 5 build (when prioritized): Phase 0 (todo-format.md :solo:/:quick: definitions
+ + task-review/task-audit enforcement) through Phase 6 (synthesis). Parent task
+ stays DOING.
+- Item 4 build (when prioritized): spec-create via the recorded decisions; ship the
+ retrofit helper + startup nudge; pilot on rulesets' own docs/design first.
+- Naming cleanup: the proposal-doc filenames still carry "fix-speedrun"; a rename
+ pass with link updates is deferred.
+- Other open carryover from startup: wrap-teardown manual validation (task 42),
+ memories-sync VERIFY (needs ratio), plus the remaining what's-next candidates.
+
+KB: promoted 0 / consulted no
+
+* Session Log
+
+** 2026-06-28 — Startup + what's-next triage
+Ran full startup: clean wrap last session (no crash anchor), repos current,
+inbox empty, no reminders/pending decisions. Roam inbox had 4 items, all for
+other projects (.emacs.d, emacs-wttrin) — none for rulesets. Surfaced 5
+what's-next candidates; Craig picked items 4 and 5 to handle, then wrap.
+
+** 2026-06-28 — Item 4 decision recorded
+Craig chose option 1 for the spec lifecycle status mechanism (org-keyword
+authoritative + Status field, drop filename suffix; adopt location split +
+org-id links). He added a requirement: existing spec/design files in ALL
+projects must be sorted into docs/specs/ vs docs/design/ — a one-time per-project
+migration template sync can't do, so the spec must design the reach mechanism
+(proposed: synced classify-and-move helper under .ai/scripts/ + startup nudge
+gated on a :LAST_SPEC_SORT: marker). Recorded both into todo.org:373.
+
+** 2026-06-28 — Item 5 (speedrun) spec revised per Craig's direction
+Craig found the eligibility criteria too soft. Revised the autonomous-batch spec
+(docs/design/2026-06-16-autonomous-batch-execution-spec.org) substantially:
+- Removed the task-size gate entirely (Craig: size shouldn't matter; large tasks
+ decompose into per-commit chunks; speedrun is the away-from-desk mode and size
+ gating forced him to stay at the desk). I agreed; only caveat is the unattended
+ loop's cost ceiling, handled by the vNext token budget.
+- Recast act-vs-file as a crisp 4-item defer checklist keyed on test-writability
+ ("can I write the failing test from the task text without inventing a
+ requirement"), an enumerated data-loss operation list, already-satisfied, and
+ design-deliberation. Replaces the old adjectives.
+- Eligibility simplified to status TODO AND :solo: (size gone, so :quick: drops to
+ an effort hint, not a gate). :solo:/:quick: get hard definitions in
+ todo-format.md, applied at creation + enforced as a mandatory step in
+ task-review and task-audit (Craig's ask).
+- Added the speedrun pre-flight decision-gathering step: gather → classify → order
+ → intro → batch-ask the quick decisions → "skip this" drops a task → run
+ hands-off. Makes "no approvals" = all approvals front-loaded. The unattended
+ loop has no kickoff human, so it still defers decision-needing tasks.
+- Naming: "fix speedrun" → "no-approvals speedrun" in spec prose + todo.org:394
+ heading/body. Proposal-doc filenames keep their on-disk names (rename pass is
+ separate). Spec Status stays draft pending ratification of the revised decisions.
+Spec opened in emacs for Craig's review. Companion build edits still pending:
+todo-format.md definitions + task-review/task-audit enforcement (Phase 0).
+
+** 2026-06-29 — Item 5 ratified
+Craig ratified all 8 decisions. Spec Status → ready, cookie → [8/8], all 8
+decision headings DONE, ratification entry added to iteration history. The
+*** VERIFY "Review the autonomous-batch execution spec" (todo.org) rewritten to a
+dated event-log entry. Parent task stays DOING (build pending: Phase 0–6).
+Items 4 and 5 both handled. Ready to wrap.
diff --git a/.ai/sessions/2026-06-30-13-55-pager-mcp-ssh-alias-and-emacsd-proposals.org b/.ai/sessions/2026-06-30-13-55-pager-mcp-ssh-alias-and-emacsd-proposals.org
new file mode 100644
index 0000000..3cc8501
--- /dev/null
+++ b/.ai/sessions/2026-06-30-13-55-pager-mcp-ssh-alias-and-emacsd-proposals.org
@@ -0,0 +1,101 @@
+#+TITLE: Session Context
+#+DATE: 2026-06-30
+
+* Summary
+
+** Active Goal
+Startup, then process the .emacs.d inbox: chase how the 2026-06-29 Signal page was sent, clean up the dead page-signal path, and review + ship the three shared-asset proposals. Verify ratio's roam-sync, then wrap + tear down.
+
+** Decisions
+- Pager: no CLI wrapper. The signal-mcp tool is one call for in-session agents (the only paging caller); a signal-cli wrapper only if a non-session caller ever needs Signal paging. Pager guidance = notify --persist at the machine, signal-mcp send_message_to_user (to Craig's UUID) when away.
+- SSH alias drift: restore the bare =cjennings= alias in ~/.ssh/config (root fix, both daily drivers) rather than repointing one repo's remote.
+- green-baseline: accept with 3 changes (no-suite guard, cross-ref When You Cannot Verify, placement as start-work 0.3).
+- todo-cleanup aging: accept retain 7, default ON; ADD script self-protect so the archive inherits the todo file's gitignore status (Craig's rule).
+- lint-org checkers: accept with indented-heading tightened to 2+ stars (1+ false-positives on valid `*` list bullets).
+
+** Data Collected / Findings
+- The 2026-06-29 page went out via the signal-mcp tool (transcript 924ea200), not a revived account. The pager account (+15045173983) was never deregistered — the 2026-06-12 "deregistered" premise conflated the page-signal *script* removal with account loss. signal-cli registered, signal-mcp connected, live page reached Craig's phone.
+- "Network down" was a misread: internet was up, DNS fine, cjennings.net resolves. Real cause = git@cjennings remote alias missing from ~/.ssh/config (only cjennings.net defined). Fixed; both repos' remotes work again.
+- gitignore-mode code projects (chime, pearl, archangel, .emacs.d) all gitignore todo.org → the aging archive must follow or it leaks private task history to a tracked path on public repos.
+- indented-heading 1+-star regex flagged 3 valid `*` list bullets (reproduced); 2+ stars is the correct, collision-free target.
+- ratio confirmed over tailscale: roam clone + roam-sync.timer live and syncing; both daily drivers now set up.
+
+** Files Modified (all committed + pushed)
+- a266250 — Makefile prune step for dangling bin symlinks + protocols.org pager section.
+- dotfiles 3119bbb — ~/.ssh/config: restore bare =cjennings= alias (propagated to ratio over tailscale).
+- d0ab047 — verification.md green-baseline section + start-work Phase 0.3.
+- 324a52b — daily-drivers.md reframed around direct tailscale reach + mechanics section.
+- f67e724 — todo-cleanup.el Resolved-section aging + tc--ensure-archive-gitignored self-protect + 2 tests.
+- d9d8ce7 — lint-org.el four structural checkers, indented-heading at 2+ stars + negative-case test.
+- a1f87b1 — inbox-process marker. 08772c5 — ratio roam-sync confirmed (todo log + cleared daily-drivers open instance).
+- Three proposal source notes preserved under docs/design/.
+
+** Next Steps
+- Memory-sync VERIFY (todo.org:225) cross-machine half done; only its manual-validation child (work/unknown-project refusal checks, needs Craig's eyes) remains before DONE.
+- This wrap runs --archive-done with the new aging step: rulesets is track-mode, so ~most of its 65 Resolved entries move to a tracked archive/task-archive.org (intended).
+- Pager: still-open caveat is signal-cli's 27-day receive gap (send unaffected; long gaps can desync eventually).
+
+KB: promoted 0 / consulted no
+
+* Session Log
+
+** 2026-06-30 ~08:50 EDT — Startup
+Ran startup workflow. Network is down for git remotes (host =cjennings= unresolvable) — both the rulesets pull and the project-repo fetch failed; session continues on local state, no remote sync this session. No crash anchor (clean prior wrap). notes.org: no active reminders, no pending decisions. =make install= nothing new to link. =.ai/= synced from templates (no tracked change).
+
+Roam inbox: 5 items, all =emacs:= / =wttrin:= prefixed — none owned by rulesets; nothing to claim.
+
+Local inbox: 9 unprocessed handoffs, all from .emacs.d:
+- green-baseline-proposal.org — shared-asset change (verification.md + start-work skill): add a "green baseline before starting work" gate.
+- todo-cleanup.el + test-todo-cleanup.el + rulesets-note-archive-aging.org — shared-asset change to synced =.ai/scripts/todo-cleanup.el=: add Resolved-section file-aging to =--archive-done=.
+- lint-org.el + test-lint-org.el + rulesets-note-lint-checkers.org — shared-asset change to synced =.ai/scripts/lint-org.el=: four structural heading checkers.
+- dangling-page-signal note (2026-06-30 0112) — FYI + ask: =~/.local/bin/page-signal= is a dead symlink (canonical removed 2026-06-12); ask make install/uninstall to clean it on both daily drivers + confirm pager = =notify --persist=.
+- follow-up page-signal note (0115) — FYI: GV pager account (+15045173983) may be live again as of 2026-06-29; capture the revival method, update pager guidance.
+
+Three of these are shared-asset proposals → skeptical review + Craig approval required (no silent apply). The two page-signal notes are one actionable cleanup + FYIs. Confirmed =~/.local/bin/page-signal= target is dead (exit 2). Working tree clean except the 9 untracked inbox files.
+
+** 2026-06-30 ~09:05 EDT — Signal pager investigation (Craig's option 2)
+Craig asked how the 2026-06-29 Signal page was done. Traced it: the .emacs.d session (transcript 924ea200) called the signal-mcp tool =send_message_to_user= at 19:19 to Craig's UUID b1b5601e-…, after desktop notify couldn't reach his session bus. No re-registration / captcha — the pager account was never down.
+
+Finding: the "GV account deregistered 2026-06-12" premise in the follow-up note (and the PROCESSED-2026-06-12 note) is WRONG. Only the =page-signal= shell script was removed from canonical on 6-12; its =~/.local/bin/page-signal= symlink dangles. The account stayed registered the whole time. Verified live: signal-cli listAccounts shows +15045173983 registered; =claude mcp list= shows signal-mcp ✔ Connected. Two independent paths, opposite status: page-signal script = dead; signal-mcp tool = alive (the working pager). Caveat: signal-cli warns last *receive* 27 days ago — send unaffected, but long receive gaps can desync eventually.
+
+Updated the =project_signal_pager_account= memory to correct the page-signal-vs-signal-mcp distinction and kill the deregistration premise.
+
+** 2026-06-30 ~09:25 EDT — page-signal cleanup + pager guidance (Craig directed: proceed)
+Craig confirmed the live signal-mcp page reached his phone. Directed: point pager guidance at signal-mcp for when away from laptop/desktop, clean up the dead script, commit+push when remote is reachable.
+
+Found page-signal already fully removed from the repo on 2026-06-12 (13256aa) — Makefile, workflow, mcp/README all clean. Only the on-disk =~/.local/bin/page-signal= symlink dangled. Root gap: the install bin loop links =claude-templates/bin/*= but never prunes orphans, so any removed script leaks a dangling symlink. Fixed durably: added a prune step to the install bin section (Makefile) that removes symlinks in =~/.local/bin= pointing into =claude-templates/bin/= whose target is gone. Ran =make install= — pruned page-signal on velox; ratio self-cleans on its next session's make install.
+
+Pager guidance: added a "Paging Craig — desktop vs. away" section to protocols.org (canonical), distinguishing =notify --persist= (at-machine) from signal-mcp =send_message_to_user= to the UUID (away), and explicitly retiring the page-signal script. sync-check --fix synced the mirror.
+
+Files modified: Makefile (prune step), claude-templates/.ai/protocols.org + .ai/protocols.org (pager section). Reviewed (/review-code --staged → Approve), /voice personal on the message. Committed a266250 (authored as Craig).
+
+** 2026-06-30 ~09:55 EDT — Decision: no CLI wrapper; push diagnosis (SSH alias drift)
+Craig's design call: don't rebuild a page-signal-style script. The signal-mcp tool is one call for in-session agents (the only paging caller we have); a thin signal-cli wrapper gets added only if a non-session caller (hook/cron) ever needs Signal paging. protocols wording already reflects this.
+
+Push diagnosis: the "network down" read this session was WRONG. Internet is up, DNS works, cjennings.net resolves (IPv6). The real cause: =origin= is =git@cjennings:rulesets.git= (bare host alias =cjennings=), but =~/.ssh/config= only defines =Host cjennings.net= — no bare =cjennings= alias — so every git remote op failed to resolve the literal hostname. =git@cjennings.net:rulesets.git= works (ls-remote returned HEAD). Pushed a266250 via the explicit .net URL (clean fast-forward e4aeea6..a266250); remote main now at a266250.
+
+OPEN: the =cjennings= SSH alias drift is a root issue affecting every repo whose remote uses =git@cjennings:=. Durable fix is a Craig decision — restore =Host cjennings= in =~/.ssh/config= (covers all repos, dotfile-synced to both daily drivers) vs. repoint this repo's remote to =cjennings.net=. Daily-driver concern: ratio may differ. Surfaced to Craig.
+
+** 2026-06-30 ~10:05 EDT — SSH alias fix (Craig: option 1) + both pushes landed
+Craig chose the root fix. =~/.ssh/config= is a stow symlink → canonical =~/.dotfiles/common/.ssh/config= (the ~/.dotfiles git repo). Edited the cjennings block from =Host cjennings.net / LogLevel ERROR= to =Host cjennings cjennings.net / HostName cjennings.net / LogLevel ERROR= so the bare alias maps to the real host. Verified: =ssh -G cjennings= → hostname cjennings.net, identity id_ed25519; =git ls-remote origin= (the original git@cjennings URL) returns a266250 — unbroken.
+
+Committed in dotfiles as 3119bbb (fix(ssh): restore the bare cjennings host alias, authored as Craig) and pushed via git@cjennings (alias works). rulesets a266250 also confirmed on remote.
+
+DAILY-DRIVER (ratio): ratio likely has the same alias drift. Once ratio runs =git -C ~/.dotfiles pull=, the stow symlink picks up the fix immediately (no re-stow needed if the symlink exists). Until then ratio's pushes to cjennings.net keep failing. Flagged to Craig per daily-drivers.md.
+
+NOTE: dotfiles repo has its own .ai/ scope + 5 pending inbox handoffs — left untouched (cross-project boundary; they belong to a dotfiles session).
+
+** 2026-06-30 ~10:15 EDT — ratio synced over tailscale; alias drift fully closed
+Craig: do the ratio pull now, it's on tailscale (ratio = 100.71.182.1). Confirmed ratio had the same alias drift (=ssh -G cjennings= → hostname cjennings). ratio's dotfiles remote already used the .net form (=git@cjennings.net:dotfiles.git=), so a plain ff-only pull worked there — no bootstrap problem. Pulled 995f7d7..3119bbb (clean ff; ratio's untracked cross-agent-comms WIP left untouched). ratio's =~/.ssh/config= is a stow symlink, so the fix went live immediately: =ssh -G cjennings= now → cjennings.net, and =git@cjennings:rulesets.git= ls-remote returns a266250. Both daily drivers fixed.
+
+Page-signal arc fully closed. Replied to .emacs.d (inbox file 2026-06-30-1310-from-rulesets) and deleted both page-signal handoffs from rulesets inbox. 7 inbox handoffs remain = the 3 shared-asset proposals.
+
+** Reviewing shared-asset proposals in turn (Craig's direction)
+1. green-baseline — DONE. Accepted with 3 changes (no-suite guard, cross-ref When You Cannot Verify, placement as start-work 0.3). Implemented in verification.md + .claude/commands/start-work.md, proposal preserved to docs/design/2026-06-29-green-baseline-proposal.org. Committed d0ab047, pushed.
+4. daily-drivers tailscale correction (arrived mid-session 13:20) — DONE. Reframed daily-drivers.md from "can't reach, flag it" to "CAN reach over tailscale, sync/verify/repair directly" + a tailscale-mechanics section; my tweak: bare hostname resolves only with MagicDNS (ssh ratio worked from velox), so IP/MagicDNS is the reliable path. Note preserved to docs/design/2026-06-30-daily-drivers-tailscale-correction.org. Committed 324a52b, pushed. Replied to .emacs.d on both.
+2. todo-cleanup.el Resolved-section aging — DONE. Accepted (retain 7, default ON, Craig confirmed). Applied proposed .el + tests to canonical. ADDED self-protect (Craig's gitignore rule): confirmed gitignore-mode code projects (chime/pearl/archangel/.emacs.d) all gitignore todo.org, so the archive must too or it leaks private task history to a tracked path on public repos. tc--ensure-archive-gitignored appends the archive path to .gitignore when the todo file is ignored but the archive isn't; track-mode leaves both tracked. +2 ERT tests (temp git repo, per branch). 36 todo-cleanup tests green, full make test green. Committed f67e724, pushed. Note preserved to docs/design/2026-06-29-todo-cleanup-aging-proposal.org. Replied to .emacs.d (incl. heads-up: its next --archive-done sheds backlog + auto-adds the .gitignore line). NOTE: rulesets is track-mode → its next wrap sheds ~most of 65 Resolved entries to a tracked archive/task-archive.org.
+3. lint-org.el four structural heading checkers — DONE. Accepted with a fix: tightened indented-heading from one-or-more stars to TWO-or-more. The 1+ regex false-positives on valid org plain-list bullets (indented single `*` is a list bullet, not a demoted heading — reproduced: a normal `*` list flagged 3 valid bullets). `**`+ is never a bullet, so an indented one is unambiguously a demoted invisible heading. Added a negative-case test. 45 lint-org ERT green, full make test green. Committed d9d8ce7, pushed. Note preserved to docs/design/2026-06-29-lint-org-structural-checkers-proposal.org. Replied to .emacs.d.
+
+ALL inbox handoffs processed (inbox empty). Commits this session: a266250 (page-signal/paging), d0ab047 (green-baseline), 324a52b (daily-drivers tailscale), f67e724 (todo-cleanup aging + self-protect), d9d8ce7 (lint-org checkers); dotfiles 3119bbb (ssh alias). All pushed.
+
+OPPORTUNITY (now actionable): daily-drivers.md "Current open instance" wants ratio's roam clone + roam-sync timer verified — now doable directly over tailscale rather than waiting on Craig. Flagged, not yet done.
diff --git a/.ai/sessions/2026-07-02-09-29-docs-lifecycle-speedrun-autonomous-loop.org b/.ai/sessions/2026-07-02-09-29-docs-lifecycle-speedrun-autonomous-loop.org
new file mode 100644
index 0000000..8fc23e9
--- /dev/null
+++ b/.ai/sessions/2026-07-02-09-29-docs-lifecycle-speedrun-autonomous-loop.org
@@ -0,0 +1,110 @@
+#+TITLE: Session Context — Docs-lifecycle build, wrap-up router, speedrun build
+#+DATE: 2026-07-02
+
+* Summary
+
+** Active Goal
+(Session closed 2026-07-02 ~09:30 — the standing loop directive below ran through the night and morning and ended with this wrap; cron job 752624d1 deleted at wrap.)
+Standing directive (Craig, 2026-07-02 ~01:30): run auto-inbox-zero every 30 minutes with a STANDING YES — execute on all items found each cycle. Per-item disposition: feature-level task → write a spec (spec-create); decisions I can't confidently guess → file a VERIFY; well-defined → implement with the full quality bar. Autonomous commit + push under rulesets' :COMMIT_AUTONOMY: waiver. Auto-flush (/flush auto, self-inject) at clean boundaries when context grows heavy. Earlier directive items all DONE tonight: wrap-up routing, speedrun Phases 1-6 (work-the-backlog.org), roam inbox zero, auto-flush implementation + speedrun incorporation + disposition rule.
+
+** Decisions
+- Inbox: Craig approved all five dispositions (convert-subtasks bundle + planning-line fix, task-audit C.6, sweep security fix + public-reachability convention + broadcast, spec-review UI-traps promotion, KB orphan report → [#C] task).
+- Wrap-teardown: Craig ran all five manual tests, all passed — feature closed DONE, armed as the default (a bare "wrap it up" tears the session down; "with summary" keeps the buffer).
+- Speedrun Phase 0: hard :solo:/:quick: definitions are fixed cross-project in todo-format.md; review/audit tag assessment is mandatory; task-review gate 3 realigned to no-deliberation (1-2 upfront-answerable quick decisions allowed).
+- Docs-lifecycle spec: ratified through two independent review rounds (Codex + fresh-context Claude agent; 14 findings, all fixed, verify passes held); Codex flipped READY; spec-response decomposition flipped DOING. Key forks: two-sequence keyword header; :SPEC_ID: parent-keyword binding; fail-safe --apply; file: links through the pilot with id: conversion gated on the .emacs.d id-index mechanism; evidence-based status confirmation.
+- Lesson: never flip a lifecycle state in the same pass that authored the fixes — the reviewer owns the flip.
+
+** Data Collected / Findings
+- Spec: docs/specs/2026-07-01-docs-lifecycle-spec.org, status DOING, :ID: 80b0787b-4a60-4c82-8a16-b383d3e3c8f2; build parent in todo.org carries :SPEC_ID: (task "Spec storage location + lifecycle-status convention", ~line 361).
+- Pilot surface: docs/design has 41 files (3 spec-spine candidates: Decisions AND Implementation phases); 2 stray root specs (agent-knowledge-base-spec.org, inbox-workflow-consolidation-spec.org); docs/design/task-review.org is the note counter-case (Metadata only).
+- Validations: KB check 1 verified (55 rg = 55 org-roam DB); check 4 velox half verified (probe node agents/20260701214910-kb-sync-validation-probe.org pushed f0252bb, 0 conflicts) — ratio half blocked (ssh times out; probe left in place; confirm command logged in todo.org). KB checks 2+3 need work/unknown-project sessions. Roam inbox holds 2 rulesets items (ai-term colors → .emacs.d territory; "wrap it up closes window" → already delivered by wrap-teardown).
+- Bare [N/N] tokens in org prose get mangled by cookie updates — spell counts in words.
+
+** Files Modified
+All committed and pushed through 9ad415d. Tonight: 80ca5d0 spec-sort helper + 33-test bats; f4b64d6 pilot (5 specs sorted to docs/specs/, board live, :LAST_SPEC_SORT: stamped); 21639cb startup nudge (find-based probe, compgen was bash-only) + .emacs.d convention-live/id-index note; 7c12007 wrap-up router (route-batch + 13-test bats, inbox.org :ROUTE_CANDIDATE: stamp, wrap-it-up router step, cross-project.md); 9ad415d speedrun decomposition + spec DOING flip. Earlier (2026-07-01): d0c92d0 docs-lifecycle Phase 1 and everything before it.
+
+** Next Steps
+SESSION CLOSED — the auto-inbox-zero loop ended with the wrap (job deleted; a future session re-arms it only on a fresh directive from Craig). What carries forward: Craig's parked [#C] wrap-summary keep-or-cut think-through; ratio probe confirm (ssh was timing out — verify agents/20260701214910-kb-sync-validation-probe.org landed on ratio); KB refusal checks 2+3 (need work/unknown-project sessions); docs-lifecycle leftovers (Craig's manual tests: nudge visibility + Emacs id-link click-through; 4 anomaly renames; id-conversion gated on .emacs.d id-index). All build work from this session is DONE and pushed.
+
+Original loop contract (historical): continue the hourly auto-inbox-zero loop (cron job 752624d1, fires at :37, session-only). Each cycle: inbox-status + roam scan (capture-guard before roam writes); quiet → one acknowledgement line; finds → file, then execute ALL under Craig's standing yes with autonomous-commit + push (:COMMIT_AUTONOMY: + :LOOP_MAY_COMMIT: both stamped in notes.org Workflow State), disposition feature→spec / unguessable→VERIFY / well-defined→implement, full quality bar, metrics JSONL per task, session-log update per state-mutating cycle, /flush auto at heavy-context clean boundaries. Parked for Craig at his choosing: wrap-summary keep-or-cut think-through ([#C] in todo.org). Carryovers: ratio probe confirm (ssh), KB refusal checks (work/unknown sessions), docs-lifecycle leftovers (Craig's manual tests: nudge visibility + Emacs id-link click-through; 4 anomaly renames). Everything else from tonight is DONE and pushed (speedrun spec IMPLEMENTED after live trial; auto-flush + self-inject shipped; inbox-send collision fix; page info-styling; host-identity rule; template-sync policy; id-link conversion).
+
+KB: promoted 1 / consulted no
+(Node: agents/20260702093025-reviewer-owns-the-lifecycle-flip.org — the don't-flip-your-own-fixes lesson from the docs-lifecycle spec review.)
+
+OLD (completed): build the speedrun / autonomous-batch phases per the spec docs/specs/2026-06-16-autonomous-batch-execution-spec.org (DOING, :ID: 90f623cd-fdbe-4f5c-b63d-b2f84d9151cf; build parent "No-approvals speedrun" in todo.org carries :SPEC_ID:, children Phases 1-6 + live-trial + flip). Read the spec's Design section (lines ~66-177: loop at both altitudes, eligibility gate, defer checklist, pre-flight Q&A, session modes/preset, run cap + kill switch, paging) before writing. Phase 1: write claude-templates/.ai/workflows/work-the-backlog.org (eligibility gate, defer checklist, per-task quality bar, run-cap; inputs task set + session mode + cap) AND revert inbox.org's "auto inbox zero" per-cycle item 3 yes-path to routing-only in the same commit (one home for execution). Phase 2: wire both callers (auto-inbox-zero yes-path → work-the-backlog tag-query/file-only/cap-1; speedrun preset → explicit list/autonomous-commit/always-push/paging after pre-flight Q&A). Phases 3-6 per the child task bodies. Canonical-side edits, sync-check --fix, make test, commit per phase. THEN: roam inbox zero (inbox.org roam mode; 2 known rulesets-related items: ai-term colors → .emacs.d territory, wrap-it-up-closes-window → already delivered) and react. Docs-lifecycle leftovers for Craig: flip decision pending his manual tests (nudge visibility + Emacs link click-through), 4 anomaly renames, id-conversion gated on .emacs.d. Carryovers: ratio probe confirm (ssh still timing out), KB refusal checks.
+
+* Session Log
+
+** 2026-07-02 Thu @ 09:30 -0400 — Session wrapped (teardown mode)
+Morning loop cycles after the 07:51 auto-flush were all quiet (0 pending handoffs, roam inbox empty; last manual cycle ~09:15). Craig called the wrap. Cron job 752624d1 deleted; one KB node promoted (reviewer-owns-the-lifecycle-flip); todo cleanup + lint ran; wrap commit pushed. Teardown sentinel dropped per the validated default.
+
+** 2026-07-02 Thu @ 07:51:13 -0400 — flushed (auto-flush, self-injected)
+Clean boundary: 07:50 loop cycle came back empty, tree clean, everything pushed through the metrics commit after a6b534f, suites green. Nothing in flight. First live use of the auto-flush mechanism shipped tonight (self-inject via tmux run-shell -b). Post-clear resume: read this Summary and continue the hourly loop per Next Steps — the cron job survives the clear (session-only, not conversation-only). Craig's standing yes remains in force (Active Goal).
+
+** 2026-07-02 Thu @ 06:00 -0400 — Loop cycle executed 3 finds (commits through a6b534f + metrics, pushed)
+First executing loop cycle under the marker-granted autonomy. Found 3 archsetup handoffs + 2 rulesets roam entries (duplicates of the routed ones). Shipped: inbox-send collision fix (uniquify -2/-3 suffix, 4 red-first deterministic tests, 30/30 — a wild data-loss find, graded [#B] P2), page styling alarm → info --persist (page-me.org + work-the-backlog page; status-check untouched), dupre-blue ai-term refinement forwarded to .emacs.d (#67809c). Roam inbox: swept the 2 rulesets entries (archsetup's 3 left, roam-sync owns the git). Local inbox 0 pending; archsetup replied. Suites green, sync clean (one drift caught by the pre-commit check and re-synced). Loop continues hourly at :37 (job 752624d1).
+
+** 2026-07-02 Thu @ 05:30 -0400 — Live trial validated, spec IMPLEMENTED, loop now hourly (5eae9e0, pushed)
+Craig's "1" granted :LOOP_MAY_COMMIT: (stamped in notes.org Workflow State) and validated the run. Closed the live-trial + flip children dated, parent "No-approvals speedrun" DONE + CLOSED, spec flipped DOING → IMPLEMENTED (board: 2 IMPLEMENTED / 2 READY / 2 DOING). Wrap-summary keep-or-cut think-through PARKED (stays filed [#C] in todo.org). Auto-inbox-zero rescheduled: old 30-min job deleted, new hourly job 752624d1 (at :37, off-minute per fleet guidance), same standing-yes contract, now with loop commits authorized by the marker rather than only the directive.
+
+** 2026-07-02 Thu @ 05:25 -0400 — First no-approvals speedrun complete: 3/3 (78bbaae, b6a977c, ed75d3c + metrics commit, all pushed)
+Live-trial run c726f526 over Craig's ordered set. Pre-flight Q&A fired once (2 questions, Craig took both recommendations, answers stamped into task bodies as dated lines). Task 1 id-link conversion: 13 links → id: form, Review-findings heading got its own :ID: for the 2 search-target links, residue zero, all ids verified. Task 2 host-identity: claude-rules/host-identity.md (linked machine-wide, verified) + startup probe 13 (fixture-verified bash+zsh) + Phase C flag line. Task 3 template-sync: freshness policy in startup Phase A.0 (dirty = tracked-only; WIP-guard named as deliberate exception), monitor-inbox precondition fixed to --untracked-files=no + close-out symmetrized. Every task: /review-code, /voice, make test green, sync clean, one JSONL record (3 records in .ai/metrics/work-the-backlog.jsonl). End-of-set page fired via notify --persist. AWAITING: Craig's read on the run → then close the live-trial child dated + flip the autonomous-batch spec DOING → IMPLEMENTED. Then his two parked decisions: :LOOP_MAY_COMMIT: grant, wrap-summary keep-or-cut think-through.
+
+** 2026-07-02 Thu @ 01:40 -0400 — Inbox pass done; auto-flush shipped (d4f132b, 794b248, pushed); 30-min executing loop armed
+Local inbox zero + roam inbox zero (roam was already empty — archsetup routed it). Processed: .emacs.d org-id delivery → id-conversion task ungated (:solo: now); archsetup's three roam items → template-sync-gitignored filed [#C], ai-term colors forwarded to .emacs.d, wrap-summary keep-or-cut filed [#C]; auto-flush bundle → self-inject.sh canonicalized + 6-test bats, flush skill auto mode (gate order preserved: verify anchor write BEFORE arming), work-the-backlog auto-flush section + preset step 6 + per-item disposition rule (feature→spec, unguessable→VERIFY, well-defined→implement). Design note preserved at docs/design/2026-07-02-auto-flush-mechanism-note.org. Replies sent to .emacs.d (x2) + archsetup (x2). All suites green, sync clean. NOW: /loop 30m auto-inbox-zero with Craig's standing yes (see Active Goal).
+
+** 2026-07-02 Thu @ 01:30 -0400 — Speedrun Phases 2-6 shipped (263138a, 8d790c0, 04561b2, eea93f1, 44c8cc2 — all pushed)
+Phase 2: both callers wired (auto-mode chain ask scoped to the queued batch — a deliberate judgment, logged in the commit; speedrun preset section + trigger routing, "speedrun" always beats "no approvals" with disambiguation in no-approvals.org + INDEX). Phase 3: waiver pinned as :COMMIT_AUTONOMY:/:LOOP_MAY_COMMIT: markers in notes.org Workflow State; rulesets stamped :COMMIT_AUTONOMY: yes, :LOOP_MAY_COMMIT: deliberately left for Craig; .emacs.d told to stamp its own (inbox-send 0118). Phase 4: VERIFY-filing dedup (existing-sibling check), quick-question discriminator, batch-ask contract, page finalized. Phase 5: JSONL field table (+failed outcome, +manual caller, +comma-separated commit_sha — three traceable spec gaps closed). Phase 6: synthesis section + "synthesize backlog metrics" trigger. Every phase: /review-code, /voice, make test green, sync clean, todo.org child flipped dated. Parent stays DOING pending Craig's live trial + the flip task. NEXT: local inbox (2 handoffs: .emacs.d org-id delivery unblocks the docs-lifecycle id-conversion task; archsetup roam-routed item unread), then roam inbox zero + react.
+
+** 2026-07-02 Thu @ 01:15 -0400 — Speedrun Phase 1 shipped (d379a23, pushed)
+work-the-backlog.org created (canonical + mirror + INDEX entry): caller contract, five-outcome vocabulary, mechanical eligibility gate (TODO + :solo:, no-scheme-header → don't run), four-item defer checklist, quality bar, cap semantics, Phase 3-5 stubs. inbox.org auto-mode item 3 reverted to routing-only. Review fixed a cap-default contradiction (explicit set defaults to list length, not 1) pre-commit. make test green twice, sync clean, todo.org Phase 1 child rewritten dated. NOTE: new inbox handoff from .emacs.d (2026-07-02-0056) — org-id resolution delivered, the gated id-conversion task is unblocked; process during the inbox pass after the speedrun build. Next: Phase 2 (wire the two callers).
+
+** 2026-07-02 Thu @ 00:47:00 -0400 — flushed
+Clean boundary: wrap-up router shipped (7c12007 — route-batch helper + 13-test bats, inbox.org marker stamp, wrap-it-up router step, cross-project.md note; review fixed a reproduced nested-candidate data-loss bug pre-commit) and the speedrun spec decomposed + flipped DOING (9ad415d). Nothing half-edited; tree clean, pushed through 9ad415d, suite green. Post-clear resume goes straight to speedrun Phase 1 (contract pointers in Next Steps). Remaining under wrapup-routing: manual e2e (Craig's) + vNext transcript task.
+
+** 2026-07-02 Thu @ 00:25 -0400 — Phases 3 + 4 shipped: pilot ran, nudge live, .emacs.d notified
+Craig confirmed all five pilot keywords as-is (option 1) plus the IMPLEMENTED reason for agent-knowledge-base-spec. Applied with --allow-dirty (only the untracked session anchor was dirty): 5 specs moved to docs/specs/, 12 todo.org links + the moved specs' outbound links rewritten, :LAST_SPEC_SORT: 2026-07-02 stamped, residue zero, board live (6 specs: 1 IMPLEMENTED, 3 READY, 2 DOING). Phase 4: spec-sort probe added to startup.org Phase A + Phase C nudge line; replaced the spec's compgen sketch with a find-based check (compgen is bash-only, zsh false-negatived on stray root specs) — fixture-verified both shells, four project shapes; fixed startup.org's stale path to the moved encourage-kb spec; sent .emacs.d the convention-live note + id-index ask (2026-07-02-0022 handoff). f4b64d6 + 21639cb pushed, make test green, sync clean.
+
+** 2026-07-02 Wed @ 00:15 -0400 — Phase 2 shipped: spec-sort + 33-test bats suite (80ca5d0, pushed)
+TDD'd the retrofit helper (bats red-first, then the Python implementation). A fresh-context review agent found 4 real issues, all fixed pre-commit: acknowledged bare mentions weren't mapped through moves (a self-mention turned a successful apply into a false FAILURE with a destructive recovery recipe — regression-tested), real OSError mid-apply lost the applied-ops list (both failure paths now share ApplyFailure), "incomplete" status proposed terminal IMPLEMENTED (word-boundary matching now), and file-relative vs root-anchored link ambiguity now blocks validation as AMBIGUOUS. Real-data dry run matches predictions (5 candidates / 4 anomalies / 30 notes / 1 self bare mention / 10 report-only incl. the Codex-flagged startup.org case). make test green, sync clean. Next: Phase 3 pilot — candidate keywords + anomaly dispositions need Craig (see Next Steps).
+
+** 2026-07-01 Wed @ 23:41:36 -0400 — flushed
+Clean boundary after docs-lifecycle Phase 1 (d0c92d0, pushed, tree clean, suite green). In flight: nothing half-edited. Post-clear resume goes straight to Phase 2 — the spec-sort build (contract pointers in Next Steps above).
+
+** 2026-07-01 21:05 EDT — Session resumed after interruption; inbox first, then validations/specs
+The 2026-06-30 session died right after the validation pre-flight (nothing lost beyond the plan itself — no edits had landed). New session startup found this anchor plus 15 pending inbox handoffs. Craig's call: process the inbox first, then return to the validations + spec writing goal above. Also committed the one-line .claude/settings.json change from his /model command (c976f5b, "chore: set fable as project default model"). Green baseline confirmed before the commit: make test exit 0 — pytest 370+67+12 passed, all ERT suites 0 unexpected, 0 bats failures.
+
+Inbox inventory (15): the .emacs.d convert-subtasks bundle (10 files — todo-cleanup/lint-org + tests + 5 workflow/rule wirings), the .emacs.d task-audit Phase C.6 follow-on (2 files), the .emacs.d sweep-gitignore anchored-pattern security fix ask, the archsetup spec-review UI-traps promotion proposal, and a KB hygiene report (42 orphan agent nodes). Plan: skeptical review via diff-against-canonical for the bundles, then surface dispositions for approval.
+
+** 2026-07-01 ~21:45 EDT — Inbox items A, B, C shipped
+Craig approved all five recommendations (A1 B1 C1 D1 E1). Shipped so far:
+- A (19ba7cb): convert-subtasks bundle applied to canonicals + mirror — todo-cleanup --convert-subtasks, lint-org subtask-done-not-dated checker, wiring in wrap-it-up/clean-todo/open-tasks/task-review/todo-format.md. TDD'd the planning-line edge on top (CLOSED removal now preserves a DEADLINE/SCHEDULED sharing the line; red then green). Suites 45/45, 49/49, full make test green.
+- B (356b905): task-audit Phase C.6 (retire completed parents / promote stragglers) applied as sent.
+- C (909b21b + bac3fe4): sweep-gitignore-tooling.sh now recognizes anchored /.ai/ (mode detection + per-pattern presence + style-matched append) and WARNs on tracked tooling reachable via a non-cjennings.net remote; bare cjennings ssh-alias counts as private (false positive on rulesets itself caught by the real-data dry run, fixed with a 13th bats test). protocols.org gained the public-reachability convention. Real sweep run: 6 projects backfilled (archsetup, chime, emacs-wttrin in anchored style), archsetup's tracked CLAUDE.md flagged. Broadcast sent to 14 projects via inbox-send.py (the inbox-send wrapper isn't on PATH here — used the .py directly); archsetup got an extra tailored note about its tracked CLAUDE.md.
+Still open in this pass: D (spec-review UI-traps), E (KB orphan task), replies to .emacs.d + archsetup, inbox file cleanup, LAST_INBOX_PROCESS stamp, push.
+
+** 2026-07-01 ~21:55 EDT — Inbox pass complete: D, E, replies, cleanup
+D (9814b94): archsetup's six UI-traps checks promoted into spec-review.org Phase 4 as the conditional "Operational-panel UI traps" dimension; accept reply sent to archsetup. E: KB orphan review filed as [#C] :chore: in todo.org (orphan-ness isn't a defect; periodic prune/merge/link pass, regenerate the list before running); report deleted, no reply (script source). Full reply to .emacs.d sent covering all three of its handoffs, including the planning-line divergence from what it sent. All 15 inbox files deleted; inbox-status 0 pending; :LAST_INBOX_PROCESS: stamped 2026-07-01. Next: commit todo/notes, push all five commits, then return to the interrupted goal — manual validations (wrap-teardown task 42, Agent-KB refusal checks task 309) and the queued specs.
+
+** 2026-07-01 ~22:00 EDT — Validations: agent-runnable half done
+Inbox pass pushed (7 commits, e36e932..6ec05bb). Resumed the validations goal. Done tonight: wrap-teardown plumbing re-verified fresh (Stop hook + symlink + no stale sentinel + companion fns (t t t); 3 live aiv-* sessions available for the gate test); Agent-KB check 1 verified (55 :agent: nodes in rg inventory AND 55 in the live org-roam DB — match); check 4 velox half verified (probe node agents/20260701214910-kb-sync-validation-probe.org committed + pushed by roam-sync in seconds, f0252bb, 0 conflicts). BLOCKED: ratio ssh times out (tailscale ping pongs via DERP, TCP:22 unreachable — likely suspended); probe left in place for later confirmation. Still needing Craig: wrap-teardown 5-test checklist (scratch session), KB checks 2+3 (work/unknown-project refusal sessions), work-machine-no-clone check. Evidence logged in todo.org under both tasks. Next: surface status + spec-writing choice to Craig.
+
+** 2026-07-01 ~22:00 EDT — Wrap-teardown feature validated and closed
+Craig ran all five manual tests live — teardown-after-valediction, both summary qualifiers, the multi-session shutdown refusal, the cancellable countdown + stubbed shutdown, and the push-failure guard. All passed ("works great"). Rewrote the checklist sub-task to a dated entry and closed the parent DONE + CLOSED [2026-07-01 Wed] (archives on the next --archive-done). Sent .emacs.d a one-line FYI since its companion functions are half the feature. NOTE for this session's own wrap: teardown is now the validated default — a bare "wrap it up" here will tear this session down; use "with summary" to keep the buffer. Remaining validation carryover: KB refusal checks (work/unknown project sessions), ratio probe confirmation, work-machine-no-clone check.
+
+** 2026-07-01 ~22:15 EDT — Speedrun Phase 0 + docs-lifecycle spec drafted
+Craig picked "2 then 1". Phase 0 shipped (2a45f07): hard :solo:/:quick: definitions in todo-format.md (fixed cross-project; :solo: = buildable + agent-verifiable + no deliberation with 1-2 upfront-answerable quick decisions allowed; :quick: = ≤30-min effort hint, never a gate), mandatory-assessment language in task-review + task-audit, and task-review gate 3 realigned from "no upfront decision" to the ratified no-deliberation form. Then the docs-lifecycle spec drafted at docs/specs/2026-07-01-docs-lifecycle-spec.org from the five 2026-06-28 decisions, dogfooding itself (first resident of docs/specs/, status heading DRAFT, :ID: link). Design call made in the draft: the authoritative keyword lives on a prepended top-level status heading (vocabulary DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED) — additive, retrofittable, grep- and agenda-scannable. lint-org --check on the spec: 0 mechanical, 6 judgment items that are todo-shape checkers misreading spec conventions (spec DONE decisions carry no CLOSED; review-history dated headers are template shape) — noted, no action. Task flipped DOING with a dated entry. Awaiting Craig's spec review to flip DRAFT → READY.
+
+** 2026-07-01 ~22:30 EDT — Dual review of the docs-lifecycle spec; all nine findings fixed
+Craig had two independent agents review the spec: Codex (4 blocking findings, recorded in the spec's Review findings section) and my dispatched fresh-context reviewer (9 findings: same top blocker as Codex plus 5 unique, incl. the unowned DOING→IMPLEMENTED flip). Both rated Not ready; both independently caught that my #+TODO replacement destroyed the decision-task keyword machinery — the spec's own cookie was hand-faked. Craig approved fixing all nine. Responder pass done: merged ledger [9/9] with per-finding responses, two-sequence keyword header (verified: org computes [5/5] and [9/9]), transition-ownership table + spec-response flip task + task-audit safety net, single classification predicate, -spec.org rename, full relink contract, marker/nudge contract, compatibility rule, org-id prerequisite, three-line transition. Lint judgments on the spec are known todo-shape false positives (spec DONE entries and dated history headers). Status stays DRAFT; Craig decides the READY flip (option: send the fixed spec back to the reviewer agent, a082bea09c72a4e15, for a verify pass).
+
+** 2026-07-01 ~22:55 EDT — Second review round: five more findings, fixed and verified
+After the first nine fixes, my premature READY flip raced Codex's re-review — no data lost (commit 642be35 carries both my flip and Codex's demotion + five new blocking implementation-readiness findings; twelve seconds of READY). Craig approved a second responder pass, including the fork on the org-id finding (keep file: links through the pilot; id: conversion gated on a concrete .emacs.d id-index mechanism). Fixed all five (b163637): canonical-placement contract, :SPEC_ID: parent-keyword binding for task-audit (dissolves the flip-task chicken-and-egg, survives --convert-subtasks), fail-safe --apply (preflight/plan/recovery), staged id conversion, evidence-based status confirmation. Also de-cookified bracket [N/N] prose tokens org's cookie updater would mangle. My reviewer's second verify pass: ready, all held, nothing regressed, three minor nits — folded in (43cecd4): scoped id-link criterion, untracked-copy cleanup in recovery, two stale prose spots. Spec parked at DRAFT [14/14]; the authoritative READY flip is left to Codex's rerun or Craig. Lesson recorded: don't flip a lifecycle state the same pass that authored the fixes.
+
+** 2026-07-01 ~23:40 EDT — Spec READY (Codex flip), decomposed, Phase 1 built
+Codex's rerun flipped the spec READY at 23:22 (all fourteen findings closed). Craig: run with it, then flush and do Phase 2. Committed the reviewer flip, then ran spec-response Phase 6 as the first live exercise of the convention: :SPEC_ID: stamped on the build parent in todo.org, six child tasks (Phases 1-4, the gated id-conversion pass, the flip-to-IMPLEMENTED task) plus a manual-testing child (nudge visibility, link click-through), spec flipped READY→DOING (328ca18). Phase 1 then built and committed: claude-rules/docs-lifecycle.md (new rule, linked machine-wide), spec-create location + template updates (two-sequence header, DRAFT status heading with :ID:), spec-review location expectation + compatibility rule + READY flip ownership (incl. the demote path), spec-response DOING flip + :SPEC_ID: + mandatory flip task, task-audit :SPEC_ID: reconcile query. Mirror synced, make test green. NEXT AFTER FLUSH: Phase 2 — build claude-templates/.ai/scripts/spec-sort + bats per the spec's retrofit contract (classify predicate, evidence panel, plan/validate/apply, preflight, recovery, relink, marker stamp); the spec section "The retrofit" and the Phase 2 task body carry the full contract.
+
+** 2026-06-30 ~14:25 EDT — Startup + validation plan (interrupted session's log)
+Fresh session, clean startup (no crash anchor, clean tree, repos current). Craig: "do all the validations now, then write all the specs." Two validation items pending: wrap-teardown (task 42, 5-test checklist) and Agent-KB refusal checks (task 309, work/unknown project refusal — the cross-machine half is already confirmed on velox+ratio).
+
+Pre-flight on wrap-teardown plumbing: Stop hook wired in settings.json, hook script present+executable, all three companion functions (cj/ai-term-quit, -live-count, -shutdown-countdown) live in the daemon (t t t). Four live aiv-* sessions right now (aiv-_emacs_d, aiv-archsetup, aiv-rulesets [this], aiv-work). Read the hook + wrap-it-up Teardown mode + the three functions: cj/ai-term-quit is a safe idempotent no-op on a nonexistent project; shutdown-countdown aborts when >1 session live. So the gate + hook-wiring pieces are safely verifiable from this session without endangering it; only the buffer-teardown/geometry-restore and the countdown render/C-g need Craig's scratch session + eyes.
diff --git a/.ai/sessions/2026-07-04-12-57-audit-closeouts-and-startup-sync-guard.org b/.ai/sessions/2026-07-04-12-57-audit-closeouts-and-startup-sync-guard.org
new file mode 100644
index 0000000..c0d569b
--- /dev/null
+++ b/.ai/sessions/2026-07-04-12-57-audit-closeouts-and-startup-sync-guard.org
@@ -0,0 +1,68 @@
+#+TITLE: Session Context
+#+DATE: 2026-07-04
+
+* Summary
+
+** Active Goal
+
+Work a short list: commit the fable→opus model switch, run a task review, double-check the speedrun and finish it, then the routing feature. The review turned into a full task audit ("how many are actually completed"), which drove the rest.
+
+** Decisions
+
+- Craig redirected task-review → full task audit, scoped to completion.
+- Closed out all three code-complete tasks (docs-lifecycle, wrap-up routing, memory-sync): flip spec to IMPLEMENTED, close the parent, promote pending manual validation to standalone tasks. Same shape each time.
+- Accepted + applied home's startup skip-sync-when-behind proposal (shared-asset change, skeptical-reviewed, Craig-approved).
+- Pushed all three session commits to origin.
+
+** Data Collected / Findings
+
+- Of 18 open tasks, none were sitting done-and-verified; three were code-complete pending validation. The speedrun (autonomous-batch) was already fully done — IMPLEMENTED, live-trialed, archived in 5dc7da3.
+- Cross-machine memory sync verified as a live round-trip: on ratio (this machine) the velox-created probe was present with no conflicts; then over tailscale to velox, the probe deletion had propagated, both clones at HEAD 8c5ee01, both roam-sync timers active. Probe cleaned up.
+- Spec board now carries no DOING specs.
+- The commit no-attribution rule caught an accidental Claude-Session trailer on the first model-switch commit (amended out before it went anywhere).
+
+** Files Modified
+
+- .claude/settings.json — model fable→opus (73835a2).
+- todo.org + docs/specs/{docs-lifecycle,wrapup-routing}-spec.org + notes.org — three tasks closed, two specs flipped IMPLEMENTED, :LAST_AUDIT: stamped (44af1b2).
+- claude-templates/.ai/workflows/startup.org + mirror — Phase A proj_behind guard skips the template rsync when the project branch is behind upstream (bc4befa).
+
+** Next Steps
+
+- The three promoted validation tasks (docs-lifecycle manual checks; wrap-routing manual e2e; work/unknown-project KB write-refusal) need live cross-project or in-Emacs sessions to close.
+- home is applying the startup fix as a local stopgap; it can drop the stopgap after pulling rulesets.
+
+KB: promoted 0 / consulted no
+
+* Session Log
+
+** 2026-07-04 Sat @ 11:06 CDT — Startup + inbox
+
+Ran startup. Session-context was absent — previous session wrapped cleanly. Project repo clean/current (no ff needed); rulesets skipped its pull on a dirty tree (uncommitted =.claude/settings.json= model switch fable→opus, plus this session's own drops). =.ai/= synced from templates clean.
+
+One pending inbox handoff: =inbox/2026-07-01-0041-kb-hygiene-report.org=, a report from the automated =kb-hygiene.sh= script listing 41 orphan =:agent:= KB nodes. Checked todo.org — already filed as =** TODO [#C] KB orphan-node review pass :chore:= (line 42), whose body cites this exact report and instructs regenerating the list rather than trusting the snapshot. So the inbox file is a leftover; deleted it (script source, already captured, no reply owed). Stamped =:LAST_INBOX_PROCESS:=.
+
+Noted but not touched: =.claude/settings.json= carries an uncommitted model switch (fable→opus) — Craig's local config, left as-is. =inbox/lint-followups.org= is a task-review-health nudge (2 top-level tasks unreviewed >30 days), correctly excluded by inbox-status.
+
+** 2026-07-04 Sat @ 11:xx CDT — Model commit, task audit, three closeouts
+
+Committed the fable→opus switch (73835a2; first amend removed an accidental Claude-Session trailer that violates the no-attribution rule). Then ran a full task audit (Craig redirected from task-review to "how many are actually completed").
+
+Audit finding: of 18 open tasks, zero were sitting done-and-verified, but three were code-complete with only validation pending. Also surfaced that the no-approvals speedrun (Craig's item 3) was already fully done — autonomous-batch spec IMPLEMENTED, live-trialed (metrics commits), task archived in 5dc7da3.
+
+Closed out all three code-complete tasks (Craig approved A, then walked B and C):
+- A: docs-lifecycle convention — flipped spec to IMPLEMENTED, closed parent, promoted manual-validation + vNext to standalone tasks.
+- B: wrap-up routing (Craig's item 4) — same shape; build was fully shipped + green, flipped spec IMPLEMENTED, closed parent, promoted manual e2e + transcript vNext.
+- C: memories-sync VERIFY — implementation was IMPLEMENTED already; verified check 4 live this session. We're on ratio, so the ratio half was a local check (probe from velox present, zero conflicts). Then velox came back reachable — did the full round-trip over tailscale: velox pulled ratio's probe deletion, both clones at the same HEAD (8c5ee01), zero conflicts, both timers active. Deleted the spent probe. Closed the VERIFY; promoted the residual work/unknown-project write-refusal checks to a standalone task.
+
+Spec board now has no DOING specs. Stamped :LAST_AUDIT: 2026-07-04.
+
+Committed the audit closeouts as 44af1b2 (todo.org + both specs + notes.org), voice-passed. Two commits now sit unpushed ahead of origin (73835a2 model switch, 44af1b2 audit).
+
+** 2026-07-04 Sat @ ~12:53 CDT — Inbox: home's startup behind-guard fix applied
+
+Two inbox handoffs from home arrived mid-session (intro + proposal): startup Phase A step 3 rsyncs templates onto a project's stale committed .ai/ baseline when the project branch is behind/diverged, producing phantom drift that conflicts on reconcile. Ran the skeptical review (correct, complete, safe, universal); recommended accept; Craig approved applying now.
+
+Applied in canonical claude-templates/.ai/workflows/startup.org: a second guard computes proj_behind from git rev-list @{u}...HEAD and skips the rsync when behind>0, composing with the rulesets-clean guard. Added companion prose (step-3 heading + a rsync-notes bullet, incl. the no-auto-discard reasoning and the 6/22 flashcard-revert precedent). sync-check --fix propagated to the mirror; make test green (all ERT 0-unexpected, bats ok). Committed bc4befa. Replied to home (delivered to their inbox), deleted both inbox files, updated :LAST_INBOX_PROCESS:. Inbox clean.
+
+Three unpushed commits now: 73835a2, 44af1b2, bc4befa.
diff --git a/.ai/sessions/2026-07-11-02-30-inbox-clearout-six-proposals-shipped.org b/.ai/sessions/2026-07-11-02-30-inbox-clearout-six-proposals-shipped.org
new file mode 100644
index 0000000..3555f8f
--- /dev/null
+++ b/.ai/sessions/2026-07-11-02-30-inbox-clearout-six-proposals-shipped.org
@@ -0,0 +1,74 @@
+#+TITLE: Session Context
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-11
+
+* Summary
+
+** Active Goal
+
+Process the full project inbox (11 handoffs) and then the roam inbox. Craig chose to work the shared-asset proposals in order rather than deferring them, so the session became a full inbox clear-out with six proposals shipped.
+
+** Decisions
+
+- Craig approved all six shared-asset proposals after skeptical review, one batch at a time: triage-intake gmail cap, PR-review drop-praise, ui-prototyping rule, desktop-capture rule, bug-matrix double-count, staleness org-date parsing.
+- Placement calls: ui-prototyping and desktop-capture each got a standalone claude-rules/ file (not folded into existing workflows), matching the docs-lifecycle.md pattern spec-create already points at.
+- chime's staleness fix used the "both" approach: accept the org-native bracketed stamp AND warn loudly on genuinely-unparseable values, rather than silently counting them stale.
+- settings.json model flip (opus→fable, harness-written at 02:12) reverted to opus per Craig — matches the deliberate 2026-07-04 switch (73835a2).
+- Bare "wrap it up" → teardown mode.
+
+** Data Collected / Findings
+
+- The AI-attribution handoff's work had already landed as commit 6def7c4 (tree was clean); the 15 remaining "& Claude" files are all in .ai/sessions/ (historical, correctly held back). Confirmed agreement with everything the work session held back.
+- personal-gmail is the only gmail-family plugin rulesets owns, so the 100-cap fix touched one plugin.
+- The sent triage-intake copy carried the old "& Claude" author line; applied only the two Scan blocks, did not reintroduce it.
+- ~/.claude/settings.json is a symlink to the repo file, so a git checkout reverts both at once (not a hardlink).
+- Roam inbox churns fast (12 items at 01:01 startup, 6 at 02:16 process). Git history confirmed exactly 3 rulesets-owned items ever passed through and all 3 were filed — no rulesets item was missed; the count drop was foreign-item/other-machine churn.
+
+** Files Modified
+
+- Six proposal commits (see Next Steps for shas). Rules/skills: review-code/SKILL.md, voice/SKILL.md, voice/references/voice-profile.org, claude-rules/{commits,todo-format,verification}.md; new claude-rules/{ui-prototyping,desktop-capture}.md. Workflows: triage-intake.personal-gmail.org, spec-create.org, spec-review.org, task-review.org (+ mirrors). Script: task-review-staleness.sh + its bats (+4 tests) + mirrors.
+- todo.org: 5 tasks filed (org-table bug [#B], Signal pager [#C], 3 roam ideas [#C]). notes.org: :LAST_INBOX_PROCESS: stamped 2026-07-11. 11 inbox handoffs renamed PROCESSED-*.
+
+** Next Steps
+
+- All inbox work done; nothing carried forward. Filed backlog tasks await future sessions: org-table helpers data-loss bug ([#B] :bug:solo:, has a repro test), Signal pager doc ([#C], reconcile signal-mcp vs signal-cli first), and the 3 roam ideas.
+- Shipped shas: 81ca16f, 99bd213, 53f6ce6, 8ddd3d3, 06ca6c7, cfbabf1 (+ housekeeping b2aeaf6, 95dd661).
+
+KB: promoted 0 / consulted no
+
+* Session Log
+
+** Startup (2026-07-11 01:01 CDT)
+
+Ran startup.org Phases A.0/A/B. rulesets pull clean, make install nothing new, project repo current. .ai/ synced from templates. No session-context anchor (prior session wrapped cleanly). Read the 5 most recent session summaries and all 11 inbox handoffs. Verified the AI-attribution handoff's work already landed as commit 6def7c4 (tree clean). Created this anchor.
+
+** Inbox processing (Craig chose: file the 2 tasks, work proposals in order)
+
+Filed two todo.org tasks: [#B] :bug:solo: org-table helpers corrupt example blocks (graded Critical×rare = P2 = [#B]); [#C] :feature:spec: document/own the Signal pager (flagged the signal-mcp vs signal-cli/account-404211 reconcile). Replied to work (bug + attribution) and home (signal), marked those 3 inbox files PROCESSED.
+
+Skeptical-reviewed the two work proposals:
+- triage-intake.personal-gmail.org: clean 2-block Scan addition (100-cap date-slice walk + mandatory backlog-residue probe). Only other diff is the "& Claude" author-line regression, which I will NOT reintroduce. Low-risk correctness fix. Verdict: approve. personal-gmail is the only gmail-family plugin in rulesets.
+- PR-review tightening: (1) no praise at all on approvals — drop the "bare positive" carve-out at review-code/SKILL.md:453, :278, examples :458-465, and /voice pattern #40; lead with substantive pointer then verdict. (2) always print full inline prose at the gate — tighten review-code Phase 5 (:276) + commits.md Shape 1. Craig's own ruling from a real session. Verdict: approve.
+
+Both awaiting Craig's approve/park.
+
+** Work proposals shipped (Craig approved both)
+
+- 81ca16f fix(triage-intake): personal-gmail 100-cap date-slice walk + backlog-residue probe (canonical + mirror). Did NOT reintroduce the "& Claude" author line from the sent copy.
+- 99bd213 docs(review): drop praise from approve summaries (removed bare-positive carve-out from review-code Posted Summary Voice + voice #40 + voice-profile §40 examples/history + quick-ref table), require full inline prose at the gate (review-code Phase 5 + commits.md Shape 1).
+- make test green (269 ok, pytest 12), both authored as Craig, not pushed yet.
+- Replied to work on both; marked the 3 work inbox files PROCESSED.
+
+Remaining inbox proposals to work (Craig's chosen order): archsetup ui-prototyping (2 files), archsetup off-workspace-captures rule, chime staleness-parse, chime bug-matrix double-count.
+
+** Full inbox cleared — 6 proposals shipped, all 11 handoffs processed
+
+Shipped commits: 81ca16f (triage 100-cap), 99bd213 (drop-praise), 53f6ce6 (ui-prototyping rule), 8ddd3d3 (desktop-capture rule), 06ca6c7 (bug-matrix), cfbabf1 (staleness org-date TDD, +4 tests, suite 273 ok), b2aeaf6 (housekeeping: 2 tasks filed + processed markers + inbox marker). Replied to every sender (work, home, archsetup, chime). notes.org :LAST_INBOX_PROCESS: stamped 2026-07-11. Nothing pushed yet.
+
+FLAG: .claude/settings.json model flipped opus→fable at 02:12 (hardlinked to ~/.claude/settings.json, harness-written, not my edit). Reverts the deliberate 2026-07-04 opus switch (73835a2). Left unstaged — needs Craig's call (revert to opus, or accept fable).
+
+** Roam inbox-zero (Craig: "then inbox-zero again")
+
+Roam inbox had 6 items at process time (roam-sync churns fast). Bucketed: 3 claimed (rulesets:), 3 foreign (emacs.d ×1, archsetup ×2), 0 unowned, 0 empty. Filed the 3 rulesets items into todo.org (all [#C]: roam-only-startup investigation :spec:, keep-WIP-from-blocking-sync-gate :feature:, install-ai-on-PATH :chore:) — committed 95dd661. Removed the 3 rulesets items from the roam inbox (foreign left untouched), handed git to roam-sync → adeb011, pushed clean. Phase E done.
+
+Session state: 8 rulesets commits this session (6 inbox proposals + 2 housekeeping), roam inbox reconciled. Nothing pushed to rulesets origin yet. settings.json model-flip still unstaged/unresolved.
diff --git a/.ai/sessions/2026-07-13-23-48-runtime-portability-pager-local-llm.org b/.ai/sessions/2026-07-13-23-48-runtime-portability-pager-local-llm.org
new file mode 100644
index 0000000..8e55623
--- /dev/null
+++ b/.ai/sessions/2026-07-13-23-48-runtime-portability-pager-local-llm.org
@@ -0,0 +1,191 @@
+#+TITLE: Session Context — 2026-07-13
+#+AUTHOR: Craig Jennings
+
+* Summary
+
+** Active Goal
+
+A marathon (05:14–23:59) that became four arcs: (1) the runtime-portability children — all eight resolved, ending with a working local-LLM agent lane; (2) the EAT sixel investigation — persistent terminal images plus two dead .emacs.d agents root-caused (OOM from an EXDATE infinite loop, fixed cross-project); (3) the agent pager — Signal paging reconciled, verified, wrapped in a universal agent-page command; (4) full backlog hygiene — inbox zero on both surfaces repeatedly, task-review staleness driven to zero.
+
+** Decisions
+
+- Craig kept the fable model pin (d5bc9b3 accidentally committed only a case change — harness rewrote the file pre-staging; corrected in e91073d).
+- Thin-pointer AGENTS.md over a generated monolith for non-Claude bootstrap ("1 is the way to go. definitely").
+- All four inventory decisions approved (PreCompact prose downgrade, Stop-teardown via codex notify/manual, Signal-pager portability note, knowledge-base.md capture-layer sentence).
+- Signal pager re-graded [#C]→[#B]; ntfy two-way-comms task KILLED (infrastructure torn down 07-04; agent-page + Signal runbook are successors).
+- Drop all four legacy ollama models after verification; MoE over dense for Strix Halo; Vulkan backend forced over ROCm (systemd override).
+- "Agent pager" naming (the Slough House of pagers); paging documented as two channels everywhere.
+- EXDATE fix applied from here (cross-project, Craig-directed); jotto SVG rule and takuzu EAT-visuals rule both accepted with changes; .emacs.d's validate-el.sh load-prefer-newer fix accepted verbatim.
+
+** Data Collected / Findings
+
+- tmux transmits sixel only when TERM_SIXEL is set AND the client's cell pixel size is known; stock EAT 0.9.4 silently drops the CSI 14 t query (no t case in its dispatch). EAT persists sixel in scrollback natively; through tmux, passthrough images are write-once but native-path images survive redraws.
+- Both .emacs.d agent deaths were kernel OOM kills (111GB emacs) from calendar-sync--get-exdates reading (match-end 0) after split-string clobbered the match data — infinite loop only on comma-separated EXDATE lines far enough into the string. My first two diagnoses (stale sentinel; double-reply keystroke leak) were wrong; the double-reply was real but non-lethal.
+- signal-mcp is a velox-local MCP server, not claude.ai-side; the pager identity +15045173983 lives in velox's signal-cli; ratio's signal-cli is Craig's own number (no push). Receive-staleness warnings on both accounts.
+- Strix Halo: ROCm allocates correctly but first-touch loads at <25MB/s (unusable for 61GB); Vulkan loads the same model in 44s. gpt-oss:120b: 37.9 tok/s; qwen3-coder:30b: 76 tok/s; both 100% GPU. num_ctx matters (native 128K balloons loads). codex --oss needs --local-provider=ollama explicitly (config.toml append lands in the last TOML table, silently inert).
+- The remembered old launcher was aix + hey, removed from dotfiles 2026-04-20 (dab0d5a); bin/ai is their unified successor. lint-org.el's mutate-on-lint defect self-demonstrated during this wrap (no damage — no pipes in example blocks).
+
+** Files Modified
+
+Rulesets (16 commits, all pushed through 3984392): SVG rule (cac1aa1), visuals rule + zsh note (8703150), model pin ×2 (d5bc9b3, e91073d), runtime child tasks (1b638e6), inventories doc + decisions + skill parity (44bc344, 0ee94ab), AGENTS.md entry + install paths (6cd3aa3), launcher --runtime / local lane / agent picker (04c3b29, 1b98a9e, 3eed2e1), agent-page + two-channel paging docs (49be354), validate-el.sh fix (7e61dc0), model-floor status (21d14e0), review close-outs (0de6e09, 3984392). Plus dotfiles 5055ef1 (tmux sixel conf, pushed) and .emacs.d 56a3ed86 (EXDATE fix, committed there). System: ollama 0.31.2 + Vulkan override; ~/.codex/AGENTS.md link; two KB node writes (roam-synced).
+
+** Next Steps
+
+- Craig's maiden voyage: ai → pick local:gpt-oss:120b → takuzu; watch for skipped startup/gates/session-log/commit-rules — that's the model-floor eval running live. The formal scripted eval remains on the child.
+- Signal-pager [#B]: runbook, receive timer on velox, ssh-vs-linked-device decision.
+- Watch for .emacs.d's ai-term multi-LLM work (they're shelling out to ai --print-runtimes); upstream EAT PR is their [#D].
+- Backlog is fully current; org-table [#B] bug and launcher-hardening [#C] are the natural next builds.
+
+* Session Log
+
+** 2026-07-13 Mon 15:25 — SECOND .emacs.d death root-caused: OOM from an EXDATE infinite loop (my first diagnosis was wrong)
+
+Craig reported death #2 (launch 14:32, [exited] 14:55:36) — AFTER my double-reply fix, so that diagnosis fell. Corrections to the record: the transcript "queue-operations" are harness task-notification bookkeeping, NOT leaked keystrokes (my death-1 keystroke story loses its evidence — the double-reply leak was real and measured but likely never killed anything); the final "background task killed" entries are claude's SIGHUP handling. Hard evidence chain: EAT scrollback shows no bash prompt before [exited] (session destroyed under a live claude); journalctl nails BOTH deaths to the second — kernel OOM killer at 13:41:57 (emacs, 111GB anon-rss!) and 14:55:36, each inside the session's tmux-spawn scope, which systemd then failed, killing the pane → single-window session gone. The bomb: tests/test-calendar-sync--get-exdates.el's two comma-separated tests (NEW in a9e61207, the off-anchor EXDATE fix) hang forever: calendar-sync--get-exdates sets pos from (match-end 0) AFTER split-string, whose internal string-match clobbers match data → pos becomes the last comma's offset within the VALUE; when that's smaller than the EXDATE line's position, the loop re-matches the same line forever, growing the list to OOM. Verified: exact test string hangs the loaded .elc; bytecode disassembly confirms match-end after split-string; comma-less values don't clobber (failed string-match preserves match data), which is why only the two new tests bomb. Fix shape: capture (match-end 0) into a local BEFORE split-string (or save-match-data). Daemon check: calendar-sync loaded but NO timer → no live-daemon exposure right now. Side finding for the handoff: test-ai-term--show-or-create.el leaks one real tmux list-sessions call past its mocks (read-only, benign, but a mock gap; caught with a tmux PATH shim). Suite-under-shim full run unnecessary — OOM evidence is conclusive. Instruction-bootstrap build parked mid-scoping (no edits yet). Next: Craig decides fix-from-here vs handoff to the .emacs.d agent.
+
+** 2026-07-13 Mon 23:50 — inbox cleared again + task review COMPLETE (staleness zero)
+
+Handoff round: .emacs.d's validate-el.sh load-prefer-newer fix accepted verbatim (diff = exactly the two claimed evals; suite green; commit 7e61dc0 PUSHED; reply sent); takuzu ack deleted (all five routed captures filed there as three [#B] tasks); two more acks arrived mid-work and were processed (.emacs.d multi-LLM work underway, will shell out to ai --print-runtimes — interop achieved; home recorded the pager recipe) — sent home a one-line addendum pointing at agent-page so its memory doesn't fossilize the raw ssh recipe. Then the remaining 11 reviews, Craig-delegated: ntfy two-way-comms KILLED (CANCELLED — home tore down ntfy 07-04; Signal pager + agent-page are the successors); triage-intake phone-push re-pointed at agent-page (send half buildable now, recv waits on the Signal runbook); coverage-summary docs task tagged :quick:solo:; three tasks gained missing drawers; the rest confirmed. STALENESS NOW ZERO — the whole backlog is current. Commit 3984392 PUSHED. Emacsd handoff sent earlier re ai-term alignment. Everything on origin through 3984392. Remaining: Craig's takuzu local trial (his keyboard, likely tomorrow) and the wrap.
+
+** 2026-07-13 Mon 23:22 — agent picker shipped: bare ai asks which brain first
+
+Craig asked where fzf went and wanted the flow to offer LLMs before projects. Built TDD (9/9 launcher bats, suite 441/0): build_runtime_choices (claude first, codex-as-ChatGPT, one local:<model> line per live ollama model, 3s-timeout query, dead server drops lines), pick_runtime feeding the runtime map, dispatch restructured (picker only on bare ai without --runtime/AI_RUNTIME; check_deps moved after final runtime known), --print-runtimes test seam, usage updated (sed range widened to 2,23 for the grown header). Live print verified: all four lines. Commit 3eed2e1, PUSHED. Also answered: the remembered old launcher was aix + hey in dotfiles, removed 2026-04-20 (dab0d5a) when bin/ai unified them — no shadowed script exists; and the rofi/zenity UI-picker exploration stands offered but unbuilt (rofi -dmenu -multi-select = nearest drop-in; zenity = real GTK checklist; fuzzel single-select only). Roam-ask "agentically democratic" now substantially delivered. Still open: Craig's takuzu local trial, 11 task reviews, wrap. 23:24: sent emacsd the ai-term alignment handoff (runtime map strings incl. the --local-provider gotcha, picker shape + --print-runtimes as a shell-out option, verified perf numbers + Vulkan/num_ctx caveats, AGENTS.md bootstrap note) — Craig is actively working ai-term multi-LLM with that project.
+
+** 2026-07-13 Mon 23:06 — local runtime LIVE: ai --runtime local = codex --oss over ollama
+
+Craig pushed (e91073d..21d14e0 landed) and chose to wire the local lane. Shipped TDD (7/7 launcher bats, suite 439/0): local → codex --oss --local-provider=ollama -m $AI_LOCAL_MODEL (default gpt-oss:120b), AGENT_BIN/AGENT_CMD split for the deps check. End-to-end verified: codex exec --oss through local gpt-oss returned a correct completion (~7K tokens). Gotchas learned: codex needs the provider named (flag beats config — appending oss_provider to config.toml lands in the last TOML table and silently does nothing; a stray append was made and removed, ~/.codex/config.toml back to original). Also: upstream-patch question answered — .emacs.d shipped its durable home as c53671b0 (DONE, pushed) and filed the upstream PR as its own [#D] with working/eat-sixel-patch/; nothing needed from rulesets. Recommendation given for the live local trial: takuzu (well-traveled, 78+ tests, 5 fresh small tasks, known Claude baseline); watch for skipped startup/gates/session-log/commit-rules + practical tok/s. Commit 1b98a9e PUSHED. During the edit I briefly clobbered the 16:39 dated heading in todo.org (Edit replaced a heading instead of inserting above it) — caught and repaired in the same minute. Remaining for the session: Craig's live takuzu trial (his keyboard), the leftover 11 task reviews, wrap.
+
+** 2026-07-13 Mon 20:20 — local inference stack verified end to end; model thread CLOSED
+
+The gpt-oss saga resolved: ROCm on gfx1151 allocates/offloads correctly but first-touch loads at <25MB/s — two 40-min load attempts died to client timeouts (one mine, one the ceiling; a third was killed externally ~19:00, unexplained). qwen3-coder verified first on ROCm (5s load, 76 tok/s, 100% GPU) proving the generate path, isolating the problem to big-model load speed. Fix: systemd override /etc/systemd/system/ollama.service.d/strix-halo-vulkan.conf (OLLAMA_IGPU_ENABLE=1 + ROCR_VISIBLE_DEVICES="") → Vulkan/RADV as sole inference device → gpt-oss:120b loads in 44 SECONDS, "verified", 37.9 tok/s, 100% GPU. Craig's approved drops executed: tinyllama + llama3 rm'd from the service store, the orphaned 45GB ~/.ollama user store purged. KB node rewritten with the end-state (backend, override path + revert note, per-model numbers, num_ctx gotcha, MoE-not-dense lesson) and roam-synced. Model-floor child updated: remaining work is purely the instruction-following eval (gpt-oss as target, qwen as comparator). Commit 21d14e0. Unpushed now 9: 44bc344..21d14e0. All threads from today are closed or cleanly parked; ready to wrap whenever Craig calls it.
+
+** 2026-07-13 Mon 18:32 — agent pager shipped; pager thread effectively closed for today
+
+Full sequence since 17:15: routed all 10 foreign roam items to owners via inbox-send (roam inbox at true zero, takuku typo flagged); Signal reconcile — pager identity +15045173983 lives in velox's signal-cli (signal-mcp = velox-local MCP config, NOT claude.ai-side; corrected the 14:40 inference), ratio's signal-cli is Craig's own number; live page from velox buzzed Craig's phone (confirmed). Then Craig directed universal paging + the "agent pager" rename: NEW claude-templates/bin/agent-page (velox-direct/ssh-relay, UUID baked in, fallback hint; 4 bats, PATH-stubbed; live-verified through the real script — second buzz), protocols.org Paging-Craig rewritten (two channels, signal-mcp demoted), page-me.org + work-the-backlog + INDEX updated, mirror synced, make install linked it (velox inherits next session). Suite 438/0. Commits: 0de6e09 (review+filings), 49be354 (agent pager, incl. the Signal task's dated entries + interim-recipe-to-home note). home got the recipe handoff at 18:14. Signal task remains [#B] with sharpened deliverables: runbook, receive timer, ssh-vs-linked-device. Unpushed count now 8 (44bc344..49be354). Model verification still pending: gpt-oss smoke test backgrounded (bkpjmp2iu), then qwen3-coder smoke, then the four legacy drops + KB update.
+
+** 2026-07-13 Mon 17:15 — task review finished (Craig delegated the close)
+
+Craig said "finish the task review" — applied my judgments across the remaining 6 (+1 from the morning = 8 of 18 total; 11 remain for future daily rotations). Changes: Signal pager re-graded [#C]→[#B] (paging is a real capability gap post-inventory — flagged for veto); install-ai-on-PATH tagged :quick:solo:; wrap-up-routing validation gained its missing PROPERTIES drawer; the rest confirmed as-is, all stamped 2026-07-13. Committed 0de6e09 together with the inbox-zero filings. Next per Craig: "handle some of those roam inbox items" — all 10 remaining are foreign, so the sanctioned shapes are inbox-send routing or launching owner agents; presenting options.
+
+** 2026-07-13 Mon 17:08 — inbox zero (roam mode) run
+
+Craig called inbox zero mid task-review (review paused at item 2 of 7, roam-only startup, recommendation pending). Local inbox clear. Roam scan: 12 total, 2 rulesets-claimed, 9 foreign (4 archsetup, 4 takuzu, 1 emacsd), 1 mis-prefixed "takuku:" (plainly takuzu's — left in place, typo flagged to Craig), 0 unowned, 0 empty. Claimed items deduped against today's work: the agent-switching ask is partly shipped (--runtime claude|codex 04c3b29; codex = the ChatGPT CLI; ollama/qwen = the reserved local runtime on the model-floor child) — folded as a dated entry on the generic-agent-runtime parent. The new work filed as [#C] "ai launcher hardening — bug hunt + refactor pass" :refactor:solo: (interactive runtime picker rides it). Roam edit: both claimed items removed under a clean capture-guard, roam-sync triggered (10 items remain, all foreign). todo.org edits uncommitted. Monitor for pulls retired at Craig's cue earlier; pull ETA ~18:30 at 12MB/s.
+
+** 2026-07-13 Mon 16:45 — session plumbing closed with a no-build verdict (8 of 8 children resolved; model-floor eval pending pulls)
+
+Assessed the four plumbing pieces: session-context-path runtime-aware by design; anchor cycle plain-files-portable; self-inject harness-agnostic (payload is the only Claude-shaped part — codex auto-flush = payload variant, second injected line carries the resume instruction, no hook needed); session-clear-resume.sh stays Claude-only. Section added to the inventories doc, child closed, commit 3ef5580 (suite 434/0). ALL EIGHT runtime children are now resolved or reduced: 6 shipped/closed, launcher shipped, model-floor's remaining work is the evaluation itself, blocked only on the pulls (~19:00 ETA). Unpushed: 44bc344, 0ee94ab, 6cd3aa3, 04c3b29, 3ef5580.
+
+** 2026-07-13 Mon 16:42 — launcher runtime flag shipped (7 of 8 children done)
+
+While the pulls run (rate settled ~0.5GB/min, ETA ~19:00): built ai --runtime claude|codex TDD — 6-test bats file driving a new --print-launch seam (prints the pane command, no tmux), runtime→CLI map with local reserved (clear error naming the model-eval dependency), AI_RUNTIME env support, runtime-aware deps check, CLAUDE_CMD→AGENT_CMD at all three launch sites. Verified: codex --help confirms positional-prompt parity with claude; live smoke prints correct commands both runtimes. Suite 434/0. Commit 04c3b29; child closed as dated entry. NOTE: Craig's real sessions launch via .emacs.d's ai-term (aiv-<proj> per-project sessions), NOT bin/ai (shared "ai" session, window per project) — the Emacs-side multi-LLM support is .emacs.d's June handoff, unchanged here. Craig also approved dropping tinyllama + llama3:8b ("annoying little pest") — all four legacy models go after verification. Remaining child: session plumbing, local model floor (eval blocked on pulls). Unpushed: 44bc344, 0ee94ab, 6cd3aa3, 04c3b29.
+
+** 2026-07-13 Mon 16:40 — ollama upgraded 0.17.7→0.31.2, model pulls running
+
+Craig picked option 1 (drop both llamas after the new models verify) and directed the full sequence. Research (WebSearch): Strix Halo wants MoE models (bandwidth-bound decode; dense 70B ≈ single-digit t/s) — gpt-oss:120b (~55 t/s reported on this chip) + qwen3-coder:30b (~100 t/s, 256K ctx) are the picks; ollama 0.17.7 was far behind. Upgraded via the official installer (sudo, NOPASSWD): 0.31.2, systemd unit recreated, service active. GPU discovery healthy out of the box: ROCm sees gfx1151 iGPU, 107.4GiB available; Vulkan RADV device present but gated behind OLLAMA_IGPU_ENABLE=1 (community says Vulkan outperforms ROCm on gfx1151 — A/B during the model-floor eval, not now). DISCOVERY: the service uses /usr/share/ollama/.ollama (system store) — holds two MORE models my earlier "complete" inventory missed (tinyllama 637MB, llama3:8b 4.7GB); the 45GB ~/.ollama store is separate and invisible to the service. KB node needs both corrections (stores + new versions) after pulls complete. Pulls running in background (btkrc1y0n, sequential: gpt-oss:120b 65GB then qwen3-coder:30b 19GB) into the system store; du-based Monitor armed (be6nrzl5d). Disk: 3.0T free, no constraint (df is aliased to dfc — use command df). Pending Craig: extend the drop list to tinyllama + llama3? Pending me after pulls: smoke-test both models w/ GPU offload check, ollama rm the drops, purge the orphaned ~/.ollama user store, update KB node + model-floor child, session log.
+
+** 2026-07-13 Mon 16:15 — local-runtime inventory gathered and promoted to the KB
+
+Craig asked about ollama; gathered the full local-inference picture. ratio: ollama 0.17.7 (/usr/local/bin, no systemd unit, on-demand), 45GB store with llama3.1:8b + llama3.3 70B, Strix Halo Radeon 8060S iGPU + 125GiB unified RAM — a real inference box, and the on-disk 70B matches the model-floor estimate. velox: no ollama, Iris Xe, 60GiB — out of scope for local inference. codex CLI installed + authed on ratio. Promoted to the KB as agents/20260713161222-local-llm-inference-inventory-daily-drivers.org (committed, roam-sync triggered — first KB write in a long while, the contribute nudge finally earned). Folded the inventory into the model-floor child's body: the evaluation can run TODAY on ratio against the on-disk 70B (scripted startup + publish-flow transcript, grade instruction-following) — no procurement needed. todo.org edit uncommitted.
+
+** 2026-07-13 Mon 16:10 — instruction bootstrap shipped (6 of 8 children done)
+
+Memory sentry ran clean during Craig's relaunch (zero alerts, daemon flat 1.97GB, third .emacs.d session alive since 15:44 and healthy). Queue resumed: built the thin-pointer AGENTS.md TDD (bats red→green): canonical claude-templates/AGENTS.md; make install CODEX_DIR stanza (→ ~/.codex/AGENTS.md, live on ratio now; velox auto-picks-up via startup's make install); install-ai.sh seed-only copy at bootstrap; rulesets root tracked symlink as dogfood. NEW scripts/tests/install-agents-entry.bats (3 tests) + 2 install-ai.bats tests. Riders: todo.org line-1357 prose reworded (lint misread "file:→id:" as a dead link), lint-followups.org cleared. Suite 428/0. Commit 6cd3aa3 (also closes the child + adds a dated entry on the agent-source task's thread 2). Inbox: two .emacs.d acks processed + deleted (the dead 14:34 one, and the live session's 15:47 — HEAD 56a3ed86 confirmed, staged eat work intact, mock-leak filed [#C] there, no action needed). Unpushed rulesets commits: 44bc344, 0ee94ab, 6cd3aa3. Remaining children: launcher runtime flag (scoping next — codex mapping clear, local mapping blocked on the model-floor/default-runtime decision), session plumbing, local model floor. Then the org-table bug + 6 task reviews.
+
+** 2026-07-13 Mon 15:32 — EXDATE bomb fixed, .emacs.d suite green end-to-end, agent cleared to relaunch
+
+Craig chose fix-from-here. Applied the capture-line-end-before-split fix to calendar-sync-recurrence.el (with a why-comment naming the clobber mechanism), deleted + recompiled the .elc, reloaded into the daemon (live Emacs calendar-safe again). Verification: the two previously-infinite tests pass in microseconds; all 12 exdates tests green; all 56 calendar-sync files green; then the FULL suite ran to completion for the first time today — all 599 elisp files + 18 bats green, including the dead agent's uncommitted eat-config guard tests. Committed 56a3ed86 in .emacs.d with a pathspec so the agent's staged eat work stayed untouched; not pushed. Superseding handoff sent to .emacs.d inbox (2026-07-13-1530): corrected diagnosis (journalctl OOM evidence), fix details, the show-or-create mock-leak finding, the double-reply note downgraded to latent-bug status, explicit safe-to-relaunch incl. the suite. Craig can relaunch the .emacs.d agent now. Resume thread: instruction-bootstrap build (thin-pointer AGENTS.md, Craig approved shape) was mid-scoping when death #2 interrupted — Makefile install stanza + install-ai.sh seeding + canonical claude-templates/AGENTS.md + bats, TDD.
+
+** 2026-07-13 Mon 14:55 — pin pulled: four decisions applied, skill parity done
+
+Craig pulled the pin and approved all four inventory decisions. Applied: knowledge-base.md capture-layer sentence (non-Claude runtimes capture into the session log), Signal-pager task gained the runtime-portability motivation entry, decisions recorded in the doc, VERIFY closed as a dated entry. Commit 44bc344. Then the skill-parity child: inventoried 11 skills + 18 commands, key finding is that one resolution sentence in the bootstrap entry file makes all 29 portable to any file-reading harness (auto-invocation degrades to by-name, which the publish flow already uses; flush excluded, session-plumbing owns it; native registration optional nicety). Section appended to the inventories doc, child closed as dated entry, commit 0ee94ab. Both commits local/unpushed (with 44bc344). 5 of 8 children done. Remaining: instruction bootstrap (needs Craig's design input, overlaps the Multiple agent-source improvements task), launcher runtime flag, session plumbing, local model floor.
+
+** 2026-07-13 Mon 14:30 — PIVOT RESOLVED: .emacs.d agent death root-caused and fixed
+
+Craig's urgent pivot: the .emacs.d agent session (launched 13:20) died — tmux session aiv-_emacs_d gone, [exited] in the launcher shell. Investigation chain: stale-sentinel theory disproved (multiple turns survived; no sentinels; hook tests clean). Real cause: DOUBLE XTWINOPS REPLIES. The .emacs.d agent had picked up our eat-xtwinops handoff, designed an advice-based fix (cj/--eat-answer-xtwinops :before eat--t-handle-output, written to modules/eat-config.el 13:33, test file 13:32), and evaluated it into the live daemon — where MY morning spike (patched parser with its own CSI t clause, loaded 10:52) was still active. Two answerers → tmux consumes reply #1, forwards reply #2 to the active pane as raw keystrokes (ESC + junk) — and Emacs ptys report 0 pixel size so EVERY frame resize re-queries. Its transcript (83070ecd, ends 13:41:57) shows the signature: four empty queue-operations, then background make test killed as claude exited. Exact final exit trigger unreconstructable from the transcript; the leak itself confirmed empirically: probe showed 2 replies per query. Fix: reloaded stock eat.elc into the daemon 14:27 (advice survives redefinition and is now sole answerer); re-probe = 1 reply. Probe buffer cleaned up. Handoff sent to .emacs.d (2026-07-13-1427): root cause, safe-to-resume, the latent no-op-claim bug in its advice comment (must guard against upstream shipping the clause or the double-reply returns), rerun its killed ERT, and the coordination lesson (our 10:58 handoff should have disclosed the live spike — our omission). Craig can relaunch the .emacs.d agent now; it resumes from its own anchor.
+
+** 2026-07-13 Mon @ 14:16:49 -0500 — SUSPENDED (urgent pivot; supersedes the 13:22 entry)
+
+Open threads, most active first:
+- ACTIVE: runtime-portability children under the generic-agent-runtime parent (todo.org ~line 231). Done: hook parity, MCP portability, memory story — all three now dated entries pointing at docs/design/2026-07-13-runtime-portability-inventories.org. AWAITING CRAIG: the *** VERIFY with four decisions (PreCompact prose downgrade; Stop-teardown via Codex notify/manual; Signal-pager portability note; knowledge-base.md capture-layer sentence) — all recommended yes, he was about to answer when the pivot hit. Next child queued: skill parity (decision matrix over voice/review-code/flush etc.: re-register per harness vs fold into .ai/workflows/), then instruction bootstrap, launcher flag, session plumbing, local model floor.
+- PINNED (from 13:22, still queued): org-table helpers bug ([#B] :bug:solo:, todo.org ~line 66) + the todo.org "→id" dead-link rider from lint-followups.org.
+- PINNED: task review, 6 of 7 unwalked; they stay at the head of the next batch.
+- SET ASIDE: sixel/EAT — .emacs.d has the patch filed [#B] with the durable-home design call coming to Craig; after it lands + velox pulls dotfiles and .emacs.d, run the velox test (client_cell_width nonzero, then magick sixel in a tmux window, Craig's eyes).
+
+Pending decisions: the four VERIFY items above. Nothing else blocks.
+
+Shipped since the 13:22 suspend (all pushed): 828f70a..1b638e6 (SVG rule cac1aa1, visuals rule 8703150, model d5bc9b3, todo folds dc13bce, child-task filing 1b638e6) and e91073d (model fix — d5bc9b3 had committed opus→Opus from a harness rewrite, not the fable flip; e91073d lands fable as Craig chose).
+
+Uncommitted work: todo.org (three children converted to dated entries + the new *** VERIFY), docs/design/2026-07-13-runtime-portability-inventories.org (new file, untracked), this anchor. Commit shape when resumed: docs+todo together as the inventories chore/feat commit.
+
+Key findings not recorded elsewhere: none — everything is in the inventories doc, the todo entries, or this log. Inbox is clear (the 13:23 .emacs.d ack was read and deleted).
+
+Background work: none running.
+
+Resume hint: read the *** VERIFY under the generic-agent-runtime parent, get Craig's four answers, apply the two tiny approved edits (Signal-pager note + knowledge-base.md sentence), commit todo.org + the inventories doc, then start the skill-parity matrix.
+
+** 2026-07-13 Mon @ 13:22:43 -0500 — SUSPENDED
+
+Craig stuck a pin in everything mid task-review walk. Resume-weighted state:
+
+Open threads, most active first:
+- PINNED: org-table helpers bug ([#B] :bug:solo:, todo.org line 66). Craig queued it as the next work item ("2 then 1" — review done-ish, this is the 1). Fully specified in the task body: two defects (line-based table detection mangles example/src blocks; lint-org.el writes without --fix), repro from work 2026-07-09, regression test named. TDD it: canonical scripts live in claude-templates/.ai/scripts/, mirror-synced; tests are ERT via make test. Start here.
+- PINNED: task review, stopped after 1 of 7. Batch was the staleness --list top 7; task 1 (build-to-prototype [#B]) kept + stamped 2026-07-13. Items 2-7 (roam-only startup, WIP sync gate, install-ai PATH, org-table bug, Signal pager, KB orphan pass) unreviewed — they stay at the head of the next run's batch automatically.
+- SET ASIDE: sixel/EAT thread — rulesets + dotfiles halves are DONE and shipped; waiting on Craig to fast-track the eat-xtwinops.patch handoff sitting in .emacs.d/inbox/ (2026-07-13-1058-from-rulesets-*). After .emacs.d lands it and velox pulls dotfiles + .emacs.d, run the velox test: restart EAT/tmux server on velox, tmux display -p '#{client_cell_width}' nonzero, then magick logo: sixel:- into a tmux window, Craig eyeballs. Note ratio's working state is runtime-only until the .emacs.d patch lands (Emacs restart loses it; tmux server restart loses the feature flags — the dotfiles conf now covers that half).
+- DEFERRED: lint-followups.org rider — dead file link at todo.org (was line 1314 pre-edit; re-grep for "→id"), promised as a rider on the next work item.
+- POST-SUSPEND ADDITION (13:27): Craig asked for the ChatGPT/local-LLM gap assessment to be filed. Found the existing parent (** TODO [#D] Generic agent runtime support — Codex spec v0) and added 8 child tasks under it (instruction bootstrap, skill parity, hook parity, launcher flag, session plumbing, memory story, MCP portability, local model floor) plus a dated decomposition entry. todo.org edit uncommitted, riding with the review stamp.
+
+Pending decisions / open questions for Craig: none blocking. (Model flip resolved: keep fable, committed.)
+
+Shipped this session (all local, UNPUSHED except dotfiles):
+- rulesets cac1aa1 — SVG-first rendering rule (emacs.md + ui-prototyping.md cross-link), jotto proposal accepted, reply sent.
+- rulesets 8703150 — Showing Craig Visuals rule (interaction.md) + zsh word-split note (protocols.org canonical+mirror), takuzu proposal accepted with corrections, reply sent.
+- rulesets d5bc9b3 — model opus→fable (Craig: keep). rulesets dc13bce — todo.org Signal-pager fold.
+- dotfiles 5055ef1 — tmux terminal-features sixel lines, COMMITTED AND PUSHED to origin; inbox handoff there rewritten as applied-FYI.
+- Handoffs delivered: jotto (acceptance), takuzu (acceptance + capability correction), emacsd (eat-xtwinops.patch + intro), dotfiles (superseded by direct apply).
+
+Uncommitted work: todo.org (the :LAST_REVIEWED: 2026-07-13 stamp on the build-to-prototype task — one drawer line, safe to commit with the next housekeeping/wrap). .ai/session-context.org is this live anchor (untracked, stays).
+
+Key findings not recorded elsewhere: none — the sixel investigation's durable facts live in the two rule commits, the emacsd/dotfiles handoffs, and this log. The four unpushed rulesets commits are the main crash exposure; push happens at wrap.
+
+Background work: none running. tmux sixel test windows killed; *sixel-probe* buffer killed; tmux server carries runtime-only settings (terminal-features sixel entries, allow-passthrough all) that are now redundant with the pushed dotfiles conf (passthrough) or pending the .emacs.d patch (features).
+
+Resume hint: start the org-table helpers bug via TDD (regression test first: a #+begin_example block of pipe-prefixed lines must survive wrap-org-table.el byte-identical), and fold in the todo.org:1314 link fix as the rider.
+
+** 2026-07-13 Mon 05:14 — Startup + inbox processing begins
+
+Ran startup: rulesets pull skipped (dirty tree — .claude/settings.json shows model opus→fable, same harness flip Craig reverted on 2026-07-11), nothing new to link, no branches behind, .ai/ synced from templates. No crashed session. Notes.org: no reminders, no pending decisions. Staleness: 18 top-level tasks unreviewed >7 days. Roam inbox: 8 items, none rulesets-owned (4 takuzu, 3 archsetup, 1 emacs.d).
+
+Inbox had 3 handoffs. Processed the home FYI (2026-07-11-1208, Signal pager ack): folded as a dated sub-entry into the [#C] "Document (and own) the Signal pager" task body — home confirms rulesets owns it, and reports signal-mcp wasn't connected in its 2026-07-09 session and its signal-cli is registered as Craig's own number (note-to-self pushes nothing), so home currently has no live page channel. Deleted the inbox file. No reply sent — the handoff was itself home's ack of our earlier reply; nothing to close.
+
+Remaining: two shared-asset proposals (jotto SVG-rendering rule, takuzu EAT-image-display rule) — skeptical reviews done, surfacing to Craig for approval next.
+
+** 2026-07-13 Mon 05:25 — jotto SVG rule shipped
+
+Craig approved option 1 (accept with changes). Added the SVG Rendering section to claude-rules/emacs.md (consider-by-default phrasing, full constraint sheet, hybrid guidance verbatim) and the svg.el port-target paragraph to claude-rules/ui-prototyping.md section 3. /review-code --staged: approve, one Minor wording nit self-fixed (1:1 repetition). make test green (273 ok, ERT clean — the 9 grep hits were test names containing "error"). Commit cac1aa1 as Craig. Replied to jotto via inbox-send.py (landed in jotto's inbox), deleted the inbox file. Note: bare inbox-send isn't on PATH in this shell — used .ai/scripts/inbox-send.py directly. Settings.json model flip still undecided (Craig hasn't answered keep-or-revert). Next: takuzu EAT-image proposal.
+
+** 2026-07-13 Mon 09:40 — EAT sixel capability verified live (takuzu claim corrected)
+
+Craig challenged takuzu's "EAT can't display images" claim; verified it's wrong on capability. Findings: EAT 0.9.4 fully implements Sixel — advertises it in DA (\e[?12;4c, feature 4 = sixel), answers XTSMGRAPHICS, renders via eat-sixel-render-formats (xpm first; daemon has xpm+svg). Live test: magick logo → sixel, wrapped in tmux passthrough DCS (ESC Ptmux; with doubled ESCs), cat from a VISIBLE pane → full-color 640px image rendered in the EAT buffer (Craig's screenshot 09:36:42). Caveats found: (1) image is a transient screen-anchored overlay — any scroll/redraw drops it; (2) tmux allow-passthrough "on" drops passthrough from invisible panes — set allow-passthrough "all" globally at RUNTIME ONLY (needs tmux.conf for persistence, both daily drivers); (3) img2sixel/libsixel 1.10.5 on ratio is broken — exit 0, zero bytes, all formats; use magick ... sixel: instead; (4) writes into the Claude Code TUI pane get instantly repainted over — draw in a separate visible pane. Test window killed, cleanup done. Next: re-present takuzu disposition with corrected facts.
+
+** 2026-07-13 Mon 09:55 — Root cause of sixel vanishing: tmux grid, not EAT
+
+Craig asked why the image is irretrievable once offscreen. Isolated it: created a background plain-EAT buffer (*sixel-probe*, eat-make, no window shown) in the daemon, fed it the RAW sixel (no tmux), scrolled 60 lines, re-scanned: all 60 image cells persist at pos 163, before display-begin 654 — EAT keeps sixel in scrollback natively (each cell is a buffer char with an XPM-tile display property, per eat--t-sixel-flush-line). So tmux is the killer: passthrough bypasses tmux's grid, tmux's model of those lines is text-only, and every repaint (scroll/window-switch/output) rewrites them, destroying EAT's image cells. Nothing upstream stores the pixels → irretrievable. Follow-up lead: tmux 3.7b binary has sixel compiled in, but client_termname is xterm-256color (TERM override somewhere in the launch chain — EAT would normally set eat-truecolor), whose terminfo doesn't advertise sixel, so tmux's native sixel path never engages. Native path would need terminal-features 'xterm*:sixel' + client reattach to test. *sixel-probe* buffer left alive so Craig can see the scrollback image; kill at wrap.
+
+** 2026-07-13 Mon 11:00 — EAT XTWINOPS spike: full native-sixel fix, verified, handed off
+
+Craig chose the spike (option 1). Found the final gate in tmux source (cloned 3.7b shallow to scratchpad): tty_cmd_sixelimage falls back to the text placeholder when TERM_SIXEL is unset OR tty xpixel/ypixel are 0. First was cleared by terminal-features 'xterm*:sixel'; second was the blocker — Emacs never fills pty pixel fields, and tmux's fallback query (CSI 18t + CSI 14t at attach) is silently swallowed by EAT (no 't' case in its CSI dispatch; confirmed client_cell_width 0x0). Spike: copied eat.el to scratchpad, added eat--t-send-window-size-report (answers CSI 14/16/18 t from char-width/height + disp dims, mirroring the XTSMGRAPHICS reply) + one dispatch case. Byte-compiled clean, loaded into the live daemon (survived — this session runs in EAT), probe confirmed replies (R14 4;576;880, R16 6;24;11, R18 8;24;80). After Craig's reattach: client_cell_width 11x24, and the invisible-window grid-redraw test PASSED — wizard image drawn from tmux's grid, survives window switches, scrolling, AND resizing (Craig verified visually). Note: first probe buffer vanished mid-session (unexplained; daemon uptime 1.5d rules out restart), recreated fine. img2sixel (libsixel 1.10.5) broken on ratio — magick sixel: is the working encoder. Velox verified over tailscale: same sixel-enabled tmux 3.7b + imagemagick — ready once configs land. Handoffs sent: emacsd got the eat-xtwinops.patch + intro (durable home, TDD productionization, upstream to codeberg akib/emacs-eat); dotfiles got the two tmux.conf lines. Everything on ratio is RUNTIME-ONLY until those land (Emacs restart loses the patch, tmux server restart loses the flags). Remaining: takuzu disposition (rule now reflects working persistent sixel), model-flip question, cleanup (sixel-final window, *sixel-probe* buffer).
+
+** 2026-07-13 Mon 12:00 — takuzu rule shipped, dotfiles committed + pushed
+
+Craig approved option 1 and directed the dotfiles commit+push. Applied: (1) rulesets 8703150 — Showing Craig Visuals section in interaction.md (never rely on SendUserFile display; browser/imv lane per desktop-capture.md; sixel quick-glance lane behind the client_cell_width capability gate) + zsh no-word-split note in protocols.org canonical + mirror (sync-check --fix). Suite green 423/0. Corrective acceptance reply sent to takuzu (its capability claim corrected with the CSI 14 t findings); inbox file deleted — local inbox now clear. (2) dotfiles 5055ef1 committed as Craig and pushed to origin (cjennings.net): two terminal-features sixel lines in common/.tmux.conf (allow-passthrough already on); dotfiles suite 10/10 OK; the pending rulesets handoff in dotfiles/inbox rewritten as an already-applied FYI. Cleanup done earlier this turn-series: sixel-final tmux window killed, *sixel-probe* buffer killed. Plan: Craig fast-tracks the emacsd patch, then a velox end-to-end test after velox pulls both repos. Still open: model flip (settings.json, unanswered), todo.org Signal-pager fold uncommitted.
+
+** 2026-07-13 Mon 13:35 — resumed: runtime-portability work begun, three inventories done
+
+Craig pulled the pin ("commit and push that, then let's work on all those remaining tasks"). Pushed 828f70a..1b638e6 (5 commits incl. the child-task filing), then found and fixed a defect: d5bc9b3 had committed opus→Opus (harness rewrote the file to "Opus" before staging), NOT the fable flip Craig chose — corrected + pushed as e91073d. Mid-work inbox arrival: .emacs.d ack of the EAT patch (filed [#B] there, design call going to Craig) — pure FYI, deleted. Then the three inventory children: wrote docs/design/2026-07-13-runtime-portability-inventories.org (hooks: only PreCompact + Stop carry porting work, AskUserQuestion moot, validators ride githooks; MCP: nine local servers portable, signal-mcp is claude.ai-side only so paging has NO off-Claude path — Signal-pager task is the fix; memory: KB already cross-agent, one sentence gap in knowledge-base.md). Converted the three children to dated entries, added a *** VERIFY carrying the four decisions (all recommended yes). todo.org + new doc uncommitted. Next children: skill parity (decision matrix), instruction bootstrap, launcher flag, session plumbing, local model floor.
+
+** 2026-07-13 Mon 12:15 — task review stopped early at Craig's call
+
+Craig picked "2 then 1" (task review, then org-table bug). Review batch was the 7 oldest-unreviewed; task 1 (build-to-prototype rule extension [#B]) kept as-is and stamped :LAST_REVIEWED: 2026-07-13 (assessed not-quick, not-solo — placement decision pending). Craig then said "let's stop here" — review closed early, 6 tasks untouched for the next run. The org-table bug (queued item 1) not started. todo.org edit uncommitted; ambiguity noted: "stop here" might mean stop-review or end-session — asked Craig which.
+
+** 2026-07-13 Mon 12:05 — model flip kept, housekeeping committed
+
+Craig chose to keep the fable model flip (deliberate this time, unlike the 2026-07-11 revert). Committed d5bc9b3 (settings model opus→fable) and dc13bce (todo.org Signal-pager fold). Both mechanical, subject-only messages; the 11:45 green suite run covers this tree state. Working tree now clean except the live session-context file. Rulesets has 4 unpushed commits (cac1aa1, 8703150, d5bc9b3, dc13bce); push at wrap per usual flow. Session remaining: nothing pending from the inbox; awaiting Craig's next move (emacsd fast-track, then velox test).
diff --git a/.ai/sessions/2026-07-14-02-50-sentry-spec-review-lint-org-fix.org b/.ai/sessions/2026-07-14-02-50-sentry-spec-review-lint-org-fix.org
new file mode 100644
index 0000000..280ab5b
--- /dev/null
+++ b/.ai/sessions/2026-07-14-02-50-sentry-spec-review-lint-org-fix.org
@@ -0,0 +1,84 @@
+#+TITLE: Session Context — 2026-07-14
+#+DATE: 2026-07-14
+
+* Summary
+
+** Active Goal
+
+Take the work project's sentry proposal (one supervisor loop running a project's hygiene passes overnight, with locking) from inbox arrival to a fully-dispositioned spec: decision walk, spec-create, adversarial review, findings walk. Mid-walk, fix the lint-org data-corruption bug the review surfaced as a sentry prerequisite. Plus two inbox waves (8 handoffs total) processed along the way.
+
+** Decisions
+
+- Sentry design, all resolved live with Craig (recorded as 10 DONE decisions + 12 DONE findings in the spec): host-suffixed sentry branch (Craig's call — unpushed commits on main diverge across two daily drivers); host-local locks under /run/user/<uid>/agent-locks/ (a lock inside the roam repo would ride roam-sync's git add -A to the other machine); mkdir-atomic lock helper with heartbeat refresh + bounded-wait-then-defer (flock can't span tool calls); interactive entry gates, NO report-only mode anywhere (Craig's explicit direction — passes run fully or skip); roam-sync stays the roam repo's only committer; :COMMIT_AUTONOMY: yes gates sentry entirely; detection-over-configuration pass portability; launch contract with in-place checkout (worktrees rejected: untracked inbox drops invisible there); inbox pass under the no-approvals contract; suite at entry + conditional fire-end run only; KB promotion pass cut to vNext (heuristic task filed); wrap-up refuses during a live loop, "stop sentry" owns shutdown; entry ff-only reconcile; persistent stall notify after 2 skips.
+- Craig corrected a misread mid-turn: the .emacs.d signel request meant "remove signel mentions from live workflows," not "run a broadcast + question smoke's scope." No broadcast ran; smoke's scope stands; nothing had been sent.
+- Spec stays DRAFT deliberately — Craig reviews deeply before the READY flip.
+
+** Data Collected / Findings
+
+- The 2026-07-09 org-file corruption's true mechanism: wrap-org-table.el's load-time CLI dispatch fired when lint-org.el merely required it, running the table reformatter over lint-org's file arguments (--check runs were immune because the flag made wot's dispatch decline). The filed defects (block-blind scanning, mutate-by-default) were real but secondary.
+- flock binds to a living process; agent Bash calls are short-lived shells, so locks for agent workflows need mkdir-atomicity + staleness semantics, never flock.
+- rg skips hidden directories by default — a grep over .ai/ paths silently misses everything without --hidden (bit this session during the signel audit).
+- No locking existed anywhere in .ai/scripts/ (verified); the roam sync-conflict forks are the observed cost.
+
+** Files Modified
+
+- docs/specs/2026-07-14-sentry-workflow-spec.org — NEW: the sentry spec, DRAFT, 10/10 decisions, 12/12 findings resolved, voice-passed. docs/design/2026-07-14-sentry-workflow-proposal.org — NEW: preserved origin proposal + cover note.
+- Commit 951b6fc — lint-org/wrap-org-table corruption fix, test-first (6 new tests): block-type-aware scanning, lint-org CLI report-only by default (--fix required; wrap-it-up + /lint-org command updated), wot entry-script dispatch guard. Commit 32afa41 — work's .flac threading adopted into process-meeting-transcript.org. Plus the signel history-line chore. All pushed (ad5bb07).
+- todo.org — sentry build task [#B] + cross-host [#D] + /schedule [#D] + KB-heuristic [#D] filed; org-table bug closed DONE and archived to Resolved.
+- notes.org — :LAST_INBOX_PROCESS: 2026-07-14; Active Reminder: Craig reviews the sentry spec before ending the next session.
+
+** Next Steps
+
+- FIRST: Craig's deep read of the sentry spec (his explicit ask — surface the reminder at startup). READY flip, then spec-response decomposes the build.
+- The paused task review: batch of 7 selected, task 1 (sentry build) walked but no action picked; 6 untouched. The sentry build task is also the one 30-day-staleness hit (never-reviewed, filed tonight) — the next review cycle covers it.
+- Sentry build follows the spec's 4 phases once READY (agent-lock helper first).
+
+KB: promoted 2 / consulted no
+
+* Session Log
+
+** Startup + inbox processing
+
+Session started 2026-07-14 00:11 CDT on ratio. Fresh session, templates synced, no reminders, staleness 0, roam inbox 0.
+
+Inbox: three items. (1+2) Sentry workflow proposal from work (intro + full proposal) — passed value gate (Q2), skeptical review found: real problem (verified no flock anywhere in .ai/scripts/), sound invariants, but four issues: roam lock held across commit-push conflicts with inbox.org's edit-then-roam-sync discipline; flock can't span short-lived tool-call shells (needs PID/staleness lockfile or mkdir semantics); unattended per-pass commits need gating on :COMMIT_AUTONOMY:; passes 3 and 10 are sender-specific (need auto-skip/config). Craig picked option 2: work it now, walk decisions, spec/build tonight. Then a task review. (3) archsetup FYI (maint update-queue CVE item done, dotfiles 6757b51) — acknowledged, deleted. Craig also directed deleting the standing inbox/lint-followups.org.
+
+Proposal preserved to docs/design/2026-07-14-sentry-workflow-proposal.org (cover note merged in). Reply sent to work: accepted, being worked now.
+
+** Sentry decision walk (all decided with Craig, one by one)
+
+- D1 commit target: dedicated sentry branch, host-suffixed (sentry/YYYY-MM-DD-<host>). Craig's call, overriding my current-branch rec: unpushed commits on main across two daily drivers = diverged main, breaks startup ff + template sync. Morning flow: review, squash-merge, delete branch. Prior unmerged sentry branch at entry → skip fire.
+- D2 roam-write lock scope: host-local only; cross-host conflict-surfacing filed as follow-up.
+- D3 interval: hourly default, config knob.
+- D4 gates: interactive at entry (Craig is at terminal when he launches) — dirty tree stops and surfaces what's dirty + options (finish the job / stash / rollback which changes); red suite same shape. NO report-only mode anywhere (Craig's explicit direction): unattended fire hitting unsafe state skips with one digest line; next fire retries. Morning-approval queue for destructive/judgment actions stays (permanent division of labor, not a degraded mode).
+- D5 roam git: roam-sync stays the only committer; lock wraps edit + trigger. Reconciliation surfaced: knowledge-base.md + roam-sync.sh header still say agents pull/commit/push inline — join companion-file list.
+- D6 locks: shared .ai/scripts/ lock helper, mkdir atomicity, PID/host/timestamp metadata, age-based staleness reclaim, acquire/release/status, bats-tested. (flock can't span tool calls.)
+- D7 autonomy: :COMMIT_AUTONOMY: yes gates sentry entirely — absent/no → declines to start, names the marker.
+- D8 portability: auto-skip by detection (existence probes per pass), no config.
+- D9 path: spec-first (spec-create → docs/specs/2026-07-14-sentry-workflow-spec.org).
+
+** Sentry spec authored
+
+Wrote docs/specs/2026-07-14-sentry-workflow-spec.org (DRAFT, ID f6c51f27-d7a2-4b63-9ff9-5ba005a66dfb): full spine — summary, problem, goals/non-goals/scope, two-altitude design (user + implementer), 5 alternatives with why-nots, 9 DONE decisions, 4 implementation phases, acceptance criteria, readiness dimensions, risks. Voice pass ran (/voice general per spec-create Phase 4): ~20 em-dashes reduced to structural-only, 2 prose fragments completed, passives activated, one correctness catch (design said "never pulls" roam but pass 1 is an ff-only pull — reconciled). Self-rubric: Ready. Cross-linked: todo.org gets the [#B] build task + two [#D] vNext deferrals (cross-host conflict surfacing, unattended /schedule variant). Next: spec-review gate, then Craig's queued task review.
+
+** Adversarial spec review (Craig's weakness-loop method)
+
+Task review started (batch of 7, task 1 = sentry build) but Craig redirected: run the spec through review now via a weakness loop — 20 weaknesses, check each against the spec, loop until <5 new, then notify him for deep review. Ran 3 rounds (20 → 7 → <5). Recorded 10 blocking + 2 non-blocking findings in the spec's Review findings section. Heavy hitters: overnight working-tree ownership (sentry checkout vs Craig's live session — biggest open design question); roam-write lock is sentry-only unless inbox.org/knowledge-base.md writers also acquire it; lock dir inside roam repo would be committed by roam-sync's git add -A (relocate to /run/user); sentry's session-context writes trip its own dirty-tree skip; pass 8 depends on the filed lint-org mutate-on-lint [#B] bug; unattended inbox semantics; per-pass commits vs verification.md suite rule (needs a recorded decision); KB lesson source undefined; lock wait/heartbeat mechanics; wrap-up-during-loop undefined. Rubric: Not ready; status stays DRAFT. Iteration history entry added. Craig notified for his deeper review. Task review batch (7 tasks) still pending — resumes after.
+
+** Findings walk + lint-org fix shipped mid-walk
+
+Walking the 12 findings with Craig one by one. Decisions so far: F1 = launch contract (in-place checkout, sentry owns the repo overnight, dirty-skip backstop, Emacs revert caveat documented). F2 = all roam writers acquire the lock (inbox.org core §5 + knowledge-base.md write recipe gain acquire/release; graceful degradation when helper absent; bounded-wait-then-surface for interactive callers). F3 = both locks under /run/user/<uid>/agent-locks/, helper owns path scheme, ~/.cache fallback. F4 = spine-set exclusion in dirty check + fire-end digest commit + path via session-context-path.
+
+At F5 Craig redirected: fix the lint-org bug now. Done, TDD (red shown first): block-type-aware scanning in wrap-org-table.el wot-process-file + lint-org.el lo--check-tables (type-matched end marker so literal inner markers can't re-expose a block); lint-org CLI report-only by default, writes behind --fix (--check kept as alias; wrap-it-up + /lint-org command updated to pass --fix). Root cause discovered during red-phase debugging: wrap-org-table's load-time CLI dispatch fired on lint-org's require and ran the reformatter over lint-org's file args — THE 2026-07-09 corruption mechanism (--check runs were immune because the flag made wot's dispatch decline). Entry-script guard added (dispatch only when -l names wrap-org-table.el itself). /review-code --staged ran: one Important (boolean in-block flag cleared by literal inner end-marker) fixed in-review + regression test; verdict Approve. Suite 0 failures. Committed 951b6fc (unpushed). todo.org bug task closed DONE; spec finding F5 completed DONE. Minor follow-up noted (not filed): lo--check-indented-headings still has the naive boolean flag, benign (suppression-only).
+
+Second archsetup FYI (tailscale-amber fixed, dotfiles 4840463 + archsetup aed151b, velox TOML hand-refresh is archsetup's item) acknowledged + deleted. Inbox clean.
+
+Next: resume walk at F6 (unattended inbox semantics), then F7-F12, then Craig's deep spec review, then the paused task review (task 1 of 7 pending — sentry build task, now also needs the F5 resolution reflected... its body references the spec; fine as-is).
+
+** Findings walk completed — all 12 dispositioned, spec updated
+
+F6 = inbox pass under the no-approvals contract, everything to the morning queue. F7 = suite policy recorded as Decision 10 (entry run + conditional fire-end run when non-org files changed, no per-pass runs). F8 = KB promotion pass CUT to vNext (Craig), lesson-detection-heuristic task filed [#D] :spec: (blocks re-adding). F9 = bounded-wait-then-defer + heartbeat refresh, reclaims surfaced. F10 = wrap-up REFUSES during live sentry (Craig picked simple over shutdown-pre-step); "stop sentry" operation owns loop-cancel + branch/queue disposition. F11 = entry ff-only reconcile, diverged joins interactive gate. F12 = persistent notify after 2nd consecutive unmerged-branch skip, then at most daily. All 12 findings flipped DONE in place; design/scope/decisions/acceptance/risks updated to match (pass list now 10, Decisions [10/10], locks at /run/user/<uid>/agent-locks/, launch-contract paragraph in user altitude). Status stays DRAFT for Craig's deep read; responder entry added to iteration history.
+
+** Second inbox wave (5 handoffs) + Craig's broadcast correction
+
+Two more archsetup FYI closures (memory-killer = live-refresh hairline artifact, filed [#C] their side; net-panel speedtest history filed [#C] their side) deleted. work's process-meeting-transcript .flac update: diff-verified verbatim (12 hunks, pure .flac threading), adopted into canonical, committed 32afa41, confirmation sent to work. .emacs.d's signel-deprecation broadcast request: I misread it as run-a-broadcast + flag smoke's scope; Craig corrected mid-turn — NO broadcast, smoke's scope is well defined and needed, and the actual ask was removing signel mentions from live workflows. Broadcast draft deleted from /tmp before anything sent; nothing reached smoke. Grep gotcha: rg skips hidden dirs by default, so .ai/ paths need --hidden. One live-workflow mention found (triage-intake.org:417 incident history line) — rewritten to keep the lesson, drop the client name (commit next to 32afa41: chore(workflows): drop retired Signal-client name). Session archives + frozen design doc keep theirs as history. Reply sent to .emacs.d. Inbox clean (0 pending).
diff --git a/.ai/sessions/2026-07-17-10-55-inbox-fixes-emacs-wayland-sentry-prep.org b/.ai/sessions/2026-07-17-10-55-inbox-fixes-emacs-wayland-sentry-prep.org
new file mode 100644
index 0000000..da2ca40
--- /dev/null
+++ b/.ai/sessions/2026-07-17-10-55-inbox-fixes-emacs-wayland-sentry-prep.org
@@ -0,0 +1,238 @@
+#+TITLE: Session Context
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-16
+
+* Summary
+
+** Active Goal
+
+Process the project inbox (5 handoffs across the session), then remove a dormant clone, switch ratio's Emacs to the pgtk build, and prep the sentry spec for Craig's READY-gate review. Two of the five handoffs were bug fixes that shipped; two were design questions filed for a Monday scouting pass; one was rejected on evidence.
+
+** Decisions
+
+- *gmail residue-probe fix (home):* took home's one-word epoch fix, but hoisted the rule to cover both anchored queries once and scoped it to the anchor windows so the deliberate day-resolution date-slicing isn't implicated. Dropped home's proposed count-check — it depends on resultSizeEstimate + a 100-cap the same file documents as untrustworthy.
+- *secret-scan fix (takuzu):* accepted their case-sensitivity split (measured: ~6% false-positive on 100KB base64 → 0), rejected the =;base64,= line-skip (guards a ~1-in-1.7M event, costs a real blind spot on minified data-URI lines — reproduced). All three variants (elisp/bash/go).
+- *subproject promotion (home):* deferred, not rejected. N=1 (home is the only project with subprojects across 27 scopes); promoting 282 lines into the always-on rules layer contradicts the proposal's own thin-always-on principle. Scheduled 2026-07-20 for a scouting pass.
+- *polyglot bundle collision (home):* fixed the silent half (install-lang guard refuses a colliding second bundle), deferred the design half. Reframed: the guard is NOT "polyglot unsupported" — non-overlapping pairs (bash+python) compose today. The line is overlap-vs-not, and nobody chose it. Scheduled 2026-07-20, paired with subprojects (same scouting question).
+- *wl-copy proposal (clock-panel):* REJECTED on evidence. Their diagnosis was inverted — plain =wl-copy= persists (while its process lives); =--foreground= is what disables the fork. Failed review Q1 (is it right), not the value gate.
+- *settings.json model:* fable → opus, Craig's call (bd76d98). Reverses the deliberate 07-13 fable pin; the field has flip-flopped 4x. =~/.claude/settings.json= symlinks here, so machine-wide.
+- *gloss:* removed the local clone (Craig's direction), remote deliberately left alive. Verified fully pushed before deleting.
+- *ratio Emacs:* installed =emacs-wayland= (pgtk), restoring archsetup's own intent (line 2605, unconditional since 2026-04-11); ratio had drifted for ~3 months. Craig restarted; verified pgtk/native-Wayland.
+
+** Data Collected / Findings
+
+- *gloss:* 24 commits all pushed, live bare remote at 7073b16; local-only =.ai/=/=todo.org=/=inbox/= all regenerable or moot. Safe to delete.
+- *Emacs on ratio was the X11 build* (=extra/emacs=, =:pgtk nil :x t=), on XWayland via DISPLAY=:0 while WAYLAND_DISPLAY sat unused — why xclip worked despite the pure-Wayland rule. archsetup line 2605 says =emacs-wayland=; ratio drifted. archsetup's package tracking is a before/after run diff (=comm -13=, "Statistics"), never an intent-vs-reality audit — the class of drift that hid this.
+- *=emacs.service= failed since 2026-07-09* (start-limit-hit, 5x exit-1, cause undiagnosed — no Emacs stderr in journal). Live daemon is a bare =emacs --daemon= PPID 1, unsupervised. Unit is dotfiles-owned (stow symlink into =~/.dotfiles=, its own repo/scope), package is archsetup's.
+- *I broke Craig's clipboard* testing clock-panel's claim — killed wl-copy owners repeatedly; each kill emptied the clipboard (owner holds the selection only while alive). Restored via Emacs. This corroborated clock-panel's symptom while disproving their mechanism.
+- *sentry spec bookkeeping was broken:* Decision 10 (suite policy) was misfiled under Review findings; Decisions cookie hand-typed [10/10], findings cookie dead =[/]=. Moved it, let org recompute → [10/10] and [12/12], both live. Zero new lint findings vs HEAD.
+- *pkill -f self-matches the shell running it* (exit 144, twice) — promoted to KB.
+
+** Files Modified
+
+- Commits (all pushed to origin/main): cf3eadc (gmail epoch fix + mirror), 794a8bd (secret-scan case-split, 3 variants + 9 bats), c98fda5 (install-lang collision guard + 9 bats + report preserved), the subproject-deferral docs commit, bd76d98 (opus model), 0cc1256 (sentry spec cookie fix).
+- =todo.org=: 2 tasks filed [#C] :spec: SCHEDULED 2026-07-20 (subproject promotion, polyglot support), both carrying the review findings as decision inputs.
+- =docs/design/=: 3 files preserved (subproject proposal + home instance, polyglot collision report).
+- KB: =agents/20260717105401-pkill-f-self-match.org= written + pushed.
+- Replies sent: home ×3 (probe, subprojects, polyglot), takuzu ×1 (secret-scan), clock-panel ×1 (wl-copy reject), archsetup ×2 (emacs drift + package-audit follow-up). gloss local clone removed.
+
+** Next Steps
+
+- *FIRST (Craig's standing ask):* deep-read the sentry spec (=docs/specs/2026-07-14-sentry-workflow-spec.org=), now structurally clean. Flip DRAFT → READY, then spec-response decomposes the build. The load-bearing decision to weigh: launching sentry hands it the repo until the morning merge.
+- *2026-07-20 scouting pass:* subproject promotion + polyglot support, one conversation — "which projects would actually be polyglot / have subprojects, and why."
+- *Owed, unsent:* home's applicability line for github-prs + personal-calendar (residue probe N/A there — one query, no seam). Wording-only.
+- *Unread inbound:* clock-panel's wl-copy review correction (arrived 14:56) — deferred to next session, non-urgent.
+- *Open elsewhere:* velox unverified for the emacs-wayland drift (offline); =emacs.service= failure undiagnosed (dotfiles' + archsetup's, handed off); archsetup asked whether it wants an intent-vs-reality package audit.
+
+KB: promoted 1 / consulted yes
+
+* Session Log
+
+** 2026-07-16 09:11 CDT — Startup
+
+Ran the startup workflow. Phase A.0: rulesets pull skipped (dirty tree — =.claude/settings.json= carries an uncommitted =fable= → =opus= model flip against a committed =fable= pin Craig deliberately kept on 07-13). =make install= had nothing new to link. Project fetch/reconcile clean.
+
+Phase A: no crash anchor (previous session wrapped cleanly). =.ai/= synced from templates, no churn. Staleness zero, language-bundle silent, spec-sort and host-identity probes silent. Roam inbox: 1 item. KB: 87 =:agent:= nodes, no project matches, best-practices node path not resolved.
+
+Phase B: read the three most recent session summaries. Inbox carries 6 new files = 3 handoffs (home's triage-intake probe-gap fix + edited plugin; takuzu's pre-commit secret-scan false-positive fix + fixed hook; home's subproject-pattern promotion proposal + reference =subprojects.org=). All three are shared-asset proposals, so all three go through the inbox engine's skeptical review rather than acting silently.
+
+Craig picked: work the inbox in order (two bug fixes first, subproject proposal last).
+
+** 2026-07-16 10:44 CDT — Inbox item 1: home's gmail residue-probe gap (landed cf3eadc)
+
+Home's diagnosis verified against the canonical and accepted. The probe cut off at =before:<anchor-YYYY/MM/DD>= (day resolution) while the scan started at =after:<anchor-epoch>= (second resolution), and Gmail's =before:<date>= excludes the named day, so the anchor day belonged to neither query and each sweep advanced past it without revisiting. An evening anchor hid most of a day. Exactly the failure the probe was added 2026-07-08 to prevent.
+
+Skeptical review (cross-project battery) produced three findings beyond home's ask:
+
+1. *No engine change needed* — the plugin declares =ANCHOR: epoch= and the engine already computes + substitutes every format, so =before:<anchor-epoch>= works as-is. Verified in =triage-intake.org= (the ANCHOR contract at line 266).
+2. *Answered home's "do other plugins share this?" — no.* Checked all five: cmail and telegram are =ANCHOR: none= (state-based, no since-window, no second query to gap); github-prs and personal-calendar are =ANCHOR: iso8601= but carry no residue probe, so there's only one query. personal-gmail is the only source pairing an anchored scan with a residue probe.
+3. *Declined home's count-equality verification check.* It depends on =resultSizeEstimate= and a full unread count, and the same file documents at lines 28/31 that the estimate is unreliable (stuck at "201" across queries whose union exceeded 300) and the MCP caps at 100 with no =pageToken=. The check would have pointed the next reader at numbers the plugin already says not to trust.
+
+Applied home's one-word fix, plus the rule consolidation home themselves suggested: the epoch rule now states once that it governs *both* anchored queries, and the probe cross-references rather than restating. Critically, the rule had to be *scoped to the anchor windows* — the date-slice walk at line 30 uses =before:<oldest-full-day-seen>= at day resolution deliberately, safe because slices overlap and dedupe by message id. A blanket "always epoch" rule would have implicated it.
+
+=/review-code --staged= surfaced one Minor I fixed before committing: the two epoch examples were ~2 months apart (=after:1778856990=, =before:1784177122=) in a paragraph whose whole point is that both queries must meet at the same instant. Now both show =1784177122=.
+
+Canonical edited first, mirror synced via =scripts/sync-check.sh --fix=, both sides identical. =make test= green (374 pytest + 67 + 12 + ERT + bats, zero failures). Commit cf3eadc, *not yet pushed*.
+
+Replied to home: unblocked (they'd filed it =[#B] :infra:blocker:=), answered both their questions, named the two declines with reasons, and raised one thing for their eyes — github-prs and personal-calendar have an anchored scan and no residue probe at all, so they're blind to pre-anchor items with nothing to catch it. Different question from home's gap; may be fine for open-PR/calendar state.
+
+Deleted both inbox files for the item.
+
+*Observation, not filed:* =cross-project.md= shows bare =inbox-send <target>= invocations, but it's not on PATH — only =.ai/scripts/inbox-send.py= exists, and =claude-templates/bin/= carries just =agent-page= and =ai=. The prose names the script path correctly; the examples imply a command that doesn't resolve. Worth a decision (link it in =make install=, or fix the examples).
+
+** 2026-07-16 11:20 CDT — Inbox item 2: takuzu's pre-commit secret-scan fix (landed 794a8bd)
+
+takuzu reported the hook false-positiving on an embedded PNG sprite data URI, forcing =--no-verify=. They diagnosed two causes and sent a fixed elisp-variant copy. I accepted cause 1, *measured and rejected cause 2*.
+
+*Cause 1 (accepted, correct).* =grep -iE= applied case-insensitivity to the fixed-case AWS token, so =AKIA[0-9A-Z]{16}= matched any mixed-case 20-char run (the =[0-9A-Z]= class also goes case-insensitive under =-i=). Their fix splits fixed-case tokens (AKIA, sk-, PEM) into a case-sensitive pass, leaving only keyword=value patterns under =-i=.
+
+*Cause 2 (rejected on evidence).* Their claim: "even case-sensitive, a 100KB base64 blob can contain AKIA+16 uppercase by chance," justifying a skip of any line containing =;base64,=. I generated ~10MB of real random base64 (100 sprite-sized blobs) and measured:
+
+- case-insensitive (the shipped hook): *7 matches, 6 of 100 blobs would false-positive* (~6%) — this is the live bug.
+- case-sensitive (their fix): *0 matches.*
+- theoretical rate for the residual: ~1 in 1.7M per 100KB blob ((1/64)^4 x (36/64)^16 x 1e5 positions).
+
+So cause 1 is the entire bug; cause 2's justification is ~1e-6. And the skip *costs* something: it drops whole lines. I verified with a minified-bundle line carrying both a data URI and a live =api_key= — under their fix the line is skipped and the key sails through; without it, the key is caught. Minified JS/CSS is exactly where data URIs live and where a bundle is one enormous line, so the blind spot lands precisely where both can co-occur.
+
+Craig picked option 1: case-sensitivity split only, drop the skip, with tests.
+
+*TDD.* Wrote =scripts/tests/pre-commit-secret-scan.bats= first (9 cases across all three variants). Red confirmed: cases 5 and 6 failed on the unfixed hooks, reproducing the live failure. Case 7 (a real credential sharing a line with a data URI) is the guard against anyone re-adding the skip. Green after the fix. Verified the new bats file is discovered by the Makefile glob rather than assuming it (=verification.md='s "a passing gate can skip your new file").
+
+*Self-review caught one regression.* Splitting one grep into two meant a line matching both passes printed twice, reading as two separate leaks. Verified empirically, wrote a 10th test (red), fixed with =awk '!seen[$0]++]'= (green). Verdict was Request Changes on my own diff until that landed.
+
+Applied to all three canonical variants (=languages/{elisp,bash,go}/githooks/pre-commit=); scan block verified byte-identical across them by md5. =bash -n= clean, shellcheck clean modulo two *pre-existing* findings I confirmed against HEAD and left alone: SC2164 on the unguarded =cd "$REPO_ROOT"= at line 8 of elisp+go (bash has =|| exit 1=), and SC2016 on go's gofmt printf. =make test= exit 0, zero failures.
+
+Replied to takuzu with the measurement, the blind-spot repro, the honest caveat (I don't have their actual sprite — if it holds a genuine uppercase AKIA run, case-sensitivity alone won't clear it), credit that their two-pass plumbing is what landed, and the two pre-existing findings.
+
+*Follow-up candidates (not filed yet):* the SC2164 inconsistency across the three variants, and go's em-dash.
+
+*Two commits unpushed:* cf3eadc, 794a8bd.
+
+*Inbox grew mid-item:* two new handoffs from home arrived (polyglot-coverage note at 10:06, reply-probe at 11:14).
+
+** 2026-07-16 11:35 CDT — Pushed cf3eadc + 794a8bd
+
+Craig picked push-then-continue. Pre-push reconcile clean (0 behind, 2 ahead), one remote (=git@cjennings.net:rulesets.git=). Pushed f14dd87..794a8bd, verified upstream at 794a8bd, 0/0.
+
+** 2026-07-16 11:58 CDT — Inbox item 3: home's subproject-pattern promotion (deferred, filed [#C])
+
+home proposed promoting its subproject pattern into =claude-rules/=. I recommended a thin rule (~50 lines) over their 282-line doc; *Craig chose option 3 — defer entirely and scout first.*
+
+His framing, which is the task's actual content: scout which projects would plausibly get subprojects, and why, before shaping any rule. If he hasn't scouted by the time it comes up, offer to do it together — brainstorm candidates, explore the reasons behind each. That evidence decides drop-vs-adjust. Don't shape the rule before the scouting.
+
+*Review findings* (recorded in the task body so the decision has its inputs):
+
+- *N=1, verified.* Scanned all 27 =.ai= scopes: home is the only project with subprojects, all nine from the 2026-06-11 fold.
+- *The placement contradicts the proposal's own principle.* =claude-rules/*.md= loads into every session of every project — the always-on layer. home's doc argues that layer is "a tax paid whether or not it's relevant today" and depth belongs "one open away". At 282 lines it'd be the 3rd-largest rule and +11% on the layer (20 files, 2491 lines), so every unrelated project's session would carry a one-project convention.
+- *Precedent for the shape if it ever promotes:* =patterns.md= (29 lines, explicit "don't carry the catalog in context") and =docs-lifecycle.md= (75 lines, depth in a spec).
+- *Dangling reference:* the doc cites =claude-rules/git-hosting-privacy-model=; no such file (verified). Real content is protocols.org's gitignore-vs-track + public-reachability decision.
+- *Instance vs rule:* metrics, self-improvement log, kill criteria, rollout dates, adoption table are home's.
+
+Content preserved: both files moved to =docs/design/2026-07-15-subproject-pattern-proposal.org= and =docs/design/2026-07-15-subprojects-convention-home-instance.org=, linked from the task.
+
+*Route-candidate: deliberately NOT stamped.* =route_recommend.py= returned "home strong", but that's a false positive — it matched on "home" appearing throughout as the proposer. The work is rulesets' own (deciding rulesets' rules layer, scheduled for Craig here), so it's a local keeper and stays unstamped per core §3.
+
+Replied to home: deferred not rejected, the N=1 + placement reasoning, the two fix-regardless items (dangling ref, instance-vs-rule split), and an ask — if they have a view on which other projects would plausibly fold, that's exactly the scouting input. Also answered their probe reply: agreed on their github-prs/personal-calendar analysis (a gap needs two queries with a seam; those have one query and no seam), taking their suggested doc line to Craig with today's batch, and owned half the =:blocked:=/=:blocker:= mix-up (my reply took their tag at face value rather than checking it against the convention).
+
+*Still open:* home's polyglot bundle-collision note (the last inbox item), and home's suggested applicability line for github-prs + personal-calendar.
+
+** 2026-07-16 13:43 CDT — Inbox item 4: home's polyglot bundle collision (guard landed c98fda5)
+
+home reported that python + typescript both ship =coverage-makefile.txt=, so the second install prints =[skip]= and silently drops its fragment. *The investigation found two worse collisions home hadn't checked.*
+
+*The full collision map* (5 shared filenames, 3 real):
+
+| file | bundles | behavior |
+|------|---------|----------|
+| =gitignore-add.txt= | 5 | appended + deduped → *composes* ✓ |
+| =CLAUDE.md= | 3 | seed-only; its fallback comment shows multi-bundle was already considered ✓ |
+| =coverage-makefile.txt= | 4 (elisp,go,python,typescript) | visible =[skip]=, fragment dropped ✗ |
+| =claude/settings.json= | 3 (bash,elisp,go) | =cp -rT= *silent overwrite*, prints =[ok]= ✗✗ |
+| =githooks/*= | 3 (bash,elisp,go) | =cp -rT= *silent overwrite*, prints =[ok]= ✗✗ |
+
+*Reproduced* (elisp then bash into one project): settings.json flips validate-el.sh → validate-bash.sh; pre-commit loses check-parens; validate-el.sh left orphaned on disk; output prints =[ok] .claude/= and =[ok] githooks/=. So home's severity read was inverted — the =[skip]= they flagged at least announces itself, while these two claim success while removing quality gates. Same class as the 794a8bd secret-scan bug: a gate you believe you have and don't.
+
+*No live damage:* checked every project with a bundle; the colliding trio isn't doubled anywhere, and clock-panel (python+typescript) could only ever have hit the coverage fragment.
+
+*The fix* (Craig picked option 1: fix the silent half, defer the design half). Guard runs *before any write*, so a refusal can't half-install. Detection reuses =sync-language-bundle.sh='s rule fingerprint (project has bundle X iff one of X's rule files is in =.claude/rules/=) — no marker file, works on pre-guard installs. Refuses when the incoming bundle would overwrite a file another shipped bundle also ships, naming each file. =FORCE=1= still overrides and the message now says out loud that FORCE also re-seeds CLAUDE.md (home's catch).
+
+*TDD:* 9 bats cases, red-first. Notably *test 8 initially false-passed* — it asserted the message contained "FORCE=1" and "CLAUDE.md", both of which already appear in the pre-existing =[skip] CLAUDE.md already exists (use FORCE=1 to overwrite)= line, so it passed against an unguarded install. Tightened to assert the refusal fired first and to exclude =[skip]= lines.
+
+*Self-review caught a coverage gap.* The guard had an untested path: a *detected* bundle sharing *no* overwritten file (bash ships settings+githooks, python ships coverage only → zero overlap), which hits =[ -n "$files" ] && ...= under =set -euo pipefail=. Verified by hand that it works, then added test 7 because "worked when I ran it once" isn't coverage and a refactor to a plain =if= could turn it into a false refusal with nothing to catch it.
+
+*That gap changed the framing.* The guard is NOT "polyglot unsupported" — =bash= + =python= composes cleanly today and yields a real polyglot project with both rule sets. The line is *overlap-vs-not*, not polyglot-vs-not, and nobody chose it; it fell out of which bundle happens to ship what. Corrected the task body before filing so Monday's decision doesn't run on a wrong premise.
+
+Also caught a factual error in my own commit draft pre-voice: it credited takuzu for home's report.
+
+*Filed:* =[#C] Polyglot projects — supported, or refused? :spec:= SCHEDULED 2026-07-20, paired with the subproject task (same scouting question). Report preserved at =docs/design/2026-07-16-polyglot-bundle-collision.txt=.
+
+*Separate finding, unfiled:* gloss has =validate-el.sh= on disk but *no settings.json at all*, so its hook has never fired. Single bundle, so not this bug — a different gap.
+
+Replied to home with the full map, the inverted-severity finding, the overlap-vs-polyglot correction, and the gloss note. *Inbox is empty.* Pushed c98fda5.
+
+** 2026-07-16 14:05 CDT — Removed the gloss clone; committed the opus model switch
+
+*gloss removed* (=rm -rf /home/cjennings/code/gloss=) on Craig's direction. Verified before deleting rather than trusting the description: 24 commits all pushed (0 behind / 0 ahead after fetch), one branch, no stashes, no tags, clean tree, and =git ls-remote git@cjennings.net:gloss.git= confirmed a live bare repo at 7073b16. Code is fully recoverable by re-cloning.
+
+Local-only content assessed before the delete (=.ai/=, =todo.org=, =inbox/= are all gitignored there, so none of it was on the remote):
+- =.ai/= 1.5M but *zero* archived sessions and no live context; contents are template-synced from rulesets, so regenerable. Not a real loss.
+- =todo.org= 3 open tasks — all gloss-scoped (integration tests, gloss-core refactor, shakedown), so moot once gloss is gone.
+- =inbox/= one unprocessed handoff: rulesets' own 2026-06-12 priority-scheme ask. Also moot.
+
+Craig's "not a real work in progress" checked out (last commit 2026-05-07, no sessions). *Remote deliberately left alive* — he confirmed "leave the remote", so the project is still re-clonable; =rm -rf= on the clone is not retirement.
+
+*settings.json: fable → opus committed* (bd76d98, pushed). Craig's call. Investigated provenance rather than assuming: the file was last written *2026-07-14 15:32*, i.e. NOT this session and after the last archived session ended (07-14 02:50) — so my initial "the harness wrote it this session" read was wrong and I corrected it before reporting. The field has now flip-flopped four times (73835a2 opus→, d5bc9b3 →fable, e91073d "pin ... as intended", bd76d98 →opus). =~/.claude/settings.json= symlinks here, so it's the machine-wide default.
+
+*Process slip, owned:* I invoked the trivial-one-liner exception to skip =/voice personal= on bd76d98, then wrote a three-line body containing an em-dash. The body took it out of trivial territory; the pass should have run. Not force-pushing over punctuation, but the exception is for subject-only commits.
+
+*Correction Craig is owed on the earlier gloss finding:* I reported "gloss has validate-el.sh but no settings.json, so its hook never fired" and passed that to home. gloss is now deleted, so the note is moot — but the underlying question (does anything else have an orphaned hook?) stands unanswered for the remaining projects.
+
+*New handoff mid-work:* clock-panel sent a Wayland clipboard finding (=wl-copy= selection owner dies with the command). Its own src block arrived with literal =\n= sequences instead of newlines — the sender's formatting is mangled and worth flagging back.
+
+** 2026-07-16 14:30 CDT — Inbox item 5: clock-panel's wl-copy proposal (REJECTED on evidence)
+
+clock-panel proposed adding a =setsid zsh -c '... | exec wl-copy --foreground' &= form to the shared Wayland rules, claiming a plain =printf | wl-copy= "leaves the desktop clipboard empty once the command exits, because the wl-copy selection owner no longer survives."
+
+*Tested all three forms on Craig's machine* (backed up his clipboard first — it held "jotto" — and restored it after). Read back from a *separate* Bash call each time so the originating shell was gone:
+
+| form | result |
+|------|--------|
+| =printf %s "..." \vert wl-copy= (plain) | *works* — survived a 3s wait + unrelated commands; owner alive throughout |
+| their =setsid ... --foreground &= form | *works* — but only by re-creating the default behavior |
+| =wl-copy --foreground= alone, no setsid/=&= | *FAILS* — blocks until killed, clipboard empty after. Their exact symptom. |
+
+*The diagnosis is inverted.* =wl-copy --help=: "-f, --foreground Stay in the foreground instead of forking." wl-copy forks and persists *by default*; =--foreground= is what disables it. So =--foreground= is the *cause* of their symptom, not the cure. Their fix works only because =setsid= + =&= restore the daemonization plain wl-copy already does for free.
+
+*Failed the review, not the value gate.* Wayland clipboard guidance is in scope (gate Q2 passes). It fails review Q1 (is it right) — adopting it would ship a more complex incantation that lands back at default behavior, plus a false mechanism in the rules layer, which is what the next person reasons from.
+
+*The real kernel:* =--foreground= is a genuine trap; if reached for, it must be detached. Roughly the inverse of the proposal. Craig chose option 1 — hold even that line until we know what clock-panel actually hit.
+
+*My own test bug worth remembering:* =pkill -f 'wl-copy'= self-matched (the Bash tool's own command line contains the string "wl-copy"), killing the shell running it — exit 144. Used =pgrep -x= + =kill <pid>= instead. =pkill -f= on a string that appears in your own command line is a self-kill.
+
+Replied to clock-panel with the three test results, the =--help= quote, the inverted-mechanism read, an explicit "this isn't a value-gate failure" note, three specific things that would change my mind (verbatim failing command — my guess is it already had =--foreground=; how they observed the empty clipboard, since a same-shell read can race the fork; any timeout/trap/sandbox reaping the process group), and the mangled-transport flag with a =--file= heredoc suggestion.
+
+*Inbox back to zero.*
+
+** 2026-07-16 15:18 CDT — Clipboard incident (self-inflicted), then the Emacs XWayland finding
+
+*I broke Craig's clipboard.* Testing clock-panel's claim on his live machine, I killed wl-copy owner processes repeatedly. wl-copy holds the selection only while its process lives, so each kill emptied the clipboard. His original content ("jotto") died with the first one. Symptoms he reported: =C-y= showing "Mark set" with nothing inserted (that's =yank= on an empty string) and Chrome greying out paste (empty clipboard). I had backed up + restored once, then kept testing on the same live clipboard *without re-checking* — the discipline was applied once and abandoned. Restored "jotto" via Emacs (=gui-set-selection=) rather than wl-copy, since Emacs is long-lived and won't evaporate.
+
+*Owed to clock-panel: a partial retraction.* I rejected their proposal partly on "I tested it, plain wl-copy works" — but I read the clipboard back *seconds* after copying, before the owner had any chance to die, and then reproduced their exact failure on Craig's machine. The =--foreground=-is-backwards half of the analysis holds; the confidence didn't. *Not yet sent.*
+
+*The real finding.* Craig's Emacs reported =framep= = =x=. Investigation:
+
+- Installed package was =extra/emacs 30.2-3= (X11 build): =(:pgtk nil :x t)=, =window-system x=, =DISPLAY=:0= with =WAYLAND_DISPLAY=wayland-1= unused. On XWayland, contradicting the documented pure-Wayland setup — and why =xclip= reads his clipboard fine despite protocols.org saying never to use it.
+- =archsetup= line 2605 has run =pacman_install emacs-wayland= *unconditionally since bb9c9bb (2026-04-11, "fix: emacs-wayland package ... velox config sync")*. So *ratio drifted from archsetup's own intent for ~3 months*; the commit named velox, so the fix likely only ever reached velox. This also explains =emacs.md='s "Verified on Emacs 30.2 pgtk" claim — true on velox, false on ratio.
+- Host derived at runtime per the host-identity rule: =uname -n= → ratio. velox offline 2d, unverified.
+
+*Installed =emacs-wayland 30.2-3= on ratio* (Craig's call, option 2). =emacs= had =Required By: None=; =emacs-wayland= =Provides: emacs= so ledger/mu optdeps stay satisfied. =--noconfirm= *failed* (it takes the default N on a conflict prompt) — used =yes | sudo pacman -S emacs-wayland= for one atomic remove+install. Snapper snapshots 7680/7681 taken. New binary verified =pgtk:t x:nil=; running daemon still X11 (holds the deleted inode). *Restart is Craig's to make.*
+
+Blast radius checked before advising: this Claude session is tmux =aiv-rulesets= and *survives* an Emacs restart (as do 6 other agent sessions); his 4 frames and EMMS's mpv don't.
+
+*Bonus finding — =emacs.service= failed since 2026-07-09 06:00:04* (1 week), =start-limit-hit= after 5x exit-1. Cause *undiagnosed* (journal has no Emacs stderr; my untested hypothesis is a daemon-already-running conflict). The live daemon (PID 3004861, bare =emacs --daemon=, started Jul 14) has PPID 1 and nothing in the Hyprland config starts it. So =Restart=on-failure= isn't protecting anything and a reboot wouldn't restore the intended shape.
+
+*Ownership split, flagged to Craig:* the unit is a stow symlink into =~/.dotfiles/= — its own repo (=git@cjennings.net:dotfiles.git=) with its own =.ai/= scope. Package = archsetup's; unit = dotfiles'. Craig routed the handoff to archsetup; I sent there only and named the boundary, leaving the dotfiles side his call.
+
+Handoff sent: =archsetup/inbox/2026-07-16-1518-from-rulesets-handoff-archsetup-emacs.org=. The open question posed to them: nothing appears to detect "archsetup says X, machine has Y" on an *existing* box — intent is expressed for fresh installs only.
+
+*Test-methodology lessons worth keeping:* (1) =pkill -f '<pattern>'= self-matches when the pattern appears in your own command line — it killed the shell running it (exit 144), twice. Use =pgrep -x= + explicit kill. (2) Verifying a persistence claim seconds after the write proves nothing about persistence.
diff --git a/.ai/sessions/2026-07-19-21-15-launcher-hardening-inbox-hook-colloquialisms.org b/.ai/sessions/2026-07-19-21-15-launcher-hardening-inbox-hook-colloquialisms.org
new file mode 100644
index 0000000..8f5207c
--- /dev/null
+++ b/.ai/sessions/2026-07-19-21-15-launcher-hardening-inbox-hook-colloquialisms.org
@@ -0,0 +1,238 @@
+#+TITLE: Session Context
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-18
+
+* Summary
+
+** Active Goal
+
+Three tasks shipped this session, all closed, pushed, velox synced (HEAD f6a2701). No active goal remaining; next is a fresh backlog pick.
+
+1. ai-launcher-hardening [#C] (113e8d8, 2b619f1, closed 33c6d7b): hardened =claude-templates/bin/ai= per its measurable acceptance criteria. Launcher tests 9 → 42, four pure decision cores extracted (=_git_prep_action=, =_order_windows=, =_match_window_id=, =_git_is_dirty=) each N/B/E, footgun audit + /refactor pass fully dispositioned, shellcheck clean + shfmt -i2 -ci + suite green before/after + live smoke correct.
+2. inbox-boundary-check hook [#B] (94e54f6, closed f6a2701): soft-nudge Stop hook enforcing the task-boundary inbox check. 6 bats, wired in settings.json + snippet, protocols note. Takes effect next session.
+3. colloquialisms / "the list" convention [#B] (8fd9e39, closed f6a2701): protocols.org Colloquialisms section + wrap-it-up Step 1 Before-Close Queue sub-step. 4 bats. Live now.
+
+KB: promoted 0 / consulted no
+
+** Decisions
+
+- The task's acceptance criteria are FIXED and live in the task body (todo.org, the 4 moves per todo-format.md's "Making an open-ended task measurable"). Follow them; don't re-derive scope.
+- Interactive runtime picker is OUT of this :solo: scope (design call) — file separately if wanted.
+- Characterization discipline per testing.md (refined this session, commit 179c495): record-not-spec, full Normal/Boundary/Error set per unit, negative/boundary cases are the bug-finders, extract-pure-core-when-IO-blocks IS the hardening.
+
+** Data Collected / Findings
+
+- Surface (22 fns). COVERED (via scripts/tests/ai-launcher-runtime.bats, 9 tests, runtime path): resolve_agent_cmd, build_runtime_choices, pick_runtime, build_instructions, print modes. UNCOVERED (17 to net): usage, check_deps, attach_session, create_window, maybe_add_candidate, build_candidates, fetch_candidates, git_status_indicator, annotate_candidates, auto_pull_if_clean, read_selections, sort_windows, find_window_id, prep_git_single, attach_mode, single_mode, multi_mode, print_launch_mode.
+- Pure/near-pure (characterize directly, N/B/E): git_status_indicator, maybe_add_candidate (dedup), annotate_candidates (format), read_selections (parse), usage. tmux/git-coupled (extract pure core + thin wrapper): sort_windows (ordering), create_window, attach_session, find_window_id, prep_git_single, auto_pull_if_clean. ~3 functional tests over single_mode/multi_mode/attach_mode against a throwaway tmux session.
+- Canonical bin/ai is claude-templates/bin/ai; installed via make install's bin loop (symlink to ~/.local/bin/ai). Tests live in scripts/tests/ (not .ai mirror). shellcheck + shfmt present; kcov NOT installed.
+
+** Files Modified
+
+This session (all pushed to origin/main, velox synced to d49be09): sentry build a8b6cf4/ccc9c26/c6383e9/8c0a56b, trial fix beb7f0b, gui-open b3195e9, flashcard apkg converter a143679 + multi-tag a14e43b, task closes a760d8e, knowledge-arch landing 179c495/2e19048/94df71e, flake fix 94015e6, launcher scoping d49be09. Nothing in flight — tree clean at the flush.
+
+** Next Steps
+
+Launcher hardening is done and pushed. Pick the next backlog task. Strong candidates surfaced this session: the two [#B] :feature: shared-asset proposals (todo-cleanup dated-seal already shipped ddbd47f; remaining backlog includes the inbox-boundary-check hook, the colloquialisms/the-list convention, and the build-to-prototype ui-prototyping extension). Manual: the sentry overnight live-trial on ratio (4-part task) stays Craig's to run.
+
+STANDING (still in force this session): run the suite as its OWN step and read it green before each commit; sync velox (git pull + make install, ssh 100.127.238.103) after each push; /review-code + /voice personal per commit; keep velox current.
+
+* Session Log
+
+** 2026-07-18 Sat 17:52 CDT — Startup + inbox inventory
+
+Ran startup.org. Phase A.0: rulesets pull skipped (dirty tree — =.claude/settings.json= carries a harness-written model flip opus → claude-fable-5[1m] against the committed opus pin bd76d98; needs Craig's call). =make install= nothing new. Project fetch clean. Phase A: no crash anchor (clean prior wrap), =.ai/= synced from templates, staleness 3 tasks >7 days, roam inbox 18 items (all foreign — archsetup/takuzu/home/clock-panel, none rulesets-claimed), KB 97 nodes / no relevant titles / best-practices path unresolved, spec-sort + host-identity probes silent.
+
+Inbox: 7 pending. Acted on the two trivial ones immediately:
+- website priority-scheme FYI — deleted (pure FYI, loop already closed by their "done").
+- website roam-KB-hosting-moved — applied the factual origin-URL fix in =claude-rules/knowledge-base.md= (git@cjennings.net:roam.git → cjennings@cjennings.net:git/roam.git), verified ratio's =~/org/roam= remote already points at the new URL. Deleted the inbox file. Commit + reply to website pending. Surfaced to Craig: rulesets.git itself is still publicly browsable on cgit (his open decision, per website's note).
+
+** 2026-07-18 Sat 18:02 CDT — Triage-intake redesign applied (item 1 of 3)
+
+Craig approved option 1 (apply as sent). Copied both sent canonicals over =claude-templates/.ai/workflows/{triage-intake,daily-prep}.org=, synced the mirror via =sync-check.sh --fix=, full suite green (374 + 67 pytest, all ERT expected, all bats ok, make exit 0). Review ran (/review-code --staged, inline): verdict Approve — the two remaining "suggested-actions" mentions are dated history entries (correct to keep), ORDER still governs the on-request long form, daily-prep line 58 keeps old "Action items" wording but is behaviorally accurate (Minor, not fixed). Commit 6e48714 (voice-passed, gate skipped — .ai/ tracked). Deleted the 3 inbox files, replied to work (delivered to their inbox 18:02).
+
+Three substantive shared-asset proposals queued for skeptical-review surfacing:
+1. work 07-17: todo-cleanup.el --archive-done dated-seal model (retain 7→31, unparseable-CLOSED archiving, --seal rename). Craig ratified the design at work. Verified: todo-cleanup.el still has retain default 7, no seal — proposal premise current.
+2. home 07-17: strip SCHEDULED/DEADLINE on dated-rewrite completion (todo-format.md + todo-cleanup --convert-subtasks) + new lint-org checker dated-log-heading-active-timestamp.
+3. work 07-18: triage-intake Phase C/D redesign (three-section digest TASKS/FYI/MISC, close-by-default, reroute modifier) + daily-prep 3b one-liner — Craig's 2026-07-18 ruling, edited canonicals attached. Verified: diffs coherent, author lines clean (no "& Claude" regression), no other template references the retired format, plugins untouched by design.
+
+** 2026-07-18 Sat 18:11 CDT — Items 2 & 3 filed; 2 new home handoffs arrived
+
+Craig said "proceed" → filed both items 2 and 3 as [#B] :feature:solo: in todo.org, each with the design preserved to docs/design/ and cross-linked (both touch todo-cleanup.el, build as one batch). Replied to work (filed) for item 2. Route checks: both none (local keepers). settings.json resolved itself — Craig's /model opus set it back to match the committed opus pin (bd76d98), tree now clean on that file.
+
+Task-boundary inbox check caught 5 NEW home handoffs (17:53 + 18:04), two distinct proposals:
+- Item 4: upcoming-birthdays feature promotion (cover + upcoming_birthdays.py + test + full daily-prep copy). Reviewed the code — clean, stdlib-only, 19 pytest cases (Normal/Boundary/Error, leap-day, placeholder-year 1900→None, window boundaries, CLI). Isolated the daily-prep reconcile: home's copy PREDATES my triage change, so only two birthday hunks apply (Heads-Up item 2 + Phase A source 8) — NOT a wholesale overwrite, which would revert my 3b triage edit. Home flagged this correctly.
+- Item 5: "Colloquialisms and Expansions" + "the list" before-close-queue convention. Touches protocols.org + wrap-it-up.org (both synced) — a norm-adoption design call.
+
+** 2026-07-18 Sat 18:22 CDT — Item 4 (birthdays) applied
+
+Craig approved. Copied upcoming_birthdays.py + test into claude-templates/.ai/scripts/, applied ONLY the two birthday hunks to canonical daily-prep (verified my 3b triage edit survived — both copies still carry "three-section digest"), synced mirror. Suite green pytest 374→393 (19 new, all pass). Commit 80ebb74. Deleted 4 inbox files, replied to home (delivered 18:22). Also answered Craig's side question: all 27 projects carry the task-boundary inbox-check instruction (protocols.org synced), but it's a prose behavioral rule, not an enforced hook.
+
+Remaining: item 5 (colloquialisms/the-list convention). Still uncommitted from earlier: knowledge-base.md roam-URL fix + todo.org filings (items 2,3) + 2 docs/design proposals — batch at close-out. Craig also asked to explore a hook design for the inbox check after item 4.
+
+** 2026-07-18 Sat 19:56 CDT — Hook exploration, item 5 filed, inbox closed, all committed
+
+Explored the inbox-check hook with Craig. Key framing that emerged: two rails split by "must fire at a wall-clock instant, or just needs to be seen soon?" — boundary rail (Stop hook / UserPromptSubmit, fires when the agent yields, never interrupts mid-task) vs cron/at (must-fire-now: meeting alarms, meds, deadlines). The Stop event maps onto the rule's own boundary definition ("before reporting back"). Recommended soft-nudge Stop hook (stop_hook_active guard, once per turn) + UserPromptSubmit visibility injector. Filed as [#B] :feature: with full design. Craig extended the idea to a general boundary-check reminder rail (session-save nudge, uncommitted-drift nudge, soft reminders coming due) — offered a design note, he declined (not restarting, so no cold-start concern).
+
+Craig aborted the wrap (not restarting after all) and said complete the tasks. Actions:
+- Filed item 5 (colloquialisms / "the list" convention) as [#B] :feature: — a cross-project convention adoption touching protocols.org + wrap-it-up.org, so filed not applied; needs Craig's adoption decision + a short design pass. Replied to home, design preserved to docs/design/.
+- New archsetup FYI arrived mid-work (package-drift audit accepted on their side) — pure loop-close, deleted, no reply owed.
+- Committed everything: 6523ed5 (knowledge-base roam URL fix), bc1d81d (4 backlog task filings + 3 docs/design proposals). Plus earlier 6e48714 (triage redesign), 80ebb74 (birthdays).
+
+STATE: inbox at zero, tree clean, suite green. 4 commits ahead of origin/main (6e48714, 80ebb74, 6523ed5, bc1d81d), 0 behind — UNPUSHED, awaiting Craig's push call. :LAST_INBOX_PROCESS: stamped 2026-07-18.
+
+Backlog filed this session (all [#B]): todo-cleanup dated-seal, dated-log planning-line strip + lint checker (batches with the seal), inbox-boundary-check hook, colloquialisms/the-list convention.
+
+** 2026-07-18 Sat 20:06 CDT — Two [#C] :quick:solo: closeouts (power-through)
+
+Craig picked the quick wins first. Both DONE + CLOSED, task-shaped (top-level):
+- coverage-summary.el local-only doc (2cb7c1b, pushed after this): stated local-only status in the .el commentary header + elisp-testing.md "Measuring it" section (the gitignored .claude/scripts/ install is by-design, not a CI gap — Craig's 2026-06-28 decision). Sent emacs-wttrin a handoff to revert its contradicting header claim.
+- install-ai on PATH (d2b1bef): new claude-templates/bin/install-ai thin launcher, resolves its own symlink chain and execs scripts/install-ai.sh. make install's existing bin loop links it to ~/.local/bin/install-ai (same as ai/agent-page) — resolves the task's open question (no dedicated sync, no dotfiles copy; the symlink is the canonical). 3 launcher bats incl. symlink-invocation. Verified live: install-ai --help runs from PATH. Suite green pytest 393, all bats/ERT pass.
+
+Remaining top solo work: the two batched todo-cleanup [#B] :feature:solo: tasks (dated-seal + planning-line strip) — the strongest next power-through target.
+
+** 2026-07-18 Sat 21:04 CDT — Batched todo-cleanup build (both [#B] :feature:solo: DONE)
+
+Craig said yes to the batch. Built both TDD in one commit ddbd47f (10 files, +746/-79), both DONE+CLOSED.
+
+Task A (dated-seal): retain default 7→31 via new defconst tc-archive-retain-days-default; unparseable-CLOSED archiving made explicit in the contract/commentary; new --seal mode (tc-seal-archive-file) renames task-archive.org → resolved-YYYY-MM-DD.org beside it, dated by seal run (not quarter — avoids late-quarter mislabeling), next --archive-done recreates fresh working file; sealed file inherits gitignore status; --seal flag + dispatch + report + CLI recognizer. 5 ERT tests.
+
+Task B (planning strip + lint): --convert-subtasks now strips the WHOLE planning line (CLOSED+SCHEDULED+DEADLINE) via tc--strip-planning-lines-in-entry, reversing the old CLOSED-only behavior (the enshrining test tc-convert-preserves-deadline... was rewritten to assert stripping). Stops at first non-planning line so body prose survives. New lint-org checker dated-log-heading-active-timestamp (flags active <..> SCHEDULED/DEADLINE on a keyword-less dated heading, ignores inactive [..]). todo-format.md sub-task rule step 5 + VERIFY path. 5 convert + 5 lint tests.
+
+Debugging note: seal tests failed only in the FULL suite — root cause was a latent bug, tc-test--reset never cleared tc-convert-subtasks, so a convert test's mode flag leaked and won the tc-process-file cond over seal. Fixed the reset + made the seal harness set all mode flags explicitly. Full suite green (todo-cleanup 52, lint-org 57, pytest 393, all bats). Replied to work + home (delivered).
+
+STATE at 21:04: ddbd47f committed, NOT yet pushed (2 quick-win commits 2cb7c1b/d2b1bef already pushed earlier). Reconcile clean before commit (0/0).
+
+** 2026-07-19 Sun 04:35 CDT — SENTRY BUILD STARTED (no-approvals + auto-flush)
+
+Craig approved the sentry spec after his deep read and told me to build it in no-approvals + auto-flush mode. Big pivot from the launcher-hardening [#C] (that was read-only, no edits to bin/ai — left as-is, still TODO).
+
+Spec flipped DRAFT → READY → DOING (docs/specs/2026-07-14-sentry-workflow-spec.org, ID f6c51f27-d7a2-4b63-9ff9-5ba005a66dfb). Metadata Status → doing. Build task decomposed into 4 phase sub-tasks under "** DOING [#B] Sentry workflow" with :SPEC_ID: binding.
+
+BUILD PLAN (resume anchor — re-read the spec's Design + Decisions + Implementation phases if context was cleared):
+- Phase 1: =agent-lock= helper (canonical claude-templates/.ai/scripts/) — mkdir-atomic acquire, PID/host/ISO metadata, age-staleness reclaim (surfaced), bounded-wait contention (~30s), heartbeat refresh, acquire/release/status subcommands. Home: /run/user/<uid>/agent-locks/<name>/ with ~/.cache/agent-locks/ fallback. bats-tested. TDD. NOTHING calls it yet.
+- Phase 2: =sentry.org= engine (.ai/workflows/, mirror synced) — entry ticket :COMMIT_AUTONOMY:, interactive entry gates, ff-only reconcile, sentry/YYYY-MM-DD-<host> branch, 10-pass runner (probe→work→session-context→commit), digest + approval queue, skip-and-note, spine-exclusion + fire-end digest commit, stall-notify after 2 skips, stop-sentry op. INDEX.org.
+- Phase 3: companion reconciliations — knowledge-base.md, inbox.org core §5, roam-sync.sh header, triage-intake.org, wrap-it-up.org active-sentry guard.
+- Phase 4: make test green; file live-trial follow-ups (Craig runs the trial); flip spec → IMPLEMENTED, close parent.
+
+Mode: no-approvals (commit+push per logical unit, Step 0 reconcile + pre-push check still run), auto-flush armed. COMMIT_AUTONOMY: yes + LOOP_MAY_COMMIT: yes in notes.org.
+
+Starting Phase 1 (agent-lock, TDD).
+
+** 2026-07-19 Sun @ 04:38:35 -0500 — flushed
+Auto-flush before the sentry build. Clean boundary: spec flipped DRAFT→READY→DOING and decomposed (495e36b, pushed), nothing half-edited. Resuming into Phase 1 (agent-lock helper, TDD) per Summary Next Steps.
+
+** 2026-07-19 Sun @ 04:52 CDT — PHASE 1 DONE (agent-lock), pushed
+
+Built the agent-lock helper TDD (canonical claude-templates/.ai/scripts/agent-lock + mirror). Subcommands acquire/refresh/release/status/path. mkdir-atomic acquire, PID/host/ISO/TTL metadata, mtime-based staleness reclaim (surfaced note), heartbeat refresh, XDG_RUNTIME_DIR/agent-locks home with ~/.cache fallback. 18 bats tests.
+
+Self-review (review-code --staged, inline) caught one real concurrency bug pre-commit: the stale-reclaim path was rm -rf + mkdir in two steps, letting two acquirers who both see a lock stale double-acquire (loser deletes winner's fresh dir). Fixed with atomic-rename claim (mv wins-or-fails, then mkdir stays sole grant); strengthened test 7 to assert fresh metadata after reclaim. Verdict cleared to Approve.
+
+Full suite green (make test exit 0, 0 not-ok; agent-lock 18/18). Commit a8b6cf4, pushed to origin/main (0 behind, pre-push reconcile clean). /voice personal ran on the message.
+
+NEXT: Phase 2 — sentry.org engine (.ai/workflows/, mirror synced). Entry ticket :COMMIT_AUTONOMY:, interactive entry gates (dirty-tree / red-suite), ff-only reconcile, sentry/YYYY-MM-DD-<host> branch, pass runner (probe→work→session-context→commit contract), digest + morning-approval queue, skip-and-note semantics, spine-exclusion + fire-end digest commit, stall-notify after 2 unmerged-branch skips, stop-sentry op, INDEX.org entry. Re-read spec Design paragraphs (branch mechanics, locks, roam writes, unattended safety, pass list, digest) + the 10 decisions.
+
+** 2026-07-19 Sun @ 05:06 CDT — SENTRY BUILD COMPLETE (all 4 phases, pushed)
+
+All four sentry phases shipped, committed, pushed to origin/main, suite green throughout:
+- Phase 1 a8b6cf4 — agent-lock helper (18 bats). Pre-commit review caught + fixed a reclaim double-acquire race (atomic-rename claim).
+- Phase 2 ccc9c26 — sentry.org engine + INDEX entry. All 10 decisions / 12 findings reflected.
+- Phase 3 c6383e9 — roam writers (knowledge-base.md, inbox.org §5) acquire roam-write lock + edit-plus-trigger (roam-sync sole committer); roam-sync.sh header; triage-intake note; wrap-it-up Step 0 active-sentry guard. Lock-name derivation pinned identically in sentry.org + wrap-it-up (sentry-<repo-basename>).
+- Phase 4 8c0a56b — make test green at HEAD; spec DOING→IMPLEMENTED (dated history + Metadata mirror); build task + phase sub-tasks closed (sub-tasks → dated event-log entries); live trial filed as structured "Manual testing and validation" task.
+
+FINAL STATE: tree clean (only untracked spine), 0/0 vs origin/main, spec IMPLEMENTED, sync-check clean. Live agent-lock smoke test on the real runtime dir passed (acquire→held on ratio→release). Each commit ran /review-code (inline for docs) + /voice personal.
+
+REMAINING (Craig's, not agent-buildable): the overnight live-trial night on ratio — arm sentry, exercise the entry gates, observe one fire end to end, run the morning branch review. Filed as the 4-part manual-testing task in todo.org. Its findings become follow-up tasks. The launcher-hardening [#C] (read-only bin/ai) is still TODO, untouched (was pre-empted by the sentry build).
+
+** 2026-07-19 Sun @ 15:38 CDT — Live trial feedback #1 processed (roam-denylist mis-park)
+
+Craig ran a full sentry session from the WORK project. First trial finding: the inbox-zero pass (P2), running from ~/projects/work, parked the whole 19-item roam inbox as a cross-project boundary crossing and refused to tidy it — over-reading knowledge-base.md's work-denylist as "don't touch roam from work." Craig ruled: roam is a shared resource; the denylist only ever gated durable agents/ KB-node writes (a confidentiality guard). Reading roam + tidying the roam inbox are allowed from any project, work included.
+
+Fix (commit beb7f0b, pushed): applied work's prepared knowledge-base.md "Scope of the denylist — durable KB-node writes only" paragraph (naming the 2026-07-19 mis-park), plus one-line companion notes in sentry.org P2 and inbox.org roam mode pointing at the rule. Suite green. Replied to work (inbox-send, delivered 15:35), deleted the two work inbox items.
+
+Commits so far: a8b6cf4, ccc9c26, c6383e9, 8c0a56b (build), beb7f0b (trial fix #1). All pushed, tree clean.
+
+PENDING: 2 untracked archsetup items in inbox/ (11:47, gui-open proposal — unrelated to sentry, separate pass). Awaiting any further trial findings from Craig.
+
+** 2026-07-19 Sun @ 15:44 CDT — Inbox cleared: archsetup gui-open proposal applied
+
+Processed the two archsetup items (gui-open proposal). gui-open is a dotfiles-shipped launcher (a34d479, on PATH via ~/.dotfiles) that shows a file from a short-lived agent shell reliably — detaches through systemd-run --user (no shell reap), resolves the Hyprland instance after a restart, verifies a visible client. Cleared the value gate (fixes documented fragility both rules already flagged).
+
+Fix (commit b3195e9, pushed): interaction.md "Showing Craig Visuals" + desktop-capture.md "Showing the user something" now launch via gui-open instead of google-chrome-stable ... & / hyprctl dispatch exec imv. Guidance-only (tool is dotfiles-owned, not a rulesets script); added an "if not on PATH, needs a dotfiles pull" note. Replied to archsetup (delivered 15:42), deleted both inbox files. Inbox now clear.
+
+DAILY-DRIVER NOTE: gui-open ships via dotfiles a34d479 — confirmed on ratio (symlinked ~/.local/bin/gui-open → ~/.dotfiles). velox needs a dotfiles pull to have it, or the new rule guidance references a missing tool there. Flag to Craig.
+
+Session commits total: a8b6cf4, ccc9c26, c6383e9, 8c0a56b (sentry build) + beb7f0b (trial fix #1) + b3195e9 (gui-open guidance). All pushed, tree clean, inbox zero.
+
+** 2026-07-19 Sun @ 15:52 CDT — velox brought current (standing: keep it synced this session)
+
+Craig: do the dotfiles pull on velox now, and keep velox up to date with our changes for the rest of the session. Reached velox over tailscale (100.127.238.103, cjennings@).
+- dotfiles: git pull → c3ef604 (was 9914a25), make restow hyprland (clean, no conflicts). gui-open symlinked ~/.local/bin/gui-open → dotfiles common tier; verified: direct exec prints usage, ~/.local/bin on interactive-login PATH (positions 1-2), interactive shell resolves it. Earlier "MISSING" was only the non-interactive SSH PATH.
+- rulesets: git fetch + merge --ff-only → b3195e9 (was 3ae71be; also caught up on upcoming_birthdays, install-ai, docs/design proposals). make install relinked install-ai + agent-page, rest skipped.
+
+STANDING INSTRUCTION (rest of session): after each push, also update velox — git pull + make install on ~/code/rulesets, and a dotfiles pull + restow if dotfiles changed. ssh via 100.127.238.103. zsh gotcha: don't word-split an unquoted $VAR for the ssh command; run ssh inline. Non-interactive SSH PATH lacks ~/.local/bin — test tools via resolved path or zsh -lic.
+
+** 2026-07-19 Sun @ 18:42 CDT — Speedrun complete (2 flashcard tasks shipped)
+
+No-approvals speedrun over the 2 flashcard :solo: tasks, both TDD + review + voice, each its own commit, pushed, velox synced after each.
+- apkg-to-orgdrill.py (a143679): inverse converter, stdlib zipfile+sqlite3, 17 tests + real-genanki round-trip. Grounded the apkg schema by generating/inspecting a real one first.
+- flashcard multi-tag reconcile (a14e43b): broadened CARD_RE in to-anki + stats for :fundamental:drill:, added --tag-filter + --guid-salt + drill-membership guard; re-derived against current canonical (kept the #+TITLE fix); parse() 3rd element now anki-tag list. End-to-end verified (2→1 with --tag-filter). 465/100 real-deck check needs the work deck (not runnable here).
+- Closed both as dated event-log entries (a760d8e).
+
+Suite green throughout (419 pytest). Session commits now: sentry build (4) + trial fix + gui-open + flashcard (2) + closes. All pushed, velox current at a760d8e.
+
+NEXT (Craig asked): discuss the ai launcher hardening [#C] task (todo.org line ~208). Remind him what it is.
+
+** 2026-07-19 Sun @ 19:04 CDT — Landed the testing/acceptance knowledge-architecture change
+
+Craig's directive: build the characterization-test + measurable-acceptance discipline into the workflows, and decide the KB-vs-rule boundary. Answer settled in discussion: apply-every-time discipline → rules (single source, auto-loaded, review-gated contributions via inbox); pull-when-relevant cross-project facts → KB. Almost none of this is KB material (it has a rule home); the KB is the capture/holding-pen, promotion moves it OUT into rules.
+
+Two commits:
+- 179c495 (characterization discipline): testing.md — defined a characterization test (record-not-spec, Feathers recipe), required the full Normal/Boundary/Error set per unit, and the key lens that negative/boundary cases are the bug-finders; + the "extracting the pure core IS the hardening" framing in the refactor-for-testability section. code-quality.org gained the precondition that behavior-preserving rests on a characterization net.
+- 2e19048 (measurable acceptance): todo-format.md new subsection "Making an open-ended task measurable (so it can be :solo:)" — bound surface / characterization net / disposition findings / objective floor; qualifying answer = dispositioned report. work-the-backlog's keystone defer item now recognizes absence-phrased open-ended goals and routes them to get criteria.
+
+Scope note: start-work and /refactor are NOT rulesets files (not installed at ~/.claude/skills/; registered elsewhere). Their pointers are a follow-up in whatever repo defines them — flagged to Craig. add-tests + review-code (rulesets skills) already well-wired to testing.md.
+
+Suite green, lint 0/0, velox synced to 2e19048.
+
+** 2026-07-19 Sun @ 19:14 CDT — Flaky-test fix + a discipline note
+
+While landing the start-work pointer (94df71e), chained make test into the commit command and the commit went through on a RED run — the bundling verification.md warns against. The red turned out flaky (rename-ai-artifact.bats teardown race, unrelated to the md edit; suite green on re-run, change sound), but the gate failed open. Correction going forward: run the suite as its own step, read it, then commit.
+
+Root-caused + fixed the flake (94015e6): rename-ai-artifact.bats teardown hit "rm: cannot remove .git: Directory not empty" because git background auto-maintenance (maintenance run --auto after commit) wrote into .git after the body, racing rm -rf. Fix: gc.auto 0 + maintenance.auto false in the throwaway repo before any git command arms them. Diagnose-not-mask (no rm retry). Verified 30 file runs + 2 full-suite runs green.
+
+Also corrected an earlier wrong claim: start-work + refactor ARE rulesets files (.claude/commands/, symlinked to ~/.claude/commands), already wired to testing.md for characterization. The acceptance-pattern landing is fully complete — no out-of-repo follow-up.
+
+** 2026-07-19 Sun @ 19:20 CDT — Scoped the ai-launcher-hardening task (dogfooded the pattern)
+
+Applied the measurable-acceptance pattern to the launcher task (d49be09). Rewrote the open-ended body with the 4 moves: bounded surface (22 fns, 5 covered via ai-launcher-runtime.bats / 17 uncovered, all named), characterization-net plan (pure-core extraction called out for the tmux/git-coupled fns), dispositioned audit (per-fn footgun matrix + /refactor pass), objective floor (shellcheck/shfmt/green suite/per-fn coverage + ~3 functional tests over single/multi/attach_mode). Split the interactive runtime picker OUT of :solo: (design call). Task is now genuinely :solo:-ready. :LAST_REVIEWED: bumped to 2026-07-19.
+
+The actual hardening work (bring 17 fns under characterization tests, extract pure cores, run the audit + refactor) is a substantial next-session speedrun candidate — now bounded.
+
+** 2026-07-19 Sun @ 20:19:15 -0500 — flushed
+Interactive flush at a clean boundary (launcher task scoped + committed d49be09, tree clean, nothing half-edited). Resuming into option 2: work the ai-launcher-hardening characterization sweep per the refreshed Summary Next Steps.
+
+** 2026-07-19 Sun @ ~21:30 CDT — LAUNCHER HARDENING COMPLETE (2 commits + close, pushed, velox synced)
+
+Worked the ai-launcher-hardening [#C] :refactor:solo: end to end per its measurable acceptance criteria. Craig present, chose push+sync+close.
+
+Sequence:
+- Grounded in bin/ai (540 lines), confirmed green baseline (make test exit 0), pinned the gate targets: shellcheck-clean (the bash hook enforces shellcheck only; shfmt is deliberately NOT hook-gated), shfmt house style = -i 2 -ci (agent-page conforms, bin/ai did not).
+- Commit 113e8d8 (net + source-testable): wrapped dispatch in main() behind a run-vs-sourced guard so the file is sourceable for unit tests; cleared 4 pre-existing shellcheck warnings on touched paths (@{u} quoted ×3 = SC1083, literal display tilde disabled = SC2088). New scripts/tests/ai-launcher-characterization.bats, 20 tests: pure/near-pure N/B/E (git_status_indicator, maybe_add_candidate, annotate_candidates, read_selections, build_candidates, usage) + 4 functional (create_window, find_window_id, sort_windows, attach_mode) against a PRIVATE tmux socket (TMUX_TMPDIR + unset TMUX) so nothing touches Craig's live ai session; throwaway repos disable gc.auto/maintenance.auto (the rename-ai flake).
+- Commit 2b619f1 (refactor): extracted four pure cores — _git_prep_action (git-prep none/pull/report classifier, shared by prep_git_single + auto_pull_if_clean), _order_windows (sort_windows ordering), _match_window_id (find_window_id), _git_is_dirty (DRY the dirty triad ×3). Added 13 unit tests for the cores incl. the safety case (dirty repo never auto-pulled). shfmt -i2 -ci applied. 42 launcher tests green, live black-box smoke correct for claude+codex.
+- Commit 33c6d7b: closed the task DONE+CLOSED (top-level ** → task-shaped) with the dispositioned resolution note.
+
+Footgun audit fully dispositioned (report delivered inline to Craig): unquoted-expansion/word-split — clean; errexit-off — intentional/documented, kept; subshell state loss — none (globals set in main shell); exit-code propagation — intentional; sort_windows race — none (two-pass park-then-reassign is correct); create_window sleep 0.1 — declined (intentional); git-prep error paths (dirty/detached/no-upstream) — all correct, dirty-safety now a named test. /refactor: 4 extractions + shfmt applied; build_candidates exit-status + create_window sleep declined.
+
+Honest coverage limit stated: attach_session and full end-to-end of single/multi/fetch stay partially covered (terminal step attaches/blocks on fzf, can't run headless); their decision logic was extracted into the netted cores.
+
+STATE: 33c6d7b pushed, 0/0 vs origin/main, velox synced to 33c6d7b (make install: ai linked), tree clean (only untracked session anchor), suite 367 ok / 0 not-ok. Interactive runtime picker stayed OUT of scope (design call, on the generic-agent-runtime parent). No memory promotion needed (nothing durable/cross-project beyond the repo record).
+
+** 2026-07-19 Sun @ ~22:00 CDT — No-approvals speedrun: inbox hook + colloquialisms (both [#B] DONE)
+
+Craig said "no approvals speedrun time" after the launcher task. Built two [#B] backlog tasks, each TDD + inline review + /voice personal, committed + pushed + velox-synced.
+
+Task 1 — inbox-boundary-check hook (94e54f6): soft-nudge Stop hook backing the "check inbox/ at every task boundary" rule (was prose-only). Blocks the yield once + injects the pending count when =inbox-status -q= exits 1, steps aside on =stop_hook_active= re-entry (soft, not hard-block, so a mid-task pause never wedges), self-skips on no-inbox/no-inbox-status/clean. Prefers project-local =.ai/scripts/inbox-status=, PATH fallback. Wired ahead of =ai-wrap-teardown= in =.claude/settings.json= (which the live =~/.claude/settings.json= symlinks to → live global on both boxes) + =settings-snippet.json= + README row; glob-installed by =make install-hooks=. protocols.org Inbox Monitoring Cadence notes the enforcement. 6 bats. Live-verified on ratio (clean no-op, pending fixture blocks). Loads next session (hooks read at session start), so it didn't interfere with this wrap.
+
+Task 2 — colloquialisms / "the list" (8fd9e39): new =* Colloquialisms and Expansions= section in protocols.org. "put X on the list" → session-scoped Before-Close Queue (=* Before-Close Queue= heading in the session anchor, resets on archive, todo.org for must-outlive); "tell <project> <msg>" → =inbox-send=. wrap-it-up Step 1 gained a "Work the Before-Close Queue (before the Summary)" sub-step so queued work rides the wrap commit. Design calls: reference in protocols.org not per-project notes.org (synced = shared norm); queue in the session anchor as home did; wrap step at front of Step 1 not a half-step (keeps "Steps 1-5" framing). 4 documentation-integrity bats. Canonical+mirror synced.
+
+Both closes in f6a2701. Full suite green throughout (0 not-ok). velox synced to f6a2701 (make install ok). Then Craig said wrap it up.
diff --git a/.ai/sessions/2026-07-20-23-36-signal-pager-and-triage-phone-push.org b/.ai/sessions/2026-07-20-23-36-signal-pager-and-triage-phone-push.org
new file mode 100644
index 0000000..fe43f61
--- /dev/null
+++ b/.ai/sessions/2026-07-20-23-36-signal-pager-and-triage-phone-push.org
@@ -0,0 +1,185 @@
+#+TITLE: Session Context — Sentry live trial (ratio, 2026-07-19)
+#+AUTHOR: Craig Jennings
+
+* Summary
+
+** Active Goal
+
+Post-flush session. Two items shipped: (1) the archsetup voice-#46 (comma-budget) handoff, reviewed and committed rulesets-side (2ea5d9a); (2) the triage-intake auto-mode phone-push, built onto agent-text, signal-only per Craig's ruling (e27aea2). Items 2 (reply-correlation spec) and 3 (token-rotation discussion) from the prior next-steps list remain untouched, carried forward.
+
+** Decisions (this session, all shipped)
+
+- Voice pattern #46 (comma budget, max two per sentence): personal mode only, in the attestation high-recurrence set. Craig's 2026-07-20 direction via archsetup; committed rulesets-side 2ea5d9a.
+- Triage auto-mode phone-push is signal-only (Craig's option 1): a quiet sweep's "nothing" heartbeat never pushes to the phone — silent-until-signal governs the phone channel too. Send half shipped (e27aea2); reply-polling deferred to the reply-correlation spec.
+- Sentry live trial: ran on ratio, 8 hourly fires, stopped on "sentry off", branch fast-forwarded to main and deleted.
+- working/ is tracked-from-creation + gitignored temp/ in both modes (Shape A). Committed.
+- Triage source activation: general (synced) plugins gate on a per-project =:TRIAGE_SOURCES:= declaration; project-specific plugins active by presence; gate in triage-intake Phase 0 (interactive + unattended). Spec IMPLEMENTED. Migrations done (home + work declared their sources).
+- Silent-until-signal: a POLICY not a mechanism — an in-session monitor fire heartbeats =<workflow> at HH:MM: nothing= on an empty check; detection stays in-session (MCP-safe). Applied to sentry, auto triage-intake, auto inbox-zero. Spec IMPLEMENTED.
+- Suspend detach change applied to canonical. Sentry cluster consolidated (merged /schedule tasks, added cross-host-coordination task). Task audit stamped 2026-07-20.
+- Polyglot: case-by-case, no option-2 machinery (bundles already compose; only coverage-makefile.txt collides, and it's a manual paste). Subprojects: don't promote (N=1). Both closed.
+
+** Data Collected / Findings — Signal pager (what the resume needs)
+
+RECONCILED (2026-07-13, in the task's dated log): there is ONE pager identity, =+15045173983=, registered in *velox's* signal-cli (account file 465310). =signal-mcp= is a velox-local MCP server (invisible from ratio). ratio's signal-cli holds only Craig's personal number (note-to-self, no push). =agent-page= already shipped (=claude-templates/bin/agent-page=): runs signal-cli directly on velox, ssh-relays from anywhere else, desktop-fallback on failure, 4 bats, live-verified. protocols.org "Paging Craig" was already rewritten around the two channels (notify desktop + agent-page phone). Craig's Signal UUID: =b1b5601e-6126-47f8-afaa-0a59f5188fde=. Reliability finding: both signal-cli accounts throw receive-staleness warnings (velox ~40 days, ratio ~26) — the Signal protocol wants regular receives, and a systemd receive timer on velox is the roam-sync-shaped fix.
+
+REMAINING deliverables (this is the work to do): (1) the runbook proper — send + read-replies + receive-timer + signal-cli account/setup notes, the Signal equivalent of the retired ntfy runbook, canonical home in rulesets docs/; (2) a systemd receive timer on velox (roam-sync-shaped) so receives don't go stale; (3) the ssh-over-tailnet-only vs register-ratio-as-a-linked-device decision (a design call for Craig); (4) confirm protocols.org "Paging Craig" is accurate (agent-page did most of it — verify, don't redo). Task body has full history. Source: home handoff 2026-07-04.
+
+** Files Modified (this session — all committed + pushed to main)
+
+Post-flush commits: 2ea5d9a (voice #46 comma budget — SKILL.md + voice-profile.org), 9721a49 (chore: mark archsetup handoff PROCESSED), e27aea2 (feat: triage phone-push via agent-text — canonical + mirror triage-intake.org, todo.org task closed). Left unstaged: .claude/settings.json model change (opus→fable, not this session's work — deferred).
+
+Earlier same-session (pre-flush): f625cf5 (working/temp feat), b02eade + 4d87f35 (triage source activation spec + build), 986d6ca + 0767af8 + 0e9958a (silent-until-signal spec + Phases), af565ba (suspend detach), 70fbe01 (link fixes), 93a2e6d (working-dir filing), c82b625 (sentry cluster), up to fecdf8c, plus the pager work (302b062, 6145489) and overnight sentry commits.
+
+** Next Steps
+
+Item 1 (triage phone-push) shipped this session. Two remain from the prior list:
+
+1. *Spec the reply-correlation follow-up.* The two-way gap: with the account linked on velox + ratio, a Signal reply reaches BOTH devices (one account, Signal fans out per-device, independent queues) and neither knows which page/session it answers. Only bites when two sessions page-and-wait at once (fire-and-forget is fine). Options laid out to Craig: (1) correlation tag stamped on the page, echoed in reply; (2) quote-reply matching (needs verifying signal-cli surfaces quotes); (3) single reply-owner (velox-only). Overlaps the helper-instance work (same shared-channel problem). This spec also owns the deferred triage-intake reply-polling (phone-recv) half. Also fix the runbook "Reading replies" section, which oversells it. Write as a spec in docs/specs/ (spec-create spine).
+
+2. *Discuss the token-rotation helper* (todo.org =[#C] Token-rotation helper for @a-bonus/google-docs-mcp OAuth refresh=, :feature:quick:) — a discussion first, not a build. Read the task body.
+
+Publish flow (review + voice + suite) for any commits.
+
+KB: promoted 0 / consulted no
+
+* Session Log
+
+** 2026-07-19 23:52 CDT — Sentry armed (entry)
+
+Startup ran clean: rulesets pull no-op, make install nothing new, project repo up to date on main (0/0). Session-context was absent (prior session 2026-07-19-21-15 wrapped cleanly).
+
+Sentry entry gates, all passed with Craig present:
+- Autonomy ticket: =:COMMIT_AUTONOMY: yes= in notes.org Workflow State.
+- Host: ratio (intended live-trial machine).
+- Dirty-tree gate: tracked tree clean; only two untracked inbox files (=2026-07-19-2141-from-.emacs.d-version-working-directories.org=, =2026-07-19-2350-from-home-craig-approved-the-colloquialisms-the.org=) — do not block; sentry's inbox-zero pass (pass 2) handles them.
+- Green-suite gate: =make test= exit 0, 377 bats ok, all pytest + ERT green.
+- Prior sentry branch: none.
+- Reconcile: main 0 behind / 0 ahead of upstream.
+- =agent-lock= helper present.
+
+Created daily branch =sentry/2026-07-19-ratio= from HEAD (main @ f76bf40). Working tree now sits on this branch overnight. Arming the hourly loop next.
+
+Two inbox items pending at entry, both to be handled by pass 2 (inbox zero):
+- .emacs.d working/ version-control ruling — a shared-asset change (working-files convention + install .gitignore behavior across projects). Parks for morning approval; does not fire unattended.
+- home colloquialisms wiring approval — already shipped on the rulesets side in the prior session (protocols.org Colloquialisms section + wrap-it-up Before-Close Queue step, commits 8fd9e39/f6a2701). Inbox zero replies-and-files, confirming it's already wired.
+
+** 2026-07-20 00:06 CDT — Fire 1 digest (manual first fire, Craig present)
+
+Ran end to end on branch =sentry/2026-07-19-ratio=. Single-runner lock =sentry-rulesets= acquired at fire start, on the sentry branch, tree clean. Pass-by-pass:
+
+- P1 roam pull — SKIPPED: =~/org/roam= working tree dirty; roam-sync owns that case (pass is read-only ff-only otherwise). No write.
+- P2 inbox zero — RAN. Two project handoffs processed. (a) home's colloquialisms + "the list" before-close-queue: already wired canonically before it arrived — replied to home confirming, marked PROCESSED. (b) .emacs.d's working/ tracked-from-creation ruling: shared-asset + convention change with a temp/-placement design decision → PARKED (see approval queue). Staged proposal at =working/sentry-2026-07-19-working-files-ruling/proposed.org=, filed a [#B] VERIFY, replied to .emacs.d with two findings. Committed 727a900.
+- P3 triage intake — SKIPPED (probe-too-loose; queued as a finding). The template-synced general plugins (personal Gmail/calendar/cmail/Telegram/GitHub-PRs) are present in *every* project, so the "plugins present" probe self-activates triage everywhere — including rulesets, which is not a triage target. Running it would file Craig's personal action items into rulesets' todo.org and run trash/mark-read/star hygiene on his real accounts unattended. Skipped for scope + safety. No write.
+- P4 todo cleanup — RAN: hygiene 0 fixes, --convert-subtasks 0, --archive-done 0. todo.org already clean. No write.
+- P5 task audit — RAN (mechanical subset). No unambiguous autonomous staleness fixes: all open tasks carry a priority cookie + type tag except the intentional "Manual testing and validation" container; spec-lifecycle clean (P7); no dead file: links in todo.org. The full reconciliation (per-task fact-check, consolidation, parent-retirement, judgment flags) is interactive → queued. Did NOT bump :LAST_AUDIT: (only the mechanical subset ran). No write.
+- P6 working-files hygiene — RAN: flagged =working/inbox-zero-phase-e/= (tracked, backs the now-IMPLEMENTED autonomous-batch-execution spec) as a filing candidate. Filing is a judgment move (3 inbound spec links need updating) → queued. No write.
+- P7 spec status board — RAN: clean. All specs READY or IMPLEMENTED; the sentry spec is IMPLEMENTED; no DOING spec with a closed/missing bound parent. No write.
+- P8 link integrity — RAN (report-only): ~4 candidate broken file: links in live docs — the autonomous-batch spec's =../../.ai/workflows/inbox-zero.org= (renamed to inbox.org), two folded-in spec-review siblings (wrapup-routing, pattern-catalog), and =subprojects-log.org=. Some may be intentional (review docs folded in and deleted). No unattended rewrites. No write.
+- P9 git health — RAN: clean. main = origin/main (0/0), no unpushed commits on other branches, no stale merged branches, sentry branch is the only local extra. No write.
+- P10 prep + symlink freshness — SKIPPED: no prep dir (work/home only). No write.
+
+Fire-end: only org/spine files touched this fire (todo.org, inbox renames, working/ proposal — all org) → conditional suite skipped per contract. Session-context spine is untracked in rulesets, so no digest commit needed; tracked tree is clean. Single-runner lock released.
+
+Branch state after fire: one commit ahead of main (727a900). Next scheduled fire at :17.
+
+** 2026-07-20 00:21 CDT — Fire 2 digest (scheduled, :17 cron)
+
+Lock =sentry-rulesets= acquired, on the sentry branch, tree clean. Little changed in the 14 min since fire 1 — one new inbox handoff, everything else steady.
+
+- P1 roam pull — RAN: =~/org/roam= clean this fire (was dirty in fire 1), ff-only pull → already up to date. No write.
+- P2 inbox zero — RAN: one new handoff. .emacs.d routed three sentry-workflow design considerations from its own hand-run trial (codebase-gated bug/enhancement logging pass; system-health pass; sibling-machine freshness pass). Filed as =** TODO [#C] Sentry vNext passes — from live-trial design input= rather than applied (design input for the Living Document; two hinge on an unresolved cross-driver coordination question). Replied to .emacs.d, marked PROCESSED. Committed 067ed55.
+- P3 triage intake — SKIPPED: same probe-too-loose reason as fire 1; the finding is already in the approval queue, not re-queued.
+- P4 todo cleanup — RAN: 0 fixes. No write.
+- P5 task audit — no change since fire 1; the full-audit-due item stays in the queue, not re-added.
+- P6 working-files hygiene — RAN: =working/inbox-zero-phase-e/= still a filing candidate (already queued fire 1, not re-queued). =working/sentry-2026-07-19-working-files-ruling/= is an *active* working dir backing the open VERIFY task — correctly not a candidate. No new queue item.
+- P7 spec status board — RAN: clean, no DOING specs. No write.
+- P8 link integrity — not re-scanned: no doc changes since fire 1's scan except the new backlog task, whose one internal =file:todo.org::*...= link resolves (KB lesson-detection heading exists). Fire 1's report stands. No write.
+- P9 git health — RAN: clean, main = origin/main (0/0), branch now 2 ahead. No write.
+- P10 prep freshness — SKIPPED: no prep dir. No write.
+
+Fire-end: only org files touched → conditional suite skipped. Spine untracked → no digest commit; tracked tree clean. Lock released. Branch 2 commits ahead of main (727a900, 067ed55). No new approval-queue items this fire.
+
+** 2026-07-20 01:20 CDT — Fire 3 digest (scheduled, :17 cron)
+
+Quiet read-only fire — nothing changed in the hour since fire 2. All passes no-op or skip; no writes, no commit, no new approval-queue items.
+
+- P1 roam pull — RAN: clean, ff-only → already up to date.
+- P2 inbox zero — RAN: 0 new items.
+- P3 triage intake — SKIPPED: same probe-too-loose reason (finding already queued fire 1).
+- P4 todo cleanup — RAN: 0 fixes.
+- P5 task audit — no change; full-audit-due item stays queued.
+- P6 working-files hygiene — RAN: same two dirs (inbox-zero-phase-e already queued; sentry-ruling active for the open VERIFY). No new item.
+- P7 spec status board — RAN: clean, no DOING specs.
+- P8 link integrity — not re-scanned: no doc changes since fire 1.
+- P9 git health — RAN: clean, main = origin/main, branch 2 ahead.
+- P10 prep freshness — SKIPPED: no prep dir.
+
+Branch unchanged at 2 commits ahead. Lock released.
+
+** 2026-07-20 02:20 CDT — Fire 4 digest (scheduled, :17 cron)
+
+Quiet read-only fire, same as fire 3 — nothing changed in the hour. All passes no-op or skip; no writes, no commit, no new queue items. P1 roam up to date; P2 inbox 0 new; P3 triage skipped (finding already queued); P4 0 fixes; P5 no change (full audit still queued); P6 same two working dirs (inbox-zero-phase-e queued, sentry-ruling active); P7 clean, no DOING specs; P8 not re-scanned (no doc changes); P9 clean, branch 2 ahead; P10 skipped (no prep dir). Lock released.
+
+** 2026-07-20 03:20 CDT — Fire 5 digest (scheduled, :17 cron)
+
+Quiet read-only fire, same as fires 3-4. No changes in the hour; all passes no-op or skip; no writes, no commit, no new queue items. P1 roam up to date; P2 inbox 0 new; P3 triage skipped (queued); P4 0 fixes; P5 no change; P6 same two working dirs; P7 clean; P8 not re-scanned; P9 clean, branch 2 ahead; P10 skipped. Lock released.
+
+** 2026-07-20 04:20 CDT — Fire 6 digest (scheduled, :17 cron)
+
+Quiet read-only fire, same as fires 3-5. No changes in the hour; all passes no-op or skip; no writes, no commit, no new queue items. P1 roam up to date; P2 inbox 0 new; P3 triage skipped (queued); P4 0 fixes; P5 no change; P6 same two working dirs; P7 clean; P8 not re-scanned; P9 clean, branch 2 ahead; P10 skipped. Lock released.
+
+** 2026-07-20 05:20 CDT — Fire 7 digest (scheduled, :17 cron)
+
+Quiet read-only fire, same as fires 3-6. No changes in the hour; all passes no-op or skip; no writes, no commit, no new queue items. P1 roam up to date; P2 inbox 0 new; P3 triage skipped (queued); P4 0 fixes; P5 no change; P6 same two working dirs; P7 clean; P8 not re-scanned; P9 clean, branch 2 ahead; P10 skipped. Lock released.
+
+** 2026-07-20 06:20 CDT — Fire 8 digest (scheduled, :17 cron)
+
+Quiet read-only fire, same as fires 3-7. No changes in the hour; all passes no-op or skip; no writes, no commit, no new queue items. P1 roam up to date; P2 inbox 0 new; P3 triage skipped (queued); P4 0 fixes; P5 no change; P6 same two working dirs; P7 clean; P8 not re-scanned; P9 clean, branch 2 ahead; P10 skipped. Lock released.
+
+* Sentry approval queue (2026-07-19)
+
+Judgment/parked items from the overnight fires. Review top to bottom; run or discard each.
+
+** [Fire 1 · P2] Apply the .emacs.d working/ tracked-from-creation ruling
+- *What:* update the canonical working-files convention + add a gitignored temp/ pattern across projects.
+- *Why:* Craig's ruling relayed from .emacs.d (2026-07-19). Parked because it's a shared-asset + convention change and carries a design decision (temp/ must be ignored in both track and gitignore modes — it's orthogonal to the personal-tooling set).
+- *Prepared:* =working/sentry-2026-07-19-working-files-ruling/proposed.org= (exact edits + the two findings). Filed as the =** VERIFY [#B] working/ tracked-from-creation + gitignored temp/= task in todo.org.
+- *Fires on approval:* edit =claude-rules/working-files.md= (tracked-from-creation + temp/ subsections), =scripts/install-ai.sh= + =scripts/sweep-gitignore-tooling.sh= (emit temp/ ignore in both modes — decide shape (a) vs (b) in the proposal), =.ai/protocols.org= Working-Files Convention (one-line mirror); then =scripts/sync-check.sh --fix= for the .ai mirror, run =make test=, commit.
+
+** [Fire 1 · P3] Tighten the sentry triage-intake pass probe — SPECCED during morning review
+- *What:* the narrow "fix the probe" framing grew, on Craig's flexibility question, into a per-project source-activation model for triage-intake.
+- *Resolution:* written up as a spec for review — =docs/specs/2026-07-20-triage-source-activation-spec.org= (DRAFT), linked from the =** TODO [#B] Triage source activation= task. General plugins gate on a per-project =:TRIAGE_SOURCES:= declaration; project-specific plugins stay active by presence; the activation layer lives in triage-intake Phase 0 so it fixes the interactive over-pull too; sentry's pass-3 probe reads the same signal.
+- *Next:* Craig's deep read → DRAFT → READY → spec-response decomposes the 5 phases. Two decisions still open (declaration format, whether interactive adopts the same gate). No code landed — this item is now tracked by the spec + task, not the queue.
+
+** [Fire 1 · P5] A full interactive task-audit is due
+- *What:* run task-audit.org interactively (its consolidation / parent-retirement / judgment-flag phases need Craig).
+- *Why:* :LAST_AUDIT: is 2026-07-04 (~16 days); task-review-staleness flags 2 top-level tasks unreviewed >7 days. Fire 1's mechanical subset found nothing unambiguous to auto-fix, so the marker was deliberately not bumped.
+- *Fires on approval:* "let's do a task audit" (or task-review for the lighter pass).
+
+** [Fire 1 · P6] File working/inbox-zero-phase-e/ to a permanent home
+- *What:* file the three artifacts under =working/inbox-zero-phase-e/= per working-files.md and update inbound links.
+- *Why:* the backing work (autonomous-batch-execution spec) is IMPLEMENTED, so the working dir is a filing candidate. Filing is a judgment move — 3 inbound =file:= links in =docs/specs/2026-06-16-autonomous-batch-execution-spec.org= point at it and need updating in the same move.
+- *Fires on approval:* rename+move the 3 files flat into their permanent home, update the spec's 3 links, delete the empty working subdir.
+
+** [Fire 1 · P8] Resolve ~4 candidate broken file: links in live docs (report-only)
+- *What:* fix or confirm-intentional the broken links P8 flagged.
+- *Why:* report-only pass; no unattended rewrites. The clearest real one: =docs/specs/2026-06-16-autonomous-batch-execution-spec.org= links =../../.ai/workflows/inbox-zero.org= (renamed to inbox.org). Others (=wrapup-routing-spec-review.org=, a pattern-catalog inbox source, =subprojects-log.org=) may be folded-in/deleted review docs — Craig's call.
+- *Fires on approval:* update the inbox-zero.org→inbox.org link; decide the rest.
+
+** 2026-07-20 Mon @ 15:28:41 -0500 — flushed (auto)
+Auto-flush checkpoint mid-session. Session's shipped work is all committed + pushed (main == origin == velox). In flight: resuming into finishing the Signal pager task ([#B] in todo.org) — the runbook, velox receive timer, and the ssh-vs-linked-device decision. See Summary → Next Steps and Data Collected for the reconciled facts needed to resume blind.
+
+** 2026-07-20 Mon @ 16:48 CDT — Notification vocabulary split (page/text) + agent-page → agent-text rename
+Craig's follow-on from the reply-ambiguity discussion: reserve trigger words by channel. "page me" = desktop notify, "text me" = Signal, "text and page me" = both. Renamed the tool agent-page → agent-text to match (deprecated agent-page shim delegates to it, removable later). Rewrote protocols.org "Paging Craig" → "Reaching Craig", page-me.org, work-the-backlog's away-run logic, INDEX, and the runbook; updated install-ai doc comments; bats renamed agent-page.bats → agent-text.bats (5 pass incl. shim delegation). /review-code + /voice ran; make test exit 0, shellcheck clean, no new lint. Commit 6145489, pushed; make install re-run on both ratio + velox, agent-text + shim resolve on both. Vocabulary decision arc (this session): first "page=Signal, notify=desktop" (rejected, inverted current default), then "page=desktop, message=Signal" (Craig's), then settled on "text me" for Signal (agent-text). NOTE excluded from staging: .claude/settings.json shows model opus→fable (not mine — left unstaged).
+
+** 2026-07-20 Mon @ 23:36 CDT — Post-flush resume: voice #46 handoff + triage phone-push (item 1)
+Resumed from the auto-flush anchor. Two things landed.
+
+Inbox handoff (archsetup, voice pattern #46 comma budget): archsetup edited voice/SKILL.md + voice-profile.org through the shared ~/.claude/skills/voice symlink (points into rulesets) at Craig's 2026-07-20 direction — "no more than two commas per sentence" in the /voice personal pass. Reviewed both files (count 45→46 everywhere, personal-only set four→five, #46 in the attestation high-recurrence set, before/after from the Hyprland draft). /review-code approved, /voice personal on the message, make test green. Committed 2ea5d9a; replied to archsetup; marked PROCESSED (chore 9721a49). The .claude/settings.json model change (opus→fable) was left unstaged — not this session's work.
+
+Item 1 — triage-intake auto-mode phone-push, built. Folded a "Phone delivery" subsection into canonical triage-intake.org: a full-three-section sweep pushes to Craig's phone over Signal via agent-text, with a pointer from End-of-sweep output and a Living Document note. Craig's design ruling (option 1, signal-only): a quiet sweep's "nothing" heartbeat never reaches the phone — silent-until-signal governs the phone channel too — while the in-session heartbeat stays as proof-of-life. Send half only; reply-polling (old phone-recv) deferred to the reply-correlation follow-up (item 2), which owns the recv wiring. sync-check --fix synced the mirror; make test green; /review-code approved; /voice personal. Committed e27aea2. todo.org task closed DONE (send half shipped, recv deferred).
+
+Next: items 2 (reply-correlation spec) and 3 (token-rotation discussion) remain untouched — carried to the next session.
+
+** 2026-07-20 Mon @ 16:14 CDT — Signal pager DONE (all four deliverables + topology decision)
+Finished the task end to end. (1) Runbook: docs/design/2026-07-20-signal-pager-runbook.org — send, read-replies, receive timer, signal-cli account/setup notes, the resolved decision. (2) Receive timer: scripts/signal-receive.sh + scripts/systemd/signal-receive.{service,timer} (roam-sync-shaped, 15-min, no-ops where the account isn't local, 4 bats); stowed via dotfiles common; enabled + verified on both ratio and velox; a manual drain cleared the 47-day staleness live. (3) Topology decision (Craig, option 2 — linked device over ssh-relay-only): ratio linked as Device 2 "ratio-pager", direct send verified; agent-page generalized from a velox-only check to "any machine holding the account sends directly, else relay" (bats updated). (4) protocols.org "Paging Craig": verified accurate, no edit. Publish flow: /review-code caught a runbook/script contradiction (runbook claimed no-op-clean; script errored) → added the account-presence guard + a bats case. make test green, shellcheck clean. Commits: rulesets 302b062 (feat(pager)), dotfiles d8b0462 (feat(systemd)); both pushed; velox synced. todo.org task closed DONE. Craig's follow-up worry addressed: the pager only fires on an explicit page-me or an away-run's end-of-set page; the receive timer is silent (receive-only, no push).
diff --git a/.ai/sessions/2026-07-24-18-05-sentry-implement-pass-review-and-hook-fixes.org b/.ai/sessions/2026-07-24-18-05-sentry-implement-pass-review-and-hook-fixes.org
new file mode 100644
index 0000000..6c60a8f
--- /dev/null
+++ b/.ai/sessions/2026-07-24-18-05-sentry-implement-pass-review-and-hook-fixes.org
@@ -0,0 +1,386 @@
+#+TITLE: Session Context — 2026-07-23
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-23
+
+* Summary
+
+** Active Goal
+
+A long session (2026-07-23 into 2026-07-24) spanning several arcs: applied two Craig-ordered sentry amendments, ran a no-approvals speedrun over four solo tasks, shipped the sentry implement-pass feature itself, ran eight overnight sentry fires that found and fixed real bugs, then subjected the whole night's output to an adversarial review that found defects in the fixes, repaired those, absorbed a repo-wide fail-open security fix from .emacs.d, and processed the inbox to zero. Ended with the branch merged to main (unpushed by Craig's choice) and the session wrapped.
+
+** Decisions
+
+- Sentry gains an opt-in solo-implementation pass (pass 12, gated on =:SENTRY_MAY_IMPLEMENT:=, separate from =:COMMIT_AUTONOMY:=) plus refactor-finding in pass 11. Craig's direction, after the discussion that the branch already contains blast radius and a skeptical premise-first review is the fact-checker that makes fixing-on-a-branch safe. Shipped to main.
+- Reviews must fact-check the *premise* (reproduce the bug) before judging the diff, not just check the diff is clean. Craig's correction; saved as harness memory =feedback-reviews-verify-premise=. Across the night the premise check killed roughly one wrong hypothesis per real bug.
+- Craig chose NOT to push main at wrap. The hook fail-open fix therefore stays undelivered to consuming projects until he pushes. Flagged and reaffirmed.
+
+** Data Collected / Findings
+
+- The dominant defect class across the session, five instances over two projects: a quality gate that enumerates its inputs instead of discovering them, so a new input is silently skipped and the green check reads as covered. Promoted to a KB node this wrap.
+- The secret-scan pre-commit hook failed open on any git error, in ALL FIVE language bundles (not just the elisp one .emacs.d reported). Two of the five were hooks I wrote the day before by copying bash — I propagated the defect. Fixed across all eleven sites; graded [#A].
+- The adversarial review round found four real defects in my overnight fixes (mode-widening and symlink-clobbering in cj-remove-block's atomic write, a same-second backup collision in two tools, a test that deleted real backups from shared /tmp) plus one I'd left: the cj-block range check still can't prove it's deleting the block that was scanned. All repaired except the last, which needs a CLI-contract decision and is filed [#B].
+- I repeated my own worst mistake pattern three times: shipping a change whose correctness depended on shared /tmp state (the backup tests), and twice concluding causation from a single-sample measurement (the audit flake A/B, the /tmp-copy comparison). The re-run/isolation habit caught each.
+
+** Files Modified
+
+- Merged to main (267d1de): cj-remove-block range guard + atomic write, todo-cleanup backup, route_recommend dedupe, audit.bats flake fix, lint.sh bin/ coverage, plus the review-round repairs.
+- Hook fail-open fix (f0c1bc4): all five bundles' pre-commit, the elisp validate-el cap removal, the cross-bundle test now discovering variants, two adopted .emacs.d bats suites.
+- Earlier: sentry.org pass 11/12 + marker (pushed), the speedrun's four fixes, the two .dotfiles amendments, voice #47, four approved parked proposals.
+- KB: =agents/20260724180443-enumerate-vs-discover-gate-failure.org=.
+
+** Next Steps
+
+- Push main (5+ commits ahead, all local). The hook security fix is the load-bearing one.
+- The cj-block wrong-block design question: content assertion vs re-scan vs bottom-up removal. Filed [#B] at the top of todo.org.
+- Seven parked VERIFYs await Craig, [#A] account-binding guard from home first, then the telegram down-is-launch fix and its engine sibling.
+- Optional: ~1600 backup files accumulated in /tmp from the night's runs (harmless, cleared on reboot).
+
+KB: promoted 1 / consulted no
+
+* Session Log
+
+** 02:34 — Startup
+
+Ran startup.org. Rulesets already current; =make install= had nothing new to link; project repo clean and up to date. =.ai/= synced from templates. No prior =session-context.org= — last session (2026-07-20 23:36, signal-pager + triage phone-push) wrapped cleanly.
+
+Findings surfaced: 13 top-level tasks unreviewed >7 days; roam inbox holds 9 items; 8 unprocessed project inbox handoffs plus =inbox/lint-followups.org=; KB has 101 =:agent:= nodes but no best-practices node resolved and nothing matching "rulesets". Active Reminder from 2026-07-14 (Craig's deep read of the sentry spec) reads stale — sentry has since shipped and dogfooded live in takuzu and archangel.
+
+Craig's instruction on arrival: finish startup, then begin sentry.
+
+** 02:35 — Craig-ordered amendments applied before launching sentry
+
+Two of the eight inbox handoffs are Craig's own orders relayed from the dotfiles session, and both gate a correct sentry run, so they land before the loop starts. The rest of the inbox stays for sentry's own inbox pass.
+
+Amendment 1 — sentry pass list (=.dotfiles=, 2026-07-21). Sentry never checks email or messengers, and gains a bug-finding pass. The handoff shipped dotfiles' whole edited =sentry.org=, but that copy forked from a pre-silent-until-signal canonical: rsyncing it would have reverted the heartbeat/digest split committed 2026-07-20. Applied their three intended changes onto current canonical by hand instead, and kept the =:TRIAGE_SOURCES:= activation-probe language their Pass 3 rewrite had dropped.
+
+Reviewed the staged diff before committing and found three defects in my own edit. Pass 11 as Craig worded it ran "the project's linters and suite," which would have violated sentry's own anti-pattern 5 (no per-pass suite run) eleven times a night; it now runs linters and static analysis only and reads the entry baseline's suite result. The Living Document line still said "ten mechanical passes." And the KB-deferral parenthetical called KB promotion "the proposal's eleventh pass," which collided with the real pass 11 — reworded. Suite green (exit 0) before and after. Committed 33949c5, unpushed pending Craig's call.
+
+** 02:43 — Sentry armed, fire 1 (working)
+
+Entry gates all passed: =:COMMIT_AUTONOMY: yes=, tracked tree clean, suite green (exit 0, run twice on this content), no prior =sentry/*= branch, zero behind upstream. Branch =sentry/2026-07-23-ratio= created from HEAD. Loop armed hourly at :07 (job ef8eb9d4, session-only, auto-expires in 7 days). Craig's repo working tree belongs to sentry until the morning merge — reclaiming it early means saying "stop sentry". If he has rulesets files open in Emacs, buffers want reverting after the merge.
+
+Fire 1 digest:
+
+- P1 roam pull — ran, already up to date.
+- P2 inbox zero — ran. Project inbox: 1 of 5 executed (org-drill's memory-sweep completion note, informational, marked PROCESSED + reply sent). 4 park (below). Roam inbox: 9 items, 8 belong to other projects and route at wrap-up, not here; the 1 rulesets item ("every project should have a working and a temp directory") is shipped except for one clause, queued below.
+- P3 triage intake — skipped: no project-specific plugin and no =:TRIAGE_SOURCES:= declaration, so no active source. Correct behavior under the activation gate.
+- P4 todo cleanup — ran, no-op.
+- P5 task audit — deferred to a later fire; takuzu's dogfood note says the full audit is too heavy hourly, and fire 1 already carries the bug-finding load.
+- P6 working-files hygiene — skipped: no =working/= directory.
+- P7 spec status board — ran. No =DOING= spec with a closed build parent.
+- P8 link integrity — ran, and the checker itself turned out to be broken (see P11).
+- P9 git health — ran. =main= is 1 ahead of =origin/main= (the amendments commit, unpushed pending Craig's call). Sentry branch correctly has no upstream. No stale branches.
+- P10 prep freshness — skipped: no =daily-prep/=.
+- P11 bug finding — ran. Rotating area this fire: =lint-org.el= and the synced =.ai/= templates. Three defects verified and filed as graded tasks (commit dc7791d): link resolution against cwd rather than the linted file's directory [#B], todo-format checkers firing on spec files [#C], and the notes.org template tripping four flags in every project [#C]. Two of the three were independently reported by takuzu and smoke tonight, which is what prompted checking them.
+
+The three bug tasks record their severity x frequency arithmetic in the body, including where the two inputs disagreed, so the grades can be argued rather than just overridden.
+
+** 02:47 — Finished the park path the fire skipped
+
+The inbox-boundary Stop hook caught a real gap. Fire 1 wrote the four proposals into the approval queue and stopped there, but =inbox.org='s park path is three things, not one: move the proposal into =working/<slug>/= beside a prepared diff, file a =[#B]= VERIFY carrying the decision package, and reply to the sender. I'd done the queue entry and none of the rest, so the senders were waiting on nothing and the decisions lived only in a session anchor that gets archived.
+
+Completed all four. Each =working/= dir now holds the original proposal, a =proposed.diff=, and the full proposed file. Both org-file diffs were verified by linting the proposed version: the notes.org template goes from four flags to zero, and the fix turned out to need one thing smoke didn't identify — the =invalid-block= pair isn't the example content confusing the parser generally, it's the literal =** Feature Name or Topic= line inside the block, which org reads as a heading. A comma-escape on that one line clears both. Also found two more column-0 bold lines beyond the two reported.
+
+Replies sent to takuzu, archangel, and smoke. dotfiles' ack needed nothing back and is marked PROCESSED. Inbox is at zero.
+
+** 03:35 — Fire 2 (working)
+
+Lock acquired, branch state verified. Digest:
+
+- P1 roam pull — ran, already up to date.
+- P2 inbox zero — ran, no-op. Project inbox at zero. Roam inbox unchanged at 9; 8 belong to other projects and route at wrap-up, the 1 rulesets item is already queued.
+- P3 triage intake — skipped: no active source surviving the mail/messenger exclusion.
+- P4 todo cleanup — ran in check mode across all three passes, zero fixes, zero conversions, zero archives.
+- P5 task audit — mechanical subset only, per the parked guidance and takuzu's note that the full audit is too heavy hourly. Staleness is 20, up from 13 at startup, which is just tonight's 7 new tasks arriving never-reviewed. Not a defect.
+- P6 working-files hygiene — ran, and this fire is the first where its probe fires, since fire 1's park work created =working/=. All four dirs have open backing VERIFY tasks, so nothing to flag. The pass works.
+- P7 spec status board — ran. Seven IMPLEMENTED, two READY (inbox-workflow-consolidation, encourage-kb-contribution). No DOING spec with a closed build parent.
+- P8 link integrity — ran across todo.org, notes.org, the anchor, and all nine specs. Two findings, both in the docs-lifecycle spec, both prose containing a bare =file:= that org parses as a bracketless link (=file:→id:= in a sentence about converting link types, and =keep-file:-links-through-pilot= in a hyphenated phrase). Org behaving as documented rather than a defect, so a digest line and no task.
+- P9 git health — main still 1 ahead of origin, unpushed, awaiting the approval-queue item. Sentry branch correctly has no upstream. No stale branches.
+- P10 prep freshness — skipped: no =daily-prep/=.
+- P11 bug finding — rotating area this fire: the =.ai/scripts/= shell helpers. Shellcheck clean on all seven of the sentry-critical ones (agent-lock, agent-roster, capture-guard, flashcard-sync, inbox-status, self-inject.sh, session-context-path). One SC2034 pair in task-review-staleness.sh, verified as a false positive — the two names are positional fields in a =read -r= that exist to put =value= in the right slot. Digest line, not a task.
+
+*** Retraction: the [#B] I filed in fire 1 was wrong
+
+The main result of this fire is negative. Fire 1 filed a =[#B]= claiming lint-org resolves =file:= links against the process cwd rather than the linted file's directory. It doesn't, and the task is CANCELLED.
+
+The claimed evidence was that linting the notes.org template from the repo root reports two siblings missing while linting from its own directory doesn't. The first half was never run. Those findings came from the =/tmp= copy I made while preparing smoke's diff, and in =/tmp= the siblings genuinely are missing, so org-lint was right. I compared two different files and read the difference as a bug, then wrote a fix direction for a mechanism I hadn't checked.
+
+Caught it here only because P8 surfaced =link-to-local-file= again and I went looking for the checker in lint-org.el to fix it. It isn't there — it's org-lint's own, which contradicted my stated fix and forced the retest. Three runs plus a direct =default-directory= probe settled it.
+
+Correction sent to takuzu, who had been told about it. Their own finding is unaffected and still filed. The useful residue: both =link-to-local-file= and the todo-format checkers are upstream org-lint checkers, so scoping them means filtering org-lint's output, not editing a local checker — which changes the fix direction on the [#C] task too.
+
+** 04:35 — Fire 3 (working)
+
+Lock acquired, branch state clean. Digest:
+
+- P1 roam pull — ran, already up to date.
+- P2 inbox zero — ran, no-op. Both inboxes unchanged.
+- P3 triage intake — skipped: no active source surviving the mail/messenger exclusion.
+- P4 todo cleanup — ran. Hygiene and convert-subtasks no-op; =--archive-done= moved one subtree, the =CANCELLED= lint-org task fire 2 retracted. Committed 962f3c0.
+- P5 task audit — mechanical subset. Staleness 19, down one from the archive.
+- P6 working-files hygiene — ran. All four =working/= dirs still have open backing VERIFY tasks. Nothing to flag.
+- P7 spec status board — ran. Two READY specs, no DOING with a closed parent.
+- P8 link integrity — ran across todo.org, notes.org, and all nine specs. Same two known prose false positives in the docs-lifecycle spec, unchanged. No new findings.
+- P9 git health — main 1 ahead of origin, unpushed, still queued. No stale branches.
+- P10 prep freshness — skipped: no =daily-prep/=, no broken symlinks.
+- P11 bug finding — rotating area this fire: the Python scripts under =.ai/scripts/=. All 15 compile; no Python linter on this box, so the pass became a targeted read of =inbox-send.py=, the script this session has exercised hardest. Three defects verified, filed as two tasks (8995016).
+
+*** The inbox-send finding
+
+The one that matters: =inbox-send= writes straight to the destination path, and =write_text= truncates on open, so any mid-write failure leaves a zero-byte =.org= in *another project's* inbox. That phantom isn't inert — =inbox-status= counts it, so it trips the receiving project's boundary hook and blocks a turn there over a file with no content and no sender context. Meanwhile the sender saw an error and retries, so the target collects a second one.
+
+Reproduced end to end with a non-ASCII message under a C locale with UTF-8 mode disabled: send fails, zero-byte file lands in the destination, =inbox-status= there reports it pending. The encoding case is just the trigger I could force; the defect is the non-atomic write, which a full disk or an interrupted process reaches the same way. Graded [#B], fix direction is temp-file-plus-=os.replace= with =encoding="utf-8"= pinned.
+
+Two smaller ones grouped as [#C]: =send_file= raises an uncaught =PermissionError= traceback because =main= catches only =ValueError= and =FileNotFoundError=, and =discover_projects= doesn't dedupe, so a roots config naming both a parent and its child lists the same project at two indices.
+
+Notable that this is the first fire where the rotating area produced findings in code the night's own work depended on. Fire 1 read the linter, fire 2 the shell helpers, fire 3 the script that carried every reply I sent.
+
+** 05:35 — Fire 4 (working)
+
+Passes 1 through 10 all quiet: roam already current, both inboxes unchanged, todo cleanup no-op across all three passes, all four =working/= dirs still backed by open VERIFY tasks, two READY specs and nothing stuck, the same two known prose false positives in the docs-lifecycle spec, main still 1 ahead and unpushed, no prep dir and no broken symlinks. Staleness 21, up two from fire 3's new tasks.
+
+P11 rotating area: =claude-templates/bin/= — the launcher and paging scripts. Shellcheck clean on all four. The findings came from reading and from a config-sanity check, and one of them is a live condition on this machine rather than a latent code defect.
+
+*** ratio's second Signal account has been cold for 8 days
+
+=signal-cli listAccounts= on ratio warns that messages were last received 8 days ago. The natural read is that the pager channel has gone stale, which would matter — it's the "text me" path. It hasn't. Established by elimination: ratio holds two accounts, =signal-receive.sh= hardcodes the pager (=+15045173983=) as its only target, and the timer's own journal shows it draining that account cleanly every 15 minutes, the last run 2 minutes before the warning printed. So the cold account is Craig's personal number, and nothing on this machine keeps it warm.
+
+What I can't establish is whether that matters. The session history describes that registration as note-to-self with no push, and Signal's tolerance for a quiet linked device isn't something I verified. Filed as [#C] with the uncertainty stated rather than graded up on a guess. If it does matter, the fix is a second timer instance rather than a code change, since =signal-receive.sh= already takes an account argument — which makes it a dotfiles handoff, that repo owning the unit. Not sent: the content is speculative and a 05:35 handoff asserting a problem I haven't confirmed would land as noise.
+
+*** agent-text's direct send is unbounded
+
+The relay path passes =ConnectTimeout=10= to ssh; the local-account path calls =signal-cli send= with no bound. Since agent-text is invoked by agents, a stall blocks the calling turn with no output and no way to tell a hang from a slow send. Not hypothetical contention — the receive timer holds the same account for ~16 seconds every cadence. Filed in the same [#C].
+
+** 06:35 — Fire 5 (working, but the bug hunt came up empty)
+
+Passes 1 through 10 all no-op: roam current, both inboxes unchanged, todo cleanup clean across all three passes, four =working/= dirs still backed, two READY specs, the same two known prose false positives, main 1 ahead unpushed, no broken symlinks. Staleness 22, up one from fire 4's new task.
+
+P11 rotating area: =hooks/= and =scripts/= — the machine-wide hooks and the repo's own install and maintenance scripts, the last major uncovered surface. *Zero real bugs.* All five shell hooks shellcheck clean, all Python hooks compile, =hooks/tests= present, =hooks/__pycache__= correctly gitignored with nothing tracked. Six shellcheck findings across =scripts/=, every one dispositioned as a non-bug:
+
+- =SC2094= (read and write the same file in one pipeline) in =install-ai.sh= and =sweep-gitignore-tooling.sh= — false positive both times. The pattern is a =>>= append with a =[ -s "$gi" ]= stat inside the group. Appending doesn't truncate, and a stat isn't a content read, so the leading-blank-line logic is correct in both the file-exists and file-absent cases.
+- =SC2088= (tilde doesn't expand in quotes) in =doctor.sh= and =audit.sh= — both are display strings printed to the user, not paths used for I/O. =audit.sh= says so in a comment on the line above.
+- =SC2295= (unquoted expansion inside =${..}=) in =audit.sh= and =diff-lang.sh= — technically correct, no live trigger; =$HOME= carries no glob characters.
+- =SC2164= (=cd= without =|| exit=) in =lint.sh= and =status.sh= — =status.sh= already guards its path upstream. =lint.sh= is genuinely unguarded and, since it sets =-u= but not =-e=, a failed =cd= would let it lint the invocation directory instead of the repo root. Confirmed =cd ""= fails rather than silently succeeding, so the hazard is real in shape but needs the running script's own parent to be unreachable. Not reachable in practice; a one-line =|| exit= would close it if anyone touches the file.
+
+This is the first fire whose hunt found nothing, which is the expected shape — takuzu's dogfood report predicted the rotating hunt goes quiet after the first few fires clear the standing defects. Recording the area covered so the next fire doesn't re-read it.
+
+*** Two measurement errors this fire, both caught before they became findings
+
+Worth flagging as a pattern, since fire 1's retraction was the same class. First, a probe loop passed =--convert-subtasks --check= through an unquoted variable; zsh doesn't word-split, so it arrived as one argument and =todo-cleanup= printed a bare =normal-top-level()=. That looked like a tool crash and was my loop. Protocols warns about exactly this.
+
+Second, chasing that, I read =exit=0= from a bad-flag run and nearly filed a silent-pass defect — a cleanup tool that exits 0 on a bad flag would let =wrap-it-up= believe a pass ran when it didn't. The =0= was =tail='s exit code, not emacs'. Measured without the pipeline, both =todo-cleanup= and =lint-org= exit 255 on an unknown flag and on a missing file, and 0 on success. No defect.
+
+Three times tonight a conclusion came from a bad measurement. Twice it was caught in the same fire; once (fire 1) it reached a filed task and a sent handoff before the retest caught it.
+
+** 07:35 — Fire 6 (working) — the night's most consequential finding
+
+Passes 1 through 10 all no-op again: roam current, inboxes unchanged, todo cleanup clean, working dirs backed, two READY specs, the same two prose false positives, main 1 ahead unpushed, no broken symlinks. Staleness 22, flat.
+
+P11 rotating area: =languages/= and the bundle install machinery — the last major uncovered surface. Not quiet.
+
+*** The python and typescript bundles ship no secret-scan hook
+
+=bash=, =elisp=, and =go= each carry =githooks/pre-commit=, =claude/hooks/validate-*.sh=, =claude/settings.json=, and a seed =CLAUDE.md=. =python= and =typescript= carry none of the four. The pre-commit hook is the credential scanner, so installing either of those bundles gives a project no secret scan on commit — while README's Bundle structure section documents all four as what every bundle follows.
+
+Verified live rather than inferred, by scanning every project with =.claude/rules/=: =work= (python) and =clock-panel= (python + typescript) both have no =githooks/= and no settings. All four elisp projects have both. =work= is the one that matters — a work repo is where a leaked credential costs most and is likeliest to reach a company remote.
+
+The history settles intent. Both bundles were added 2026-05-31; =go='s githooks landed 2026-06-02 and =bash='s 2026-06-23. The rollout swept the bundles added *after* these two and skipped these two. And =install-lang.sh= guards its copy with =[ -d "$SRC/githooks" ]=, so the install succeeds and reports nothing missing, which is how this stayed invisible for almost two months.
+
+Filed [#A], SCHEDULED today — the only [#A] of the night. The fix is find-not-fix per the pass contract, so nothing was changed. The task carries a second half worth as much as the port itself: make =install-lang= warn when a bundle lacks a component the README documents, so the next partial bundle announces itself rather than installing quietly.
+
+*** Same drift shape, smaller
+
+=bash='s pre-commit has =cd "$REPO_ROOT" || exit 1=; the =go= and =elisp= copies of that line dropped the guard. Impact is genuinely low — git chdirs to the working-tree root before running a hook, so the checks run against the right tree regardless, and neither script sets =-e=. Filed [#C] for the two-character fix, mostly because the shape is the same as the [#A]: a fix landed in one bundle copy and stopped there.
+
+Two fires' worth of evidence now says bundle-to-bundle propagation is where this repo leaks changes.
+
+** 00:51 — Fire 1 of the 2026-07-24 run (working) — first fire with pass 12 live
+
+Lock acquired, branch =sentry/2026-07-24-ratio= verified, tree clean outside the spine. Digest:
+
+- P1 roam pull — ran, already up to date.
+- P2 inbox zero — ran, no-op. Inbox at zero (the question-capture proposal was parked during entry).
+- P3 triage intake — skipped: no active source surviving the mail/messenger exclusion.
+- P4 todo cleanup — ran with real work. =--archive-done= moved three completed speedrun tasks out of Open Work into Resolved; =--convert-subtasks= normalized the tree. Committed.
+- P5 task audit — mechanical subset. Staleness 13, back to the pre-speedrun baseline now that the four solo tasks closed.
+- P6 working-files hygiene — ran. All =working/= dirs still have open backing VERIFY tasks (the parked proposals). No orphans.
+- P7 spec status board — ran. Two READY specs, nothing stuck.
+- P8 link integrity — ran. The docs-lifecycle spec still shows its two known prose false positives (=file:→id:= and =keep-file:-links-through-pilot=, bare =file:= tokens org parses as bracketless links). Correctly unaffected by tonight's spec-scoping, since =link-to-local-file= is org-lint's own checker, not a todo-format one. No new findings.
+- P9 git health — main level with origin, sentry branch correctly has no upstream, no stale branches.
+- P10 prep freshness — skipped: no =daily-prep/=.
+- P11 bug and refactor finding — ran, first fire under the widened pass. Rotating area: the cross-project routing scripts (=route_recommend.py=, =broadcast.py=), untouched by the previous six areas. One verified latent bug filed [#D] (below). Refactor note not worth a task: =recommend='s two weak-tier branches collapse to a single =_tiebreak= call, since =_tiebreak= on a one-element list returns that element. Two lines, no behavior change, so it's a digest line rather than backlog noise.
+- P12 solo-task implementation — *active this run* (=:SENTRY_MAY_IMPLEMENT: yes=) but a correct no-op: zero eligible tasks. All four open =:solo:= TODOs were completed in tonight's speedrun, so the ready bucket is empty. Nothing to implement, nothing deferred.
+
+*** The finding
+
+=route_recommend='s =discover_destination_names= collapses projects to bare basenames, so two projects sharing a basename across roots would both literal-match and read as an ambiguous tie, downgrading a correct strong match to weak. Reproduced by direct probe. Latent rather than live: 27 projects, 27 distinct basenames today. The destination stays right, only the tier is wrong, so the cost is an extra routing prompt. Filed [#D] with the order-preserving dedupe as the fix.
+
+** 01:33 — Fire 2 (working) — pass 12's first real implementations
+
+Digest:
+
+- P1 roam pull — ran, already up to date.
+- P2 inbox zero — ran. Project inbox at zero. The roam inbox dropped 9 → 5 (another session filed some); all 5 remaining belong to archsetup or work, none rulesets-claimed, so roam mode is correctly a no-op here.
+- P3 triage intake — skipped: no active source.
+- P4 todo cleanup — ran, all three checks clean (fire 1 did the archiving).
+- P5 task audit — mechanical subset. Staleness 13, flat.
+- P6 working-files hygiene — ran. All =working/= dirs still backed by open VERIFYs.
+- P7 spec status board — ran. Two READY specs, nothing stuck.
+- P8 link integrity — ran. Only the docs-lifecycle spec's two known prose false positives.
+- P9 git health — main level with origin, sentry branch correctly upstream-less.
+- P10 prep freshness — skipped: no =daily-prep/=.
+- P11 bug and refactor finding — rotating area: the cj-comment tooling (=cj-scan.py=, =cj-remove-block.py=). Two findings filed, one of them serious.
+- P12 solo-task implementation — *two tasks implemented and committed to the branch* (17f5d48, 1b0f284). Both premise-checked before a line was written.
+
+*** The serious find: cj-remove-block destroys content
+
+=looks_like_cj_range= validated only the first and last lines of a range. A span from one cj block's opener to a *later* block's closer passed, and the removal then deleted everything between — prose, headings, whole tasks — silently, exit 0. That is exactly the failure the check exists to prevent, and drift is its normal case, since =respond-to-cj-comments= edits the file while processing and a file under cj review usually holds several blocks. Reproduced on a two-block fixture that lost a heading and two content lines. Filed [#B], then implemented in pass 12 after an independent re-verification on a different fixture shape.
+
+Same file, second defect fixed alongside: =remove_range= rewrote the org file with a bare =write_text= (truncates on open) and took no backup, so a mid-write failure would leave =todo.org= truncated with nothing to recover from. =lint-org.el= already backs these files up to =/tmp= before mutating; cj-remove-block now matches that and writes atomically.
+
+*** The measurement lesson, third time tonight
+
+A full-suite run went red on =audit.bats= test 4. I stashed my changes, saw it pass clean, and had a one-sample A/B pointing straight at my own diff. Re-ran three times with the changes restored and it passed every time. The failure is an intermittent teardown flake (=rm -rf= racing something still writing into a fixture =.git/objects=), not my change, and I nearly filed the wrong cause off a single sample. Filed [#C] with the git-background-gc theory explicitly labelled a lead rather than a verified cause.
+
+Committed only on a genuinely green re-run, not on the red with a hand-wave.
+
+** 02:33 — Fire 3 (working) — the flake's cause traced, and my own lead disproved
+
+Digest:
+
+- P1 roam pull — ran, already up to date.
+- P2 inbox zero — ran, no-op. Project inbox at zero; roam holds 5, none rulesets-claimed.
+- P3 triage intake — skipped: no active source.
+- P4 todo cleanup — ran. =--archive-done= moved fire 2's two completed tasks to Resolved. Committed.
+- P5 task audit — mechanical subset. Staleness 13, flat.
+- P6 working-files hygiene — ran, all dirs backed.
+- P7 spec status board — ran, two READY, nothing stuck.
+- P8 link integrity — ran, only the known docs-lifecycle prose false positives.
+- P9 git health — main level with origin, branch upstream-less, nothing stale.
+- P10 prep freshness — skipped.
+- P11 bug and refactor finding — no new findings. The fire's whole investigative budget went to confirming fire 2's filed flake, which is the honest place for it; a hunt that finds nothing new is a result.
+- P12 solo-task implementation — one task implemented and committed (7f45d4b).
+
+*** Confirming the cause before fixing, and disproving my own lead
+
+Fire 2 filed the =audit.bats= flaky teardown with a stated theory: =gc.auto='s loose-object threshold. Pass 12's premise check went after that theory rather than the fix, and killed it — a fixture holds five objects against a default threshold of 6700, so that mechanism cannot fire.
+
+The real cause came from a =GIT_TRACE= run: =git commit= spawns =git maintenance run --auto --quiet --detach= on git 2.55. The commit returns while the detached process is still writing a pack, and teardown's =rm -rf= races it. Every failed run had left a =tmp_pack_*= behind, which is the thread that led there.
+
+Fix: =maintenance.auto false= plus =gc.auto 0= in the fixture, killing the background writer rather than retrying the delete (a retry loop hides a live process instead of removing it). Validated over 20 consecutive clean runs against a ~1-in-8 baseline, and recorded honestly in the task that 20 clean runs alone would be ~7% likely by luck — the trace is the evidence, the runs confirm.
+
+This is the second night running where the premise check changed the outcome. Fire 2 it stopped a wrong causation call; here it stopped me implementing a fix for a mechanism that was never operating. Both times the cost of checking was minutes and the cost of not checking would have been a plausible, wrong, committed change.
+
+** 03:33 — Fire 4 (working) — two hypotheses killed, one real gap closed
+
+Digest:
+
+- P1 roam pull — ran, up to date. P2 inbox zero — no-op, both surfaces clean. P3 triage — skipped, no active source.
+- P4 todo cleanup — ran. Archived fire 3's completed task to Resolved. Committed.
+- P5 task audit — mechanical subset. Staleness 13, flat all night.
+- P6 working-files — all dirs backed. P7 spec board — two READY, nothing stuck. P8 link integrity — only the known prose false positives. P9 git health — clean. P10 — skipped, no prep dir.
+- P11 bug and refactor finding — rotating area: the org-file mutators, chosen because =cj-remove-block= yielded a serious find in that same class last fire. One gap filed.
+- P12 solo-task implementation — one task implemented and committed (0686784).
+
+*** What the area review actually found, and what it disproved
+
+The lead was that =todo-cleanup.el= might lose data. It rewrites =todo.org=, creates archive files, and moves subtrees *between* files, which is the shape that bit =cj-remove-block=. Two specific hypotheses, both tested, both wrong:
+
+- *"A mid-move failure loses a subtree from both files."* No. The order is delete-from-buffer, write-archive, save-todo.org-last, so an archive-write failure aborts before the save. Verified by making the archive directory unwritable: exit 255, =todo.org= byte-identical, content intact.
+- *"Errors are swallowed on the mutation path."* No. The only =ignore-errors= in the file wrap =call-process "git"=, never a write.
+
+What survived was narrower and real: todo-cleanup mutates with *no backup*, while both sibling mutators (=lint-org.el=, =wrap-org-table.el=) copy to =/tmp= first, and =cj-remove-block= joined them last fire. It is also the one that runs most often. Confirmed empirically that Emacs's own backup does not fire under =--batch -q=, so there was genuinely no undo short of git. Filed [#C], then implemented in P12.
+
+Also checked my own change for regression rather than assuming: a missing input file exits 255 and creates nothing, identical to the pre-change version tested from git.
+
+*** Running tally on the premise habit
+
+Three fires, three times it changed the outcome. Fire 2 it stopped a wrong causation call. Fire 3 it disproved my own filed =gc.auto= theory before I could implement against it. Here it killed two data-loss hypotheses before they became tasks, leaving only the gap that was actually there. The pattern is consistent: the cheap check keeps a plausible story from becoming a committed change.
+
+** 04:33 — Fire 5 (working) — the first real defer, and last fire's fix proving itself
+
+Digest:
+
+- P1 roam pull — ran, up to date. P2 inbox zero — no-op, both surfaces clean. P3 triage — skipped, no active source.
+- P4 todo cleanup — ran, archived fire 4's completed task. *Confirmed last fire's backup fix working live*: the real =--archive-done= run left =/tmp/todo.org.before-todo-cleanup.20260724-043331=. Dogfooded within an hour of shipping.
+- P5 task audit — staleness 13, flat all night. P6 working-files — all backed. P7 spec board — two READY. P8 link integrity — only the known prose false positives. P9 git health — clean. P10 — skipped.
+- P11 bug and refactor finding — rotating area: the attachment/email handlers, chosen because they parse genuinely untrusted input, unlike every internal-tooling area covered so far. One finding filed.
+- P12 solo-task implementation — *deferred*, and correctly. First defer of the run.
+
+*** The find: attachment filenames are partly sanitized, in two different ways
+
+Both writers derive on-disk names from the =filename= an email declares, and both sanitize incompletely, covering *different* gaps. =eml-view= cleans the name but interpolates the extension raw. =gmail-fetch='s =safe_filename= handles path separators and leading =..= and nothing else. Probed both with the same adversarial set: a =; rm -rf ~= extension survives in both, a literal newline survives in both, a 300-character extension produces a 314-character filename in both.
+
+Two things I checked so this doesn't get over-graded later. Not RCE — files are written through Python =open=, never a shell. Not path traversal — =splitext= only returns an extension when the last dot follows the last separator, so =ext= can never hold a slash, and the traversal case is neutralized in both scripts. The genuine harms are narrower: a newline in a filename breaks downstream tooling that reads the directory as a line-delimited list, and an unbounded extension blows the 255-byte limit so a crafted attachment aborts extraction. Graded [#C] on that honest read rather than the scarier one.
+
+I also corrected my own framing mid-investigation. I first read this as the familiar "one sibling hardened, the other not" pattern from the last two fires. It isn't — =safe_filename= is narrower than it looks, and the two scripts are *differently* incomplete. Neither handles newlines or length.
+
+*** Why pass 12 deferred instead of implementing
+
+The fix itself is clear, but where the shared sanitizer lives is a design call: a shared helper module (clean, but a new synced template file plus =importlib= gymnastics for kebab-named scripts), duplicate it in both (self-contained, but drift — the exact defect class that produced three separate findings tonight), or patch each in place (smallest diff, permanent divergence). That is deliberation, not a quick factual question, so checklist item 4 fires and the unattended loop defers. Filed the VERIFY with the three options and my lean, and left the task *un-=:solo:=-tagged* so a later run doesn't pick it up and guess.
+
+This is the checklist discriminating rather than rubber-stamping. Four fires implemented; this one correctly didn't.
+
+sentry at 05:33: nothing (bug hunt swept =scripts/*.py=; two candidate gaps both disproved — =workflow-integrity.py= *is* gated, its bats runs the real checker against the real canonical tree under =make test=, and =update-skills.py= is an on-demand maintenance command rather than a gate. Recorded so neither gets re-investigated.)
+
+sentry at 06:33: nothing (bug hunt swept =wrap-org-table.el=, the third org-file mutator and the one with prior history — its load-time dispatch caused the 2026-07-09 corruption. Clean on every probe: the entry-script guard correctly refuses to dispatch when lint-org merely =require='s it, it backs up before writing like its siblings, and it is block-aware — a table inside =#+begin_example= stayed verbatim while a real over-budget table wrapped onto continuation rows with rules. Second consecutive quiet hunt.)
+
+** 07:33 — Fire 8 (working) — a time bomb I planted four fires ago went off
+
+Digest: P1-P10 all no-op (roam current, both inboxes clean, todo cleanup clean, staleness 13, working dirs backed, two READY specs, only the known prose link false positives, git clean). P11 found a real coverage gap. P12 implemented it, and the suite caught a regression of my own making.
+
+*** The find: the only ungated shell in the repo
+
+=scripts/lint.sh= sweeps =scripts/*.sh=, the language hooks, and the language githooks. It never touched =claude-templates/bin/= — zero references. Those four scripts (=ai=, =agent-text=, =agent-page=, =install-ai=) are the ones =make install= symlinks onto PATH, which makes them the *most* exposed shell in the repo and left them the only shell with no gate over it. All four are clean today, so the guard is a no-op by design; it exists so a future regression can't pass silently. The =ai= launcher was hardened to 42 tests recently and nothing enforced that going forward.
+
+The test pins *coverage*, not cleanliness: it plants a broken file in each swept location and asserts lint complains, so a location that stops being swept fails the suite rather than passing quietly.
+
+Surfaced a bigger question I did *not* answer overnight: rulesets ships shellcheck enforcement to consuming projects (the bash bundle's pre-commit, =validate-bash.sh=) and runs none on itself. Filed as a VERIFY, because turning it on would surface the false positives dispositioned earlier this session and choosing between fixing them or adding disable-directives is a preference, not a fact.
+
+*** The regression: my own test, detonating on schedule
+
+The full suite went red on the two backup tests I added in fire 4. Not a flake — a time bomb. They globbed =/tmp/todo.org.before-todo-cleanup.*=, but the backup name derives from the file's *basename*, and the real =todo.org= shares it. So a live sentry run's genuine backup was indistinguishable from the test's own artifact, and the check-mode test (which asserts *no* backup exists) failed the moment fire 5's real archive pass created one.
+
+They passed when written only because no real backup existed yet. Three now sit in =/tmp=. Both tests rebind =temporary-file-directory= to a private dir, and I verified they pass with the real backups present rather than by clearing them.
+
+Worth naming plainly: I shipped a test whose correctness depended on the state of a shared directory that the code under test writes to in production. That is the "no shared mutable state" rule in =testing.md=, and I broke it while fixing a different durability bug. The suite caught it two fires later, which is the argument for running the full suite every fire rather than only the touched file.
+
+* Sentry approval queue (2026-07-23)
+
+Five items. Each names what, why, and the exact edit. Items 2 through 5 are also filed as =VERIFY [#B]= tasks in =todo.org= with prepared diffs under =working/=, so they survive this anchor being archived — say "approve the parked <topic>" for any of them.
+
+** 1. Push main to origin
+
+What: =git switch main && git push origin main= (then switch back, or leave main checked out if sentry is done).
+
+Why: commit 33949c5 (the two .dotfiles amendments) is on main and unpushed. The sentry spec called this out — an unpushed commit on main diverges across ratio and velox and breaks the next startup fast-forward. It's ahead-only, so the push is clean. Held because =commits.md= requires explicit confirmation before any push.
+
+** 2. interaction.md — remove the fenced-code-block carve-out (org-drill)
+
+What: in =claude-rules/interaction.md=, the "No Reverse-Video Highlighting in Chat Output" rule currently says fenced code blocks "are acceptable when the user explicitly wants a block to copy". Replace that sentence with a plain-text-always statement covering fences as well.
+
+Why: org-drill relayed Craig's 2026-05-30 direction ("always always list it out without markup"), saved there as a project memory. The rule as written contradicts it, and the directive belongs in the shared rule rather than one project's memory. It's a convention change to an always-on rule, so it parks rather than lands.
+
+Note: this session violated the tightened form of the rule several times already (fenced blocks in chat), which is evidence for the proposal rather than against it.
+
+** 3. sentry.org Living Document — fold in the two dogfood runs (takuzu, archangel)
+
+What: append to sentry.org's pass list and notes: (a) make the rotating-angle bug hunt an official pass — already done tonight as pass 11, so this reduces to noting takuzu's corroboration; (b) add randomized property sweeps as a sanctioned quiet-fire activity; (c) note that =todo-cleanup --archive-done= touches =.gitignore= on its first archive, so an "org-only" pass can produce a real commit and trigger the fire-end suite; (d) note that in a project gitignoring =.ai/=, quiet fires produce zero commits, so =git log main..sentry/*= understates the night and the anchor's heartbeat list is the only record; (e) replace the per-fire full task audit with a mechanical subset hourly plus judgment items queued once nightly.
+
+Why: two independent first-live-run reports, no engine defects in either. Item (e) matches what fire 1 did by instinct (P5 deferred). sentry.org is a shared synced asset, so the edit parks.
+
+** 4. notes.org template — clear the four lint flags (smoke)
+
+What: in =claude-templates/.ai/notes.org=, rephrase the two column-0 =**bold**= lines so neither starts with =**=, and comma-escape the =#+begin_example= block's own markers. Then =scripts/sync-check.sh --fix=.
+
+Why: filed as a [#C] bug task tonight with the reproduction. The edit itself parks because it changes a template every project inherits. Related finding worth Craig's attention: those two flags are mechanical, not judgment, so =lint-org --fix= run anywhere would rewrite the template and create drift against canonical.
+
+** 5. wrap-it-up.org — delete temp/ at wrap (from Craig's own roam capture)
+
+What: add a cleanup sub-step to =.ai/workflows/wrap-it-up.org= that removes the project's =temp/= contents during teardown.
+
+Why: Craig's roam inbox item asks for exactly this ("temp is ... deleted as a part of the wrap up sequence"). The rest of that capture shipped on 2026-07-20 (=working/= tracked from creation, =temp/= gitignored, the sweep backfilling it), but =wrap-it-up.org= has no mention of =temp/= at all, so this clause was missed. Parks as both a shared-asset edit and a destructive one.
+
+** 02:35 — Craig-ordered amendments applied before launching sentry (earlier)
+
+Amendment 2 — KB personal roots (=.dotfiles=, 2026-07-21). =~/.dotfiles= classified Unknown under =knowledge-base.md='s personal-roots list, blocking KB writes from that project twice (the sentry pass-list rule tonight, the xdg-desktop-portal gotcha on 2026-07-04). Grepped for other copies of the enumeration: =knowledge-base.md:25= is the only place that lists roots for *classification*. The other three hits (=triggers.md=, =broadcast.org=, =session-harvest.org=, =work-the-backlog.org=) enumerate roots for project *discovery*, a different question, and =~/.dotfiles= reaches those through the machine-local =~/.claude/inbox-roots.txt=. Edited the one line.
diff --git a/.ai/sessions/2026-07-25-09-24-invalid-block-filter-push-task-review.org b/.ai/sessions/2026-07-25-09-24-invalid-block-filter-push-task-review.org
new file mode 100644
index 0000000..d0814f9
--- /dev/null
+++ b/.ai/sessions/2026-07-25-09-24-invalid-block-filter-push-task-review.org
@@ -0,0 +1,83 @@
+#+TITLE: Session Context — 2026-07-24
+#+AUTHOR: Craig Jennings
+
+* Summary
+
+** Active Goal
+
+An evening of backlog clearing in the order Craig picked: push the delivery-blocking commits, fix the lint-org =invalid-block= false positive that home had just unblocked, then run a task-review cycle. All three finished.
+
+** Decisions
+
+- The =invalid-block= fix is a filter on org-lint's output, not a local checker edit. Home settled the open question and I re-verified both halves before acting: the string appears nowhere in =lint-org.el=, and =org-lint--checkers= enumerates it in batch Emacs. Same resolution as the earlier =link-to-local-file= episode.
+- Left the =,**= comma-escape in =claude-templates/.ai/notes.org= in place. The task said the fix "lets that escape be reverted," but the escape is correct org for a literal =**= inside a verbatim block, so reverting trades correctness for nothing once the finding is suppressed. Reasoning recorded in the closed task rather than acting on the permission silently.
+- Task review: all seven in the batch kept as-is, no new =:quick:= or =:solo:= tags, confirmed by Craig in one pass.
+- Work's sentry loop was left alone. Craig said "sentry stop" here, but the running loop belongs to the work project (branch =sentry/2026-07-25-ratio=, firing hourly into =aiv-work:1.1=), and the stop procedure's later steps — lock release, branch disposition, approval queue — need that session's context. Surfaced rather than half-executed.
+
+** Data Collected / Findings
+
+- =invalid-block= is org-lint's own checker. Reproduced the false positive three ways: a paired example block with a heading-shaped body line flags both delimiters, a paired src block holding a literal =#+end_example= flags three lines, and a genuinely unterminated block flags once and must keep doing so.
+- Verified the fix against home's live fixture in both directions: its =.ai/notes.org= produced exactly the two reported findings (lines 386 and 398) under the pre-change script and zero under the new one, file untouched.
+- The uppercase-delimiter path (=#+BEGIN_EXAMPLE=) was handled but untested. Confirmed it was a real trigger — 2 findings before, 0 after — before adding the boundary test.
+- Part of the task-review staleness count measures work that can't be delegated rather than work nobody read. This batch was the deliberation-heavy tail, where every task needs a decision from Craig mid-stream, which is why the speedruns kept stepping past them.
+- The KB orphan task cites a 2026-07-01 snapshot of 53 agent nodes. Startup counted 104, so the KB has doubled and the snapshot is worth even less than the task body assumed.
+- Five =[#D]= tasks carry no =:LAST_REVIEWED:= at all. The staleness script excludes =[#D]= but =lint-org= flags them, so the two tools disagree permanently. Left alone; needs a decision about which is right.
+
+** Files Modified
+
+- =claude-templates/.ai/scripts/lint-org.el= (+ mirror) — =lo--matched-block-regions= pairs blocks by line scan under org's real rule, memoized on the buffer modification tick; =lo--handle-item= drops an =invalid-block= finding inside a paired region.
+- =claude-templates/.ai/scripts/tests/test-lint-org.el= (+ mirror) — five tests: heading-in-example, literal =#+end_example= in src, uppercase delimiters, unterminated block still reports, and one file with both proving per-block scoping.
+- =todo.org= — closed the =invalid-block= task with its verification record, stamped seven review dates, inserted a missing properties drawer.
+- KB: =agents/20260725093500-parser-cannot-verify-its-own-misreading.org=.
+
+** Next Steps
+
+- Work's sentry is still running on ratio and untouched. Stopping it properly means saying "sentry stop" in the work session (pane =aiv-work:1.1=), which handles the branch disposition and the overnight approval queue.
+- Home is waiting on this push to re-run its =invalid-block= fixture.
+- Next review batch starts with the agent-source improvements and flashcard tooling tasks.
+- Seven parked VERIFYs still await Craig, the =[#A]= account-binding guard from home first.
+- The sentry spec still wants Craig's deep read before the READY flip.
+
+KB: promoted 1 / consulted no
+
+* Session Log
+
+** Startup
+
+Startup ran clean: rulesets already current, =make install= had nothing new to link, project repo up to date with origin (5 commits ahead, unpushed — carried over from the prior session by Craig's choice). =.ai/= synced from templates with no churn. No crash anchor. Task staleness reported 13 top-level tasks unreviewed for >7 days.
+
+** Inbox — home's answer on invalid-block
+
+One pending handoff: home answering the open question left in the =[#C] lint-org invalid-block false-positives= task — is =invalid-block= lint-org.el's own checker or org-lint's? Home says org-lint's, and I re-verified both halves rather than taking it: =grep invalid-block= over =claude-templates/.ai/scripts/lint-org.el= returns nothing, and =org-lint--checkers= enumerates =invalid-block= in batch Emacs alongside =link-to-local-file=. So the fix is a filter on org-lint's output, matching the =link-to-local-file= episode.
+
+Folded into the existing task as a dated sub-entry rather than filing anything new — the task was already filed and only needed its open question closed. Also recorded home's regression fixture: its own =.ai/notes.org= PENDING DECISIONS block (lines 386-398) is left unescaped on purpose and trips both delimiters, so the filter should take those two findings to zero without touching the file. Replied to home confirming, and asked them to keep the fixture unescaped pending a ping when the filter lands. Inbox back to zero.
+
+** Pushed main
+
+Craig picked the evening's order: push, then the invalid-block fix, then a task review. Pre-push reconcile showed ahead-only by 5, no divergence, so =git push origin main= went out and verified at 0/0. The secret-scan fail-open fix is now delivered to consuming projects.
+
+** invalid-block filter
+
+Built test-first. Reproduced the failure three ways before writing anything: a paired example block with a heading-shaped body line flags both delimiters; a paired src block holding a literal =#+end_example= flags three lines; a genuinely unterminated block flags once and must keep doing so. Wrote four ERT tests covering those plus a mixed file, watched three fail for the right reason, then implemented.
+
+The fix is =lo--matched-block-regions=: a line scan that pairs blocks under org's real rule (once open, only the block's own =#+end_TYPE= closes it), memoized on the buffer's modification tick. =lo--handle-item= then drops an =invalid-block= finding falling inside a matched region, delimiters included, since org-lint reports at the delimiters. Line-scanning instead of asking org is the whole point — org's parser is the thing that's confused.
+
+One test failed after the implementation on an off-by-one in my own expectation (the unterminated opener is line 7, not 8); the code was right and I corrected the test. Verified against home's live fixture both ways: two findings under the pre-change mirror copy, zero under the new canonical, home's file untouched. Synced canonical → mirror, full =make test= green at exit 0.
+
+Review caught two things I fixed rather than filed: the uppercase-delimiter path was handled but untested (confirmed it was a real trigger — 2 findings before, 0 after — then added the boundary test), and the cache-tick comparison used =eq= where =eql= is strictly correct. Verdict Approve, committed 8822b0d after the voice pass. Pinged home that the filter landed and the fixture can be re-run.
+
+Left the =,**= comma-escape in the notes template alone. The task said the fix "lets that escape be reverted," but the escape is correct org for a literal =**= in a verbatim block, so reverting buys nothing now that the finding is suppressed. Recorded the reasoning in the closed task rather than acting on the permission silently.
+
+** Task review
+
+Batch of 7 from the staleness script, oldest first. Every one came back Keep with no new =:quick:= or =:solo:= tag, and Craig confirmed the batch in one pass. Backed todo.org up to /tmp first (matching the mutator convention) and stamped by exact line number rather than a global replace, since ten other tasks carried the same 2026-07-13 date and only six were in the batch. The Sentry vNext task had no properties drawer at all, so it got one.
+
+Two observations worth keeping. The uniform Keep-with-no-tags result isn't the review going soft: this batch is the deliberation-heavy tail, where every task needs a decision from Craig somewhere in the middle, which is exactly why the speedruns kept stepping past them and why they aged. So part of the staleness count is measuring work that can't be handed off rather than work nobody read. And the KB orphan task cites a 2026-07-01 snapshot of 53 agent nodes; tonight's startup counted 104, so the KB has doubled and the snapshot is worth even less than the task body already assumed.
+
+** Inbox — home's acknowledgment
+
+Home replied confirming both messages landed, the fixture stays unescaped, and they'll pick the filter up after the rulesets push and their next clean startup sync. A pure FYI asking nothing, so it skipped the skeptical review and got no reply (acking an ack loops). Deleted; inbox back to zero. It does corroborate that the unpushed commit is the only thing standing between home and the fix.
+
+** Task review (cont.)
+
+Staleness went 13 → 6. Lint flags five more tasks missing =:LAST_REVIEWED:= entirely (lines 327, 336, 434, 593, 602) — all =[#D]=, which the staleness script excludes but the checker doesn't. Pre-existing, not touched.
diff --git a/.ai/sessions/2026-07-25-15-34-clean-wrap-and-inbox-safe-sync.org b/.ai/sessions/2026-07-25-15-34-clean-wrap-and-inbox-safe-sync.org
new file mode 100644
index 0000000..737b1c5
--- /dev/null
+++ b/.ai/sessions/2026-07-25-15-34-clean-wrap-and-inbox-safe-sync.org
@@ -0,0 +1,85 @@
+#+TITLE: Session Context — 2026-07-25
+#+AUTHOR: Craig Jennings
+
+* Summary
+
+** Active Goal
+
+Guarantee that rulesets can only report a successful wrap with a completely clean Git worktree, while allowing other projects to refresh rulesets when its only residue is untracked inbox deliveries. Implemented, reviewed, fully tested, and prepared for a strict self-hosted wrap.
+
+** Decisions
+
+- One executable, =git-worktree-gate=, owns both repository-state policies. =strict= means no staged, unstaged, untracked, dirty-submodule, or in-progress-operation state; =sync-safe= permits only untracked paths beneath =inbox/=.
+- Cleanup failure is not a degraded wrap. It leaves the session open and must report each path, its Git state, why it cannot be resolved safely, and the decision Craig needs. Dirty-file deferrals, valediction, and teardown are forbidden in that state.
+- Final wrap verification is a HEAD-bound certificate stored in the Git directory and freshly rechecked inside the existing teardown hook. Integrating the check avoids the concurrent-hook race documented by Codex.
+- Another project's structured Edit/Write call may not resolve through an installed symlink into rulesets. The runtime hook denies it and points the sender to =inbox-send rulesets=.
+- The two MCP-registry handoffs were consolidated into one parked =[#B]= specification decision. The memory auditor remains separate; no machine-owned MCP configuration was promoted into canonical rulesets.
+
+** Data Collected / Findings
+
+- The prior rulesets archive completed at 09:24; the inherited three tracked changes were written at 10:27. The repository was dirtied after wrap through a later write path, which is why wrap certification alone needed the symlink-aware boundary guard.
+- The previous wrap prose contradicted itself: clean Git state was an exit criterion, but the leftover and inbox sections allowed explicit deferral. The teardown hook checked only the sentinel.
+- Adversarial review found a fail-open process-substitution edge: a low-level =git status= failure could appear as an empty stream. The gate now captures status and its exit code in Git-directory temporary files and blocks on failure.
+- Current Codex hooks support Stop blocking with =continue=false= and run matching commands concurrently. A user-level =hooks.json= is installed; a new Codex session must complete the normal hook review/trust step.
+- Full =make test= passed twice on the implementation. The final run includes 436 core Python tests, 72 hook tests, language suites, ERT suites, and all Bats suites. Focused additions cover inbox-only pulls, staged/unstaged/untracked/submodule/operation states, status failure, certificate/HEAD drift, Claude and Codex Stop outputs, installation, and realpath-based write denial.
+- The wrap roam sweep found Craig's 120-column table question. It was already enforced by =org-tables.md=, =lint-org='s =org-table-standard= judgment, and =wrap-org-table.el=; the live lint pass flagged the existing over-wide table. Removed the duplicate capture and synced roam.
+
+** Files Modified
+
+- =claude-templates/bin/git-worktree-gate= — shared strict/sync-safe classifier plus certificate/verify modes.
+- Startup protocol/workflow mirrors and =claude-templates/bin/ai= — inbox-only state remains visible but no longer blocks fast-forward refresh.
+- Wrap protocol/workflow mirrors and =hooks/ai-wrap-teardown.sh= — no deferral escape, actionable hard blocker, final certificate, fresh teardown verification.
+- =hooks/rulesets-write-boundary.py=, Claude/Codex hook configuration, cross-project rule, Makefile, and hook documentation — prevent structured writes through installed symlinks and install the enforcement on both runtimes.
+- Bats and pytest suites — repository-state, launcher, teardown, installer, and cross-project boundary regressions.
+- =todo.org= and workflow state — parked the consolidated MCP registry spec decision and recorded inbox processing.
+- Three inherited backlog files — retain the post-09:24 definition that speedrunnable means =:solo:=.
+
+** Next Steps
+
+- Start a new Codex session and review/trust the new user-level hooks when =/hooks= prompts; Claude already reads the linked hook configuration.
+- Say "spec the MCP registry sync" when ready to design the separate host-level registry reconciler.
+- The existing =inbox/lint-followups.org= pipeline retains its 14 current judgment items, including the over-wide table and older missing review stamps/links; they do not represent uncommitted work after this wrap.
+
+KB: promoted 0 / consulted no
+
+* Session Log
+
+** Startup and clean-worktree investigation
+
+Ran the required startup workflow. The canonical rulesets pull and template sync were blocked by three tracked modifications; two untracked inbox handoffs were also pending. Read the project and global behavioral rules, recent session archive, wrap workflow, relevant hook and launcher code, and the current Codex MCP and hook documentation needed to evaluate the inbox proposals.
+
+Investigated Craig's requirement that rulesets finish with an absolutely clean worktree while still allowing downstream projects to sync when rulesets has received inbox deliveries. The current wrap workflow states a clean exit criterion but later permits explicitly deferred dirty files, and its teardown hook checks only the wrap sentinel rather than Git state. The startup shell's tracked-change check already ignores untracked inbox files, but the general launcher dirty check does not distinguish inbox deliveries from other untracked residue.
+
+The three tracked files now dirty were written at 10:27, after the latest archived rulesets session wrapped at 09:24. That establishes a post-wrap write path: a strict wrap gate can guarantee the state at completion, but preventing later contamination also needs a realpath-aware cross-project write guard because globally installed rules and workflows are symlinks into this repository.
+
+The proposed design is one shared repository-state classifier with two policies. Strict wrap requires no staged, unstaged, or untracked entries and no dirty submodules. Inbox-safe sync permits an otherwise clean tracked/index state with untracked entries only below =inbox/=; those deliveries do not block pull or template sync. The strict check should run after the final push and again inside the ordered teardown hook, tied to the checked HEAD, while the inbox-safe policy should drive both startup and the =ai= launcher. Tests should cover every Git state, unusual path names, inbox-only sync, hook behavior, and the canonical/template mirrors.
+
+Two pending inbox handoffs both propose a shared Claude-to-Codex MCP registry mirror. They pass the value gate but overlap. The recommendation is to consolidate them into one =[#B]= specification with Codex-only entries preserved, atomic and redacted updates, dependency and health checks, both-machine verification, and the Claude-memory audit split into a separate task. No inbox disposition or project implementation has been applied.
+
+Completed the investigation plan without changing product code. The worktree proposal is ready for Craig's approval; implementation, tests, and inbox filing remain deliberately pending.
+
+** Clean-wrap invariant clarified
+
+Craig confirmed that cleanup failure must prevent wrap-up entirely. The agent must keep the session open and report exactly what remains in the Git worktree, why it could not resolve each item safely, and the specific action or decision Craig needs to supply. A warning, deferred-file exception, valediction, archived-as-complete status, or teardown is not an acceptable substitute for a clean tree.
+
+** Clean-wrap enforcement implemented
+
+Craig approved implementation and asked for a full wrap when it is done. Added =git-worktree-gate= as the single policy executable: strict mode rejects every staged, unstaged, untracked, dirty-submodule, and in-progress-operation state; sync-safe mode permits only untracked =inbox/= deliveries. Certificate and verify modes bind the final clean check to HEAD inside the Git directory.
+
+Wired sync-safe behavior into both startup workflow copies and the =ai= launcher. The picker labels inbox-only state distinctly and now fast-forwards a behind repository with inbox deliveries present while refusing other untracked residue.
+
+Made wrap cleanup fail closed: removed every dirty-file and inbox deferral escape, added the post-push clean certificate as a hard prerequisite to valediction, and required an exact path/state/needed-decision report when cleanup cannot finish. The existing teardown hook now freshly verifies the certificate and HEAD before consuming either sentinel, emits the runtime-appropriate Claude or Codex Stop blocker, and leaves the sentinel/session intact on failure. Added global Codex hook configuration and installed its symlink; Codex will require its normal hook review/trust on a new session.
+
+Added =rulesets-write-boundary.py= and configured Claude and Codex Edit/Write hooks. It resolves targets through symlinks and denies another project's write when the real path lands in rulesets, directing the proposal through =inbox-send rulesets=. The cross-project rule now states the installed-symlink case explicitly.
+
+Focused verification is green: 10 state-gate Bats cases, 13 teardown-hook cases including dirty/changed-HEAD/missing-certificate and both runtime outputs, 36 launcher cases including inbox-only pull, 5 installer cases, and 5 Python write-boundary cases.
+
+** Inbox — MCP registry proposals consolidated
+
+Processed the two pending 2026-07-25 handoffs from work and home. They were duplicate evidence for a host-level Claude-to-Codex MCP registry reconciler, not project-level implementation requests. Filed one =[#B]= parked specification decision in =todo.org= with preservation, redaction, atomicity, transport, health-check, two-machine, malformed-input, and token-rotation gates; split the memory auditor into separate future work. Deleted both inbound handoffs, stamped =:LAST_INBOX_PROCESS:=, sent acknowledgements to both source projects, and verified zero pending project handoffs.
+
+** Review, full verification, and wrap cleanup
+
+The first full =make test= run passed. Adversarial self-review then found the Git-status process-substitution fail-open and a path-resolution fail-open in the cross-project hook; fixed both, added status-error and sequencer regressions, aligned =protocols.org= with the new policies, and reran the complete suite successfully.
+
+Executed wrap cleanup: no sentry lock, todo hygiene/convert/archive/priority passes made no changes, lint-org applied zero mechanical changes and refreshed the 14-item judgment pipeline, project inbox remained at zero, route-batch had no candidates, and 30-day staleness was zero. The shared roam inbox held one rulesets capture about 120-column tables; verified the enforcement already exists in three layers, removed the duplicate, and pushed the roam update with the repository's sync helper.
diff --git a/.ai/sessions/2026-07-27-17-02-context-engineering-rightsizing.org b/.ai/sessions/2026-07-27-17-02-context-engineering-rightsizing.org
new file mode 100644
index 0000000..54020e4
--- /dev/null
+++ b/.ai/sessions/2026-07-27-17-02-context-engineering-rightsizing.org
@@ -0,0 +1,176 @@
+* Summary
+
+** Active Goal
+
+Review three Anthropic posts on context engineering against what this repo ships downstream, then act on the findings. Ended with the always-loaded rules surface cut from ~57,800 tokens to ~28,949 (plus 13,461 path-scoped), two mechanisms proven in a live session, and two of my own bugs found and fixed — one of which had killed Craig's work session.
+
+** Decisions
+
+- *Split rules by blast radius, not by size.* What must hold whether or not you're publishing, and where a violation is permanent and reaches other people, stays always-loaded. Everything recoverable can ride a trigger. That's what let =commits.md= and =testing.md= ship without waiting on any pilot.
+- *Goal is output quality first, tokens second* (Craig's correction). Anthropic's 80% was a finding, not a target. P4/Phase 8 (effort reduction) dropped outright for trading quality for cost; P5 (positive framing over prohibition) promoted as the lever that actually targets guardrails working against output.
+- *Don't apply the posts additively.* The harness system prompt already carries most of what the Opus 5 guide recommends adding, near-verbatim. Adding it to =claude-rules/= would worsen the duplicate-and-conflict problem the first post opens with. The posts' value here is subtractive.
+- *Everything authored in or about the repo is first person* (Craig's instruction), with one carve-out: a comment describing what the code does stays third person, since there the code is the actor.
+- *No unmerging home.* Its domains are already separated by tag (24 finances, 18 kit, 17 jrestate). The real problem is a tag namespace flattening four orthogonal axes, and cross-domain priority is a judgment no scheme can make — splitting projects hides the question rather than answering it.
+
+** Data Collected / Findings
+
+- *Path-scoping works at user level*, confirmed by =/context= in a live work session: 17 generic rules listed, the three path-scoped ones absent. Deterministic glob match, so no trial needed for that tier.
+- *User-level and project-level rules both load, project wins.* work and =.emacs.d= carried 19 byte-identical duplicates, and a stale project copy silently overrode the fresh global rule.
+- *My token estimates were 45% low.* Real ratio 2.28 tok/word. =commits.md= was 12,800 tokens, not the ~7,000 I claimed. =/context= reported the true per-file numbers the whole time and I used a word-count estimate because it was easier to compute from inside the repo.
+- *Two loading paths, not one.* Memory files arrive via the harness; =protocols.org= and the workflows are read by startup and land in Messages. They shrink by editing the workflow, not by scoping a rule.
+- *41 execution/hygiene workflows against 6 discovery/design.* The system is heavily built on the half of the problem that got easier.
+- *The instructions don't practice what they demand:* =commits.md= argued terseness at 5,561 words, =interaction.md= bans bold while the rules carry 591 bold markers, =testing.md= argues TDD across eight more rows of rationalizations.
+- *Two of my own mechanical guards failed the same day, both certifying success while doing damage.* =wrap-org-table.el= reflowed a table into a worse shape and =lint-org= then passed it; the wrap-teardown hook consumed a two-hour-old sentinel and killed Craig's live work session.
+
+** Files Modified
+
+Seven commits, all pushed, velox synced throughout. =6c1ea8b= peer-reasoning rule + Chrome convention + KB probe fix. =0adcb1a= =paths:= frontmatter on the three file-type rules + the lint checker that catches prose/frontmatter mismatch. =7ea1d7b= generic rules no longer ship per project, sweep + gitignored session anchor. =79ed3b0= the three rightsizing docs. =2c664cb= =hooks/session-start-disarm.sh= for the sentinel bug. =d74d98d= docs corrected against live measurements. =931f364= =commits.md= → invariant core + =publish= skill. =2f45b6e= =testing.md= → directive core + =testing-standards= skill, approval-gate signal fixed, first-person directive.
+
+** Next Steps
+
+Everything remaining needs Craig's decisions rather than execution — see the =[#B] Finish context-engineering rightsizing= task and the =[2026-07-27]= reminder. In order: reconcile the three working docs (one commit behind), then C1 (=verification.md='s honesty core vs the over-verification warning), =interaction.md=, the TDD rationalization table, and D3 (which approval gates are preference vs guardrail).
+
+Also open: the work sentry triage split and the recurring-loop proposal, both filed =[#B]= with their reviews. The sentry spec review is still waiting, now two weeks old.
+
+KB: promoted 2 / consulted no
+
+* Session Log
+
+** Startup — 2026-07-27 10:25 CDT
+
+Ran startup. Rulesets already current; =make install= had nothing new to link; project repo clean at f2609d9 with no upstream drift. =.ai/= synced from templates (no churn — the sync is a no-op mirror refresh in this repo). Previous session wrapped cleanly (no session-context anchor present).
+
+Startup signals: 6 top-level tasks unreviewed for >7 days; roam inbox empty; KB at 106 =:agent:= nodes but the best-practices node path resolved empty (=rg -l 'agent-kb-best-practices'= found nothing — worth checking whether that node exists); no spec-sort or host-identity flags; language-bundle sync silent.
+
+Five new inbox handoffs arrived since the last wrap. Read all five and ran the skeptical review on each before surfacing dispositions.
+
+Disposed of one without asking: home's 07-26 10:21 file was a pure FYI acknowledging that the parked MCP-registry spec decision and the separate memory-auditor track matched its handoff. It asked for nothing, so it needed no reply and no approval — deleted it. Four remain, all shared-asset or convention changes, all waiting on Craig's approval per the inbox engine's core §2.
+
+Skeptical-review conclusions on the four pending:
+
+- *archsetup publish-lock* — design sound, acceptance checks testable. Three gaps: TTL sizing has to survive an indefinite human approval wait (refresh-on-re-entry covers the agent, not a Craig who steps away); the blocked second session's behavior (wait / defer / stop) is unspecified; and it's a real build, not a quick edit. Recommend filing =[#B] :feature:=.
+- *home Chrome convention* — protocols.org already prefers google-chrome-stable over xdg-open, so the new parts are =--new-tab=, multi-URL, and the confirmation line. The confirmation half contradicts the existing =&>/dev/null &= form, which discards exactly the message to be verified. Recommend applying with a foreground-when-running / background-on-cold-start reconciliation.
+- *work sentry triage correction* — Craig's 07-27 correction supersedes his 07-21 ruling; today's work fire missed a Hayk DM and a Kostya PR-review request. The gap is that the current rule excludes by category (mail / messenger) and the new split is work-vs-personal, which category can't express. Shipping plugins: cmail, personal-gmail, personal-calendar, telegram, github-prs — no general work-mail plugin, so work's source is project-specific. A denylist of personal plugin names fails open on the next personal source added; a per-plugin eligibility declaration is the durable shape, and that's a design call. Recommend filing =[#B] :bug:= (Major × most-users-frequently = P2).
+- *work peer-reasoning rule* — approved exact text, well-formed. Two notes: it's a reasoning contract in a file scoped to communication style (the framing line should widen), and "process serves the outcome" sits one reading away from licensing deviation from the mandatory gates. Its own wording says surface-before-proceeding, so no edit needed, but that's the line to watch. Recommend installing as written.
+
+** Inbox pass applied — commit 6c1ea8b
+
+Craig approved all four dispositions plus the probe fix. Two corrections from him along the way: I had inverted the render-merge guard (numerals belong to the options list, dashes to every other enumeration in the same message — I did the reverse), and processed items shouldn't be left sitting in =inbox/=.
+
+Shipped: the peer-reasoning section at the top of =claude-rules/interaction.md= with the file's framing line widened; the Chrome convention rewritten in canonical =protocols.org=; the KB best-practices probe switched from a content grep to a filename =find=. Filed two =[#B]= tasks (sentry triage split, repository publish-lock), both stamped =:LAST_REVIEWED: 2026-07-27=. Swept the 40-file =PROCESSED-*= backlog out of =inbox/= along with the four handoffs; =inbox-status= now reports 0. Replies sent to archsetup, home, and work.
+
+*The review caught my own error.* I had written that Chrome's confirmation line prints to stderr and told every project to capture it with =2>&1=. It prints to *stdout*; stderr is empty. Verified both directions on ratio before correcting. An agent following the original text would have captured stderr, seen nothing, and concluded the tab failed to open — the exact silent-failure shape as the KB probe it shipped alongside. I asserted a stream rather than checking it, inside the same change that told others to verify. Side effect: four =about:blank= tabs opened in Craig's live browser during the check.
+
+Deliberate departure recorded: =route_recommend= returned =work strong= for the sentry task, but the work happens in rulesets' canonical =sentry.org=, so it's a local keeper and I left it unstamped rather than letting the wrap router offer to ship it away.
+
+Possible KB promotion at wrap: an identifier that lives only in a filename is invisible to a content search, and the lookup fails *silently* rather than erroring. Adjacent to the existing enumerate-vs-discover node but a distinct failure surface. Two instances in one commit (the KB probe, the Chrome stream) argue it generalizes.
+
+Startup-workflow bug found while checking the KB nudge: Phase A resolves the best-practices node with =rg -l 'agent-kb-best-practices' "$ra"=, which greps file *content*. The node's slug lives in its filename, so the probe returns empty and the contribute nudge points at nothing — in every project, every session. The node exists at =~/org/roam/agents/20260620232112-agent-kb-best-practices.org=. Synced-workflow change, so it waits on approval too.
+
+** Pushed and synced velox
+
+Pushed 6c1ea8b to origin/main (ahead-only, reconciled immediately before). On ratio, so velox needed the pull: it fast-forwarded and =make install= linked three things it had been missing since 2026-07-25 — the Codex =hooks.json=, =rulesets-write-boundary.py=, and =git-worktree-gate=. That drift is exactly the one-time-setup case =daily-drivers.md= names: the files traveled with the pull, but nothing re-runs the installer, so the symlinks only land where someone runs it. A new Codex session on velox will now hit the hook review/trust prompt, which was already on the 2026-07-25 next-steps list.
+
+** Context-engineering rightsizing — analysis and rollout plan
+
+Craig supplied three Anthropic posts (the 2026-07-24 Claude 5 context-engineering post, the Opus 5 prompting guide, the 2026-07-06 Fable field guide) and asked for a review, proposals, and a consistency audit of what this repo ships downstream. Then he reframed twice, and both reframes were better than the question I'd been answering.
+
+*First reframe:* consider the files as *his prompts*, not my context. That changed the finding. My first pass measured the always-loaded surface (32,123 words — =claude-rules/= 25,386 + =protocols.org= 6,620 + CLAUDE.md 117, roughly 40k tokens before the user's first word) and proposed shrinking it. Read as a map he hands every project, the finding is different: 41 execution/hygiene workflows against 6 discovery/design, seven to one. That ratio was right when the risk was the model doing things wrong. The field guide's claim is the bottleneck moved to the human's ability to clarify unknowns, so the system is heavily built on the half that got easier.
+
+*Second reframe:* metrics per claim, not one go/no-go. Turns the rollout into a set of separable testable claims rather than one bet.
+
+Three checkable "doesn't practice what it demands" findings: =commits.md= argues terseness at 5,561 words (longest file in the set); =interaction.md= bans bold in chat while the rules carry 591 bold markers; =testing.md= mandates TDD then argues eight more rows against rationalizations.
+
+*The finding that changed the plan:* the harness system prompt already carries most of what the Opus 5 guide recommends adding — its task-scope block, correction-narration block, and subagent cap are present nearly verbatim, and post 1's replacement comment guidance is present as the post's own new wording. So applying the posts additively would make the duplicate-and-conflict problem worse. The posts' value here is subtractive. It also exposes a third dedup axis nobody has audited: =claude-rules/= against the harness prompt, invisible from inside the repo.
+
+*Pilot selection rule* (the part that matters more than the list): the six pilot files were chosen because a silent miss is *detectable*, not because they're small. Four have a mechanical checker (=lint-org= =org-table-standard=, spec-board grep, =spec-review=), two produce an error Craig sees in seconds. =daily-drivers.md= and =emacs.md= were considered and held back — low risk, but a miss surfaces too slowly to learn from inside the trial window.
+
+Artifacts in =working/context-engineering-rightsizing/=: =proposals.org= (P1-P6, conflicts C1-C2, the from-your-side-of-the-desk section), =rollout.org= (Phases 0-8, decisions D1-D7, target trajectory), =metrics.org= (claim-by-claim testability, pilot go/no-go with the denominator rule, turn-back vs abandon triggers).
+
+Two honesty notes carried into the docs: I have a stake in arguing my own instructions should be shorter, so the plan weights mechanical detectors over my self-report; and about half the posts' claims aren't testable here without an eval harness, so those are labelled judgment rather than measurement so a future session doesn't mistake an adopted opinion for a tested result.
+
+Not started. Awaiting D1 (confirm pilot set) and D2 (skill index in the core).
+
+** Path-scoping shipped (0adcb1a) and work pre-synced
+
+The session's biggest finding: Claude Code scopes a rule by a =paths:= field in YAML frontmatter, and none of the 20 rules had one — even though three already declared a file-type scope in their =Applies to:= prose line. So =todo-format.md= (4,494), =org-tables.md= (464), and =emacs.md= (923) loaded into every session in every project, contradicting their own first line. 5,896 words. Fixed by adding the frontmatter, plus a =lint.sh= checker that warns when prose names a concrete extension without matching frontmatter (flags exactly those three, nothing else), plus teaching the heading check to skip a frontmatter block. Always-loaded rules surface: 25,386 → 19,505.
+
+Also confirmed from the docs: user-level and project-level rules *both* load, and project rules take priority. So work and =.emacs.d= carry 19 byte-identical duplicate copies, and a stale project copy overrides a fresh global one — which is exactly what was happening to work's =interaction.md= between this morning's commit and its next startup.
+
+Pre-synced work via =scripts/sync-language-bundle.sh ~/projects/work= (rulesets' own installer, run early rather than waiting for work's startup) so Craig's next work session is a valid test rather than one running the set it loaded before the sync. Verified: all four files now match canonical, frontmatter present, and work's =.claude/= is gitignored there so nothing was dirtied.
+
+Open question the next session answers: does =paths:= frontmatter apply to *user-level* rules or project-level only? The docs don't draw the distinction. =/context= in a fresh session settles it — if =todo-format.md= is absent from Memory files until an org file is opened, it works. If it's listed, the frontmatter is inert (no harm) and semantic skills are the only route.
+
+Not done: the double-load fix. Removing the 19 duplicates means changing what =install-lang= pushes into projects, and there may be a teammate-facing reason for them. Surfaced as Craig's call, not urgent — wasteful, not harmful.
+
+** De-duplicated the rules layer, unblocked sync (7ea1d7b, 79ed3b0)
+
+Craig confirmed no teammates depend on the per-project rule copies, so I removed them. =install-lang.sh= no longer copies the generic rules; =sync-language-bundle.sh= sweeps the ones earlier installs left, guarded on the global rule existing so a machine mid-bootstrap isn't stranded with none. Swept 20 files each from work and =.emacs.d=, leaving only their language rules plus work's =publishing.md= overlay. Three existing tests encoded the old contract and were rewritten; the generic-drift test now asserts sweep-not-repair, which is the stronger fix since the drifted copy outranked the global rule while it existed. Four new tests cover the sweep, the two keep-cases, and the no-global-rule guard.
+
+Also gitignored =.ai/session-context.org= and =.ai/session-context.d/=. This repo tracks =.ai/=, so the live anchor read as untracked all session and =git-worktree-gate= reported rulesets sync-blocked — meaning every other project skipped its rulesets pull until wrap, every session. Craig spotted the blocked state and inferred it was why I pre-synced work; it wasn't (rules load at launch, before the startup sync runs, which was the real reason), but chasing his inference found the anchor problem, which was the better bug.
+
+Corrections from Craig this stretch: the goal is output quality first, token reduction second — my docs led with the wrong number and P4 (effort reduction) should be demoted or dropped since it trades quality for cost. And all authored prose goes first person; I amended the first commit rather than leaving it. Code comments stay third-person by agreement, since they describe what the code does for the next reader.
+
+Docs not yet updated for either the goal reordering or the last two hours of findings (path-scoping, the double-load, the harness overlap). That's the next task.
+
+** Inbox: archsetup ack
+
+archsetup acknowledged the publish-lock acceptance and the three implementation gaps, confirming the decision stays closed on its side. Pure FYI, nothing asked, no reply owed. Deleted it. Inbox back to zero.
+
+** Killed Craig's work session with my own hook, then fixed it (2c664cb)
+
+Craig's 13:20 work session was blocked repeatedly and then had its terminal closed under it. The cause was mine, from Saturday's clean-wrap work.
+
+=wrap-it-up= drops =/tmp/ai-wrap-teardown-<project>= so the =Stop= hook tears down once the wrap certifies clean. I deliberately made a failed certification *preserve* the sentinel, so a wrap blocked by a dirty tree could retry on a later stop. I never bounded that retry to the session. work's 11:37 wrap left an uncertified sentinel armed; the 13:20 session's stops were all blocked by it failing certification; then startup's two commits (task filing, template sync) made the tree clean, the next stop certified, and =cj/ai-term-quit= killed the tmux session mid-work.
+
+Two others were armed and dangerous at the same moment: archsetup's since Saturday 15:02 on a live attached terminal, and home's from 13:21 on a live session. Disarmed all three by hand (backed up to =/tmp/disarmed-sentinels=) before writing any fix, since both were minutes from the same fate.
+
+Fix: =hooks/session-start-disarm.sh= clears the project's sentinels at =SessionStart= — a new session means the wrap that armed one is gone. Within-session retry is untouched (the hook only runs at session start) and a test pins that so the deliberate behavior isn't lost to the fix. Four tests on the disarm including project-scoping, one on the retry. Wired into =.claude/settings.json=, installed on both machines, =wrap-it-up.org= documents the session-scoping with the worked failure.
+
+Diagnostic note worth keeping: I found it by reading work's own crashed session anchor, which showed startup completing normally and then stopping dead, plus its git log showing two commits at 13:21 — the exact moment the tree went clean. The anchor being left behind by the interrupted session is what made the timeline reconstructable. That's the crash-recovery purpose earning itself.
+
+** /context settled both open questions; docs corrected (d74d98d)
+
+Craig ran =/context= in work. Memory files lists 17 generic rules; =todo-format.md=, =org-tables.md=, and =emacs.md= are absent, and only =python-testing.md= and =publishing.md= come from the project's own rules dir. So *path-scoping works at user level* and *the de-duplication holds*. Both were open.
+
+Three corrections the live numbers forced:
+
+1. *My token figures were low by ~45%.* Real ratio is 2.28 tok/word, not the ~1.3 I assumed. =commits.md= is 12,800 tokens (I said ~7,000); =claude-rules/= was ~57,800/session before today, now 44,410, with 13,390 path-scoped out. Worth naming the actual error: =/context= reports per-file token counts and I used a word-count estimate instead because it was easier to compute from inside the repo. The instrument existed the whole time.
+2. *Two loading paths, not one.* Memory files arrive via the harness at session start. =protocols.org= and the workflows are *read by startup*, so they land in Messages and never appear under Memory files. They shrink by editing the workflow, not by scoping a rule. My "always-loaded surface" number conflated them.
+3. *The harness's own suggestion* names =commits.md=, =testing.md=, =MEMORY.md= as the top three to prune — independently the same Phase 4 list I'd proposed.
+
+Because a glob match is deterministic, the remaining work splits: path-scopable rules ship with no trial (=docs-lifecycle.md= on =docs/**= is next), and only semantic-condition rules need the skills route and the stop conditions. =commits.md= is the real test there — largest single item, and almost all publish machinery that only applies when a commit is in play.
+
+Recorded a caution the confirmation doesn't cover: path-scoping fires on a *read* of a matching file, so creating a new org file from scratch never triggers =todo-format.md=. Edits are safe (Edit requires a prior read).
+
+Also folded in Craig's goal correction (quality first, tokens second): P4/Phase 8 dropped outright since lowering effort trades quality for cost, P5 promoted since positive-framing-over-prohibition is what targets guardrails working against output.
+
+** Split commits.md: 12,800 tokens → 2,342 always-loaded (931f364)
+
+Craig picked the commits.md split over docs-lifecycle after I checked the latter and found I'd overstated it — =docs-lifecycle.md= scopes to "any project carrying a docs/ tree," a *project-level* condition a glob can't express, and 2 of its 6 trigger points are creation cases a read-triggered path rule misses. Only three rules ever named a concrete extension and all three are already converted, so there is no other clean path-scope candidate.
+
+The split line is *blast radius*, not size. Stayed always-loaded (1,027 words / ~2,342 tokens): author identity, the no-AI-attribution ban, the generated-document byline rule, the public-artifact content-scope rules, and "If You Catch Yourself." Moved to =publish/SKILL.md= (4,871 words): message format, Voice and Focus, PR description structure, Review and Publish Steps 0-2, the three review shapes, hook authorization, merge strategy, the pre-commit checklist.
+
+Why that line: if the skill fails to trigger I don't know the flow and have to be told — visible and recoverable. I don't silently commit with AI attribution, because that guard never moved. Only the recoverable half rides the skill-triggering bet, which is what let this ship without waiting on the pilot.
+
+*Verified by using it.* The skill registered mid-session and I invoked =/publish= to publish its own commit; it loaded with the full flow present. Content conserved and checked rather than assumed: 5,561 words in, 5,898 across both files (delta = frontmatter + the pointer added to the core). Repointed five cross-references in =voice=, =review-code=, =inbox.org=, and =no-approvals.org= that named moved sections.
+
+Always-loaded rules surface: 44,410 → ~33,950 tokens. Started the day at ~57,800.
+
+Noted and deliberately not done: =publish/SKILL.md= is a single 4,871-word blob, and both posts argue a long skill should use progressive disclosure internally. It loads on demand now, which is the win worth taking; splitting it further is its own change.
+
+Also surfaced: the Step 2 =.ai=-tracking heuristic misfires here. It reads tracked =.ai/= as "shared team repo → skip the approval gate," but rulesets tracks =.ai/= as a committed mirror while being a private single-user repo. I kept asking rather than skipping, and flagged it to Craig.
+
+** testing.md split, gate fixed, first-person directive added (2f45b6e)
+
+Three changes. *testing.md split* the same way as commits.md — by what has to be resident, not by size. Core keeps TDD-is-default and the three-category requirement (347 words), because those fire *before* any code is written, which is exactly when no skill has been summoned. Everything else → =testing-standards= skill (2,903 words): characterization recipes, per-category detail, property/mutation testing, pyramid, integration rules, naming, test-quality and mocking rules, coverage targets, spike exception, anti-patterns.
+
+*Approval-gate fix.* The publish flow decided whether to ask by checking whether =.ai/= is tracked, as a proxy for "team repo." Wrong in the direction that matters: rulesets, home, and work all track =.ai/= and all three are private single-user repos, so the rule skipped the gate on Craig's three most-used projects. Now checks whether any remote is on a host other than cjennings.net. Verified both directions including a synthetic GitHub remote. Every current project → gate applies, which matches how the flow has actually been run all session.
+
+*First-person directive* added to the always-loaded core, at Craig's instruction. One existed for commit bodies/PR prose but it moved into the publish skill, and it never covered code comments at all. Now: everything authored in or about the repo is first person, with one carve-out — a comment describing what the code *does* stays third person, since there the code is the actor.
+
+Also split =publish/SKILL.md= internally: PR descriptions + the three review shapes → =references/pull-requests.md=, since a plain commit never needs them. SKILL.md 5,012 → 3,888 words.
+
+*Surface: ~57,800 tokens this morning → ~28,949 always-loaded now* (plus 13,461 path-scoped). Largest remaining: =interaction.md= 3,828, =verification.md= 3,388, =commits.md= core 2,804, =subagents.md= 2,373, =cross-project.md= 2,305.
+
+Risk recorded rather than buried: testing.md's margin is thinner than commits.md's. If =testing-standards= fails to trigger mid-test-writing I lose the mocking-boundary rules — a quality regression, visible in review, but a real bet where commits.md's moved half was purely procedural. Also moved the TDD rationalization table rather than cutting it; the posts say that kind of over-argument is counterproductive now, but deleting Craig's defense against me skipping TDD is his call.
diff --git a/.ai/sessions/2026-07-29-06-36-adversarial-review-flow-and-telegram-fixes.org b/.ai/sessions/2026-07-29-06-36-adversarial-review-flow-and-telegram-fixes.org
new file mode 100644
index 0000000..12cbe3d
--- /dev/null
+++ b/.ai/sessions/2026-07-29-06-36-adversarial-review-flow-and-telegram-fixes.org
@@ -0,0 +1,514 @@
+#+TITLE: Session Context — 2026-07-28
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-28
+
+* Summary
+
+** Active Goal
+
+Started as inbox triage on a telegram-plugin bug report and became two things: shipping the cross-project fixes that arrived overnight, then designing and dogfooding a mandatory isolated adversarial review before every commit — which immediately found real defects in its own design and in everything filed afterward.
+
+** Decisions
+
+- *Merge colliding fixes rather than sequence them.* The parked down-is-launch diff still carried the bad =loadChats= call and cited the segfault gotcha the other fix rewrites. Applying either alone would have shipped a file arguing against itself.
+- *"Adversarial", not "hostile"* (Craig). An agent told to attack manufactures findings, so the stance carries a substantiation floor: a finding not substantiated against the diff is dropped.
+- *Re-review until the reviewer approves* (Craig's addition, the thing I had missed). Same reviewer continued, not a fresh one — a fresh reviewer can't tell an addressed finding from one that never existed. Bounded at three rounds or first recurrence.
+- *Dispatch on every commit*, with the reviewer's own Phase 0 ruling triviality. A floor written as "small" or "mechanical" puts the judgment back with the author, whose judgment is the thing being checked.
+- *Pass the requirement source, withhold the rationale.* A ticket is not the author's model; it was written first and by someone else, so it's the only input that can contradict the author's claim.
+- *=cycle=, not =pass=, for one sentry loop* (Craig). home proposed =pass=; =sentry.org= already uses it as a numbered noun for the eleven hygiene passes, so =Pass 11= would have collided with =pass 12=.
+- *Drop the =references/= link rather than sync the directory.* The four calendar workflows already travel and already carry the recipes.
+- *Strip the wrap-org-table task to Verified / Open questions* after its review loop bounded out. The measurements were never what failed.
+
+** Data Collected / Findings
+
+- *The telegram bug.* =(telega--loadChats 'main)= sends a bare symbol on the wire; =tdat_plist_value= (=telega-dat.c=) accepts only =(=, =[=, ="=, =-=, digit, =t=, =:=, =n= and calls =assert(false)= on =m=. Verified against telega's source at four points rather than trusting the handoff. Exposure was manual triage only — sentry excludes messengers, so home's eleven overnight cycles never loaded the plugin.
+- *=wrap-org-table.el= splits logical rows*, and =lint-org= doesn't merely miss it — it *causes* it. =lint-org.el:424= calls the same broken predicate, so it reports the tool's own correct output as "missing rule between rows — wrap-org-table.el reflows it" when nothing is missing, then reports the corrupted result clean. Idempotence is broken: the tool corrupts its own output on a second run.
+- *The isolated reviewer earned its keep on its first four uses*, finding: that withholding the ticket made my claim self-certifying; that my =subagents.md= override reaffirmed the Prompt Contract field that would destroy the isolation; a fourth verdict (=Needs Discussion=) I'd asserted didn't exist; and four successive wrong root-cause analyses on the table bug.
+- *rulesets is itself exposed* to the table bug: =todo.org='s four-row attachment-sanitization table. Don't reflow until fixed.
+- *Seven =../../= link sites* across four synced workflows resolve only in rulesets. =scripts/lint.sh='s =check_md_links= was built for that class and misses them because it matches markdown syntax only.
+
+** Files Modified
+
+Seven commits, all pushed, velox synced after each. =bff0138= merged telegram fixes. =43a4cf7= post-load liveness check. =614e3b1= removed finished working dirs. =ca508a1= filed three handoff findings. =f3f5bfd= the =fire= → =cycle= rename (72 sites plus three that had leaked outside =sentry.org=). =3a933a2= the =references/= analysis. =5999f88= dropped the dead link and deleted the stale file behind it. =ecd5d7b= filed the wrap-org-table bug.
+
+Rules changed: =publish/SKILL.md= Step 1 rewritten (dispatch contract, four defined verdicts, the loop, bounds), =review-code/SKILL.md= (two levels of dispatch, adversarial contract, re-review mode), =claude-rules/subagents.md= (Isolation Override; Prompt Contract field 2 inverts), and the three unattended callers taught to park.
+
+** Next Steps
+
+- *Three of the four items Craig queued are untouched*: the winvm =[#C]= lint defects, the context-engineering rightsizing (needs his four decisions), and the sentry spec deep read (two weeks old).
+- *Rule gap found by using the rule*: =Needs Discussion= exits to the user, but nothing says what happens after the user answers — whether the round counter resets. I treated it as a fresh review; that judgment isn't written down.
+- *Item 2 is now qualified*: its fix says "run =wrap-org-table.el=", and that tool has a live corruption bug. The specific table is safe, but verify the output rather than trust it.
+- Four =[#B]= bugs filed tonight and unstarted: the table splitter, the =../../= links, plus the two carried in.
+
+KB: promoted 1 / consulted no
+
+* Session Log
+
+** 11:55 — Startup
+
+Ran startup. Rulesets already current, project repo clean and current, =make
+install= had nothing new to link, =.ai/= synced from templates. No crash anchor
+— previous session (context-engineering rightsizing, 2026-07-27 17:02) wrapped
+cleanly.
+
+Findings: 6 tasks unreviewed >7 days; roam inbox holds 4 items; KB at 108
+=:agent:= nodes with nothing matching this project. Spec-sort and host-identity
+probes silent. Language-bundle check silent.
+
+** 12:05 — Inbox: the telegram segfault root cause
+
+Four new inbox files from =.emacs.d=, two pairs: a 06:15 intro note + plugin
+file, then a 07:21 correction + superseding plugin file. The correction retracts
+one secondary claim from the 06:15 write-up (that the "19 of ~50 chats" reading
+was truncation caused by the bug — it wasn't; 19 is the real account size,
+measured by work at the wire level). Root cause and fix unchanged.
+
+The proposal: =triage-intake.telegram.org= Step 1 calls =(telega--loadChats
+'main)=, and that bare symbol kills =telega-server= outright.
+
+I verified the whole chain against telega's own source rather than taking the
+handoff's word for it (=elpa/telega-20260706.2147/=):
+
+- =telega--loadChats= (telega-tdlib.el:2190) drops its argument straight into
+ the request as =:chat_list= with no conversion. Confirmed.
+- The C parser =tdat_plist_value= (server/telega-dat.c:466) accepts only =(=,
+ =[=, ="=, =-=, a digit, =t=, =:=, or =n= to start a value; anything else
+ prints "Unexpected char '%c' in plist value" and calls =assert(false)=.
+ =main= starts with =m=. Confirmed, and the accepted-char list in the handoff
+ is exactly right.
+- telega's own callers all pass the object: telega.el:290, telega.el:295,
+ telega-tdlib-events.el:516. Confirmed.
+- The symbol shorthand lives in a different layer — telega-filter.el:472 and
+ telega-folders.el:99 convert =(eq cl-fspec 'main)= to ='(:@type
+ "chatListMain")=. Confirmed.
+
+Every claim holds. Grepped all workflows for other bare-symbol arguments to
+=telega--*= wrappers: only the two sites this diff already fixes.
+
+*The complication I found.* There is already a parked =[#B]= VERIFY (todo.org:136,
+from =.emacs.d= 2026-07-24) proposing a different fix to this same file — the
+down-is-launch fix, with a prepared diff in
+=working/triage-telegram-down-launch/=. Its proposed file still carries the bad
+=loadChats= call at its lines 52 and 122, and its third hunk adds prose citing
+"tdlib segfaults in native mode (SEGFAULT gotcha below)" — pointing at the very
+section this new handoff rewrites to say the crashes were our bug, not tdlib's.
+Applying either one alone leaves the file half-right. They need merging, not
+sequencing.
+
+Also: home and work both declare =telegram= in =:TRIAGE_SOURCES:=, so both have
+been killing telega-server on every sweep. Work already patched locally and
+verified at the wire level. Home has not been told.
+
+** 11:58 — Staged the batch, closed the loops
+
+The Stop hook counted the four files still sitting in =inbox/=. The triage was
+done; what remained was Craig's merge decision, not mine. So I staged all four
+into =working/triage-telegram-segfault-rootcause/= (the 07:21 pair as
+=.proposed= + correction note, the 06:15 pair kept as superseded context — its
+blast-radius and follow-up sections aren't repeated in the correction). Inbox
+back to clean, =inbox-status= rc=0.
+
+Left =working/triage-telegram-down-launch/= untouched: the parked VERIFY's body
+links its =proposed.diff= by path, and moving it would break the link before
+Craig has decided anything. If he takes the merge, that consolidates then.
+
+Replied to =.emacs.d= — confirmed the verification, and told them the one thing
+they couldn't see: their own 2026-07-24 parked diff collides with this, both in
+the stale =loadChats= call it still carries and in the native-mode prose it adds
+citing the gotcha this rewrites. Also declined their suggested bare-symbol lint:
+one consumer file, two call sites both fixed, so the stated rule is cheaper than
+a checker with a single input.
+
+Notified home directly with the mechanism and the two-site fix, told it to patch
+locally rather than wait on the canonical, and carried the correction forward
+explicitly so it doesn't inherit the retracted truncation story.
+
+Nothing applied to the canonical. That still waits on Craig.
+
+** 12:10 — Craig picked the merge. Applied and committed.
+
+Merged both fixes into one version rather than sequencing them. Base was the
+segfault-corrected file, then the three parked hunks on top: the down-is-launch
+directive, the SCAN-FAILED-only-after-launch-attempted rewording, and =(setq
+telega-use-docker t)= restored to the Step 1 code block.
+
+The reconcile that made merging worth doing. The parked hunk's new comment said
+"tdlib segfaults in native mode (SEGFAULT gotcha below)", pointing at the
+section the other fix rewrites to say those deaths were our own bad argument.
+Left alone the file would have argued against itself. I changed the Step 1
+comment to state plainly that the two are separate concerns (the deaths happened
+*in* docker mode, so docker mode is neither a defense against the loadChats bug
+nor evidence for itself), and reworded the Quick Reference line from "tdlib
+segfaults outside docker mode" to "crashed in native mode (2026-06-09)" with the
+same disambiguation.
+
+That reword also removed a host-identity violation I hadn't gone looking for.
+The original asserted "Craig's daemon currently has telega-use-docker nil" — a
+mutable machine fact stated as fixed in a synced doc. I checked the actual
+default (=telega-customize.el:514=, =defcustom telega-use-docker nil=) and wrote
+the durable claim instead.
+
+Verified: both live call sites use the TL object, the two remaining ='main=
+occurrences are inside the gotcha prose describing the bug, lint-org clean on the
+changed file, mirror synced, =make test= green before (exit 0) and after (exit
+0).
+
+todo.org: closed the parked =**= VERIFY as =DONE= + =CLOSED:= per todo-format.md
+with the merge rationale in the body. Promoted its =***= engine child (SCAN
+FAILED must not advance the sentinel) to top-level =**= VERIFY so it doesn't get
+buried under a DONE parent. Kept its =:LAST_REVIEWED: 2026-07-24= rather than
+stamping today — I moved it and judged it separate, but nobody re-derived its
+content, so the older date keeps it honest.
+
+Review: Approve, no Critical or Important. Two Minor, both surfaced rather than
+fixed. The gotcha now advises a post-load =(process-live-p ...)= check that the
+Step 1 recipe doesn't actually do, and adding it would extend the recipe past the
+two fixes Craig approved.
+
+Committed =bff0138=. Not pushed — that's a separate confirmation.
+
+** 12:20 — home replied, and corrected my urgency read
+
+home accepted, patched both call sites locally, and re-verified the diagnosis
+independently rather than trusting it. Useful correction back: home declares
+telegram in =:TRIAGE_SOURCES:=, but sentry excludes messengers from triage
+intake, so the plugin never loaded on a sentry fire. Eleven overnight fires ran
+clean against the broken file. I had assumed the sweeps were affected; the real
+exposure is manual triage intake only.
+
+Told home its stopgap won't be reverted into a broken state — the next rsync
+replaces it with canonical content carrying the same fix. On this machine that
+lands as soon as its next startup runs, since the rsync reads the local rulesets
+working tree. velox needs the push.
+
+Inbox back to zero.
+
+** 14:10 — Pushed, and closed the velox gap
+
+Pushed =1675613..bff0138= to origin after the pre-push reconcile (still 1 ahead,
+0 behind).
+
+=.emacs.d= replied while the push was in flight. It had already verified bff0138
+here, re-run its own workflows rsync, and confirmed the corrected form landed.
+It withdrew the bare-symbol lint suggestion, conceded the home omission, and
+flagged one real gap: bff0138 was committed but not pushed, so velox stayed
+exposed. That was true when written and stale by the time it arrived.
+
+Checked velox over tailscale (this host is ratio, per =uname -n=). It was 1
+behind / 0 ahead and sync-safe, so I fast-forwarded it through the same
+=git-worktree-gate sync-safe= check startup uses. Velox is now at bff0138 and its
+workflow file carries the corrected call, with the only bare ='main= occurrences
+inside the gotcha prose. Both daily drivers covered.
+
+Corrected read carried into both replies: the exposure was manual triage intake
+only, not the automated sweeps, because sentry excludes messengers.
+
+** Open follow-ups (surfaced to Craig, not acted on)
+
+1. The gotcha tells callers to check =(process-live-p (telega-server--proc))=
+ after a load, but the Step 1 recipe doesn't do it. Now that the corrected call
+ shouldn't kill the server, that check is what would catch a regression. Left
+ out deliberately as scope creep past the two approved fixes.
+2. Both =working/triage-telegram-*= dirs are completed-task artifacts and want
+ filing per working-files.md. Revised read after checking: delete both
+ outright. Every file is tracked (=b19d420= and =bff0138=), so git holds them
+ permanently and a copy in =assets/= would only duplicate history. Nothing
+ links to them.
+
+** 15:00 — Second inbox round: .emacs.d self-correction + winvm lint findings
+
+=.emacs.d= wrote back to say it had overcorrected on home: it accepted "home was
+in the blast radius" and then recorded that home "had been killing telega-server
+on every sweep too", which home's own sentry data refutes. It fixed its task
+record rather than leaving it. I told it the pattern wasn't one-sided — I made
+the same move this morning, estimating home's blast radius instead of measuring
+it, and home's data is what corrected me. It also offered the emacs-side half of
+a completed-vs-truncated signal, which it has filed as =[#C]=, if the
+=process-live-p= recipe change lands.
+
+=winvm= sent a link-integrity pass with three findings, all reproduced against
+the rulesets source rather than only its local copy. I verified all three:
+
+1. =protocols.org:273= links =references/calendar-reference.org=, but the rsync
+ set is only =protocols.org=, =workflows/=, =scripts/=. Dead link in every
+ consuming project. home and =.emacs.d= have no =.ai/references/= at all.
+2. =retrospectives/PRINCIPLES.org:38= violates the org-table standard.
+ =lint-org= confirms, checker =org-table-standard=.
+3. =protocols.org= lints 8 mechanical + 19 judgment =misplaced-heading=, all
+ from Markdown =**bold**= in an org file. 48 bold spans, 14 line-initial.
+
+The check winvm couldn't run: adding =references/= to the sync brings =--delete=
+with it, which would destroy project-owned files there. work is the only project
+with a =.ai/references/= and its copy is byte-identical to the source, so nothing
+would be lost. That clears the path to sync-it over drop-the-link.
+
+One refinement back to winvm on finding 3: the 8 mechanical hits are would-fix,
+and the 14 line-initial bold spans are exactly what =lint-org --fix= would
+rewrite without knowing they're emphasis rather than headings. So that half is
+not safe to run mechanically even though the count suggests it is.
+
+Filed as =[#B] :bug:= (the dead link, Minor x every-project-every-sync = P2) and
+=[#C] :chore:= (the two cosmetic defects, P3), both stamped =:LAST_REVIEWED:=
+today since I graded them today. Verified todo.org's lint counts are unchanged
+from the pre-edit baseline (mechanical=2, judgment=12, both pre-existing and
+already in =inbox/lint-followups.org=).
+
+Inbox back to zero. todo.org is left uncommitted deliberately — Craig has a
+pending decision that will likely touch it again, so batching beats a churn
+commit.
+
+** 18:45 — Craig took both follow-ups. Liveness check staged.
+
+Added the liveness check to the Step 1 load call. It now returns ='server-died=
+instead of an unconditional ='loaded=, and new prose routes that into the
+existing SCAN FAILED definition rather than inventing a category: a server that
+dies during load leaves a thin hash, which reads exactly like a quiet account,
+which is the same false all-clear the down/not-loaded rule already prevents.
+
+Self-review caught a real defect in my own first draft. I called
+=(telega-server-live-p)= bare, but Step 0 guards the same predicate with
+=fboundp= because telega may not be loaded. A launch that failed outright would
+have signalled void-function instead of returning the clean contract. Added the
+guard, matching Step 0's idiom. Verified the predicate is exactly the
+=process-live-p= expression the gotcha names (=telega-server.el:221=).
+
+Verified: parens balance at depth 0, lint-org 0/0, mirror identical, =make test=
+green (exit 0) on the final state.
+
+Not committed — waiting on the approval gate.
+
+** 18:48 — Third inbox round
+
+=.emacs.d= sent a closing FYI marked no-action, agreeing my framing of the shared
+failure (blast radius estimated rather than measured) named the trigger rather
+than the failure. Deleted without reply, since replying to "nothing owed back"
+is noise.
+
+home proposed renaming sentry.org's noun-sense "fire" to "pass", after Craig read
+its "nine fires" as nine emergencies: "I assume you mean nine crises, not nine
+loop cycles and I begin to get scared." The problem is real, well-evidenced, and
+reaches Craig directly through digest headings.
+
+But the proposed term is wrong, and the reason home gave for it is the
+disqualifier. =sentry.org= already uses "pass" as a precise numbered noun — the
+pass list, the Pass Runner, "eleven finding/hygiene passes", "pass 12". With
+exactly eleven hygiene passes, home's proposed =** Pass 11= heading collides with
+an existing referent. That trades a term Craig misreads as urgent for one that is
+genuinely ambiguous.
+
+Counter-proposed *cycle*: zero occurrences in the file, and Craig's own word in
+the quote home cited. Checked and rejected "sweep" (3 uses) and "run" (used as a
+noun). Filed =[#C] :chore:= with the grading and the collision analysis; replied
+to home with the counter-proposal.
+
+** 18:50 — Three commits, and the collision confirmed from live evidence
+
+Craig approved both follow-ups and the =cycle= term. Three commits:
+
+- =43a4cf7= the liveness check.
+- =614e3b1= removed both telegram working dirs. Filing by deletion, since every
+ file was already in git via =b19d420= and =bff0138= and an =assets/= copy would
+ only duplicate history.
+- =ca508a1= filed the three handoff findings with their gradings.
+
+home wrote back confirming the "pass" collision was real, and that it had already
+walked into it: its anchor now carries =** Pass 11= meaning the eleventh cycle,
+three lines from =pass 12 (solo-task implementation)= meaning the twelfth item in
+the pass list. Same file, two referents, introduced by its own normalization an
+hour earlier. It found that in live evidence faster than reading the file would
+have caught it.
+
+It was blocked on Craig's confirmation and had written a memory saying "pass", so
+I sent the confirmation immediately. The memory was the urgent half — a stale one
+teaches every future home session the ambiguity, where the anchor is one file.
+
+Recording Craig's approval flipped the sentry task to =:solo:=. The term was the
+only judgment it carried, and the completion check is objective, so it can ride a
+backlog run rather than waiting for someone to touch =sentry.org=.
+
+Pushed =bff0138..ca508a1=, velox fast-forwarded to match.
+
+** 19:45 — The cycle rename, and the leak home's scope missed
+
+home did its side first: renormalized its anchor by restoring the
+pre-normalization backup and re-running fire→cycle from clean, rather than
+reverse-mapping pass→cycle. That was the right call — reverse-mapping would have
+needed a judgment on every instance to separate its own conversions from genuine
+pass-list references, where re-running from clean makes it structural. It also
+corrected its memory and handed the canonical back.
+
+Did the canonical rename with a script rather than by eye: protect the verb sites
+by explicit pattern, assert zero unclassified =-ed/-ing= forms survive, then
+substitute. 72 noun instances converted, 4 verb sites untouched (=/loop= fires
+again, two "fires on approval", "record of what fired").
+
+*The leak home's scope missed.* The term wasn't confined to =sentry.org=.
+=wrap-it-up.org= said "a crashed fire", and =todo-cleanup.el= and its test both
+said "every sentry fire" — all three naming a sentry cycle. Renaming only
+=sentry.org= would have split the vocabulary across files. Found by grepping
+every file that mentions sentry, then re-grepping without a context window after
+the first pass truncated short-line matches and hid them.
+
+Verified: exactly 3 verb instances left in sentry.org, no placeholder leaked, the
+digest commit template now reads =<date> <time> cycle=, capitalized plurals
+handled, lint 0/0 on sentry.org, suite green (exit 0 — load-bearing here, since
+=todo-cleanup.el= and its test are under test). The two =wrap-it-up.org= lint
+findings are pre-existing and identical at HEAD.
+
+Committed =f3f5bfd=, pushed, velox fast-forwarded and verified. Notified home
+(with the leak it hadn't seen) and =.emacs.d=. Closed the task =DONE=.
+
+Four commits this session, all pushed, both daily drivers current, inbox at zero.
+
+** 20:00 — Roam inbox zero, then the adversarial-review design
+
+Roam scan: 3 items, 1 claimed (=rulesets:= prefix), 2 unowned gear links left
+for Craig. Filed the claimed one, removed it from roam under capture-guard +
+roam-write lock, triggered =roam-sync=. A local =.emacs.d= FYI also cleared.
+
+The claimed item: "code reviews must occur before every commit an agent does,
+and they should be hostile reviews from a subagent without the agent's context."
+
+Craig's decisions: *adversarial* rather than hostile (he took my push-back that
+an agent told to attack manufactures findings), plus a requirement I had missed —
+a re-review loop that runs until the reviewer approves — and that the rules must
+not contradict afterward. Unopposed recommendations I proceeded on: dispatch on
+every commit with the reviewer's own gate deciding triviality, and the flow lands
+in the =publish= skill.
+
+Wrote it into three files: =publish/SKILL.md= Step 1 (dispatch contract,
+adversarial-with-substantiation, the loop, bounds), =review-code/SKILL.md= (the
+two levels of dispatch, the adversarial contract, re-review mode), and
+=claude-rules/subagents.md= (a new Isolation Override section, since three
+separate size rules there said don't dispatch small work).
+
+** 20:30 — Dogfooded it, and the reviewer found nine things
+
+Ran the new flow on its own diff: dispatched an isolated adversarial reviewer
+with the diff plus a one-line claim, withholding everything else. It returned
+REQUEST CHANGES with seven Important and two Minor, every one substantiated. I
+checked each against the files rather than accepting them, and all nine were
+real:
+
+1. =Skipped= from Phase 0 is not =Approve=, so trivial diffs dead-ended at a gate
+ with no defined pass.
+2. =no-approvals.org= line 73 (the actual execution step) still described the old
+ inline unbounded flow; I had only updated the preamble at line 49.
+3. =work-the-backlog.org= and =sentry.org= — the unattended callers — had no
+ receiver for "stop and surface to the user". The speedrun routes to
+ work-the-backlog, not the file I updated.
+4. The Step 2 exception still said the review runs "when it applies", which my
+ rewrite had made false.
+5. *The sharpest one.* Withholding the ticket/plan makes the author's claim
+ self-certifying and strands =review-code='s Intent-vs-Delivery criterion — the
+ one aimed at exactly the inherited-scope error this gate exists to catch. A
+ ticket is not the author's model; it is the independent record of what was
+ asked. I had not considered this.
+6. The loop turned on the verdict token, so a single Minor could burn all three
+ rounds and escalate.
+7. My override said it "doesn't relax the Prompt Contract" while field 2 of that
+ contract says paste your context verbatim — which would destroy the isolation
+ the whole change is built on.
+8. todo.org carried an unrelated =references/= rewrite, and the new task body
+ listed as open the questions the same commit answered.
+9. =subagents.md= still says subagent output is a claim to verify, unreconciled
+ with "approval is the reviewer's to give".
+
+All nine fixed. The =references/= hunk is split out as =3a933a2=. Round 2 sent
+back to the same reviewer, which is the loop working as designed.
+
+** 21:30 — The loop closed at three rounds
+
+Round 2 (three findings): =review-code='s adversarial contract still said two
+inputs, so the round-1 fix landed in =publish= but not in the text the reviewer
+reads — the two files disagreed at the one seam the change was about. A
+count/list regression my own fix introduced ("exactly two things" over a
+three-item list). And a fourth verdict I had missed entirely: =review-code=
+emits =Needs Discussion=, which is exactly what an adversarial reviewer reaches
+for on an architectural objection, and the flow had no handling for it.
+
+Fixing =Needs Discussion= reproduced round 1's finding 3 one level down, which I
+caught myself: all four unattended parking clauses keyed on "the loop hits its
+bound", and a first-round =Needs Discussion= is not a bound hit. Rewrote all four
+to trigger on "the review can't reach approval" with the three causes named, and
+asked the reviewer to verify that independently rather than take it from me.
+
+Round 3: APPROVE, with one Minor — a recurrence of the count/list mismatch, since
+my fourth-verdict fix left the lead-in saying "three outcomes" above four
+bullets. Fixed the numeral. The committed diff therefore differs from the
+approved one by exactly that word, which the flow's own Minor-only rule permits
+rather than spending a fourth round.
+
+The reviewer also verified things I had not asked about and would not have
+checked: that =failed= is already a legal outcome slug in work-the-backlog's
+metrics table, that the three workflow mirrors carry identical blob hashes rather
+than merely similar text, and that no fifth verdict token exists anywhere in
+=review-code=. It filed one follow-up correctly rather than fixing it in-diff —
+=start-work.md= Phase 7 still summarizes the publish flow instead of pointing at
+it, and was stale before today. Filed =[#C] :chore:solo:=.
+
+Convergence shape across the three rounds: nine findings, three, one Minor. The
+residue in later rounds was integration error from the previous round's fixes
+rather than new design problems, which is the shape the bound is calibrated for.
+
+Committed =8062460=, pushed, velox fast-forwarded and verified.
+
+Six commits this session, all pushed, both daily drivers current, inbox at zero,
+tree clean.
+
+** 2026-07-29 — Walking the remaining items, and a long lesson
+
+*Item 1, the =references/= dead link* (=5999f88=). Craig picked drop-the-link.
+The adversarial review returned =Needs Discussion= — the fourth verdict, on its
+first real use — and widened the fix twice, both correctly. My replacement prose
+said credentials "live in the rulesets repo" without naming a file, which would
+have sent readers to =calendar-reference.org=, whose three paths had been dead
+since May. Now names =mcp/README.org=. And the file itself was orphaned by the
+link removal, so both copies and the empty =references/= dirs are gone. Round 2
+approved with three Low findings, all against the task record: my count was wrong
+(seven sites, not five) and =scripts/lint.sh='s =check_md_links= already exists
+for that class, missing them only because it matches markdown syntax.
+
+*Rule gap found by using the rule.* =Needs Discussion= exits to the user, but I
+never wrote what happens after the user answers. Treated it as a fresh review on
+the reasoning that Craig adjudicated and the scope changed. Flagged to Craig; not
+yet written into the skill.
+
+*The wrap-org-table bug* (=ecd5d7b=). work reported that the tool splits a logical
+row and lint passes the result. Everything I *measured* held. Everything I
+*inferred* on top was refuted, four times:
+
+1. Root cause "absence of rules in the input" — refuted by the double-run repro
+ (the tool corrupts its own correct, rule-delimited output).
+2. "Tested and killed the empty-cell hypothesis" — the fixture was confounded.
+ With no hlines the code short-circuits at =:184= before the predicate is
+ reached, so I varied the empty cell while the path that reads it was switched
+ off, got a negative, and wrote it down as settled.
+3. "The reporter's row-below observation discriminates between the paths" — it
+ doesn't; both produce that signature. Claimed twice.
+4. "A static scan can't see primary-path exposure, needs simulation" — wrong, and
+ worse, I sent it to work, who built on it. Then my *corrected* advice (run the
+ predicate over rule-delimited groups) was also wrong: work implemented it and
+ showed it can't discriminate at any threshold, with worked examples where the
+ same structural signature has opposite correct verdicts.
+
+Three corrections sent to work, plus a fourth acknowledging their disproof. The
+review loop *bounded out* at three rounds — the first bound-out under the new
+rule, working as designed. Craig adjudicated by stripping the task to Verified /
+Two-fixes-that-work / Open-questions, which is the right shape: the measurements
+were never the problem.
+
+The fix that survived is work's read, not mine: check idempotence (reflow twice,
+diff) rather than build a detector. It's true by construction and needs nobody to
+decide what a group means.
+
+*The lesson, stated plainly.* Every refutation across three rounds landed on
+inference, never on a measurement. I ran experiments and then over-read them,
+repeatedly, at full confidence. work — after I'd sent them three wrong analyses —
+labelled their own uncertain number as a floor on a population they couldn't
+cleanly define, unprompted. That discipline is the thing to copy.
+
+I also found the 2026-07-27 note where I'd already observed this exact failure
+and left it in a session summary instead of filing it. work's framing: a correct
+observation recorded and then read as fine is the same failure as a green check
+on a corrupted table.
diff --git a/.ai/workflows/INDEX.org b/.ai/workflows/INDEX.org
index 17963ef..f18d953 100644
--- a/.ai/workflows/INDEX.org
+++ b/.ai/workflows/INDEX.org
@@ -1,5 +1,5 @@
#+TITLE: Workflow Index
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-04-25
* Purpose
@@ -15,10 +15,15 @@ This index must list every =.org= file in =.ai/workflows/= except this one and e
** Session lifecycle
- =startup.org= — runs automatically at session start. No manual trigger.
+- =helper-mode.org= — role contract for a helper instance (a second Claude in the same project as a live primary). No manual trigger; the spawn paths route to it, "you are a helper" is the manual fallback.
- =first-session.org= — initialize =.ai/= for a brand-new project.
- Triggers: "this is a new project", "let's set this project up". Auto-runs if =.ai/sessions/= is empty.
-- =wrap-it-up.org= — end-of-session: write summary, archive, commit, push.
+- =wrap-it-up.org= — end-of-session: write summary, archive, commit, push, then a phrase-dependent Step 6 teardown. Bare "wrap it up" tears the session down (kills the ai-term buffer + =aiv-<project>= tmux session via a =Stop=-hook sentinel, after the valediction flushes); a "with summary" / "and summarize" wrap keeps the buffer; "and shutdown" gates on being the only live ai-term session, then powers the machine off via an abort-able Emacs countdown.
- Triggers: "wrap it up", "that's a wrap", "let's call it a wrap"
+ - No-teardown triggers: "wrap it up with summary", "wrap it up and summarize"
+ - Shutdown trigger: "wrap it up and shutdown"
+- =suspend.org= — capture-only mid-session pause for an abrupt departure: append a resume-weighted =SUSPENDED= entry to the Session Log, note uncommitted work, and LEAVE =.ai/session-context.org= in place so the next startup resumes from it. The capture-only counterpart to =wrap-it-up= (which archives + tears down) and to =flush= (=/flush=, which prompts =/clear= and resumes the same session). Provides only the capture half; startup's interrupted-session path is the resume half.
+ - Triggers: "suspend the session", "suspend", "I need to go", "stick a pin in everything"
- =retrospective.org= — post-mortem after a tough session.
- Triggers: "let's do a retrospective", "retrospective time"
@@ -43,12 +48,19 @@ This index must list every =.org= file in =.ai/workflows/= except this one and e
- Triggers: "let's do a journal entry", "create a journal entry"
- =clean-todo.org= — tidy =todo.org=: hygiene pass + =--archive-done=, then summarize. Wrap-up does this automatically; this is the manual entry point.
- Triggers: "clean up todo.org", "clean-todo", "tidy the todo file", "archive the done items in todo.org", "run the todo cleanup"
-- =process-inbox.org= — evaluate each inbox item against a three-question value gate (advances an existing TODO / improves the project / serves the mission), then implement, fold, file, defer, or reject per source (Craig / project handoff / script). Auto-invoked by startup when inbox is non-empty. Source-aware rejection flow: handoff rejections write a response back via =inbox-send= naming the failed gate question and any reconsideration condition.
- - Triggers: "process inbox", "process the inbox", "handle the inbox", "what's in inbox", "what's in the inbox", "let's clear the inbox", "let's process the inbox items"
-- =monitor-inbox.org= — the cadence + act-vs-file + reply layer over process-inbox: check =inbox-status= at every task boundary, decide act-now (just do it) vs file (ask, file = option 1), and confirm back to handoff senders. Includes the opt-in background-monitor =/loop= recipe.
- - Triggers: "monitor the inbox", "watch the inbox", "respond to the handoffs", "handle the handoffs"
-- =inbox-zero.org= — route the *global roam inbox* (=~/org/roam/inbox.org=) to owning projects by =<project>:= heading prefix. Distinct from =process-inbox.org= (the project's own =inbox/= dir). The current session claims only its own prefixed items, files them into =todo.org=, removes them from the shared inbox, and leaves foreign/unowned items. Every scan reports the total item count plus how many appear related to this project. v1 is single-destination (prefix-claim only); domain-aware whole-inbox routing is deferred. Called read-only from startup (count + offer) and as a wrap-up Step 3 sub-step.
- - Triggers: "inbox zero", "empty the inbox", "process the roam inbox", "triage my roam inbox"
+- =inbox.org= — one engine for the project's inbox surfaces, with the shared value gate / skeptical review / disposition ladder / reply discipline / capture-guard / priority-scheme check in one place, plus thin per-surface modes. *Process mode* evaluates each project-local =inbox/= item against the three-question value gate, then implements / folds / files / defers / rejects per source (auto-invoked by startup when inbox is non-empty). *Monitor mode* runs process mode now then loops it every 15 min, gating on a clean tree + green suite and adding the act-vs-file + no-approvals-execute + reply discipline. *Roam mode* routes the global roam inbox (=~/org/roam/inbox.org=) to owning projects by =<project>:= prefix (read-only nudge at startup, sweep at wrap-up). *Auto inbox zero* runs roam mode on an interactive =/loop= at a Craig-chosen interval. Distinct from =triage-intake.org= (external accounts), which stays separate.
+ - Process-mode triggers: "process inbox", "process the inbox", "handle the inbox", "what's in inbox", "what's in the inbox", "let's clear the inbox", "let's process the inbox items"
+ - Monitor-mode triggers: "monitor the inbox", "watch the inbox", "respond to the handoffs", "handle the handoffs"
+ - Roam-mode triggers: "inbox zero", "empty the inbox", "process the roam inbox", "triage my roam inbox"
+ - Auto-mode trigger: "auto inbox zero" (match before "inbox zero")
+
+- =work-the-backlog.org= — the autonomous task-execution loop, the single home for working a batch of marked tasks unattended: takes an ordered task set (explicit list or tag query) + session mode (=file-only= default / =autonomous-commit= + paging) + a hard run cap; each candidate passes the mechanical eligibility gate (status =TODO= + =:solo:= per the project's scheme header) and the four-item defer checklist, then is implemented to the full quality bar (TDD, =/review-code=, =/voice=) as its own logical commits. Fed by the inbox auto-loop's chain step (yes-gated, file-only, cap 1) and the no-approvals speedrun preset (pre-flight Q&A → autonomous-commit + always-push + end-of-set page over an explicit ordered list).
+ - Speedrun triggers: "speedrun", "no approvals speedrun", "speedrun these: <task set>" — any phrase containing "speedrun" routes here (the preset), never to =no-approvals.org=
+ - Manual triggers: "work the backlog", "work the backlog with <task set>" (file-only defaults)
+ - Synthesis trigger: "synthesize backlog metrics" — read the per-project metrics logs, compute trends + the corrections signal, write one =:agent:metrics:= KB node (personal projects only)
+- =sentry.org= — the overnight hygiene supervisor: an interval loop (default hourly) that walks a fixed pass list (roam pull, inbox zero, triage, todo cleanup, task audit, working-files hygiene, spec board, link integrity, git health, prep freshness), commits each writing pass to a throwaway =sentry/<date>-<host>= branch (never pushed), and parks every judgment call and destructive action in a morning-approval queue. Gated on =:COMMIT_AUTONOMY: yes= plus interactive entry gates (clean tree, green suite) with Craig present. Locks via =agent-lock=; morning teardown (review, squash-merge, delete) is Craig's, never automated.
+ - Triggers: "start sentry", "run sentry", "arm sentry", "sentry mode", "start sentry every <interval>"
+ - Stop trigger: "stop sentry", "stand down sentry", "sentry off"
** Calendar
@@ -80,11 +92,18 @@ This index must list every =.org= file in =.ai/workflows/= except this one and e
- =spec-create.org= — author a design/feature spec before non-trivial work (more than ~6 hours, or real trade-offs): a when-to-spec gate, then problem-first framing, design + alternatives + inline mini-ADR decisions, implementation phases + acceptance criteria + a readiness-dimensions menu, a terseness pass, and a hand-off self-check against the review rubric. The *author* side that starts the trio; its output feeds =spec-review.org=.
- Triggers: "let's write a spec", "spec this out", "create a spec for X", "spec-create workflow"
-- =spec-review.org= — review a design/feature spec for implementation-readiness: run the readiness gate, read the code first, evaluate across dimensions, assign a rubric (Ready / Ready-with-caveats / Not-ready / Needs-research), and write a =<spec>-review.org= file when not ready. The *reviewer* side; its output feeds =spec-response.org=.
+- =spec-review.org= — review a design/feature spec for implementation-readiness: run the readiness gate, read the code first, evaluate across dimensions, assign a rubric (Ready / Ready-with-caveats / Not-ready / Needs-research), and record findings as =TODO= tasks in the spec's =* Review findings= section when not ready. The *reviewer* side; its output feeds =spec-response.org=.
- Triggers: "review the spec", "is this spec implementation-ready?", "spec-review workflow", "review the design"
-- =spec-response.org= — fold an external spec review back in: decide accept / modify / reject for every recommendation, weave accepts into the spec body, document modifies and rejects in a "Review dispositions" section, reconcile cross-spec tensions, iterate to implementation-ready. The *author* side; consumes the =<spec>-review.org= file =spec-review.org= produces.
+- =spec-response.org= — fold a spec review back in: decide accept / modify / reject for every finding, weave accepts into the spec body, complete each finding task in place (the reason recorded on modifies and rejects), reconcile cross-spec tensions, iterate to implementation-ready. The *author* side; consumes the =* Review findings= =spec-review.org= produces.
- Triggers: "respond to the review", "process the spec reviews", "spec-response workflow", "fold in the review"
+** Code quality
+
+- =code-quality.org= — one trigger that sequences every behavior-preserving quality pass over a scope of existing code: =/refactor= (complexity, duplication, dead-code, simplification) then =readability-audit= (comments, headers, names, organization), then surfaces the =:refactor:= tasks readability filed and any deferred =/refactor= findings. A thin orchestrator — each pass keeps its own gate. Excludes =/simplify= (that's for the current diff, not existing code).
+ - Triggers: "code quality sweep", "quality sweep", "run every quality pass on <scope>", "give me every pass on <scope>"
+- =readability-audit.org= — make code readable to a future maintainer: audit file-top commentary, inline comments (why-not-what), names (intention-revealing), and organization (co-location / stepdown / cohesion). The cheap comment- and name-only fixes (dimensions A/B/C) land inline, verified by a green suite; the structural findings (dimension D — split a module, rename a public symbol) are *filed* as =:refactor:= tasks, not done here. Language-agnostic. Feeds =/refactor= (which executes the filed structural work); distinct from =/refactor='s metric scans and =/simplify='s diff cleanup.
+ - Triggers: "let's run the readability-audit workflow", "audit the comments and commentary in <area>", "clean up the structure/organization of <module>", "readability audit"
+
** Tools and meta
- =process-meeting-transcript.org= — record → transcript → labeled archive.
@@ -94,8 +113,8 @@ This index must list every =.org= file in =.ai/workflows/= except this one and e
- Situational triggers: "broadcast the <event> to all projects", "broadcast that <situation>", "let every project know I'll be away ..."
- =flashcard-review.org= — review an org-drill flashcard file, restructure cards to question-form headings (no answer hints), audit content accuracy against project source-of-truth via subagent, rewrite source preserving SRS state, regenerate the Anki =.apkg= to =~/sync/phone/anki/=. Person cards use "Who is X? Tell me about their Y."; talking-points cards stay as-is. Script behavior: =flashcard-to-anki.py= strips =:PROPERTIES:= drawers + =SCHEDULED:= / =DEADLINE:= planning lines from Anki output.
- Triggers: "review the flashcards", "update the flashcards", "review the drill deck", "update the drill deck", "refresh the Anki cards", "let's run the flashcard-review workflow"
-- =page-me.org= — set a timed notification.
- - Triggers: anything containing the word "page" used as a verb ("page me", "page me in 10 minutes", "page me at 3pm")
+- =page-me.org= — set a timed notification. "page me" desktop =notify=, "text me" phone via =agent-text=, "text and page me" both.
+ - Triggers: anything containing the word "page" used as a verb ("page me", "page me in 10 minutes", "page me at 3pm", "page my phone")
- =status-check.org= — proactive long-running-job updates.
- Triggers: "keep me posted on this", "provide status checks on this job", "let me know when it's done", "monitor this for me". Auto: any job estimated 10+ min.
- =create-workflow.org= — define a new workflow.
@@ -106,7 +125,7 @@ This index must list every =.org= file in =.ai/workflows/= except this one and e
- Triggers: "session harvest", "harvest the sessions", "let's run the session-harvest workflow", "monthly harvest", "mine the sessions"
- =no-approvals.org= — drop the interaction-level approval gates for a pre-agreed batch while keeping engineering-discipline gates (=/review-code=, =/voice personal=, tests, session-log updates, subagent reviews, destructive-action consent). Mode stays on until Craig turns it off, a real question arises, the queue empties, or the conversation switches topics.
- Triggers: "no-approvals mode", "no approvals", "no-approval", "no need for approval gates", "stop asking, just keep going", "I'll check back in when you're done or stuck", "do all =<selector>= with no-approval"
-- =cross-agent-comms.org= — protocol for cross-project agent coordination via =inbox/from-agents/= (file-based IPC, GPG-signed, supports cross-machine over Tailscale). Auto: when =cross-agent-watch= detects a new inbound message, or when an agent decides to initiate a cross-project conversation. Operational scripts (=cross-agent-send=, =-recv=, =-watch=, =-status=, =-discover=, =-halt=, =-resume=) and their READMEs live at =.ai/scripts/cross-agent-comms/=.
+ - Exception: any phrase containing "speedrun" routes to =work-the-backlog.org='s no-approvals speedrun preset instead
* Living Document
diff --git a/.ai/workflows/add-calendar-event.org b/.ai/workflows/add-calendar-event.org
index 2650fb7..5dd6c42 100644
--- a/.ai/workflows/add-calendar-event.org
+++ b/.ai/workflows/add-calendar-event.org
@@ -1,5 +1,5 @@
#+TITLE: Add Calendar Event Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-01
* Overview
diff --git a/.ai/workflows/broadcast.org b/.ai/workflows/broadcast.org
index 1be07d2..cc14f00 100644
--- a/.ai/workflows/broadcast.org
+++ b/.ai/workflows/broadcast.org
@@ -1,5 +1,5 @@
#+TITLE: Broadcast Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-29
* Overview
@@ -159,11 +159,11 @@ broadcast, not a task and not tailored to this project.
- Ask Craig any follow-up questions then — this message is deliberately general.
#+end_example
-The "For the receiving agent" block is fixed text — it travels with every situational broadcast so the message is self-describing. A receiving project's =process-inbox= reads it and acts on those instructions without needing any special-casing; the value gate accepts it as situational awareness that improves how the project works.
+The "For the receiving agent" block is fixed text — it travels with every situational broadcast so the message is self-describing. A receiving project's =inbox.org= process mode reads it and acts on those instructions without needing any special-casing; the value gate accepts it as situational awareness that improves how the project works.
** Receiving behavior (what a project does with an incoming situational broadcast)
-When =process-inbox= encounters a =Broadcast:= item, the disposition is *record-and-hold*, not file-as-task:
+When =inbox.org= process mode encounters a =Broadcast:= item, the disposition is *record-and-hold*, not file-as-task:
1. Add a dated entry to =notes.org= Active Reminders capturing the situation and its end date (if any).
2. If the event bears on an open task, note the connection in that task's body.
diff --git a/.ai/workflows/clean-todo.org b/.ai/workflows/clean-todo.org
index dd33056..48d3084 100644
--- a/.ai/workflows/clean-todo.org
+++ b/.ai/workflows/clean-todo.org
@@ -1,5 +1,5 @@
#+TITLE: Clean-Todo Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-11
* Overview
@@ -27,7 +27,17 @@ Deletes bogus =- State "X" from "X" [date]= log lines (state didn't actually cha
To preview without writing, run =--check= first: =emacs --batch -q -l .ai/scripts/todo-cleanup.el --check todo.org=.
-** Step 2: Archive completed work
+** Step 2: Convert done sub-tasks to dated entries
+
+#+begin_src bash
+emacs --batch -q -l .ai/scripts/todo-cleanup.el --convert-subtasks todo.org
+#+end_src
+
+Rewrites every heading at level 3 or deeper whose TODO state is DONE/CANCELLED/FAILED into a dated event-log entry (=<stars> YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <text>=), dropping the keyword, priority cookie, and tags, and removing the =CLOSED:= line. Enforces the depth rule that a completed sub-task becomes dated history — a shape interactive org closes and =--archive-done= (level-2 only) leave unapplied. Timestamp comes from each entry's =CLOSED= cookie; heading text kept verbatim; idempotent; a done sub-task with no parseable =CLOSED= is flagged and left alone. Run before archiving so a parent's sub-tasks are already dated when it moves. Capture the output.
+
+To preview without writing: =emacs --batch -q -l .ai/scripts/todo-cleanup.el --convert-subtasks --check todo.org=.
+
+** Step 3: Archive completed work
#+begin_src bash
emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done todo.org
@@ -37,10 +47,11 @@ Moves every level-2 subtree whose TODO state is DONE or CANCELLED out of the "Op
To preview the moves without writing: =emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done --check todo.org=.
-** Step 3: Summarize
+** Step 4: Summarize
-Report to Craig from the two captured outputs:
+Report to Craig from the three captured outputs:
- Hygiene: how many bogus state-log lines were deleted; any orphan-planning warnings (file:line + heading), or "none".
+- Convert: how many done sub-tasks were rewritten to dated entries (heading + line), any flagged for no =CLOSED= date, or "nothing to convert".
- Archive: how many subtrees moved and which (heading + line), or "nothing to move" / the skip reason if a section was missing or ambiguous.
- If the file changed, note that =todo.org= now has an uncommitted edit — review =git diff -- todo.org= and commit it (in this repo's commit style) if it looks right. If nothing changed, say so and stop.
@@ -49,7 +60,7 @@ Don't auto-commit. The summary is the review point; Craig decides whether the di
* Principles
- *Both passes apply, not just preview.* The workflow is invoked because cleanup is wanted. Use the =--check= variants only when Craig asks for a dry run.
-- *Two passes, two invocations.* =--archive-done= is its own mode and does not run the hygiene pass; run both.
+- *Separate modes, separate invocations.* =--convert-subtasks=, =--archive-done=, and the hygiene pass are each their own mode and don't run the others; run all three.
- *Never auto-commit todo.org.* Surface the diff and let Craig commit it. The cleanup is a working-tree change, fully reversible until committed.
- *Trust the script.* It's fast and idempotent; if there's nothing to do, it reports zero and exits clean. No pre-checks.
diff --git a/.ai/workflows/code-quality.org b/.ai/workflows/code-quality.org
new file mode 100644
index 0000000..3c4ed8f
--- /dev/null
+++ b/.ai/workflows/code-quality.org
@@ -0,0 +1,90 @@
+#+TITLE: Code-Quality Sweep Workflow
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-06-28
+
+* Overview
+
+One trigger that runs every behavior-preserving quality pass over a scope of
+*existing* code, in order, then surfaces what got filed for later. It's a thin
+orchestrator — each pass keeps its own discipline and its own confirm gate; this
+workflow only sequences them and collects the residue.
+
+*Behavior-preserving rests on a test net.* The passes below claim to preserve
+behavior, but a refactor on untested code is a guess, not a preservation. Where
+the scope has no tests, bring it under a characterization net first
+(Normal/Boundary/Error per unit, per the =testing-standards= skill's "Adding Tests to Existing
+Untested Code") — that net is what turns "behavior-preserving" from an assertion
+into something the green suite actually verifies across each pass.
+
+The passes it chains:
+
+1. =/refactor= — structural and logic cleanup on measurable metrics (complexity,
+ duplication, dead-code) plus the simplification lens.
+2. =readability-audit= ([[file:readability-audit.org][readability-audit.org]]) — prose and human-reader clarity
+ (comments, file headers, names, organization).
+
+It deliberately does *not* run =/simplify=: that works the current uncommitted
+diff, not existing committed code, so it belongs to the moment you've just made a
+change, not to a sweep of code already in the tree (see "The /simplify boundary"
+below).
+
+* When to Use This Workflow
+
+- "code quality sweep" / "quality sweep"
+- "run every quality pass on <scope>" / "full quality pass on <scope>"
+- "give me every pass on <file/module/tree>"
+
+Do NOT use it for:
+- *In-flight diff cleanup* — that's =/simplify= on the change you just made.
+- *Bug hunting* — these passes are behavior-preserving; for defects use =debug=
+ or =/review-code=.
+- *Performing the structural refactors it files* — those become =:refactor:=
+ tasks; work them later via =/refactor rename= / =/refactor simplification= or
+ =/start-work=.
+
+* Steps
+
+** 1. Scope
+
+Pick the target: one file, a named module set, or the whole tree (honor
+=.aiignore=). The same scope is passed to both passes so they cover the same
+code.
+
+** 2. /refactor <scope>
+
+Run =/refactor= on the scope. Its default full scan covers complexity,
+duplication, dead-code, and simplification. It presents findings and applies
+only what's approved (its own gate) — structure and logic first, so the
+readability pass audits the cleaned-up code.
+
+** 3. readability-audit on <scope>
+
+Run the readability-audit workflow on the same scope. Its cheap comment- and
+name-only fixes (dimensions A/B/C) land inline and are verified by a green
+suite; its structural findings (dimension D — split a module, rename a public
+symbol) are *filed* as =:refactor:= tasks rather than done here.
+
+** 4. Surface the residue
+
+Collect and report what the sweep left behind for later work:
+
+- The =:refactor:= tasks readability-audit filed (the structural backlog).
+- Any =/refactor= findings deferred rather than applied in step 2.
+
+That residue is the "do this next" list the sweep produces; it's not a failure
+to finish, it's the structural work that needs its own design and test pass.
+
+* The /simplify boundary
+
+=/simplify= and this sweep don't overlap: =/simplify= cleans the *current diff*
+and applies its fixes directly, so reach for it right after making a change,
+before committing. This sweep works *existing committed code* and runs the
+scan-and-present passes. One trigger can't sensibly do both — a diff you're
+holding and a tree you're auditing are different inputs.
+
+* Verification
+
+Each pass owns its verification (=/refactor= runs the suite after applying;
+readability-audit verifies inline fixes against a green suite). The umbrella
+adds nothing beyond sequencing, so when both passes report green, the sweep is
+clean — confirm that before reporting done rather than assuming it.
diff --git a/.ai/workflows/create-workflow.org b/.ai/workflows/create-workflow.org
index 6060df1..393fce5 100644
--- a/.ai/workflows/create-workflow.org
+++ b/.ai/workflows/create-workflow.org
@@ -195,7 +195,7 @@ After the Q&A, ask together:
Decide on a name for this workflow.
*Naming convention:* Action-oriented (verb form)
-- Examples: "refactor", "inbox-zero", "create-workflow", "review-code"
+- Examples: "refactor", "clean-todo", "create-workflow", "review-code"
- Why: Shorter, natural when saying "let's do a [name] workflow"
- Filename: =.ai/workflows/[name].org=
@@ -240,10 +240,10 @@ Update =notes.org=:
Example entry:
#+begin_src org
-,** inbox-zero
-File: =.ai/workflows/inbox-zero.org=
+,** journal-entry
+File: =.ai/workflows/journal-entry.org=
-Workflow for processing inbox to zero:
+Workflow for capturing a daily journal entry:
1. [Brief workflow summary]
2. [Key steps]
diff --git a/.ai/workflows/cross-agent-comms.org b/.ai/workflows/cross-agent-comms.org
deleted file mode 100644
index 430b4b0..0000000
--- a/.ai/workflows/cross-agent-comms.org
+++ /dev/null
@@ -1,334 +0,0 @@
-#+TITLE: Cross-Agent Communication Workflow (v5)
-#+AUTHOR: Craig Jennings & Claude (homelab + career sessions)
-#+DATE: 2026-04-27
-#+VERSION: 5
-
-* Status
-
-Draft. Iterating between the homelab and career sessions through a multi-round design discussion. Awaiting Craig's review for promotion to =~/code/rulesets/claude-templates/.ai/workflows/=.
-
-v5 changes from v4:
-- *Script absorption.* Seven operational scripts (=cross-agent-send=, =cross-agent-recv=, =cross-agent-watch=, =cross-agent-status=, =cross-agent-discover=, =cross-agent-halt=, =cross-agent-resume=) now own most implementation detail. Their READMEs are the operational source of truth. The spec stays declarative.
-- *Failsafe halt.* Layered HALT-file mechanism stops all cross-agent activity on a machine within ~5 min, without visiting individual sessions or restarting Claude Code. =cross-agent-halt= and =cross-agent-resume= are the convenience entry points; every other component checks the HALT file independently.
-- *Identity.* Messages are GPG-signed by sender and verified by receiver. Combined with POSIX permissions on =from-agents/= and Tailscale-level network auth, identity becomes a three-layer story.
-- *Atomic writes.* Writers MUST use temp-file + rename. =cross-agent-send= handles this; the spec just states the contract.
-- *Dedup.* Sequence-collision dedup is now binary SHA-256 equality, not a fuzzy ">90% match" threshold.
-- *Cold-start handling.* Layered: =cross-agent-watch= (push notifications via =inotifywait=) is the primary mechanism; startup-workflow check and user-direct-injection are coverage layers.
-- *Spec stays roughly the same length but does more protocol work.* Operational detail (rsync retry numbers, inotifywait recipes, peers.toml schema, GPG flags, dedup mechanics) moved to the script READMEs. The spec adds new protocol elements (identity layer, atomic-writes contract, SHA-256 dedup, =escalate= type, =RELEASE_STATUS= values, =REQUIRES_TOOLS= optional field) in the freed space. Total documentation surface (spec + seven READMEs ≈ 1000 lines) is larger than v4's 259 lines, but the spec and the READMEs serve different audiences — protocol-thinkers and CLI-users — and a reader of just the spec can comprehend the protocol without consulting any README.
-
-* When to use
-
-When two Claude sessions in different projects (same machine or different machines on the same Tailscale tailnet) need to coordinate on a shared task that one session can't complete alone — typically because one has tooling, context, or MCP access the other doesn't.
-
-Examples that fit:
-- Session A asks session B to apply a workflow patch in B's project, then verify it.
-- Session A runs a long task and needs session B to monitor results in B's domain.
-- Two sessions co-design a workflow.
-
-Examples that don't fit:
-- A simple file handoff that doesn't require iteration.
-- A task one session can do alone.
-- Cross-tailnet or cross-organization. The protocol is local-tailnet-scoped.
-
-* Protocol
-
-** File location
-
-Each project has =inbox/from-agents/= as its agent-comms mailbox. Create the directory if it doesn't exist; set permissions =chmod 700= and ownership to the user.
-
-- Sender writes to receiver's =inbox/from-agents/=.
-- Receiver polls (or watches) =inbox/from-agents/=, *not* the parent =inbox/=.
-- The parent =inbox/= stays reserved for human-triage items.
-- Out-of-band artifacts (PDFs, datasets) live at =inbox/from-agents/artifacts/=. Reference by relative path in the message body.
-
-The user does NOT write directly to =from-agents/=. To inject input into a running conversation, the user tells one of the agents in that agent's session; the agent writes the input as a normal message attributed to the user.
-
-** File naming
-
-=YYYYMMDDTHHMMSSZ-from-<sender>-<short-conv-id>.org=
-
-- Timestamp is UTC ISO 8601 compact. The trailing =Z= is mandatory.
-- =from-<sender>= prefix.
-- =<short-conv-id>= is a stable kebab-case slug across the back-and-forth. Reusable across time; ordering relies on filename timestamps.
-
-Frontmatter =#+TIMESTAMP= carries the same instant in local time with explicit offset. The two MUST refer to the same instant.
-
-The implementation (=cross-agent-send=) generates the canonical filename from the message's frontmatter (=CONVERSATION_ID=, current UTC time) and the sender's project context. Senders supply only the message body file; the script handles naming. Senders MUST NOT pre-name files in this format and pass them through; the script overwrites with its own canonical name to ensure consistency and enable the sender-side max-seen sequence-collision-reduction scan.
-
-GPG signatures live in a sibling file =YYYYMMDDTHHMMSSZ-from-<sender>-<short-conv-id>.org.asc=. Receivers verify before processing. See =* Writes are atomic= for the two-file delivery ordering rule.
-
-** Frontmatter
-
-Required:
-
-#+begin_example
-#+TITLE: <human-readable subject>
-#+CONVERSATION_ID: <stable across the thread>
-#+MESSAGE_TYPE: <see types below>
-#+SEQUENCE: <integer hint>
-#+TIMESTAMP: <ISO 8601 with explicit offset>
-#+PROTOCOL_VERSION: 5
-#+end_example
-
-Optional:
-
-#+begin_example
-#+REQUIRES_TOOLS: <comma-separated tool/MCP slugs, e.g. gmail-mcp, slack-mcp>
-#+RELEASE_STATUS: <see release-statuses; valid only on MESSAGE_TYPE: release>
-#+WORKFLOW_VERSION: <sender's version of cross-agent-comms.org; informational only in v5 — no enforcement>
-#+end_example
-
-Receiver sanity-checks frontmatter before acting. Missing or malformed frontmatter → surface to user, don't proceed. Mismatched =PROTOCOL_VERSION= → receiver writes a =query= asking the originator to upgrade.
-
-** Identity
-
-Messages are GPG-signed by the sender. Receivers verify the detached signature before processing the message body.
-
-The implementation (=cross-agent-send=) signs automatically with the sender's configured key (the user's primary GPG key by default; configurable via =--key= flag or environment). Receivers verify automatically against the keys in their GPG keyring.
-
-Identity is a three-layer story:
-
-1. *Tailscale layer.* Only tailnet members can reach the rsync-over-SSH endpoint at all.
-2. *POSIX layer.* =chmod 700= on =from-agents/= means only processes running as the directory's owner can write.
-3. *GPG layer.* Sender's signature on each message proves the message originated from a process holding the key.
-
-Three independent layers. Per-user GPG (using existing keys) gives a correctness check more than a security boundary — unsigned messages are almost certainly bugs, not attackers. That's still load-bearing.
-
-** Writes are atomic
-
-Writers MUST use a temp-file + rename pattern (=mktemp= + =mv= within the same filesystem) so receivers never see partial files. The implementation script (=cross-agent-send=) handles this.
-
-Receivers ignore =.tmp.*= files, processing only the final renamed name.
-
-*Two-file ordering.* When a message has a sibling GPG signature file (=.org.asc=), the writer MUST rename the =.asc= to its final name *before* renaming the =.org=. Two =mv= operations are not atomic together — without this ordering, a receiver could read the =.org= in the window between the two renames and fail GPG verify because the =.asc= hasn't landed yet. The rule: receiver only acts on =.org= files, and a =.org= without a corresponding =.asc= means the signature is genuinely missing (not still in flight).
-
-** Sequence numbering
-
-=#+SEQUENCE= is a *hint*, not a strict counter. Canonical order is =#+TIMESTAMP=. Sequences may collide under rapid back-and-forth (both sides write what they think is sequence N near-simultaneously). Treat collision as a normal protocol event.
-
-*Receiver-side dedup rule.* When a new file shares =CONVERSATION_ID= + =SEQUENCE= with an already-processed message, compare SHA-256 hashes. Identical hashes → silent dedup, treat as a retry. Different hashes → process both, ordered by =#+TIMESTAMP=.
-
-*Sender-side collision-reduction (best-effort).* Before picking sequence, scan the receiver's =from-agents/= for the highest existing sequence in this conversation across both sender prefixes. Use =max(seen) + 1=.
-
-** Message types
-
-- *request* — a side asks for work, input, or a decision. Sequence 1 is always =request=.
-- *progress* — work-in-progress checkpoint. "Here's where I am, no action needed from you, more coming." Originator's poll loop should NOT page the user on progress messages.
-- *query* — either side asks a clarifying question that blocks further work. Originator's poll loop SHOULD surface this immediately. Originator answers and work continues.
-- *pushback* — receiver formally disagrees with the request and has *not* started the work. Carries reasoning. Distinct from =query= because the originator's response path differs.
-- *complete* — receiver signals the requested work is done. Triggers verification.
-- *release* — terminal type. Originator writes after verifying =complete=. Carries =RELEASE_STATUS= to disambiguate the closure mode.
-- *escalate* — punts the conversation to the user for adjudication. Both sides pause polling on =escalate=; the user resolves.
-
-Reply expectation is implied by type: =request=, =query=, =pushback=, =escalate= expect a reply; =progress=, =complete=, =release= don't.
-
-** Conversation lifecycle
-
-A conversation is a directed loop between an originator (issued sequence 1) and a receiver:
-
-1. Originator writes =request= (sequence 1). Begins polling for replies.
-2. *Optional acknowledgment.* Receiver may write a =progress= at sequence 2 to acknowledge receipt and set expectations. Required if work will take >5 minutes (so the originator's poll loop doesn't waste wakes).
-3. *Optional echo-back.* For ambiguous or large requests, receiver writes a =progress= that restates work items and announces "starting now unless you push back within N minutes."
-4. Receiver works. May write =progress= updates. =query= mid-work if blocked. =pushback= if the request is wrong.
-5. Receiver writes =complete=. Begins polling for =release=.
-6. Originator reads, *verifies the deliverable directly*. For subjective deliverables, verification is the originator's editorial accept.
-7. If verified: =release= with =RELEASE_STATUS: complete=. If problems: new =request= (next sequence number).
-8. Receiver sees =release=, stops polling.
-
-The verification step is load-bearing. =complete= is a *claim*; =release= is *verification*.
-
-** Pushback path
-
-On receiving a =pushback=, the originator chooses:
-
-1. *Revise* — new =request= with adjusted scope.
-2. *Insist* — new =request= addressing the pushback's reasoning, standing by direction.
-3. *Withdraw* — =release= with =RELEASE_STATUS: withdrawn-after-pushback=.
-
-*Deadlock cap.* After two pushback-insist exchanges, the next message MUST be =MESSAGE_TYPE: escalate=. Both agents pause polling; the user resolves.
-
-** =RELEASE_STATUS= values
-
-| Status | Meaning |
-|---+---|
-| =complete= | Goal achieved, originator verified |
-| =cancelled= | Originator changed their mind mid-conversation |
-| =withdrawn-after-pushback= | Originator chose option 3 on receiver's =pushback= |
-| =abandoned-after-escalation= | User adjudicated and chose to close the conversation |
-| =abandoned-after-timeout= | Receiver auto-closed after originator never returned to verify |
-
-** Async fallback
-
-If the originator session ends between =request= and =complete=, the receiver's =complete= goes unverified. Receiver behavior:
-
-- Polls for =release= up to ~24 hours of cycles (implementation default).
-- After timeout, writes a final =progress= message ("treating as terminal-without-verification; originator never returned to release") and stops polling. Receiver does NOT write =release= itself — that would contradict the lifecycle rule that =release= is the originator's terminal action.
-- Next time the originator project starts, the unreleased =complete= is surfaced as a startup item. The user can issue a late =release= (with whichever =RELEASE_STATUS= fits) or open a fresh conversation to revisit. =RELEASE_STATUS: abandoned-after-timeout= is used at that point if the user wants to formally close the orphaned thread.
-
-** Escalation
-
-A side writes =escalate= when:
-- Pushback-insist deadlock cap reached.
-- Conversation has stalled (no productive movement in N exchanges).
-- A reply-expecting message has gone unanswered past timeout.
-
-Body summarizes both sides' positions in 60 seconds of reading. Both agents pause polling; the user resolves.
-
-* Implementation notes
-
-This sub-section describes how to operate the protocol. Operational detail lives in the seven scripts' READMEs.
-
-** Recommended scripts
-
-| Script | Replaces user action | README |
-|---+---+---|
-| =cross-agent-send <dest> <msg>= | Filename generation, GPG sign, atomic write, peer lookup, rsync push, retry+backoff, failure surfacing — seven mechanical sender-side steps. Frontmatter and message body are still author-supplied. | =cross-agent-send.md= |
-| =cross-agent-recv <msg>= | Frontmatter sanity-check, =PROTOCOL_VERSION= verify, GPG verify, SHA-256 dedup, =REQUIRES_TOOLS= check — five mechanical receiver-side steps. Output is a structured decision (=process= / =dedup= / =query= / =reject=) the agent acts on. | =cross-agent-recv.md= |
-| =cross-agent-watch= | Manually checking inboxes; "did I get a message?" | =cross-agent-watch.md= |
-| =cross-agent-status= | Walking each project to count pending messages | =cross-agent-status.md= |
-| =cross-agent-discover= | Remembering project topology and reachability | =cross-agent-discover.md= |
-| =cross-agent-halt [reason] [--tailnet]= | Visiting each session to stop polling, restarting Claude Code, or hand-killing processes when comms go runaway. =--tailnet= propagates HALT to all peers. | =cross-agent-halt.md= |
-| =cross-agent-resume [--tailnet]= | Manually clearing the HALT state and restarting the watcher. Per-session polling does NOT auto-resume — the user re-engages each session explicitly. | =cross-agent-resume.md= |
-
-The scripts are tools the user runs from any terminal. They do not depend on agent context — =cross-agent-status= run from a fresh shell works.
-
-A reader can comprehend this protocol from this spec alone. Script READMEs add operational detail that makes the protocol practical to use, but understanding the protocol's semantics requires only this document.
-
-** Polling
-
-Default cadence: 270 seconds (≈4.5 min). Sits just under the 5-minute prompt-cache TTL.
-
-If a side needs to slow down (heads-down work, idle wait), it writes a =progress= message saying so in prose. The other side adapts. There are no named polling modes.
-
-After ~12 empty polls in a row, the poll loop surfaces the silence to the user.
-
-A future runtime with native filesystem-event support could replace polling for active sessions; =cross-agent-watch= already provides event-driven notifications outside active sessions.
-
-** User multi-tasking
-
-- *Deferral.* If the user's last message in the agent's session was less than 60 seconds ago AND a poll fires, queue the inbox check until either the user sends another message OR 5 minutes pass without further input.
-- *Surfacing.* On the next user-facing response: "While we were working on X, a cross-agent message landed from <project>. It's a =<type>= — want me to handle it now or after we finish?"
-- *Mid-question.* Answer the user first.
-- *Project switch.* If the user moves to the receiver project mid-conversation, the receiver agent surfaces the in-flight thread on first user prompt.
-- *Conversation state.* Always include in any response that mentions a cross-agent thread: "<conv-id> at sequence N, awaiting <event>."
-
-** Failure modes
-
-The seven scripts surface most failures with concrete error messages. Spec-level failure modes:
-
-- *Malformed frontmatter on a received file.* Surface to user; do not act.
-- *Mismatched =PROTOCOL_VERSION=.* Receiver writes =query= asking originator to upgrade.
-- *Missing or invalid GPG signature.* Receiver surfaces "unsigned/unverified message"; refuses to act.
-- *Sequence collision* with non-matching SHA-256. Process both, ordered by timestamp.
-- *Required tool unavailable.* Receiver checks =REQUIRES_TOOLS= during frontmatter-sanity-check (before any work begins). On a missing tool, receiver writes =query= asking the originator to reframe the request to avoid the unavailable tool. Originator may revise (new =request=) or withdraw (=release= with =RELEASE_STATUS: cancelled=). =query= is the right type rather than =pushback= because missing-tool is a capability gap, not disagreement.
-- *Runaway resource usage.* User invokes =cross-agent-halt= globally (or =cross-agent-halt --tailnet= for cross-machine). HALT file stops all components within one polling cycle (~5 min). See =* Halt mechanism= for the layered checks.
-- *User halts mid-conversation.* Both sides write a final =progress= note ("HALT fired; pausing"); polling stops within one cadence; conversations resume on explicit per-session re-engage after HALT clears.
-- *HALT file accidentally created* (typo, errant =touch=). =cross-agent-status= prominently flags HALT active; user clears with =cross-agent-resume=. Cost: no messages send during the typo window.
-- *HALT file unreadable* (perms wrong, partial write). Each component fails-closed (treats as halted) and reports "HALT file present but unreadable; treat as halted." Safer than fail-open.
-
-Operational failures (rsync push fails, watcher dies, peer unreachable) live in the script READMEs' failure-mode tables.
-
-* Halt mechanism
-
-A failsafe to stop all cross-agent activity on a machine without visiting individual sessions or restarting Claude Code. Designed for the runaway-polling case: an agent has spun up conversations with N other agents, polling is eating CPU, and the user needs to stop everything *now*.
-
-** The HALT file
-
-Path: =~/.config/cross-agent-comms/HALT=.
-
-Existence triggers halt across all components on the machine. The file's body may carry an optional human-readable reason (reviewed by the user later when deciding to resume).
-
-User commands:
-
-#+begin_example
-$ touch ~/.config/cross-agent-comms/HALT # halt
-$ rm ~/.config/cross-agent-comms/HALT # resume
-#+end_example
-
-Or via convenience scripts (=cross-agent-halt= / =cross-agent-resume=) that also handle the watcher service and cross-machine propagation.
-
-** Layered checks (the failsafe property)
-
-Every component MUST check the HALT file. The "any one component stops the system independently" property is what makes this failsafe — the system doesn't depend on a single point doing the right thing.
-
-| Component | Check timing | Behavior on HALT |
-|---+---+---|
-| =cross-agent-send= | At start of send + between =.asc= and =.org= rsync + between retry iterations | Refuse to start new send; complete current step then exit. Worst case: one in-flight send finishes within a few seconds. |
-| =cross-agent-recv= | Before any verify or dedup | Leave inbound message in place — do NOT dedup, reject, or move. Resume picks it up via cold-start handling. |
-| =cross-agent-watch= | At iteration start | Suppress notifications; log only. Continues running, no-op until HALT clears. |
-| =cross-agent-status= | At start | Print prominent "⚠ HALT ACTIVE" banner before normal output. Read-only, continues. |
-| =cross-agent-discover= | At start | Print HALT banner; continue read-only enumeration. |
-| Agent polling loop | First action on every wake | Write a final =progress= note to any active conversation ("HALT fired; pausing"), do NOT reschedule, surface "halt active" to user. Polling decays within one cadence (~5 min). |
-| Agent user-facing responses | Every response while HALT is set | Append "(HALT active; cross-agent comms paused)" to the response. On HALT clear, the next response says "(HALT cleared; cross-agent comms ready to resume — say so to re-engage polling)." Persistent, not just first-response — keeps awareness alive. |
-| Conversation initiator | Before writing sequence 1 of any new conversation | Refuse and surface to user. |
-| Startup workflow | Phase A on session start | If HALT exists, surface immediately and skip cross-agent inbox checks. |
-
-The agent polling-loop check is the load-bearing one for "stops eating CPU." Wake-ups already scheduled fire, but each wake on-HALT is a no-op + reschedule-prevention. Within one polling cadence (~5 min) all polling stops.
-
-*Fail-closed on unreadable HALT.* If the HALT file exists but is unreadable (wrong permissions, partial write), components MUST treat as halted. Safer than fail-open.
-
-** Resume asymmetry (deliberate)
-
-Halt is automatic everywhere. Resume requires explicit user intent per-session.
-
-When the user removes HALT (or runs =cross-agent-resume=), components stop refusing to act, but agent polling does NOT auto-resume. The user must open each session and tell that agent to resume polling for its conversations.
-
-The asymmetry exists because:
-
-1. Auto-resume could silently invert intentional kills. If the user halted because a session was misbehaving, removing HALT shouldn't quietly revive it.
-2. Per-session resume forces the user to look at each session and confirm the situation is resolved before re-engaging.
-
-** Cross-machine halt
-
-=cross-agent-halt --tailnet= iterates =peers.toml= and SSH-touches HALT on each peer. Same shape for resume.
-
-Reports per-peer status with non-zero exit on partial halt:
-
-#+begin_example
-$ cross-agent-halt --tailnet
-Halting velox.local ✓ (HALT file written)
-Halting bastion.local ✗ (ssh exit 255: no route to host)
-Halting locally ✓ (HALT file written)
-
-PARTIAL HALT: 2/3 machines halted. bastion.local needs manual halt.
-Exit 1.
-#+end_example
-
-Scripting can detect partial halt via the exit code. Same pattern for =--tailnet= on resume.
-
-* Limitations
-
-- *Local-tailnet only.* Filesystem IPC + rsync over SSH. Cross-tailnet or cross-organization is out of scope.
-- *Identity has three layers (Tailscale + POSIX + GPG)* but no message-content encryption. Confidentiality is not the goal; signing is correctness, not secrecy.
-- *Single-receiver per conversation.* Fan-out to multiple receivers requires manually orchestrating multiple parallel conversations.
-- *Polling is best-effort.* A wake may be delayed by an in-flight tool call until the runtime is idle. =cross-agent-watch= mitigates by offering event-driven notifications.
-- *Project-extension drift.* If two projects' =.ai/project-workflows/= modify shared workflow definitions in incompatible ways, cross-agent assumptions can diverge silently. The optional =#+WORKFLOW_VERSION= advisory field is informational only in v5 — no implementation reads or acts on it. A future version may add enforcement on mismatch (e.g. receiver writes =query= asking which side is stale). Today, alignment is verified manually before high-stakes conversations.
-
-* Persistence after release
-
-Conversation files persist by default. The conversation log is the audit trail.
-
-Manual archival is fine if the inbox grows unmanageable. Suggested cadence: once the conversation has been =release='d AND the work it produced has shipped, archive both projects' message files into =.ai/sessions/cross-agent/= as a flat directory — no per-conversation subdirectories. Rename each archived file to lead with the conversation-id so messages from the same conversation cluster on =ls=: =<conv-id>-<TIMESTAMP>-from-<sender>.org= (and the matching =.asc= sibling, if present). Inbox filenames lead with the timestamp because chronological arrival is what matters in =from-agents/=; archives invert that because grouping by conversation is what matters when reading history. Keep the =.asc= signatures alongside the =.org= files in archive — they're small and document the GPG verification chain.
-
-Old messages don't affect protocol behavior (=cross-agent-status='s pending semantics correctly ignore released messages) but the =from-agents/= directory grows indefinitely without manual archival. =cross-agent-status= performance degrades noticeably when a project's =from-agents/= exceeds a few hundred files. =cross-agent-init= (deferred to v6) would include an archival sub-command.
-
-* Open questions
-
-- *=cross-agent-init= and =cross-agent-compose= helper scripts.* =-init= would be one-command project bootstrap (creates =inbox/from-agents/= with =chmod 700=, installs the =cross-agent-watch= systemd path unit, validates peer config, runs a discovery probe). =-compose= would be interactive frontmatter authoring (prompts for required fields, produces a draft message file). Both deferred to v6. Current onboarding requires manual =mkdir= + systemd setup per =cross-agent-watch.md='s install recipe; current message authoring requires writing the file by hand or via a small in-agent template.
-- *Hard conversation timeout.* The async-fallback timeout is implementation-default ~24 hours. Right number depends on use case; tighten as patterns emerge.
-- *=paused= polling state.* Today there's no clean signal for "pause without ending." Add when first user complaint surfaces.
-- *Multi-LLM context.* If we ever bring in a non-Claude agent, the protocol's natural-language framing may need formalization.
-
-* Examples
-
-** =prep-fixup= conversation (2026-04-26 → 2026-04-27)
-
-Eleven exchanges between homelab and career produced the v4 spec by iterative critique-and-simplification. Three real-time sequence collisions during the conversation drove the sequence-as-hint rule that landed in v4 and persists in v5.
-
-Files at =~/projects/{homelab,career}/inbox/from-agents/= named =*-prep-fixup.org=. Worth re-reading when designing future cross-agent flows.
-
-** =comms-cold-start-discovery= conversation (2026-04-27)
-
-The follow-up that produced this v5 spec. Cold-start, watcher tooling, agent discovery, GPG identity, sha256 dedup, atomic writes, POSIX perms, script absorption, and process-vs-text simplification. Tonight's first cold-start in real time (career session went dormant after =prep-fixup= release; Craig's user-injection re-engaged it) is the worked demonstration of the v5 user-injection rule.
-
-Files at =~/projects/{homelab,career}/inbox/from-agents/= named =*-comms-cold-start-discovery.org=.
diff --git a/.ai/workflows/daily-prep.org b/.ai/workflows/daily-prep.org
index b6989e7..3f21214 100644
--- a/.ai/workflows/daily-prep.org
+++ b/.ai/workflows/daily-prep.org
@@ -1,5 +1,5 @@
#+TITLE: Daily Prep Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-06-11
* Overview
@@ -74,7 +74,7 @@ The separate =* Standup Briefs= and =* Upcoming Deadlines= sections are *retired
Items that frame the day. Four standing items are *always* present, plus the look-ahead:
1. *Meeting-density framing.* One line on the day's shape, e.g. "Meeting-dense morning: 09:00 team discussion, 10:00 standup, 11:00 general standup. Real focus time only opens at noon."
-2. *Calendar events from BOTH calendars* (work + personal): birthdays, holidays, anniversaries, vacations, trips, big events. "Your trip begins Friday." "It's <person>'s birthday today."
+2. *Calendar events from BOTH calendars* (work + personal): birthdays, holidays, anniversaries, vacations, trips, big events. "Your trip begins Friday." "It's <person>'s birthday today." Fold in the =upcoming_birthdays.py= block from Phase A (source 8) for contact birthdays the calendar doesn't carry; keep its callout entries (within 7 days) so a gift or plan gets prompted.
3. *Reminders due, imminently due, or past trigger* — from notes.org Active Reminders plus scheduled/deadline tasks. A reminder tied to a rescheduled meeting reports against the new date.
4. *Requested metrics* — a slot for any metric Craig has asked to track in the daily prep. Render the slot only when a metric is active; none are active by default. (Metric design is a separate discussion — don't invent metrics.)
5. *5-Day Look-Ahead* — one day per line, format =Fri 12:= / =Mon 15:=, including clear days marked =clear=. Built by Phase 1's forward scan with the invite quick-read and decline gate applied.
@@ -192,6 +192,7 @@ Pull every source in a *single batch of parallel tool calls*:
5. Pull the project tracker's view of Craig's plate (assigned issues, items in review, blocked items) where the project has one.
6. List + read the *previous* prep doc. Glob =daily-prep/*-daily-prep.org=, sort by date, take the file *before* the one the root =daily-prep.org= symlink resolves to. If the symlink doesn't resolve yet, take the most recent file. The standup lookback anchors on this file's date.
7. Read the most recent =.ai/sessions/= summary (for the standup brief's lookback).
+8. Run =.ai/scripts/upcoming_birthdays.py= (reads =~/sync/org/contacts.org=). Its block feeds the Heads-Up birthday line — contacts carry birthdays the calendar doesn't, and anything inside 7 days comes back flagged so a gift/plan gets prompted.
This fetch *is* the live calendar read for build time. In Update mode, re-run the calendar fetches — never reuse the build-time snapshot.
@@ -239,7 +240,7 @@ Assemble priorities from both sources; no mid-flow confirmation (the gate handle
*** Sub-step 3b: Triage external sources (delegate to triage-intake.org)
-Don't scan email / Slack / tracker / PRs inline. Run the =triage-intake.org= engine (if the mode's freshness check already ran one this hour, use its synthesis). It classifies everything new against the four-bucket model, writes every Action item into =todo.org= as its own =:quick:reactive:= task, and executes the routine actions on confirmation. Source coverage comes from its Phase 0 plugin load — both =.ai/workflows/triage-intake.*.org= (general) and =.ai/project-workflows/triage-intake.*.org= (project-specific). A missing source is a missing plugin, not a daily-prep regression.
+Don't scan email / Slack / tracker / PRs inline. Run the =triage-intake.org= engine (if the mode's freshness check already ran one this hour, use its synthesis). It surfaces the three-section digest (==TASKS== / ==FYI== / ==MISC==) and then closes by default per its Phase D — filing every TASKS item into =todo.org= as its own =:quick:reactive:= task and running the routine mail hygiene without itemized confirmation. Source coverage comes from its Phase 0 plugin load — both =.ai/workflows/triage-intake.*.org= (general) and =.ai/project-workflows/triage-intake.*.org= (project-specific). A missing source is a missing plugin, not a daily-prep regression.
*** Sub-step 3c: Build the entries
diff --git a/.ai/workflows/delete-calendar-event.org b/.ai/workflows/delete-calendar-event.org
index 5bb92a1..7de0086 100644
--- a/.ai/workflows/delete-calendar-event.org
+++ b/.ai/workflows/delete-calendar-event.org
@@ -1,5 +1,5 @@
#+TITLE: Delete Calendar Event Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-01
* Overview
diff --git a/.ai/workflows/edit-calendar-event.org b/.ai/workflows/edit-calendar-event.org
index 662f0b4..27a9dd3 100644
--- a/.ai/workflows/edit-calendar-event.org
+++ b/.ai/workflows/edit-calendar-event.org
@@ -1,5 +1,5 @@
#+TITLE: Edit Calendar Event Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-01
* Overview
diff --git a/.ai/workflows/email-assembly.org b/.ai/workflows/email-assembly.org
index 003459c..699dbc0 100644
--- a/.ai/workflows/email-assembly.org
+++ b/.ai/workflows/email-assembly.org
@@ -1,5 +1,5 @@
#+TITLE: Email Assembly Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-01-29
* Overview
diff --git a/.ai/workflows/extract-email.org b/.ai/workflows/extract-email.org
index 3a70bea..c68bafe 100644
--- a/.ai/workflows/extract-email.org
+++ b/.ai/workflows/extract-email.org
@@ -1,5 +1,5 @@
#+TITLE: Extract Email Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-06
* Overview
diff --git a/.ai/workflows/find-email.org b/.ai/workflows/find-email.org
index 0ef9615..d71ed3e 100644
--- a/.ai/workflows/find-email.org
+++ b/.ai/workflows/find-email.org
@@ -1,5 +1,5 @@
#+TITLE: Find Email Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-01
* Overview
diff --git a/.ai/workflows/first-session.org b/.ai/workflows/first-session.org
index 60118a2..147026f 100644
--- a/.ai/workflows/first-session.org
+++ b/.ai/workflows/first-session.org
@@ -1,5 +1,5 @@
#+TITLE: First Session Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
Run this workflow on the first Claude Code session for a new
project. It establishes the git/.ai policy, orients Claude to the
diff --git a/.ai/workflows/flashcard-review.org b/.ai/workflows/flashcard-review.org
index 31027b3..09af348 100644
--- a/.ai/workflows/flashcard-review.org
+++ b/.ai/workflows/flashcard-review.org
@@ -1,5 +1,5 @@
#+TITLE: Drill Deck Review Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-30
* Overview
diff --git a/.ai/workflows/helper-mode.org b/.ai/workflows/helper-mode.org
new file mode 100644
index 0000000..a6acfa7
--- /dev/null
+++ b/.ai/workflows/helper-mode.org
@@ -0,0 +1,101 @@
+#+TITLE: Helper Mode Workflow
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-06-15
+
+* Overview
+
+The role contract for a *helper instance*: a second Claude session running in the same project as a live primary session, spawned to look something up or make a scoped task update while the primary keeps working. This file is the single canonical home of the helper rules. [[file:../protocols.org][protocols.org]] carries a one-paragraph pointer here; the spawn paths and the wrap-up reference this file rather than restating it.
+
+A helper is not a subagent. When the work fits a dispatched subagent (the Agent tool) — a bounded lookup or analysis the primary folds back into its own context — use that instead; no second session exists and none of this applies. A helper is for interactive, long-lived parallel work Craig drives himself in a second terminal.
+
+The governing fact behind every rule below: the session-context split isolates each agent's /session state/ (=.ai/session-context.d/<id>.org=), but everything else in the project — =todo.org=, =.ai/notes.org=, =inbox/=, docs, the git index — is shared mutable state. Two Edit-tool writers on one org file lose updates silently: both read, both write, last write wins. The helper's job is to stay useful without ever being the second writer that clobbers the primary.
+
+* When to Use This Workflow
+
+No operator trigger phrase. A helper reaches this contract one of three ways:
+
+- The =ai --helper= launcher routes here after the roster confirms a live agent (the deterministic path).
+- Startup's roster check finds the session is not alone and routes here instead of running normal startup (the safety net for a raw =claude= launch).
+- An explicit "you are a helper, follow helper-mode.org" instruction (the manual fallback).
+
+If none of those applies — the roster shows the session is alone — this is a primary session. Run normal [[file:startup.org][startup.org]], not this.
+
+* Identity
+
+A helper is =helper-<rand4>= (four random hex/alphanumeric characters, e.g. =helper-a83f=).
+
+- If the launcher exported =AI_AGENT_ID=, use it. Otherwise self-assign =helper-<rand4>= now.
+- Record the chosen id as the *first line* of the session-context file, so it survives across tool calls (shell state does not). The id lives in the file, never in ambient shell state.
+- The active session-context path is =.ai/session-context.d/<id>.org=. Resolve it by prefixing the id explicitly wherever a script consumes =AI_AGENT_ID=:
+
+ #+begin_src bash
+ AI_AGENT_ID=<id> .ai/scripts/session-context-path
+ #+end_src
+
+ The id must be unique per run; =helper-<rand4>= and the launcher's epoch-tailed id both satisfy that (see protocols.org "Agent-scoped path").
+
+* Read/Write Contract
+
+Four tiers, by how much coordination the write needs.
+
+** Always safe
+
+Any read, anywhere in the project. Writes to the helper's own =.ai/session-context.d/<id>.org=.
+
+** Safe by discipline — scoped task updates (the case Craig named)
+
+Scoped writes to shared org files (=todo.org=, =.ai/notes.org=) are allowed under four rules, together:
+
+1. Re-read the file region immediately before each edit.
+2. Anchor the edit on a unique heading.
+3. Scope each edit to that single heading's subtree.
+4. Never reflow, restructure, or sweep the file.
+
+Appending a new =**= task at a section's end and editing one existing task's body or state both qualify. The race window collapses to seconds, and a collision corrupts one heading rather than the whole file. Log the intended edit first (see Write-ahead journal below).
+
+** Primary-only — never as a helper
+
+- File-wide passes: =todo-cleanup.el=, =lint-org.el=, =wrap-org-table.el=, archive sweeps.
+- Inbox processing and template sync.
+- ALL git mutation: commit, push, pull, stash. Two committers in one worktree contend on the index lock and interleave staging.
+- Startup's Phase A.0 pulls and the =.ai/= rsync. The primary already did them, and a concurrent pull-under-edit is exactly the race the startup guards exist to prevent.
+- Memory writes. =MEMORY.md= is a shared read-modify-write index with no heading anchors, so it has the lost-update shape of =todo.org= with none of the scoped-edit protection.
+
+The git ban is concurrency-scoped. /Helper wrap-up/ below lifts it for exactly one case: an orphaned helper that finds itself alone.
+
+** Escalation
+
+Anything the contract blocks gets reported to Craig, or — for a cross-project handoff — routed through =inbox-send= to the owning project's =inbox/=. The helper leaves its tree changes for the primary's next commit, or describes them in a note to Craig.
+
+* Data-Integrity Rules
+
+The scoped-edit discipline covers helper-vs-primary edits on /different/ headings. Four loss windows remain. They matter doubly in a consolidated project where one =todo.org= carries every task and corruption has maximal blast radius.
+
+1. *Primary file-wide passes vs a live helper.* A whole-file rewrite run while a helper is mid-edit clobbers the helper's just-written change. Enforced primary-side (the live-helper gate before any hygiene pass), but the helper's part is to keep its own session-context file current so the primary's gate can see it is live.
+2. *A new primary starting while a helper runs.* The helper's uncommitted edits make the tree dirty; Phase A.0 already skips pulls on a dirty tree, and startup surfaces live =session-context.d/= files so the new primary knows /why/ the tree is dirty rather than treating it as mess to resolve.
+3. *Write-ahead edit journal.* Before applying any shared-file edit, log it — file, heading, one-line intent — to the helper's own session-context file. This tightens the Session Log discipline to log-before-write for shared files specifically, so after a crash the journal shows which edits landed.
+4. *The memory dir.* Helpers don't write memory at all. Candidate memories go into the helper's session log; the primary (or its wrap-up promotion check) writes them.
+
+One collision nit: =inbox-send= filenames carry minute-resolution timestamps, so a helper and primary sending to the same target in the same minute with the same slug would collide. Helper-originated sends include the agent id in the slug.
+
+* Light Startup
+
+A helper does not run normal startup. It runs a light version:
+
+1. Self-assign or adopt the identity (above) and create =.ai/session-context.d/<id>.org= with the id on the first line.
+2. Read [[file:../protocols.org][protocols.org]] and this file. Read =.ai/notes.org= for project context if useful.
+3. Do NOT run Phase A.0 pulls, =make install=, or the =.ai/= rsync — the primary owns those.
+4. Do NOT process the inbox — primary-only.
+5. Begin the work Craig spawned the helper for.
+
+* Helper Wrap-Up
+
+When the helper's work is done:
+
+1. Re-run the roster (=.ai/scripts/agent-roster=) to learn whether a primary is still live.
+2. *Primary still live (the normal case):* finalize the Summary in the helper's own =.ai/session-context.d/<id>.org=, archive it to =.ai/sessions/YYYY-MM-DD-HH-MM-<id>-<description>.org=, and stop. Do NOT commit, push, or run hygiene — the primary's next commit picks up the archived file and any scoped edits the helper left in the tree.
+3. *Orphaned helper (roster shows the helper is now alone):* the primary already exited, so the helper assumes full closing duties — the git ban lifts because the concurrency that justified it is gone. Commit and push the tree (including the helper's own edits, which would otherwise strand as a dirty tree), per the normal wrap-up flow in [[file:wrap-it-up.org][wrap-it-up.org]].
+
+* Status
+
+Phase 1.5 of the generic-agent-runtime spec. This contract is the canonical home; the spawn paths (=ai --helper=, startup's roster branch) and the [[file:wrap-it-up.org][wrap-it-up.org]] helper branch route here. Those wiring pieces ship behind the spec's bats-then-drills-then-pilot gate and are not yet live; until then, the manual "you are a helper" instruction is how a session adopts this contract.
diff --git a/.ai/workflows/inbox.org b/.ai/workflows/inbox.org
new file mode 100644
index 0000000..6faa20f
--- /dev/null
+++ b/.ai/workflows/inbox.org
@@ -0,0 +1,524 @@
+#+TITLE: Inbox Workflow (Engine)
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-06-23
+
+* Overview
+
+One engine for the project's inbox surfaces. Inbox items are *ideas to evaluate*, not orders to execute — each is a proposal that earns a place in =todo.org= or git history only when it passes the value gate. The engine holds the shared disposition machinery once: the three-question value gate, the skeptical review, the disposition ladder, the reply-to-sender discipline, the capture-guard before a roam write, and the priority-scheme check. Each *mode* is a thin section that names which surface it reads, how it enters and exits, and which core steps it runs.
+
+Two surfaces feed a project, and there is a recurring check over the second:
+
+1. *Project-local =inbox/= dir* — handoffs from other projects (via =inbox-send=), from scripts, and from Craig (typed directives saved as files). Handled by *process mode*; watched on a cadence by *monitor mode*.
+2. *Global roam inbox* (=~/org/roam/inbox.org=) — Craig's cross-project GTD capture, one shared file every project can see. Handled by *roam mode*, which claims only the items this project owns. *Auto inbox zero* runs roam mode on a recurring interactive loop.
+
+A *third* surface — external accounts (email / calendar / PRs) — is a different domain and stays in its own engine: =triage-intake.org= and its source plugins are *not* part of this engine. "Deal with my inbox dirs" is here; "what's new across my accounts" is there.
+
+*Two altitudes.* For the user, the trigger phrase picks the mode and the phrases are unchanged (see When to Use). For the implementer, this is one file: the core sections are written once, and each mode references them by name ("run the value gate (core §1) on each item") rather than restating them.
+
+* When to Use This Workflow
+
+The trigger phrase selects the mode. Every phrase below still works; it now routes to a mode of this engine.
+
+** Process mode — the local =inbox/= dir
+
+- "process inbox" / "process the inbox"
+- "handle the inbox"
+- "what's in inbox" / "what's in the inbox"
+- "let's clear the inbox" / "let's process the inbox items"
+
+Auto-invocation: startup Phase C delegates here when the local inbox is non-empty — don't ask, just run it.
+
+** Monitor mode — process mode on a cadence
+
+- "monitor the inbox" / "watch the inbox" — *the defined meaning:* one process pass now, then loop every 15 minutes (see the Monitor mode Cadence section). The phrase *is* the loop, not an opt-in extra.
+- "respond to the handoffs" / "handle the handoffs" — a single pass now, no loop.
+
+Ambient (always on, even with no loop running): the =inbox-status= task-boundary check (Monitor mode).
+
+** Roam mode — the global roam inbox
+
+- "inbox zero" / "empty the inbox" / "process the roam inbox" / "triage my roam inbox"
+
+Called read-only from startup (count + offer) and as a wrap-up Step 3 sub-step.
+
+** Auto inbox zero — recurring interactive roam check
+
+- "auto inbox zero"
+
+Match this before "inbox zero" — the auto phrase contains the roam phrase as a substring, so the longer match wins. Starts a recurring =/loop=-driven roam-mode pass; see the Auto inbox zero mode.
+
+** Boundary
+
+Do *not* invoke this engine for an inbox item that is clearly out-of-scope for the project — that is a cross-project routing problem, handled per the cross-project boundary rule in =protocols.org=. And do not invoke it for external-account triage ("what's new in email/cal/PRs") — that is =triage-intake.org=.
+
+* Core §1 — The value gate
+
+Every inbox item (local or roam) passes through three questions. One *yes* is enough to accept.
+
+1. *Does it advance an existing TODO?* Look up by topic in =todo.org='s open work. If the item extends a filed task, fold it in. If it implements a filed task, do the work.
+2. *Does it improve how the project works?* Architecture cleanup, workflow refinement, tooling, rule hygiene, drift detection — anything that makes the project itself more effective.
+3. *Does it serve the project's stated mission?* Read =notes.org= *Project-Specific Context* if the mission isn't obvious from the working directory and current task. The item should advance that mission, not orbit it.
+
+Three *no*s means reject. The rejection isn't lazy — an idea that doesn't help any current task, doesn't improve the system, and doesn't serve the mission is genuine noise, and accepting it inflates =todo.org= without payoff.
+
+* Core §2 — The skeptical review
+
+The value gate decides whether an item is worth taking. This review decides whether what it proposes is *right*, *complete*, and *as simple as it should be*. Run it on every task and file that arrives — not only shared-asset change proposals. Pure FYIs and replies that ask for nothing skip it.
+
+Approach the file with curiosity and skepticism. Work through, in writing — the core pass on every item:
+
+1. Is the request actually right — does it do what it claims, and is the claim correct for this project?
+2. Is it complete, or does it leave a gap — an unhandled case, a missing step, an untested path?
+3. Should it be simpler?
+4. Can it be enhanced to be more effective than as proposed?
+5. Does it conflict with any existing instruction — workflows, skills, rules, protocols, CLAUDE.md?
+
+When the item proposes a change to *shared assets* — template workflows, rules, skills, scripts, anything synced to consuming projects — or to a substantive convention, add the cross-project battery. It arrived from one project's context; you're evaluating it for all of them:
+
+6. Does this make sense for *all* consuming projects, or just the sender's situation?
+7. How does it change a common activity Craig performs — better, worse, or differently than the sender assumed?
+8. Plus at least three more questions specific to this change — what breaks for artifacts already using the old shape, what tooling interacts with it, what's underspecified, what the sender's worked example doesn't exercise.
+
+Output: a short summary of the thinking and a recommendation (do it / do it with named changes / file / reject). For shared-asset and convention changes the recommendation is surfaced to Craig for approval before applying; for ordinary tasks and files it feeds the act-vs-file and no-approvals-execute decision (Monitor mode).
+
+** In a no-approvals session: shared-asset changes defer and stage
+
+Shared-asset and convention changes still don't self-apply when Craig has put the session in no-approvals mode — they need his decision, so they fail the *solo* test in Monitor mode's executing-in-no-approvals criteria. Ordinary tasks and files that pass the review and are quick + solo execute under that criteria instead; this defer-and-stage path is for the shared-asset and convention changes that don't qualify. Run the review, prepare the edits in =working/<task-slug>/= (a patch file or the worked-out diff), file a =[#B]= VERIFY carrying the decision package, and reply to the sender that it's parked. The sender's local stopgap (per =cross-project.md='s propagation process) means the delay costs nothing — the canonical update is about durability, not speed.
+
+Wording-only fixes — no consuming project acts differently — may proceed even then, logged in the session log.
+
+The VERIFY shape (top-level, =[#B]= so startup's A/B surfacing catches it; no =SCHEDULED= unless the proposal names a real deadline):
+
+#+begin_example
+** VERIFY [#B] Parked: <proposal topic> (from <sender>)
+What arrived: <one line — what the handoff proposes>.
+Recommendation: <accept as-is / accept with changes / reject> — <2-3 line
+skeptical-review summary: what's right, what to change, what was checked>.
+Prepared diff: [[file:working/<slug>/proposed.diff]] — apply is mechanical on
+your go.
+Say "approve the parked <topic>" (or adjust / reject) and it gets applied.
+#+end_example
+
+The full question-battery answers live in the session log and the =working/= dir, not the task body — the body carries the conclusion, with the trail one link away.
+
+* Core §3 — The disposition ladder
+
+Every item that clears the value gate gets one disposition. The first six are the per-item outcomes; *park* is the no-approvals shared-asset path from core §2.
+
+** Implement now
+Small, scoped, clear, no design call required. The work is the disposition. Do the work, commit per the project's commit flow, delete the inbox file. The commit message references the inbox item by filename so the provenance lands in =git log=.
+
+** Fold into existing TODO
+The item extends a task already filed. Update the parent TODO's body with a dated reconciliation sub-entry per =todo-format.md= (=*** YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <what landed>=). Move substantive content to =docs/design/<date>-<topic>.<ext>= if it's worth keeping; reference from the TODO body. Delete the inbox file.
+
+** File as TODO
+Substantive but waits, or needs design/triage before implementation. Add the TODO under =* <Project> Open Work= with priority + tags per the priority-scheme check (core §6). Body summarizes the proposal and links the inbox content if it's been moved to =docs/design/=. Delete the inbox file (or move it to =docs/design/= first if the content survives).
+
+*Route-candidate marking (feeds the wrap-up router).* After filing, check whether the keeper's inferred home is a different project:
+
+#+begin_src bash
+python3 .ai/scripts/route_recommend.py --item "<the keeper's heading + body text>" --exclude "$(basename "$PWD")"
+#+end_src
+
+On a =<destination>\tstrong= or =<destination>\tweak= result, stamp the new TODO's property drawer with =:ROUTE_CANDIDATE: <destination>= (create the drawer if the task has none). A =none= result stamps nothing, and a local keeper stays unstamped. The marker is the wrap-up router's entire candidate set — =wrap-it-up.org= Step 3 surfaces exactly the =:ROUTE_CANDIDATE:=-tagged tasks and offers to deliver each to its destination's inbox, never scanning the standing backlog. Stamping is cheap and reversible (the router's skip leaves the task in place; a wrong marker is one property line to delete), so prefer stamping on any plausible match — the human reviews the batch at wrap time.
+
+*Blocking-dependency handoff.* A special shape: another project sends a note that *this* project's work is blocking one of theirs ("your task X is blocked on us — we need Y"). File or link the owning task, tag it =:blocker:=, and name the requesting project in the body (see the cross-project dependency convention in =todo-format.md=). The =:blocker:= tag makes =open-tasks.org= surface that task *first*, since clearing it unblocks the other project. Dedup against an existing task rather than filing a duplicate. When the work later lands, drop =:blocker:= and notify the waiting project (=inbox-send <their-project> --text "Delivered: <what> — you're unblocked."=) so it can lift its own =:blocked:=.
+
+** Defer
+Rename in place to =inbox/PROCESSED-<original-filename>= and add a brief comment line at the top: =# Deferred YYYY-MM-DD: <condition>=. Don't accumulate deferred items indefinitely — sweep them on a future process pass when the condition is met or the deferral has aged out.
+
+** Reject — by source
+- *From Craig* — push back honestly in chat. State why you won't implement; offer the conditions under which you would, if any. The inbox file stays until Craig confirms — override re-enters as accept, acknowledgment deletes the file. Don't theatre the pushback: if you don't genuinely think Craig is wrong, just do the work.
+- *From another project (handoff)* — write a response file at =/tmp/inbox-response-<topic>.org=: a heading naming the original handoff and date, one paragraph on the rejection rationale (*which* value-gate question failed and why), one paragraph on the condition under which you'd reconsider (or "never, this misreads the project's mission" if that's the truth). Deliver via =inbox-send <sender> --file /tmp/inbox-response-<topic>.org=. Delete the local inbox file after the response lands. Silent rejection on a handoff trains the sender to escalate around the channel — always close the loop.
+- *From a script or automated system* — just delete. No notification.
+
+** Park (skeptical review in a no-approvals session)
+Move the proposal file into =working/<task-slug>/= alongside the prepared diff, file the =[#B]= VERIFY per core §2, reply to the sender that it's parked for Craig's review, and delete the inbox file. On Craig's approval the apply is mechanical: apply the prepared edits, run the normal verify-and-publish flow, close the parked =**= VERIFY per =todo-format.md= (a top-level VERIFY resolves to =DONE= + =CLOSED:=, not a dated header), and send the acceptance reply. On rejection, the reject-from-another-project flow above runs unchanged.
+
+* Core §4 — Reply-to-sender discipline
+
+A handoff came from another project's agent (or the user). Close the loop:
+
+- *Accepted and acted on* — send a confirmation to the sender via =inbox-send <sender> --text "..."=, naming what landed and the commit, so they're not left guessing (they can't see this project's git log). =inbox-send= excludes the current project as a target, so a self-sourced item is handled in-session, not sent.
+- *Accepted and filed* — a short confirmation that it's filed and where, so the sender knows it wasn't dropped.
+- *Rejected* — always state the why (which value-gate question failed), per the reject-by-source ladder (core §3).
+
+Cross-project boundary: never act on a file under another project's =.ai/= scope from here — route it back as a handoff (see =cross-project.md=).
+
+* Core §5 — Capture-guard before a roam write
+
+Before *any* read-modify-write of =~/org/roam/inbox.org=, run the capture-guard. This runs first because the Phase D edit rewrites the file on disk, and editing underneath a live capture wedges it just as a stray hand edit would.
+
+*Wait-and-retry, not bounce.* Use the poll mode so a *transient* capture clears itself instead of immediately kicking the work back to the caller:
+
+#+begin_src bash
+.ai/scripts/capture-guard --wait "$HOME/org/roam/inbox.org"
+#+end_src
+
+An org capture is usually only a few seconds of mid-finalize state, so =--wait= (default 30s, re-checking every ~10s) returns the instant it clears and reports blocked only if it's *still* open at the deadline. The common case — nothing capturing — returns instantly without sleeping. This is the fix for the guard bouncing a caller over a capture that would have cleared on its own a moment later. (The bare single-shot form — no =--wait= — stays available for a caller that genuinely must not block.)
+
+- *Exit 0* → no live capture, or it cleared during the wait (or no reachable Emacs). Proceed with the edit.
+- *Exit 1* → an indirect org-capture buffer is *still* cloned from the roam inbox after the wait (the script prints the offending buffer name). Editing underneath it would leave the capture pointing at stale state and unable to finalize with =C-c C-c= (see =emacs.md=). Only now does the per-caller fallback fire:
+ - *On-demand / interactive run* → stop and surface: "You have a live org-capture session open against the roam inbox (=<buffer>=) — finalize it (=C-c C-c=) or abort it (=C-c C-k=) and I'll continue." Re-run the guard and resume once it returns clean.
+ - *Auto inbox zero (=/loop=) cycle* → don't surface or wait further; defer the roam reconcile to the next cycle, which is itself the retry at loop cadence. The items were already filed in Phase C, so the next cycle's Phase C status-check drops the duplicates and its Phase D removes them. Note one line: "roam reconcile deferred — a capture is still open; next cycle catches it."
+ - *Wrap-up sub-step* → don't block the wrap. Skip the roam reconcile for this run and surface one line: "Skipped roam-inbox reconcile — a live org-capture is open against it; claimed items stay and get caught next run." The items were already filed into =todo.org= in roam mode Phase C, so the next roam run's Phase C status-check drops the duplicates and its Phase D removes them — the skip self-heals.
+
+*The roam-write lock (around the Phase D edit).* Capture-guard protects against a live *human* capture; the roam-write lock protects against a concurrent *agent* writer (a sentry inbox pass, a KB promotion) editing =~/org/roam= at the same time. Acquire it after the capture-guard clears and release it after the edit-and-trigger, so the two guards nest — capture-guard underneath, the agent lock around the write:
+
+#+begin_src bash
+if [ -x .ai/scripts/agent-lock ]; then
+ .ai/scripts/agent-lock acquire roam-write --wait || { echo "roam-write busy; deferring roam reconcile" >&2; exit 1; }
+fi
+# capture-guard (above), then the Phase D read-modify-write of ~/org/roam/inbox.org,
+# then trigger the sync — roam-sync stays the only committer:
+systemctl --user start roam-sync.service
+[ -x .ai/scripts/agent-lock ] && .ai/scripts/agent-lock release roam-write
+#+end_src
+
+Degrade gracefully when =agent-lock= isn't installed (an older checkout mid-sync): the write proceeds unlocked, today's behavior. A *present* helper reporting the lock busy after its bounded wait defers the roam reconcile (the auto-loop and wrap-up paths already defer-and-retry per the fallback list above); an *absent* helper never blocks it.
+
+* Core §6 — Priority-scheme check
+
+This gates filing whenever there are accept-and-file items. Check whether =todo.org= has a top-of-file priority scheme (an explicit legend defining =[#A]= through =[#D]= semantics and mandatory/optional tag conventions — a =* <Project> Priority Scheme= section or similar).
+
+- *Scheme present* — file new TODOs per the scheme. Every TODO gets a priority cookie matching the legend's rules, the mandatory type tag, and any applicable effort/autonomy tags.
+- *Scheme absent* — surface one sentence: "This project has no priority scheme. We should adopt one before filing the new TODOs from this inbox pass — want me to propose one based on the rulesets scheme?" If Craig says yes, do that first (the =/research-priority-scheme= research subagent pattern in rulesets is the reference). If Craig says no, file the TODOs without grading but flag in the commit message that they're un-prioritized pending a scheme.
+
+The point is to avoid adding ungraded =TODO= entries to a project that's never agreed on what =[#A]= means.
+
+* Mode: process
+
+Reads the project-local =inbox/= dir. Entry: a trigger phrase, or startup Phase C on a non-empty inbox. Exit: inbox empty (excluding =.gitkeep= and intentional =PROCESSED-*=), session log updated, =:LAST_INBOX_PROCESS:= stamped.
+
+** Phase A — Inventory (one parallel batch)
+
+Issue these reads in one parallel batch:
+
+1. List =inbox/= excluding =.gitkeep= and =PROCESSED-*= prefixes (use =\ls -la inbox/= per the protocols.org exa-alias note).
+2. Read =notes.org= *Project-Specific Context* if mission isn't already loaded in the session.
+3. Read =todo.org='s top-of-file priority scheme if present.
+
+For each inbox file, parse the filename for sender. Two common patterns:
+
+- =YYYY-MM-DD-HHMM-from-<sender>-<topic>.<ext>= — from another project via =inbox-send=.
+- =<topic>.org= — typically from Craig directly, or from a script.
+
+Note the file type. =.eml= files need the extract script (not raw =Read=):
+
+#+begin_src bash
+# View mode
+python3 .ai/scripts/eml-view-and-extract-attachments.py inbox/<file>.eml
+
+# Pipeline mode (extract attachments to a directory)
+python3 .ai/scripts/eml-view-and-extract-attachments.py inbox/<file>.eml --output-dir assets/<target>/
+#+end_src
+
+Everything else, read directly.
+
+** Phase B — Evaluate each item
+
+For each inbox file:
+
+1. *Read it.* Full read for substantive proposals (org files with TODO entries, design notes, multi-section docs); skim short FYIs and one-liner asks.
+2. *Identify the shape.* Instruction, question, proposal, FYI, or handoff — shapes guide disposition.
+3. *Apply the value gate* (core §1). One yes → candidate accept. Three nos → candidate reject.
+4. *Run the skeptical review* (core §2) on the item before classifying — the core pass on every accepted task and file, plus the cross-project battery when it proposes a shared-asset or convention change. Its summary + recommendation rides along to Phase C; in a no-approvals session it gates whether the item self-applies (quick + solo + agreed, per Monitor mode) or, for shared-asset and convention changes, defers and stages.
+5. *Within accept, classify* by the disposition ladder (core §3): implement now / fold into existing TODO / file as TODO.
+6. *Within reject, classify by source* (core §3): from Craig / from another project / from a script.
+
+** Phase B.1 — Priority-scheme check
+
+Run core §6. This gates Phase C filing when there are accept-and-file items.
+
+** Phase C — Surface dispositions
+
+Numbered options inline per =interaction.md= (no popup). Recommendation at item 1.
+
+Batch trivial items (one-line rejections of script noise, obvious file-as-TODO accepts where the scheme is already settled) into a single confirm-all prompt. Walk substantive items one at a time so the decision is visible.
+
+Per-item template:
+
+#+begin_example
+<filename> from <sender>: <one-line summary>
+Value-gate read: <yes/no on each of the three questions, one phrase each>
+Disposition recommendation: <implement / fold into <TODO> / file [#X] :tags: / reject>
+
+1. <recommendation as item 1>
+2. <alternative>
+3. Defer — leave in inbox under PROCESSED-<topic>.<ext> until <condition>
+4. Something else
+#+end_example
+
+For items that went through the skeptical review, the surfaced disposition includes its summary + recommendation, and approval here is what authorizes the apply. In a no-approvals session those items are reported as parked (the =[#B]= VERIFY) rather than surfaced for live approval.
+
+For pure FYIs that need no action, surface as a single line and recommend delete-with-acknowledgment.
+
+** Phase D — Apply
+
+Apply each disposition per the ladder (core §3). The flow is autonomous past Craig's Phase C approval.
+
+** Phase E — Close out
+
+Verify =inbox/= is empty (excluding =.gitkeep= and any intentional =PROCESSED-*= files). Run =\ls -la inbox/= and confirm.
+
+Update the session log per =protocols.org= with one short paragraph: count processed, count accepted (implement/fold/file split), count rejected (Craig/handoff/script split), and the commit SHA if a commit landed.
+
+Stamp =:LAST_INBOX_PROCESS:= in =notes.org='s *Workflow State* section if it exists, so future workflows that gate on freshness can read it. Same format as =:LAST_AUDIT:= (=YYYY-MM-DD=).
+
+* Mode: monitor
+
+Process mode on a cadence. This is the *when, how-often, and act-vs-file* layer; the per-item disposition mechanics are the core sections, run via process mode — not restated here. Monitor decides *that* an item gets handled and *how I respond*; the core decides *what disposition* each item gets.
+
+The gap it closes: handoffs that arrive mid-session used to sit unseen until the user asked or the next startup ran. A handoff the sender can't see being handled trains them to escalate around the inbox channel.
+
+** Preconditions — before starting
+
+Never begin monitoring on a dirty worktree or a failing test suite. A dirty tree means the auto-commit at the end of an executed item sweeps up unrelated changes; a red suite means you can't tell whether the monitor broke something. At the start:
+
+1. =git status --porcelain --untracked-files=no= is empty (no tracked modifications). Untracked and gitignored files never block — an inbox drop is exactly what this mode processes, and a scratch file is none of its business (the template-freshness policy in =startup.org= Phase A.0). The tracked-only gate is safe because the per-item commit stages its files explicitly (=commits.md=: only intended changes staged) — never =git add -A=, which would sweep untracked files and is the failure this gate guards against.
+2. A full test run is all green (=make test= here, or the project's full-suite command).
+
+If *dirty*: offer to commit the pending changes in discrete, logical batches before starting. If *red*: offer to investigate the failures first. Surface the blocker with inline numbered options per =interaction.md= and wait — monitoring does not start until the tree is clean and the suite is green.
+
+** Cadence — how often to check
+
+*"Monitor the inbox" = run now, then loop every 15 minutes.* Do one process pass over any pending handoffs immediately, then start the loop:
+
+#+begin_src
+/loop 15m check the inbox with inbox-status and run inbox.org process mode over any pending handoffs
+#+end_src
+
+Each firing runs the cheap =inbox-status= check first and only does a full process pass when items are pending. The loop is the monitoring; it runs until Craig stops it or the session ends. Honor the Preconditions gate before the first pass and the Close-out gate when the loop stops.
+
+*Ambient task-boundary check (always on, even without a loop).* After finishing a unit of work, before reporting back or asking "what's next," run the cheap status check:
+
+#+begin_src bash
+.ai/scripts/inbox-status -q
+#+end_src
+
+Exit 1 means handoffs are pending — list them (drop =-q=) and run process mode. Exit 0 means clean; say nothing. This is one =find=; it costs nothing to run often, and it's the fix for handoffs piling up unseen during long sessions.
+
+*Startup and wrap-up already cover their ends.* Startup Phase C processes a non-empty inbox; the wrap-up sanity check refuses to wrap with unprocessed handoffs. The task-boundary cadence fills the middle.
+
+*Mid-task arrivals.* If a handoff lands while you're mid-task and it's urgent (blocks the current work, or is time-sensitive), surface it right away. Otherwise batch it to the next task boundary so the current work isn't thrashed.
+
+** The act-vs-file decision
+
+Every accepted handoff (one that clears the value gate) is then either acted on now or filed as a task.
+
+*Act immediately — and just do it, no asking — when all of these hold:*
+- *Clear* — the action is unambiguous; no design decision or option-choice is needed.
+- *Bounded* — small, finishable this session, ideally a tight file set.
+- *Low-risk and verifiable now* — not a risky change to load-bearing infra (or trivially revertible), and testable/lintable this session.
+- *In-scope and safe* — within this project, not destructive or outward-facing without confirmation, not across a project boundary.
+- *Cheaper than deferring* — doing it now costs less than filing plus re-triaging later.
+
+When you decide to act, queue the work and do it. Don't ask first.
+
+*Exception:* a proposal to change a shared asset (template workflow, rule, skill, synced script) or a substantive convention never qualifies for silent act-now, however clear and bounded it looks — it routes through the skeptical review (core §2), which carries its own approval (or, in a no-approvals session, park) step.
+
+*File a task when any of these hold:*
+- It needs a judgment call, a design decision, or an option the user would pick.
+- It's large, multi-session, or sprawls across many files.
+- It's blocked (a dependency, an external thing, the user is away).
+- It's risky enough to want the user's eyes before it lands.
+- It's off the session's active goal and acting now would derail it (file and keep going, unless it's urgent).
+
+When you decide to file, *ask first* — inline numbered options per =interaction.md=, with *filing as option 1 (the recommendation)* and *"do it now" as option 2*:
+
+#+begin_example
+<handoff> wants <X>. My read: file it (needs <reason>).
+
+1. File as a TODO ([#?] :tags:) — Recommended
+2. Do it now instead
+3. Something else
+
+Pick a number.
+#+end_example
+
+*Always ask if you're unsure* which side of the line an item falls on. Decisiveness on clear act-now items is the point of the rule; the ask is for genuine ambiguity and for filing.
+
+** Executing in no-approvals mode
+
+When Craig has put the session in no-approvals mode, an accepted item may be implemented automatically — but only when all three of these hold:
+
+1. *Agreed* — you've run the value gate and the full skeptical review and concluded the change should be done, not merely that it's harmless.
+2. *Quick* — the whole implementation, including verification, is under ~15 minutes.
+3. *Solo* — you can carry it end to end without a decision from Craig. Manual verification you perform yourself is fine; needing Craig to choose an option, approve a design, or resolve an ambiguity is not.
+
+All three → implement it, verify, then commit and push at the end of that item (the Step 0 reconcile and pre-push check from the =publish= skill still run). Miss any one and it doesn't self-apply: a shared-asset or convention change needs Craig's decision, so it fails *solo* and routes to the defer-and-stage park (core §2 / core §3); an oversized item fails *quick* and gets filed.
+
+** Replying to handoffs
+
+Close the loop per the reply-to-sender discipline (core §4): confirm what landed (accepted-and-acted), confirm where it's filed (accepted-and-filed), or state the why (rejected).
+
+** The inbox-status script
+
+=.ai/scripts/inbox-status= lists unprocessed handoffs and exits nonzero when any are pending. Exclusions match the wrap-up sanity check (=.gitkeep=, =lint-followups.org=, =PROCESSED-*=). Exit 0 = clean, 1 = pending, 2 = no inbox/ or bad usage. Use =-q= for the count-only form the cadence check calls.
+
+** Close out — before finishing
+
+End the way it started: clean worktree, green suite. Before stopping the loop or reporting the pass done:
+
+1. Commit or revert every tracked modification left in the worktree — no tracked change remains uncommitted. Untracked files (unprocessed inbox drops, scratch) are not the monitor's to sweep.
+2. Run the full test suite once more and confirm all green.
+
+If either can't be satisfied — a half-done item, a failure introduced during the pass — surface it rather than leaving it. The next monitor run assumes a clean, green starting state (the Preconditions gate).
+
+* Mode: roam
+
+Reads the *global roam inbox* (=~/org/roam/inbox.org=), Craig's cross-project GTD capture: one shared file every project can see. This mode routes each roam item to the project that owns it. The current session claims only the items belonging to THIS project, files them into the project's =todo.org=, and removes them from the shared inbox. Everything it doesn't own stays.
+
+*Allowed from any project, work included.* Tidying the shared roam inbox is housekeeping on a shared resource, not a cross-project boundary crossing and not a durable KB-node write, so the =knowledge-base.md= work-denylist doesn't gate it (a sentry inbox-zero pass mis-parked the whole inbox as a boundary crossing from the work project on 2026-07-19 — the error this note closes). Reading roam and tidying its inbox are fine from work; only promoting a durable =agents/= node stays work-denylisted.
+
+The aspiration is inbox zero: after this mode runs, the current project's local handoff inbox has been processed (Phase A delegates to process mode) and the shared roam inbox no longer contains items explicitly owned by this project.
+
+This is distinct from the wrap-up inbox/transcript routing feature (which moves session-filed keepers between projects). This routes the shared roam capture file by ownership prefix.
+
+** Scope: single-destination (v1)
+
+Routes each item to its one owning project, identified by an explicit =<project>:= heading prefix. The multi-project domain-aware mode (guess the owner of every unprefixed item and empty the whole inbox in one run) is deferred — see "Deferred: domain-aware routing" at the end. v1 claims only what's prefixed for the current project, surfaces the rest, and never guesses.
+
+** Callers
+
+The steps live here so three callers reuse them:
+- *Startup* (read-only nudge) — count the items, identify which appear related to this project, surface both numbers, offer processing as one of the startup options. Never auto-files.
+- *Wrap-up* (Step 3 sub-step) — sweep items that belong here before the cleanup scripts, so imported tasks lint and ride the wrap commit.
+- *On demand* — the roam-mode trigger phrases.
+
+Each project touches the roam inbox at least twice a session this way: once at startup, once at wrap-up.
+
+** The ownership rule (the coordination primitive)
+
+The inbox is shared, so the mode must never let two projects fight over an item or let one grab another's. Ownership is by explicit prefix:
+
+- =<project>: ...= heading → owned by that project. The current project claims only items prefixed with its own identifier.
+- Prefixed for *another* project → leave untouched (cross-project boundary, =protocols.org=).
+- *No prefix* → unowned. Never auto-claim. Surface as candidates a human can claim or prefix.
+
+The prefix partition is what makes concurrent triage across projects safe: each project only ever removes its own items, so two sessions editing the inbox touch disjoint lines.
+
+*Resolving this project's identifier (v1).* Use the project root basename plus its common aliases (=.emacs.d= ↔ =emacs=, and the obvious ones: =rulesets=, =work=, =home=). A project may override the inferred set with an =:INBOX_PREFIX:= line in =notes.org='s *Workflow State* section when the basename is fragile. The explicit override is optional in v1; the durable multi-project resolution is part of the deferred domain-aware mode.
+
+** Phase A — Process the project-local inbox
+
+1. Check the project-local =inbox/= with =.ai/scripts/inbox-status -q=.
+2. If pending handoffs exist, run *process mode* before touching the roam inbox. Project handoffs are already addressed to this project, so they are higher-confidence and cheaper to clear than shared roam captures.
+3. If =inbox-status= reports no =inbox/= directory, note it and continue to the roam inbox. Some projects only participate in the shared roam capture flow.
+4. If process mode cannot finish because it needs Craig's decision, stop after surfacing that decision. Do not remove roam items in the same run; the project still does not have a clean inbox.
+
+** Phase B — Identify, count, and match roam items
+
+1. Resolve the current project's identifier and aliases (above).
+2. Read =~/org/roam/inbox.org=. If absent, silent no-op (the file lives only on machines with the roam clone).
+3. Bucket every item under the inbox heading:
+ - *claimed* — prefixed for this project
+ - *foreign* — prefixed for another project → leave
+ - *unowned* — no project prefix
+ - *empty* — a heading with no title and no body: just stars, optionally a =TODO=/keyword, and whitespace (e.g. =** =, =** TODO =, =*** TODO =). These are aborted or accidental captures, owned by nobody, and safe to delete regardless of project. A heading with any title text or any body content is never empty.
+4. *Summarize the scan* (Craig's requirement, every scan): report the total item count in the inbox, then the count that appears related to this project. "Appears related" is the union of claimed items (exact prefix) and any unowned item whose topic plainly concerns this project's domain (a content judgment, surfaced as a candidate, never auto-claimed). Foreign-prefixed items are not "related" — they belong to their owner. Note the empty count separately.
+5. If claimed, related-unowned, *and* empty are all absent, report the total and stop (the common case for most wraps). Empty entries on their own are enough to enter Phase D — the cleanup runs even when this project owns nothing else, since empties belong to nobody and removing them is what "check the inbox" should always do.
+
+** Phase C — File each claimed roam item into todo.org
+
+Apply the core disposition discipline against the project's =todo.org=; don't reinvent it:
+
+1. *Status check first.* Already done, or already a task in =todo.org=? → drop it, or fold into the existing task (dated sub-entry per =todo-format.md=). Don't duplicate.
+2. *Rewrite* to terse-heading + body per =todo-format.md=.
+3. *Priority + tags from THIS project's scheme* (core §6) — the legend at the top of its =todo.org=, tags from that scheme's allowed set only. The project expresses someday-maybe with =[#D]=; there's no special someday-maybe routing.
+4. *File* under the project's Open Work section.
+
+** Phase D — Reconcile the shared roam inbox
+
+The roam inbox lives in a git repo (=~/org/roam=, auto-synced every 15 minutes by the =roam-sync= timer). Craig captures into it constantly, so its working tree is dirty most of the time — which is exactly why this mode never runs =git pull= itself. A pull on a dirty tree fails, and that would block triage on nearly every run. Instead, edit the file and hand the git work to =roam-sync=, which already commits-first-then-rebases and so handles the dirty tree correctly.
+
+1. *Guard against a live org-capture session* — run the capture-guard in poll mode (=capture-guard --wait=, core §5) before the edit, so a transient capture clears itself rather than bouncing the run. On a still-blocked exit 1 the caller-specific fallback (interactive stop-and-surface / auto-loop defer-to-next-cycle / wrap-up skip-and-self-heal) is in core §5.
+2. *Remove the claimed items and the empty entries* from the working-tree file. Never touch foreign or unowned (titled) items. Empty entries (Phase B's =empty= bucket) are removed on every triage regardless of who would own a titled version, since an aborted capture belongs to nobody. The claimed-item removal and the empty sweep happen in the same edit.
+3. *Hand the commit + push to =roam-sync=.* Don't =git pull=, =git commit=, or =git push= here. Trigger the sync so the edit lands promptly rather than waiting up to 15 minutes for the next timer tick:
+
+ #+begin_src bash
+ systemctl --user start roam-sync.service
+ #+end_src
+
+ =roam-sync= commits the edit (under its generic auto-sync message), rebases onto the remote, and pushes. The removal is safe to land without a pull-first because only this project ever touches =<project>:=-prefixed lines (the ownership partition), so =roam-sync='s rebase can't conflict on the edit. Provenance for the routed tasks lives in the project's =todo.org= and session log, not the roam commit message. If =systemctl= isn't available, leave the edit for the next timer tick — it still lands.
+
+ Don't pull or stash the roam tree to "clean" it first: that fights =roam-sync= for ownership of the repo's git state. The edit-then-sync handoff is the whole point.
+
+** Phase E — Surface
+
+Report: local project inbox disposition first (processed count and whether it is clear), then roam disposition: moved (with their new priorities and tags), folded, dropped-as-done, and empty entries swept (count). Then the residue: foreign items (left for their owners, count only) and unowned items (count plus the headings that appear related to this project, for manual claim or prefix). Same "summarize what we kept" shape.
+
+If triaging this batch surfaced a durable, cross-project fact (a reference pointer worth keeping, a pattern worth recording), consider writing it to the agent KB as one =:agent:= node (see =knowledge-base.md=; personal projects only). Skip silently when nothing durable came up — never pad an empty run with a KB line.
+
+** Skip conditions
+
+- No project-local =inbox/= and no =~/org/roam/inbox.org= → silent no-op.
+- Project-local =inbox/= exists but has no pending handoffs → continue to roam scan.
+- No =~/org/roam/inbox.org= after the local inbox check → report the local inbox disposition and stop.
+- No claimed, no related-unowned, and no empty roam entries → report the total, stop.
+- Live org-capture against the roam inbox (capture-guard exit 1) → surface (interactive) or skip-and-self-heal (wrap-up), per core §5.
+
+** Caller integration
+
+*Startup (read-only nudge).* Startup already checks the project-local =inbox/= via =inbox-status= and processes it through process mode when needed. It also reads =~/org/roam/inbox.org= and produces the roam scan summary; one line surfaces: "Roam inbox: N items total, M appear related to this project (K empty entries to sweep) — say 'inbox zero' to file them." Offered as one of the priority options. The empty count rides along so a clean-up-only run still gets offered. Startup never auto-files or auto-sweeps roam items; it counts and offers (the read-only nudge never edits, so empties are reported, not removed, until a real triage runs).
+
+*Wrap-up (Step 3 sub-step).* A sub-step at the start of wrap-up Step 3 (before the cleanup scripts, so imported tasks get linted and ride the wrap commit) delegates here for the claimed set. Skip-fast when nothing matches.
+
+** Deferred: domain-aware routing (future work, multi-project)
+
+v1 handles the single-destination case via the prefix rule. The multi-project parts are deferred until the need is real:
+
+1. *Domain-aware empty-it-all mode.* If rulesets held a description of each project's domain, one run could guess the owner of every item (prefixed or not) and empty the whole inbox at once, delivering each item to its owning project's =inbox/= via =inbox-send= (where that project's process-mode gate still decides whether to file it). Open: where the domain map lives, how confident a guess must be before auto-routing vs surfacing, and whether a low-confidence item stays put.
+2. *Explicit per-project =:INBOX_PREFIX:= as the durable resolver*, replacing basename inference.
+3. *Unowned-item lifecycle* once domain-aware routing exists (no item stays unrouted indefinitely).
+4. *Concurrent push contention* on the shared roam repo: triage now hands its commit + push to =roam-sync=, which already aborts-and-surfaces on a rebase conflict. If multi-machine contention ever makes that abort frequent, =roam-sync= may want a retry-once-after-rebase.
+
+Take these up when the single-destination version is in use and the multi-project pain is concrete.
+
+* Mode: auto inbox zero
+
+A recurring, *interactive* roam check. Trigger phrase: "auto inbox zero" (match before "inbox zero" — the longer phrase wins). On invocation, *ask Craig for the interval* (e.g. 30 min, 2 hours), then drive the loop with =/loop <interval>= running roam mode. It is in-session and interactive by design — each cycle reports what it found and filed.
+
+** Per cycle
+
+1. Run roam mode's scan (Phase A local check + Phase B roam scan), read-only — no =git pull=. The capture-guard still gates any write: use =capture-guard --wait= (core §5) so a transient capture clears itself; if it's still open after the wait, *defer this cycle's roam reconcile to the next cycle* rather than surfacing — the loop cadence is the retry, and the filed items get swept next time. The rare write hands its git to =roam-sync= (roam Phase D).
+2. *Nothing found* → no inbox summary. One heartbeat line: =inbox zero at HH:MM: nothing= (HH:MM local, from =date=) — the silent-until-signal policy, see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=. Nothing else. Keeping a quiet inbox quiet is the whole point.
+3. *Items found* → summarize the found items, file them as tasks (roam Phase C), and *append them to a displayed queue* — the harness task list, via =TaskCreate= — so the queue accumulates across cycles. Then ask: "run this batch next?"
+ - *Yes* → chain into =work-the-backlog.org= as an explicit second step after routing completes: pass it the eligibility query over the queued items (status =TODO= + =:solo:= per the scheme header, priority-ordered), =file-only= mode, paging off, cap 1. The highest-priority eligible candidate runs; the rest wait for the next tick or a later yes.
+ - *No* → they stay queued for a later go.
+ This mode never implements anything itself — routing ends here, and the execution loop lives in =work-the-backlog.org=, its one home.
+4. *Cross-cycle dedup.* Subsequent cycles add only *newly-found* items to the same displayed queue, never re-surfacing what's already there. Dedup against the queue (the =TaskCreate= list), not against what's already been implemented — a find that was queued-but-not-yet-run must not reappear, and one already filed into =todo.org= is dropped by roam Phase C's status check.
+
+A find is always surfaced and filed; execution happens only through the =work-the-backlog.org= chain and waits for Craig's yes. A quiet inbox produces only the =inbox zero at HH:MM: nothing= heartbeat. =auto inbox zero= is inherently in-session because its chain step waits for that yes.
+
+** Fully-unattended pass (=/schedule=) — vNext, not v1
+
+A fully-unattended cron pass (firing while Craig is away) is a *different contract* and is deferred. It can't wait for a yes, so it has to decide up front whether it may mutate =todo.org= and the roam inbox or stays read-only, how a find reaches Craig asynchronously, how dedup state survives across runs that don't share a session, and what session/auth context a cron run carries.
+
+The =/schedule= recipe, once that contract is designed, would look like:
+
+#+begin_src
+/schedule <cron-expression> run inbox.org roam mode read-only, and <surface-mechanism> any finds
+#+end_src
+
+v1 ships only the interactive =/loop= shape above; the unattended contract is logged to =todo.org= for its own design pass. Don't invent the unattended behavior here — route a request for it to that task.
+
+* Common Mistakes
+
+1. *Treating items as orders.* Inbox content is a proposal. The value gate is the rule. Implementing every item without evaluation inflates =todo.org= and trains senders to keep sending noise.
+2. *Filing without applying the value gate.* "File as TODO" is not a default — it's the disposition for proposals that pass the gate but wait. A reject is also a valid answer.
+3. *Filing raw TODOs when the project has a priority scheme.* Core §6 is mandatory when the scheme exists. An un-graded TODO in a project with a legend is a defect.
+4. *Silently deleting a project handoff.* Send a response naming which value-gate question failed. Silent rejection trains the sender to escalate to Craig instead of through the inbox channel.
+5. *Pushing back on a Craig directive only to immediately implement it anyway.* If you genuinely think Craig is wrong, say so and wait. If you don't, just do the work — don't theatre the pushback.
+6. *Skipping the implement-vs-fold-vs-file classification.* Defaulting every accept to "file as TODO" turns the inbox into a queue that flows into =todo.org= without filtering.
+7. *Not propagating value-gate failure to the response.* When you reject a handoff, name *which* gate question failed so the sender can recalibrate, not just resend.
+8. *Forgetting to delete the inbox file after acting.* The local inbox should be empty when process mode ends. Files left behind become noise on the next startup.
+9. *Applying a shared-asset change proposal without the skeptical review.* The value gate alone asks whether to take the change, never whether the change is right, complete, or as simple as it should be. (Worked example: the 2026-06-12 spec-decisions handoff was applied as-is and the after-the-fact review surfaced a lost state, a vacuous gate pass, and an enhancement — all catchable up front.)
+10. *Editing the roam inbox without the capture-guard.* A disk write under a live org-capture wedges the capture (core §5). Guard first, every roam write.
+11. *Auto inbox zero re-surfacing queued items.* The loop must dedup against the displayed queue, not just against what's been implemented — or every cycle re-lists the same un-run finds.
+
+* Living Document
+
+Refine the value gate's three questions if the project's mission sharpens. Tune the per-source rejection-response template if =inbox-send= response loops surface a pattern. Tune the monitor cadence if task-boundary checking proves too frequent or too sparse. Capture the auto-loop interval that worked once the pattern recurs.
+
+If a mode wants real depth — enough that it bloats the core — it can become an =inbox.<mode>.org= plugin under this engine's namespace (the pattern =triage-intake= uses) rather than swelling this file. The principle that inbox items are *ideas to evaluate* is the part that doesn't change.
diff --git a/.ai/workflows/journal-entry.org b/.ai/workflows/journal-entry.org
index 3f476a7..c70dfe8 100644
--- a/.ai/workflows/journal-entry.org
+++ b/.ai/workflows/journal-entry.org
@@ -1,5 +1,5 @@
#+TITLE: Journal Entry Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2025-11-07
* Overview
diff --git a/.ai/workflows/meeting-prep.org b/.ai/workflows/meeting-prep.org
index 162ae30..563328b 100644
--- a/.ai/workflows/meeting-prep.org
+++ b/.ai/workflows/meeting-prep.org
@@ -1,5 +1,5 @@
#+TITLE: Meeting-Prep Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-06-10
* Overview
diff --git a/.ai/workflows/meeting-prep.pre-wire.org b/.ai/workflows/meeting-prep.pre-wire.org
index 6a156c0..3e27c2a 100644
--- a/.ai/workflows/meeting-prep.pre-wire.org
+++ b/.ai/workflows/meeting-prep.pre-wire.org
@@ -1,5 +1,5 @@
#+TITLE: Meeting-Prep — Pre-Wire Method (supporting doc)
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-06-10
Supporting document for the [[file:meeting-prep.org][meeting-prep workflow]]'s Phase 3.5. The workflow carries the condensed, in-flow version of pre-wiring; this file is the full Manager Tools method, kept beside the workflow (same name + =.pre-wire= suffix) so it travels with the workflow. Source casts: "How to Prewire a Meeting" (2007) and "Peer Prewire" (2015).
diff --git a/.ai/workflows/monitor-inbox.org b/.ai/workflows/monitor-inbox.org
deleted file mode 100644
index 1639f3b..0000000
--- a/.ai/workflows/monitor-inbox.org
+++ /dev/null
@@ -1,94 +0,0 @@
-#+TITLE: Monitor Inbox Workflow
-#+AUTHOR: Craig Jennings & Claude
-#+DATE: 2026-05-31
-
-* Overview
-
-Keep the project's =inbox/= responsive: notice handoffs on a cadence, triage each one, decide whether to act now or file it, and reply to the sender. This workflow is the /when, how-often, and act-vs-file/ layer. The per-item disposition mechanics — the value gate, the implement/fold/file classification, the per-source rejection flow — live in [[file:process-inbox.org][process-inbox.org]] and are not duplicated here. Think of it as: monitor-inbox decides /that/ an item gets handled and /how I respond/; process-inbox decides /what disposition/ each item gets.
-
-The gap this closes: handoffs that arrive mid-session used to sit unseen until the user asked or the next startup ran. A handoff the sender can't see being handled trains them to escalate around the inbox channel.
-
-* When to Use This Workflow
-
-Trigger phrases:
-
-- "monitor the inbox" / "watch the inbox"
-- "respond to the handoffs" / "handle the handoffs"
-
-Cadence auto-trigger (the main mechanism — see Cadence below): check at every task boundary during a session, not only when asked.
-
-* Cadence — how often to check
-
-*Default: check at every task boundary.* After finishing a unit of work, before reporting back or asking "what's next," run the cheap status check:
-
-#+begin_src bash
-.ai/scripts/inbox-status -q
-#+end_src
-
-Exit 1 means handoffs are pending — list them (drop =-q=) and process per process-inbox.org. Exit 0 means clean; say nothing. This is one =find=; it costs nothing to run often, and it's the fix for handoffs piling up unseen during long sessions.
-
-*Startup and wrap-up already cover their ends.* Startup Phase C processes a non-empty inbox; the wrap-up sanity check refuses to wrap with unprocessed handoffs. The task-boundary cadence fills the middle.
-
-*Mid-task arrivals.* If a handoff lands while you're mid-task and it's urgent (blocks the current work, or is time-sensitive), surface it right away. Otherwise batch it to the next task boundary so the current work isn't thrashed.
-
-*Unattended / background monitoring (opt-in).* When the user is working elsewhere and wants rulesets handoffs handled without being present, run a polling loop:
-
-#+begin_src
-/loop 15m check the inbox with inbox-status and process any handoffs per process-inbox.org
-#+end_src
-
-This is opt-in, not the default — continuous polling has a cost, and most handoffs aren't urgent. Pick an interval matched to how fast handoffs actually arrive (a burst of cross-project work warrants a tighter loop; a quiet day warrants none).
-
-* The act-vs-file decision
-
-Every accepted handoff (one that clears process-inbox's value gate) is then either acted on now or filed as a task. The rule, and how to surface it:
-
-*Act immediately — and just do it, no asking — when all of these hold:*
-- *Clear* — the action is unambiguous; no design decision or option-choice is needed.
-- *Bounded* — small, finishable this session, ideally a tight file set.
-- *Low-risk and verifiable now* — not a risky change to load-bearing infra (or trivially revertible), and testable/lintable this session.
-- *In-scope and safe* — within this project, not destructive or outward-facing without confirmation, not across a project boundary.
-- *Cheaper than deferring* — doing it now costs less than filing plus re-triaging later.
-
-When you decide to act, queue the work and do it. Don't ask first.
-
-*Exception:* a proposal to change a shared asset (template workflow, rule, skill, synced script) or a substantive convention never qualifies for silent act-now, however clear and bounded it looks — it routes through process-inbox's *Skeptical Review*, which carries its own approval (or, in a no-approvals session, park) step.
-
-*File a task when any of these hold:*
-- It needs a judgment call, a design decision, or an option the user would pick.
-- It's large, multi-session, or sprawls across many files.
-- It's blocked (a dependency, an external thing, the user is away).
-- It's risky enough to want the user's eyes before it lands.
-- It's off the session's active goal and acting now would derail it (file and keep going, unless it's urgent).
-
-When you decide to file, *ask first* — inline numbered options per =interaction.md=, with *filing as option 1 (the recommendation)* and *"do it now" as option 2*:
-
-#+begin_example
-<handoff> wants <X>. My read: file it (needs <reason>).
-
-1. File as a TODO ([#?] :tags:) — Recommended
-2. Do it now instead
-3. Something else
-
-Pick a number.
-#+end_example
-
-*Always ask if you're unsure* which side of the line an item falls on. Decisiveness on clear act-now items is the point of the rule; the ask is for genuine ambiguity and for filing.
-
-* Replying to handoffs
-
-A handoff came from another project's agent (or the user). Close the loop:
-
-- *Accepted and acted on* — send a confirmation to the sender via =inbox-send <sender> --text "..."=, naming what landed and the commit, so they're not left guessing (they can't see this project's git log). =inbox-send= excludes the current project as a target, so a self-sourced item is handled in-session, not sent.
-- *Accepted and filed* — a short confirmation that it's filed and where, so the sender knows it wasn't dropped.
-- *Rejected* — always state the why (which value-gate question failed), per process-inbox's per-source rejection flow.
-
-Cross-project boundary: never act on a file under another project's =.ai/= scope from here — route it back as a handoff (see =cross-project.md=).
-
-* The inbox-status script
-
-=.ai/scripts/inbox-status= lists unprocessed handoffs and exits nonzero when any are pending. Exclusions match the wrap-up sanity check (=.gitkeep=, =lint-followups.org=, =PROCESSED-*=). Exit 0 = clean, 1 = pending, 2 = no inbox/ or bad usage. Use =-q= for the count-only form the cadence check calls.
-
-* Living Document
-
-Tune the cadence if task-boundary checking proves too frequent or too sparse in practice. Refine the act-vs-file criteria as edge cases recur. If the background-monitor loop becomes a common pattern, capture the interval that worked. The decision rule itself — act-now is silent, filing asks with file-as-option-1, ambiguity asks — is the stable core (set by Craig, 2026-05-30).
diff --git a/.ai/workflows/no-approvals.org b/.ai/workflows/no-approvals.org
index 1efce82..6b5c7fa 100644
--- a/.ai/workflows/no-approvals.org
+++ b/.ai/workflows/no-approvals.org
@@ -1,5 +1,5 @@
#+TITLE: No-Approvals Mode
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-28
* Overview
@@ -22,6 +22,8 @@ Craig activates the mode with any of:
- Queuing several tasks in =todo.org= followed by any phrase above
- Any equivalent phrasing that signals he doesn't want to be re-asked between items
+*Not this mode:* any phrase containing "speedrun" ("speedrun", "no approvals speedrun") routes to =work-the-backlog.org='s no-approvals speedrun preset — an autonomous batch over an explicit ordered task set, with a pre-flight Q&A, autonomous commits, always-push, and an end-of-set page. This mode is the general interaction-gate suspension for whatever work is already underway; the speedrun is the dedicated backlog-batch workflow.
+
Mode resets when:
- Craig says approvals are back on
@@ -33,7 +35,7 @@ Mode resets when:
The interaction gates that step the workflow back to Craig for an "OK to proceed?" check:
-- The commit-message gate in =commits.md= Step 2. Print the final commit message inline, then commit immediately. No "approve / request changes / open in editor" prompt.
+- The commit-message gate in the =publish= skill, Step 2. Print the final commit message inline, then commit immediately. No "approve / request changes / open in editor" prompt.
- The PR-description gate. Print the final body, then create the PR.
- The PR-review-reply gate. Print the final reply, then post.
- "Ready to start?" / "Plan looks like X, proceed?" gates before implementation work begins.
@@ -44,10 +46,10 @@ The interaction gates that step the workflow back to Craig for an "OK to proceed
The engineering-discipline gates protect quality, not Craig's interaction time. They remain in force:
-- =/review-code= against the staged diff before every commit. Critical and Important findings still block. Minor findings still surface. No "proceed anyway" override unless Craig has given it explicitly for this batch.
+- =/review-code= against the staged diff before every commit, dispatched as an isolated adversarial reviewer per the =publish= skill's Step 1 — no-approvals removes *interaction* gates, never the isolation. Critical and Important findings still block, and the re-review loop still runs to approval. Minor findings still surface. No "proceed anyway" override unless Craig has given it explicitly for this batch. If the review can't reach approval — three rounds, a recurring finding, or a =Needs Discussion= verdict — that is a genuine question: park the item per step 4 and move to the next one rather than committing past a standing finding.
- =/voice personal= on every publish artifact (commit messages, PR titles + bodies, PR review comments). The full pattern walk happens. The printed result just doesn't wait for approval.
- The full test suite + lint + compile before commit (per =verification.md=).
-- Fetch-and-reconcile in =commits.md= Step 0.
+- Fetch-and-reconcile in the =publish= skill, Step 0.
- Session Log updates per =protocols.org=. Every state-mutating turn writes to =.ai/session-context.org= before the closing message. The log is the crash-recovery anchor while Craig is away. Missing entries lose work.
- Subagent review-gate cadence (=subagents.md=). Review each subagent's output before the next dispatch.
- Destructive or irreversible operations per =CLAUDE.md='s "Executing actions with care": force-push, =rm -rf=, dropping a column, dropping a branch, package removal. These need explicit consent regardless of mode. No-approvals is for *interaction* gates, not destructive-action consent.
@@ -68,7 +70,7 @@ For each item:
- Do the work.
- Update the Session Log per the rules in =protocols.org=.
-- Before any commit: run =/review-code= against the staged diff. Surface Critical and Important findings inline; fix them and re-review until clean. Minor findings show but don't block.
+- Before any commit: dispatch the isolated adversarial reviewer per the =publish= skill's Step 1 — never review your own staged diff inline. Surface Critical and Important findings; fix them and send the updated diff back to the *same* reviewer until it approves. Minor findings show but don't block and never earn another round. If the review can't reach approval — three rounds, a finding that recurs after being reported fixed, or a =Needs Discussion= verdict — park the item per step 4 with the standing findings and move on; don't commit past a blocking finding.
- Draft the commit message. Run =/voice personal= (the skill, or walk the patterns inline if unavailable). Print the final message inline before committing so the log shows it.
- Commit and push.
- One-line status between items ("Task X done, on to Y.") so Craig knows what's happening when he checks back in.
diff --git a/.ai/workflows/open-tasks.org b/.ai/workflows/open-tasks.org
index fe782d6..205d95c 100644
--- a/.ai/workflows/open-tasks.org
+++ b/.ai/workflows/open-tasks.org
@@ -1,5 +1,5 @@
#+TITLE: Open Tasks Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-04-25
* Overview
@@ -23,15 +23,16 @@ Don't route "task review" / "review tasks" here — those trigger the hygiene ha
* Phase A: Data Gathering (both modes)
-** Phase A pre-step — archive any freshly-DONE tasks
+** Phase A pre-step — normalize freshly-closed tasks
-Before reading =todo.org=, run the cleanup script's archive-done sweep so completed level-2 subtrees move from =* $Project Open Work= to =* $Project Resolved=:
+Before reading =todo.org=, run two cleanup sweeps so the read reflects current state. First convert any done sub-tasks to dated entries, then archive completed level-2 subtrees from =* $Project Open Work= to =* $Project Resolved=:
#+begin_src bash
+emacs --batch -q -l .ai/scripts/todo-cleanup.el --convert-subtasks todo.org
emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done todo.org
#+end_src
-Costs a few hundred milliseconds. Without it, a task that completed earlier in the session sits as =** DONE= under Open Work until the next =clean-todo= or wrap-up pass, and Next Mode would surface it as a "what's next" candidate. The sweep makes Phase A's read of =todo.org= reflect current state.
+Costs a few hundred milliseconds. Without the archive sweep, a task that completed earlier in the session sits as =** DONE= under Open Work until the next =clean-todo= or wrap-up pass, and Next Mode would surface it as a "what's next" candidate. The convert sweep runs first so a completed parent's sub-tasks are already dated when it archives; it also keeps interactive level-3 closes from lingering as DONE keywords. Together they make Phase A's read of =todo.org= reflect current state.
Skip the sweep if the workflow is invoked in an explicit read-only or dry-run context. Default is to run it.
@@ -176,6 +177,10 @@ Next Mode answers two questions in one output: "what matters most right now?" (t
Apply the prioritization cascade in order. Stop at the first matching step. This is the importance/urgency answer.
+*Exclude blocked tasks.* A task tagged =:blocked:= has an unmet cross-project dependency (its body names the project and the work owed, per =todo-format.md=). It can't be worked until that other project delivers, so it is *never* the cascade recommendation — skip it at every cascade step below. Blocked tasks are surfaced on their own in Step 3 so the stalled dependency stays visible instead of silently dropping out of view.
+
+*Surface blocking tasks first.* The mirror of the above: a task tagged =:blocker:= is holding up work in *another* project (its body names which project and what's owed, per =todo-format.md=). Clearing it unblocks that project, so it carries borrowed urgency — surface it at the *top* of the cascade recommendation regardless of its own priority cookie, ahead of the normal In-Progress / deadline / priority order. When several =:blocker:= tasks exist, lead with the one blocking the most, or the longest. This is the "do the thing that unblocks someone else first" rule; a =:blocker:= task left at its own low priority is exactly how a cross-project dependency stalls.
+
**** 1. In-Progress Tasks
- Look for tasks marked =DOING= or partially complete.
- *If found:* Recommend that task (always finish what's started).
@@ -228,11 +233,22 @@ Within each row, pick a single task per the same-level tie-breakers above (block
The friction filter is the override path. When the cascade winner is partially blocked, hardware-dependent, or simply too large for the user's current state, one of the friction rows is what they pick instead.
+*** Step 3 — Blocked-on-other-projects surface
+
+Independently of the cascade and the friction filter, collect every open task tagged =:blocked:=. These are tasks this project can't advance until another project delivers; surfacing them keeps a cross-project dependency from rotting at low priority on the other side — the exact failure the tag exists to prevent (a blocked task whose blocker is a =[#D]= in another project sits forever otherwise).
+
+For each blocked task, read its body for the blocking project and what's owed, and present one line: the task, the blocking project, and what that project owes. Then offer — per blocked task — to nudge the blocker: an =inbox-send <project> --text= note naming what's needed and why it's blocking, so the dependency gets attention in the project that owns it. Don't send without the user's go.
+
+If no =:blocked:= tasks exist, omit this surface entirely (the common case).
+
*** Output Format
-Pair the cascade recommendation with the friction block beneath it. Recommendation-at-item-1 convention applies to the friction rows — quick+solo first, since it's the strongest low-friction pick.
+Pair the cascade recommendation with the friction block beneath it, and the blocked-on-other-projects surface (Step 3) beneath that when any blocked task exists. Recommendation-at-item-1 convention applies to the friction rows — quick+solo first, since it's the strongest low-friction pick.
#+begin_example
+Unblocks other projects (do these first):
+- ai-term wrap-teardown companion — :blocker:, unblocks rulesets (the three ai-term functions)
+
Cascade recommendation (importance/urgency):
- Fix org-noter reliability — [#A], Method 1, 8/18 complete, blocks daily reading/annotation
@@ -240,17 +256,25 @@ If you want lower friction instead:
1. Quick + solo: Bump linter config — [#C] :quick:solo:, ~15 min
2. Quick: Confirm new dirvish setup — [#B] :quick:, needs your eye
3. Solo: Refactor config-utilities — [#B] :solo:, bounded but multi-hour
+
+Blocked on other projects (can't advance until the blocker delivers):
+- Wrap-teardown feature — blocked by emacsd: ai-term companion functions — nudge?
#+end_example
+The =:blocker:= surface sits at the very top — clearing one of those is the highest-leverage thing on the list, since it frees work in another project. Omit it when no =:blocker:= task exists (the common case).
+
Include for each row:
- Task name / description.
- Priority + tag cluster.
- One-line reasoning. For the cascade row, name which cascade step matched. For friction rows, an effort hint when one is obvious.
- Progress indicator (for V2MOM-structured todos) on the cascade row only.
+- For a =:blocker:= row: the project it unblocks and what's owed (from the task body).
+- For a blocked row: the blocking project and what it owes (from the task body), plus the nudge offer.
**** Edge cases
- *Empty friction block.* If no =:quick:= or =:solo:= tagged tasks exist in the open set, omit the friction block entirely. Present only the cascade recommendation.
+- *No =:blocker:= tasks.* Omit the "Unblocks other projects" surface entirely (the common case) — show it only when a task carries the =:blocker:= tag.
- *Dedupe.* If the cascade recommendation IS the same task as one of the friction rows (e.g. it's =:quick:solo:= and also won the cascade), show it once at the top with both labels. Don't list it twice.
- *Decline behavior.* If the user declines the cascade recommendation, drop straight to the friction block as the natural next prompt. Do not fall through to lower-cascade-tier tasks; the friction filter IS the override.
diff --git a/.ai/workflows/page-me.org b/.ai/workflows/page-me.org
index 607ed51..7a3b792 100644
--- a/.ai/workflows/page-me.org
+++ b/.ai/workflows/page-me.org
@@ -1,21 +1,29 @@
#+TITLE: Page Me Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-01-31
#+UPDATED: 2026-02-27
* Overview
-This workflow enables Claude to set timers and alarms that reliably notify Craig, even if the terminal session ends or is accidentally closed. Notifications are distinctive (audible + visual with alarm icon) and persist until manually dismissed.
+This workflow enables Claude to set timers and alarms that reliably notify Craig, even if the terminal session ends or is accidentally closed. Notifications are distinctive (audible + visual with the blue info icon) and persist until manually dismissed.
-Uses the =notify= command (alarm type) for consistent notifications across all AI workflows.
+Uses the =notify= command (info type) for consistent notifications across all AI workflows. Info-level on purpose: the earlier alarm styling read as all-red urgency, and Craig's verdict was that a page "should be a persistent info notification" — noticeable, never crash-scary (2026-07-02).
* Trigger Phrase
Craig says *"page me"* (or variations like "page me in 10 minutes", "page me at 3pm").
-The word "page" is the trigger for this workflow. It means: set a timed notification.
+The word "page" is the trigger for this workflow. It means: set a timed notification on the *desktop* channel (=notify=).
-Previously called "set-alarm" -- renamed to "page-me" for a distinctive, short trigger phrase that won't collide with common words like "remind" or "alert."
+Two sibling triggers pick a different channel; the timed =at= machinery below is identical for all three, only the fired command changes:
+
+- *"page me"* — desktop =notify= (this workflow's default).
+- *"text me"* — a Signal push to Craig's phone via =agent-text= (the away channel).
+- *"text and page me"* — both, for when he might be either place.
+
+Scope the triggers to the reflexive "me": "page me" and "text me", not a bare "page" or "text" in prose. The full channel vocabulary lives in protocols.org "Reaching Craig".
+
+"page" was chosen (renamed from the old "set-alarm") for a distinctive, short trigger that won't collide with common words like "remind" or "alert".
* Problem We're Solving
@@ -63,8 +71,8 @@ Craig tells Claude when and why:
Claude schedules the alarm using the =at= daemon with =notify=:
#+begin_src bash
-echo "notify alarm 'Page' 'Time to call the dentist' --persist" | at 3:30pm
-echo "notify alarm 'Page' 'Meeting starts' --persist" | at now + 45 minutes
+echo "notify info 'Page' 'Time to call the dentist' --persist" | at 3:30pm
+echo "notify info 'Page' 'Meeting starts' --persist" | at now + 45 minutes
#+end_src
The =at= daemon:
@@ -89,30 +97,43 @@ Craig dismisses the notification and acts on it.
** Setting Alarms
-Use the =at= daemon to schedule a =notify alarm= command:
+Use the =at= daemon to schedule a =notify info= command:
#+begin_src bash
# Schedule for specific time
-echo "notify alarm 'Page' 'Meeting starts' --persist" | at 3:30pm
+echo "notify info 'Page' 'Meeting starts' --persist" | at 3:30pm
# Schedule for relative time
-echo "notify alarm 'Page' 'Check the build' --persist" | at now + 30 minutes
+echo "notify info 'Page' 'Check the build' --persist" | at now + 30 minutes
# Schedule for tomorrow
-echo "notify alarm 'Page' 'Call the dentist' --persist" | at 3:30pm tomorrow
+echo "notify info 'Page' 'Call the dentist' --persist" | at 3:30pm tomorrow
#+end_src
** Notification System
-Uses the =notify= command with the =alarm= type. The =notify= command provides 8 notification types with matching icons and sounds.
+Uses the =notify= command with the =info= type. The =notify= command provides 8 notification types with matching icons and sounds.
#+begin_src bash
-# Immediate alarm notification (for testing)
-notify alarm "Page" "Your message here" --persist
+# Immediate page notification (for testing)
+notify info "Page" "Your message here" --persist
#+end_src
The =--persist= flag keeps the notification on screen until manually dismissed. All page-me notifications should use =--persist= by default.
+** Texting Craig's phone (the "text me" channel)
+
+The timed =notify= alarm above is the desktop channel. When Craig says "text me" (or a run expects him away from the machine), use =agent-text= instead, a Signal push to his phone from any machine or agent runtime:
+
+#+begin_src bash
+agent-text "Build finished, ready for your eyes"
+
+# Timed phone message: same at-daemon pattern, different channel
+echo "agent-text 'Meeting starts in 5'" | at 3:25pm
+#+end_src
+
+Channel selection and the mechanics live in protocols.org "Reaching Craig". On "text and page me", fire both: the desktop notification persists for whenever he returns, the phone push reaches him now.
+
** Managing Alarms
#+begin_src bash
@@ -139,10 +160,10 @@ The alarm must fire. Use the =at= daemon which is designed for exactly this purp
Simple invocation - Claude runs one command. No complex setup required per alarm.
** Fail Audibly
-If the alarm fails to schedule, report the error clearly. Don't fail silently.
+If the page fails to schedule, report the error clearly. Don't fail silently.
** Testable
-The =notify alarm= command can be called directly to verify notifications work without waiting for a timer.
+The =notify info= command can be called directly to verify notifications work without waiting for a timer.
** Non-Alarming
Use normal urgency, not critical. The notification should be noticeable but not imply something has gone horribly wrong.
diff --git a/.ai/workflows/process-inbox.org b/.ai/workflows/process-inbox.org
deleted file mode 100644
index 86df4c2..0000000
--- a/.ai/workflows/process-inbox.org
+++ /dev/null
@@ -1,215 +0,0 @@
-#+TITLE: Process Inbox Workflow
-#+AUTHOR: Craig Jennings & Claude
-#+DATE: 2026-05-28
-
-* Overview
-
-Inbox items are *ideas to evaluate*, not orders to execute. They arrive from Craig (typed directives saved as files), from other projects (handoffs via =inbox-send=), and from scripts/automated systems. Each is a proposal. An item earns a place in =todo.org= or git history only when it passes the value gate: it advances an existing task, improves how the project works, or serves the project's stated mission.
-
-The workflow is the disposition discipline. Read each item, evaluate honestly, apply the decision, then notify the sender if it was a project handoff and you're rejecting. Silent rejection on a handoff is worse than no reply.
-
-* When to Use This Workflow
-
-User triggers:
-
-- "process inbox" / "process the inbox"
-- "handle the inbox"
-- "what's in inbox" / "what's in the inbox"
-- "let's clear the inbox" / "let's process the inbox items"
-
-Auto-invocation:
-
-- Startup =Phase C step 2= delegates here when the inbox is non-empty. Don't ask Craig — just run it.
-
-Do *not* invoke this for inbox items that are clearly out-of-scope for the project — those are cross-project routing problems, handled per the cross-project boundary rule in =protocols.org=.
-
-* The Value Gate
-
-Every inbox item passes through three questions. One *yes* is enough to accept.
-
-1. *Does it advance an existing TODO?* Look up by topic in =todo.org='s open work. If the item extends a filed task, fold it in. If it implements a filed task, do the work.
-2. *Does it improve how the project works?* Architecture cleanup, workflow refinement, tooling, rule hygiene, drift detection — anything that makes the project itself more effective.
-3. *Does it serve the project's stated mission?* Read =notes.org= *Project-Specific Context* if the mission isn't obvious from the working directory and current task. The item should advance that mission, not orbit it.
-
-Three *no*s means reject. The rejection isn't lazy — an idea that doesn't help any current task, doesn't improve the system, and doesn't serve the mission is genuine noise, and accepting it inflates =todo.org= without payoff.
-
-* The Skeptical Review (change proposals)
-
-The value gate decides whether an item is worth taking. This review decides whether the proposed change is *right*. It applies to any item proposing a change to shared assets — template workflows, rules, skills, scripts, anything synced to consuming projects — and to any substantive convention change. FYIs, replies, and routine task filings skip it.
-
-Approach the file with curiosity and skepticism. It arrived from one project's context; you're evaluating it for all of them. Work through, in writing:
-
-1. Does this make sense for *all* consuming projects, or just the sender's situation?
-2. Does it conflict with any existing instruction — workflows, skills, rules, protocols, CLAUDE.md?
-3. How does it change a common activity Craig performs — better, worse, or differently than the sender assumed?
-4. Can it be enhanced to be more effective than as proposed?
-5. Should it be simpler?
-6. Plus at least three more questions specific to this change — e.g. what breaks for artifacts already using the old shape, what tooling interacts with it, what's underspecified, what does the sender's worked example not exercise?
-
-Output: a short summary of the thinking and a recommendation (accept as-is / accept with named changes / reject), surfaced to Craig for approval before applying.
-
-** In a no-approvals session: defer and stage
-
-Behavior-changing proposals still don't self-apply when Craig has put the session in no-approvals mode. Run the review, prepare the edits in =working/<task-slug>/= (a patch file or the worked-out diff), file a =[#B]= VERIFY carrying the decision package, and reply to the sender that it's parked. The sender's local stopgap (per =cross-project.md='s propagation process) means the delay costs nothing — the canonical update is about durability, not speed.
-
-Wording-only fixes — no consuming project acts differently — may proceed even then, logged in the session log.
-
-The VERIFY shape (top-level, =[#B]= so startup's A/B surfacing catches it; no =SCHEDULED= unless the proposal names a real deadline):
-
-#+begin_example
-** VERIFY [#B] Parked: <proposal topic> (from <sender>)
-What arrived: <one line — what the handoff proposes>.
-Recommendation: <accept as-is / accept with changes / reject> — <2-3 line
-skeptical-review summary: what's right, what to change, what was checked>.
-Prepared diff: [[file:working/<slug>/proposed.diff]] — apply is mechanical on
-your go.
-Say "approve the parked <topic>" (or adjust / reject) and it gets applied.
-#+end_example
-
-The full question-battery answers live in the session log and the =working/= dir, not the task body — the body carries the conclusion, with the trail one link away.
-
-* Phase A — Inventory (one parallel batch)
-
-Issue these reads in one parallel batch:
-
-1. List =inbox/= excluding =.gitkeep= and =PROCESSED-*= prefixes (use =\ls -la inbox/= per the protocols.org exa-alias note).
-2. Read =notes.org= *Project-Specific Context* if mission isn't already loaded in the session.
-3. Read =todo.org='s top-of-file priority scheme if present (look for a =* Priority and Tag Scheme= section or similar between the intro and the first =* <Project> Open Work= header).
-
-For each inbox file, parse the filename for sender. Two common patterns:
-
-- =YYYY-MM-DD-HHMM-from-<sender>-<topic>.<ext>= — from another project via =inbox-send=.
-- =<topic>.org= — typically from Craig directly, or from a script.
-
-Note the file type. =.eml= files need the extract script (not raw =Read=):
-
-#+begin_src bash
-# View mode
-python3 .ai/scripts/eml-view-and-extract-attachments.py inbox/<file>.eml
-
-# Pipeline mode (extract attachments to a directory)
-python3 .ai/scripts/eml-view-and-extract-attachments.py inbox/<file>.eml --output-dir assets/<target>/
-#+end_src
-
-Everything else, read directly.
-
-* Phase B — Evaluate each item
-
-For each inbox file:
-
-1. *Read it.* For substantive proposals (org files with TODO entries, design notes, multi-section docs), the full read is the right move. For short FYIs and one-liner asks, skim.
-2. *Identify the shape.* Is it an instruction, a question, a proposal, an FYI, or a handoff? Shapes guide disposition.
-3. *Apply the value gate.* Three questions above. One yes → candidate accept. Three nos → candidate reject.
-4. *Run the Skeptical Review* (section above) on any accepted item that proposes a shared-asset or convention change, before classifying. Its summary + recommendation rides along to Phase C; in a no-approvals session its defer-and-stage path replaces implement-now for behavior-changing proposals.
-5. *Within accept, classify:*
- - *Implement now* — small, scoped, clear, no design call required. The work is the disposition.
- - *Fold into existing TODO* — the item extends a task already filed; update the TODO body and link the inbox content if substantive.
- - *File as TODO* — substantive but waits, or needs design/triage before implementation.
-6. *Within reject, classify by source:*
- - *From Craig* — push back honestly in chat. State why you won't implement. Offer the conditions under which you would, if any. Wait for Craig to override or accept.
- - *From another project* — write a response file naming the rejection rationale and (optionally) the condition under which you'd reconsider. Deliver via =inbox-send <sender> --file <response>= per the cross-project handoff convention.
- - *From a script or automated system* — just delete; no notification needed.
-
-* Phase B.1 — Priority-scheme check
-
-This gates Phase C filing when there are accept-and-file items.
-
-Check whether =todo.org= has a top-of-file priority scheme (an explicit legend defining =[#A]= through =[#D]= semantics and mandatory/optional tag conventions).
-
-- *Scheme present* — file new TODOs per the scheme. Every TODO gets a priority cookie matching the legend's rules, the mandatory type tag, and any applicable effort/autonomy tags.
-- *Scheme absent* — surface one sentence: "This project has no priority scheme. We should adopt one before filing the new TODOs from this inbox pass — want me to propose one based on the rulesets scheme?" If Craig says yes, do that first (the =/research-priority-scheme= research subagent pattern in rulesets is the reference). If Craig says no, file the TODOs without grading but flag in the commit message that they're un-prioritized pending a scheme.
-
-The point is to avoid adding ungraded =TODO= entries to a project that's never agreed on what =[#A]= means.
-
-* Phase C — Surface dispositions
-
-Numbered options inline per =interaction.md= (no popup). Recommendation at item 1.
-
-Batch trivial items (one-line rejections of script noise, obvious file-as-TODO accepts where the scheme is already settled) into a single confirm-all prompt. Walk substantive items one at a time so the decision is visible.
-
-Per-item template:
-
-#+begin_example
-<filename> from <sender>: <one-line summary>
-Value-gate read: <yes/no on each of the three questions, one phrase each>
-Disposition recommendation: <implement / fold into <TODO> / file [#X] :tags: / reject>
-
-1. <recommendation as item 1>
-2. <alternative>
-3. Defer — leave in inbox under PROCESSED-<topic>.<ext> until <condition>
-4. Something else
-#+end_example
-
-For items that went through the Skeptical Review, the surfaced disposition includes its summary + recommendation, and approval here is what authorizes the apply. In a no-approvals session those items are reported as parked (the =[#B]= VERIFY) rather than surfaced for live approval.
-
-For pure FYIs that need no action, surface as a single line and recommend delete-with-acknowledgment.
-
-* Phase D — Apply
-
-Apply each disposition. The flow is autonomous past Craig's Phase C approval.
-
-** Implement-now
-
-Do the work. Commit per the project's commit flow. Delete the inbox file. The commit message references the inbox item by filename so the provenance lands in =git log=.
-
-** Fold into existing TODO
-
-Update the parent TODO's body with a dated reconciliation sub-entry per =todo-format.md= (=*** YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <what landed>=). Move substantive content to =docs/design/<date>-<topic>.<ext>= if it's worth keeping; reference from the TODO body. Delete the inbox file.
-
-** File as TODO
-
-Add the TODO under =* <Project> Open Work= with priority + tags per Phase B.1. Body summarizes the proposal and links the inbox content if it's been moved to =docs/design/=. Delete the inbox file (or move it to =docs/design/= first if the content survives).
-
-** Reject from Craig
-
-State the rejection in chat clearly: what you won't implement, why, and the conditions (if any) under which you would. Wait for Craig's override or acknowledgment. The inbox file stays until Craig confirms — if he overrides, re-enter Phase D as accept; if he acknowledges the rejection, delete the file.
-
-** Reject from another project (handoff)
-
-Write the response file at =/tmp/inbox-response-<topic>.org=. Contents:
-
-- Heading naming the original handoff and date
-- One paragraph: the rejection rationale (which value-gate question failed and why)
-- One paragraph: the condition under which you'd reconsider, if such a condition exists. If the answer is "never, this misreads the project's mission," say so directly.
-
-Deliver via =inbox-send <sender> --file /tmp/inbox-response-<topic>.org=. The =inbox-send= script (per =cross-project.md=) handles the from-prefix, date stamp, and target inbox path.
-
-Delete the local inbox file after the response lands in the sender's inbox.
-
-** Reject from script or automated system
-
-Just delete. No notification.
-
-** Defer
-
-Rename in place to =inbox/PROCESSED-<original-filename>= and add a brief comment line at the top: =# Deferred YYYY-MM-DD: <condition>=. Don't accumulate deferred items indefinitely — sweep them on a future =process-inbox= run when the condition is met or the deferral has aged out.
-
-** Park (Skeptical Review in a no-approvals session)
-
-Move the proposal file into =working/<task-slug>/= alongside the prepared diff, file the =[#B]= VERIFY per the Skeptical Review section, reply to the sender that it's parked for Craig's review, and delete the inbox file. On Craig's approval the apply is mechanical: apply the prepared edits, run the normal verify-and-publish flow, rewrite the VERIFY to a dated log entry per =todo-format.md=, and send the sender the acceptance reply. On rejection, the reject-from-another-project flow above runs unchanged.
-
-* Phase E — Close out
-
-Verify =inbox/= is empty (excluding =.gitkeep= and any intentional =PROCESSED-*= files). Run =\ls -la inbox/= and confirm.
-
-Update the session log per =protocols.org= with one short paragraph summarizing this pass: count processed, count accepted (implement/fold/file split), count rejected (Craig/handoff/script split), and the commit SHA if a commit landed.
-
-Stamp =:LAST_INBOX_PROCESS:= in =notes.org='s *Workflow State* section if it exists, so future workflows that gate on freshness can read it. Same format as =:LAST_AUDIT:= (=YYYY-MM-DD=).
-
-* Common Mistakes
-
-1. *Treating items as orders.* Inbox content is a proposal. The value gate is the rule. Implementing every item without evaluation inflates =todo.org= and trains senders to keep sending noise.
-2. *Filing without applying the value gate.* "File as TODO" is not a default — it's the disposition for proposals that pass the gate but wait. A reject is also a valid file-as-TODO answer to nothing.
-3. *Filing raw TODOs when the project has a priority scheme.* Phase B.1 is mandatory when the scheme exists. An un-graded TODO in a project with a legend is a defect.
-4. *Silently deleting a project handoff.* Send a response. The sender's next session sees the response in their inbox and learns the rejection rationale. Silent rejection trains the sender to escalate to Craig instead of through the inbox channel.
-5. *Pushing back on a Craig directive only to immediately implement it anyway.* If you genuinely think Craig is wrong, say so and wait for his call. If you don't, just do the work — don't theatre the pushback.
-6. *Skipping the implement-vs-fold-vs-file classification.* Defaulting every accept to "file as TODO" turns the inbox into a queue that flows into =todo.org= without filtering. Small, scoped, clear items get implemented now; substantive proposals get filed; extensions to existing work get folded.
-7. *Not propagating value-gate failure to the response.* When you reject a handoff, the response should name *which* gate question failed (advances no current task / doesn't improve the project / doesn't serve the mission) so the sender can recalibrate, not just resend.
-8. *Forgetting to delete the inbox file after acting.* The inbox should be empty when this workflow ends. Files left behind become noise on the next startup.
-9. *Applying a shared-asset change proposal without the Skeptical Review.* The value gate alone asks whether to take the change, never whether the change is right, complete, or as simple as it should be. A proposal that's clear and bounded can still carry a design gap — the review is where that surfaces, before the change syncs to every consuming project. (Worked example: the 2026-06-12 spec-decisions handoff was applied as-is and the after-the-fact review surfaced a lost state, a vacuous gate pass, and an enhancement — all catchable up front.)
-
-* Living Document
-
-Refine the value gate's three questions if the project's mission sharpens. Tune the per-source rejection-response template if =inbox-send= response loops surface a pattern. Add new auto-classification shortcuts if certain item shapes (e.g. routine FYIs from a script) become common.
-
-The workflow is shaped by use. The principle that inbox items are *ideas to evaluate* is the part that doesn't change.
diff --git a/.ai/workflows/process-meeting-transcript.org b/.ai/workflows/process-meeting-transcript.org
index 4dd340f..d0806ad 100644
--- a/.ai/workflows/process-meeting-transcript.org
+++ b/.ai/workflows/process-meeting-transcript.org
@@ -1,5 +1,5 @@
#+TITLE: Process Meeting Transcript Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-03
* Overview
@@ -10,16 +10,16 @@ This workflow defines the process for processing meeting recordings from start t
Trigger this workflow when:
- Craig says "process the transcript" or "process the recording" or similar
-- New recording files (.mkv) appear in ~/sync/recordings/ after meetings
+- New recording files (.mkv, .m4a, or .flac) appear in ~/sync/recordings/ after meetings
- Craig wants to process meeting recordings into labeled transcripts
* Prerequisites
-- Recording file(s) exist in ~/sync/recordings/ (*.mkv)
+- Recording file(s) exist in ~/sync/recordings/ (*.mkv, *.m4a, or *.flac)
- Calendar files available at ~/.emacs.d/data/*cal.org for meeting titles
- AssemblyAI transcription script at ~/.emacs.d/scripts/assemblyai-transcribe
- AssemblyAI API key stored in ~/.authinfo.gpg (machine api.assemblyai.com)
-- ffmpeg available for audio extraction
+- ffmpeg available for audio extraction (video .mkv only; .m4a and .flac skip extraction)
* The Workflow
@@ -43,13 +43,13 @@ Classification is per recording, not per session — a single run can carry this
Find and match recording files with calendar events. *Run sub-steps 1 and 3 (recording list + calendar dump) as a single parallel batch* — they're independent. Sub-step 2 (parse timestamps) and sub-step 4 (matching) work from those two outputs in-memory, so they're sequential after the batch.
-1. **List recordings:** Find all recording files in ~/sync/recordings/ (video .mkv or audio-only .m4a)
+1. **List recordings:** Find all recording files in ~/sync/recordings/ (video .mkv or audio-only .m4a / .flac)
#+begin_src bash
- ls -la ~/sync/recordings/*.mkv ~/sync/recordings/*.m4a 2>/dev/null
+ ls -la ~/sync/recordings/*.mkv ~/sync/recordings/*.m4a ~/sync/recordings/*.flac 2>/dev/null
#+end_src
- Audio-only recordings (.m4a) are used when no screen content is expected. These skip Step 3 (audio extraction) since they're already in a transcribable format.
+ Audio-only recordings (.m4a or lossless .flac) are used when no screen content is expected. These skip Step 3 (audio extraction) since they're already in a transcribable format. FLAC is the recorder's current audio format — its frames are self-contained, so an interrupted recording still decodes.
-2. **Extract timestamps:** Parse date/time from each filename (format: YYYY-MM-DD-HH-MM-SS.mkv or .m4a)
+2. **Extract timestamps:** Parse date/time from each filename (format: YYYY-MM-DD-HH-MM-SS.mkv, .m4a, or .flac)
3. **Match with calendar:** Check ~/.emacs.d/data/*cal.org for meetings at those times
#+begin_src bash
@@ -74,7 +74,7 @@ Per =cross-project.md=: transcribe and label here, then deliver the *labeled tra
** Step 3: Extract Audio (video recordings only)
-*Skip this step for .m4a files* — they are already audio and can go directly to transcription.
+*Skip this step for .m4a and .flac files* — they are already audio and can go directly to transcription.
For .mkv video recordings, extract audio for transcription:
@@ -96,11 +96,11 @@ Output: /tmp/FILENAME.m4a (temporary, deleted after transcription)
#+begin_src bash
# For .mkv files (audio was extracted to /tmp/):
~/.emacs.d/scripts/assemblyai-transcribe /tmp/FILENAME.m4a > ~/sync/recordings/FILENAME.txt
- # For .m4a files (transcribe directly):
+ # For .m4a or .flac files (transcribe directly — AssemblyAI accepts both natively):
~/.emacs.d/scripts/assemblyai-transcribe ~/sync/recordings/FILENAME.m4a > ~/sync/recordings/FILENAME.txt
#+end_src
-2. **Clean up:** Delete intermediate .m4a file after successful transcription (only for .mkv extractions — do NOT delete original .m4a recordings)
+2. **Clean up:** Delete intermediate .m4a file after successful transcription (only for .mkv extractions — do NOT delete original .m4a or .flac recordings)
#+begin_src bash
rm /tmp/FILENAME.m4a
#+end_src
@@ -181,13 +181,13 @@ Present the speaker identification table to Craig for confirmation:
** Step 9: Copy Recording to Meetings Folder
-1. Ensure engagement meetings folder exists and patterns are in .gitignore (~*/meetings/*.mkv~ and ~*/meetings/*.m4a~)
+1. Ensure engagement meetings folder exists and patterns are in .gitignore (~*/meetings/*.mkv~, ~*/meetings/*.m4a~, and ~*/meetings/*.flac~)
2. Copy the recording file with descriptive name:
#+begin_src bash
# Video recordings:
cp ~/sync/recordings/YYYY-MM-DD-HH-MM-SS.mkv {engagement}/meetings/YYYY-MM-DD_HH-MM-meeting-name.mkv
- # Audio-only recordings:
+ # Audio-only recordings (.m4a or .flac — copy with the original extension):
cp ~/sync/recordings/YYYY-MM-DD-HH-MM-SS.m4a {engagement}/meetings/YYYY-MM-DD_HH-MM-meeting-name.m4a
#+end_src
Example: ~deepsat/meetings/2026-02-03_11-02-standup-ipm-grooming.mkv~
diff --git a/.ai/workflows/read-calendar-events.org b/.ai/workflows/read-calendar-events.org
index be66bf4..5eac529 100644
--- a/.ai/workflows/read-calendar-events.org
+++ b/.ai/workflows/read-calendar-events.org
@@ -1,5 +1,5 @@
#+TITLE: Read Calendar Events Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-01
* Overview
diff --git a/.ai/workflows/readability-audit.org b/.ai/workflows/readability-audit.org
new file mode 100644
index 0000000..90ad366
--- /dev/null
+++ b/.ai/workflows/readability-audit.org
@@ -0,0 +1,242 @@
+#+TITLE: Readability Audit Workflow
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-06-28
+
+* Overview
+
+A pass over one file, a set of modules, or the whole tree that makes the code
+*readable to a future maintainer*. It checks four things and fixes the cheap
+ones in place: the file-top commentary, the inline comments, the names, and the
+physical organization of the code. Structural changes that need a real refactor
+(splitting a module, renaming a public symbol) are not done here — they are
+filed as =:refactor:= tasks so they get their own design and test pass.
+
+This is language-agnostic. Where a step names a language-specific tool or
+convention, it's stated as "the project's <X>, if it has one" — read the
+project's =CLAUDE.md= / =notes.org= and the language bundle to resolve the
+concrete tool.
+
+* Where it sits among the code-quality tools
+
+These tools are a pipeline, not duplicates. Knowing which to reach for:
+
+- *readability-audit* (this workflow) — prose and human-reader clarity:
+ comments, file headers, names, and physical organization. Judgment-driven
+ (does this comment lie? does this name reveal intent? can a newcomer place
+ this file in a minute?).
+- =/refactor= — structure on measurable metrics: complexity, duplication,
+ dead-code, the =simplification= lens (behavior-preserving logic/size
+ reduction), and =rename= (executes a codebase-wide symbol rename).
+- =/simplify= — behavior-preserving cleanup of the current diff, applied
+ directly.
+
+The link that keeps them from overlapping: when this audit finds a structural
+problem too big for a comment/name fix — a module to split, a *public* symbol to
+rename across call sites — it *files* a =:refactor:= task rather than doing it
+here. =/refactor= (rename, simplification) or =/start-work= then executes that
+filed task with a proper design and test plan. Readability finds and files;
+=/refactor= transforms.
+
+* Problem We're Solving
+
+Source files drift toward two opposite failure modes, and both hurt the next
+person to open the file:
+
+- *Documentation rot and noise.* Headers carry stale user-manual content
+ (quick-starts, full option matrices, setup walkthroughs) that belongs in user
+ docs; comments restate what the next line already says; comments go out of
+ date and start lying; placeholder =TODO=/=FIXME= stubs and conversational
+ asides accumulate. A blank summary or a missing file-top description leaves a
+ reader with no map.
+- *Structural fog.* Names that don't reveal intent force the reader to decode
+ them; related functions scatter; a public entry point sits far from the
+ private helpers it calls; a file grows to hold several unrelated
+ responsibilities.
+
+Left alone, opening a file costs more every month. The fix is a repeatable audit
+with a clear, checkable standard, run on demand or as files are touched.
+
+* Exit Criteria
+
+For the audited scope:
+
+1. *Every file has an accurate top section* that states what the file does and
+ how it fits the rest of the codebase — terse, no user-manual content, and
+ carrying the project's file-header convention where it has one.
+2. *Every surviving comment earns its place* — it explains a *why* the code
+ can't (a constraint, a workaround and its reason, an ordering dependency, a
+ warning), it is accurate against the current code, and it is terse. Obvious
+ "describe the next line" comments are gone.
+3. *Names reveal intent* — no cryptic abbreviations; the project's
+ public/private visibility convention is applied consistently.
+4. *Related code is co-located* — a public function's private helpers sit right
+ after it; the file reads top-to-bottom by descending abstraction; sections
+ group what belongs together.
+5. *Structural problems too big to fix in a comment pass are filed* as
+ =:refactor:= tasks, not left as a vague note and not half-done inline.
+6. *Nothing broke* — the build is clean and the test suite is green
+ (comment/name edits are behavior-preserving, so this should always hold; it
+ is the proof, not a hope). See "Graceful degradation" for projects without a
+ suite.
+
+* When to Use This Workflow
+
+- "Let's run the readability-audit workflow."
+- "Audit the comments and commentary in <file/area>."
+- "Clean up the structure/organization of <module>."
+- After landing a feature, on the files it touched, before moving on.
+- On a single file you just found hard to read.
+- As a tree-wide sweep: inventory all the source files, audit each, batch the
+ fixes.
+
+Do NOT use this to *perform* the structural refactors themselves (use
+=/refactor= or =/start-work= against a filed task) or to hunt for bugs /
+complexity / duplication (that is =/refactor=, not a readability pass).
+
+* Approach: How We Work Together
+
+** Phase 1 — Scope and inventory
+
+Pick the target: one file, a named module set, or the whole tree. For a sweep,
+list the source files (honor =.aiignore=) and decide coverage. Lean on the
+language's own doc linters as a first filter where they exist — many flag a
+missing or blank file summary and malformed headers; run the project's lint
+target first.
+
+** Phase 2 — Audit each file against the four dimensions
+
+Record findings as =file:line — issue — proposed fix=. The four dimensions:
+
+*** A. File-top commentary (the map)
+
+- Present, and *accurate* against what the file now does.
+- States purpose, the file's role/architecture, and key entry points —
+ *tersely*. A reader should learn what this is and how it connects in a few
+ lines.
+- Carries the project's file-header convention where it has one (a metadata
+ block, a module docstring, a standard header comment). If the project has no
+ header convention, skip this sub-check — don't invent one.
+- Does *not* carry user-manual content — quick-starts, full option matrices,
+ step-by-step setup. That belongs in user docs; move it, don't keep it in the
+ source header.
+- Mechanics are correct for the language: a filled summary line (not blank), the
+ expected section markers, the expected footer.
+
+*** B. Inline comments (why, not what)
+
+- Explains a *why* the code cannot: a workaround *and its reason*, an ordering
+ or load dependency, business-logic rationale, a real warning ("do not reorder
+ these — deadlock").
+- Is *accurate* — matches the current code. A wrong comment is worse than none;
+ fix or delete on sight.
+- Is *terse and useful*. Delete the obvious "describe the next line" comment
+ unless it names a non-obvious constraint. Replace a stale placeholder or a
+ rambling aside with the real one-line reason, or remove it.
+- Convert a comment that's only restating the code into a better *name* instead
+ (see C).
+
+*** C. Names (carry the what/how so comments don't have to)
+
+- Intention-revealing variable and function names; no cryptic single letters or
+ abbreviations outside tight local scopes.
+- The project's public/private convention is applied consistently and correctly:
+ a helper only called within the file is private; a user-facing or
+ intentionally-reusable symbol is public. (Resolve the concrete convention from
+ the language and the project — a naming prefix, an export list, an
+ access modifier.)
+- When a comment exists only to explain a name, rename instead.
+
+*** D. Organization (co-location and ordering)
+
+- Related functions sit together. A public function's private helpers come
+ *right after* it (stepdown / proximity / "reads like a newspaper").
+- The file reads top-to-bottom by descending abstraction.
+- Sections group what belongs together.
+- *Cohesion check:* if the file holds several unrelated responsibilities, or has
+ grown large enough that the top no longer describes one coherent thing, flag a
+ split into layered owners — but see Phase 4: that's a filed refactor, not an
+ inline fix.
+
+** Phase 3 — Apply the cheap, safe fixes inline
+
+Dimensions A, B, and C are *comment- and name-only* and *solo* (no design or
+preference call): apply them directly. After each file (or a batch), verify with
+the project's gates: parse/syntax check, a clean build (no new warnings), and a
+green test suite. Comment/name edits can't change behavior, so green is the proof
+the edit was clean, not a behavior check.
+
+For a tree-wide sweep, drive the uniform rewrites mechanically and verify the
+whole batch at once: a *mechanical applier with a boundary assertion* that
+replaces a well-defined header span is reliable and fast, then one suite run
+covers the batch. Keep the varied cases (header-line fixes, summary fixes that
+must preserve surrounding metadata, inline-comment surgery, generated-file
+headers) as careful per-file edits. (The boundary markers are language-specific;
+the principle — mechanical applier + assert + one suite run for uniform
+rewrites, per-file judgment for varied cases — is not.)
+
+** Phase 4 — File the structural refactors, don't do them here
+
+Dimension D's bigger findings — split a module, rename a *public* symbol across
+call sites, move a function to a different file — are real refactors with their
+own risk and test surface. Do *not* slip them into a readability pass. File each
+as a =:refactor:= task in =todo.org= with the specific finding, so it gets
+=/refactor= or =/start-work= with a proper design and test plan. This is the
+line between the cheap clarity win and the structural change; keeping it sharp is
+what lets the audit stay safe and fast.
+
+** Phase 5 — Verify and commit in logical batches
+
+Full suite green, build clean. Commit the doc/comment changes as =docs:= (or
+=refactor:= where a header/structure normalized) in cohesive batches — one
+commit per coherent slice (a set of condensed commentaries, the
+generated-file-header fixes, the obvious-comment prune), not one mega-commit and
+not one-per-file. Generated files are fixed *in their generator* and then
+regenerated, so the next regen stays compliant.
+
+* Graceful degradation
+
+The audit adapts to what the project provides:
+
+- *No file-header convention* → skip dimension A's metadata sub-check; still
+ check the summary/description for accuracy and terseness.
+- *No test suite* → the green-suite proof in Phases 3 and 5 is unavailable. Fall
+ back to the strongest gate the project has (compile/byte-compile, parse check,
+ linters) and *flag the weaker proof as a known limit* — a behavior-preserving
+ edit is lower-risk, but say plainly that there's no suite to confirm it.
+- *No doc linter* → do the Phase 1 first-filter by reading instead; the audit
+ still runs, just without the cheap pre-pass.
+
+* Principles to Follow
+
+- *Comments explain why; code explains what.* If a comment restates the code,
+ delete it or turn it into a better name.
+- *Accuracy beats completeness.* A wrong or stale comment is worse than no
+ comment. When in doubt, delete.
+- *Terse and useful.* Every comment and every header line earns its place. The
+ source header is not the user manual — move manuals to user docs.
+- *Readable means the next person, fast.* The test of the top-section and the
+ organization is whether a maintainer who has never seen the file can place it
+ and navigate it in under a minute.
+- *Keep the cheap pass cheap.* Comment/name fixes are solo and land inline.
+ Structural splits and public renames are not — they get filed, designed, and
+ tested separately.
+- *Preserve legal and attribution headers verbatim.* Vendored / GPL / copyright
+ notices are never condensed away by a readability pass.
+- *Manual validation is still Craig's.* Solo means no input is needed to *do*
+ the work; visual/behavior confirmation afterward is expected where relevant.
+
+* Living Document
+
+Update this with what real runs teach. Lessons worth keeping as the standard
+sharpens:
+
+- *Interpretation default for "fix blank summary":* when a rewrite shows only a
+ header + summary and omits a metadata block the file already has, keep the
+ existing metadata and replace only the header line and the summary. Its
+ absence from the rewrite means "leave it," not "delete it."
+- *Generated files:* fix the *generator*, then regenerate. Editing the generated
+ file directly is reverted on the next regen.
+- *Vendored files:* preserve the copyright/attribution; do not auto-condense a
+ licensed header.
+- *Mechanical applier + assert + one suite run* is the safe way to do a
+ many-file uniform rewrite; per-file judgment is for the varied cases.
diff --git a/.ai/workflows/rename-artifact.org b/.ai/workflows/rename-artifact.org
index 7b9f15b..a8d1246 100644
--- a/.ai/workflows/rename-artifact.org
+++ b/.ai/workflows/rename-artifact.org
@@ -1,5 +1,5 @@
#+TITLE: Rename an .ai Artifact
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-31
* Summary
diff --git a/.ai/workflows/send-email.org b/.ai/workflows/send-email.org
index 065f925..82d2286 100644
--- a/.ai/workflows/send-email.org
+++ b/.ai/workflows/send-email.org
@@ -1,5 +1,5 @@
#+TITLE: Email Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-01-26
* Overview
diff --git a/.ai/workflows/sentry.org b/.ai/workflows/sentry.org
new file mode 100644
index 0000000..b25fc14
--- /dev/null
+++ b/.ai/workflows/sentry.org
@@ -0,0 +1,227 @@
+#+TITLE: Sentry — Overnight Hygiene Supervisor
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-19
+
+* Overview
+
+Sentry is an interval loop that keeps a project's hygiene current while Craig is away. Each cycle walks a fixed list of passes — roam pull, inbox zero, triage (no mail or messengers), todo cleanup, task audit, working-files hygiene, spec board, link integrity, git health, prep freshness, bug and refactor finding, and (opt-in) solo-task implementation — and commits each pass's writing to a throwaway daily branch. Nothing pushes. In the morning Craig reviews the branch, squash-merges what he wants, and deletes it.
+
+The design goal is a project that greets the morning already tidy, with every judgment call and every destructive action parked in an approval queue rather than executed unattended. Sentry does the mechanical sweeping; Craig does the deciding.
+
+This file is the engine. It owns the entry gates, the branch mechanics, the lock model, the per-cycle pass runner, the digest and approval queue, the skip semantics, and the stop-sentry shutdown. The =agent-lock= helper (=.ai/scripts/agent-lock=) provides the locks. The passes reuse existing workflows (=inbox.org=, =triage-intake.org=, =clean-todo.org=, =task-audit.org=) under sentry's unattended contract.
+
+* When to Use This Workflow
+
+Craig arms sentry at the end of a session, with the machine left running, to have overnight hygiene done by morning.
+
+Triggers:
+
+- "start sentry", "run sentry", "arm sentry", "sentry mode"
+- "let sentry watch this overnight", "keep this tidy overnight"
+- "start sentry hourly", "start sentry every <interval>" (sets the loop interval)
+
+Stop trigger (see Stop Sentry below):
+
+- "stop sentry", "stand down sentry", "sentry off"
+
+Sentry is deliberately *not* auto-armed. Running it in a project is a per-project grant (the =:COMMIT_AUTONOMY:= marker) plus a deliberate launch with Craig at the terminal for the entry gates.
+
+* Prerequisite — the autonomy ticket
+
+Sentry commits unattended. =commits.md= gates commits on Craig's approval, so sentry needs standing, per-project authorization to run at all. Before anything else, read the project's =.ai/notes.org= Workflow State block for:
+
+: :COMMIT_AUTONOMY: yes
+
+If the marker is absent or not =yes=, decline to start and name the marker:
+
+: Sentry needs ":COMMIT_AUTONOMY: yes" in .ai/notes.org Workflow State to run — it commits unattended. Add it to grant, or run the hygiene passes by hand.
+
+No half-running mode: a project without the grant doesn't run sentry's read-only passes either. The grant is one line away, so this is a deliberate opt-in, not a barrier.
+
+A second, *independent* marker gates the solo-task implementation pass (pass 12):
+
+: :SENTRY_MAY_IMPLEMENT: yes
+
+=:COMMIT_AUTONOMY:= lets sentry commit its hygiene sweeps to the branch; =:SENTRY_MAY_IMPLEMENT:= additionally lets it implement solo, decision-free backlog tasks on the branch. The split exists because the two carry different morning costs: hygiene is a two-minute merge, implemented code is a review session. A project can run hygiene-only sentry without the implement pass, and most should until sentry has quiet weeks behind it. Absent =:SENTRY_MAY_IMPLEMENT:=, pass 12 skips; sentry still runs every other pass. Requires =:COMMIT_AUTONOMY:= alongside it — implementing implies committing.
+
+* Entry — interactive, with Craig present
+
+Craig types the sentry trigger, so the first moves run with him at the terminal. Do them in order; each gate that fails stops entry until Craig answers.
+
+1. *Autonomy ticket* — the prerequisite above. Absent → decline and stop.
+
+2. *Dirty-tree gate.* =git diff --quiet HEAD= (tracked modifications only; untracked and gitignored files never block — an inbox drop or scratch file is not in-progress work). If the tracked tree is dirty, describe what's dirty and offer, inline-numbered per =interaction.md=:
+
+ 1. Finish the job — commit the in-progress work first (recommended if it's a coherent unit)
+ 2. Stash it — =git stash= and start sentry on a clean tree
+ 3. Roll back named changes — discard specific files (names them)
+
+ Wait for an answer. Sentry can't start unattended from a dirty state; that's the point.
+
+3. *Green-suite gate.* Run the project's full suite (=make test=, or the project's equivalent — detect it). Read the output. If anything is red, describe the failures and offer to investigate before arming. The loop starts only on a green baseline, because every unattended cycle measures itself against "did I break this?" and a pre-existing red poisons that check.
+
+4. *Prior sentry branch.* =git branch --list 'sentry/*'=. An unmerged =sentry/*= branch from a previous night means the morning review didn't happen. Surface it and offer to squash-merge or delete it now (Craig is present); don't stack a second sentry branch on the first.
+
+5. *Reconcile the project branch.* Fetch and fast-forward-only against upstream — the same reconcile =startup= runs:
+
+ : git fetch --all --prune
+ : git rev-list --left-right --count @{u}...HEAD
+
+ Zero-behind → continue. Behind-only and clean → =git merge --ff-only @{u}=. Diverged → surface to Craig (he's present); don't auto-resolve.
+
+6. *Create the daily branch.* From HEAD:
+
+ : git switch -c "sentry/$(date +%F)-$(uname -n)"
+
+ The host suffix (=uname -n=) stops a same-date collision between the two daily drivers. The working tree now sits on this branch overnight — the launch hands the repo to sentry until the morning merge. Reclaiming it mid-night means stopping sentry first (see Stop Sentry). Note the Emacs buffer-revert caveat to Craig if he has the repo open: files change on disk under him overnight, so buffers want reverting after the morning merge (see =emacs.md=).
+
+7. *Arm the loop.* Start =/loop= at the interval (default hourly; Craig's "every <interval>" phrase overrides) with the per-cycle body being one sentry cycle (the Pass Runner below). Confirm the arming in one line: interval, branch name, project.
+
+* The lock model
+
+Two locks, both served by =.ai/scripts/agent-lock= (names only; the helper owns the paths, which live on tmpfs under =$XDG_RUNTIME_DIR/agent-locks/=, host-local and cleared on reboot).
+
+*Single-runner lock* (=sentry-<project>=, where =<project>= is the repo-root basename: =basename "$(git rev-parse --show-toplevel)"= — the same derivation =wrap-it-up.org='s guard uses, so the two agree on the lock name). Each cycle acquires it at cycle start and releases it at cycle end, and refreshes it between passes (the heartbeat, so a live cycle's lock never ages past one pass). If =/loop= fires again while a previous cycle still holds it, the new cycle's acquire fails and the cycle skips with one digest line — no two cycles run at once. The bounded wait is short (a few seconds); a live cycle means defer, not queue.
+
+*Roam-write lock* (=roam-write=). A pass that edits a file under =~/org/roam= acquires it, runs =capture-guard --wait= (the human-capture layer stays underneath), edits the working tree, triggers =systemctl --user start roam-sync.service=, and releases. The lock spans only edit-plus-trigger. Sentry never runs =git= against =~/org/roam= — roam-sync stays the repo's only committer (the 2026-06-24 one-git-owner rule). Pass 1's =pull --ff-only= is the sole, read-only exception.
+
+Every reclaim of a stale lock surfaces in the digest — the helper prints the reclaim note, and the cycle records it. A reclaim during a genuinely slow pass is possible, so it's never silent.
+
+* The Pass Runner — one contract per pass
+
+Each cycle, after acquiring the single-runner lock and verifying branch state (below), walks the pass list in order. Every pass follows the same four-step contract:
+
+1. *Probe* — a cheap existence check for the pass's target (named per pass below). Absent → the pass is one skip line in the digest and nothing more. This is what makes the pass list portable: passes self-activate where their target exists and stay silent elsewhere, with zero per-project configuration.
+
+2. *Work* — run the pass under the unattended contract. Quick, solo, already-agreed mechanical actions execute. Anything destructive or requiring judgment does *not* execute — it appends to the morning-approval queue (what, why, the exact command or edit that fires on approval). A pass runs fully or not at all; there is no reduced-form pass.
+
+3. *Session-context entry* — a pass that does or queues work appends its digest line to the =session-context.org= Session Log (path resolved via =.ai/scripts/session-context-path=) before its commit, so a crash between them still leaves the trail. Per-pass lines for an all-quiet cycle (every pass probe-skipped or no-op) are not written one by one — the cycle collapses to a single heartbeat at cycle-end (below), so an idle cycle doesn't spray one skip line per pass.
+
+4. *Commit* — if the pass wrote to disk, commit it: =chore(sentry): <pass> — <what changed>=. One commit per writing pass. A probe-skip or a no-op pass writes nothing and commits nothing.
+
+Between passes, refresh the single-runner lock (=agent-lock refresh sentry-<project>=) — the heartbeat.
+
+** Branch-state verification (cycle start, before the passes)
+
+After acquiring the lock, confirm the cycle is safe to run:
+
+- *On the right branch* — HEAD is =sentry/<today>-<host>=. If the loop was armed on a prior day and crossed midnight, the branch keeps the arming date; that's fine, morning teardown handles it. If HEAD is somehow *not* a sentry branch (an interrupted stop, a manual checkout), skip the whole cycle with a digest line rather than committing onto main.
+- *Clean of foreign changes* — =git diff --quiet HEAD= excluding the spine set (=session-context.org= / =session-context.d/=, resolved via =session-context-path=). Sentry's own spine writes must not trip this; a genuinely unexpected dirty tree (something outside the spine changed and wasn't committed by a prior pass) poisons the cycle — skip it with a digest line, the next cycle retries.
+
+* Unattended safety — skip, never degrade
+
+With no one at the terminal, any unsafe state makes the affected scope skip with one digest line, and the next cycle retries. Unsafe states and their scope:
+
+- *Unexpected dirty tree* (non-spine) → skip the whole cycle.
+- *Lost or un-acquirable single-runner lock* → skip the cycle (another cycle holds it, or the helper is missing).
+- *A pass's own precondition unmet* (its probe fails, or a dependency is dirty) → skip that pass only.
+- *Red suite at cycle-end* (see below) → the commits stay on the branch, flagged in the digest for morning review; the cycle doesn't roll back.
+
+Skips are never silent and never partial. Inside a *working* cycle, a pass line means the pass fully ran and a skip line names why it didn't. An *all-quiet* cycle is not a silent skip either: its single =sentry at HH:MM: nothing= heartbeat is the explicit record that every pass found nothing, standing in for a wall of identical skip lines. The anti-silence rule targets a pass that hides work it should have surfaced; a quiet cycle has surfaced that there was none.
+
+** Multi-day stall notification
+
+An unmerged prior =sentry/*= branch at cycle start (the morning review never happened) skips the cycle. After the *second consecutive* cycle skipped for this reason, send one persistent desktop notification naming the project and branch:
+
+: sentry stalled: <branch> unmerged — merge or delete to resume
+
+Then repeat at most daily. Persistent notify matches the paging convention — it stays on screen until dismissed. A multi-day stall never stays silent.
+
+* The pass list (v1)
+
+In order. Each names its detection probe. A pass whose probe fails is one skip line.
+
+1. *Roam pull* — =git -C ~/org/roam pull --ff-only=. Probe: =~/org/roam= is a git clone. Skipped when the roam tree is dirty (roam-sync owns that case) or the clone is absent. Read-only and ff-only — the one narrow exception to "don't touch roam git," so later passes read a fresh tree.
+
+2. *Inbox zero* — run =inbox.org= roam mode under the no-approvals contract: quick+solo+agreed items execute, shared-asset and convention proposals park (prepared diff, =VERIFY= task, sender reply) in the approval queue. Edits to =~/org/roam/inbox.org= take the roam-write lock + =capture-guard=. Probe: the roam clone or a project =inbox/= exists. Tidying the shared roam inbox is allowed from *any* project session, work included — it's housekeeping on a shared resource, not a durable KB-node write, so the work-denylist doesn't gate it (=knowledge-base.md=). Never park it as a cross-project boundary crossing.
+
+3. *Triage intake — mail and messenger sources excluded.* Run =triage-intake.org=, loading only its non-mail, non-messenger source plugins (calendar, PR/ticketing). The mail and messenger plugins — cmail, any Gmail variant, Telegram, Signal, chat DMs — are never loaded by a sentry cycle: Craig ruled 2026-07-21 that sentry doesn't check email or messengers. A manual "triage intake" still scans everything. Probe: the project has at least one *active* triage source that survives that exclusion — a project-specific plugin (=.ai/project-workflows/triage-intake.*.org=), or a non-empty =:TRIAGE_SOURCES:= declaration naming general plugins that exist. Mere presence of the template-synced general plugins does *not* activate the pass; a project that declares no sources, or whose only declared sources are mail or messengers, probe-skips (see =docs/specs/2026-07-20-triage-source-activation-spec.org=). Destructive actions (deleting, archiving, sending) queue; they never cycle unattended.
+
+4. *Todo cleanup* — the =clean-todo.org= mechanics (hygiene pass + =--archive-done= + =--convert-subtasks=). Probe: a root =todo.org=. Note that =--archive-done= is not purely an org-file pass on its first run in a project: it creates =archive/task-archive.org= and appends a =.gitignore= entry, so it produces a real tracked-file commit and correctly trips the cycle-end conditional suite. (archangel, first live run 2026-07-21.)
+
+5. *Task audit* — the *mechanical subset* of =task-audit.org= hourly (staleness counts, structural checks, cookie recomputation); the judgment half (priority regrades, consolidations, merge candidates) runs *once per night* and queues its findings rather than repeating them every cycle. Probe: a root =todo.org=. A full audit every hour is too heavy and re-surfaces the same judgment calls all night. (takuzu, first live run 2026-07-21.) Factual staleness fixes that are unambiguous still execute.
+
+6. *Working-files hygiene* — flag =working/<slug>/= directories whose backing task is closed (a filing candidate per =working-files.md=). Probe: a =working/= directory exists. The filing itself queues (it's a judgment move).
+
+7. *Spec status board* — the =docs-lifecycle= grep for spec keywords, surfacing any =DOING= spec whose bound build parent is closed. Probe: =docs/specs/= exists.
+
+8. *Link integrity* — broken =file:= links in the project's org files, via =lint-org.el=. Probe: =lint-org.el= present. Report-only into the digest; no unattended rewrites.
+
+9. *Git health* — uncommitted drift, unpushed commits on other branches, stale branches, main-behind-origin. Probe: =.git=. Report into the digest.
+
+10. *Prep + symlink freshness* — stale daily-prep docs, broken symlinks. Probe: the prep dir / symlinks exist (work and home only, in practice).
+
+11. *Bug and refactor finding* — hunt for real bugs and worthwhile refactoring opportunities in the project's codebase: static analysis (=shellcheck= for shell, the project's own linters for its languages), config sanity checks, plus one targeted code-reading area per cycle. Rotate the area across cycles and name it in the digest, so coverage accumulates over a night instead of re-reading the same corner. Randomized property sweeps (generate-and-verify against an engine's own invariants) are good quiet-cycle work here, reaching past a frozen test corpus. Expect the pass to go honestly quiet after the first few cycles find the standing defects; a quiet hunt is a result, not a failure. (takuzu, first live run 2026-07-21: three real fixes in the first four cycles, then quiet.) This pass does *not* run the test suite — the entry baseline already ran it, and re-running it hourly is anti-pattern 5; read the entry result instead. Probe: the project carries a codebase — source under version control beyond its org and tooling files. File each verified bug as a graded task in =todo.org= per the severity × frequency matrix (=todo-format.md=), and each refactoring opportunity as a =:refactor:= task, deduped against existing tasks; an unverifiable suspicion is a digest line, not a task. *Find, never fix in this pass* — the finding files a task and stops. A fix happens only in the opt-in implementation pass below, and only after the finding is a filed task that pass then re-verifies from scratch (see the premise rule there). A freshly-found "bug" can be a misread — one was filed and retracted two cycles apart on 2026-07-23 — so the file-then-verify-then-fix pipeline is deliberate: the task is the checkpoint, not a same-breath fix. (Added at Craig's order 2026-07-21, first dogfooded in dotfiles; refactor-finding added 2026-07-24.)
+
+12. *Solo-task implementation (opt-in — =:SENTRY_MAY_IMPLEMENT:=)* — work the backlog's solo, decision-free tasks on the branch. Probe: =.ai/notes.org= Workflow State carries =:SENTRY_MAY_IMPLEMENT: yes= *and* the project holds =:COMMIT_AUTONOMY:= (the implement pass commits). Absent the marker, skip — this pass is off by default, because it turns the morning from a two-minute merge into a code review, and that's the project owner's call. When on: invoke =work-the-backlog.org= under its unattended-loop contract (no pre-flight Q&A — there's no Craig overnight), eligibility =TODO= + =:solo:=, with the defer checklist deciding act-vs-file. The overnight-only tightening: only the *ready* bucket implements (clears every checklist item with zero open decisions); a task needing even one quick decision defers to a =VERIFY= rather than guessing, exactly as the loop caller already does. Commit each logical change to the sentry branch; *never push* — the morning review and merge is the gate, same as every other pass. The full quality bar holds (TDD, suite green before each commit, the isolated adversarial review per =publish= Step 1 with its re-review loop, =/voice=), and the review here runs the *premise check first*: reproduce the bug or confirm the problem is real before judging the diff. The review is the fact-checker that a filed claim never got, and it is what makes fixing-on-a-branch safe (Craig, 2026-07-24). A task that fails its premise check is not implemented — the finding was wrong, and that outcome is a digest line, not a commit. A task whose review never reaches approval — three rounds, a recurring finding, or a =Needs Discussion= verdict — is the same shape: no commit, and a digest line naming the standing findings, so the morning review sees what the reviewer would not pass rather than finding the task silently absent. (Added at Craig's direction 2026-07-24: overnight implement-on-branch, gated and never-pushed.)
+
+(KB lesson promotion — the pass the original proposal listed eleventh — is deferred to vNext. An unattended judgment pass writing to the shared knowledge base waits until sentry has quiet weeks behind it and a designed detection heuristic. See the filed lesson-detection-heuristic task.)
+
+* Cycle-end — conditional suite, then the digest commit
+
+After the passes:
+
+1. *Conditional suite run.* If any pass this cycle modified files *outside* the org/spine set (a code-touching pass, rare but possible via fixtures), run the full suite once. A green run confirms the cycle's commits are safe; a red run flags the digest for morning review — the commits stay on the branch (nothing is pushed, so the morning gate catches it). No per-pass suite runs: the entry run is the green baseline, and hourly per-commit runs would turn a seconds-long cycle into minutes all night. Cycles that only touched org/spine files skip this.
+
+2. *Heartbeat or digest, then commit.* Decide quiet vs working. A *quiet* cycle — every pass probe-skipped or no-op, nothing added to the approval queue — writes a single heartbeat line to the Session Log, =sentry at HH:MM: nothing= (HH:MM local, from =date=), and no per-pass digest block. A *working* cycle — any pass ran, wrote, or queued — writes its full per-pass digest block. Then commit any accumulated spine writes in one sweep: =chore(sentry): digest — <date> <time> cycle= for a working cycle, =chore(sentry): heartbeat — <date> <time>= for a quiet one, so even a quiet cycle leaves a clean tree for the next branch-state check (where the spine is untracked, the mirror-only case, there is nothing to commit and the heartbeat line stays in the working-tree anchor). This is the silent-until-signal policy (see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=): an all-quiet night collapses from a wall of no-op digests to a list of one-line heartbeats, while a cycle that actually did or queued something still writes the full record.
+
+3. *Release the single-runner lock.*
+
+* The digest and the approval queue
+
+*Digest.* A *working* cycle appends its block to the =session-context.org= Session Log (the spine the cycle already writes), so it survives a crash, rides the session archive, and is on screen in the running session. One block per working cycle: the timestamp, then one line per pass (ran + what, or skipped + why), plus any lock reclaim notes. A *quiet* cycle (nothing done or queued) writes no block — just the one heartbeat line =sentry at HH:MM: nothing= (the silent-until-signal policy). The per-pass block is a working-cycle artifact; it still carries one line per pass so a real skip inside a working cycle is never hidden.
+
+*Approval queue.* Destructive and judgment actions accumulate under one heading in the same file — =* Sentry approval queue (<date>)= — newest last. Each item carries three things: *what* (the action), *why* (what triggered it), and the *exact command or edit* that fires on approval. The morning review is Craig reading this heading top to bottom and running or discarding each item.
+
+* Morning teardown — Craig's, documented not automated
+
+Sentry never merges its own branch. In the morning Craig:
+
+1. Reviews the digest and the approval queue in =session-context.org=.
+2. Runs or discards each approval-queue item.
+3. Reviews the branch: =git log main..sentry/<date>-<host>= and the diff.
+4. Squash-merges what he wants (=git switch main && git merge --squash sentry/<date>-<host>=, then one clean commit) or cherry-picks selectively.
+5. Deletes the branch: =git branch -D sentry/<date>-<host>=.
+6. Reverts any Emacs buffers still showing the pre-merge on-disk state (=emacs.md= buffer-revert caveat).
+
+A bad night is discarded by deleting one branch — nothing reached main, nothing was pushed.
+
+In a project that gitignores =.ai/=, the whole spine is untracked, so quiet cycles produce no commits at all and =git log main..sentry/<date>-<host>= understates the night's activity. There the anchor's heartbeat list is the only record of what fired. Read the anchor, not just the log. (archangel, first live run 2026-07-21.)
+
+* Stop Sentry
+
+Trigger: "stop sentry" (and synonyms above). Sentry owns its own shutdown:
+
+1. *Cancel the loop* — stop the =/loop= (=ScheduleWakeup= stop / the loop's stop path). No further cycles.
+2. *Release the single-runner lock* if this context holds it.
+3. *Branch disposition* — offer, inline-numbered:
+ 1. Squash-merge the day's branch into main now (walk the morning teardown steps 3-5 interactively)
+ 2. Leave it named for later review (=sentry/<date>-<host>= stays; review at leisure)
+4. *Approval queue* — offer to walk the queued items now, or carry them (they stay under the heading for whenever Craig reviews).
+
+Stopping sentry is the only way to reclaim the working tree mid-night. The entry gate fronts the handoff; stop-sentry ends it.
+
+* Wrap-up interaction
+
+=wrap-it-up.org= refuses while sentry is live: it detects the single-runner lock (=agent-lock status sentry-<project>= → held) and stops with "sentry is active — say 'stop sentry' first." The shutdown logic lives here, not in wrap-up; wrap-up carries only the one guard.
+
+* Common Mistakes
+
+1. *Running without the =:COMMIT_AUTONOMY:= grant* — sentry commits unattended; the marker is the entry ticket, and its absence is a hard stop, not a degrade.
+2. *Starting from a dirty or red tree* — the entry gates exist because an unattended cycle can't tell Craig's in-progress work from a regression. Answer the gate; don't bypass it.
+3. *Committing onto main* — every writing pass commits to the daily =sentry/*= branch. A cycle that finds HEAD off the sentry branch skips rather than commits.
+4. *Running a =git= write against =~/org/roam=* — roam-sync is the only committer. Sentry edits the tree under the roam-write lock and triggers the sync; it never commits or pushes roam.
+5. *A per-pass suite run* — the suite runs at entry (baseline) and conditionally at cycle-end (only when a pass touched non-org files). Hourly per-commit runs all night is the anti-pattern the suite policy exists to prevent.
+6. *Executing a judgment or destructive action unattended* — those queue for the morning with their exact command. The pass did its detection; Craig makes the call. The one sanctioned exception is pass 12's solo-task implementation, and only because it inherits work-the-backlog's full defer checklist (data-loss and irreversible actions defer, never execute) plus a premise-verifying review, and it commits to the branch rather than acting on anything live.
+7. *A silent skip* — inside a working cycle, every skip writes a digest line naming why; a missing pass with no line reads as "ran clean" when it didn't. The one exception is not a violation: an all-quiet cycle collapses to a single =sentry at HH:MM: nothing= heartbeat instead of one skip line per pass — the heartbeat is the explicit "nothing to do" record, per the silent-until-signal policy.
+8. *Degrading a pass to a reduced form* — a pass runs fully or skips. No half-passes.
+9. *Letting an unmerged branch stall silently* — after two consecutive unmerged-branch skips, the persistent desktop notify cycles. Don't suppress it.
+10. *Merging sentry's branch automatically* — the morning teardown is Craig's. Sentry creates and commits; it never merges or deletes its own branch.
+
+* Living Document
+
+Sentry ships with eleven finding/hygiene passes, one opt-in implementation pass, and a deferred KB pass. The pass list, the interval default, the =:SENTRY_MAY_IMPLEMENT:= default, and the queue-vs-execute line for each pass are the knobs most likely to move with dogfooding. The implement pass especially is new (2026-07-24) and unproven at scale — watch the corrections signal (work-the-backlog's metric for autonomous commits later reverted or hand-fixed) before widening it past the projects that opt in. Fold in what the live trial surfaces — a pass that queues too eagerly, a probe that misfires, a digest line that wants more detail. Refine as the signal arrives.
+
+* History
+
+Built 2026-07-19 from the sentry spec (=docs/specs/2026-07-14-sentry-workflow-spec.org=, ID f6c51f27-d7a2-4b63-9ff9-5ba005a66dfb) — 10 decisions and 12 review findings resolved before the build. Phase 1 shipped the =agent-lock= helper (commit =a8b6cf4=); this file is Phase 2, the engine. Phase 3 reconciles the roam writers (=inbox.org=, =knowledge-base.md=) to acquire the roam-write lock and adds the =wrap-it-up.org= guard.
diff --git a/.ai/workflows/session-harvest.org b/.ai/workflows/session-harvest.org
index c48d689..54a7c09 100644
--- a/.ai/workflows/session-harvest.org
+++ b/.ai/workflows/session-harvest.org
@@ -1,5 +1,5 @@
#+TITLE: Session-Harvest Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-06-11
* Overview
diff --git a/.ai/workflows/spec-create.org b/.ai/workflows/spec-create.org
index f90c511..39758a0 100644
--- a/.ai/workflows/spec-create.org
+++ b/.ai/workflows/spec-create.org
@@ -1,5 +1,5 @@
#+TITLE: Spec-Create Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-06-09
* Overview
@@ -10,7 +10,7 @@ The guiding principle, drawn from how Google, Oxide, Amazon, Basecamp, and the A
It is the front of a trio:
- =spec-create.org= (this one) — author writes the spec.
-- =spec-review.org= — a reviewer gates the spec for implementation-readiness and writes =<spec-basename>-review.org=.
+- =spec-review.org= — a reviewer gates the spec for implementation-readiness and records findings in the spec's =* Review findings= section.
- =spec-response.org= — the author folds the review back in.
The spec this workflow produces has to *pass spec-review's gate* — that gate is the definition of done. So the structure below is built to answer the reviewer's questions up front. Keep it lightweight anyway: a short required spine plus a *readiness-dimensions menu* where each item is either answered or explicitly marked "N/A because…". The best spec is the shortest one that still lets an engineer build it, test it, and ship behavior that matches the user's mental model.
@@ -47,6 +47,8 @@ Capture, in this order:
** Phase 2 — Design, alternatives, decisions
1. *Design* — overview first, then detail. Write the reasoning as *prose, not bullet dumps* — prose exposes weak logic that bullets let you hide. Use bullets only for genuinely enumerable lists. When the thing has an interface, use the *two-altitude* split (Rust RFC): explain it once for a user/caller, once for an implementer.
+
+ *Non-trivial UI.* When the deliverable is a real UI (a panel, a multi-control surface, an interacting visual layout — not a single dialog, a CLI flag, or a one-off prompt), the design isn't settled on the page. Run the research → ~5 distinct working-prototype directions → iterate-one-to-final process in =claude-rules/ui-prototyping.md= before treating the UI design as done, and add a =Prototype iterations= subsection under the spec's status heading linking every iteration (final linked in the design section). A UI design decision moves to =DONE= only once it's been seen working in a prototype.
2. *Alternatives considered* — the load-bearing section authors skip and reviewers need most. For each option, force a why-not with the MADR grammar: "Good, because… / Bad, because… / Neutral, because…". Even one rejected option, with the reason, beats presenting one path as inevitable.
3. *Decisions* — capture each real choice as an org =TODO= task carrying an inline mini-ADR (Nygard's spine):
- The heading is =** TODO <Decision name>=. It flips to =DONE= when the decision-maker agrees with the call; until then it stays =TODO=.
@@ -59,7 +61,7 @@ Capture, in this order:
This is where the spec earns a "Ready" from review: an engineer must be able to build it in steps, know when it's done, and never have to invent product behavior mid-implementation.
-1. *Implementation phases* — decompose the work into phases each small enough to finish in one focused session and each leaving the tree in a working (not half-broken) state. =spec-review= lifts this section straight into =todo.org= tasks, so a spec that can't be phased fails the gate — the absence is itself a finding.
+1. *Implementation phases* — decompose the work into phases each small enough to finish in one focused session and each leaving the tree in a working (not half-broken) state. =spec-review= checks this section decomposes cleanly and =spec-response= lifts it into =todo.org= tasks, so a spec that can't be phased fails the gate — the absence is itself a finding.
2. *Acceptance criteria* — the observable conditions that mean the feature works, written as checkable items. The review's test-surface task mirrors these.
3. *Readiness dimensions* — walk this menu and, for each, either define the behavior or write "N/A because…". The escape hatch keeps a simple spec short; the prompt keeps a hidden decision from slipping into implementation:
- *Data model & ownership* — what's user-authored / generated / cached / remote; who owns each editable region; what persists vs refreshes.
@@ -82,8 +84,9 @@ This is where the spec earns a "Ready" from review: an engineer must be able to
** Phase 5 — Wire it up (conventions)
-- *Filename + location:* =docs/<problem-slug>-spec.org=. Org-mode. The slug names the *problem/feature*, not a date. Must end in =-spec.org=.
-- *Metadata header:* a small table at the top — Status, Owner, Reviewer(s), Date, Related (link to the task/ticket).
+- *Filename + location:* =docs/specs/YYYY-MM-DD-<problem-slug>-spec.org= — formal specs live in =docs/specs/=, never =docs/design/= (that's for notes, brainstorms, inventories; see =claude-rules/docs-lifecycle.md=). Org-mode. The slug names the *problem/feature*; no status suffixes ever — status lives in the file. Must end in =-spec.org=.
+- *Status heading (first element after the file header):* a top-level heading carrying the lifecycle keyword, stamped =DRAFT= at authoring — spec-create owns this flip. It holds an =:ID:= UUID (generate with =uuidgen=) and dated history lines, newest first. The keyword is authoritative; the Metadata =Status= field mirrors it in lowercase. Transitions are three lines in one file (keyword + history line + mirror): spec-review flips =READY=, spec-response flips =DOING= at decomposition, the final build task flips =IMPLEMENTED=. Terminal states always record a reason.
+- *Metadata header:* a small table at the top — Status (the lowercase mirror), Owner, Reviewer(s), Date, Related (link to the task/ticket).
- *Review-and-iteration-history stub:* add a =Review and iteration history= section at the bottom and seed it with the author's first entry. =spec-review= and =spec-response= append provenance entries here, so the heading shape is a contract: =YYYY-MM-DD Day @ HH:MM:SS -ZZZZ — Contributor — Role=, body fields What / Why / Artifacts.
- *Cross-link both ways:* the spec links its task; the task links the spec (replace the task's inline plan with a terse description + a =file:= link to the spec).
@@ -103,7 +106,14 @@ Then it's ready for =spec-review.org=. Snapshot-vs-living rule: keep the spec li
,#+TITLE: <Feature> — Spec
,#+AUTHOR: <author>
,#+DATE: <YYYY-MM-DD>
-,#+TODO: TODO | DONE SUPERSEDED CANCELLED
+,#+TODO: TODO | DONE
+,#+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED
+
+,* DRAFT <spec short name>
+:PROPERTIES:
+:ID: <uuid — generate with uuidgen>
+:END:
+- <YYYY-MM-DD Day @ HH:MM:SS -ZZZZ> — drafted.
,* Metadata
| Status | draft |
diff --git a/.ai/workflows/spec-response.org b/.ai/workflows/spec-response.org
index 2686cf8..7628e49 100644
--- a/.ai/workflows/spec-response.org
+++ b/.ai/workflows/spec-response.org
@@ -5,9 +5,9 @@
* Overview
-The spec-response workflow processes external reviews of a design spec and folds them into the spec until it is implementation-ready. A reviewer (human or another agent) leaves a review file next to the spec — typically produced by its counterpart, the *spec-review* workflow, which writes =<spec-basename>-review.org= and assigns a readiness rubric. Claude works through every recommendation, deciding accept / modify / reject for each, updates the spec for the accepted ones, documents the modified and rejected ones with reasons, then deletes the review file. Repeat for each spec under review.
+The spec-response workflow processes a review's findings and folds them into the spec until it is implementation-ready. A reviewer (human or another agent) records findings in the spec's =* Review findings= section — typically via its counterpart, the *spec-review* workflow, which writes one =TODO= task per finding (=[/]= cookie on the heading) and assigns a readiness rubric. Claude works through every finding, deciding accept / modify / reject, updates the spec body for the accepted ones, and completes each finding task in place: accept and modify finish =DONE= (the modify's change noted in the body), reject finishes =CANCELLED= with the reason. Repeat for each spec under review.
-The output is a spec a reader could implement from, plus a durable record — inside the spec — of why any recommendation was changed or declined, so the reviewer can find the reasoning later without re-litigating.
+The output is a spec a reader could implement from, plus a durable record — inside the spec — of why any finding was changed or declined, so the reviewer can find the reasoning later without re-litigating. The reasoning lives on the completed finding task, not a separate file.
This workflow was first run on 2026-05-23 against the linear-emacs =issue-query-spec.org= and =issue-representation-spec.org= reviews, and written up from that run.
@@ -27,25 +27,25 @@ A review is only useful if every point in it gets a decision. Without a defined
A spec's review is fully processed when:
-1. *Every recommendation has an explicit disposition* — accepted, modified, or rejected. None dropped.
-2. *Accepted recommendations are woven into the spec body* — the spec reads as if they were always there, not appended as a changelog.
-3. *Modified and rejected recommendations are documented in a bottom section* (e.g. "Review dispositions") with a one-paragraph reason each, so the reviewer can find the reasoning.
+1. *Every finding has an explicit disposition* — accepted, modified, or rejected, and its task completed (=DONE= or =CANCELLED=). None dropped.
+2. *Accepted findings are woven into the spec body* — the spec reads as if they were always there, not appended as a changelog.
+3. *Modified and rejected findings carry a one-paragraph reason in the completed task's body*, so the reviewer can find the reasoning.
4. *Review/response provenance is documented in the spec* — iteration count/date, contributor, role, what changed, and why.
5. *Pre-agreed decisions are flipped to =DONE=* — each settled decision's =TODO= becomes =DONE=, and the =[/]= cookie on the spec's =* Decisions= heading reflects the tally.
6. *Cross-spec tensions are reconciled in writing* when related specs were reviewed together.
-7. *The review file is deleted* once 1-6 hold.
+7. *Every finding task is completed* — the =* Review findings= =[/]= cookie reads complete (each finding =DONE= or =CANCELLED=).
8. *Tracking is updated* — the spec's VERIFY/task body notes "review incorporated" and whether it's implementation-ready.
9. *Implementation tasks exist* — once the author confirms the spec is Ready, the project's =todo.org= carries the full implementation-task breakdown (Phase 6), reviewed for completeness, with =:solo:= marked and a Manual-testing task for everything else.
-The whole run is done when no =*-review.org= files remain and each spec is judged implementation-ready (or its remaining blockers are named).
+The whole run is done when every spec's =* Review findings= cookie reads complete and each spec is judged implementation-ready (or its remaining blockers are named).
-*Measurable validation:* a reader scanning the review against the revised spec can find, for every review point, either the change in the body or its disposition at the bottom. Nothing is unaccounted for.
+*Measurable validation:* a reader scanning the spec can find, for every finding, either the change in the body or its disposition on the completed finding task. Nothing is unaccounted for.
* When to Use This Workflow
Trigger when:
-- A reviewer drops a review file alongside a spec — convention: same basename with a =-review.org= suffix (=foo.org= → =foo-review.org=).
+- A reviewer records findings in the spec's =* Review findings= section — typically via the spec-review workflow.
- Craig says "respond to the review" / "let's run the spec-response workflow" / "process the spec reviews."
- Any time a spec needs to absorb structured external feedback and converge to implementation-ready.
@@ -63,7 +63,7 @@ the file should be renamed first. Spec workflows require the -spec.org suffix as
guard against pointing the workflow at tutorial, inventory, or setup docs.
#+end_example
-The review file the response consumes follows the convention =<spec-basename>-review.org=, so a misnamed spec produces a mis-pointed review file too. Fix the spec name first.
+The findings the response consumes live in the spec's own =* Review findings= section, so the spec filename is the handle the workflow keys on — point it at the right =-spec.org= file. Fix the spec name first.
The user resolves the mismatch and re-invokes the workflow. Do not proceed with the response against a misnamed spec.
@@ -71,7 +71,7 @@ The user resolves the mismatch and re-invokes the workflow. Do not proceed with
** Phase 0: Orient
-1. List the review files (=ls docs/*-review.org= or wherever they live). Process them one at a time in whatever order; the user may name an order.
+1. Find specs with open findings — a =* Review findings= section whose =[/]= cookie isn't complete (=TODO= findings remain). Process them one at a time in whatever order; the user may name an order.
2. Re-read the *current* spec, not your memory of it — it may have changed since you wrote it (the user or a linter may have edited it, and pre-agreed decisions may already be encoded in the tracking file).
3. Note any *pre-agreed decisions* the reviewer or user has already settled — in the review's own "Agreed decisions" section, or in the spec's tracking task. These are settled inputs. Don't reopen them; bake them in.
@@ -79,15 +79,15 @@ The user resolves the mismatch and re-invokes the workflow. Do not proceed with
Read the entire review first. Recommendations interact — an early "medium" finding may be subsumed by a "high" one, or two findings may point at the same edit. Decide dispositions with the whole picture in view, not finding-by-finding as you scroll.
-** Phase 2: Decide a disposition for every recommendation
+** Phase 2: Decide a disposition for every finding
-For each recommendation, choose one:
+For each finding, choose one:
-- *Accept* — the recommendation is right as written. Plan the edit.
-- *Modify* — the recommendation is right in spirit but wrong in detail or scope. Adjust it, and record what you changed and why.
-- *Reject* — the recommendation doesn't fit. Record why.
+- *Accept* — the finding is right as written. Plan the edit.
+- *Modify* — the finding is right in spirit but wrong in detail or scope. Adjust it, and record what you changed and why.
+- *Reject* — the finding doesn't fit. Record why.
-*Engage critically.* Rubber-stamping is a failure mode. On a strong review most points are accepts, but actively look for the genuine modify/reject cases — they are where your judgment earns its place. Examples from the first run:
+*Engage critically.* Rubber-stamping is a failure mode. On a strong review most findings are accepts, but actively look for the genuine modify/reject cases — they are where your judgment earns its place. Examples from the first run:
- *Modify:* the review proposed an automatic cache-TTL defcustom. Accepted the goal (fresh data) but deferred TTL to vNext because for a single-user tool an explicit force-refresh + clear-cache command covers it without invalidation complexity.
- *Reject:* the review floated a separate =default-issue-filter= defcustom. Rejected as redundant — a fixed default command plus a default-view preference already covered it.
@@ -101,8 +101,8 @@ When related specs were reviewed together, two reviews can recommend opposite th
** Phase 4: Update the spec
-1. *Weave accepted recommendations into the body.* The spec should read naturally — a new "Selector semantics" section, a revised phase plan, an added test-strategy section — not a list of "review said X so I did Y." The body reflects the decisions; it doesn't narrate them.
-2. *Add a bottom "Review dispositions" section* listing only the *modified* and *rejected* recommendations, each with a short reason. Close it with a one-line "everything else accepted as written" so the reader knows the omissions from this section are accepts, not gaps.
+1. *Weave accepted findings into the body.* The spec should read naturally — a new "Selector semantics" section, a revised phase plan, an added test-strategy section — not a list of "review said X so I did Y." The body reflects the decisions; it doesn't narrate them.
+2. *Complete each finding task in place.* Accept → =DONE=, body noting where it was folded; modify → =DONE=, body noting what you changed and why; reject → =CANCELLED=, body giving the reason. The =[/]= cookie on =* Review findings= tracks progress. The reason on a modified or rejected finding is the durable record — accepted findings are recorded by the body change itself, so they need no separate note. The asymmetry is deliberate.
3. *Update or add a bottom "Review and iteration history" section.* Every response pass gets an entry, even when all findings are accepted. Each entry is an org subheading with a compound id followed by three body fields:
Heading format: =YYYY-MM-DD Day @ HH:MM:SS -ZZZZ — Contributor — Role=
@@ -113,26 +113,28 @@ When related specs were reviewed together, two reviews can recommend opposite th
- *What changed:* compact summary of accepted, modified, and rejected work.
- *Why:* the rationale or decision pressure behind the changes.
- - *Artifacts:* review filename, disposition section, task IDs, source checks, or commits when useful.
+ - *Artifacts:* the relevant findings, task IDs, source checks, or commits when useful.
4. *Flip settled decisions to =DONE=.* Each decision the decision-maker has agreed flips its =TODO= to =DONE=; the =[/]= cookie on the =* Decisions= heading tracks the tally. A contested decision stays =TODO= with the back-and-forth under its =*** Discussion= child header. Decisions still =TODO= should be only what genuinely still blocks, each with an owner and a by-when.
-5. *Raise the spec to implementation-ready:* consolidate decisions up front, add any implementation prerequisites the review surfaced (e.g. a schema-verification checklist), a consolidated test strategy, and a phased plan ordered so dependencies (like an output model everything depends on) come early. *Gate:* the spec Status cannot move past =draft= to implementation-ready while any decision is still =TODO= — the =[/]= cookie must read complete, or the author consciously accepts and records the risk of building with one open.
+5. *Raise the spec to implementation-ready:* consolidate decisions up front, add any implementation prerequisites the review surfaced (e.g. a schema-verification checklist), a consolidated test strategy, and a phased plan ordered so dependencies (like an output model everything depends on) come early. *Gate:* the spec Status cannot move past =draft= to implementation-ready while any decision or any =:blocking:= finding is still =TODO= — both =[/]= cookies must read complete, or the author consciously accepts and records the risk of building with one open. *If this response expanded scope* — folding a finding in added new phases, decisions, or external-dependency assumptions — re-run spec-review's readiness rubric against the *expanded* spec, and file any new gap as a finding or decision before claiming =Ready=. Disposition-completeness gates the *review*; the readiness rubric gates the *spec*. A response can resolve every finding and still be less ready than before, because the answers introduced unproven obligations — the cookies only protect you if the new obligation is actually filed.
6. *Update the status line* to note "review incorporated (<reviewer>, <date>)."
** Phase 5: Close out and iterate
-1. *Delete the review file* — only after every recommendation has a disposition. Its deletion is the signal the review is fully processed.
-2. *Update tracking* — the spec's VERIFY/task body gets a line noting review incorporated, what changed at a high level, which recommendations were modified (pointing at Review dispositions), and whether it's now implementation-ready pending final go.
+1. *Confirm every finding is completed* — the =* Review findings= =[/]= cookie reads complete (every finding =DONE= or =CANCELLED=). The complete cookie is the signal the review is fully processed; there is no file to delete.
+2. *Update tracking* — the spec's VERIFY/task body gets a line noting review incorporated, what changed at a high level, which findings were modified or rejected (pointing at the completed findings), and whether it's now implementation-ready pending final go.
3. *Update the session log* (state changed this turn).
-4. *Move to the next review file.* Repeat Phases 1-5 until none remain.
+4. *Move to the next spec with open findings.* Repeat Phases 1-5 until none remain.
5. *Report* what was accepted-wholesale, what was modified/rejected and why, any cross-spec reconciliations, and the implementation-ready verdict per spec.
** Phase 6: On Ready, build the implementation-task breakdown
This is the *last* step of the workflow, and it runs *only after the author confirms the spec is Ready* — never during review iterations. A Ready spec nobody can act on is unfinished; this phase turns it into tracked work. It applies to every project type (library, application, service, docs set).
-1. *Decide where the tasks live.* If the work is spinning off into its own project/repo, move the parent task into that project's =todo.org= (and relocate the spec with it); otherwise use the current project's =todo.org=. One parent task owns the effort; the phase tasks hang under it.
+*This phase owns the =READY= → =DOING= lifecycle flip* (docs-lifecycle convention): when the decomposition below lands, update the spec's top-level status heading keyword to =DOING=, add a dated history line, and set the Metadata =Status= mirror to =doing= — three lines, one file.
-2. *Create one task per implementation phase* from the spec's =Implementation phases=, in dependency order, so the task set as a whole describes the *full* milestone (e.g. v1) with no gaps. Each task body names the deliverable, its tests, and how it is verified. Carry over deferred/vNext work and any publish/release steps as their own tasks.
+1. *Decide where the tasks live.* If the work is spinning off into its own project/repo, move the parent task into that project's =todo.org= (and relocate the spec with it); otherwise use the current project's =todo.org=. One parent task owns the effort; the phase tasks hang under it. *Stamp the binding:* the parent task's =:PROPERTIES:= drawer gets a =:SPEC_ID:= line holding the spec's status-heading UUID. That property is the durable join task-audit uses to police =DOING= specs (a =DOING= spec whose bound parent is closed, archived, or missing gets flagged).
+
+2. *Create one task per implementation phase* from the spec's =Implementation phases=, in dependency order, so the task set as a whole describes the *full* milestone (e.g. v1) with no gaps. Each task body names the deliverable, its tests, and how it is verified. Carry over deferred/vNext work and any publish/release steps as their own tasks. *Always end the set with the flip task:* a final "flip the spec to IMPLEMENTED (+ dated history line + mirror)" task under the same parent — the tracked obligation that closes the lifecycle loop when the build finishes. Never skip it; "a human remembers" is the failure mode this exists to prevent.
3. *Turn a critical eye on completeness.* Re-read the spec — every phase, every acceptance criterion, every named deliverable, every data-safety/principle rule — and confirm each has a home in a task. The work is not done when the tasks merely exist; it is done when nothing in the spec is left untracked. This completeness pass is mandatory regardless of project type.
@@ -150,7 +152,7 @@ The workflow is complete when these tasks exist, the completeness pass confirms
Accept, modify, or reject — but never silently drop. The reviewer must be able to account for every point.
** Document the no's, not the yes's
-Accepted recommendations live in the spec body (the change *is* the record). Modified and rejected ones need an explicit written reason at the bottom, because the change is invisible and the reasoning would otherwise be lost. The asymmetry is deliberate.
+Accepted findings live in the spec body (the change *is* the record). Modified and rejected ones need an explicit written reason on the completed finding task, because the change is invisible and the reasoning would otherwise be lost. The asymmetry is deliberate.
** Critique, don't rubber-stamp
A review you accept entirely without finding a single thing to push on probably wasn't read critically. Your judgment — including a well-reasoned no — is the value you add.
@@ -164,11 +166,11 @@ When reviews conflict, find the framing where both are right. Silently honoring
** A reject goes back to the reviewer, not just into the file
Recording a reasoned reject is the floor, not the close. Communicate the rejection and its reason to the reviewer — a reject is a two-party event, not a unilateral call. If the reviewer disagrees, that's a discussion: weigh the counter, and if you still can't agree, escalate to whoever owns the decision rather than letting the author's "no" stand by default. "I'm not doing that" with no reason the reviewer can engage is the failure mode. (For a tight solo author-reviewer loop this is lightweight; for a team it's the difference between a review and a rubber-stamp-in-reverse.)
-** The spec reads forward, the dispositions read backward
-The body is written for the implementer (no review archaeology). The dispositions section is written for the reviewer (the reasoning trail). Keep the two audiences separate.
+** The spec reads forward, the findings read backward
+The body is written for the implementer (no review archaeology). A completed finding's reason is written for the reviewer (the reasoning trail). Keep the two audiences separate.
** The history explains provenance, not implementation behavior
-The spec body should still be the implementation contract. The bottom =Review and iteration history= section is for provenance: number of iterations, dates, contributors (including agents), roles, what each pass contributed, and why. Keep it short enough that future readers can understand how decisions evolved without rereading chats, deleted review files, or session logs.
+The spec body should still be the implementation contract. The bottom =Review and iteration history= section is for provenance: number of iterations, dates, contributors (including agents), roles, what each pass contributed, and why. Keep it short enough that future readers can understand how decisions evolved without rereading chats or session logs.
** Re-read before editing
The spec may have changed since you last saw it. Edit the current file, reconcile against the latest tracking state.
@@ -213,3 +215,8 @@ Update this workflow as we learn what works. Capture new disposition patterns, b
- *What:* Reconciled this workflow to spec-create's new Decisions convention (each decision is an org =TODO= task that flips to =DONE= on agreement, with a =[/]= cookie on the =* Decisions= heading and a =*** Discussion= child for disputes). Exit Criterion 5, Phase 2's pre-agreed-decisions step, and Phase 4 steps 4-5 now speak in flip-to-=DONE= terms, and the implementation-ready step gates on the all-=DONE= cookie.
- *Why:* The convention change landed in spec-create.org via an .emacs.d handoff (originated in its keymap-consolidation spec); this workflow still described the retired =State: proposed | accepted | superseded= model.
- *Artifacts:* Handoff =inbox/2026-06-12-1906-from-.emacs.d-spec-create-decisions-todo-note.org=. Paired spec-create.org and spec-review.org edits in the same commit.
+
+** 2026-06-21 Sun @ 23:16:06 -0400 — Claude Code (rulesets) — responder
+- *What:* Folded the review into the spec. Findings are now =* Review findings= =TODO= tasks the responder completes in place (accept/modify → =DONE=, reject → =CANCELLED= with the reason) instead of a "Review dispositions" section; the response is done when the =[/]= cookie reads complete, not when a review file is deleted. Phase 0 finds open work by an incomplete findings cookie; the Phase 4 implementation-ready gate now also requires the findings cookie, and rerun-the-readiness-rubric-on-expanded-scope is folded into that gate (a scope-expanding response must file new obligations as findings or decisions before claiming =Ready=).
+- *Why:* Deleting the review file left the iteration-history =Artifacts= line dangling and lost the verbatim review; keeping the file collided with this workflow's file discovery and its "no review files remain" done-condition. Craig's call: incorporate the review into the document, reusing the decisions machinery so the readiness signal is a cookie. The scope-expansion rerun closes a real gap — a response can resolve every finding and still introduce unreviewed obligations.
+- *Artifacts:* Paired spec-review.org edits in the same commit. Inbox handoffs =2026-06-20-2339-from-home-spec-response-readiness-gate-proposal.org= and =2026-06-21-0156-from-home-companion-to-tonight-s-spec-response.org=.
diff --git a/.ai/workflows/spec-review.org b/.ai/workflows/spec-review.org
index d956f00..0da8e65 100644
--- a/.ai/workflows/spec-review.org
+++ b/.ai/workflows/spec-review.org
@@ -5,9 +5,9 @@
* Overview
-The spec-review workflow evaluates a feature/specification document before implementation and decides one thing: can an engineer implement it confidently, test it thoroughly, and ship behavior that matches the user's mental model? If yes, say so and stop. If no, write a review file next to the spec naming every blocking gap and the concrete change that closes it.
+The spec-review workflow evaluates a feature/specification document before implementation and decides one thing: can an engineer implement it confidently, test it thoroughly, and ship behavior that matches the user's mental model? If yes, say so and stop. If no, record every blocking gap and the concrete change that closes it as findings in the spec's own =* Review findings= section.
-This is the *reviewer* side of a pair. Its counterpart is the spec-response workflow, which the spec's author runs to fold a review back in. The contract between them is the review file: =<spec-basename>-review.org= (e.g. =docs/issue-query-spec.org= → =docs/issue-query-spec-review.org=). spec-review produces it; spec-response consumes it.
+This is the *reviewer* side of a pair. Its counterpart is the spec-response workflow, which the spec's author runs to disposition the findings. The contract between them lives *in the spec*: a =* Review findings= section carrying one =TODO= task per finding, with a =[/]= cookie — the same shape the spec's =* Decisions= section already uses. spec-review writes the findings; spec-response completes them. No separate review file is written, so nothing dangles when a review is processed and the full review/response trail stays in the spec.
The goal is not to prove the spec is clever. It is to leave the implementer with *fewer* hidden decisions, not more prose.
@@ -28,7 +28,7 @@ A review is complete when:
1. *The implementation-readiness gate has been evaluated* and a rubric label assigned (=Ready= / =Ready with caveats= / =Not ready= / =Needs research=).
2. *If ready:* the user is told plainly ("This spec is implementation-ready. I have no further blocking review notes."), and the review stops — no churn for its own sake.
-3. *If not ready:* a =<spec>-review.org= file is written next to the spec, in the standard structure, with every finding specific and actionable (current behavior named, risk explained, change recommended, blocking-or-not stated).
+3. *If not ready:* findings are recorded in the spec's =* Review findings= section as =TODO= tasks (one per finding, =[/]= cookie on the heading), each specific and actionable (current behavior named, risk explained, change recommended, blocking-or-not stated).
4. *The spec's review history is updated* with who reviewed it, when, which iteration it was, what changed or was recommended, and why.
5. *Deferred work is logged* to =todo.org= (v1 = =[#B]=, vNext/someday = =[#D]=), not left only in chat.
6. *Implementation tasks are enumerated* — the spec's =Implementation phases= section is lifted into a drop-in =todo.org= block (one entry per phase plus a test-surface entry), or, if the spec has no phase decomposition, that gap is raised as a finding.
@@ -50,6 +50,11 @@ Run it *early* — design review exists to catch viability problems and costly m
Before Phase 1, verify the file under review ends with =-spec.org=. Every design, decision, or planning document under a project's =docs/= directory carries that suffix as its identifier. The =.org= extension alone is not enough because =docs/= holds non-spec org files too (tutorials, frozen inventories, reference material).
+*Location expectation (docs-lifecycle convention).* Formal specs live in =docs/specs/=. Whether that's enforced depends on whether the project has run its one-time =spec-sort= retrofit:
+
+- =:LAST_SPEC_SORT:= present in =.ai/notes.org= Workflow State → the project has sorted; a =-spec.org= file outside =docs/specs/= fails this precondition. Surface it: "this spec sits outside docs/specs/ — move it (and update inbound links) before review."
+- Marker absent → legacy locations (=docs/= root, =docs/design/=) stay reviewable; add one nudge line to the review output ("this project's docs pile has never been spec-sorted — say 'run spec-sort' to sort it") and proceed. No legacy spec is ever unreviewable during the transition.
+
If the file does not end with =-spec.org=, stop immediately and surface the mismatch:
#+begin_example
@@ -93,19 +98,19 @@ Mark the spec implementation-ready only if *all* of these hold:
- The plan can be phased without shipping broken intermediate states, and phases are small enough to reach a clean stopping point in one focused work session.
- External API assumptions are verified or explicitly listed as prerequisites.
-If all true → tell the user it's ready and stop unless they ask for more. If any false → continue and write the review file. A "ready" at this phase is provisional; confirm it at Phase 3 after the code read.
+If all true → tell the user it's ready and stop unless they ask for more. If any false → continue and record findings (Phase 5). A "ready" at this phase is provisional; confirm it at Phase 3 after the code read.
** Phase 2: Required reading order
Never review a spec in isolation.
1. *Read the existing implementation first.* The code paths the spec would touch: public commands and entry points, internal helpers/boundaries, current data representation, persistence/write-back, async/sync, caching, error handling, existing tests, naming/style. Capture current-state facts with function names and file paths. Don't recommend designs that ignore how the package works today.
-2. *Read related specs and task tracking.* Companion specs, relevant =todo.org= tasks, README/testing docs, prior review files. Record which tasks the spec absorbs, which stay separate, which decisions are already made, which are still open.
+2. *Read related specs and task tracking.* Companion specs, relevant =todo.org= tasks, README/testing docs, prior reviews (in each spec's =* Review findings= and =Review and iteration history=). Record which tasks the spec absorbs, which stay separate, which decisions are already made, which are still open.
3. *Read the target spec end to end — twice.* First for its problem/behavior/phases/assumptions; second looking only for gaps. The second read asks: "What would an implementer still have to invent?"
** Phase 3: Re-run the gate (authoritative)
-After reading code and spec, re-run the Phase 1 gate — this is the pass that counts, because now you can actually judge the items that needed the code: architecture fit, API verification, integration points. If now ready, don't manufacture churn. If not, write the review file.
+After reading code and spec, re-run the Phase 1 gate — this is the pass that counts, because now you can actually judge the items that needed the code: architecture fit, API verification, integration points. If now ready, don't manufacture churn. If not, record findings (Phase 5).
** Phase 4: Evaluate across dimensions
@@ -128,6 +133,8 @@ Work the spec against these. Each is a source of concrete findings, not a box to
- *Performance & scale.* Expected counts (issues/comments/labels/teams/projects/views)? Server-side filtering where possible? Bounded, visible pagination? Cached name→ID lookups? Sync calls in the command path acceptable? Could a save hook or whole-file scan make N network calls? Rendering linear? Full-file rewrites avoided? Long-running operations async/cancellable/observable? Is concurrency/queueing/backpressure defined? Are high-output process filters throttled and cheap? Is progress/ETA exposed only when defensible, and are hung/stalled operations detectable and killable? Identify UI freezes, repeated network calls, unbounded pagination — without premature optimization.
- *Security & privacy.* API keys safe? Debug logs leaking secrets or private issue text? Confirmations before mutating shared workspace objects? Personal vs shared distinguished? Local files holding sensitive descriptions/comments? Anything to redact from messages/logs? Any work-tracker integration may handle private company data.
- *UX & accessibility.* Discoverable commands? Recoverable mistakes? Prompts ordered to the task? Safe, useful defaults? Informative-not-noisy status messages? Does the UI avoid implying unsupported actions are supported? Match the upstream product's permissions/concepts? Are customizations named in user language, with clear defaults and docstrings? For Emacs packages, command names, completion candidates, buffer layout, defcustom names, and message wording *are* the UX.
+- *Operational-panel UI traps.* Applies when the spec covers a user-facing panel, dialog, or control surface; skip otherwise. Lists that mix saved, current, and generated items must name each item's source. Refresh or scan actions must not gate data that could be shown immediately. Add-forms must not ask the user to retype values the system already discovered. Destructive confirmations read in future tense before the action and verified-result tense after it. Diagnostics, performance, logging, and repair affordances are reviewed as one coherent flow before extra pages or buttons are added. A popup launched from a bar, tray, or tool surface should visually belong to that launcher. (Promoted from archsetup's Waybar network-panel review, 2026-06-30.)
+- *Prototype process for non-trivial UI.* Applies when the deliverable is a real UI (a panel, a multi-control surface, an interacting visual layout — not a single dialog or CLI flag); skip otherwise. Verify the =claude-rules/ui-prototyping.md= process ran: category research is cited in Goals/Design, the final prototype is linked in the design section, a =Prototype iterations= subsection under the status heading lists every pass, and each UI design decision is backed by a prototype it was seen working in rather than asserted on the page. A non-trivial-UI spec with decisions but no prototype evidence is a =:blocking:= finding.
- *Test strategy and coverage.* Characterization tests before behavior changes? Pure functions to unit-test? API responses needing fixtures? Command flows needing stubs? Regression tests for prior bugs? Boundary/error cases? What's covered elsewhere and shouldn't be re-tested? Which existing tests must change? How is coverage generated, summarized, and used to find untested/refactor-worthy code? Prefer tests that lock contracts: representation shape, query compilation, sync no-op, conflict refusal, pagination, dirty-buffer protection, log redaction, and long-running/slow-operation behavior via fakes rather than flaky live dependencies.
- *Observability & operations.* How does a user see what the package is doing? Progress messages for long ops? Useful, safe debug logging? Are logs structured enough to isolate issues from a bug report? Are commands provided to inspect/clear caches, test connectivity, diagnose backends/tools, copy redacted debug info, or reproduce command invocations? How are terminal states discovered: completion, failure, partial success, stalled/hung, cancelled, cleanup-unverified, and "needs user action"? Does the product notify only when useful, avoid noisy success spam, and keep non-success states visible until acknowledged? For generated org files, headers should often carry source, filter/view name, refresh time, count, truncation state.
- *Comparable-product sentiment.* When there are obvious adjacent products, research what users love and hate about them from official docs plus current community reports. Do not cargo-cult their feature set; translate findings into the spec's scope. For each loved behavior, say whether the spec provides it, intentionally omits it, or defers it. For each hated behavior, say whether the spec avoids, resolves, inherits, or accepts it.
@@ -136,66 +143,41 @@ Work the spec against these. Each is a source of concrete findings, not a box to
- *Development tooling.* Does the repo give contributors obvious commands for setup, fast tests, specific tests, compile, lint, coverage, cleanup, slow/manual tests, and release checks? Are optional/live tests gated by explicit environment variables? Is the Makefile/script surface consistent with sibling projects?
- *Small enhancement radar.* Are there low-complexity, high-value affordances already provided by the platform that should be surfaced now or explicitly deferred? Examples: archive/compress commands in file managers, built-in history, previews, diagnostics, or doctor commands. Keep the hot path simple; capture the opportunity rather than accidentally losing it.
-** Phase 5: Write the review file
+** Phase 5: Record findings in the spec
-Use this structure for =<spec-basename>-review.org= unless the spec calls for something different:
+Findings live in the spec, not a sibling file. Add (or append to) a =* Review findings= section near the spec's =* Decisions= section, with a =[/]= cookie on the heading. Each finding is a =** TODO= task: the heading is the smallest noun phrase naming the gap; the body names current behavior, the risk, and the recommended change. Tag a blocking (high-priority) finding =:blocking:= — it holds the rubric at =Not ready= until dispositioned; leave non-blocking findings untagged. Findings accumulate across review rounds the way decisions do, and the responder completes each one in place (Phase 4 of spec-response), so the section becomes the full review/response trail.
#+begin_src org
-,#+TITLE: Review: <Spec Title>
-,#+AUTHOR: <reviewer>
-,#+DATE: <date>
-,#+STARTUP: showall
-
-,* Scope reviewed
-What code, tests, docs, and specs you read.
-
-,* Implementation-readiness
-Whether the spec is ready. If not, summarize the blockers.
-
-,* Overall assessment
-The short senior-engineering read: what's right, what's risky, what must be clarified.
-
-,* High-priority findings
-Concrete headings. Each: why it matters and what to change.
-
-,* Medium-priority findings
-Important improvements that shouldn't block all progress.
-
-,* UX observations
-,* Architecture observations
-,* Robustness and performance observations
-,* Test strategy recommendations
-Specific test cases, not generic "add tests".
-,* Documentation and tooling recommendations
-README/user/developer docs, Makefile/package scripts, coverage, debug tools, and customization surface.
-
-,* Suggested spec edits
-Concrete edits to make the spec implementation-ready.
-
-,* Agreed decisions
-Answers reached during review. Omit if none.
-
-,* Open questions
-Only questions that truly block or materially affect implementation.
-
-,* vNext candidates
-Deferred features to capture in task tracking.
+,* Review findings [/]
+,** TODO Comment edit-back is undefined :blocking:
+The spec says fetched comments render as subheadings but doesn't define whether
+editing one syncs back. Linear only lets users edit their own comments. V1 should
+treat fetched comments as remote-owned display content and support only adding new
+comments; editing own comments can be vNext. (blocking)
+,** TODO Empty result and fetch error render identically
+A failed fetch and a successful-but-empty fetch produce the same buffer, so the
+user can't tell "no issues" from "the query broke." Define a distinct empty-state
+message. (non-blocking)
#+end_src
+Where the old review-file sub-sections go now: the scope-reviewed and overall-assessment narrative goes in the =Review and iteration history= entry (Phase 6); suggested spec edits are the recommended-change line in each finding's body; agreed decisions flip the spec's own =* Decisions= tasks; open questions are =:blocking:= findings or open decisions; vNext candidates are logged to =todo.org= as =[#D]= (Phase 6). The Phase 4 review dimensions are where findings come *from* — not headings to reproduce in the spec.
+
** Phase 6: Assign the rubric and update tracking
Assign one label consistently:
-- =Ready= — no blocking open questions; implementation can start. Requires no decision in the spec's =* Decisions= section to still be =TODO= (the =[/]= cookie reads complete; =SUPERSEDED= and =CANCELLED= count as resolved) — a decision still =TODO= holds the rubric at =Not ready=, or =Ready with caveats= if the author consciously accepts and records the risk.
+- =Ready= — no blocking open questions; implementation can start. Requires both cookies complete: no decision in =* Decisions= and no =:blocking:= finding in =* Review findings= still =TODO= (the =[/]= cookies read complete; =SUPERSEDED=/=CANCELLED= and a completed or rejected finding count as resolved) — a still-=TODO= decision or =:blocking:= finding holds the rubric at =Not ready=, or =Ready with caveats= if the author consciously accepts and records the risk. A non-blocking finding left =TODO= is author's discretion and does not hold the rubric.
- =Ready with caveats= — can start if the caveats are accepted and tracked.
- =Not ready= — blocking ambiguity / missing decisions would force implementers to invent product behavior.
- =Needs research= — external API/library/platform assumptions must be verified first.
The most useful reviews move a spec from =Not ready= to =Ready with caveats= or =Ready= once decisions are captured.
+*The =Ready= verdict flips the spec's lifecycle status.* spec-review owns the =DRAFT= → =READY= transition (docs-lifecycle convention): on assigning =Ready= (or =Ready with caveats= the author accepts), update the spec's top-level status heading keyword to =READY=, add a dated history line under it naming the review that passed, and set the Metadata =Status= mirror to =ready= — three lines, one file. Any other rubric label leaves the keyword where it stands (a re-review that finds new blockers on a =READY= spec demotes it back to =DRAFT= the same three-line way, with the reason in the history line).
+
Finding severity maps to blocking power: *high-priority findings block =Ready=* — they hold the rubric at =Not ready= (or =Ready with caveats= if the author accepts and tracks them) until dispositioned; *medium-priority findings are the author's discretion* and don't block. State the blocking status on each finding so the author running spec-response knows which ones gate the rubric.
-Then update the spec's review history. Specs should carry a bottom section named =Review and iteration history= (or the nearest existing equivalent) that tracks each material author/reviewer pass. Add a concise entry for this review even when the spec is ready and no review file is written.
+Then update the spec's review history. Specs should carry a bottom section named =Review and iteration history= (or the nearest existing equivalent) that tracks each material author/reviewer pass. Add a concise entry for this review even when the spec is ready and no findings are recorded.
Each entry is an org subheading with a compound id followed by three body fields.
@@ -207,27 +189,11 @@ Body fields:
- *What changed or was recommended:* high-signal summary, not a duplicate of the whole review.
- *Why:* the decision pressure or rationale that caused the contribution.
-- *Artifacts:* links to the review file, response/disposition section, commits, task IDs, or source checks when useful.
+- *Artifacts:* links to the relevant findings, commits, task IDs, or source checks when useful.
If the spec has no such section, add it at the bottom. Keep the history short and cumulative; it is provenance for future readers, not a session transcript.
-*Emit implementation tasks (drop-in for =todo.org=).* Read the spec's =Implementation phases= section and turn it into a paste-ready block in the review file, under a heading =Implementation tasks (drop-in for todo.org)=. One =** TODO= entry per phase, plus a final entry for the test surface. The point: the handoff to whoever implements is one paste, not a re-read of the spec, and a spec that can't be decomposed into phases fails this step, surfacing a shape problem before =Ready=.
-
-Per-phase entry, following =todo-format.md= (terse heading names the phase; body holds the one-line deliverable plus a pointer back to the spec; tags on the heading):
-
-#+begin_example
-** TODO [#B] <phase name — smallest noun phrase> :feature:
-<what this phase delivers, one line>. Spec: [[file:<spec path>]] (Implementation phases, phase N).
-#+end_example
-
-Final test-surface entry, mirroring the spec's =Acceptance criteria= when present:
-
-#+begin_example
-** TODO [#B] <feature> — test surface :test:
-Unit: <...>. Integration: <...>. E2e / manual-verify: <acceptance criteria as checkable items>. Spec: [[file:<spec path>]] (Acceptance criteria).
-#+end_example
-
-Priority and tags follow the deferred-work rule below. Emit the block in the review file; the author pastes it into =todo.org= during spec-response, or you log it directly when you're also closing the loop. If the spec has no =Implementation phases= section, don't invent one — that absence is the finding, and the step becomes the prompt to ask the author to add a phase decomposition before the spec can be =Ready=.
+*Check the spec decomposes into phases.* A =Ready= spec needs an =Implementation phases= section an implementer can turn into one task per phase plus a test surface. Confirm it's present and decomposable — each phase small enough to reach a clean stopping point in one focused session, with no broken intermediate states. If it's missing or can't be phased, file that as a =:blocking:= finding; don't invent the phases. The phase-to-task breakdown itself is spec-response's job (its Phase 6 reads =Implementation phases= directly once the author confirms =Ready=); the reviewer only verifies the section exists and is sound.
Then log deferred work to =todo.org=: v1 implementation = =[#B]= (unless urgent or speculative); vNext/someday = =[#D]=. Tag =:feature:= / =:bug:= / =:refactor:= / =:test:= / =:quick:= / =:solo:= only when accurate. Don't leave important deferred decisions only in chat.
@@ -256,8 +222,11 @@ Every material comment should be tagged by force: blocking, should-fix, or optio
** Make feedback author-usable
Review comments should be specific, neutral, and actionable: quote or name the spec behavior, explain the risk, recommend the smallest concrete change, and say how the author can verify the fix. Avoid personal language, rhetorical questions, vague "this needs work" comments, and comments that require the author to infer the desired edit.
+** Keep review and response roles explicit
+If the user asks for review plus "enhance the spec" in the same turn, produce the findings first. Make only low-risk provenance and tracking edits unless the user clearly wants the reviewer to respond too. Don't silently resolve product decisions on the author's behalf — a proposed default belongs in a finding until it's accepted, modified, or rejected.
+
** Preserve iteration provenance
-Future reviewers and implementers need to know not just the current decision, but how the spec got there: how many review/response loops happened, who contributed, what they changed or recommended, and why. Keep that record in the spec itself under =Review and iteration history= so the trail survives deleted review files, chat loss, and agent handoffs.
+Future reviewers and implementers need to know not just the current decision, but how the spec got there: how many review/response loops happened, who contributed, what they changed or recommended, and why. Keep that record in the spec itself under =Review and iteration history= so the trail survives chat loss and agent handoffs.
** Be strict about ownership
Especially for org-mode features: a user treats visible text as editable unless the representation says otherwise. Make generated-vs-editable explicit.
@@ -265,6 +234,9 @@ Especially for org-mode features: a user treats visible text as editable unless
** Never depend on an unverified API shape
If the spec assumes fields/mutations/enums, they're verified against current schema/docs/live responses, or listed as a research prerequisite. =Needs research= is a real, useful verdict.
+** Source external-dependency checks in the finding
+When a finding turns on a current external-dependency fact (release version, API capability, platform behavior, package availability, hosted-service terms), cite the checked source in the finding body. Stale dependency assumptions are common, and the next reviewer needs to tell "verified this pass" from "remembered from prior context."
+
** Favor small pure cores and thin IO layers
Push findings toward separable, unit-testable pure functions surrounded by thin command/transport layers.
@@ -354,3 +326,8 @@ Sources:
- *What:* Two refinements to the same-day decisions convention after Craig's review: the gate item and =Ready= rubric now read "no decision is still =TODO=" with =SUPERSEDED= and =CANCELLED= counting as resolved (spec-create's template defines them as done-class keywords via a =#+TODO:= header), and a spec still on the retired =State:= field model explicitly fails the gate item until converted — closing the vacuous-pass hole on old specs.
- *Why:* Review of the freshly-landed convention flagged that TODO/DONE alone lost the old model's superseded state and that the gate as written would silently pass a spec with no decision tasks at all. Craig chose the two done-class keywords and the auto-added =#+TODO:= header (the in-file header is what makes custom keywords portable).
- *Artifacts:* Paired spec-create.org edits (keyword scheme + template header) in the same commit.
+
+** 2026-06-21 Sun @ 23:16:06 -0400 — Claude Code (rulesets) — responder
+- *What:* Moved findings from a sibling =<spec>-review.org= file into the spec itself. Findings are now =** TODO= tasks under a =* Review findings= section with a =[/]= cookie, mirroring =* Decisions=; =:blocking:= marks high-priority. Phase 5 records findings in the spec instead of writing a review file; the Phase 6 =Ready= rubric gates on both the decisions and the findings cookie; the implementation-task drop-in (which lived in the review file) is gone, leaving the reviewer to verify the spec decomposes into phases and spec-response to build the breakdown. Also added two reviewer-practice principles harvested from a home spec-review: keep review and response roles explicit, and source external-dependency checks in the finding.
+- *Why:* The delete-the-review-file convention left the iteration-history =Artifacts= line dangling and dropped the verbatim review; keeping the file instead collided with spec-response's file discovery and its "no review files remain" done-condition. Craig's call: incorporate the review into the document, reusing the decisions machinery so the readiness signal is a cookie, not a file's presence or absence. The role-explicit and source-checking practices came in from the home finance-report spec via inbox handoffs.
+- *Artifacts:* Paired spec-response.org edits in the same commit. Inbox handoffs =2026-06-20-2339-from-home-spec-response-readiness-gate-proposal.org=, =2026-06-21-0156-from-home-companion-to-tonight-s-spec-response.org=, and the home-edited =2026-06-21-0156-from-home-spec-review.org=.
diff --git a/.ai/workflows/startup.org b/.ai/workflows/startup.org
index 59c9c54..2262eea 100644
--- a/.ai/workflows/startup.org
+++ b/.ai/workflows/startup.org
@@ -1,5 +1,5 @@
#+TITLE: Startup Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-04-25
* Summary
@@ -10,8 +10,8 @@ The workflow is structured into four phases. *Phase A.0* is a sequential pre-fli
Quick contract — runs / produces:
- *Phase A.0* (sequential): refresh rulesets, then the project repo.
-- *Phase A* (parallel batch): timestamp, session-context check, guarded =.ai/= sync, recent sessions, inbox-status, cross-agent status, notes.org, staleness, language-bundle freshness.
-- *Phase B* (parallel batch): read the crash-recovery anchor if present, the recent session summaries, new inbox items, pending cross-agent messages.
+- *Phase A* (parallel batch): timestamp, session-context check, guarded =.ai/= sync, recent sessions, inbox-status, notes.org, staleness, language-bundle freshness.
+- *Phase B* (parallel batch): read the crash-recovery anchor if present, the recent session summaries, new inbox items.
- *Phase C* (interactive): surface findings, process the inbox, run project startup-extras, ask priorities.
* Execution
@@ -29,10 +29,16 @@ Inside a rulesets session, the project-repo refresh below covers this — the ru
#+begin_src bash
rs="$HOME/code/rulesets"
if [ -d "$rs/.git" ]; then
- if (cd "$rs" && git diff --quiet --ignore-submodules HEAD -- 2>/dev/null); then
+ gate="$rs/claude-templates/bin/git-worktree-gate"
+ if [ -x "$gate" ] && "$gate" sync-safe "$rs"; then
+ (cd "$rs" && git pull --ff-only origin main 2>&1) | tail -3
+ elif [ ! -x "$gate" ] \
+ && (cd "$rs" && git diff --quiet --ignore-submodules HEAD -- 2>/dev/null); then
+ # Bootstrap fallback for a checkout old enough not to have the shared
+ # gate yet. The pull that follows installs it for subsequent starts.
(cd "$rs" && git pull --ff-only origin main 2>&1) | tail -3
else
- echo "rulesets: dirty working tree — using as-is, skipping pull"
+ echo "rulesets: changes beyond untracked inbox deliveries — using as-is, skipping pull"
fi
else
echo "rulesets: not a git checkout — skipping"
@@ -40,10 +46,12 @@ fi
#+end_src
Behavior:
-- *Clean working tree* → fast-forward pull. =git pull --ff-only= refuses any merge or rebase, so the operation is either a no-op (already current) or a clean advance.
-- *Dirty working tree* → skip the pull. Don't auto-stash and don't auto-merge — those would either lose work or invite conflicts at the worst possible moment (session start).
+- *Clean working tree, or untracked deliveries only beneath =inbox/=* → fast-forward pull. =git pull --ff-only= refuses any merge or rebase, so the operation is either a no-op (already current) or a clean advance. Inbox files are queue input, not source-tree work, and do not block other projects from receiving rulesets updates.
+- *Any staged or tracked change, dirty submodule, Git operation in progress, or untracked file outside =inbox/=* → skip the pull. Don't auto-stash and don't auto-merge — those would either lose work or invite conflicts at the worst possible moment (session start).
- *Non-fast-forward history* → =--ff-only= aborts with an error. Surface that to the user; the rsync still proceeds against the working tree as-is.
+*Template-freshness policy (applies to every dirty-check in the synced workflows).* The shared =git-worktree-gate sync-safe= policy is the source of truth: untracked files beneath =inbox/= and gitignored files do not block a pull, fast-forward, or monitoring gate; every other staged, tracked, untracked, submodule, or in-progress-operation state does. Projects must not fall behind merely because somebody sent them a task, but an arbitrary scratch file is not silently treated as safe. One deliberate exception remains: the rsync WIP-guard below is narrower than the repository gate and counts untracked files within rulesets' own synced source paths, because an untracked half-written template is exactly the WIP it exists to hold back.
+
*** Install rulesets symlinks into ~/.claude (idempotent)
A skill, rule, or bin script added to rulesets and pushed reaches each machine's *files* on the next pull, but not its =~/.claude= *symlink* — =make install= only links what isn't already linked, and =git pull= doesn't run it. So a newly-added skill stays silently uninstalled until someone re-runs =make install= by hand. The flush skill sat in that gap from 2026-06-02 until a manual install on 2026-06-05. Running =make install= here, right after the rulesets pull, closes it: "add a skill, commit, push" becomes enough for it to reach every machine on the next session.
@@ -72,8 +80,11 @@ if [ -d .git ]; then
current=$(git symbolic-ref --short HEAD 2>/dev/null)
dirty=0
- if ! git diff --quiet --ignore-submodules HEAD -- 2>/dev/null \
- || [ -n "$(git status --porcelain --untracked-files=no)" ]; then
+ gate="$HOME/code/rulesets/claude-templates/bin/git-worktree-gate"
+ if [ -x "$gate" ]; then
+ "$gate" sync-safe "$PWD" >/dev/null 2>&1 || dirty=1
+ elif ! git diff --quiet --ignore-submodules HEAD -- 2>/dev/null \
+ || [ -n "$(git status --porcelain --untracked-files=no)" ]; then
dirty=1
fi
@@ -105,8 +116,8 @@ fi
#+end_src
Behavior, per branch:
-- *Behind only, current branch, clean tree* → =git merge --ff-only= advances HEAD.
-- *Behind only, current branch, dirty tree* → fetched but not advanced. Surface so Craig can ff manually after dealing with the dirty state.
+- *Behind only, current branch, sync-safe tree* → =git merge --ff-only= advances HEAD. An untracked =inbox/= delivery is sync-safe.
+- *Behind only, current branch, sync-blocking tree* → fetched but not advanced. Surface so Craig can ff manually after dealing with the reported state.
- *Behind only, non-checkout branch* → =git fetch . upstream:branch= advances the ref without touching the working tree.
- *Diverged* (ahead and behind) → leave alone. Surface for Craig to resolve. Don't auto-rebase or auto-merge.
- *Ahead only* or *up to date* → silent no-op.
@@ -124,7 +135,7 @@ These calls have no dependencies on each other. Issue them all together in one m
sc=$(.ai/scripts/session-context-path 2>/dev/null || echo .ai/session-context.org)
[ -e "$sc" ] && echo "present: $sc" || echo "absent: $sc"
#+end_src
-3. *Sync =.ai/= from templates — but only when the synced source paths in rulesets are clean.* Guard the three rsyncs behind a check that =claude-templates/.ai/{protocols.org,workflows/,scripts/}= have no uncommitted changes. Otherwise Phase A copies in-flight rulesets WIP (tracked edits or new untracked files) into this project's =.ai/workflows/= and =.ai/scripts/=, where it shows up as drift the user didn't author. Skipping once is cheap — the next session with rulesets clean catches up. The check is scoped to the synced paths, so unrelated rulesets dirt (a stray =session-context.org=, scratch files) doesn't needlessly block the sync.
+3. *Sync =.ai/= from templates — but only when the synced source paths in rulesets are clean.* Guard the three rsyncs behind a check that =claude-templates/.ai/{protocols.org,workflows/,scripts/}= have no uncommitted changes. Otherwise Phase A copies in-flight rulesets WIP (tracked edits or new untracked files) into this project's =.ai/workflows/= and =.ai/scripts/=, where it shows up as drift the user didn't author. Skipping once is cheap — the next session with rulesets clean catches up. The check is scoped to the synced paths, so unrelated rulesets dirt (a stray =session-context.org=, scratch files) doesn't needlessly block the sync. A second guard skips the same rsyncs when the *project* branch is behind its upstream (=git rev-list --left-right --count @{u}...HEAD= with =behind > 0=): syncing templates onto a stale committed =.ai/= baseline measures the diff against old content, so it comes out huge and conflicts when the branch later reconciles to upstream, whose history already carries the newer templates. It composes with the rulesets-clean guard — a stable rulesets source and a current project branch are both required before the sync runs.
#+begin_src bash
rs="$HOME/code/rulesets"
@@ -132,26 +143,75 @@ These calls have no dependencies on each other. Issue them all together in one m
claude-templates/.ai/protocols.org \
claude-templates/.ai/workflows/ \
claude-templates/.ai/scripts/ 2>/dev/null)
- if [ -z "$synced_dirty" ]; then
+ # Skip the sync when the project branch hasn't reached its upstream. Syncing
+ # templates onto a behind baseline measures the diff against stale committed
+ # .ai/, producing confusing drift that conflicts when the branch reconciles —
+ # the newer .ai/ is already in upstream. behind==0 (up-to-date or ahead-only)
+ # means HEAD contains all of upstream, so the baseline is current. No upstream
+ # (new/unpushed branch) → rev-list fails → proj_behind stays 0, sync runs.
+ proj_behind=0
+ if [ -d .git ]; then
+ counts=$(git rev-list --left-right --count '@{u}...HEAD' 2>/dev/null) \
+ && [ "$(printf '%s' "$counts" | cut -f1)" -gt 0 ] 2>/dev/null \
+ && proj_behind=1
+ fi
+
+ if [ -n "$synced_dirty" ]; then
+ echo "rulesets has uncommitted changes under the synced template paths — skipping .ai/ sync this session (catches up when rulesets is clean):"
+ echo "$synced_dirty" | sed 's/^/ /'
+ elif [ "$proj_behind" -eq 1 ]; then
+ echo "project branch is behind upstream — skipping .ai/ sync this session (templates never land on a stale baseline; the sync runs once the branch is current)"
+ else
rsync -a "$rs/claude-templates/.ai/protocols.org" .ai/protocols.org
rsync -a --delete "$rs/claude-templates/.ai/workflows/" .ai/workflows/
rsync -a --delete --exclude='__pycache__' --exclude='.pytest_cache' --exclude='*.pyc' \
"$rs/claude-templates/.ai/scripts/" .ai/scripts/
echo ".ai/ synced from templates"
- else
- echo "rulesets has uncommitted changes under the synced template paths — skipping .ai/ sync this session (catches up when rulesets is clean):"
- echo "$synced_dirty" | sed 's/^/ /'
fi
#+end_src
4. =\ls -t .ai/sessions/ 2>/dev/null | head -5= — list 5 most recent session files. The backslash bypasses any =ls= alias in the user's profile. Without it, bare =ls -t= silently returns no output under =exa= (a common =ls= replacement) — which makes a sessions directory full of files look empty, and the agent then skips Phase B step 2.
5. =\ls -la inbox/ 2>/dev/null= — inventory the inbox. Same reason for the backslash escape, applied uniformly across the Phase A =ls= calls.
-6. =cross-agent-status 2>/dev/null || true= — snapshot of pending cross-agent messages across local projects. This is layer A of the cold-start design from =cross-agent-comms.org=: pending messages from other agents (delivered while no session was active here) get surfaced on session start. The =|| true= keeps Phase A from failing if =cross-agent-status= isn't installed yet — older projects without the script still boot cleanly. If HALT is active, =cross-agent-status= prints a banner; surface that prominently in Phase C.
-7. Read =.ai/notes.org= — Project-Specific Context, Active Reminders, Pending Decisions sections (skip About This File).
-8. Read =.ai/project-workflows/startup-extras.org= if it exists.
-9. =[ -f todo.org ] && .ai/scripts/task-review-staleness.sh todo.org 7 || true= — count top-level tasks overdue for review (the daily task-review habit's startup nudge). The =[ -f todo.org ]= guard skips projects without a root todo.org; =|| true= keeps Phase A from failing if the script isn't synced yet. Threshold 7 days is one review cycle of slack — softer than the wrap-up health check's 30-day alarm.
-10. =bash ~/code/rulesets/scripts/sync-language-bundle.sh "$PWD" 2>/dev/null || true= — language-bundle freshness for the current project. Fingerprint-detects which bundle (if any) the project has, auto-fixes drifted rulesets-owned files (=.claude/rules/*.md=, =.claude/hooks/*=, =githooks/*=), and surfaces drift in =settings.json= without writing it (a project may have customized it). =CLAUDE.md= is deliberately left untracked — it's seed-only in =install-lang= and project-owned afterward, mirroring how =diff-lang= skips it. Quiet when there's no bundle or everything's clean. Hardcodes the rulesets path because =languages/= is the canonical source and lives only there — the same absolute-path dependency the rsyncs already carry. =|| true= keeps Phase A from failing on older checkouts where the script isn't present yet. The =.ai/= rsyncs and this call write to disjoint paths (=.ai/= vs =.claude/=/=githooks/=), so the batch stays parallel-safe.
-11. =[ -f "$HOME/org/roam/inbox.org" ] && grep -cE '^\*\* ' "$HOME/org/roam/inbox.org" || true= — count items in the roam global inbox (=~/org/roam/inbox.org=), the inbox-zero startup nudge. Silent if the roam clone isn't on this machine. Phase C reads the file when the count is non-zero, splits total vs items related to this project, and surfaces the offer (see =inbox-zero.org=). Read-only; never files at startup.
+6. Read =.ai/notes.org= — Project-Specific Context, Active Reminders, Pending Decisions sections (skip About This File).
+7. Read =.ai/project-workflows/startup-extras.org= if it exists.
+8. =[ -f todo.org ] && .ai/scripts/task-review-staleness.sh todo.org 7 || true= — count top-level tasks overdue for review (the daily task-review habit's startup nudge). The =[ -f todo.org ]= guard skips projects without a root todo.org; =|| true= keeps Phase A from failing if the script isn't synced yet. Threshold 7 days is one review cycle of slack — softer than the wrap-up health check's 30-day alarm.
+9. =bash ~/code/rulesets/scripts/sync-language-bundle.sh "$PWD" 2>/dev/null || true= — language-bundle freshness for the current project. Fingerprint-detects which bundle (if any) the project has, auto-fixes drifted rulesets-owned files (=.claude/rules/*.md=, =.claude/hooks/*=, =githooks/*=), and surfaces drift in =settings.json= without writing it (a project may have customized it). =CLAUDE.md= is deliberately left untracked — it's seed-only in =install-lang= and project-owned afterward, mirroring how =diff-lang= skips it. Quiet when there's no bundle or everything's clean. Hardcodes the rulesets path because =languages/= is the canonical source and lives only there — the same absolute-path dependency the rsyncs already carry. =|| true= keeps Phase A from failing on older checkouts where the script isn't present yet. The =.ai/= rsyncs and this call write to disjoint paths (=.ai/= vs =.claude/=/=githooks/=), so the batch stays parallel-safe.
+10. =[ -f "$HOME/org/roam/inbox.org" ] && grep -cE '^\*\* ' "$HOME/org/roam/inbox.org" || true= — count items in the roam global inbox (=~/org/roam/inbox.org=), the roam-mode startup nudge. Silent if the roam clone isn't on this machine. Phase C reads the file when the count is non-zero, splits total vs items related to this project, and surfaces the offer (see =inbox.org= roam mode). Read-only; never files at startup.
+11. KB surface prep (the read + contribute startup nudges; see =docs/specs/2026-06-16-encourage-kb-contribution-spec.org=). Gated on the agent KB clone. Counts =:agent:= nodes, lists up to 5 whose content matches the current project basename (titles only; a few most-recent nodes as a fallback when nothing matches), and resolves the best-practices node path. Read-only; silent when the clone is absent. Phase C surfaces the relevant titles (consult) and the best-practices link (contribute).
+
+ The best-practices lookup matches the node's *filename*, not its content. A roam node's slug lives only in its filename, so the earlier content-grep (=rg -l 'agent-kb-best-practices'=) matched nothing and the contribute nudge silently pointed at an empty path in every project, every session, for as long as it shipped. =find= rather than a glob keeps the probe identical under bash and zsh (zsh aborts on an unmatched glob) — the same reason the spec-sort probe below uses =find=.
+
+ #+begin_src bash
+ ra="$HOME/org/roam/agents"
+ if [ -d "$ra" ]; then
+ proj=$(basename "$PWD")
+ echo "kb-total: $(rg -l '#\+filetags:.*:agent:' "$ra" 2>/dev/null | wc -l)"
+ echo "kb-bestpractices: $(find "$ra" -maxdepth 1 -name '*agent-kb-best-practices*.org' -print -quit 2>/dev/null)"
+ matches=$(rg -il "$proj" "$ra" 2>/dev/null | head -5)
+ [ -z "$matches" ] && matches=$(\ls -t "$ra"/*.org 2>/dev/null | head -3)
+ echo "kb-relevant-titles:"
+ for f in $matches; do rg -m1 '^#\+title:' "$f" 2>/dev/null | sed 's/^#+title:/ -/'; done
+ fi
+ #+end_src
+
+12. Spec-sort probe (the docs-lifecycle retrofit nudge; see the docs-lifecycle spec in =docs/specs/=). Read-only; prints one line when the project has an unsorted docs pile — a =docs/design/= directory or stray =docs/*-spec.org= root files — and no =:LAST_SPEC_SORT:= marker in =.ai/notes.org=. Silent for projects with nothing to sort or an already-stamped marker (the marker permanently clears it).
+
+ #+begin_src bash
+ { [ -d docs/design ] || [ -n "$(find docs -maxdepth 1 -name '*-spec.org' -print -quit 2>/dev/null)" ]; } \
+ && ! grep -qs ':LAST_SPEC_SORT:' .ai/notes.org \
+ && echo "spec-sort: unsorted docs present" || true
+ #+end_src
+
+ The stray-root check uses =find= rather than a glob so the probe behaves identically under bash and zsh (=compgen= is bash-only, and zsh aborts on an unmatched glob).
+
+13. Host-identity probe (see the host-identity rule in =claude-rules/=). Read-only; flags fixed machine-identity claims in the project's tracked/synced docs — the "This machine is ratio" trap, false on every machine but the one that wrote it. Silent when nothing matches.
+
+ #+begin_src bash
+ grep -inE '\b(this|the current) (machine|host|box|laptop|workstation) is ' \
+ CLAUDE.md .ai/notes.org 2>/dev/null | head -3 || true
+ #+end_src
+
+ Fleet descriptions ("the fleet is ratio and velox") and runtime derivations ("run =uname -n= to find the hostname") don't match — only current-identity assertions do. Fixture-verified under bash and zsh.
Notes on the rsync commands:
- Trailing slashes on both source and destination matter — they tell rsync to sync /contents/ rather than nest a directory inside.
@@ -159,6 +219,7 @@ Notes on the rsync commands:
- protocols.org is a single file, no =--delete= needed.
- The =scripts/= sync excludes Python build artifacts (=__pycache__/=, =.pytest_cache/=, =*.pyc=). Running rulesets' own pytest leaves these in =claude-templates/.ai/scripts/tests/=, and =rsync -a= copies by disk presence regardless of =.gitignore=, so without the excludes every consuming project's tree gets polluted with machine-specific cache files. The excludes also protect existing dest copies from =--delete= cleanup, so a project that already received the cache must remove it once by hand.
- The sync is guarded to skip when rulesets has uncommitted changes under the synced source paths. =rsync -a --delete= copies the working tree by disk presence, so without the guard a downstream session started while rulesets had in-flight WIP would pull that WIP into its =.ai/workflows/= and =.ai/scripts/=, surfacing as drift the user never authored (and tempting a fake "chore: sync .ai tooling" commit). The guard is scoped to the synced paths, not the whole repo, so unrelated rulesets dirt doesn't block the sync. From the jr-estate handoff 2026-05-29.
+- The sync is also guarded to skip when the *project* branch is behind its upstream (=proj_behind=). Phase A.0 correctly declines to fast-forward a diverged or behind-and-dirty branch, but the rsync would then land templates on the stale committed =.ai/= baseline — a huge diff measured against old content that conflicts once the branch reconciles to upstream's newer templates. Skipping is safe: the sync runs next session once the branch is current. Not an auto-discard — startup never =git checkout=s drift away, because a legitimate local stopgap in a synced file is indistinguishable from accidental drift by content alone (home reverted an intentional =flashcard-to-anki.py= fix this way on 2026-06-22). Prevention is safe; blind cleanup-after is not. Phase C's template-sync-churn safety net still surfaces any pre-existing dirt for a human decision. From the home handoff 2026-07-04.
- The sync touches only =protocols.org=, =workflows/=, and =scripts/=. The project-owned dirs =project-workflows/= and =project-scripts/= are deliberately *outside* the synced set, so a project's own workflows and scripts survive startup. This is why a project script that a workflow imports must live in =.ai/project-scripts/=, never =.ai/scripts/= — the latter is wiped to match the template by =--delete= on every startup. Naming: a script imported as a Python module needs an importable name (underscores, e.g. =zlibrary_api.py=); a CLI-invoked script can stay kebab-case like the template tooling (=cmail-action.py=).
Rationale: Every call in Phase A is read-only or writes to a distinct path. Running them sequentially wastes round-trips; running them in parallel gives Claude the complete starting picture in one round-trip.
@@ -170,7 +231,6 @@ These calls depend on Phase A outputs, but are independent of each other. Issue
1. *Read =.ai/session-context.org= if Phase A reported it exists.* The file is the crash-recovery anchor — if it's there, the previous session was interrupted and the context lives only in this file.
2. *Read each of the 5 most recent session files* from Phase A's =\ls -t .ai/sessions/= output. Read just the =* Summary= section of each — not the full file. The Summary gives Active Goal / Decisions / Data Collected / Findings / Files Modified / Next Steps. That's enough to pick up where things left off. Drill into a specific =* Session Log= later only if you need the /why/ or sequence on something. *If Phase A's listing came back empty, sanity-check with =\ls -la .ai/sessions/= before treating empty as definitive — sessions/ should normally be populated, and an empty result usually means the listing got swallowed somewhere, not that the directory is genuinely empty.*
3. *Read each new inbox file* from Phase A's =\ls -la inbox/= output. For =.eml= files, defer to Phase C — those need the extract script (below) rather than a raw Read.
-4. *Process pending cross-agent messages.* For each project with a pending count >0 in Phase A's =cross-agent-status= output (typically the current project; cross-project pending is surfaced too but only acted on if the user asks), run =cross-agent-recv <message-file>= on the file path =cross-agent-status= named. The script returns a structured decision (=process= / =dedup= / =query= / =reject=) per the protocol. For =process=, read the message body to determine the action. For =query=, prepare a clarifying reply. For =reject=, surface to user with the reason. For =dedup=, no action — silent retry already handled. Surface all decisions in Phase C alongside other findings.
Rationale: Reads are independent and benign. Batching them means the whole session-history view + inbox view lands in one round-trip instead of one per file.
@@ -184,7 +244,11 @@ This phase touches the user and runs sequentially:
- Mention Pending Decisions from notes.org.
- Briefly note significant template updates noticed during sync (new workflows, protocol changes).
- *Task-review nudge.* If the Phase A staleness count (step 11) is greater than zero, surface one line: "=<N>= top-level tasks unreviewed for >7 days — say 'let's do a task review' to run a cycle." If zero, say nothing.
- - *Roam inbox nudge.* If the Phase A roam-inbox count is greater than zero, read =~/org/roam/inbox.org=, split total vs items related to this project (claimed by the =<project>:= prefix, plus any unprefixed item whose topic plainly concerns this project), and surface one line: "Roam inbox: =<N>= total, =<M>= appear related to this project — say 'inbox zero' to file them." Offer it as a priority option; never auto-file. If the count is zero or the file is absent, say nothing. See =inbox-zero.org=.
+ - *Roam inbox nudge.* If the Phase A roam-inbox count is greater than zero, read =~/org/roam/inbox.org=, split total vs items related to this project (claimed by the =<project>:= prefix, plus any unprefixed item whose topic plainly concerns this project), and surface one line: "Roam inbox: =<N>= total, =<M>= appear related to this project — say 'inbox zero' to file them." Offer it as a priority option; never auto-file. If the count is zero or the file is absent, say nothing. See =inbox.org= roam mode.
+ - *KB consult nudge (read side).* If the Phase A KB-surface prep returned any =kb-relevant-titles=, surface one line listing them (capped 5): "KB lessons that may be relevant: =<title>=; =<title>=… — open the node before related work." The titles are declarative, so the list alone tells you whether to open one. Gated on the roam clone; silent when the clone is absent or nothing relevant surfaced. See the best-practices node and =knowledge-base.md=.
+ - *KB contribute nudge (write side).* Once per session, surface one line pointing at the best-practices node (the =kb-bestpractices= path from Phase A): "Learned something durable? See =<path>= for how to write a KB node — contributing cross-project facts is welcome (personal projects only; work/unknown projects never write per =knowledge-base.md=)." Light encouragement, never a gate. Gated on the roam clone; silent when absent.
+ - *Spec-sort nudge.* If the Phase A spec-sort probe printed =spec-sort: unsorted docs present=, surface one line: "this project's docs pile has never been spec-sorted — say 'run spec-sort' to sort it." If the probe was silent, say nothing. A project with nothing to sort never sees the line; a stamped =:LAST_SPEC_SORT:= marker permanently clears it. See the docs-lifecycle rule and the spec in =docs/specs/=.
+ - *Host-identity flag.* If the Phase A host-identity probe printed any match, surface it with the file:line and the fix: "this doc asserts a fixed machine identity — false on every other machine; replace with a runtime derivation (run =uname -n=), per the host-identity rule." The probe flags for judgment, never blocks. Silent when the probe is silent.
- *Language-bundle sync.* If the Phase A step-12 call (=sync-language-bundle.sh=) printed anything, surface it. =fixed= lines are informational — the drift was already repaired (note that =.claude/= is now dirty if the project commits it). A =drift= line on =settings.json= is surface-only and needs the printed =make install-<lang> PROJECT=.= to reconcile; flag it so the user can decide. If the call was silent, say nothing.
- *Newly-installed symlinks.* If the Phase A.0 =make install= step printed any =link= / =relink= / =WARN= line, surface it. A =link= line means a skill, rule, hook, or script added to rulesets is now linked into =~/.claude= for the first time on this machine. For a newly-linked *skill*, check the agent's available-skills list: if the harness already registered it mid-session, note it's available and move on; if it's absent, stop and tell Craig to restart the agent so it loads (whether a mid-session reload works is harness-version-dependent). For a newly-linked *hook*, note that the harness reads hooks at session start — it fires from the next session (or after Craig opens =/hooks= once); its settings.json wiring travels with the tracked file, so the link is usually the only missing piece. A =WARN ... not a symlink= line is a real collision at the target path — surface it; it needs a human. If the step printed only "nothing new to link", say nothing.
- *Template-sync churn (safety net).* Check whether Phase A's rsync left uncommitted churn in the synced =.ai/= paths — accumulated from a prior session that crashed before wrap-up, or freshly added this session when rulesets advanced. Without surfacing, it builds up silently until it blocks Phase A.0's auto-ff (git won't ff a dirty tree). Skip in the rulesets repo itself (there =.ai/= is a committed mirror, kept honest by the pre-commit hook). The check is sequential here, after the rsync has finished — not a Phase A step, to keep that batch race-free.
@@ -197,8 +261,7 @@ This phase touches the user and runs sequentially:
#+end_src
If it reports a count, surface one line: wrap-up's Step 4.0 will commit it as =chore: sync .ai tooling from templates=, or offer to commit it now. If silent, say nothing. This is the crashed-session counterpart to the wrap-up commit step (the primary fix). From the 2026-05-31 jr-estate + work handoffs.
- - *Surface pending cross-agent messages.* If =cross-agent-status= reported any pending messages, list them with their =cross-agent-recv= decision (process / query / reject) per file. For =process= messages in this project's inbox, propose handling now or after the current task. For pending in other projects, mention the count so the user knows to switch projects when ready. If HALT was active, surface that prominently — cross-agent activity is paused until =cross-agent-resume= clears it.
-2. *Process inbox if non-empty.* Mandatory — don't ask, just delegate to [[file:process-inbox.org][process-inbox.org]]. That workflow owns the value gate (advances an existing TODO / improves the project / serves the mission), the per-source rejection flow (Craig / project handoff / script), the priority-scheme check before filing, and the =.eml= extraction path. Single source of truth for the discipline.
+2. *Process inbox if non-empty.* Mandatory — don't ask, just delegate to [[file:inbox.org][inbox.org]] process mode. That mode owns the value gate (advances an existing TODO / improves the project / serves the mission), the per-source rejection flow (Craig / project handoff / script), the priority-scheme check before filing, and the =.eml= extraction path. Single source of truth for the discipline.
3. *Execute project-specific startup extras* (the contents of =.ai/project-workflows/startup-extras.org= read in Phase A). If the file didn't exist, skip.
4. *Ask about priorities.* "What would you like to work on, or is there something urgent you need?"
- If urgent: proceed immediately.
diff --git a/.ai/workflows/status-check.org b/.ai/workflows/status-check.org
index efff16d..4a9972c 100644
--- a/.ai/workflows/status-check.org
+++ b/.ai/workflows/status-check.org
@@ -1,5 +1,5 @@
#+TITLE: Status Check Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-02
* Overview
diff --git a/.ai/workflows/summarize-emails.org b/.ai/workflows/summarize-emails.org
index 6ac5e6f..c9c7001 100644
--- a/.ai/workflows/summarize-emails.org
+++ b/.ai/workflows/summarize-emails.org
@@ -1,5 +1,5 @@
#+TITLE: Summarize Emails Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-14
* Overview
diff --git a/.ai/workflows/suspend.org b/.ai/workflows/suspend.org
new file mode 100644
index 0000000..166f9c9
--- /dev/null
+++ b/.ai/workflows/suspend.org
@@ -0,0 +1,143 @@
+#+TITLE: Session Suspend Workflow
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-06-28
+
+* Overview
+
+This workflow captures the live state of a session when Craig must leave
+abruptly, so a future session resumes with nothing lost. It is the fast,
+capture-only workflow for departure: it writes down where every thread stands,
+notes any uncommitted work, then STOPS — no cleanup, no archive, no teardown.
+
+Triggered by Craig saying "suspend the session," "suspend," "I need to go,"
+"stick a pin in everything," or similar. "I need to go" is broad — if it reads
+as a conversational aside rather than a request to suspend, confirm before
+running.
+
+* Where suspend sits among its neighbors
+
+Three workflows touch the session anchor (=.ai/session-context.org=); keep them
+straight:
+
+- =flush= ([[file:../../flush/SKILL.md]] / =/flush=) — *stay and sharpen.*
+ Refreshes the anchor in place, prompts Craig to type =/clear=, and a hook
+ resumes the *same* logical session in a fresh context. Craig is still here.
+- *suspend* (this workflow) — *leave.* Captures richly into the anchor, leaves
+ the file in place, detaches the tmux client so the session parks in the
+ re-attachable set, and Craig walks away. The next session is a cold startup
+ that detects the present anchor and resumes from it — or Craig re-attaches the
+ still-live session directly.
+- =wrap-it-up= ([[file:wrap-it-up.org][wrap-it-up.org]]) — *end.* Writes the
+ Summary, archives the anchor into =.ai/sessions/=, commits + pushes, and runs
+ the phrase-dependent teardown.
+
+Suspend and flush share one core — capture into the anchor, leave it in place.
+They differ in the exit (leave vs clear-and-continue) and the resume path
+(startup vs the =/clear= hook). Suspend reuses flush's capture discipline (its
+Phase 1 anchor-refresh) rather than restating it, and adds a richer,
+resume-weighted Session Log entry because it's written for a cold resume after a
+gap, not a same-session reset.
+
+* Suspend vs wrap-up — the one structural difference
+
+=wrap-it-up= ARCHIVES =.ai/session-context.org= (renames it into
+=.ai/sessions/=); its absence at the next startup is the signal that the last
+session ended cleanly.
+
+Suspend does the opposite: it LEAVES =.ai/session-context.org= in place. Its
+presence at startup is exactly the signal that the previous session was
+interrupted, so the startup workflow reads it and resumes. Suspend provides only
+the *capture* half — startup's existing interrupted-session path (Phase A checks
+for the anchor, Phase B reads it, Phase C offers to resume) is the *resume* half,
+already built.
+
+So: never archive, never rename the context file in a suspend. Capture into it
+and leave it.
+
+* What gets captured
+
+The point is zero lost information, weighted toward RESUME. Into the
+=* Session Log= of =.ai/session-context.org=, append one dated
+=** YYYY-MM-DD ... — SUSPENDED= entry holding:
+
+1. *Open threads — resume here.* For each active or pending thread: the topic,
+ its status (ACTIVE / PINNED / SET ASIDE / DEFERRED), the immediate next
+ step, and the pointers needed to act on it cold (files + line numbers,
+ commit SHAs, the specific finding or decision). This is the core; spend the
+ most words here. Order newest / most-active first.
+2. *Pending decisions / open questions* awaiting Craig — anything blocked on
+ his input, with enough context that the answer is actionable.
+3. *Shipped this session* — a terse list of what landed, each with its commit
+ SHA, so the resume knows what is already done and need not re-derive it.
+4. *Uncommitted work* — anything modified on disk but not committed, named
+ file by file, so the resume knows what state the tree is in.
+5. *Key findings not yet recorded elsewhere* — anything learned this session
+ that isn't already in a commit, a file, or memory, so it survives.
+6. *Background work* — any running task, agent, or job, and how to check it.
+7. *Resume hint* — the single most likely "start here" next action.
+
+Also update the top of =* Summary= (Active Goal) with a one-line SUSPENDED
+pointer to the entry, so startup reading the top sees the current state even
+when the Summary body is from an earlier thread.
+
+* Steps
+
+1. *Write the SUSPENDED entry* into the Session Log, per "What gets captured"
+ above. Timestamp with =date "+%Y-%m-%d %a @ %H:%M:%S %z"=.
+2. *Update the Active Goal pointer* at the top of =* Summary=.
+3. *Record uncommitted work, don't force-commit it.* A suspend records state, it
+ does not tidy it. Name every uncommitted change in the SUSPENDED entry and
+ leave the tree as it is — on an abrupt departure, a dirty tree (like any
+ crash) is safer than a blind commit of arbitrary mid-work state. (If a
+ project defines a standing always-commit set in its own workflow, commit only
+ that set — but the default shared behavior is to leave the tree alone.)
+4. *Leave =.ai/session-context.org= in place.* Do not archive it.
+5. *Brief handoff* — one or two lines: what was captured, where the resume
+ pointer is, the most-active thread. This is the last thing Craig sees before
+ the view detaches (Step 6), so deliver it complete.
+6. *Detach the tmux client.* As the final action, detach the client viewing the
+ =aiv-<project>= session so it drops out of Craig's active view while staying
+ alive in the background. This is a DETACH, not a teardown: the session and the
+ agent process keep running, nothing is killed, no context is lost.
+
+ #+begin_src bash
+ sess=$(tmux display-message -p '#S' 2>/dev/null)
+ [ -n "$sess" ] && tmux detach-client -s "$sess"
+ #+end_src
+
+ Run it as the very last tool call, after the handoff text has rendered — tmux
+ preserves the pane, so Craig sees the full handoff when he re-attaches. Unlike
+ wrap-up's teardown (which must defer to a =Stop= hook because it kills the
+ session the agent runs in, which would cut off the valediction), detach runs
+ inline: it disconnects the view but leaves the agent's session alive, so
+ nothing is cut off. Degrade gracefully — if not inside tmux (=$TMUX= unset, no
+ session), skip silently and the session simply stays attached.
+
+ Why detach on every suspend: Craig cycles his live agent sessions in Emacs
+ with alt-space, and rotates through everything — including re-attaching
+ detached ai-term sessions — with shift+alt+space. A suspended session left
+ attached clutters the active rotation; detaching parks it in the
+ re-attachable set, which is what makes suspend-and-walk-away work. Re-attach
+ is one keystroke (shift+alt+space) or =tmux attach -t aiv-<project>=.
+
+* What suspend does NOT do
+
+Speed over completeness. A suspend deliberately skips everything wrap-it-up
+does beyond capture:
+
+- No =* Summary= rewrite beyond the one-line Active Goal pointer.
+- No todo.org cleanup / archive-done.
+- No KB / memory promotion sweep.
+- No Linear / board reconciliation.
+- No session-record archive (the file stays live).
+- No teardown. Suspend DETACHES the tmux client (Step 6) but never kills the
+ session: the =aiv-<project>= session and the agent process stay alive in the
+ background, only the view disconnects. It drops no =Stop=-hook teardown
+ sentinel, so the wrap-teardown hook stays dormant. Teardown — killing the
+ session — is wrap-it-up's job, not suspend's; detach is the lighter move that
+ parks a still-live session.
+- No blind commit of working files (step 3).
+- No valediction. A suspend is a pause, not a goodbye.
+
+If Craig later wants the clean end, he runs wrap-it-up, which picks up the
+captured state and finishes the job.
diff --git a/.ai/workflows/sync-email.org b/.ai/workflows/sync-email.org
index 52a7caf..863b400 100644
--- a/.ai/workflows/sync-email.org
+++ b/.ai/workflows/sync-email.org
@@ -1,5 +1,5 @@
#+TITLE: Sync Email Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-01
* Overview
diff --git a/.ai/workflows/task-audit.org b/.ai/workflows/task-audit.org
index 67ce496..aa50176 100644
--- a/.ai/workflows/task-audit.org
+++ b/.ai/workflows/task-audit.org
@@ -1,5 +1,5 @@
#+TITLE: Task Audit Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-22
* Overview
@@ -61,6 +61,8 @@ For each open task, read its body and cross-check its claims against the actual
- *Calendar* — did a scheduled event happen; is a SCHEDULED/DEADLINE date now past.
- *Meeting recordings* — if a task hinges on "did this conversation happen / what was said," check the recording queue (e.g. =~/sync/recordings/=) and transcribe via =process-meeting-transcript.org= if the answer lives in an un-transcribed recording. (This is exactly how a "did the interview happen?" task gets resolved instead of guessed.)
+*Spec lifecycle reconcile (docs-lifecycle convention).* If the project has a =docs/specs/=, run the =:SPEC_ID:= query as part of this phase: for each spec whose top-level status heading reads =DOING=, find the =todo.org= task whose =:SPEC_ID:= property matches the spec's =:ID:=. Flag the spec NEEDS-USER when that bound parent is =DONE=/=CANCELLED=, archived, or missing — the build finished (or evaporated) without the =IMPLEMENTED= flip, exactly the drift this check exists to catch. Check the parent's own keyword, not its children (completed children become dated entries and the final flip task is a child, so child-counting misleads).
+
Assign each task a bucket (CURRENT / STALE / NEEDS-USER) and, for STALE, the specific factual update.
*Scale tactic.* For a large open-task set, dispatch read-only investigation sub-agents over batches of tasks (parallel-safe per =subagents.md= — independent read-only domains). Each returns a per-task bucket + suggested update. *Never* let sub-agents write to =todo.org= concurrently — apply all edits serially in the main thread (concurrent writes to one file race and lose work).
@@ -79,11 +81,41 @@ For every STALE task, edit it in the main thread:
- *Ensure priority is set per the project scheme.* The top of the project's =todo.org= should carry the priority legend (=[#A]= through =[#D]=). Every task should carry an explicit priority cookie. If a cookie is missing, or no longer matches the reconciled facts, assign the right level per the legend. If the level is unambiguous from the body, do it autonomously; if it's a judgment call (especially the [#A] / [#B] line for important-but-not-urgent work), flag NEEDS-USER. Also enforce the [#A]-discipline rule from the legend — an [#A] task without a =SCHEDULED:= or =DEADLINE:= line is mis-graded and is either down-graded to [#B] (when reconciled facts say "important but not urgent") or surfaced as NEEDS-USER for the user to date.
- *Ensure a type tag is set.* Every task carries one type tag from the project's tag legend (typically =:feature:= / =:chore:= / =:spec:= / =:bug:=). If missing or wrong, assign or correct it from the body when the type is unambiguous. If two tags fit (a refactor that also fixes a bug; a spec that's also a chore), flag NEEDS-USER rather than picking one silently.
- *Enforce the project's declared tag vocabulary.* If the project's tag legend declares an *exhaustive* set of allowed tags, strip from each task any tag outside that set — the heading and parent section already carry topic/scope context, so ad-hoc tags only fragment the vocabulary and defeat tag-based filtering. Normalize near-duplicate spellings to the canonical tag (a plural to its singular, say). Where the legend does not declare the set closed, leave existing tags alone; this step applies only where the allowed set is exhaustive by design.
-- *Re-assess the =:quick:= and =:solo:= tags* — reconciliation can change a task's effort or autonomy: a resolved dependency may make a stuck task =:solo:=, a scope cut may make it =:quick:=, and new complexity surfaced by the sources can invalidate either. Add or remove the tags per the definitions in the project's tag legend (and [[file:task-review.org][task-review.org]]) when the reconciled facts make the call clear. When they don't — an effort estimate you can't pin down, a =:solo:= gate you can't confirm — it's a NEEDS-USER flag, not a guess.
+- *Re-assess the =:quick:= and =:solo:= tags (mandatory — an audit that skips this is incomplete).* Reconciliation can change a task's effort or autonomy: a resolved dependency may make a stuck task =:solo:=, a scope cut may make it =:quick:=, and new complexity surfaced by the sources can invalidate either. Add or remove the tags per the hard definitions in [[file:../../claude-rules/todo-format.md][todo-format.md]] ("Hard definitions: :solo: and :quick:"; task-review carries the same three-gate walk). Autonomous execution reads =:solo:= as its eligibility gate and trusts the tag, so a stale one is a run-time hazard, not cosmetic drift. When the call isn't clear — an effort estimate you can't pin down, a =:solo:= gate you can't confirm — it's a NEEDS-USER flag, not a guess.
- Bump =:LAST_REVIEWED:= on each edited task.
Follow =todo-format.md= for completion mechanics (depth-based DONE vs dated-rewrite) and the working-files / link-hygiene rules when moving artifacts.
+** Phase C.5 — Consolidate related tasks (interactive)
+
+Phase C's *Consolidate duplicates* bullet folds tasks that track the *same* thing. This step is the broader case: tasks that aren't duplicates but are really *one effort* fragmented across the list. A spread-out effort — several tasks all circling "make the tooling agent-agnostic," say — is harder to see, plan, and finish as a whole than one task, or one parent with the pieces as children.
+
+After the Phase C edits, read the open-task set as a whole and look for *clusters*: tasks that share a goal, a subsystem, or an obvious sequence. Use judgment over the task bodies, not a keyword heuristic — adjacency is a semantic call, and a brittle title-match both misses real clusters and invents false ones.
+
+For each cluster, surface it to Craig (inline numbered options per =interaction.md=, no popup) with a recommendation, offering the two shapes:
+
+- *Merge* — fold the cluster into one task when the members are genuinely the same work split up (near-duplicates, or steps with no independent value). The merged task keeps the strongest priority, unions the type tags, and absorbs each member's body as a dated note or a short list; the absorbed tasks close per =todo-format.md= (a =**= task → =CANCELLED= + =CLOSED:= with a one-line "merged into <task>", or deletion if it carried nothing unique).
+- *Parent with children* — when the members are related but distinct (each ships independently or has its own value), promote a parent task and re-home the members beneath it as sub-tasks, so the list shows the effort as a unit without losing the individual pieces.
+
+Never merge or re-parent autonomously — which tasks belong together, and whether they're one-work or related-distinct, is a judgment only Craig ratifies. Propose, don't apply, until he picks. A cluster he declines stays as separate tasks; don't re-surface it every audit (note the decline in the session log).
+
+When no clear cluster exists, say so in one line and move on — most audits won't find one, and forcing a merge fragments worse than it consolidates.
+
+** Phase C.6 — Retire completed parents and promote stragglers (interactive)
+
+Phase C.5 consolidates related *open* tasks. This step retires parent tasks whose work is *finished*, so completed containers don't linger in Open Work as scaffolding.
+
+Run =todo-cleanup.el --convert-subtasks= first (it's part of the =clean-todo= / wrap-up cleanup, and =open-tasks.org= runs it too) so every completed sub-task is a dated event-log entry rather than a lingering =DONE= keyword. The closure logic below reads "open child" as a child heading still carrying a task keyword (=TODO=/=DOING=/=WAITING=/=VERIFY=/=NEXT=/=PROJECT=/=STALLED=/=DELEGATED=); a dated entry is correctly not open.
+
+Two shapes, both proposed to Craig (inline numbered options per =interaction.md=, no popup) before applying:
+
+- *Zero open children → close the parent.* A parent whose child *tasks* are all resolved (now dated) and that carries no open child task is finished: close it per =todo-format.md= (=**= parent → =DONE=/=CANCELLED= + =CLOSED:=), and it moves to Resolved on the next =--archive-done=. If the work resurfaces later, a fresh task is created then; a completed container shouldn't sit open as a placeholder.
+- *One or two open children → promote, then close.* When a parent has only one or two open children, pull them out and rewrite them as standalone =**= level-2 tasks — give each a priority per the project scheme, and make the heading stand alone without the parent's context — then close the now-childless parent and let it move to Resolved. The former children become first-class Open Work tasks; the retired parent stops being scaffolding for one or two stragglers.
+
+*The leaf-with-notes carve-out (important).* "Zero open children" is not the same as "done." A =**= leaf task whose only descendants are dated *notes* — a captured "Ideas", "Goals", or "Current State" entry, not a real completed sub-task — is unstarted work with a note attached, not a finished container. Do not close it. Tell the two apart by intent: a container reads as a grouping (a =PROJECT= keyword, an explicit "parent grouping ..." line, or several dated entries that were genuinely separate sub-tasks that shipped); a leaf-with-notes is a single feature/bug task whose title names unstarted work and whose lone dated child is a design note. When the call is ambiguous, flag it NEEDS-USER rather than closing.
+
+Never close or promote autonomously past the ambiguous line — surface the candidates with a recommendation and let Craig ratify, the same interactive stance as Phase C.5. Clear container completions (a =PROJECT= whose every child is dated) can be proposed as a batch; leaf-with-notes ambiguities are flagged individually. Verify open-vs-done counts against the actual headings (a real scan of the subtree), not a fragile regex that a shell's =\b= support can silently break — a miscount here closes live work.
+
** Phase D — Flag the judgment calls (interactive)
Present the NEEDS-USER bucket as a short, scannable list — one line per task, naming the decision or the fact required. Adjudicate with the user one item at a time (inline numbered options per =interaction.md=, no popup). Apply the user's calls as they come (which may itself produce more autonomous updates, or new tasks).
@@ -132,3 +164,7 @@ Two Phase C behaviors added, both surfaced by an Emacs-config =todo.org= audit:
- *Tag-vocabulary enforcement.* That project declares a closed tag set (=bug=, =feature=, =refactor=, =test=, =quick=, =solo=); the audit had to strip ~44 ad-hoc tags that had accumulated across the file. The prior workflow only checked that a type tag was *present* — it had no concept of an exhaustive allowed set. The new bullet enforces a declared closed vocabulary and leaves open-vocabulary projects untouched.
- *Code-complete-but-unverified closing.* Many tasks had shipped (tests green, live in the daemon) but stayed open awaiting a manual or visual verification, so they accumulated as half-open. Leaving them open is noise; auto-closing them would violate "never claim a fix verified before the user confirms." The fix routes the pending human check into the project's =Manual testing and validation= parent (dedup-checked) per =verification.md='s manual-verification hand-off, then closes the implementation task. The work is done and the check is tracked; a failed check promotes to a bug.
+
+** 2026-07-01 — Retire completed parents (Phase C.6)
+
+Added Phase C.6: retire a parent task once its child *tasks* are all done. Zero open children → close the parent; one or two open children → promote them to standalone level-2 tasks, then close. Surfaced by an Emacs-config =todo.org= audit where several PROJECT containers had all children complete. Depends on =todo-cleanup.el --convert-subtasks= running first so completed sub-tasks are dated (not lingering =DONE= keywords) and the open-child count is accurate. Carries a leaf-with-notes carve-out: a =**= leaf task whose only descendant is a dated design note ("Ideas"/"Goals") is unstarted work, not a finished container, and must not be closed — the ambiguous case is flagged NEEDS-USER. The step also warns against counting open-vs-done with a fragile regex (a =\b= that a given shell/awk silently drops miscounts and closes live work).
diff --git a/.ai/workflows/task-review.org b/.ai/workflows/task-review.org
index 69e172d..7ea2e8e 100644
--- a/.ai/workflows/task-review.org
+++ b/.ai/workflows/task-review.org
@@ -1,5 +1,5 @@
#+TITLE: Task Review Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-20
* Overview
@@ -57,7 +57,9 @@ Keep is the common case — most tasks are still right and just need re-stamping
*** Tagging =:quick:= — small tasks
-While reviewing each task, estimate its effort. If you judge it *30 minutes or less* and it doesn't already carry =:quick:=, add the tag to the heading line. If the heading and body don't tell you how long it'll take, *ask Craig* — don't guess. A wrong =:quick:= is worse than none: the tag exists so Craig can grab a genuinely small task in a spare moment, and a mislabeled one wastes that moment.
+The =:quick:= and =:solo:= assessments (this section and the next) are *mandatory* for every reviewed task except a Kill — a review that skips them is incomplete. The hard definitions live in [[file:../../claude-rules/todo-format.md][todo-format.md]] ("Hard definitions: :solo: and :quick:"); autonomous execution (work-the-backlog / the no-approvals speedrun) reads =:solo:= as its eligibility gate and trusts the author's tag, so the run-time gate is only as trustworthy as this pass.
+
+While reviewing each task, estimate its effort. If you judge it *30 minutes or less* and it doesn't already carry =:quick:=, add the tag to the heading line. If the heading and body don't tell you how long it'll take, *ask Craig* — don't guess. A wrong =:quick:= is worse than none: the tag exists so Craig can grab a genuinely small task in a spare moment, and a mislabeled one wastes that moment. =:quick:= is an effort hint only, never an eligibility gate — size does not decide what runs autonomously.
This is orthogonal to the action chosen — a task can be kept (or re-graded, or marked DOING) *and* tagged =:quick:= in the same pass. Skip the assessment on a Kill, since it's leaving the pool. Tags go on the heading line per [[file:../../claude-rules/todo-format.md][todo-format.md]], sharing one =:tag1:tag2:= cluster.
@@ -67,7 +69,7 @@ While reviewing each task, judge whether Claude could build *and* verify it with
1. *Buildable* — Claude has the capability and access to do the work.
2. *Verifiable by Claude* — an objective or local check exists that Claude can run itself. Craig's routine spot-checking does not count against this, and neither does handing off a residual human-in-the-loop confirmation as a structured manual-testing reminder (the =verification.md= "Handing Off Manual Verification" pattern). The disqualifier is having no verification path of Claude's own at all — when the success criterion is only judgeable by Craig's eyes or subjective taste.
-3. *No upfront decision* — no design or preference call Craig must make before Claude can begin.
+3. *No deliberation* — no open design question and no "weigh these approaches" with real tradeoffs. At most one or two *quick, upfront-answerable* factual decisions are allowed — the speedrun preset batches those into its pre-flight Q&A, so they don't break the hands-off run. A genuine design or preference call disqualifies.
If any gate is shaky, leave the tag off. Like =:quick:=, a wrong =:solo:= is worse than none — it tells Craig he can hand the task off and walk away, so a mislabeled one wastes that trust. When the heading and body don't make all three gates clear, ask Craig instead of guessing.
@@ -90,12 +92,14 @@ Set =:LAST_REVIEWED:= to today's date (from above) in the task's =:PROPERTIES:=
Body...
#+end_example
-The exact date string matters: =task-review-staleness.sh= and the wrap-up health check both parse =:LAST_REVIEWED: YYYY-MM-DD=.
+Format: =:LAST_REVIEWED:= takes a bare ISO date (=2026-05-20=) or an org-native inactive timestamp (=[2026-05-20 Tue]=, matching the =CREATED:=/=CLOSED:= cookies beside it); =task-review-staleness.sh= and the wrap-up health check normalize both to the date. A value that is neither is a data error — the staleness script warns loudly to stderr (naming the file, line, and value) and leaves the task out of the stale count rather than silently reporting a freshly-reviewed task as never-reviewed. Stamp a clean date and the warning never fires.
*** Killing a task
Follow the completion rules in [[file:../../claude-rules/todo-format.md][todo-format.md]]. A killed top-level =**= task stays task-shaped: change the keyword to =CANCELLED=, add a =CLOSED: [YYYY-MM-DD Day]= line under the heading (generate with =date "+%Y-%m-%d %a"=), and leave the priority and tags intact. It's then a candidate for =--archive-done= at the next cleanup. Don't stamp =:LAST_REVIEWED:= on a kill — it's leaving the review pool anyway.
+A killed *sub-task* (=***= or deeper, under a parent task) instead becomes a dated event-log entry per the depth rule — but you don't have to hand-format it here. =todo-cleanup.el --convert-subtasks= (run in the =clean-todo= and wrap-up cleanup passes) rewrites any level-3+ DONE/CANCELLED/FAILED heading into its dated form mechanically from the =CLOSED= cookie, so a keyword-plus-=CLOSED= close at depth gets normalized on the next cleanup rather than lingering. =lint-org.el= flags any that slip through (checker =subtask-done-not-dated=).
+
* Phase D: Close out
When the batch is done (or Craig calls it early):
diff --git a/.ai/workflows/triage-intake.cmail.org b/.ai/workflows/triage-intake.cmail.org
index d818c72..8d8abfb 100644
--- a/.ai/workflows/triage-intake.cmail.org
+++ b/.ai/workflows/triage-intake.cmail.org
@@ -1,5 +1,5 @@
#+TITLE: Triage Intake — cmail (Proton) Source
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-26
# Source plugin for the triage-intake engine. See triage-intake.org for the
diff --git a/.ai/workflows/triage-intake.github-prs.org b/.ai/workflows/triage-intake.github-prs.org
index c1bc796..644421c 100644
--- a/.ai/workflows/triage-intake.github-prs.org
+++ b/.ai/workflows/triage-intake.github-prs.org
@@ -1,5 +1,5 @@
#+TITLE: Triage Intake — Personal GitHub PRs Source
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-26
# Source plugin for the triage-intake engine. See triage-intake.org for the
diff --git a/.ai/workflows/triage-intake.org b/.ai/workflows/triage-intake.org
index b257f2d..55cc939 100644
--- a/.ai/workflows/triage-intake.org
+++ b/.ai/workflows/triage-intake.org
@@ -1,5 +1,5 @@
#+TITLE: Triage Intake Workflow (Engine)
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-01
* Summary
@@ -11,12 +11,14 @@ Think of it as the ER intake queue: every new message, invite, and PR notificati
*This file is the engine.* It carries no sources of its own. Every source it scans comes from a *source plugin* — a =triage-intake.<source>.org= file the engine loads at Phase 0. The engine is source-agnostic and project-agnostic; the project- and account-specific knowledge lives entirely in the plugins. To add a source, drop a plugin file. To change one, edit its plugin. Never wire a source into this file.
+*Which sources a project pulls is a per-project choice.* A *project-specific* plugin (=.ai/project-workflows/triage-intake.*.org=, never synced) is active by presence — dropping it is the declaration. A *general* plugin (=.ai/workflows/triage-intake.*.org=, template-synced into every project — personal Gmail, cmail, calendar, Telegram, GitHub PRs) is active only when the project names its basename in a =:TRIAGE_SOURCES:= line in =.ai/notes.org= Workflow State (space-separated basenames, e.g. =:TRIAGE_SOURCES: personal-gmail cmail=). A project that declares nothing and owns no project plugin pulls nothing. This is the Phase 0 activation gate — presence is capability, the declaration is activation (see =docs/specs/2026-07-20-triage-source-activation-spec.org=).
+
Distinct from =daily-prep.org=:
- *daily-prep* — heavier, once daily, builds the day's plan + standup brief + meeting prep + time blocks.
- *triage-intake* — fast, repeatable, just answers "what's new since last check?"
-Quick contract — what it does: fans out across source plugins, classifies every item into Action / FYI / Noise-keep / Noise-trash, synthesizes one deduped summary, writes each Action item to =todo.org= as a =:quick:reactive:= task, and executes star/mark-read/trash on confirmation.
+Quick contract — what it does: fans out across source plugins, classifies every item into Action / FYI / Noise-keep / Noise-trash, surfaces one three-section digest (==TASKS== / ==FYI== / ==MISC==) with a two-option close offer, then *closes by default*: files each TASKS item to =todo.org= as a =:quick:reactive:= task, runs the star/mark-read/trash hygiene on every scanned account, advances the sentinel, and tears down anything it started. The close runs unless Craig explicitly holds it.
** When to Use This Workflow
@@ -37,6 +39,8 @@ Typical timing:
Do *not* use when running daily-prep — daily-prep already does this as Phase 3.
+Also runs unattended as sentry's triage pass (=sentry.org=, pass 3): sentry invokes this engine under its no-approvals contract, where destructive actions (deleting, archiving, sending) queue for the morning-approval review instead of firing. The trigger phrases above are unchanged — a manual "triage intake" always routes here directly.
+
* Execution
@@ -56,16 +60,18 @@ ls .ai/workflows/triage-intake.*.org .ai/project-workflows/triage-intake.*.org 2
The glob exclude is automatic: =triage-intake.*.org= matches the plugins but not this engine file (=triage-intake.org= has no second dot-segment), so the engine never loads itself.
After globbing, for each plugin file:
-1. Read it.
-2. Evaluate its =ENABLED= precondition. If false, *announce the skip with its reason* ("skipping linear — mcp__linear not present") and move on.
-3. The surviving set is the source list for Phases A-D.
+1. *Activation gate.* A *general* plugin (from =.ai/workflows/=, template-synced into every project) is active only if its basename appears in the project's =:TRIAGE_SOURCES:= declaration (=.ai/notes.org= Workflow State — a space-separated list of source basenames). If it isn't declared, it is *inactive*: announce it ("inactive: personal-gmail — not in :TRIAGE_SOURCES:") and skip it. A *project-specific* plugin (from =.ai/project-workflows/=, never synced) is always active — dropping it there is itself the per-project declaration. This is what stops the synced general plugins from self-activating in projects that aren't triage targets: presence is capability, the declaration is activation (see =docs/specs/2026-07-20-triage-source-activation-spec.org=). An absent or empty =:TRIAGE_SOURCES:= means no general sources are active; a project with no declaration and no project plugin has no active sources, so triage no-ops there.
+2. Read it.
+3. Evaluate its =ENABLED= precondition. If false, *announce the skip with its reason* ("skipping linear — mcp__linear not present") and move on.
+4. The surviving set — active and enabled — is the source list for Phases A-D.
-*Announce the loaded set before scanning* so the omission can't hide:
+*Announce the loaded set before scanning* so the omission can't hide — inactive (undeclared) plugins are named too, so a general plugin left out of =:TRIAGE_SOURCES:= is a visible choice, not a silent drop:
#+begin_example
-Loaded 5 source plugins:
- general: personal-gmail, personal-calendar, cmail, github-prs
+Loaded 2 source plugins (:TRIAGE_SOURCES: personal-gmail cmail):
+ general: personal-gmail, cmail
project: deepsat-gmail
+ inactive (undeclared): personal-calendar, github-prs, telegram
skipped: linear (mcp__linear not present)
#+end_example
@@ -91,23 +97,27 @@ Every item lands in one bucket. Plugins refine these with source-specific bias a
Per-source bias (a work email account leans keep for audit value; a personal account leans trash on high noise volume) lives in each plugin's =Classify= section. Read it from there; don't re-derive it here.
-*** Phase C: Synthesize a single summary
+*** Phase C: Synthesize — the three-section digest
-One markdown summary surfaced inline to Craig. Order:
+One digest surfaced inline to Craig — notable items only, written as prose bullets a reader can absorb without knowing the source taxonomy. It is *not* a per-source roll-call; the plugins' =Render= shapes feed the classification, they are no longer displayed as blocks. (Format ratified by Craig 2026-07-18 after a sweep where the follow-up "summarize the notable items" digest was the report he actually wanted first.)
-0. *Scan failures — first, loud, always.* Any loaded source whose scan failed, hung, was killed, or was skipped for an operational reason renders at the very top of the summary, before Top signals:
+Order:
+
+0. *Scan failures — first, loud, always.* Any loaded source whose scan failed, hung, was killed, or was skipped for an operational reason renders at the very top of the summary, before everything else:
#+begin_example
⚠ SCAN FAILED: <source> — <reason, one line> — <what's now unknown>
#+end_example
- A failed scan is never folded into "quiet." Quiet means the scan ran and found nothing; a failure means the sweep is blind on that channel, and the reader must know which. The same applies to a precondition skip the user hasn't standing-approved (e.g. a messaging client that needs a temporary server spin-up): run the lifecycle or report the failure — don't silently narrow the sweep.
+ A failed scan is never folded into "quiet." Quiet means the scan ran and found nothing; a failure means the sweep is blind on that channel, and the reader must know which. The same applies to a precondition skip the user hasn't standing-approved (e.g. a messaging client that needs a temporary server spin-up): run the lifecycle or report the failure — don't silently narrow the sweep. Backlog banners (a plugin's pre-anchor-residue probe firing) render here too — loud, above the sections.
+
+1. *==TASKS==* — every work item that needs Craig or his sign-off. Sorted in two groups: *solo-executable first* (see the solo test in Phase D), then the rest; within each group, priority order (blocking someone / deadline inside 48h first). Each item is one short prose bullet naming who, what, and why-now, with the source link or locator.
+2. *==FYI==* — substantive work context worth seeing, no action owed. Priority order.
+3. *==MISC==* — everything outside the project: personal mail, family, household, personal calendar. Priority order. An outside-project item that needs Craig still lands here (with its action-ness stated inline), not in TASKS — TASKS is the work queue. MISC items are *not* filed into this project's =todo.org=; they surface in the digest and are gone after the close unless Craig reroutes them to their owning project (the "and reroute" modifier, below).
-1. *Top signals to act on* — bullet list of 3-7 items, ordered by urgency, *Action only*. Each bullet links to the source (permalink, thread URL, PR number).
-2. *Per-source breakdown* — one short section per loaded source *that has changes*, in =ORDER=, using that plugin's =Render= shape: Action items detailed, FYI items as a short list, Noise as a tally only ("Noise: 12 trash candidates, 4 keep, 0 starred").
-3. *Suggested actions* — explicit list of state changes Craig could take this run (trash these N messages, mark-read these M, star this Action item, respond to this invite, merge PRs #X and #Y, etc.). This line stays whenever there are queued actions, regardless of how quiet the sweep was.
+A section with nothing in it is omitted. After the sections, the offer (see Phase D). The sweep's final line is always the run timestamp (=date "+%A %Y-%m-%d %H:%M %Z"=).
-*Deltas only.* The summary reports what *changed* since the anchor: a new invite, a new/moved/cancelled calendar event, a new message needing attention. A source with no changes gets no block — no "Calendar — quiet", no "PRs — nothing new" roll-call. A sweep where nothing changed anywhere renders as a single line:
+*Deltas only.* The digest reports what *changed* since the anchor: a new invite, a new/moved/cancelled calendar event, a new message needing attention. A source with no changes contributes nothing — no "Calendar — quiet", no "PRs — nothing new" roll-call. A sweep where nothing changed anywhere renders as a single line plus the timestamp:
#+begin_example
17:39 sweep: no changes
@@ -117,11 +127,40 @@ One markdown summary surfaced inline to Craig. Order:
Scan failures are the standing exception: a failed or skipped scan always renders loudly per point 0 above and is never folded into the no-change line — "no changes" is a claim about channels the sweep could actually see.
-Format target: scannable in 30 seconds, full read in 2 minutes. Don't pad.
+Format target: scannable in 30 seconds, full read in 2 minutes. Don't pad. The old long-form report (anchor line, per-source breakdown, itemized suggested-actions list) is available *on request* — it is no longer the default surface.
+
+**** The offer — exactly this, right after the sections
+
+#+begin_example
+1. Close the triage — file todo.org tasks for the TASKS items, run the standard close
+2. Close the triage and execute the solo TASKS now, after filing the rest
+Append "and reroute" to either option to send the outside-project items to their owning projects.
+Or tell me what you'd like handled differently.
+#+end_example
+
+Option 2 renders *only when solo-executable TASKS exist*. The reroute line renders only when MISC is non-empty. No other options, no itemized action menu — the close (Phase D) owns the routine hygiene.
+
+*The reroute modifier.* "1 and reroute" / "2 and reroute" (or a bare "reroute" right after a close) means: in addition to the close, deliver every outside-project item the sweep surfaced to the project that owns it. Routing goes through =inbox-send= per the cross-project rule — a handoff into the owner's =inbox/=, never a direct write to a foreign =todo.org= — and the owner's own inbox processing files it by its conventions. This is the persistence path for MISC: without a reroute, MISC items are surfaced-only.
+
+*** Phase D: Close — the default, not an option
+
+*Closing the triage is the next action after the digest, no exceptions* — unless Craig explicitly says not to ("hold the triage", "don't close yet"). If his reply picks option 1 or 2, close per the option. If his reply is anything else — a question, a redirect, a new task — *close the triage first* (as option 1), then handle what he asked. An unclosed triage strands the noise unprocessed and the sentinel stale; Craig ruled 2026-07-18 that the close, including the mail hygiene on every scanned account, is unconditional.
+
+The close, in order:
+
+1. *File every TASKS item into =todo.org=* as its own =:quick:reactive:= task (format below). Dedupe against existing tasks first — fold into an existing task's body when one already covers the topic.
+2. *Mail and message hygiene on every scanned account*: trash the Noise-trash set, mark-read the Noise-keep set, star what was flagged for keeping. This runs *without itemized confirmation* — the digest's tallies are the notice, and the actions dispatch through each plugin's =Actions= verbs. Trash is recoverable (Gmail 30-day trash; cmail =\Deleted= flag), which is what makes the no-confirmation batch safe.
+3. *Clear unacked items* that the sweep found resolved.
+4. *If option 2: execute the solo TASKS.* The solo test — mechanical, standing-approved, and producing *no prose under Craig's name*: calendar RSVPs, ticket-state moves the publishing overlay already authorizes, mark-read/star/trash. Anything that sends words as Craig (a Slack reply, an email, a PR comment, a Linear comment) is *never* solo — it stays a filed task and goes through the normal draft gate when worked. Destructive or hard-to-reverse actions beyond mail hygiene (branch deletes, PR merges) also stay confirm-gated per their plugins.
+5. *If "and reroute": route the outside-project items.* For each MISC item (and any surfaced item this project doesn't own), send a handoff to the owning project's inbox: =inbox-send <project> --text "..."= carrying what it is, the source locator (message id, thread, event id), and why it routed. Resolve the owner against =inbox-send --list=; when ownership is ambiguous, ask before sending — a wrong-project handoff costs more than one question. Never write another project's =todo.org= directly (cross-project rule). Note in the close's status line which items went where.
+6. *Advance the sentinel* — write the held Phase A capture into the sentinel's *content* (see "Capture the Phase A timestamp"): =echo "$PHASE_A_TS $(date -d "@$PHASE_A_TS" '+%Y-%m-%d %H:%M:%S %z')" > .ai/last-triage-intake=. Do not use plain =touch= (writes mtime to /now/ and strands items posted between Phase A and end of run) and do not use =touch -d "@$PHASE_A_TS"= (correct timestamp but mtime is per-machine — won't survive a fresh clone or cross-machine sync).
+7. *Tear down anything the sweep started* (e.g. the telegram lifecycle's leave-no-trace shutdown).
+
+After the close, report one status line — what shipped, the sentinel time — and stop. The close *is* the exit; there is no separate confirmation loop.
-**** Sub-step: write each Action item into =todo.org= as its own =:quick:= task
+**** Task-filing format (=todo.org=)
-After surfacing the summary inline, append every Action item — regardless of source — to =todo.org= as its own top-level =** TODO= heading carrying the =:quick:= tag plus =:reactive:= and any relevant person/entity tag.
+Append every TASKS item — regardless of source — as its own top-level =** TODO= heading carrying the =:quick:= tag plus =:reactive:= and any relevant person/entity tag.
Each Action item is one task. Don't group items by source under =** Email Response=, =** PR Review=, etc. sub-headings. Each response is its own filterable task so Craig can re-prioritize, =SCHEDULE:= / =DEADLINE:=, or tag individually.
@@ -142,30 +181,113 @@ Rules:
- *Record the source locator in the task body* so a reply can be routed back to where the request came from — the channel + thread id for chat, the repo + PR number, the message id for mail. The general rule: a reply goes back to the *origin* of the request, not a fixed notification channel. (Project plugins may add stricter routing rules in their own files.)
- Placement: append at end of =* Work Open Work= (just before =* Work Incubate=) unless the project's =todo.org= has a designated triage section near the top (=* Triage= or =* Inbox=).
-This sub-step makes triage-intake's findings *persist* in =todo.org= instead of evaporating after the inline summary.
+The filing makes triage-intake's findings *persist* in =todo.org= instead of evaporating after the inline digest. Every close action dispatches to the owning source plugin's =Actions= verb (trash, mark-read, star, respond, merge, comment, attachment-fetch) — the engine doesn't hardcode action commands; it reads them from the loaded plugins.
-*** Phase D: Execute actions on confirmation
+*** Exit Criteria
-Wait for Craig's go-ahead before running any state changes. Default to single-confirmation for the whole batch ("yes" → run everything proposed). Craig may also pick a subset ("trash personal but hold the work account") or hand back a different plan ("trash all but star the expense thread and queue PR merges for after lunch").
+The close is the exit. Once the close completes (tasks filed, hygiene run, sentinel advanced, teardown done), report one status line — what shipped, the sentinel time — and stop. The old stay-open-until-confirmed loop is retired (Craig, 2026-07-18): the digest plus the two-option offer is the whole interaction, and the close runs by default. The only way the workflow stays open is Craig explicitly saying not to close.
-Each action dispatches to the owning source plugin's =Actions= verb (trash, mark-read, star, respond, merge, comment, attachment-fetch). The engine doesn't hardcode action commands — it reads them from the loaded plugins. Read each plugin's =Actions= section for the exact command.
+*** KB capture (only if the sweep surfaced something durable)
-After actions complete, write the Phase A capture into the sentinel's *content* (see "Capture the Phase A timestamp"): =echo "$PHASE_A_TS $(date -d "@$PHASE_A_TS" '+%Y-%m-%d %H:%M:%S %z')" > .ai/last-triage-intake=. Do not use plain =touch= (writes mtime to /now/ and strands items posted between Phase A and end of run) and do not use =touch -d "@$PHASE_A_TS"= (correct timestamp but mtime is per-machine — won't survive a fresh clone or cross-machine sync).
+If this sweep surfaced a durable, cross-project fact — a recurring pattern across sources, a reference pointer worth keeping, an environment gotcha — consider writing it to the agent KB as one =:agent:= node (see the best-practices node and =knowledge-base.md=; personal projects only, work never writes). One line of judgment, not a step: an all-quiet sweep surfaces nothing and writes nothing. Never blocking, never padded onto a no-signal run.
-*Do not close the workflow yet.* See Exit Criteria below.
-*** Exit Criteria
+* Auto mode (unattended monitoring)
+
+Auto mode is a self-running variant of the engine for when Craig is away from the desk but wants tight awareness — a loop that runs the standard sweep on a short interval, *accumulates* findings rather than mutating state, and hands Craig a gated checkpoint to commit the batch. It composes two things: the *delivery* (a =/loop= in the live session) and the *behavior* (accumulate-don't-mutate sweeps with a checkpoint). The one-shot run above is unchanged; auto mode is an additional way to run the same Phase 0 / A-D engine.
+
+** Trigger and delivery
+
+- "auto triage" / "auto triage-intake" / "watch the desk" / "monitor the triage" — start auto mode.
+- Default interval *20 minutes*; Craig sets it.
+
+Auto mode runs as a =/loop= in the *live session*, not a detached cron job:
+
+#+begin_src
+/loop 20m run an auto-mode triage-intake sweep per triage-intake.org
+#+end_src
+
+Running in the live session means MCP auth (Slack, Gmail, Linear) is inherited from the session — the headless-auth wall that blocks a detached cron run does not apply. A durable cross-session schedule is out of scope here; that belongs to the morning-ops orchestrator, which can later invoke auto mode's accumulate behavior as its triage limb. The close/stop commands below require a live session by design.
+
+*** Phone delivery — push each signal sweep via =agent-text=
+
+Auto mode exists for when Craig is away from the desk, so a sweep that surfaces something worth seeing is delivered to his phone, not just printed into a session he isn't watching. After a sweep that renders the full three sections — one with real deltas or an unacked-list change (see "End-of-sweep output" below) — send that same output to his phone over Signal with =agent-text=:
+
+#+begin_src bash
+agent-text "$SWEEP_SUMMARY"
+#+end_src
-The workflow stays open until Craig has *explicitly* either:
+The pushed text is the *fuller* three-section shape, not a terse one-liner: the per-source deltas, the responses-awaiting-acknowledgment list, and the timestamp, led by a ⚠ SCAN FAILED banner if any source failed.
-1. *Confirmed* that the executed actions are sufficient and nothing more is needed this round, or
-2. *Handed back a different plan* (e.g., "actually hold the PR merges, address #131 first").
+*Signal-only — never on a quiet sweep.* An empty sweep (the =triage intake at HH:MM: nothing= heartbeat) does *not* push to the phone. Silent-until-signal (see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=) governs the phone channel too, so the phone stays silent until a sweep has real signal. The in-session heartbeat still prints as proof the loop ran; the phone is reserved for something that actually needs Craig. (Craig's ruling, 2026-07-20: the higher-cost channel doesn't buzz with "nothing.")
-A successful Phase D run is *not* an exit signal. After the action batch returns, surface what shipped and wait. Don't volunteer "done" or "all set" — those are exit-claim phrases that pre-empt Craig's call. Use a status report ("17 actions succeeded, sentinel written at 12:19") and stop.
+If =agent-text= isn't on =PATH=, fall back to inline delivery and say so once.
-If Craig has been silent for a while after Phase D and the surface looks closed-out, *ask*: "Anything else on this triage, or are we good to close out?" Don't auto-terminate.
+*Reply polling is deferred.* The send half ships here; polling the phone for Craig's replies (the =phone-recv= half of the retired ntfy design) waits on the reply-correlation follow-up. With the Signal account linked on more than one device, a reply fans out to every device and neither knows which page it answers — that has to be resolved before auto mode reads replies back. Until then auto mode pushes but does not poll, and Craig acts on a pushed summary from wherever he picks it up.
-This rule prevents the failure mode where the workflow self-declares done and the next exchange has to relitigate what state things are in.
+** Preconditions and Close-out
+
+Auto mode borrows the inbox monitor-mode gates (=inbox.org= monitor mode): do not start on a dirty worktree or a red test suite — a close's batch commit would otherwise sweep up unrelated changes — and leave the tree clean and green when the loop stops. Surface a blocker with inline numbered options per =interaction.md= and wait.
+
+** A sweep: accumulate, don't mutate
+
+Each sweep runs Phase 0 (load *both* plugin dirs — the loud requirement still holds) and Phases A-D's scan / classify / synthesize, but performs *none* of the normal run's mutations:
+
+- Does NOT advance the sentinel. The scan window grows from the last *close* until the next close: every sweep scans from the existing sentinel up to now, so nothing between sweeps is dropped.
+- Does NOT write =todo.org= Action tasks — accumulates them for the close.
+- Does NOT take mail actions (trash / mark-read / star).
+- Does NOT commit.
+- DOES update an active daily-prep in Update mode and re-open it on change (per =daily-prep.org=).
+- DOES report, deltas-only, with loud scan-failure banners (Phase C rules unchanged).
+
+** End-of-sweep output — three sections, or one heartbeat
+
+*Silent-until-signal (see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=).* An *empty sweep* — no deltas since the previous sweep and no change to the awaiting-acknowledgment list — collapses to a single heartbeat line and nothing else: =triage intake at HH:MM: nothing= (HH:MM local, from =date=). Detection still runs in full (Phase 0 plus the A-D scan, against the session's inherited MCP auth); only the output collapses, so a long unattended run stops filling the session with identical "no changes" blocks. A sweep with real deltas or an unacked-list change prints the full three sections below, and — when away — pushes them to Craig's phone via =agent-text= (see "Phone delivery" above). The empty-sweep heartbeat is never pushed.
+
+1. *Deltas* — what changed since the *previous sweep* (the standard Phase C summary scoped to the inter-sweep delta).
+2. *Responses awaiting your acknowledgment* — every Slack reply, email, or message directed at Craig that he hasn't acknowledged or had the agent answer. A *running list carried forward across sweeps* until Craig acks each item or closes the triage. An away user's first need is "who's waiting to hear back from me," which a delta-only sweep loses the moment it scrolls past.
+3. *Timestamp* — the current date, time, and timezone on the sweep's own final line, so an away reader sees how fresh the summary is without computing it. Print it on every sweep that prints these three sections. On an *empty* sweep there is no separate timestamp line — the heartbeat (=triage intake at HH:MM: nothing=) is itself the freshness stamp and the proof the loop ran. Generate it with:
+
+ #+begin_src bash
+ date "+%A %Y-%m-%d %H:%M:%S %Z (%z)"
+ #+end_src
+
+** The unacked list — durable state
+
+The awaiting-acknowledgment list lives in =.ai/triage-intake-unacked.org=, so it survives a session crash, a =/clear=, or a restart — the away-from-desk case auto mode exists for. It's project-local state, tracked the same way as the sentinel (=.ai/last-triage-intake=), created on first need.
+
+Shape — one =** = heading per awaiting item:
+
+#+begin_example
+#+TITLE: Triage Intake — Responses Awaiting Acknowledgment
+# Maintained by triage-intake auto mode. One heading per item; acked items are removed.
+
+** Dana — 2pm reschedule invite
+:PROPERTIES:
+:SOURCE: personal-calendar
+:LOCATOR: <event id or thread url — the dedupe key>
+:SINCE: 2026-06-15 10:42
+:END:
+She's waiting on a yes/no to the move.
+#+end_example
+
+- *Add* — a sweep appends any new directed-at-Craig response not already listed, deduped on =LOCATOR=.
+- *Carry forward* — every sweep re-renders the full list in its second section, whether or not it changed this sweep.
+- *Ack* — "ack <item>" (e.g. "ack the Dana thread") removes that heading; "ack all" clears the list.
+- *Close* — a close empties the list as part of processing (each item is either actioned or filed).
+
+** Close and stop — the checkpoint
+
+The mutations are gated behind two commands:
+
+- *"close the triage"* — run the full close per Phase D: take the accumulated mail hygiene, file the accumulated TASKS items to =todo.org=, reroute if asked ("close the triage and reroute"), empty the unacked list, then *advance the sentinel* — capture the close run's Phase A timestamp, do the mutations, write that timestamp to =.ai/last-triage-intake= exactly as a normal run does (per "Capture the Phase A timestamp") — and commit + push the batch. Then *keep looping* (next sweep on the normal interval). This is the "flush the batch and carry on" checkpoint.
+- *"stop the triage"* — the same close processing, then *stop the loop* and revert to manual (on-demand) triage.
+
+A close is the only point auto mode advances the sentinel or commits. Between closes the engine state is untouched — that is what makes a 20-minute sweep cheap and non-destructive, and it preserves the engine invariant: the sentinel still means "everything before this timestamp has been scanned," it just advances once per close instead of once per run.
+
+** Why a separate mode
+
+The standard engine is one-shot and mutating — right for an at-the-desk "what's new?" glance, wrong for unattended polling: run every 20 minutes it would advance the sentinel past unprocessed items, spray reactive todos, take mail actions, and commit noise without review. Auto mode separates the cheap, frequent *watching* from the deliberate, gated *committing*, and adds the away-user's missing primitive — the running unacked-responses list.
* Reference
@@ -185,7 +307,7 @@ A plugin file declares exactly one source through a fixed shape:
*Body sections:*
- =** Scan= — the command(s) that fetch new/unread items since =<anchor>=, emitting raw items.
- =** Classify= — the source's per-bucket bias and noise patterns. *Deltas* from the engine's shared four-bucket model below, not a re-derivation.
-- =** Render= — the source's block in the Phase C summary. "Omit if empty."
+- =** Render= — the source's classification shape. Since the 2026-07-18 digest format, Render blocks are *inputs to the Phase C synthesis*, not displayed sections — the digest is source-agnostic (TASKS / FYI / MISC). Keep the shape: it defines what the source considers reportable, and the long-form breakdown (on request) still uses it.
- =** Actions= — the executable state-changes, one verb per line: =verb :: command template (parameterized by item id)=.
Template:
@@ -263,33 +385,37 @@ If both fail, fall through to the resolution order above (prep doc → session f
** Output Template
-The summary follows this shape (deltas only: a source with no changes gets no block; when *nothing* changed anywhere, the whole summary collapses to the one-line form below — plus any scan-failure banners and the suggested-actions line if actions are queued):
+The digest follows this shape (deltas only: a source with no changes contributes nothing; when *nothing* changed anywhere, the whole digest collapses to the one-line form below plus the timestamp — scan-failure banners always render regardless):
#+begin_example
17:39 sweep: no changes
#+end_example
-When there are changes, render one block per changed source in =ORDER=, using each plugin's =Render= shape:
+When there are changes:
#+begin_example
-**Anchor:** <previous run timestamp> → now (<elapsed> elapsed)
-**Loaded:** <general plugins> + <project plugins> (skipped: <disabled, with reason>)
+<⚠ SCAN FAILED / backlog banners, if any>
+
+==TASKS==
+- <solo-executable items first, then the rest; priority order within each;
+ one prose bullet each: who, what, why-now, source link>
+
+==FYI==
+- <substantive work context, no action owed; priority order>
-**Top signals to act on:**
-1. <terse Action description with link>
-2. ...
+==MISC==
+- <everything outside the project — personal mail, family, household;
+ priority order; action-ness stated inline>
-<one block per loaded source, in ORDER — see each plugin's Render>
+1. Close the triage — file todo.org tasks for the TASKS items, run the standard close
+2. Close the triage and execute the solo TASKS now, after filing the rest
+Append "and reroute" to either option to send the outside-project items to their owning projects.
+Or tell me what you'd like handled differently.
-**Suggested actions:**
-- Trash N noise items
-- Mark-read M keep items
-- Respond to <invite>
-- Merge PRs #X and #Y
-- ...
+<run timestamp — always the final line>
#+end_example
-Order matters: top-signals first because that's what Craig reads in 30 seconds between meetings. Per-source detail second. Suggested actions last because they require a decision.
+Option 2 appears only when solo-executable TASKS exist; the reroute line only when MISC is non-empty. Sections with nothing in them are omitted. No per-source blocks, no itemized suggested-actions list — the close owns the routine hygiene. The long-form per-source breakdown is available on request only.
** Common Mistakes
@@ -297,11 +423,11 @@ Order matters: top-signals first because that's what Craig reads in 30 seconds b
1. *Globbing only =.ai/workflows/= and missing the project plugins.* The single most damaging failure mode — the sweep runs with half its sources and the omission is invisible (a missing source looks identical to a quiet one). Phase 0 globs *both* =.ai/workflows/triage-intake.*.org= and =.ai/project-workflows/triage-intake.*.org=, every run, and announces the loaded set.
2. *Running Phase A sequentially.* Send every enabled source's scan in one message — the whole point is parallelism.
3. *Wiring a source into the engine.* Sources live in plugin files, never here. If you find yourself editing this file to add an account, repo, or channel, stop — write or edit a =triage-intake.<source>.org= plugin instead.
-4. *Executing actions without explicit confirmation.* Phase D runs only after Craig says "yes" or picks a subset.
+4. *Leaving the triage open, or re-asking about the routine hygiene.* The close is the default next action after the digest (Craig's 2026-07-18 ruling — no exceptions unless he says hold). Mail hygiene (trash/mark-read/star) runs at close without itemized confirmation. What still needs explicit confirmation: anything sending prose under Craig's name, and destructive actions beyond mail hygiene (branch deletes, PR merges).
5. *Forgetting to set the sentinel at the end.* Without it, the next run re-scans the same window.
6. *Using mtime instead of content for the sentinel.* Plain =touch= writes /now/ to mtime, stranding items posted between Phase A and end of run. =touch -d "@$PHASE_A_TS"= fixes the time but mtime is per-machine — git tracks content, not metadata, so the anchor doesn't survive a clone or cross-machine sync. Always write the epoch into the file's *content*.
7. *Running this alongside daily-prep.* Daily-prep already does this as Phase 3 — don't duplicate.
-8. *Mixing Action and FYI in the top-signals list.* Top signals = Action only. FYI lives in the per-source detail.
+8. *Mixing sections.* TASKS = work items needing Craig only; work FYIs never appear there. Outside-project items always land in MISC, even when they carry an action. Solo items lead TASKS; priority order inside every section.
9. *Reporting a failed or skipped scan as a quiet source.* A hung receive, a dead daemon, or a skipped spin-up looks identical to "no new messages" in the output unless it's flagged. The 2026-06-10 sweep shipped with Signal silently missing because the scan hung on an account lock. Failures lead the summary, in their own banner line.
10. *Rendering a per-source quiet roll-call.* "Calendar — quiet" / "PRs — nothing new" lines on every silent source bury the one change that matters and pad a no-change sweep into a report. Deltas only: changed sources get blocks, unchanged sources get nothing, and an all-quiet sweep is one line (Craig's 2026-06-11 ruling in Phase C).
@@ -314,6 +440,15 @@ Update the engine as the orchestration pattern evolves; update a plugin as its s
*** Updates and Learnings
+**** 2026-07-20: Phone delivery for signal sweeps (=agent-text=, send half)
+Auto mode now pushes a full-three-section sweep to Craig's phone over Signal via =agent-text=, the away-from-desk delivery the retired ntfy design carried before ntfy was torn down (2026-07-04). Transport is =agent-text= (the renamed Signal pager), not ntfy. Signal-only by Craig's 2026-07-20 ruling: a quiet sweep's =nothing= heartbeat never reaches the phone — silent-until-signal governs the phone channel too, so the higher-cost channel only fires when a sweep has real signal, while the in-session heartbeat stays as proof the loop ran. Falls back to inline when =agent-text= is absent. Only the send half ships; reply polling (the old =phone-recv=) waits on the reply-correlation follow-up, because a Signal reply fans out to every linked device and neither knows which page it answers.
+
+**** 2026-07-18: Three-section digest + close-by-default (Phase C/D rewrite)
+Craig's ruling after a 42h-gap sweep where the long-form report (top signals + per-source breakdown + 7-option action menu) was followed by "summarize the notable items" — and the digest that answered it was the report he wanted first. Phase C now renders ==TASKS== (work items needing Craig, solo-executable first, priority order) / ==FYI== (work context, no action owed) / ==MISC== (everything outside the project, actions stated inline), then exactly two options (close-and-file / close-and-execute-solo, the latter only when solo items exist), timestamp last. Per-source blocks and the itemized action menu are gone from the default surface (long form on request). Phase D became the close: it runs as the next action no matter what Craig replies (unless he explicitly holds), includes the mail hygiene on every scanned account without itemized confirmation, files the TASKS, clears resolved unacked items, advances the sentinel, and tears down started services. Solo = mechanical + standing-approved + no prose under Craig's name; prose sends and destructive non-mail actions stay gated. The stay-open-until-confirmed exit loop is retired. Same-day addendum: the "and reroute" modifier ("1 and reroute") — MISC items are surfaced-only by default (never filed to this project's todo.org); appending the modifier delivers each outside-project item to its owner's inbox via inbox-send per the cross-project rule.
+
+**** 2026-06-15: Auto mode (unattended monitoring)
+Added a self-running mode for when Craig is away but wants tight awareness — a =/loop= in the live session running accumulate-don't-mutate sweeps with "close the triage" / "stop the triage" as the gated checkpoint. Born the morning Craig cleared his day for a family emergency and wanted the desk watched while in and out. Design decisions (work-project proposal, ratified by Craig 2026-06-15): the unacked-responses list is durable in =.ai/triage-intake-unacked.org= (survives a crash/clear, the away-from-desk case it exists for); the sentinel advances only at close, preserving the scanned-before invariant; delivery is an in-session loop so MCP auth is inherited (a detached cron schedule belongs to the morning-ops orchestrator, not here, because of the headless-auth wall); it stays a mode of this engine, distinct from but reusable by that orchestrator. Same-day addendum (work, 2026-06-15): each sweep ends with a date/time/timezone stamp on its own final line (printed on quiet sweeps too, as proof the loop ran) so an away reader gauges freshness at a glance.
+
**** 2026-05-01: Initial creation
Extracted from daily-prep's Phase 3 pattern as a standalone, lightweight, between-meetings sweep.
@@ -327,7 +462,7 @@ The sentinel is checked into git, but git tracks content, not mtime — so an mt
Craig, via the work project's same-day handoff: "we only need to report if anything's changed when we do triage intake." Sweep summaries report deltas only — a new invite, a new/moved/cancelled event, a new message needing attention. Unchanged sources get no block (the "Calendar — quiet" roll-call is retired), and an all-quiet sweep renders as a single "HH:MM sweep: no changes" line. Failures keep their loud banner (never folded into the no-change line) and the suggested-actions line stays when actions are queued. Same ruling: the telegram plugin's dev-community group traffic is dropped from reports entirely unless Craig asks (see that plugin's 2026-06-11 note).
**** 2026-06-10: Loud failure surfacing (Phase C item 0 + Common Mistake 9)
-Craig: "highlight any failures in daily triage loudly. I get important communication from all these channels." Trigger: the 2026-06-10 sweep shipped with Signal silently missing — a standalone receive hung on the account lock while the signel daemon owned it, and the failure looked identical to a quiet source. Failures now lead the summary in a ⚠ SCAN FAILED banner; the telegram plugin's failure path points at this rule.
+Craig: "highlight any failures in daily triage loudly. I get important communication from all these channels." Trigger: the 2026-06-10 sweep shipped with Signal silently missing — a standalone receive hung on the signal-cli account lock while another client held it, and the failure looked identical to a quiet source. Failures now lead the summary in a ⚠ SCAN FAILED banner; the telegram plugin's failure path points at this rule.
**** 2026-05-26: Refactor into engine + source plugins
Split the monolithic workflow into a source-agnostic engine (this file) and per-source plugins named =triage-intake.<source>.org=. The engine carries the anchor/sentinel logic, the four-bucket model, the Phase A-D orchestration, the todo.org persistence convention, and the exit criteria. Each source's scan/classify/render/action knowledge moved to its own plugin. General plugins (personal-gmail, personal-calendar, cmail, github-prs) live in =.ai/workflows/= and are template-synced; project-specific plugins (a work project's Linear, work Gmail, work Slack, enterprise PRs) live in the project's =.ai/project-workflows/= and are never synced. Phase 0 globs *both* directories — the loud requirement, because missing the project dir silently halves the sweep. Naming convention: first dot is the engine/plugin boundary, deeper dots reserved for sub-adapters. This removed all DeepSat/Linear specifics from the engine; they become work-project plugins.
diff --git a/.ai/workflows/triage-intake.personal-calendar.org b/.ai/workflows/triage-intake.personal-calendar.org
index bf7d543..b5ee67a 100644
--- a/.ai/workflows/triage-intake.personal-calendar.org
+++ b/.ai/workflows/triage-intake.personal-calendar.org
@@ -1,5 +1,5 @@
#+TITLE: Triage Intake — Personal Calendar Source
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-26
# Source plugin for the triage-intake engine. See triage-intake.org for the
diff --git a/.ai/workflows/triage-intake.personal-gmail.org b/.ai/workflows/triage-intake.personal-gmail.org
index aa0554d..7fb1231 100644
--- a/.ai/workflows/triage-intake.personal-gmail.org
+++ b/.ai/workflows/triage-intake.personal-gmail.org
@@ -1,5 +1,5 @@
#+TITLE: Triage Intake — Personal Gmail Source
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-26
# Source plugin for the triage-intake engine. See triage-intake.org for the
@@ -21,10 +21,29 @@ Personal Gmail unread in the inbox since the anchor:
mcp__google-docs-personal__listMessages q="is:unread in:inbox after:<anchor-epoch>" maxResults=100
#+end_src
-⚠ *Express the cutoff as the literal UNIX epoch* — =after:1778856990=, not =after:YYYY/MM/DD=. Gmail's =after:YYYY/MM/DD= operator only supports day resolution; the =YYYY/MM/DD HH:MM:SS= form is NOT valid syntax — Gmail parses the space as a term separator, treats =HH:MM:SS= as a search term that never matches, and returns 0 results, silently masking unread mail. The engine supplies =<anchor-epoch>= because this source declares =ANCHOR: epoch=.
+⚠ *Express every anchor cutoff as the literal UNIX epoch* — =after:1784177122= and =before:1784177122= for the same anchor, never the =YYYY/MM/DD= form. This governs *both* anchored queries: the scan above and the backlog-residue probe below. They must meet at the same instant or mail falls between them permanently. Gmail's day-resolution operators fail two different ways: =after:YYYY/MM/DD HH:MM:SS= is not valid syntax at all — Gmail parses the space as a term separator, treats =HH:MM:SS= as a search term that never matches, and returns 0 results, silently masking unread mail — while =before:YYYY/MM/DD= is valid but excludes the named day entirely, so pairing it with a second-resolution scan leaves the whole anchor day covered by neither query. The engine supplies =<anchor-epoch>= because this source declares =ANCHOR: epoch=.
+
+The rule binds the *anchor* windows only. The date-slice walk below deliberately uses =before:<oldest-full-day-seen>= at day resolution — safe there because consecutive slices overlap and get deduped by message id.
⚠ *Do NOT add =-category:promotions -category:social=.* That filter masked 67 promo+social messages across two runs (2026-05-04, 2026-05-06), both needing a follow-up sweep. Pull the full unfiltered set; the trash-leaning bias in Classify handles promotions and social directly.
+⚠ *The MCP caps at =maxResults=100= and exposes NO =pageToken= parameter.* The response carries a =nextPageToken=, but the tool can't consume it, so a pile over 100 is silently truncated — the tail below the cap never gets classified, and every later anchored sweep skips it (it predates the new anchor). This is exactly how a 300+ backlog accumulated invisibly by 2026-07-08. Two consequences:
+
+- *Never treat a 100-row result as complete.* When a scan returns exactly 100, walk the tail in *date slices*: re-query with =before:<oldest-full-day-seen>= (day resolution), repeat until a page returns fewer than 100, dedupe by message id across slices (the day-resolution boundary overlaps).
+- *Never report =resultSizeEstimate= as a count.* It's unreliable — observed stuck at "201" across three different queries whose real union exceeded 300.
+
+*** Backlog-residue check (every sweep — cheap, mandatory)
+
+The anchored scan is blind to anything unread from *before* the anchor. After it, run one probe for pre-anchor residue:
+
+#+begin_src text
+mcp__google-docs-personal__listMessages q="is:unread in:inbox before:<anchor-epoch>" maxResults=5
+#+end_src
+
+The cutoff is the epoch, matching the scan's =after:<anchor-epoch>= — see the epoch rule above.
+
+If it returns any messages, surface one loud line in the sweep summary: "Backlog: unread predating the anchor exists (N+ shown; date-slice to inventory)" and offer a backlog sweep. Never fold the residue into a quiet sweep — an anchored "no changes" claim is only true for the window the scan saw. (Added 2026-07-08 after ~300 pre-anchor unread accumulated unseen; the probe returns actual messages, so it works where the estimate lies. Shipped with a day-resolution cutoff that hid the entire anchor day; fixed to epoch 2026-07-16 after a home sweep reported the backlog clear while two July-15 messages sat unread.)
+
** Classify
Bias: *trash-leaning* — personal Gmail is high noise volume.
diff --git a/.ai/workflows/triage-intake.telegram.org b/.ai/workflows/triage-intake.telegram.org
index 9caa4e1..1319da5 100644
--- a/.ai/workflows/triage-intake.telegram.org
+++ b/.ai/workflows/triage-intake.telegram.org
@@ -1,5 +1,5 @@
#+TITLE: Triage Intake — Telegram Source
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-06-09
# Source plugin for the triage-intake engine. See triage-intake.org for the
@@ -30,12 +30,27 @@ Telega does not autostart with the Emacs daemon. "Down" is its normal state
unless Craig has Telegram open in Emacs. The scan therefore runs the full
lifecycle every time, never skips because the server is down:
+⚠ *DOWN / not-loaded is the TRIGGER to launch, never a reason to skip or fail.*
+This is the exact mistake two projects (work + home, 2026-07-24) made: they
+probed telega, saw =(telega-server-live-p)= nil or telega not =featurep=, and
+reported =SCAN FAILED: telegram — not loaded= or a silent SKIP — a *blind*
+sweep — instead of running Step 1 to start it. A down or unloaded telega is the
+normal entry state; =(telega t)= both LOADS the package and STARTS the docker
+server (work confirmed: down → =(telega t)= → Ready, 18 chats). So the plugin
+MUST run Step 1's launch whenever telega is down/unloaded, wait for Ready, then
+scan. =SCAN FAILED= is reserved for a launch that was actually ATTEMPTED and did
+not reach Ready (image missing, server crash on start, daemon unreachable) —
+never for the pre-launch down state itself. The =:ENABLED:= guard above tests
+whether telega is INSTALLED (=fboundp=), not whether the server is up; a down
+server never disables the source.
+
1. Record prior state: TELEGA_WAS_RUNNING via (telega-server-live-p).
2. Launch (only if not running):
emacsclient -e "(progn (setq telega-use-docker t) (telega t) 'started)"
- The setq is mandatory defense: tdlib segfaults outside docker mode
- (2026-06-09), and Craig's daemon currently has telega-use-docker nil.
- Wait ~2s for Ready, then (telega--loadChats 'main) until telega--chats
+ The setq is mandatory defense: tdlib crashed in native mode when this was
+ set up (2026-06-09) — a separate matter from the SEGFAULT gotcha, which is
+ about the loadChats argument — and Craig's daemon defaults to nil.
+ Wait ~2s for Ready, then (telega--loadChats '(:@type "chatListMain")) until telega--chats
is populated.
3. Check messages: the maphash unread scan in ** Scan Step 2 (filters the
messageContactRegistered join-notice noise).
@@ -48,10 +63,13 @@ lifecycle every time, never skips because the server is down:
Verify: telega-server-live-p → nil, no zevlg/telega-server container in
docker ps. If Craig had it running, leave it untouched.
-If any lifecycle step fails (docker image missing, server crash, daemon
-unreachable), the sweep reports it as SCAN FAILED at the top of the summary
-per the engine's failure rule — never as a silent skip. Craig gets real
-traffic here.
+If any lifecycle step fails *after the launch was attempted* (docker image
+missing, server crash on start, daemon unreachable, Ready never reached), the
+sweep reports it as SCAN FAILED at the top of the summary per the engine's
+failure rule — never as a silent skip. This does NOT cover the ordinary
+pre-launch down state: a down server means "run Step 1," not "SCAN FAILED."
+Craig gets real traffic here, so a blind sweep that skipped the launch is worse
+than a clean failure — it hides real unread messages behind a false all-clear.
** Scan
@@ -85,22 +103,58 @@ TELEGA_WAS_RUNNING=$(emacsclient -e "(and (fboundp 'telega-server-live-p) (teleg
*** Step 1 — start (docker mode) if not already running, wait for Ready
#+begin_src bash
-# `(telega t)` starts without popping the root buffer. Docker mode (the stable
-# path — see the SEGFAULT gotcha) reconnects the persisted ~/.telega session in
-# ~2s. Then load the main chat list so telega--chats populates.
+# `(telega t)` starts without popping the root buffer. Docker mode reconnects the
+# persisted ~/.telega session in ~2s. Then load the main chat list so
+# telega--chats populates.
+#
+# The `(setq telega-use-docker t)` is mandatory and must come BEFORE `(telega t)`:
+# tdlib crashed in native mode when this was first set up (2026-06-09), and the
+# daemon's default is nil unless something (e.g. an Emacs-config :custom) has
+# already forced it. It was missing here while the Quick Reference required it —
+# a session that started telega without it on a native-mode daemon would take the
+# untested path. Match the Quick Reference exactly.
+#
+# Note this is a SEPARATE concern from the SEGFAULT gotcha below: that gotcha is
+# about the `loadChats` argument, and the deaths it explains happened in docker
+# mode. Docker mode is not a defense against it, and it is not evidence for
+# docker mode. Keep both.
emacsclient -e "(progn
+ (setq telega-use-docker t)
(unless (and (fboundp 'telega-server-live-p) (telega-server-live-p)) (telega t))
'started)"
# Poll until Ready with chats synced, or a crash/timeout. Background this with an
# until-loop so the wait doesn't block; exit on Ready-with-chats OR an abnormal
# server exit. Then force a chat-list load if the hash is thin:
-emacsclient -e "(progn (ignore-errors (telega--loadChats 'main)) (ignore-errors (telega--loadChats 'main)) 'loaded)"
+# NOTE: the chat-list argument must be a TL object, not the symbol 'main.
+# `telega--loadChats' puts it straight into the request as :chat_list, and a
+# bare symbol kills the server outright (see the SEGFAULT gotcha below).
+#
+# The liveness check on the tail is the load's only failure signal. `ignore-errors'
+# catches nothing here, because a bad argument kills the server process rather than
+# signalling in elisp, so without this the call returns 'loaded either way.
+# The `fboundp' guard matches Step 0: if the launch failed outright telega is not
+# loaded, and that should read as 'server-died like any other failure rather than
+# signalling void-function.
+emacsclient -e "(progn (ignore-errors (telega--loadChats '(:@type \"chatListMain\"))) (ignore-errors (telega--loadChats '(:@type \"chatListMain\"))) (if (and (fboundp 'telega-server-live-p) (telega-server-live-p)) 'loaded 'server-died))"
#+end_src
On a persisted session telega reaches status "Ready" within ~2s; the chat list
loads over a few more. If =(hash-table-count telega--chats)= is 0 or thin,
re-issue =telega--loadChats= and poll until it stabilizes.
+⚠ *=server-died= is SCAN FAILED, never a quiet account.* A server that dies
+during the load leaves a thin =telega--chats= hash, and a thin hash reads exactly
+like an account with little unread. That is the same false all-clear the
+down/not-loaded rule exists to prevent, arriving one step later in the lifecycle.
+It also fits the SCAN FAILED definition above: the launch was attempted and did
+not hold. So on =server-died=, report SCAN FAILED rather than scanning, and never
+report a low unread count from that run.
+
+This is the independent evidence the SEGFAULT gotcha asks for when it says to
+treat a short chat list as a real short list. Without the check there is no way
+to tell the two apart, which is how the =loadChats= crash stayed invisible
+through two investigations.
+
*** Step 2 — read unread, classified by last-message type
The single most important filter: =messageContactRegistered=. Telegram counts a
@@ -157,24 +211,61 @@ stays non-nil). =telega-server-kill= is what actually stops the server. Call
left in =docker ps=. Skipping this whole branch when =TELEGA_WAS_RUNNING= is t is
the point of Step 0: never tear down a session Craig is actively using.
-⚠ *SEGFAULT GOTCHA — crashes are spontaneous; treat server death as routine.*
-The dockerized =telega-server= (=zevlg/telega-server:latest=, image built
-2026-06-04, tdlib 1.8.64) SIGSEGVs (exit 139) *on its own*, minutes-to-hours
-into a session — 11 host coredumps between 2026-06-09 and 2026-06-11, several at
-times when no triage verb was running. The 2026-06-11 investigation reproduced
-the crash-free verbs and the spontaneous deaths side by side: coredump
-backtraces show a corrupted stack (memory corruption in the musl build), and
-no newer image exists upstream. Earlier theories — "native mode is the trigger",
-"toggle-read is the trigger" — were timing coincidences; the verbs are sound.
+⚠ *SEGFAULT GOTCHA — this was our bug, not tdlib's. Root-caused 2026-07-28.*
+=telega-server= dies with =Unexpected char 'm' in plist value= followed by
+=Assertion failed: false (telega-dat.c: tdat_plist_value: 500)=. The cause was
+this workflow: Step 1 called =(telega--loadChats 'main)=.
+
+The chain. =telega--loadChats= is a raw TL wrapper — it drops its argument into
+the request as =:chat_list= with no conversion. =telega-server--send= then
+=prin1='s the whole plist, and =telega--tl-pack= passes atoms through untouched,
+so the symbol goes out on the wire bare as =main=. The C parser
+(=server/telega-dat.c=, =tdat_plist_value=) accepts only =(=, =[=, ="=, =-=, a
+digit, =t=, =:=, or =n= to start a value. It hits =m=, prints that line, and
+calls =assert(false)=, which aborts the process. The =m= in the error is
+literally the first character of =main=.
+
+The symbol shorthand is real but belongs to a different layer:
+=telega-filter.el= and =telega-folders.el= convert =(eq cl-fspec 'main)= into
+='(:@type "chatListMain")=. The raw TL layer never does. telega's own callers
+always pass the object (=telega.el:290=, =telega-tdlib-events.el:516=).
+
+Proved by experiment, not inference (2026-07-28): from a live Ready server,
+=(telega--loadChats 'main)= killed it within seconds and added one coredump,
+with that exact assertion; a restart plus =(telega--loadChats '(:@type
+"chatListMain"))= survived three consecutive calls with no new coredump and no
+assertion.
+
+*The previous entry here was wrong and cost real time.* It recorded the deaths
+as spontaneous musl memory corruption and declared "the verbs are sound", which
+sent later investigations at the docker image and tdlib versions instead of at
+this file. The corrupted stack in the backtraces is what an =assert= abort looks
+like, not independent evidence of a memory bug. If crashes are ever seen again
+with *no* triage verb running, that is a genuinely separate cause and needs its
+own investigation — do not reuse the old spontaneous-crash story to explain it.
+
+*This crash kills a scan; it does not silently shorten one.* An earlier draft of
+this section claimed the reported "19 chats of ~50" was truncation caused by the
+bad call. That was wrong, and work disproved it at the wire level on 2026-07-28:
+with the corrected call their count is 19 before the first load and 19 after five
+(four on =chatListMain=, one on =chatListArchive=). Nineteen is the real size of
+that account. The same reading here — 19 stable across three corrected loads —
+was already sitting in the evidence and should have retired the claim before it
+was written down. Treat a short chat list as a real short list unless something
+independently shows the server died mid-sync.
+
+=ignore-errors= around the call never helped — the failure is the server process
+dying, not an elisp signal, so there is nothing for it to catch. That is why the
+death is easy to miss from inside elisp, and why a caller should check
+=(process-live-p (telega-server--proc))= after a load rather than trusting a
+returned value.
Operationally: docker mode stays mandatory (=telega-use-docker= = t; the setq
before =(telega t)= is still the right defense), and *every action batch checks
the server first* — =(process-live-p (telega-server--proc))= — restarting via
-=(telega t)= when dead and re-checking Ready before firing verbs. A mid-sweep
-death is recoverable, not an abort: restart, confirm Ready, resume. Durable-fix
-candidates if the crashing gets worse: pin a pre-2026-06 image digest, build
-=telega-server= natively against tdlib, or report upstream to zevlg with the
-coredumps (=coredumpctl list /usr/bin/telega-server=).
+=(telega t)= when dead and re-checking Ready before firing verbs. Any argument
+handed to a =telega--*= TL wrapper must be a TL object or a plain
+string/number/list, never a bare symbol.
Defense in depth: even if the server does die, the scan still works because it
reads the cached =telega--chats= hash, not a live query. A dead server is
diff --git a/.ai/workflows/work-the-backlog.org b/.ai/workflows/work-the-backlog.org
new file mode 100644
index 0000000..ea3f402
--- /dev/null
+++ b/.ai/workflows/work-the-backlog.org
@@ -0,0 +1,266 @@
+#+TITLE: Work the Backlog
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-02
+
+* Overview
+
+The single home for the autonomous task-execution loop: take a set of marked, solo-doable tasks from the project's =todo.org= and work them unattended, each held to the full quality bar, under a fixed safety contract. Spec: =rulesets/docs/specs/2026-06-16-autonomous-batch-execution-spec.org=.
+
+Two callers feed it, differing only in how they build the task set and which session mode they pass:
+
+- The *inbox auto-loop* (=inbox.org= auto mode) chains here after its routing completes, with a tag/priority query, file-only mode, cap 1.
+- The *no-approvals speedrun* preset feeds an explicit ordered list with autonomous-commit + always-push + paging-on, after a pre-flight Q&A that front-loads every decision.
+
+This workflow owns the execution logic — eligibility gate, defer checklist, quality bar, run cap. Callers own input assembly and mode selection. Capture-routing (inbox surfaces) stays entirely in =inbox.org=; this file never reads an inbox.
+
+* When to Use This Workflow
+
+Invoked by its two callers, or directly by phrase:
+
+- *Speedrun triggers:* "speedrun", "no approvals speedrun", "speedrun these: <task set>" — run the no-approvals speedrun preset (below). The word "speedrun" always routes here, even when the phrase also says "no approvals": plain =no-approvals.org= is the general session mode; the speedrun is this workflow's preset over an explicit task set.
+- *Loop caller:* =inbox.org= auto mode chains here after its routing (below). Not phrase-triggered.
+
+Manual fallback: "work the backlog" / "work the backlog with <task set>" — gather the three inputs below (ask for whichever are missing, defaulting to file-only mode; default cap is the list length for an explicit set, 1 for a query) and run the loop.
+
+* Inputs — the caller contract
+
+A caller hands this workflow three things:
+
+1. *A task set* — an ordered list of candidate task headings from the project's =todo.org=. Either an explicit ordered list (speedrun) or the result of a tag/priority query (the loop). The loop does not care how the set was assembled; it receives an ordered list of candidates.
+2. *A session mode* — two orthogonal flags:
+ - *Commit autonomy:* =file-only= (default) or =autonomous-commit=. See "Commit autonomy" below.
+ - *Paging:* on or off. End-of-set only.
+3. *A run cap* — the hard maximum number of tasks to complete this run.
+
+It returns a per-task outcome and a run summary.
+
+* Outcomes — the per-task vocabulary
+
+Every task in the set ends in exactly one of:
+
+- =implemented-committed= — implemented, committed (and pushed per the project's flow) under =autonomous-commit=.
+- =implemented-diff-surfaced= — implemented, diff surfaced, *not* committed (=file-only=).
+- =deferred-VERIFY= — a defer-checklist hit; a =VERIFY= filed naming what's missing or risky.
+- =dropped-by-craig= — removed from the run at the speedrun pre-flight Q&A ("skip this").
+- =skipped-ineligible= — failed the mechanical eligibility gate.
+- =failed= — implementation was attempted and abandoned: the tree is left working (never commit a broken state), the failure is surfaced in the run summary, and the run continues to the next task.
+
+The run summary lists each task with its outcome, plus the remaining set when the cap stopped the run.
+
+* The loop
+
+For the task set, in order, until the run cap is hit:
+
+1. *Eligibility gate* (below). Ineligible → record =skipped-ineligible=, next task.
+2. *Scope read* of the relevant code. Cheap; just enough to run the defer checklist.
+3. *Defer checklist* (below). Any hit → defer: file the =VERIFY= naming the gap and record =deferred-VERIFY= (or, under the speedrun preset, route a quick-question gap to the pre-flight Q&A), next task.
+4. *Implement* under the project's commit discipline: TDD red→green→refactor, then the isolated adversarial review (=publish= Step 1) with its re-review loop, fix all Critical/Important findings, then close the task per =todo-format.md='s completion rules. Decompose into as many logical commits as the change needs — size is not capped. If implementation fails partway, leave the tree working, record =failed=, surface it, and continue to the next task.
+5. *Commit autonomy branch:*
+ - =file-only= → surface the diff, do *not* commit. Record =implemented-diff-surfaced=.
+ - =autonomous-commit= → =/voice personal= on the message, commit individually, push per the project's flow. Record =implemented-committed=.
+6. *Record metrics* for the task (the JSONL append — see Metrics below).
+7. Decrement the cap. At zero, stop.
+
+After the set: if the paging flag is set, fire the end-of-set page (below). Surface the run summary either way.
+
+* Eligibility gate — mechanical, no judgment
+
+A task is autonomous-safe when *both* hold. This layer is a lookup, not a judgment; all the judgment lives in the defer checklist.
+
+1. *Status is =TODO=* — never =VERIFY=, =DOING=, =DONE=, or =CANCELLED=. =VERIFY= marks "awaiting Craig's input"; auto-implementing one defeats the check it represents. The do-not-implement set is safe-by-omission: anything not plainly =TODO= (plus any project-declared "hold" marker) is out.
+2. *Tagged =:solo:=* — the autonomy tag, resolved against the project's priority/tag scheme header in =todo.org= (never hardcoded). =:solo:= carries the hard definition in =todo-format.md=: completable and verifiable without Craig beyond at most one or two quick decisions answerable up front, no design deliberation. A project whose scheme declares a different autonomous-safe tag set overrides the default.
+
+Terminology: *speedrunnable means tagged =:solo:=*. It does not mean =:quick:= or require =:quick:solo:=. The =TODO= status check above is the execution-state gate over that speedrunnable set.
+
+Priority and =:next:= drive *ordering* within the eligible set, not eligibility ([#A] before [#B] before [#C], then the author's ordering). =:quick:= is an effort hint for batching and duration estimates — never a gate.
+
+Task *size* is deliberately absent from this gate. A large but well-specified, decision-free task is in scope and gets decomposed into per-logical-commit chunks during implementation. Size never sends a task away; only *deliberation* or *risk* does (the checklist below).
+
+*No scheme header → don't run.* The gate reads =:solo:= semantics from the project's scheme header; a =todo.org= without one leaves the tag undefined (=todo-format.md= makes the header mandatory). Surface that the header is missing and stop rather than guessing eligibility.
+
+* The defer checklist — act vs file
+
+After the scope read, run each eligible candidate through the checklist. Each item is a concrete, answerable question, not an adjective. *Any* hit — or any "unsure" — defers the task. Only a task that clears every item is implemented.
+
+1. *Test-writability (the keystone).* Can I write the failing test from the task text — plus any decisions gathered up front — without inventing a requirement? *No / unsure* → underspecified. Under the speedrun preset, if the gap is one or two quick answerable questions, route it to the pre-flight Q&A; otherwise file a =VERIFY= noting what's missing. Under the unattended loop, file the =VERIFY= (no one to ask). *Open-ended goals are a specific, recognizable failure of this item:* a task phrased as an absence ("find bugs until none remain," "refactor until nothing worthwhile is left," "clean it up") has no writable acceptance test and so isn't really =:solo:=, even when tagged. Don't guess a stopping point — defer it and note that it needs measurable acceptance criteria (bound the surface, characterization net, dispositioned findings, objective floor — see =todo-format.md='s "Making an open-ended task measurable"). Once those are in the task body, it becomes runnable.
+2. *Data-loss / irreversible / external operation.* Does implementing it require any of: =rm= of non-scratch data, =git reset --hard= / force-push, =DROP= / =DELETE= / =TRUNCATE=, file truncate/overwrite of persisted content, a schema or data migration, any external or shared-state mutation, any credential touch? *Yes* → do NOT implement; file a =VERIFY= naming the risk. This is the hard safety gate; an upfront answer never overrides it without an explicit checkpoint.
+3. *Already-satisfied.* Does the scope read show the desired end-state already holds? *Yes* → file a =VERIFY= noting it and move on. Don't make a no-op change.
+4. *Design deliberation.* Does the task carry an unresolved design question, a "weigh these approaches" with real tradeoffs, or a TBD that isn't a quick factual answer? *Yes* → under the speedrun preset, if it collapses to one or two quick questions, route to the pre-flight Q&A; otherwise file and surface as a =/start-work= candidate. Under the loop, file. The discriminator is *quick-answerable question* vs *deliberation* — never task size.
+
+When genuinely unsure which side a task falls on, defer — a wrong auto-implement costs a revert *and* the next-session correction.
+
+** Filing the deferral =VERIFY=
+
+Every checklist hit files a =VERIFY= in the project's =todo.org=, per =todo-format.md='s VERIFY rules:
+
+- *Dedup first.* If a =VERIFY= sibling for this deferral already exists (a prior run filed it), don't file another — record the outcome as =deferred-VERIFY= with a "previously filed" note and move on. The deferred task keeps its =TODO= status and tags, so without this check every subsequent run would re-defer and re-file.
+- *Placement:* sibling of the deferred task (the deferred task is the trigger) — a =**= task gets its =VERIFY= at =**=, a =***= sub-task gets it at =***= under the same parent, never deeper.
+- *Heading:* carries the question or risk on its own ("VERIFY <topic> — migration touches persisted rows").
+- *Body:* which checklist item hit, what's missing or risky, and what answer or action would make the task runnable. For an already-satisfied hit, the evidence that the end-state already holds.
+
+** Routing a quick-question gap (speedrun only)
+
+Under the speedrun preset, a checklist-1 or checklist-4 hit that collapses to one or two quick answerable questions routes to the pre-flight Q&A instead of deferring (see the preset section below). The discriminator: a *quick question* is a factual or preference pick answerable in one line without weighing tradeoffs ("cap at 5 or 8?", "which config key name?"); *deliberation* is anything that needs tradeoffs weighed, options explored, or code read by Craig. A task needing three or more questions isn't quick-question-gapped — it's underspecified; file the =VERIFY=. Checklist item 2 (data-loss / irreversible) never routes to the Q&A: an upfront answer doesn't override the hard safety gate.
+
+The unattended loop has no one to ask — every hit defers there.
+
+* Per-task quality bar
+
+Autonomy changes who approves, not what quality means. Per task, non-negotiable:
+
+- *TDD* per =testing.md=: red first, green, refactor. The keystone checklist item already proved the failing test is writable.
+- *Verification* per =verification.md=: fresh evidence, full suite green before any commit.
+- *Isolated adversarial review* before every commit, dispatched per the =publish= skill's Step 1 — never an inline self-review, however small the diff. Critical and Important findings block until fixed, and each fix goes back to the *same* reviewer until it approves. Minor findings never earn another round.
+ - *When the review can't reach approval* — three rounds without it, a finding that recurs after being reported fixed, or a =Needs Discussion= verdict — the unattended run has no one to ask. Record the task =failed= with the standing findings in its result, leave the tree working, and continue to the next task. Never commit past a blocking finding because nobody is awake to adjudicate — an unreviewed commit landing overnight is the outcome this gate exists to prevent.
+- *=/voice personal=* on every commit message on the =autonomous-commit= path (or the patterns walked inline if the skill is unavailable), message printed inline so the log shows what landed.
+- *Task closure* per =todo-format.md=: depth-based completion (keyword + =CLOSED:= at level 2, dated rewrite at level 3+).
+- *One logical change per commit.* A large task becomes several commits, not one omnibus.
+
+* Commit autonomy
+
+=file-only= is the default: surface the diff, never commit. =autonomous-commit= is honored only when the project carries the commit-autonomy waiver, read fresh each run — never from memory of past runs or "this project usually allows it."
+
+The waiver lives in the project's =.ai/notes.org= *Workflow State* section as marker lines, the same shape as the workflow markers already there:
+
+#+begin_example
+:COMMIT_AUTONOMY: yes
+:LOOP_MAY_COMMIT: yes
+#+end_example
+
+- =:COMMIT_AUTONOMY: yes= — the project has the waiver. An =autonomous-commit= request (the speedrun preset, or a manual run asking for it) is honored.
+- =:LOOP_MAY_COMMIT: yes= — the *unattended loop caller* may also commit. It requires =:COMMIT_AUTONOMY:= alongside it; the split exists because "Craig-initiated speedrun may commit" and "the recurring loop may commit unattended" are different levels of trust. Without this flag the loop stays =file-only= even when the project holds the waiver.
+
+An absent marker means no. Anything other than a plain =yes= value also means no. The read is one grep of the Workflow State section — a lookup, not a judgment.
+
+*The degrade contract.* When a caller requests =autonomous-commit= and the required marker is missing, degrade to =file-only= and surface it in both the run intro and the run summary: "autonomous-commit requested, no :COMMIT_AUTONOMY: waiver in notes.org — running file-only." Never honor the request without the marker, and never drop to file-only silently — the first commits into a project that didn't opt in, the second hides why nothing got committed.
+
+* Bounding the run
+
+The cap is a hard per-run task ceiling passed by the caller — the kill switch a runaway can't exceed:
+
+- *Loop caller default: 1.* Implement the highest-priority eligible candidate, record, stop; the next tick continues.
+- *Speedrun: the length of the explicit list*, capped at a ceiling — the human bounded the set by naming it.
+
+Even the speedrun stops at the cap and surfaces (and, with paging on, pages) the remainder. The cap bounds task *count*, not cost; a token budget is logged as vNext.
+
+* Context hygiene — auto-flush between tasks
+
+Task boundaries are clean boundaries by construction: the previous task is closed and committed (or filed), nothing is half-edited. When the context window grows heavy mid-run, run the flush skill's *auto mode* between tasks: checkpoint the session anchor with the remaining task set, session mode, and cap in Next Steps (so the resumed context continues the run blind), arm the self-injection (=.ai/scripts/self-inject.sh= via =tmux run-shell -b=), and end the turn. The fresh context resumes from the anchor and works on. Unattended runs only — the keystroke-collision hazard and the full mechanism live in the flush skill.
+
+* End-of-set page
+
+With paging on, fire one page when the set is done or the cap is hit — end-of-set only, never per-task:
+
+#+begin_src sh
+notify info "Page" "<project>: <N> done, <M> remaining — <one-line summary>" --persist
+#+end_src
+
+=--persist= keeps it on screen until dismissed, and =info= is the notification urgency convention (persistent but never crash-scary). The notification fires when the set completes *or* the cap stops the run, either way exactly once. The message carries the project name, the completed count, and the remaining count (with skipped tasks noted in the run summary) so Craig can confirm ready and name the next project in one reply. =notify= is the desktop channel (the "page me" surface); a run that expects Craig to be away also fires =agent-text= with the same message (the Signal phone channel, "text me"). See protocols.org "Reaching Craig".
+
+* Metrics
+
+Each task outcome appends one JSON line to the project's =.ai/metrics/work-the-backlog.jsonl= — git-tracked, append-only, =jq=-queryable. Create the directory and file on the first append. Logging is a side effect only: a failed append surfaces a warning in the run summary but never blocks, reorders, or aborts execution.
+
+One record per task, written at the moment its outcome is decided:
+
+| Field | Meaning |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =ts= | ISO-8601 timestamp of the task outcome |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =run_id= | UUID shared by every record in one run (=uuidgen= at run start) |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =project= | project basename |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =caller= | =loop= / =speedrun= / =manual= |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =task= | the task heading (slug) |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =outcome= | =implemented-committed= / =implemented-diff= / =deferred-verify= / =skipped-ineligible= / |
+| | =dropped-by-craig= / =failed= |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =defer_reason= | =underspecified= / =data-loss= / =already-satisfied= / =needs-deliberation= — set on |
+| | =deferred-verify= records only |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =upfront_decision= | =true= when a pre-flight answer was recorded and used for this task |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =wall_clock_s= | seconds from task start to outcome |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =commit_sha= | committed tasks: the commit SHA (comma-separated when the task decomposed into several); empty |
+| | otherwise |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =review_findings= | count of =/review-code= Critical + Important findings on this task |
+|--------------------+-------------------------------------------------------------------------------------------------|
+
+The =outcome= slugs map one-to-one onto the outcome vocabulary above (=implemented-diff= is =implemented-diff-surfaced=; =deferred-verify= is =deferred-VERIFY=). Per-run rollups (attempted / completed / deferred / dropped, wall-clock total, findings per commit) are computed at synthesis, not stored per record. The =commit_sha= field is what the synthesis step's corrections signal keys on — whether a later commit reverted or hand-fixed an autonomous one — so never omit it on a committed task.
+
+* Caller: the inbox auto-loop
+
+=inbox.org= auto mode chains here as an explicit second step *after* its routing completes — never as a phase inside inbox processing. When a cycle files new items and Craig answers "run this batch next?" with yes, auto mode invokes this workflow with:
+
+- *Task set:* the eligibility query over the queued/filed items — status =TODO= + =:solo:= per the scheme header, priority-ordered.
+- *Session mode:* =file-only=, paging off. (A project carrying both =:COMMIT_AUTONOMY:= and =:LOOP_MAY_COMMIT:= markers opts the loop into commits — see Commit autonomy above.)
+- *Cap: 1.* The highest-priority eligible candidate runs, gets recorded, and the loop's next tick (or the next yes) continues from there.
+
+The loop has no human at kickoff of each task, so a needs-quick-decisions task defers with a =VERIFY= — the pre-flight Q&A is a speedrun capability, not a loop one. Startup and wrap-up never invoke this workflow.
+
+* Preset: the no-approvals speedrun
+
+The named preset is a label for one flag combination, not a second code path: *explicit ordered list + =autonomous-commit= + always-push + paging-on*, with every approval front-loaded into a single pre-flight step. "No approvals" means all input first, then hands-off — not no input ever. =autonomous-commit= still requires the =:COMMIT_AUTONOMY:= waiver (Commit autonomy above); without it the preset degrades to =file-only= and says so in the pre-flight intro.
+
+When Craig names a task set and says "speedrun":
+
+1. *Gather* the named task set.
+2. *Scope-read and classify* each task against the eligibility gate + defer checklist: *ready* (clears everything), *needs-quick-decisions* (one or two upfront-answerable questions — checklist item 1 or 4), or *drop* (data-loss/irreversible, or deliberation that isn't a quick question).
+3. *Order* the list — priority, then the author's ordering / =:next:=.
+4. *Intro the work* — present the ordered plan: what will run, what was dropped and why, and the batched questions for the needs-quick-decisions tasks.
+5. *Craig answers each question or says "skip this"* — a skip removes the task (recorded =dropped-by-craig=; the task itself stays =TODO=); an answer is recorded so implementation works from the decision, not a guess.
+6. *Run the finalized list autonomously* — no further approvals until done. Cap = the list length (the human bounded the set by naming it), still one commit per logical change, always-push per the project's flow, auto-flushing between tasks when the context grows heavy (see Context hygiene above).
+7. *End-of-set page* with completed + remaining + skipped.
+
+The batch-ask (step 4-5) is one message: each question names its task, puts the recommended answer at item 1 when there is one (per =interaction.md= — inline numbered, no popup), and offers "skip this" as the last option. Before the run starts, write each answer into its task's body in =todo.org= as a dated line — the implementation works from the recorded decision, and the record survives the session. The Q&A fires only under this preset; the loop caller never asks (its decision-needing tasks defer).
+
+*** Per-item disposition rule
+
+For every item the run picks up (this holds for any executing caller, including an auto-inbox-zero run given a standing yes):
+
+- *Feature-level task* → write a spec first (=spec-create=), don't implement directly. The spec is the run's deliverable for that item.
+- *Needs decisions you can't confidently guess* → file it as a =VERIFY= carrying the question (under this preset, one or two quick questions route to the pre-flight Q&A instead).
+- *Well-defined* → implement it, taking the time it needs.
+
+This extends the defer checklist: the checklist decides *act vs file*; this rule decides the *shape* of the act.
+
+* Synthesis: metrics → org-roam KB
+
+Trigger: "synthesize backlog metrics" (optionally a weekly scheduled run). This is the read side of the metrics log — Craig's ask was "gather data and create org-roam articles we can look at later," and this step is the second half. It is read-only over the logs plus exactly one KB write.
+
+1. *Gather the JSONL union.* Discover =.ai/metrics/work-the-backlog.jsonl= across the project roots (dirs carrying =.ai/protocols.org= under =~/code=, =~/projects=, =~/.emacs.d=). Classify each project per =knowledge-base.md= (work-root denylist, never inference) before reading it into the union.
+2. *Enforce personal-only.* A work-classified or unknown project's metrics never enter the KB write — they stay in that project's own log. Report the exclusion per the KB refusal contract: the classification, a one-line redacted summary, and where the data stayed.
+3. *Compute the rollups and trends.* Per run: attempted / completed / deferred (by reason) / dropped / failed, wall-clock total, commits landed, review findings per commit. Trends across runs: completion rate over time, defer-reason distribution, findings-per-commit trend.
+4. *Compute the corrections signal* — the key metric. For each =commit_sha= in the window, check that project's history for a later commit (within ~14 days) that reverts it or carries a fix touching the same files. A clean run is one whose autonomous commits survive untouched; a flagged run is what Craig reviews by hand. This is a cheap proxy, not proof — it flags candidates, it doesn't convict.
+5. *Write one KB node* at =~/org/roam/agents/YYYYMMDDHHMMSS-backlog-metrics-<window>.org= per =knowledge-base.md=: =:agent:metrics:= filetags, a concise title, the rollup table, the trend narrative, and =[[id:...]]= links to prior synthesis nodes so the series is traceable. Pull before writing, commit and push after — the normal KB session discipline.
+
+The KB node is the artifact Craig reads later: "are the runs completing more and getting corrected less?" should read off the trend table without touching raw logs. Synthesis never mutates the JSONL, todo.org, or any project tree.
+
+* Common Mistakes
+
+1. *Implementing a =VERIFY= or =DOING= task.* The gate is status =TODO= only — a =VERIFY= exists precisely because Craig's input is pending.
+2. *Treating =:quick:= as eligibility.* It's an effort hint. =:solo:= is the gate.
+3. *Deferring on size.* A large, well-specified, decision-free task runs — decomposed into logical commits. Size is not a checklist item.
+4. *Guessing past the keystone.* If the failing test isn't writable from the task text, the task isn't ready. Inventing the requirement is the failure the checklist exists to stop.
+5. *Rationalizing through the data-loss list.* "The migration is small" doesn't clear checklist item 2. Enumerated operations defer, full stop.
+6. *Committing in =file-only= mode.* The diff is the deliverable; the commit is Craig's.
+7. *One omnibus commit for the whole run.* Every logical change is its own reviewed commit.
+8. *Skipping =/review-code= or =/voice= because nobody's watching.* Autonomy removes interaction gates, never engineering-discipline gates (same contract as =no-approvals.org=).
+9. *Running past the cap.* The cap is the kill switch; hitting it means stop and surface, even mid-set.
+10. *Paging per-task.* One page, end of set.
+11. *Honoring =autonomous-commit= from memory.* The waiver is the marker line in =notes.org=, read fresh each run. "This project usually allows it" isn't a read.
+12. *Re-filing the same deferral =VERIFY= every run.* The deferred task stays =TODO=, so a run that skips the existing-sibling check spams =todo.org= with duplicates.
+13. *Routing a data-loss hit to the pre-flight Q&A.* Checklist item 2 is the hard gate — an upfront answer never clears it without an explicit checkpoint.
+
+* Living Document
+
+Refine as the dogfooding signal arrives — the metrics log and the corrections-in-next-session signal are the feedback loop. Fold recurring adjustments in rather than accumulating caller-side workarounds.
+
+* History
+
+Created 2026-07-02 as Phase 1 of the autonomous-batch execution spec, reconciling the inbox-zero "Phase E" proposal and the =.emacs.d= speedrun proposal into one execution loop. The auto-inbox-zero execute step in =inbox.org= reverted to routing-only in the same change so this file is the loop's only home. Phases 2-6 (same day) wired both callers, pinned the commit-autonomy waiver markers, fleshed the defer/Q&A/page mechanics, and added the metrics record + KB synthesis step.
diff --git a/.ai/workflows/wrap-it-up.org b/.ai/workflows/wrap-it-up.org
index 2d79795..a9a5895 100644
--- a/.ai/workflows/wrap-it-up.org
+++ b/.ai/workflows/wrap-it-up.org
@@ -1,10 +1,10 @@
#+TITLE: Session Wrap-Up Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-04-20
* Overview
-This workflow defines the process for ending a Claude Code session cleanly. It finalizes the session record, commits + pushes all work, and provides a warm handoff.
+This workflow defines the process for ending a Claude Code session cleanly. It finalizes the session record, commits + pushes all work, and provides a warm handoff. A bare wrap also tears the session down (kills the ai-term buffer + tmux session, restoring geometry); a qualified wrap keeps the buffer, and a shutdown wrap powers the machine off. The teardown variants are set by the trigger phrase (see Teardown mode below) and act only at the very end, in Step 6.
Triggered by Craig saying "wrap it up," "that's a wrap," "let's call it a wrap," or similar.
@@ -24,15 +24,53 @@ The wrap-up is complete when:
2. *File is archived.* =.ai/session-context.org= has been renamed to =.ai/sessions/YYYY-MM-DD-HH-MM-description.org=. The old path no longer exists.
3. *todo.org is clean.* Cleanup script ran. Any auto-fixes are staged for the wrap-up commit. Orphan planning lines surfaced for manual fix if there are any.
4. *Linear board is honest* (skip if project doesn't use Linear). Any Dev-Review ticket whose PR has merged was moved to Done or PM Acceptance per the classification rule.
-5. *Git state is clean.* All changes committed + pushed to all remotes. Working tree clean.
-6. *Valediction delivered.* Brief, warm closing with key accomplishments and reminders.
+5. *Git state is certified clean.* All changes are committed + pushed to all remotes, =git-worktree-gate certify= succeeded at the current HEAD, and the working tree has no staged, unstaged, untracked, submodule, or in-progress-operation state.
+6. *Valediction delivered.* Brief, warm closing with key accomplishments and reminders, ending with =session wrapped.= on its own line as the signoff marker.
The absence of =.ai/session-context.org= is the signal that the last session wrapped up cleanly. Its presence at session start means the previous session was interrupted.
+* Teardown mode (set from the trigger phrase)
+
+The wrap itself — Steps 1 through 5 — is identical in every mode. The trigger phrase only decides what Step 6 does once commit + push and the valediction are done. Resolve the mode from the phrase before starting:
+
+- *Teardown* (the default) — bare "wrap it up", "that's a wrap", "let's call it a wrap". The full wrap, then Step 6 kills the ai-term buffer + the =aiv-<project>= tmux session (which takes =claude= with it) and restores the saved window geometry. This is Craig's typical end-of-day case.
+- *No-teardown* — "wrap it up with summary" or "wrap it up and summarize". The full wrap, but Step 6 leaves the buffer and session intact so the summary stays readable. The explicit qualifier is what opts out of teardown.
+- *Shutdown* — "wrap it up and shutdown". The full wrap, then Step 6 gates on this being the only live ai-term session and powers the machine off. Shutdown supersedes teardown (killing the buffer is moot if the box is going down).
+
+Why teardown waits for Step 6 and runs through a hook, never inline: teardown kills the very tmux session =claude= runs in, so an inline kill would cut the valediction off before it renders. Step 6 instead drops a sentinel after everything else is verified, and the =Stop= hook (=ai-wrap-teardown.sh=) does the actual teardown when this response ends — by which point the valediction has already been delivered.
+
+This depends on three functions in =.emacs.d/modules/ai-term.el= (=cj/ai-term-quit=, =cj/ai-term-live-count=, =cj/ai-term-shutdown-countdown=) and on the =Stop= hook being wired in =settings.json= (=hooks/settings-snippet.json=). If =emacsclient= or the daemon is unreachable, the sentinel is cleared and the session simply stays up — teardown degrades to a no-op, never a wedge.
+
* The Workflow
+** Step 0: Refuse if sentry is live
+
+Before anything else, check whether sentry is running in this project. Sentry holds the working tree on its =sentry/<date>-<host>= branch and commits unattended; wrapping underneath it would archive the session anchor and tear down the buffer while the loop is still firing into it. If sentry's single-runner lock is held, stop and point at the shutdown path:
+
+#+begin_src bash
+proj="$(basename "$(git rev-parse --show-toplevel 2>/dev/null || pwd)")"
+if [ -x .ai/scripts/agent-lock ] && .ai/scripts/agent-lock status "sentry-$proj" | grep -q '^held'; then
+ echo "sentry is active — say 'stop sentry' first"
+ exit 1
+fi
+#+end_src
+
+The stop-sentry operation (defined in =sentry.org=) owns the shutdown: it cancels the loop, disposes of the branch, and walks the approval queue. Wrap-up carries only this one guard; a =stale= lock (a crashed cycle) doesn't block — only a live =held= lock does.
+
** Step 1: Finalize the Summary
+*** Work the Before-Close Queue (before the Summary)
+
+If the session anchor (=.ai/session-context.org=) carries a =* Before-Close Queue= heading with items, work them now, oldest-first, before writing the Summary, so any resulting edits ride this wrap's commit and get described in it. The queue is the "put X on the list" shorthand (see =protocols.org=, Colloquialisms and Expansions): session-scoped work Craig deferred to wrap time.
+
+Per item: do it if it's clear and bounded, or promote it to a =todo.org= task if it turns out to need its own session. Never drop an item silently. Remove each line as it's handled; if one can't be finished, surface it in the valediction (Step 5) and either leave a follow-up task or state why it's dropped.
+
+If there's no =* Before-Close Queue= heading, or it's empty, this step is a silent no-op.
+
+*** Early KB reflection (capture while fresh, before the Summary)
+
+Before distilling the Summary, while the session is still fresh, ask: what did this session learn worth remembering, for yourself or a future agent? Reflect and stage any candidate durable facts — a decision and its why, an environment gotcha, a reference pointer, a transferable lesson. Self-answer silently; this adds no interactive turn (Craig already authorized the wrap). The candidates flow straight into the KB promotion check below, which does the actual writing and the receipt — this is the capture half, that is the commit half, one pipeline, one receipt. Reflecting here rather than reconstructing learnings after the Summary is the point: the early ask is what keeps the receipt from defaulting to "promoted 0" out of fatigue.
+
Read through the =* Session Log= in =.ai/session-context.org=. Populate (or refine) the =* Summary= section:
- *Active Goal* — one or two sentences describing the session's focus
@@ -84,21 +122,21 @@ idseg="${AI_AGENT_ID:+${AI_AGENT_ID}-}"
mv "$sc" ".ai/sessions/${now}-${idseg}DESCRIPTION.org"
#+end_src
-Replace =DESCRIPTION= with your picked slug. (=AI_AGENT_ID= should be filename-safe; the recommended =host.project.runtime.shortid= shape already is.)
+Replace =DESCRIPTION= with your picked slug. (=AI_AGENT_ID= should be filename-safe and unique per run; the recommended =host.project.runtime.<epoch>= shape is both. The epoch on the tail keeps a re-run of the same logical agent from resolving to a prior run's leftover anchor. See protocols.org "Agent-scoped path".)
** Step 3: todo.org cleanup (hygiene + archive completed work)
If the project has a =todo.org= at its root, run the cleanup script before committing. Two passes, both fast and idempotent: a hygiene pass and an archive pass.
-*** Roam inbox sweep (inbox-zero)
+*** Roam inbox sweep (inbox roam mode)
-Before the cleanup scripts, sweep the roam global inbox (=~/org/roam/inbox.org=) for items that belong to this project, so any imported tasks get linted and ride the wrap commit. Delegate to [[file:inbox-zero.org][inbox-zero.org]] for the claimed set.
+Before the cleanup scripts, sweep the roam global inbox (=~/org/roam/inbox.org=) for items that belong to this project, so any imported tasks get linted and ride the wrap commit. Delegate to [[file:inbox.org][inbox.org]] roam mode for the claimed set.
#+begin_src bash
[ -f "$HOME/org/roam/inbox.org" ] && grep -cE '^\*\* ' "$HOME/org/roam/inbox.org" || true
#+end_src
-Skip-fast when nothing matches: if the roam clone isn't on this machine, or no item is prefixed for this project, this is a silent no-op. When claimed items exist, run inbox-zero's Phase B–C (file each into =todo.org=, then remove them from the shared inbox in a separate roam commit). Report the total count and how many appeared related to this project, per inbox-zero's scan-summary rule.
+Skip-fast when nothing matches: if the roam clone isn't on this machine, or no item is prefixed for this project, this is a silent no-op. When claimed items exist, run roam mode's Phase B–D (file each into =todo.org=, then remove them from the shared inbox and let =roam-sync= commit + push the edit). Report the total count and how many appeared related to this project, per roam mode's scan-summary rule.
*** Hygiene pass
@@ -121,6 +159,22 @@ Run the report-only variant first if you want to see what would change without w
emacs --batch -q -l .ai/scripts/todo-cleanup.el --check todo.org
#+end_src
+*** Convert done sub-tasks to dated entries
+
+#+begin_src bash
+[ -f todo.org ] && emacs --batch -q -l .ai/scripts/todo-cleanup.el --convert-subtasks todo.org
+#+end_src
+
+=--convert-subtasks= rewrites every heading at level 3 or deeper whose TODO state is DONE/CANCELLED/FAILED into a dated event-log entry (=<stars> YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <text>=), dropping the keyword, priority cookie, and tags, and removing the now-redundant =CLOSED:= line. This enforces the =todo-format.md= depth rule that a completed *sub-task* (a heading under a parent task) becomes dated history, not a lingering DONE keyword — a shape an interactive org close (=org-log-done= → DONE + CLOSED) never applies and =--archive-done= (level-2 only) never reaches. The timestamp comes from each entry's own =CLOSED= cookie; a date-only close yields =00:00:00=. Heading text is kept verbatim. Idempotent (an already-dated heading has no keyword to match), and a done sub-task with no parseable =CLOSED= is flagged and left alone rather than stamped with a fabricated date.
+
+Run this *before* =--archive-done= so that when a completed level-2 parent is archived, its sub-tasks already carry their dated form. Any rewrites show up in the wrap-up commit's diff for review before push.
+
+Preview without writing:
+
+#+begin_src bash
+emacs --batch -q -l .ai/scripts/todo-cleanup.el --convert-subtasks --check todo.org
+#+end_src
+
*** Archive completed work
#+begin_src bash
@@ -135,6 +189,16 @@ Preview the moves without writing:
emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done --check todo.org
#+end_src
+*** Clear temp/
+
+#+begin_src bash
+[ -d temp ] && find temp -mindepth 1 -delete && echo "temp/ cleared"
+#+end_src
+
+=temp/= holds throwaway artifacts — discarded prototypes, scratch output, intermediate data (see =working-files.md=). It's gitignored in every project, so nothing here rides a commit and nothing is recoverable from git once deleted. Clearing it at wrap is what keeps ephemeral work from silting up across sessions, and it's the counterpart to =working/=, which is tracked and *never* cleared here.
+
+Two guards. Confirm before deleting if =temp/= holds anything a reasonable reader would call in-progress rather than throwaway — misfiled work belongs in =working/=, so move it there instead of deleting it. And skip the step entirely in a project where =temp/= is not gitignored, since that means the project is using the directory for something else.
+
*** Sync child priorities
#+begin_src bash
@@ -170,9 +234,14 @@ else
followups=".ai/lint-followups.org"
fi
[ -f todo.org ] && emacs --batch -q -l .ai/scripts/lint-org.el \
- --followups-file="$followups" todo.org
+ --fix --followups-file="$followups" todo.org
#+end_src
+The =--fix= flag is required for the writes: lint-org's CLI default is
+report-only (a linter reports, it doesn't write), and this wrap-up pass is
+the deliberate exception that applies fixes — its diff rides the wrap-up
+commit for review.
+
=lint-org= runs =org-lint= over =todo.org=, auto-applies four mechanical
categories (=item-number= counters, bare =#+begin_src= → =#+begin_example=,
multi-line planning-info merged onto one line, =**X.**= → =*X.*=), and
@@ -204,7 +273,7 @@ For an interactive walk of the judgments mid-day, run =/lint-org todo.org=.
*** Inbox sanity check (surface unprocessed handoffs)
-If the project has an =inbox/= directory, verify it holds nothing but =.gitkeep=, =lint-followups.org= (the lint-org pipeline file the next morning's daily-prep consumes), and any explicitly-deferred =PROCESSED-*= files before the wrap completes. An inbox that arrived at session start with handoffs from other projects, or that received handoffs mid-session, needs the =process-inbox.org= workflow to run and apply its value-gate dispositions. Wrapping with a dirty inbox silently defers the work to next session and accumulates handoff debt that the sender can't see.
+If the project has an =inbox/= directory, verify it holds nothing but =.gitkeep=, =lint-followups.org= (the lint-org pipeline file the next morning's daily-prep consumes), and =PROCESSED-*= files before the wrap completes. An inbox that arrived at session start with handoffs from other projects, or that received handoffs mid-session, needs =inbox.org= process mode to run and apply its value-gate dispositions. Wrapping with an unprocessed inbox silently defers the work to next session and accumulates handoff debt that the sender can't see.
#+begin_src bash
unprocessed=$(find inbox -maxdepth 1 -type f \
@@ -213,7 +282,7 @@ unprocessed=$(find inbox -maxdepth 1 -type f \
! -name 'PROCESSED-*' \
2>/dev/null | wc -l)
if [ "$unprocessed" -gt 0 ]; then
- echo "wrap-up: inbox/ has $unprocessed unprocessed item(s). Run process-inbox.org before wrapping, or explicitly defer each item with a one-line reason in the valediction."
+ echo "wrap-up blocked: inbox/ has $unprocessed unprocessed item(s). Run inbox.org process mode before wrapping."
find inbox -maxdepth 1 -type f \
! -name '.gitkeep' \
! -name 'lint-followups.org' \
@@ -222,11 +291,37 @@ if [ "$unprocessed" -gt 0 ]; then
fi
#+end_src
-If the count is zero or the project has no =inbox/= directory, the check is a silent no-op. If non-zero, the wrap is incomplete by default. The user resolves each item (process now, defer with reason in the valediction, or delete with rationale) before the validation checklist passes.
+If the count is zero or the project has no =inbox/= directory, the check is a silent no-op. If non-zero, the wrap is blocked. Process each item through its value-gate disposition, or delete it only when that workflow's rationale authorizes deletion, before continuing.
The check exempts =lint-followups.org= explicitly because lint-org runs earlier in the same wrap-up workflow and writes its judgment items to that file in =inbox/= by design. The file is a pipeline artifact for the next morning's =daily-prep=, not a handoff that needs the value gate.
-This integrates with =process-inbox.org=, which stamps =:LAST_INBOX_PROCESS:= in =notes.org='s *Workflow State* section on completion. Wrap-up doesn't double-stamp. It only ensures the inbox carries nothing but the expected pipeline artifacts at session end.
+This integrates with =inbox.org= process mode, which stamps =:LAST_INBOX_PROCESS:= in =notes.org='s *Workflow State* section on completion. Wrap-up doesn't double-stamp. It only ensures the inbox carries nothing but the expected pipeline artifacts at session end.
+
+*** Cross-project router (optional — route filed keepers to their home projects)
+
+Runs directly after the inbox sanity check. The split between the two: the sanity check *gates* the wrap (a dirty inbox blocks until resolved); the router is *optional* (skipping it never blocks anything — the candidates just stay local until a future wrap). Spec: =docs/specs/wrapup-routing-spec.org= (D7/D8/D9).
+
+The candidate set is exactly the local tasks carrying a =:ROUTE_CANDIDATE:= property — keepers that inbox process mode filed this session whose inferred home is another project. Never scan the standing backlog.
+
+#+begin_src bash
+.ai/scripts/route-batch --list
+#+end_src
+
+*Empty set = zero interaction.* =--list= prints nothing when there are no candidates; continue the wrap silently — no prompt, no "0 items" line.
+
+When candidates exist, surface the batch as one line per task — the task heading, the destination project, the delivery mode (=inbox-send= file handoff), and the engine's confidence — then offer exactly two options: *go* (route the whole batch) or *skip* (leave everything local). Derive each confidence label by running the engine on the task's heading + body (=python3 .ai/scripts/route_recommend.py --item "..." --exclude "$(basename "$PWD")"=); label weak matches visibly ("weak — verify the destination") so a low-confidence route gets a human glance before the keystroke.
+
+On *go*:
+
+#+begin_src bash
+.ai/scripts/route-batch --go
+#+end_src
+
+Per candidate, the helper writes the task's subtree (children ride along; =:ROUTE_CANDIDATE:= stripped, headings promoted to top level) to a one-task handoff, delivers it via =inbox-send <destination> --file= (so the =from-<this-project>= provenance is stamped and the destination's inbox process mode dispositions it as a single item), and only after a successful send removes the subtree from the local =todo.org= — a single-file local edit the wrap is already committing. A failed send leaves that task in place and exits non-zero; report it and continue the wrap. Never write the destination's =todo.org= directly; its own inbox processing files the task per its conventions.
+
+On *skip*, leave every candidate in place, marker included — they resurface next wrap.
+
+Mis-routes are recoverable: the receiving project rejects via inbox process mode's reject-from-another-project flow, which returns the item to this project's inbox with the rationale. That reject path is why removing the local source on send is safe.
*** Review-habit health check (surface a slipped daily task-review)
@@ -406,17 +501,17 @@ Behavior:
git status --short
#+end_src
-*Default policy: end every session with an empty =git status=.* The wrap is incomplete while anything remains dirty. There is no "leave it alone" default — every leftover gets an active resolution. The only way for a file to stay dirty across the wrap is the user explicitly saying "defer this one, leave it dirty." Surface each leftover with a concrete recommendation; the user has to actively opt out for the dirt to persist.
+*Hard policy: end every session with an empty =git status=.* The wrap is incomplete while anything remains dirty. There is no deferral exception and no "wrapped with known changes" state: unresolved dirt means the session remains open and wrap-up does not occur.
This inverts the older "intentional carryover" default, which let pre-existing dirty state accumulate across sessions silently. Carryover that lives for days or weeks is almost always one of: a forgotten commit from a prior wrap, a stale change that should be discarded, or genuine in-flight work that needs an explicit stash/branch home. None of those should default to "leave it dirty."
**** Three kinds of leftover
-| Pattern | What it is | Recommended action (apply unless user defers) |
+| Pattern | What it is | Recommended action |
|---+---+---|
| Generated, runtime, or lock files that no human edits — e.g., =.claude/scheduled_tasks.lock=, =.pytest_cache/=, build outputs, IDE state, editor swap files | *Runtime artifact* — created by tooling or the harness, not by the user, and shouldn't be tracked | Add the matching pattern to =.gitignore= (project-level, not =~/.gitignore_global=). For tracked files, =git rm --cached <path>=. Stage =.gitignore= and any =rm --cached= changes in *one* follow-up commit (=chore: gitignore X=), push. Re-run =git status= to confirm clean. |
| Modified or created during the session but not staged into the wrap-up commit | *Forgotten change* — real session work that should have been in the wrap commit but missed it | Stage and create a follow-up commit. Don't =--amend= the wrap-up commit once pushed (diverging history without a clear win). Push the follow-up to all remotes. |
-| Was dirty at session start and still dirty at session end — work this session deliberately didn't touch | *Pre-existing dirt that needs a decision* — could be a missed commit from a prior wrap, stale abandoned work, or real in-flight work without a home | Investigate (show diff + check the originating session). Recommend one of: (a) commit now if the work is complete, (b) stash with a descriptive message if it's genuine WIP, (c) =git checkout -- <path>= / =git clean -f <path>= if stale and unwanted, (d) move to a feature branch if it's longer-running, (e) user explicitly defers and accepts the dirt. Do not silently leave dirty. |
+| Was dirty at session start and still dirty at session end — work this session deliberately didn't touch | *Pre-existing dirt that needs a decision* — could be a missed commit from a prior wrap, stale abandoned work, or real in-flight work without a home | Investigate (show diff + check the originating session). Recommend one of: (a) commit now if the work is complete, (b) stash with a descriptive message if it's genuine WIP, (c) =git checkout -- <path>= / =git clean -f <path>= if stale and unwanted, or (d) move to a feature branch if it's longer-running. Do not silently leave dirty. |
**** Per-file flow
@@ -424,18 +519,40 @@ For each leftover line in =git status --short=:
1. Identify which of the three kinds above it matches.
2. State what the file is (one line) and the recommended action.
-3. Apply the action unless the user explicitly defers.
-4. Re-run =git status --short= after each follow-up commit until empty (or until every remaining line is an explicit user-deferred entry).
+3. Apply the action when it is safe and authorized.
+4. Re-run =git status --short= after each follow-up commit until empty.
The pre-existing-dirt case (third row) is the one this rule most cares about. Treat each pre-existing-dirty file as a question that must get an answer this session, not as "carryover that's fine to inherit." A file that was dirty for a week before this session probably isn't going to get cleaner by waiting another week. Look at the diff, check the originating session's notes, and recommend a real resolution.
-**** When the user defers
+**** When cleanup cannot be completed
+
+Stop the wrap. Do not deliver the valediction, print =session wrapped.=, drop a teardown/shutdown sentinel, or describe the session as complete. Report:
-If the user does say "leave this one dirty for now" after seeing the recommendation, that is fine — log the deferral in the valediction so the next session knows it was an explicit choice, not a miss. Format: "Deferred (per Craig's decision today): =path/to/file= — <one-line reason>". Without that note, the next session can't distinguish "we agreed to defer" from "we forgot again."
+1. Every remaining path and its exact Git state.
+2. What the file is and why the agent cannot safely resolve it alone.
+3. The concrete action or decision Craig needs to provide to make the tree clean.
+
+An explicit decision to keep a file dirty changes the outcome from "wrapping" to "leaving the session interrupted." It never satisfies this workflow.
+
+*** Final clean-tree certificate — hard gate
+
+After all commits are pushed and every leftover appears resolved, run the shared gate:
+
+#+begin_src bash
+gate="$(command -v git-worktree-gate 2>/dev/null || true)"
+[ -n "$gate" ] || gate="$HOME/code/rulesets/claude-templates/bin/git-worktree-gate"
+if [ ! -x "$gate" ]; then
+ echo "wrap blocked: git-worktree-gate is unavailable; install rulesets tooling and retry"
+ exit 1
+fi
+"$gate" certify "$PWD"
+#+end_src
+
+The certificate lives inside the Git directory, so it does not dirty the worktree. It records the exact verified HEAD. A non-zero result is a hard stop governed by "When cleanup cannot be completed" above. Step 5 is unreachable until certification succeeds.
** Step 5: Valediction
-Brief, warm closing. 3-4 sentences max.
+Only after the final clean-tree certificate succeeds, deliver a brief, warm closing. 3-4 sentences max.
Include:
- What was accomplished (specific, not generic)
@@ -444,6 +561,8 @@ Include:
Tone: warm but professional. No emoji unless Craig has explicitly requested. Acknowledge effort when session was long or difficult.
+End on a clear signoff: the *last* line of the valediction is always =session wrapped.= on its own line (lowercase, with the period, nothing after it). It's the unmistakable end-of-session marker, so don't trail it with another sentence. This is the last user-facing output — Step 6's teardown is silent.
+
Example:
#+begin_example
That's a wrap. Today we restructured the entire claude-templates
@@ -456,8 +575,49 @@ from earlier) and archsetup's layout-navigate tests. Both are
ratio-local uncommitted state.
Good session. Talk tomorrow.
+
+session wrapped.
#+end_example
+** Step 6: Session teardown (mode-dependent)
+
+The last action of the wrap, and only after Step 4's commit + push is verified and the Step 5 valediction is composed. The teardown itself happens when this response ends (via the =Stop= hook), so the valediction always renders first. Act by the mode resolved up front:
+
+*** No-teardown mode
+
+Do nothing. The buffer, the =aiv-<project>= tmux session, and =claude= all stay up so the summary stays readable. The wrap is complete.
+
+*** Teardown mode (default)
+
+Confirm commit + push and the final clean-tree certificate succeeded (Exit Criteria 5 — never tear down over unpushed or dirty work), then drop the sentinel:
+
+#+begin_src bash
+touch "/tmp/ai-wrap-teardown-$(basename "$PWD")"
+#+end_src
+
+That is the whole step. Don't run any =tmux kill-session=, =emacsclient=, or buffer kill inline — the =Stop= hook reads the sentinel when this response ends and runs =cj/ai-term-quit=, which kills the =aiv-<project>= session (taking =claude= with it), kills the vterm buffer, and restores geometry. The basename of =$PWD= is the key the hook matches, so the sentinel names the session it tears down.
+
+*The sentinel is session-scoped.* If certification fails, the =Stop= hook blocks and leaves the sentinel armed on purpose, so a wrap blocked by a dirty tree retries on a later stop without re-running this workflow. It does *not* survive the session: =session-start-disarm.sh= clears it at =SessionStart=, because a wrap that never certified is not a pending teardown once its session is gone. Before that hook existed, an uncertified sentinel sat armed indefinitely and fired in whatever session next reached a clean tree — work's 2026-07-27 11:37 wrap killed the 13:20 session mid-work, and archsetup's sat armed on a live terminal for two days. If teardown is still wanted in a new session, run this workflow again.
+
+*** Shutdown mode
+
+Confirm commit + push succeeded, then evaluate the safety gate *before* committing to the shutdown — never power the box off out from under another live session:
+
+#+begin_src bash
+emacsclient -e '(cj/ai-term-live-count)'
+#+end_src
+
+- *Count > 1* — another ai-term session is alive. ABORT the shutdown. List the other live =aiv-*= sessions, drop *no* sentinel, and tell Craig in the valediction that it fell back to a normal wrap (no poweroff, no teardown). This gate is the load-bearing safety of the whole feature.
+- *Count = 1* — this session is the only one. Drop the shutdown sentinel:
+
+ #+begin_src bash
+ touch "/tmp/ai-wrap-shutdown-$(basename "$PWD")"
+ #+end_src
+
+ The =Stop= hook fires =cj/ai-term-shutdown-countdown= when this response ends: it re-checks the gate, runs an abort-able 10→1 countdown in the Emacs echo area (=C-g= cancels), then =sudo shutdown now=. Shutdown supersedes teardown — do *not* also drop the teardown sentinel.
+
+If =emacsclient= isn't resolvable or the daemon is down, the gate can't run — abort the shutdown, fall back to a normal wrap, and say so. Don't power off on an unverifiable gate.
+
* Common Mistakes to Avoid
1. *Skipping Step 1 (Summary)* — the file becomes the record; an empty Summary makes it hard to scan at catch-up
@@ -469,7 +629,8 @@ Good session. Talk tomorrow.
7. *Leaving =.ai/session-context.org= in place* — its presence means "interrupted session", confuses next startup
8. *Long preachy valediction* — brief beats thorough
9. *Leaving runtime/generated files dirty without gitignoring them* — pollutes every future =git status= and erodes trust in "working tree clean" as a signal. Fix =.gitignore= during the wrap, not later.
-10. *Treating "was dirty at session start, still dirty now" as fine by default* — that's how a forgotten commit from two sessions ago turns into "carryover" for two weeks. Every pre-existing dirty file needs an active resolution recommendation this session. Deferral is allowed only with an explicit user choice, logged in the valediction.
+10. *Treating "was dirty at session start, still dirty now" as fine* — that's how a forgotten commit from two sessions ago turns into "carryover" for two weeks. Every pre-existing dirty file must be resolved or the wrap remains blocked.
+11. *Calling a blocked cleanup a wrap* — if the strict gate fails, report the paths and needed decisions; do not valedict, certify completion, or tear down.
* Validation Checklist
@@ -479,19 +640,23 @@ Before considering wrap-up complete:
- [ ] The Summary ends with the =KB: promoted N / consulted yes-no= line (promotion check ran)
- [ ] File renamed to =.ai/sessions/YYYY-MM-DD-HH-MM-description.org=
- [ ] =.ai/session-context.org= no longer exists
-- [ ] =todo-cleanup.el= ran — hygiene pass + =--archive-done= + =--sync-child-priority= (if =todo.org= exists at project root)
+- [ ] =todo-cleanup.el= ran — hygiene pass + =--convert-subtasks= + =--archive-done= + =--sync-child-priority= (if =todo.org= exists at project root)
- [ ] =lint-org.el= ran on =todo.org= — mechanical fixes applied, judgments appended to follow-ups file (if =todo.org= exists)
- [ ] Any orphan-planning-line warnings reviewed (fix or accept)
-- [ ] Inbox carries nothing but expected pipeline artifacts (=.gitkeep=, =lint-followups.org=, =PROCESSED-*= prefixes), OR each remaining handoff has an explicit deferral logged in the valediction
+- [ ] Inbox carries nothing but expected committed or ignored pipeline artifacts (=.gitkeep=, =lint-followups.org=, =PROCESSED-*= prefixes); any untracked inbox delivery was processed before wrap
- [ ] Linear Dev-Review sweep ran; any merged-PR tickets moved to Done or PM Acceptance (skip if project doesn't use Linear)
- [ ] Template-sync churn committed as its own =chore: sync .ai tooling from templates= (consuming projects only; skipped in rulesets), or surfaced if a synced path didn't match canonical
-- [ ] After wrap-up commit + push, =git status --short= is empty OR every remaining line has an explicit user-deferred decision logged in the valediction
+- [ ] After wrap-up commit + push, =git-worktree-gate certify "$PWD"= succeeded at the current HEAD
- [ ] Each leftover was investigated and the user saw a concrete resolution recommendation
- [ ] Runtime artifacts added to =.gitignore=, follow-up commit pushed, =git status= re-verified
- [ ] Forgotten changes committed in a follow-up and pushed
-- [ ] Pre-existing dirty files resolved (committed / stashed / discarded / moved to a feature branch) or explicitly deferred with a one-line reason in the valediction
+- [ ] Pre-existing dirty files resolved (committed / stashed / discarded / moved to a feature branch); otherwise wrap stopped with an actionable blocker report
- [ ] Current branch pushed to ALL remotes (verified with =git remote -v=)
- [ ] All other local branches with a tracking upstream pushed to their remote
- [ ] Any untracked-upstream branches surfaced for manual =git push -u=
+- [ ] Step 6 teardown matches the trigger phrase: no-teardown leaves the buffer; teardown drops only =/tmp/ai-wrap-teardown-<project>=; shutdown gates on =cj/ai-term-live-count= = 1 and drops only =/tmp/ai-wrap-shutdown-<project>=
+- [ ] No teardown/shutdown sentinel was dropped before commit + push was verified
+- [ ] The teardown hook can re-verify the clean-tree certificate before consuming a sentinel
+- [ ] Shutdown aborted (fell back to normal wrap, logged in the valediction) when another =aiv-*= session was live or the gate couldn't run
- [ ] Commit message follows format (no =session:=, no Claude attribution)
- [ ] Valediction delivered (brief, specific, warm)
diff --git a/.claude/commands/lint-org.md b/.claude/commands/lint-org.md
index 953629c..d9719fa 100644
--- a/.claude/commands/lint-org.md
+++ b/.claude/commands/lint-org.md
@@ -50,6 +50,7 @@ Out of scope (refuse, don't try to lint):
| `invalid-fuzzy-link` | (1) Repair to a `[[*Heading]]` ref if a similar heading exists. (2) Drop to `=verbatim label=` text. (3) Skip. |
| `misplaced-heading` *(verbatim-asterisk case)* | (1) Strip asterisks and rephrase to preserve semantics. (2) Convert surrounding markup to `~code~` style. (3) Skip. |
| `suspicious-language-in-src-block` | (1) Emit an Emacs init one-liner that registers the language. (2) Change the block label to `text` or `example`. (3) Skip. |
+| `level-2-dated-header` *(custom check, not org-lint)* | A `** <YYYY-MM-DD> …` heading is a completion defect per `todo-format.md` (no keyword, so `--archive-done` can't archive it). (1) Convert to `DONE`/`CANCELLED` + `CLOSED:`, keeping the heading text — the usual fix. (2) Demote to `***` if it's really a mis-leveled sub-entry. (3) Skip (a dated-log-format org file where `**` dates are intentional). |
| anything else | Surface the raw `org-lint` message and ask the user how to proceed. |
## Phase A — Run the script
@@ -59,10 +60,10 @@ Out of scope (refuse, don't try to lint):
3. Invoke the script:
```bash
- emacs --batch -q -l .ai/scripts/lint-org.el FILE
+ emacs --batch -q -l .ai/scripts/lint-org.el --fix FILE
```
- The script applies every mechanical fix, then emits structured stdout. First line is a summary; each subsequent line is a plist describing one issue:
+ `--fix` is required for the writes — the script's default invocation is report-only (a linter reports, it doesn't write). With the flag, the script applies every mechanical fix, then emits structured stdout. First line is a summary; each subsequent line is a plist describing one issue:
```
;; lint-org: file=todo.org mechanical=4 judgment=11
diff --git a/.claude/commands/refactor.md b/.claude/commands/refactor.md
index 08cbdab..97ce366 100644
--- a/.claude/commands/refactor.md
+++ b/.claude/commands/refactor.md
@@ -1,5 +1,5 @@
---
-description: Scan code for refactoring opportunities or perform a targeted refactor. Six modes — `full` (default; complexity + duplication + dead-code scans), `quick` (high-severity findings only), `complexity` (length / nesting / cyclomatic / parameter count / boolean ops with severity bands and techniques like guard clauses, extract method/predicate, parameter object, decompose conditional), `duplication` (clones / logic / constants / patterns / error-handling with extract-function / parameterize / template-method strategies), `dead-code` (imports / exports / branches / feature flags / deps with high/medium/low confidence labels), `rename old new` (codebase-wide symbol rename with reference search, preview gate, atomic commit, post-apply verification). Findings render as `[SEVERITY] Category — File / Metric / Issue / Suggestion` blocks plus a summary table and quick-wins. Structure-only — no feature work mixed in, no auto-apply without confirmation, characterization tests first when coverage is missing, small focused commits. Use for cleanup or wide renames. Do NOT use for behavior changes (`fix:` or `feat:`, not refactor), green-field design (use `/arch-design`), or single-symbol single-file renames (just edit). Companion to `/add-tests` for the characterization-test prereq.
+description: Scan code for refactoring opportunities or perform a targeted refactor. Seven modes — `full` (default; complexity + duplication + dead-code + simplification scans), `quick` (high-severity findings only), `complexity` (length / nesting / cyclomatic / parameter count / boolean ops with severity bands and techniques like guard clauses, extract method/predicate, parameter object, decompose conditional), `duplication` (clones / logic / constants / patterns / error-handling with extract-function / parameterize / template-method strategies), `dead-code` (imports / exports / branches / feature flags / deps with high/medium/low confidence labels), `simplification` (over-defensive guards / needless indirection / convoluted logic / redundant state / legibility rewrites — behavior-preserving clarity and size reduction, distinct from the metric-driven complexity scan), `rename old new` (codebase-wide symbol rename with reference search, preview gate, atomic commit, post-apply verification). Findings render as `[SEVERITY] Category — File / Metric / Issue / Suggestion` blocks plus a summary table and quick-wins. Structure-only — no feature work mixed in, no auto-apply without confirmation, characterization tests first when coverage is missing, small focused commits. Use for cleanup or wide renames. Do NOT use for behavior changes (`fix:` or `feat:`, not refactor), green-field design (use `/arch-design`), or single-symbol single-file renames (just edit). Companion to `/add-tests` for the characterization-test prereq.
argument-hint: "[scope: full|quick|complexity|duplication|dead-code|rename old new]"
---
@@ -9,11 +9,12 @@ Parse `$ARGUMENTS` to determine the operation:
| Argument | Description |
|----------|-------------|
-| `full` (default) | Run all scans: complexity + duplication + dead code |
+| `full` (default) | Run all scans: complexity + duplication + dead code + simplification |
| `quick` | High-severity issues only (critical/high across all scans) |
| `complexity` | Analyze code complexity: nesting, length, parameters, boolean expressions |
| `duplication` | Detect duplicated logic, clone blocks, repeated patterns |
| `dead-code` | Find unused imports, exports, unreachable code, dead feature flags |
+| `simplification` | Find over-complicated code: redundant guards, needless indirection, convoluted logic, redundant state, legibility rewrites |
| `rename old new` | Codebase-wide symbol rename with verification |
If a file or directory path is included in the arguments, scope the scan to that path. Otherwise scan the project source directories (exclude vendored code, node_modules, build output, test fixtures).
@@ -182,6 +183,40 @@ Group findings by category with confidence levels:
---
+## Mode: Simplification
+
+Scan for code that's more complicated than it needs to be — a plainer, smaller, more direct expression of the same behavior. Behavior-preserving like every other mode; this targets clarity and size, not metrics (complexity mode) or repetition (duplication mode).
+
+### What to Check
+
+- **Over-defensive / redundant guards** — existence checks and fallbacks for conditions that can't occur given the actual call sites.
+- **Needless indirection / unearned abstraction** — single-use closures, helpers, or wrappers; a single-use local that could inline (or a repeated expression that should be a local); a parameter always passed the same value; an option or branch never exercised.
+- **Convoluted logic expressible more directly** — a manual loop that's a map/filter/comprehension; a verbose if-chain that's a lookup table; an if/else assigning a value that's a ternary; boolean expressions that simplify; redundant intermediate computations.
+- **Redundant state** — a cache rebuilt every call anyway; two variables tracking the same thing; write-only "dead storage" (assigned, never read); dead flags.
+- **Legibility rewrites** — named locals to expose a decision matrix obscured by indexing or chaining. No structural change, large readability win.
+
+For identical-twin branches and plain deletion of unreachable code, see Mode: Dead Code; for repeated literals → named constant, see Mode: Duplication. Those modes already own that detection.
+
+### Detection Heuristics
+
+- Search for guard clauses and fallbacks (null checks, default branches), then check every call site to see whether the guarded condition can actually occur
+- Search for helpers, closures, and locals referenced exactly once
+- Search for manual index/accumulator loops that build or filter a collection
+- Search for variables assigned but never read, and for values recomputed on every call that never change
+
+### Rules
+
+- Behavior-preserving only.
+- Verify against **all** call sites before deleting a guard, dropping an option, or removing "never read" state — "can't occur" and "never read" are only true relative to every caller. Never remove a guard or fallback that's genuinely reachable.
+- Run the test suite after each change.
+- Present findings before applying.
+
+### Boundary with /simplify
+
+`/refactor simplification` sweeps the whole tree (or a scoped path), presents findings, and applies on confirmation. The built-in `/simplify` works on the current diff and applies its lenses directly. Reach for `/simplify` on a change in flight; reach for `/refactor simplification` to sweep existing code.
+
+---
+
## Mode: Rename
Perform a codebase-wide symbol rename.
diff --git a/.claude/commands/respond-to-cj-comments.md b/.claude/commands/respond-to-cj-comments.md
index 7ee3909..2f16099 100644
--- a/.claude/commands/respond-to-cj-comments.md
+++ b/.claude/commands/respond-to-cj-comments.md
@@ -1,5 +1,5 @@
---
-description: Scan an org file for cj comments — Craig's annotations wrapped in `#+begin_src cj: ... #+end_src` source blocks — and process each via subagent-delegated accuracy. Each item is classified instruction / question / both, then dispatched to an instruction subagent (proposes a file:line patch) or a question subagent (researches with explicit scope, reports answer + evidence + confidence). Main thread reviews proposals before editing; subagents don't write to the source file. Org-mode TODO parents flip to DOING; new content lands under timestamped subheadings one level deeper; on completion, top- and second-level tasks advance to `DONE` while deeper tasks get their heading rewritten to a dated action description (no DONE keyword), becoming an in-place event log. VERIFY tasks at any depth flip to dated log entries with body replaced by the answer or action taken. Public-facing writing (commits, PRs, Slack, email, public docs) gets `/voice personal`; private writing skips it. Summary lists handled instructions, answered questions with evidence + confidence, follow-ups, unresolved items, and an explicit clean / N-remain verdict. Anything needing Craig's input becomes a `VERIFY` task in `todo.org` (top-level or first-level child of a parent task — never deeper) rather than a separate summary file. File/URL references render as clickable org-mode links. Use when an org file accumulates cj comments. Do NOT use for general code review (`/review-code`), new work without cj comments, or trivial items.
+description: Scan an org file for cj comments — Craig's annotations wrapped in `#+begin_src cj: ... #+end_src` source blocks — and process each via subagent-delegated accuracy. Each item is classified instruction / question / both, then dispatched to an instruction subagent (proposes a file:line patch) or a question subagent (researches with explicit scope, reports answer + evidence + confidence). Main thread reviews proposals before editing; subagents don't write to the source file. Org-mode TODO parents flip to DOING; new content lands under timestamped subheadings one level deeper; on completion, top- and second-level tasks advance to `DONE` while deeper tasks get their heading rewritten to a dated action description (no DONE keyword), becoming an in-place event log. VERIFY tasks at `***` and deeper flip to dated log entries with body replaced by the answer or action taken; a top-level (`**`) VERIFY instead closes as `DONE`/`CANCELLED` + `CLOSED:` like any top-level task, the answer in its body. Public-facing writing (commits, PRs, Slack, email, public docs) gets `/voice personal`; private writing skips it. Summary lists handled instructions, answered questions with evidence + confidence, follow-ups, unresolved items, and an explicit clean / N-remain verdict. Anything needing Craig's input becomes a `VERIFY` task in `todo.org` (top-level or first-level child of a parent task — never deeper) rather than a separate summary file. File/URL references render as clickable org-mode links. Use when an org file accumulates cj comments. Do NOT use for general code review (`/review-code`), new work without cj comments, or trivial items.
---
# /respond-to-cj-comments — Process cj Comments in an Org File
@@ -143,9 +143,9 @@ For **instructions**:
- **Regular `TODO`/`DOING` at `*` or `**`** — advance to `DONE` + `CLOSED:` line; the keyword and original heading stay visible in the agenda.
- **Regular `TODO`/`DOING` at `***` and deeper** — rewrite the heading to `<depth> YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <past-tense description>`; drop the keyword/priority/tags.
- - **`VERIFY` at any depth** — dated-heading rewrite *and* a body replacement: replace the body with either the information Craig provided (when the VERIFY was a question) or a description of the action taken (when it was an instruction / pending-decision marker). VERIFYs at `**` follow this rule even though regular `**` DONE tasks stay task-shaped — a resolved VERIFY is an answered question, not a finished task.
+ - **`VERIFY` — depth decides the heading.** At `***` and deeper, a dated-heading rewrite. At `**`, a terminal keyword (`DONE`/`CANCELLED` + `CLOSED:`) like any top-level task — never a dated `**` header. Either way, replace the body with the information Craig provided (when the VERIFY was a question) or a description of the action taken (when it was an instruction / pending-decision marker).
- **VERIFY-answer pattern.** When a cj annotation's `parent_heading_chain` ends with a `VERIFY ...` heading (i.e., the cj sits directly inside a VERIFY task), the cj is Craig's answer to the question that VERIFY held open. The cj content is the source for the dated-rewrite body. Two shapes:
+ **VERIFY-answer pattern.** When a cj annotation's `parent_heading_chain` ends with a `VERIFY ...` heading (i.e., the cj sits directly inside a VERIFY task), the cj is Craig's answer to the question that VERIFY held open. The cj content is the source for the resolved body. Two shapes:
- *Direct answer.* The cj body IS the answer (a value, decision, link, paste from elsewhere). Lift the cj body verbatim into the new dated body; trim filler ("okay," "approved," "yes,") that isn't load-bearing.
- *Indirect answer.* The cj points at where the answer lives ("Kostya gave this in Slack — pull it from DM channel X," "see the attached doc"). Execute the instruction first (per step 3 — subagent if research is needed), then the resolved info becomes the dated body.
@@ -153,7 +153,7 @@ For **instructions**:
Both shapes land at the same end state:
1. Generate the timestamp with `date "+%Y-%m-%d %a @ %H:%M:%S %z"`.
- 2. Rewrite the VERIFY heading to its dated form (depth-preserving) with a short summary of what got answered.
+ 2. Rewrite the VERIFY heading by depth: at `***` and deeper, its dated form with a short summary of what got answered; at `**`, a `DONE`/`CANCELLED` keyword + `CLOSED:` line with the heading text kept.
3. Replace the body with the resolved info (the cj body for direct, the executed result for indirect).
4. Delete the cj annotation — it's now folded into the body. Don't keep both.
diff --git a/.claude/commands/start-work.md b/.claude/commands/start-work.md
index d146622..726cef4 100644
--- a/.claude/commands/start-work.md
+++ b/.claude/commands/start-work.md
@@ -1,12 +1,12 @@
---
-description: Pick up a task (Linear ticket, GitHub issue, todo.org task, or a described scope) and take it through Pre-work, Claim, Justify, Approach, Implement, Verify, and Hand-off. Three user-approval gates separate the phases. Pre-work covers eligibility, a fetch-and-reconcile against the base branch, and a source-code check that the problem still exists in the tree. The Justify gate weighs benefits, costs, impact, urgency, effort, alternatives, and ticket quality. The Approach gate covers root cause, risk, refactor prerequisites, test strategy (unit, integration, e2e, pairwise, characterization), migration and backwards-compat, feature flags, commit decomposition, and branch name. Implementation uses TDD (red, green, edge cases); a refactor audit then walks every touched file against a language-agnostic checklist, fixing each finding here or filing it as a ticket, never dropping one. A verify phase exercises the feature end-to-end locally (Playwright against localhost for web, scripted manual test otherwise) before the final gate hands off to the Review-and-Publish flow in commits.md. Use when starting work on a specific task where both "should we" and "how exactly" are worth deliberating. Do NOT use for open-ended bug investigation without a clear target (use debug first), for architectural paradigm exploration (use arch-design), for architectural decision recording (use arch-decide), when the task is trivial and obvious (just do it), or when requirements are still being shaped (use brainstorm).
+description: Pick up a task (Linear ticket, GitHub issue, todo.org task, or a described scope) and take it through Pre-work, Claim, Justify, Approach, Implement, Verify, and Hand-off. Three user-approval gates separate the phases. Pre-work covers eligibility, a fetch-and-reconcile against the base branch, a green-baseline suite run, and a source-code check that the problem still exists in the tree. The Justify gate weighs benefits, costs, impact, urgency, effort, alternatives, and ticket quality. The Approach gate covers root cause, risk, refactor prerequisites, test strategy (unit, integration, e2e, pairwise, characterization), migration and backwards-compat, feature flags, commit decomposition, and branch name. Implementation uses TDD (red, green, edge cases); a refactor audit then walks every touched file against a language-agnostic checklist, fixing each finding here or filing it as a ticket, never dropping one. A verify phase exercises the feature end-to-end locally (Playwright against localhost for web, scripted manual test otherwise) before the final gate hands off to the Review-and-Publish flow in commits.md. Use when starting work on a specific task where both "should we" and "how exactly" are worth deliberating. Do NOT use for open-ended bug investigation without a clear target (use debug first), for architectural paradigm exploration (use arch-design), for architectural decision recording (use arch-decide), when the task is trivial and obvious (just do it), or when requirements are still being shaped (use brainstorm).
---
# /start-work: pick up a task, justify it, plan it, build it
Three review gates separate the phases. The user can redirect or kill the work at each one.
-0. **Pre-work.** Eligibility check, fetch-and-reconcile against the base branch, source-code check that the problem still exists.
+0. **Pre-work.** Eligibility check, fetch-and-reconcile against the base branch, green-baseline suite run, source-code check that the problem still exists.
1. **Claim.** Mark in-progress, assign, label, verify project.
2. **Justify (gate 1).** Benefits, costs, impact, urgency, effort, alternatives, ticket quality. Stop for approval.
3. **Approach (gate 2).** Root cause, risk, tests, migration, flag, commit decomposition. Stop for approval.
@@ -53,7 +53,7 @@ If the reference is ambiguous, ask the user to clarify before proceeding.
## Phase 0: pre-work
-Three checks before claiming the task. All run before any state change — no assignee added, no label written, no status moved. If any of them disqualify the task, the rollback is free.
+Four checks before claiming the task. All run before any state change — no assignee added, no label written, no status moved. If any of them disqualify the task, the rollback is free.
### 0.1 Eligibility
@@ -84,7 +84,18 @@ The branch this task will be cut from must reflect the remote — otherwise the
4. If the current branch is *not* the base branch (e.g. left over from a prior task), surface and ask whether to switch before continuing. Don't auto-switch — the user may want to finish or stash WIP first.
-### 0.3 Existence check (validate the problem is real)
+### 0.3 Green baseline (confirm the tree starts known-good)
+
+Run the project's test suite now, against the reconciled base, so the baseline you build on is actually green (see the Green Baseline section in `verification.md`). This runs after 0.2 — baselining a stale tree is pointless.
+
+- **Green** — proceed to 0.4.
+- **Red** — fix the failure first, or, when it's out of scope or needs a decision, file a tracked task with the diagnosis and carry its name forward as the only tolerated failure for this work. Surface the baseline result either way so "we started from green" is on the record.
+- **No suite** — nothing to baseline. Note it and proceed (your personal/doc projects hit this).
+- **Suite can't run** (no network, missing dep, sandbox limit) — that's the "When You Cannot Verify" case in `verification.md`, not a blocker. Record what you couldn't run, name the risk, and proceed.
+
+This baseline is the green starting point; the intentional red test you write in Phase 4 (TDD) is expected and distinct from a baseline failure.
+
+### 0.4 Existence check (validate the problem is real)
The ticket may describe a problem the code no longer has — fixed independently of the ticket, made obsolete by another change, or never present in the first place. Read the source to confirm the problem exists in the tree as the ticket describes, before justifying or planning the fix.
@@ -160,7 +171,7 @@ Then produce a justification that covers all of these, concisely:
7. **Effort estimate.** S (under 1 hour), M (1 hour to 1 day), L (over 1 day). Rough is fine.
8. **Alternatives considered.** Is there a cheaper way? Can we defer? Can we address the root cause via a different path?
9. **Reasons not to do this.** A forced devil's-advocate verdict on whether the work should happen at all — distinct from Downsides (what the change costs) and Alternatives (cheaper paths). Surface the top three objections if real ones exist; when none rise to a genuine objection, say so in one line rather than manufacturing three (e.g. "Nothing material argues against this. No reason to defer or drop it."). Building the case against the work is cheapest at this gate, which is its purpose.
-10. **Ticket quality check.** Is scope clear, are acceptance criteria concrete, are reproduction steps present for bugs? If **not clear**, stop and ask the user to choose one of:
+10. **Ticket quality check.** Is scope clear, are acceptance criteria concrete, are reproduction steps present for bugs? An **open-ended goal** phrased as an absence ("find bugs until none remain," "refactor until nothing worthwhile is left," "clean it up") is the specific case where acceptance criteria aren't concrete: it has no definition of done, so there's no writable acceptance test and no clean commit at the end. Don't start it as-is. Give it measurable criteria first — bound the surface, a characterization net, dispositioned findings, an objective floor (see `todo-format.md`'s "Making an open-ended task measurable") — which is also what makes it `:solo:`-eligible. If **not clear**, stop and ask the user to choose one of:
- (a) Bounce to `/brainstorm` to refine the ticket first.
- (b) Ping the ticket author for clarification.
- (c) Supply the missing info themselves right now, if it is easy for them to do so.
@@ -337,7 +348,7 @@ Follow `commits.md` exactly. Summary of the flow:
## Anti-patterns
- **Skipping the pre-flight reconcile.** Cutting a new branch from a stale base means the whole task happens on top of yesterday's main. Conflicts surface at PR time instead of at the start; rebases later are noisier than a fetch up front.
-- **Taking the ticket's word that the problem still exists.** Tickets age. Read the source. A `git log --grep` for a fix commit is a hint, not a check — fixes ship under all kinds of commit-message wording, and the buggy behavior may be gone for reasons that never landed in a commit titled "fix." Five minutes of source-read at Phase 0.3 saves an entire Justify-and-Approach cycle on a phantom problem.
+- **Taking the ticket's word that the problem still exists.** Tickets age. Read the source. A `git log --grep` for a fix commit is a hint, not a check — fixes ship under all kinds of commit-message wording, and the buggy behavior may be gone for reasons that never landed in a commit titled "fix." Five minutes of source-read at Phase 0.4 saves an entire Justify-and-Approach cycle on a phantom problem.
- **Skipping the Justify gate.** "This is obviously worth doing" is exactly what the gate exists to verify. If the answer really is obvious, the gate takes thirty seconds.
- **Skipping the Approach gate.** Implementation without a plan is how scope creep happens. It is also how the user loses the chance to redirect.
- **Marking a personal todo task DOING before Phase 2 approval.** Personal claims carry no teammate signal, so they wait until the gate clears — a killed task then needs no rollback. Team-tracker claims (Linear, GitHub) are the exception: they happen in Phase 1 to flag intent, but only after the prior state is recorded so the gate can restore it cleanly.
diff --git a/.claude/settings.json b/.claude/settings.json
index f648e44..5e1cfcc 100644
--- a/.claude/settings.json
+++ b/.claude/settings.json
@@ -11,11 +11,20 @@
"hooks": {
"PreToolUse": [
{
+ "matcher": "Edit|Write",
+ "hooks": [
+ {
+ "type": "command",
+ "command": "~/.claude/hooks/rulesets-write-boundary.py"
+ }
+ ]
+ },
+ {
"matcher": "AskUserQuestion",
"hooks": [
{
"type": "command",
- "command": "echo '{\"hookSpecificOutput\":{\"hookEventName\":\"PreToolUse\",\"permissionDecision\":\"deny\",\"permissionDecisionReason\":\"Popup choice menus are disabled per interaction.md (No Popup Menus for Choices) — present options inline in chat as a numbered list and ask the user to reply with a number.\"}}'"
+ "command": "echo '{\"hookSpecificOutput\":{\"hookEventName\":\"PreToolUse\",\"permissionDecision\":\"deny\",\"permissionDecisionReason\":\"Popup choice menus are disabled per interaction.md (No Popup Menus for Choices) \u2014 present options inline in chat as a numbered list and ask the user to reply with a number.\"}}'"
}
]
}
@@ -37,6 +46,10 @@
{
"type": "command",
"command": "~/.claude/hooks/session-title.sh"
+ },
+ {
+ "type": "command",
+ "command": "~/.claude/hooks/session-start-disarm.sh"
}
]
},
@@ -49,6 +62,20 @@
}
]
}
+ ],
+ "Stop": [
+ {
+ "hooks": [
+ {
+ "type": "command",
+ "command": "~/.claude/hooks/inbox-boundary-check.sh"
+ },
+ {
+ "type": "command",
+ "command": "~/.claude/hooks/ai-wrap-teardown.sh"
+ }
+ ]
+ }
]
},
"statusLine": {
diff --git a/.codex/hooks.json b/.codex/hooks.json
new file mode 100644
index 0000000..e14bed5
--- /dev/null
+++ b/.codex/hooks.json
@@ -0,0 +1,30 @@
+{
+ "description": "Rulesets lifecycle enforcement shared with Claude sessions.",
+ "hooks": {
+ "PreToolUse": [
+ {
+ "matcher": "Edit|Write",
+ "hooks": [
+ {
+ "type": "command",
+ "command": "~/.claude/hooks/rulesets-write-boundary.py",
+ "timeout": 30,
+ "statusMessage": "Checking rulesets write boundary"
+ }
+ ]
+ }
+ ],
+ "Stop": [
+ {
+ "hooks": [
+ {
+ "type": "command",
+ "command": "~/.claude/hooks/ai-wrap-teardown.sh",
+ "timeout": 30,
+ "statusMessage": "Verifying clean wrap state"
+ }
+ ]
+ }
+ ]
+ }
+}
diff --git a/.gitignore b/.gitignore
index 94b983f..dd66346 100644
--- a/.gitignore
+++ b/.gitignore
@@ -20,3 +20,11 @@
# (only the .gpg counterpart is safe to commit)
mcp/secrets.env
mcp/gcp-oauth.keys.json
+
+# Live session anchor — ephemeral, archived under a different name into
+# .ai/sessions/ at wrap. Tracking .ai/ (this repo does; most projects gitignore
+# it) meant the anchor showed as untracked for the whole session, which made
+# git-worktree-gate report rulesets sync-blocked and every other project skip
+# its rulesets pull until wrap.
+.ai/session-context.org
+.ai/session-context.d/
diff --git a/AGENTS.md b/AGENTS.md
new file mode 120000
index 0000000..1e43f54
--- /dev/null
+++ b/AGENTS.md
@@ -0,0 +1 @@
+claude-templates/AGENTS.md \ No newline at end of file
diff --git a/Makefile b/Makefile
index 450dc1a..fdbb70b 100644
--- a/Makefile
+++ b/Makefile
@@ -5,6 +5,8 @@ SKILLS_DIR := $(HOME)/.claude/skills
RULES_DIR := $(HOME)/.claude/rules
HOOKS_DIR := $(HOME)/.claude/hooks
CLAUDE_DIR := $(HOME)/.claude
+CODEX_DIR := $(HOME)/.codex
+CODEX_HOOKS := $(CURDIR)/.codex/hooks.json
LOCAL_BIN := $(HOME)/.local/bin
AI_LAUNCHER := $(CURDIR)/claude-templates/bin/ai
SKILLS := $(patsubst %/SKILL.md,%,$(wildcard */SKILL.md))
@@ -15,7 +17,7 @@ HOOKS := $(wildcard hooks/*.sh hooks/*.py)
OPTIN_HOOKS := hooks/destructive-bash-confirm.py
DEFAULT_HOOKS := $(filter-out $(OPTIN_HOOKS),$(HOOKS))
CLAUDE_CONFIG := $(wildcard .claude/*.json) $(wildcard .claude/.*.json) $(wildcard .claude/*.sh)
-LANGUAGES := $(notdir $(wildcard languages/*))
+LANGUAGES := $(notdir $(patsubst %/,%,$(wildcard languages/*/)))
TEAMS := $(notdir $(wildcard teams/*))
PDFTOOLS_VENV ?= $(HOME)/.local/venvs/pdftools
@@ -220,6 +222,30 @@ install: ## Symlink skills, rules, config, hooks, and bin scripts into place
fi \
fi
@echo ""
+ @echo "Agent entry (codex):"
+ @mkdir -p "$(CODEX_DIR)"
+ @if [ -L "$(CODEX_DIR)/AGENTS.md" ]; then \
+ echo " skip AGENTS.md (already linked)"; \
+ elif [ -e "$(CODEX_DIR)/AGENTS.md" ]; then \
+ echo " WARN AGENTS.md exists and is not a symlink — skipping"; \
+ else \
+ ln -s "$(CURDIR)/claude-templates/AGENTS.md" "$(CODEX_DIR)/AGENTS.md"; \
+ echo " link AGENTS.md → $(CODEX_DIR)/AGENTS.md"; \
+ fi
+ @if [ -L "$(CODEX_DIR)/hooks.json" ]; then \
+ target=$$(readlink "$(CODEX_DIR)/hooks.json"); \
+ if [ "$$target" = "$(CODEX_HOOKS)" ]; then \
+ echo " skip hooks.json (already linked)"; \
+ else \
+ echo " WARN hooks.json links elsewhere ($$target) — skipping"; \
+ fi; \
+ elif [ -e "$(CODEX_DIR)/hooks.json" ]; then \
+ echo " WARN hooks.json exists and is not a symlink — skipping"; \
+ else \
+ ln -s "$(CODEX_HOOKS)" "$(CODEX_DIR)/hooks.json"; \
+ echo " link hooks.json → $(CODEX_DIR)/hooks.json"; \
+ fi
+ @echo ""
@echo "Hooks (default):"
@for hook in $(DEFAULT_HOOKS); do \
name=$$(basename $$hook); \
@@ -255,6 +281,15 @@ install: ## Symlink skills, rules, config, hooks, and bin scripts into place
echo " link $$name → $(LOCAL_BIN)/$$name"; \
fi \
done
+ @for link in "$(LOCAL_BIN)"/*; do \
+ [ -L "$$link" ] || continue; \
+ target=$$(readlink "$$link"); \
+ case "$$target" in "$(CURDIR)/claude-templates/bin/"*) ;; *) continue ;; esac; \
+ if [ ! -e "$$target" ]; then \
+ rm "$$link"; \
+ echo " prune $$(basename "$$link") (dangling → $$target)"; \
+ fi \
+ done
@echo ""
@echo "done"
@@ -296,6 +331,13 @@ uninstall: ## Remove global symlinks from ~/.claude/
else \
echo " skip commands (not a symlink)"; \
fi
+ @if [ -L "$(CODEX_DIR)/hooks.json" ] \
+ && [ "$$(readlink "$(CODEX_DIR)/hooks.json")" = "$(CODEX_HOOKS)" ]; then \
+ rm "$(CODEX_DIR)/hooks.json"; \
+ echo " rm codex hooks.json"; \
+ else \
+ echo " skip codex hooks.json (not our symlink)"; \
+ fi
@echo ""
@echo "ai launcher:"
@if [ -L "$(LOCAL_BIN)/ai" ]; then \
@@ -509,7 +551,7 @@ test: ## Run all test suites (pytest + ERT + bats)
echo "ert: $$(basename "$$f")"; \
emacs --batch -q -l ert -l "$$f" -f ert-run-tests-batch-and-exit; \
done
- @set -e; for f in scripts/tests/*.bats .ai/scripts/tests/*.bats; do \
+ @set -e; for f in scripts/tests/*.bats .ai/scripts/tests/*.bats languages/*/tests/*.bats; do \
[ -e "$$f" ] || continue; \
echo "bats: $$(basename "$$f")"; \
bats "$$f"; \
diff --git a/README.org b/README.org
index 067a2a1..f8e5d8f 100644
--- a/README.org
+++ b/README.org
@@ -39,7 +39,12 @@ make list-languages # show available bundles
#+end_src
What gets installed:
-- =.claude/rules/*.md= — project-scoped rules (language-specific + verification)
+- =.claude/rules/*.md= — the language's own rules only. The generic rules in
+ =claude-rules/= are *not* copied per project: =make install= links them once
+ into =~/.claude/rules/=, where they load in every session on the machine.
+ Copying them here too loaded them twice, and project rules outrank user-level
+ ones, so a stale project copy silently overrode the fresh global rule.
+ =sync-language-bundle.sh= sweeps copies left by earlier installs.
- =.claude/hooks/= — PostToolUse validation scripts
- =.claude/settings.json= — permission allowlist + hook wiring
- =githooks/= — git hooks (activated via =core.hooksPath=)
@@ -80,9 +85,18 @@ re-encrypt. See [[file:mcp/README.org][mcp/README.org]] for the full pipeline.
* Available languages
-| Language | Path | Notes |
-|----------+------------------+----------------------------------------------|
-| elisp | =languages/elisp/= | Emacs Lisp — ERT, check-parens, byte-compile |
+| Language | Path | Notes |
+|------------+-------------------------+-----------------------------------------------|
+| bash | =languages/bash/= | Shell, shellcheck validate hook, bats tests |
+|------------+-------------------------+-----------------------------------------------|
+| elisp | =languages/elisp/= | Emacs Lisp, ERT, check-parens, byte-compile |
+|------------+-------------------------+-----------------------------------------------|
+| go | =languages/go/= | Go, gofmt + go vet hook, table-driven tests |
+|------------+-------------------------+-----------------------------------------------|
+| python | =languages/python/= | Python, pytest, coverage-summary |
+|------------+-------------------------+-----------------------------------------------|
+| typescript | =languages/typescript/= | TypeScript, coverage-summary |
+|------------+-------------------------+-----------------------------------------------|
Add more by creating =languages/<name>/= with the same structure.
diff --git a/archive/task-archive.org b/archive/task-archive.org
new file mode 100644
index 0000000..9e053b4
--- /dev/null
+++ b/archive/task-archive.org
@@ -0,0 +1,2009 @@
+#+TITLE: Task Archive
+#+FILETAGS: :archive:
+
+* Resolved (archived)
+** DONE [#C] Fix =cj-scan= false positives on cj fences nested inside other =#+begin_*= blocks :bug:
+CLOSED: [2026-05-15 Fri]
+
+=cj-scan.py= was matching =#+begin_src cj:= / =#+end_src= line-by-line
+without awareness of enclosing block scopes. A cj fence embedded inside a
+=#+begin_example= block (typically when documenting what the =<cj= yasnippet
+emits) or inside =#+begin_src snippet= (the yasnippet definition itself) was
+misclassified as a live cj annotation. Surfaced from a /respond-to-cj-comments
+run against the dotemacs =todo.org= that reported two false positives in the
+=<cj= yasnippet documentation.
+
+Fix: track an active =wrapper_type= state. When the scanner sees =#+begin_<type>=
+(for any =<type>= other than =cj:= via the more-specific cj-open regex, which
+is checked first), it enters a wrapper state where every line is treated as
+content until the matching =#+end_<type>= closer fires. Inside a wrapper, cj
+fence patterns and legacy inline =cj:= lines are both suppressed.
+
+Tests: added =TestCjScanNestedFencesIgnored= (6 tests) to
+=claude-templates/.ai/scripts/tests/test_cj_scan.py= covering nesting inside
+=#+begin_example=, =#+begin_src <other-lang>=, and =#+begin_quote=, plus
+regression guards that a wrapper closes cleanly (a subsequent real cj fence
+is still detected) and that an unclosed wrapper doesn't silently swallow
+later content into false-positive cj blocks.
+
+Full =make test-scripts= equivalent (=python3 -m pytest=): 302 passed, 1
+skipped, 0 failures.
+** DONE [#A] Add =make doctor= — verify ~/.claude/ matches repo + settings.json :feature:
+
+A drift detector that scans =~/.claude/= and reports anything inconsistent with what the repo expects. Single-command answer to "is my machine consistent with rulesets?"
+
+*** Why this matters
+
+A 2026-05-06 sweep found =~/.claude/hooks/= didn't exist on this machine even though =settings.json= referenced =~/.claude/hooks/precompact-priorities.sh= as a PreCompact hook. Compaction would have silently failed to invoke the hook. The fix was =make install-hooks=, but the breakage was invisible until I happened to grep for it. =make doctor= run regularly (or even as part of session start) would catch this kind of drift in seconds instead of after the fact.
+
+*** Checks
+
+- Every entry in =settings.json= ="hooks"= block points at a file that exists.
+- Every entry in =enabledPlugins= has a matching install under =~/.claude/plugins/data/=.
+- Every skill in =$(SKILLS)= has a working symlink at =~/.claude/skills/<name>=.
+- Every rule in =$(RULES)= has a working symlink at =~/.claude/rules/<name>=.
+- Every default hook has a symlink at =~/.claude/hooks/<name>= (warn-only — opt-out is legitimate).
+- =settings.json= and =.mcp.json= symlinks resolve to the rulesets versions.
+- =mcp/install.py= state matches =claude mcp list= (every server in =servers.json= is registered).
+- No dangling symlinks anywhere under =~/.claude/=.
+
+*** Output
+
+One line per check: =ok= / =WARN= / =FAIL=. Final summary: =N ok, M warnings, K failures=. Exit non-zero on any failure so it can ride a pre-flight check.
+** DONE [#A] Build =voice= skill — combine =humanizer= with universal + personal style passes :feature:
+
+Combine =humanizer= with universal good-writing passes (Strunk & White, Orwell, Plain English) and the personal-style passes from =commits.md=. Two modes — =general= for arbitrary writing, =personal= for commits/PRs/comments — share a foundation and diverge on register.
+
+Built and shipped 2026-05-07: =voice/SKILL.md= with 39 numbered patterns walked sequentially. Patterns 1-25 carried over from humanizer, 26-31 are universal good-writing additions, 32-39 are personal-only. Migrated three callers (=commits.md=, =respond-to-cj-comments.md=, =start-work.md=). Removed the standalone =humanizer= skill since voice supersedes it.
+
+*** Why this matters
+
+Three transformations want to run together for personal-mode artifacts (commits, PR titles + bodies, PR comments) but lived in three places: =humanizer= as a skill, S&W-style universal rules nowhere (applied ad-hoc), and the personal-style passes as prose steps in =commits.md= that got re-applied by hand each time. Costs: (1) the "I forgot pass (e)" failure mode — skipping a pass without flagging is a defect but happens in practice. (2) No single-call invocation of the full transform. (3) General-mode writing (research notes, philosophy, history) got only humanizer with no universal-prose pass at all. Combining brings them under one skill with one invocation.
+
+*** Design
+
+Two modes:
+
+- *general* (default) — for arbitrary writing not bound for commit/PR/comment publishing (research notes, philosophy/history essays, emails, README prose). Runs:
+ - humanizer (current behavior — strip AI-generated-writing fingerprints)
+ - tier-1 universal passes (canonical good-writing rules)
+ - the 2 personal-style passes that have no register conflict (jargon-fragment rewrite, noun-ified verbs)
+
+- *personal* — for commits, PR titles + bodies, PR comments. Runs general PLUS:
+ - 8 personal-only passes (first-person rewrite, semicolons, contractions, sentence-split, felt-experience, sentence fragments, terse cut, public-artifact scope check)
+
+The 8 personal-only passes are explicitly *not* in general mode. They conflict with academic / literary / philosophical register. Forcing first-person on a Foucault essay or stripping felt-experience from a journal entry would damage the writing.
+
+*** Tier 1 universals (v1)
+
+From Strunk & White, Orwell's "Politics and the English Language", Plain English Campaign, and Garner's Modern English Usage. Each is a detection-pattern + rewrite-rule pair, mechanical enough to apply consistently across runs.
+
+- *Omit needless words* — curated phrase list (=the fact that= → =that=/=because=, =in order to= → =to=, =at this point in time= → =now=, =due to the fact that= → =because=, =for the purpose of= → =to=, =in spite of= → =despite=, etc.)
+- *Long word → short word* — Plain English wordlist (~150 entries: =utilize=→=use=, =commence=→=start=, =terminate=→=end=, =facilitate=→=help=, =demonstrate=→=show=, =sufficient=→=enough=, =prior to=→=before=, =subsequent to=→=after=, =in the event that=→=if=, =a great deal of=→=much=)
+- *Active over passive voice* — detect "to be + past-participle" patterns. Suggestion-only in v1 (auto-rewrite is risky in technical contexts where passive is appropriate); graduate to auto-rewrite for unambiguous cases in v2.
+- *Comma splices* — detect independent clauses joined only by comma; rewrite to period or semicolon-then-period.
+- *Cliché flag* — small curated list (=at the end of the day=, =moving forward=, =going forward=, =at this juncture=, =circle back=, =low-hanging fruit=, =deep dive=, =leverage= as verb).
+
+*** Tier 2 universals (v2)
+
+- *Positive over negative form* (S&W) — =not unlike= → =like=, =do not fail to= → =remember to=, =did not pay any attention= → =ignored=
+- *Garner-style word-pair corrections* — comprise/compose, less/fewer, that/which (restrictive vs nonrestrictive), affect/effect, principal/principle
+- *Parallelism in lists* — detect mismatched grammar in bullet items
+- *Tense consistency* — flag mid-paragraph tense shifts
+- *Acronym definition on first use* — detect uppercase tokens used before being expanded
+
+*** Tier 3 (v3, may not land)
+
+- *Concrete-over-abstract* preference
+- *Emphatic word at sentence end* (S&W rule 18)
+- *Vary sentence length / rhythm*
+- *Reading-grade-level scoring* (Hemingway-style)
+
+*** Personal-style pass placement
+
+| # | Pass | Mode | Why |
+|----+-------------------------------------+-------------------------------------+-------------------------------------|
+| 1 | First-person voice rewrite | personal only | Forces "I" voice; wrong for |
+| | | | academic prose where third-person |
+| | | | and "we" are conventional |
+|----+-------------------------------------+-------------------------------------+-------------------------------------|
+| 2 | Jargon-fragment → complete sentence | both | Universal clarity, no genre |
+| | | | conflict |
+|----+-------------------------------------+-------------------------------------+-------------------------------------|
+| 3 | Semicolon → period/comma | personal only | Semicolons are conventional in |
+| | | | long-form / academic prose |
+|----+-------------------------------------+-------------------------------------+-------------------------------------|
+| 4 | Contractions ("it's", "don't") | personal only | Academic and formal writing |
+| | | | typically avoids contractions |
+|----+-------------------------------------+-------------------------------------+-------------------------------------|
+| 5 | Sentence split on conjunctions | personal only | Foucault, Hegel, Adorno |
+| | | | deliberately use long compound |
+| | | | sentences |
+|----+-------------------------------------+-------------------------------------+-------------------------------------|
+| 6 | Felt-experience narration ("I'll | personal only | Personal essays *use* |
+| | feel this every time") | | felt-experience as content |
+|----+-------------------------------------+-------------------------------------+-------------------------------------|
+| 7 | Noun-ified verbs ("the ask", "a | both | Targets corporate-speak with |
+| | learn", "the spend") | | curated wordlist; doesn't catch |
+| | | | philosophical nominalizations like |
+| | | | "the becoming" |
+|----+-------------------------------------+-------------------------------------+-------------------------------------|
+| 8 | Sentence fragments → complete (in | personal only | Fragments are valid stylistic |
+| | prose) | | devices in literary prose |
+|----+-------------------------------------+-------------------------------------+-------------------------------------|
+| 9 | Terse cut (rhetorical padding: | personal only | Tier 1 omit-needless-words covers |
+| | "worth noting", "it's important to | | the worst offenders universally; |
+| | understand") | | aggressive cut conflicts with |
+| | | | academic register |
+|----+-------------------------------------+-------------------------------------+-------------------------------------|
+| 10 | Public-artifact scope check (local | personal only — *flag-only*, no | Operational/safety check, not |
+| | paths, private repos, personal | auto-rewrite | stylistic; auto-masking risks |
+| | tooling) | | silently editing meaningful text |
+|----+-------------------------------------+-------------------------------------+-------------------------------------|
+
+*** Inclusive-language pass — explicitly excluded
+
+Considered and rejected. Conflicts with planned writing on philosophy/history topics (Foucault on sexuality and gender, history of slavery in New Orleans). Wordlist substitutions would override deliberate vocabulary choices in those genres.
+
+*** V1 scope
+
+- [ ] Skill at =~/code/rulesets/voice/= with =SKILL.md=
+- [ ] Frontmatter with positive triggers (commit, PR, comment, "humanize", "voice pass") and negative triggers (code, structured data, plain bullet lists)
+- [X] Mode invocation: default = =general= when invoked bare; =personal= invoked explicitly by publish-context callers
+- [X] humanizer content migrated from =humanizer/= → =voice/=
+- [X] Tier 1 universal passes implemented (5 patterns: #26-30, plus #31 noun-ified verbs as a universal personal addition)
+- [X] 2 personal passes that run in both modes (#30 jargon-fragment, #31 noun-ified verbs)
+- [X] 8 personal passes that run in personal mode only (#32 first-person, #33 semicolons, #34 contractions, #35 sentence-split, #36 felt-experience, #37 fragments, #38 terse cut, #39 scope check)
+- [X] Each pass = detection-pattern + rewrite-rule pair (#39 is detection + flag-only)
+- [X] Total v1 pattern count: 31 in general mode (humanizer's 25 + 4 tier-1 + 2 universal personal); +8 personal-only = 39 in personal mode
+- [X] Update =commits.md= to invoke =/voice personal= instead of "run =humanizer= and apply five passes manually"
+- [X] Remove the existing =humanizer/= skill (no callers outside this repo, all migrated)
+- [X] =make doctor= still passes
+- [X] =make lint= clean
+
+*** v2 (deferred)
+
+- [ ] Tier 2 universals (positive form, word-pair corrections, parallelism, tense consistency, acronym definition)
+- [ ] Per-pass severity flags for Tier 1 active-voice (suggestion-only when actor is implicit; auto-rewrite when actor is named)
+- [ ] Reporting mode: list which passes fired and which were no-ops
+
+*** v3 (aspirational, may not land)
+
+- [ ] Tier 3 (concrete-over-abstract, emphatic-word position, sentence-length variation, reading-grade scoring)
+- [ ] Progressive disclosure split: =voice/SKILL.md= orchestrator + =voice/passes/<pass-name>.md= per pass with worked examples
+
+*** Migration (resolved)
+
+Decision: deleted =humanizer/= entirely. Three callers (=commits.md=, =respond-to-cj-comments.md=, =start-work.md=) all updated to invoke =/voice= directly. No alias needed since nothing outside the repo invoked humanizer.
+
+*** Naming alternatives considered
+
+- =voice= — chosen. Captures both modes; broad enough.
+- =polish= — descriptive of multi-pass nature; less prescriptive about whose voice.
+- =house-style= — signals "this is the house style"; appropriate for personal repo.
+- =commit-voice= — too narrow (passes apply to research notes, emails, etc. in general mode).
+- =humanize= (extending current) — undersells the universal + personal additions.
+
+*** Open questions before implementation
+
+Resolved during implementation:
+- Default mode when =/voice= is invoked bare: =general=. Personal-context callers (=commits.md= publish flow, =respond-to-cj-comments.md=) invoke =/voice personal= explicitly. Avoids accidentally first-person-ifying research notes.
+- Reporting: skill prints "Summary of changes" listing which patterns fired (audit value).
+- Public-artifact scope check (#39): flag-only, user resolves manually. Blocking would frustrate on legitimate path mentions.
+- Tier 1 active-voice detection: suggestion-only in v1. Auto-rewrite for unambiguous cases deferred to v2.
+** DONE [#B] Add =--archive-done= mode to =.ai/scripts/todo-cleanup.el= :feature:
+
+Opt-in mode that moves every level-2 subtree whose TODO state is DONE or CANCELLED out of the "Open Work" section and into the "Resolved" section of the same org file, subtree intact.
+
+- *Section matching.* Key on a top-level heading containing "Open Work" and one containing "Resolved" — that pairing is the only naming consistent across projects (=Work Open Work= / =Work Resolved= here; bare =Open Work= / =Resolved= elsewhere). Require exactly one match for each; otherwise skip with a clear message, no crash.
+- *Modes.* =--check= previews and writes nothing, same as the existing hygiene pass. Idempotent. Not run by default in the wrap-up flow — archiving is consequential, so it stays opt-in: =emacs --batch -q -l todo-cleanup.el --archive-done FILE=.
+- *Edge cases.* Source or target section missing; subtree at EOF; nested DONE subtree under an open parent stays put (only level-2 entries move); nothing to move → clean no-op.
+- *Tests.* TDD with ERT — the project's first elisp tests. Fixtures (synthetic) under =.ai/scripts/tests/=; run via =make test= (rulesets) or =make test-scripts= (claude-templates), which run pytest + every =tests/test-*.el= ERT suite. Cases: one DONE level-2 moves; multiple; CANCELLED also moves; structural (no-state) headings don't move; nested DONE under an open parent stays; level-2 DONE with open level-3 children moves intact; subtree at EOF; missing source/target section; ambiguous "Resolved"; lowercase headings; nothing-to-do; idempotency; =--check= preview + its idempotency; realistic-sample integration.
+
+Origin: came up while scrubbing a project's todo.org on 2026-05-11 — moving a big completed PROJECT subtree (plus a few smaller ones) into the Resolved section by hand was the cue to build a reusable tool.
+
+Built and shipped 2026-05-11: =--archive-done= added to =.ai/scripts/todo-cleanup.el= test-first; 13-test ERT suite (=tests/test-todo-cleanup.el=) + realistic synthetic fixture (=tests/fixtures/todo-sample.org=), wired into =make test= / =make test-scripts= alongside pytest. The CLI dispatch moved into =tc-main= behind a guard so the suite can =require= the file without firing it. Section matching is case-insensitive and tolerates the =<Project> Open Work= / =<Project> Resolved= naming variants. Opt-in only — not wired into the wrap-up flow. Source of truth is =~/projects/claude-templates/=; rsync'd into this repo.
+** DONE [#B] Encode follow-up filing rules into =/start-work=
+CLOSED: [2026-05-15 Fri]
+
+Phase 4 step 5 of =/start-work= ("refactor audit") says any candidate that isn't fix-now must land in one of three buckets: fold-into-related-commit, separate =refactor:= commit, or "file a ticket or todo.org entry." The third disposition doesn't say *where* — which leaves the orchestrator picking a location ad-hoc. Result: follow-ups buried under children of an epic parent get orphaned when the parent closes, or follow-ups for standalone tasks scatter across the file with no convention.
+
+Proposed placement rule (already memorized for this project as =feedback_followups_as_siblings.md=, generalizing):
+
+- *Epic-style parent task* (level-2 with multiple level-3 children) → follow-ups file as level-2 *siblings* of the parent. Stays visible after parent closure.
+- *Standalone task* (level-2 with no children, or a level-3 inside another structure) → follow-up files as a new level-2 top-level entry in the same =* Open Work= section. Don't nest under the originating task.
+
+Both cases: include a "Triggered by: <date> <task or commit>" line so a future reader sees what surfaced it.
+
+Update =.claude/commands/start-work.md= Phase 4 step 5's "Disposition for each candidate" section to spell this out. Update any cross-references in =commits.md= or other files that touch the discipline.
+
+Triggered by: 2026-05-15 fold-epic session — Craig flagged the gap mid-flight after I'd surfaced a follow-up but hadn't filed it.
+** DONE [#A] Consolidate =.ai/= template infrastructure (fold + audit + install-ai + ratio) :feature:
+CLOSED: [2026-05-15 Fri]
+
+End-state: one repo (=rulesets=) is the single source of truth for =.ai/= template content. =make audit= verifies and applies drift across every =.ai/=-using project on the machine. =make install-ai= bootstraps new projects. Same setup propagated to ratio so both machines run the same way.
+
+Today (2026-05-15) the canonical-source rule got violated again: rulesets commit =372fb76= added a wrap-up subsection to =rulesets= without going through =claude-templates= first, and the next session's startup rsync was about to silently undo it. Two-repo coordination is the root cause; fold solves it.
+
+Build order: fold first (others depend on the new canonical path), then audit + install-ai in parallel, then test, then propagate to ratio.
+
+*** DONE [#A] Fold =claude-templates= into rulesets
+CLOSED: [2026-05-15 Fri]
+
+Two repos, one source of truth. =~/projects/claude-templates/= is the canonical =.ai/= template that gets rsync'd into every project at session start. Keeping it standalone means a second =git pull= in startup Phase A.0, a second remote to push to at wrap-up, and a split history any time a change touches both. Folding it into =rulesets/claude-templates/= gives one repo to clone on a fresh machine and one place to edit templates.
+
+**** Open design choices
+
+- *History.* =git subtree add --prefix=claude-templates ~/projects/claude-templates main= preserves the 84-commit history under the new prefix. Plain content copy (=cp -a= + =git add=) is simpler but loses history. Either is fine since the standalone repo stays archived on =cjennings.net=.
+- *Layout.* =rulesets/claude-templates/= mirrors the old repo name and sits next to =claude-rules/= cleanly. Alternative: absorb =.ai/= directly under a different name (=rulesets/.ai-template/= or similar). First option is clearer.
+- *bin/ai.* The standalone Makefile symlinks =$HOME/.local/bin/ai → bin/ai=. After the move, fold that into rulesets' Makefile as another install target.
+
+**** Mechanical steps
+
+1. Subtree-merge or copy =~/projects/claude-templates/= into =rulesets/claude-templates/=.
+2. Update 3 references in rulesets:
+ - =.ai/protocols.org= line 163 — pointer in the "Let's run/do the X workflow" section.
+ - =.ai/workflows/cross-agent-comms.org= line 8 — promotion-target path.
+ - =.ai/workflows/startup.org= lines 22, 96-98 — Phase A.0 pull + Phase A rsync sources.
+3. Update Phase A.0 of =startup.org= to pull rulesets instead of claude-templates. Inside rulesets sessions, the existing project-repo pull already covers it. Outside rulesets (every other project's session), Phase A.0 needs an explicit =git pull= on =~/code/rulesets/= before the rsync — otherwise the templates will be stale.
+4. Replace =~/projects/claude-templates/= with a symlink to =~/code/rulesets/claude-templates/= for transition continuity.
+5. After every active project has had one session start (and rsync'd the new =startup.org=), drop the symlink and archive =cjennings.net:git/claude-templates.git=.
+
+**** Bootstrap gap
+
+Every project on the machine has a =.ai/workflows/startup.org= that rsyncs from =~/projects/claude-templates/=. Until each project's startup.org gets refreshed (which happens via the rsync itself), the old path needs to keep resolving. The symlink at step 4 is the bridge: old paths resolve into the new location, the rsync delivers the updated startup.org, next session uses the new path directly.
+
+*** DONE [#A] Add =make audit= — drift detector across all =.ai/=-using projects
+CLOSED: [2026-05-15 Fri]
+
+Companion to =make doctor= (single-machine scope, checks =~/.claude/=). =audit= is cross-project scope: walks every directory on the machine that has a =.ai/=, diffs the synced template files against the canonical source, and reports drift. =--apply= flag rsyncs the drift into the project's working tree (no auto-commit). Catches stale projects without forcing a session start in each one.
+
+**** Open design choices
+
+- *Scope.* Template-sync drift is the useful flavor: for each project, diff =.ai/protocols.org=, =.ai/workflows/=, =.ai/scripts/= against the canonical source.
+- *Source path.* Post-fold: =~/code/rulesets/claude-templates/.ai/=. Build =audit= against the new path from day one.
+- *Project discovery.* Walk =~/code/=, =~/projects/=, =~/.emacs.d/= up to depth 3 for any directory containing =.ai/=. Skip the canonical source itself.
+- *Default mode is report-only.* =--apply= triggers rsync; =--force= overrides the dirty-skip safety.
+
+**** Per-project flow (designed 2026-05-15)
+
+For each discovered project, in order:
+
+1. Verify =.ai/= exists (path probe). If missing → =FAIL=, skip, continue loop.
+2. Detect git tracking via =git check-ignore .ai/= → =tracked= or =gitignored=.
+3. Verify no uncommitted =.ai/= changes (=git status --porcelain .ai/=). Dirty → =WARN=, skip rsync unless =--force=.
+4. Verify content matches canonical via three =rsync -a --dry-run --itemize-changes= calls (=protocols.org=, =workflows/=, =scripts/=). Zero items = clean.
+5. Action (=--apply= only, drift detected): three =rsync -a [--delete]= calls.
+6. Verify rsync converged (re-run the dry-runs; zero now).
+7. Verify working-tree state after rsync (tracked projects). Report deltas. Do not auto-commit.
+8. Verify no unpushed =.ai/= commits (=git log @{u}..HEAD -- .ai/=). Informational only.
+
+**** Output format (mirrors =doctor=)
+
+#+begin_example
+Claude-templates source:
+ ok rulesets/claude-templates is current (origin/main)
+
+Per-project .ai/ drift:
+ ok ~/projects/work
+ applied ~/projects/homelab 3 files changed
+ skipped ~/code/winvm uncommitted .ai/ (use --force)
+ ok ~/projects/clipper
+
+Summary: 18 ok, 3 applied, 1 skipped, 0 failed
+#+end_example
+
+Exit code: =0= if all clean, no skips, no failures. =1= otherwise.
+
+**** Why not extend =make doctor= instead
+
+=doctor= has a clean meaning today: "is this machine's =~/.claude/= consistent with rulesets?" Mixing in cross-project =.ai/= drift muddies the exit code. Keep them separate. =audit= can optionally invoke =doctor= as its last check since both ask "did the symlinks keep up with the source?". A future =make all-checks= can wrap both.
+
+*** DONE [#A] Add =make install-ai PROJECT=<path>= — bootstrap =.ai/= in a fresh project
+CLOSED: [2026-05-15 Fri]
+
+Separate target from =audit= because operating on projects that lack =.ai/= is a distinct action. The absence might be intentional, so =audit= skips them. Bootstrap is explicit opt-in.
+
+**** Flow
+
+1. Refuse if =.ai/= already exists in =PROJECT=. Message: "already installed; use =make audit --apply= to update."
+2. Verify =PROJECT= is a git checkout (warn if not — works without git, loses some lifecycle benefits).
+3. Create =PROJECT/.ai/= directory.
+4. Rsync canonical content: =protocols.org=, =workflows/=, =scripts/= (same three rsyncs as =audit=).
+5. Seed =PROJECT/.ai/notes.org= from a canonical template with project-name placeholder.
+6. Create empty =PROJECT/.ai/sessions/= (with =.gitkeep= for tracked projects).
+7. Track or gitignore =.ai/=? Default: ask. Flag: =--track= / =--gitignore=.
+8. Print next-steps banner: =make install-lang LANG=<lang> PROJECT=<path>=; open Claude Code in the project.
+
+**** Symmetry with existing install targets
+
+#+begin_example
+make install-lang LANG=python PROJECT=/path # language bundle (existing)
+make install-ai PROJECT=/path # .ai/ template (new)
+make install-lang # no args → fzf-pick
+make install-ai # no args → fzf-pick from
+ # ~/projects/* + ~/code/* dirs
+ # without an existing .ai/
+#+end_example
+
+*** DONE [#A] Test plan for audit + install-ai before propagating to ratio
+CLOSED: [2026-05-15 Fri]
+
+Test against the current state of this machine before pushing changes to ratio.
+
+**** =make audit= tests
+
+1. Dry-run report only (no =--apply=). Should show: claude-templates current; per-project drift; correct =ok=/=drift= classifications; summary line and exit code match.
+2. After the fold lands, every project should be reported as drift (their =startup.org= still points at the old path). Run =--apply= → rsync converges. Re-run audit → all =ok=.
+3. Manually edit one =.ai/workflows/foo.org= in a tracked project. Re-run audit → should report =skipped: uncommitted .ai/=. Run =--apply --force= → rsync clobbers the edit. Verify the edit is gone.
+4. Manually delete one =.ai/= dir. Re-run audit → =FAIL: .ai/ missing=. Loop continues.
+5. Idempotency: =--apply= twice in a row converges to all =ok= on the second pass.
+
+**** =make install-ai= tests
+
+1. Create =/tmp/test-fresh-project= as a git repo. Run =make install-ai PROJECT=/tmp/test-fresh-project=. Verify =.ai/= structure matches canonical, =notes.org= has placeholder, =sessions/= exists.
+2. Run =make install-ai PROJECT=/tmp/test-fresh-project= again → should refuse (=.ai/= already exists).
+3. Open Claude Code in the new project. Startup workflow runs cleanly (Phase A.0 + Phase A rsync should be a no-op since the install just ran).
+4. fzf form: =make install-ai= with no args. Lists candidate dirs (=~/projects/*=, =~/code/*= without =.ai/=).
+
+**** Pass criteria
+
+- =audit= behavior matches the per-project flow spec for every classification path.
+- =install-ai= produces a project indistinguishable from one that's been running sessions for a while.
+- =make doctor= still passes 36/0/0 after all the work.
+- =make test= (pytest + ERT) passes.
+
+*** DONE [#A] Migrate projects on ratio (second machine)
+CLOSED: [2026-05-15 Fri]
+
+After local fold + audit + install-ai are working, propagate to ratio.
+
+**** Steps
+
+1. On ratio: =git -C ~/code/rulesets pull= — picks up the folded =claude-templates/= subdir and updated =Makefile= targets.
+2. On ratio: archive or =mv= the standalone =~/projects/claude-templates/= aside, replace with symlink to =~/code/rulesets/claude-templates/= (same bridge mechanic as local).
+3. On ratio: =make audit= → see drift across ratio's projects.
+4. On ratio: =make audit --apply= → rsync into each tracked/gitignored project. Surface projects with uncommitted =.ai/= drift for manual handling.
+5. On ratio: =make doctor= → catch any =~/.claude/= install drift (likely some, since ratio hasn't seen recent rulesets updates).
+6. Verify by opening Claude Code in a few ratio projects. Startup should be a no-op or near-zero rsync.
+
+**** Known unknowns
+
+- Ratio may have its own project list overlapping with this machine's but not identical. =audit= discovers projects via the walk, so this is automatic.
+- Ratio might have uncommitted =.ai/= work in some projects that this machine doesn't. =audit= surfaces them; handle case-by-case.
+- If anything goes wrong, ratio's archived =~/projects/claude-templates/= is the safety net — restore the symlink target and re-run audit.
+
+**** Adjacent: cross-machine memory sync
+
+The =[#A] DOING= memory-sync investigation (todo.org:10) is adjacent. Both involve "make my Claude setup portable across machines." Coordinate so the memory-sync stow approach (if approved) doesn't conflict with this fold's symlink mechanics.
+** DONE [#B] Document startup pull-ordering rule in protocols.org
+CLOSED: [2026-05-15 Fri]
+
+Phase A.0 of =startup.org= now pulls rulesets ff-only before the project repo
+(shipped 2026-05-15 as part of the claude-templates fold — after the subtree
+merge, there's no separate claude-templates pull, just rulesets-then-project).
+The protocols.org paragraph stating the ordering and "resolve any issues
+before proceeding" rule shipped 2026-05-15 in the =** Startup Pull Ordering=
+subsection under =IMPORTANT - MUST DO=.
+** DONE [#A] Build =/lint-org= skill + wrap-up integration
+CLOSED: [2026-05-14 Thu]
+
+Spec: [[file:.ai/specs/lint-org-skill-spec.md]]
+
+A two-mode skill (=interactive=, =mechanical-only=) that runs =org-lint=,
+auto-fixes safe categories (item-number, missing-language-in-src-block,
+misplaced-planning-info, markdown-bold → single-asterisk), and walks judgment
+items (broken local-file links, invalid fuzzy links, verbatim-asterisk false
+positives, suspicious-language blocks) inline.
+
+Wrap-up integration: =wrap-it-up.org= invokes
+=/lint-org todo.org --mode=mechanical-only= after the existing
+=todo-cleanup.el --archive-done= pass. Judgment items defer to a
+carry-forward file that the next morning's daily-prep merges in, so
+wrap-up never blocks on a judgment call.
+
+Baseline that motivated this: the 2026-05-14 manual pass took =todo.org=
+from 55 → 1 lint warnings across two commits (=0d10458= signal,
+=9ad5b30= cosmetic). A nightly mechanical sweep keeps the count near
+zero forever — each day's drift is small.
+** DONE [#C] Test harness for =make audit= + =make install-ai= edge cases :test:
+CLOSED: [2026-05-15 Fri]
+
+Three edge cases from the fold-epic test plan were not exercised because they're destructive on real projects:
+
+- =audit --force= clobbers uncommitted =.ai/= work — needs a project with intentionally dirty =.ai/= to verify the override path.
+- =audit= reports =FAIL= when =.ai/= is missing — needs a project where the directory was deleted to verify the loop continues past the failure.
+- =install-ai= fzf-pick form (no =PROJECT= arg) — needs interactive testing.
+
+Build a self-contained test harness under =.ai/scripts/tests/= that spins up =/tmp/audit-test-projects/= with a known matrix of project states (clean, dirty, missing =.ai/=, pristine, etc.), runs the audit + install-ai targets against it, and asserts expected outputs. The harness should clean up after itself.
+
+Pattern reference: bats or shell-based assertions (similar to the elisp ERT suites for =todo-cleanup= and =lint-org=, but for shell scripts).
+
+Triggered by: 2026-05-15 fold-epic, child 4 test plan; commits =94782ee= (audit) + =d364cf2= (install-ai).
+** DONE [#A] wrap it up mentions github, which isn't the remote for many projects. :chore:
+CLOSED: [2026-05-16 Sat]
+For many of them, git.cjennings.net mirrors to github.com, and github.com isn't the remote.
+For many others, git.cjennings.net is the remote with no mirror.
+Remove or replace the reference to github.com
+** DONE [#B] Phase A startup blind to =claude-templates/inbox/= post-fold :bug:fold:
+CLOSED: [2026-05-19 Tue]
+
+Resolved on inspection: the bug is moot in current state. =inbox-send.py='s discovery scans =~/code/*= and =~/projects/*= single-level only, so =claude-templates/= (two levels under =~/code/=) is never a routable target; the 2026-05-15 incident was a one-time manual workaround because =rulesets/inbox/= didn't exist yet, and that root inbox was added in =470085f=. =claude-templates/inbox/= was removed 2026-05-15 and is no longer on disk.
+
+Phase A's inbox check at =startup.org:107= runs =\ls -la inbox/= against the project root. Post-fold, the canonical's inbox sits inside the subtree at =claude-templates/inbox/= and never gets scanned. A 2026-05-15 cross-project handoff from a dotemacs session dropped a record there; the next rulesets session (this one) missed it at startup entirely. Picked up only when the working-tree drift surfaced during the publish flow.
+
+Fix: extend Phase A's discovery to also scan =claude-templates/inbox/= when the canonical lives in-repo (i.e., when =claude-templates/.ai/= exists alongside =./.ai/=). The Phase B/C inbox-processing flow already handles per-file routing once a file is surfaced; the gap is only in discovery.
+
+Adjacent question worth answering at the same time: should cross-project handoffs file into =./inbox/= at the project root (matching what Phase A already scans), or stay in =claude-templates/inbox/= and rely on the discovery fix? The =inbox-send= script's target-project logic is the place to settle that.
+
+Triggered by: 2026-05-15 evening session, surfaced when committing the test-harness work.
+** DONE [#A] Implement task-review daily-habit per spec
+CLOSED: [2026-05-20 Wed]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-05-20
+:END:
+Spec: [[file:docs/design/task-review.org]]
+
+Retires =wrap-it-up.org='s date-coverage scan and replaces it with a daily list-hygiene review (N=7 oldest-unreviewed top-level =[#A]= / =[#B]= / =[#C]= tasks per session, ~12-day rotation). Built as a pure Claude workflow — Shape B, no elisp; see the spec's Revision section for why the elisp approach was dropped.
+
+Status:
+1. [X] =task-review-staleness.sh= + bats (count + =--list= modes).
+2. [X] =wrap-it-up.org= health check (threshold 30).
+3. [-] =task-review.el= — dropped (Shape B is a pure workflow, not an Emacs mode).
+4. [X] New =task-review.org= workflow + INDEX entry (the existing listing workflow was renamed to =open-tasks.org= to free the name).
+5. [X] Startup nudge in template =startup.org= (threshold 7), not the project-only startup-extras layer.
+6. [X] Smoke test against live =todo.org= — first cycle run 2026-05-20 (7 tasks reviewed: 3 re-grades, 1 cancellation, 1 bump-and-tag).
+
+Triggered by: 2026-05-16 brainstorm on retiring the date-coverage scan.
+** CANCELLED [#B] Build =ov-1= skill for DoDAF OV-1 (High-Level Operational Concept Graphic)
+CLOSED: [2026-05-20 Wed]
+
+Cancelled during the 2026-05-20 task review.
+
+Triggered by SOFWeek (May 2026, Tampa) — DeepSat attending; DoD attendees
+may ask for architecture diagrams. OV-1 is the universal informal
+currency in DoD briefings ("show me the architecture" → OV-1 by default).
+
+Priority upgrades to =[#A]= if Craig confirms scenario 2 below (personal
+load-bearing need at the event); stays =[#B]= or drops to =[#C]= if
+scenario 1 (team already covers it, future asset only).
+
+*** Prior art (searched 2026-04-19)
+
+No existing Claude Code skill exists for DoDAF / OV-1 / SV-1 / SysML.
+
+- =anthropics/skills= — 17 skills, zero DoDAF/SysML/defense coverage.
+- =awesome-claude-code= list — zero hits for DoDAF/OV-1/SysML/UAF.
+- =mfsgr/sysml2dodaf= — empty repo (0 stars, no code). Vapor.
+- =HowardKao-1130/mini-NEXEN= — broad SE methodology skill that
+ name-drops DoDAF as a trigger keyword; no artifact generation. 0 stars.
+- =gaphor/gaphor= (Apache-2.0, 2.2k stars) — mature UML/SysML GUI
+ modeler. Not a skill; not a pipeline. Useful reference only.
+
+Nearest prior art to lean on when building:
+- DoDAF 2.02 Viewpoints & Models reference (dodcio.defense.gov) —
+ canonical OV-1 exemplars. Embed 3-5 layouts as skill =references/=.
+- Pattern from existing =c4-diagram= skill — same shape (prose → diagram
+ spec), swap the viewpoint vocabulary to DoDAF.
+- PlantUML for SV-1 (when that skill comes later); Mermaid or draw.io
+ XML for OV-1 lightweight visuals.
+
+*** Build scope (when triggered)
+
+*In scope:*
+- Input: prose description of a system + its operational context.
+- Output: structured OV-1 *spec* — performers, external actors (other
+ systems, forces, adversaries), relationships (data/control flows),
+ narrative captions, classification marking, legend requirements.
+- DoDAF 2.02 completeness checklist as a quality gate — verify the
+ produced spec contains every element a correct OV-1 requires.
+- Optional lightweight visual: draw.io XML or Mermaid approximation for
+ quick review; NOT a finished rendering.
+
+*Out of scope:*
+- Icon libraries, pictorial assets, finished PowerPoint export. OV-1
+ final art belongs to a designer or Craig in Visio/PowerPoint; the
+ skill's job is the spec and the check, not the slide.
+- SV-1, SV-2, UAF, IDEF1X, other viewpoints. Build only when a
+ concrete need triggers each.
+
+Estimate: 4-6 hours.
+
+*** Craig's investigation before kickoff
+
+1. Does DeepSat's systems-engineering or marketing team already have an
+ OV-1 (or the equivalent briefing artifact) for SOFWeek?
+2. If yes (scenario 1) — skill is a future asset, not event-load-bearing.
+ Ship after SOFWeek. Priority drops to =[#C]=.
+3. If no, or if the scenario is "Craig may need to produce/iterate an
+ OV-1 on the fly during the event" (scenario 2) — skill is load-bearing
+ for the event. Priority upgrades to =[#A]=; build before SOFWeek.
+4. Confirm the classification level the skill needs to handle
+ (unclassified-only? or FOUO markings? affects the classification
+ block in the spec).
+5. Confirm the target rendering format DeepSat uses for OV-1
+ deliverables (PowerPoint slide? Cameo? Visio? affects whether the
+ skill emits draw.io XML vs Mermaid vs pure structured spec).
+
+*** Related
+
+See also the DoD-specific notations section under the later TODO
+(=c4-*= rename revisit) — OV-1 is flagged there as the highest-value
+starting point across the DoD notation landscape (SysML, DoDAF/UAF,
+IDEF1X). This entry is the execution plan for that starting point.
+** DONE [#A] Split team-specific publishing rules out of commits.md :commits:
+CLOSED: [2026-05-22 Fri]
+Shipped 3cb467e. Moved the DeepSat publishing steps (Linear ticket-state, the Slack notification protocol + channel ID, the GHE host, the team merge norm, the Linear ticket-body structure) out of the global =claude-rules/commits.md= into =teams/deepsat/claude/rules/publishing.md=. The global file keeps the universal skeleton and uses seams ("run the project's publishing overlay here if present") like startup-extras. Added =install-team= (targeted per-project copy, keyed on PROJECT, never globally symlinked) and generalized =sync-language-bundle.sh= to keep team overlays fresh at startup (3 new bats; make test green).
+
+Remaining deploy step (cross-project, surfaced to Craig): install the overlay into the DeepSat work project — =make install-team TEAM=deepsat PROJECT=<deepsat-path>= — so it actually loads there.
+** DONE [#A] Define a /voice-unavailable fallback in the commits.md publish flow :commits:
+CLOSED: [2026-05-22 Fri]
+Added an "If =/voice= is unavailable" paragraph to the Single-skill gate in =commits.md=: walk the same patterns inline (the flow already names which matter), state the skill was unavailable and the pass was applied by hand ("/voice unavailable — patterns walked inline"), and flag the missing skill for install. The gate is the pattern walk, not the tooling. The original "=humanizer= unavailable" framing was moot (humanizer → /voice).
+** DONE [#A] wrap-it-up Step 3.5 assumes GitHub-family remote :chore:quick:
+CLOSED: [2026-05-22 Fri]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-05-20
+:END:
+Documented the assumption inline at =wrap-it-up.org= Step 3.5 (chose the lightweight path over a provider-agnostic rewrite): the =gh= lookup expects a GitHub-family host, holds today via DeepSat on GHE, flagged for update if a future Linear project lands on GitLab/Gitea/Bitbucket.
+Triggered by: 2026-05-16 wrap-it-up github.com cleanup (audit of the same file).
+
+Step 3.5 (Linear ticket-state hygiene) at =wrap-it-up.org:207= says "the project's GitHub remote — use =gh pr list ...=". Currently fine in practice: the step is Linear-gated, and the only Linear-using project is DeepSat (on =deepsat.ghe.com=, a GitHub-family host where =gh= works). Would break if a future Linear-using project lived on a non-GitHub host (gitlab, gitea, bitbucket). Either drop the GitHub-family assumption (provider-agnostic lookup, harder) or document the assumption explicitly so future projects know the step needs an update if they don't fit.
+** DONE [#C] Review pass: tighten skills and rulesets after 2026-05-04 audit
+CLOSED: [2026-05-22 Fri]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-05-20
+:END:
+All 55 grouped-index items dispositioned (2026-05-22): ~49 edited across skills, commands, rule files, hooks, and the two playwright skills; several came out moot post-audit (humanizer→voice, skills→commands, typescript ruleset added); the two commits.md items shipped as the team-overlay split + /voice fallback. Freshness-checked each item against current reality before editing.
+
+Source notes used in this pass:
+- C4 official docs: C4 is notation-independent; System Context and Container
+ diagrams are enough for most teams; every diagram needs title, key/legend,
+ explicit element types, and audience-appropriate abstraction.
+ [[https://c4model.com/diagrams][C4 diagrams]],
+ [[https://c4model.com/diagrams/notation][C4 notation]],
+ [[https://c4model.com/abstractions/component][C4 component]]
+- arc42 docs: quality requirements need measurable scenarios; section 10
+ should reference top quality goals and capture lesser quality requirements
+ with specific measures. [[https://docs.arc42.org/section-10/][arc42 section 10]],
+ [[https://quality.arc42.org/articles/specify-quality-requirements][specifying quality requirements]]
+- ADR references: ADRs capture one justified architecturally significant
+ decision and its rationale; Nygard's original guidance emphasizes short,
+ numbered, repository-stored records and superseding rather than rewriting old
+ decisions. [[https://adr.github.io/][adr.github.io]],
+ [[https://cognitect.com/blog/2011/11/15/documenting-architecture-decisions][Nygard ADR article]]
+- Playwright docs: prefer user-visible locators and web assertions; locators
+ auto-wait and retry; =networkidle= is discouraged for testing readiness.
+ [[https://playwright.dev/docs/best-practices][Playwright best practices]],
+ [[https://playwright.dev/docs/locators][Playwright locators]],
+ [[https://playwright.dev/docs/next/api/class-page][Playwright page API]]
+- OWASP references: Top 10 2021 includes Broken Access Control,
+ Cryptographic Failures, Injection, Insecure Design, Security
+ Misconfiguration, Vulnerable and Outdated Components, Identification and
+ Authentication Failures, Software and Data Integrity Failures, Security
+ Logging and Monitoring Failures, and SSRF; WSTG adds a broader testing map
+ across configuration, identity, authn/z, sessions, input validation, error
+ handling, cryptography, business logic, client-side, and API testing.
+ [[https://owasp.org/Top10/2021/][OWASP Top 10 2021]],
+ [[https://owasp.org/www-project-web-security-testing-guide/latest/4-Web_Application_Security_Testing/][OWASP WSTG]]
+- V2MOM references: Salesforce calls the last M "Measures" and emphasizes a
+ simple alignment document with prioritized Methods, explicit Obstacles, and
+ measurable outcomes. [[https://trailhead.salesforce.com/content/learn/modules/selfmotivation/get-focused-with-your-personal-v2mom][Salesforce Trailhead personal V2MOM]],
+ [[https://www.salesforce.com/blog/?p=12][Salesforce V2MOM alignment]]
+- Prompt research: the cited Meincke paper is titled "Call Me A Jerk:
+ Persuading AI to Comply with Objectionable Requests"; its scope is
+ persuasion increasing compliance with objectionable requests, not a general
+ proof that persuasion framing improves prompt quality.
+ [[https://papers.ssrn.com/sol3/papers.cfm?abstract_id=5357179][SSRN paper]]
+- Combinatorial testing references: NIST supports t-way combinatorial testing
+ and notes pairwise is one covering strength, with higher-strength arrays
+ useful for failures requiring more interacting factors.
+ [[https://www.nist.gov/publications/practical-combinatorial-testing-beyond-pairwise][NIST beyond pairwise]],
+ [[https://www.nist.gov/publications/combinatorial-software-testing][NIST combinatorial testing]]
+
+*** Grouped index (for batching by area)
+
+Each item below is a one-line summary of a sub-TODO further down. Tick the box when the matching sub-TODO is moved to =DONE=. Items are grouped by area so they can be batched (e.g., "do all Playwright items in one session").
+
+**** Browser testing
+- [X] [#A] =playwright-js=: locator/assertion-first guidance (replace raw CSS, =networkidle=)
+- [X] [#B] =playwright-js= + =playwright-py=: reconcile headless/visible defaults
+- [X] [#B] =playwright-js= + =playwright-py=: remove emoji console markers from examples
+
+**** Frontend / UI
+- [X] [#B] =frontend-design=: WCAG 2.2 alignment, accessibility non-optional
+- [X] [#B] =frontend-design=: harmonize aesthetic guidance with anti-pattern rules
+
+**** Security
+- [X] [#A] =security-check=: OWASP 2021 + WSTG coverage
+- [X] [#B] =security-check=: tooling and offline/network caveats
+
+**** Combinatorial testing
+- [X] [#B] =pairwise-tests=: t-way escalation guidance beyond pairwise
+- [X] [#B] =pairwise-tests=: clarify negative value syntax + generator availability
+
+**** V2MOM
+- [X] [#A] =create-v2mom=: rename Metrics → Measures (Salesforce alignment)
+- [X] [#B] =create-v2mom=: prevent task migration from turning V2MOM into a backlog
+- [X] [#B] =create-v2mom=: mitigation/owner fields for Obstacles
+
+**** Prompt engineering
+- [X] [#A] =prompt-engineering=: correct/narrow Meincke citation
+- [X] [#B] =prompt-engineering=: eval-harness requirement for production prompts
+
+**** Codify
+- [X] [#B] =codify=: stale-entry review + privacy checks before writing project =CLAUDE.md=
+
+**** Code review
+- [X] [#A] =review-code=: resolve local-verification vs CI boundary
+- [X] [#B] =review-code=: =CLAUDE.md= citation scope for public artifacts
+- [X] [#B] =review-code=: relax three-strengths rule for tiny/failing diffs
+
+**** PR / review responses
+- [X] [#A] =respond-to-review=: remove review-process language from commit messages
+- [X] [#B] =respond-to-review=: use unresolved threads + resolution state
+- [X] [#B] =respond-to-cj-comments=: drop personal absolute paths from public-writing (moot — already clean)
+- [X] [#B] =respond-to-cj-comments=: fallback when =humanizer= or =emacsclient= unavailable (moot — superseded by /voice + VERIFY pattern)
+
+**** Branch workflow
+- [X] [#A] =finish-branch=: fix base-branch detection
+- [X] [#B] =finish-branch=: worktree-aware pull/merge safety
+- [X] [#B] =start-work=: tool-availability + ceremony-scaling rules
+- [X] [#B] =start-work=: claim-before-justify rollback risk
+
+**** Tests / TDD
+- [X] [#B] =add-tests=: fix missing =typescript-testing.md= reference or add ruleset (moot — ruleset now exists)
+- [X] [#B] =add-tests=: explicit exceptions to "all three categories per function"
+
+**** Debugging / RCA
+- [X] [#B] =debug=: capture environment + recent-change context before hypotheses
+- [X] [#B] =root-cause-trace=: constrain defense-in-depth to trust boundaries
+- [X] [#B] =five-whys=: require evidence + counterfactual validation per why
+
+**** Brainstorming
+- [X] [#B] =brainstorm=: timebox + research/source rules for high-stakes designs
+
+**** Architecture
+- [X] [#B] =arch-decide=: timeless examples, drop unverifiable claims
+- [X] [#B] =arch-decide=: standardize statuses + immutability language
+- [X] [#B] =arch-design=: threat modeling + privacy/compliance as first-class inputs
+- [X] [#B] =arch-design=: separate paradigms from tactical patterns
+- [X] [#B] =arch-document=: arc42/Q42 quality scenarios
+- [X] [#B] =arch-document=: staleness + ownership metadata for generated docs
+- [X] [#B] =arch-evaluate=: confidence levels for framework-agnostic findings
+- [X] [#B] =arch-evaluate=: report skipped tool checks explicitly
+
+**** C4 modeling
+- [X] [#A] =c4-analyze= + =c4-diagram=: notation/output fallback (not draw.io-only)
+- [X] [#B] =c4-analyze= + =c4-diagram=: clarify abstraction boundaries
+
+**** Global rules
+- [X] [#B] =commits.md=: split DeepSat/Linear/Slack-specific from global rules → promoted to a top-level task (deferred for Craig)
+- [X] [#A] =commits.md= + publish flows: =humanizer=-unavailable fallback → promoted to a top-level task (deferred; humanizer premise moot)
+- [X] [#B] =verification.md=: explicit "unable to verify" reporting standard
+- [X] [#B] =testing.md=: property-based + mutation testing as escalation paths
+- [X] [#B] =testing.md=: soften absolute TDD with explicit spike protocol
+- [X] [#B] =subagents.md=: capability/availability + cost checks
+
+**** Languages
+- [X] [#A] =python-testing.md=: revisit in-memory SQLite guidance
+- [X] [#B] =python-testing.md=: separate "never mock ORM" from unit-test boundaries
+- [X] [#B] =elisp.md=: drop tool-specific advice
+- [X] [#B] =elisp-testing.md=: batch-mode + native-comp caveats
+
+**** Hooks
+- [X] [#A] =hooks/README.md=: include =destructive-bash-confirm.py= in install/settings snippets
+- [X] [#A] =hooks/git-commit-confirm.py= + =gh-pr-create-confirm.py=: inspect message/body files referenced by =-F= / =--body-file=
+- [X] [#B] =hooks/destructive-bash-confirm.py=: shell-aware command parsing (not regex)
+
+*** 2026-05-22 Fri @ 15:47:10 -0500 Made playwright guidance locator/assertion-first, dropped networkidle-as-readiness
+
+Rewrote the readiness guidance in both =playwright-js/SKILL.md= and =playwright-py/SKILL.md=: reconnaissance now waits for a visible app landmark via a web assertion or locator (=expect(...).toBeVisible()= / =get_by_role(...).wait_for()=), not =networkidle= (which Playwright discourages). Updated the login/form examples to =getByLabel=/=getByRole= + web assertions, the API_REFERENCE.md waiting section, and =lib/helpers.js= defaults (=waitForPageReady= now defaults to =load= and prefers a caller-supplied landmark; =authenticate= races the success indicator over a =load= navigation). node --check passes.
+
+*** 2026-05-22 Fri @ 14:23:02 -0500 Added headed/headless decision tables to both playwright skills
+
+Added matching purpose-based decision tables to =playwright-js/SKILL.md= (was "always visible") and =playwright-py/SKILL.md= Best Practices (was "always headless"). Each names its own default and points at the other skill, so the difference is deliberate, not a habit-flip: headed for interactive debugging, headless for CI/pytest. Also softened the absolutist "Always launch... headless" comment in the py example.
+
+*** 2026-05-22 Fri @ 15:47:10 -0500 Removed emoji console markers from the playwright skills
+
+Replaced every emoji status marker with a plain ASCII prefix across =playwright-js/= (run.js, lib/helpers.js, SKILL.md) and =playwright-py/= (SKILL.md, examples/*.py): 📦/⚡/📄/📥/🎭/🚀/📋/✅/❌/🔍/📸/✓/✗ → =[setup]=/=[run]=/=[ok]=/=[error]=/=[fail]= etc. Post-change emoji grep is clean (excluding node_modules); node --check and py_compile pass.
+
+*** 2026-05-22 Fri @ 14:35:16 -0500 Made accessibility a non-optional WCAG 2.2 gate in frontend-design
+
+Added an "Accessibility Gate (required before handoff)" section to =frontend-design/SKILL.md= covering keyboard operation, focus visibility, focus-not-obscured (2.2), target size (2.2), contrast, reduced motion, labels, and semantic structure — a baseline for all frontend work, not just interactive components. Rewrote the Build/Review phases to build accessibly as you go and clear the gate before handoff, and bumped =references/accessibility.md= from WCAG 2.1 to 2.2 with backing detail for the new criteria.
+
+*** 2026-05-22 Fri @ 14:35:16 -0500 Added a "creative but bounded" section to frontend-design
+
+Added a subsection under Frontend Aesthetics framing the bold/maximalist directions as tools, not obligations: domain fit, readability first, responsive stability, and no decorative effect that degrades the workflow. Reconciles rather than contradicts the maximalist encouragement (maximalism stays on the table as deliberate usable density), and ties the readability bullet to the new accessibility gate.
+
+*** 2026-05-22 Fri @ 14:35:16 -0500 Updated security-check to OWASP Top 10 2021 + WSTG mapping
+
+Replaced the older six-category list in =.claude/commands/security-check.md= with the full Top 10 2021 set, each finding mapped to a 2021 category or WSTG area. Added the four missing categories (Insecure Design, Software and Data Integrity Failures, Security Logging and Monitoring Failures, SSRF) plus explicit checks for object/function-level authorization, SSRF on URL-fetch paths, update/plugin/dependency integrity, and logging/monitoring gaps.
+
+*** 2026-05-22 Fri @ 14:35:16 -0500 Added scanner tooling + network caveats to security-check
+
+Added an optional configured-scanners step (=gitleaks=/=trufflehog= secrets, =semgrep= source patterns, OSV scanner, lockfile-diff review) that supplements the manual scans, plus a network caveat: dependency audits that can't run (offline, tool absent, DB unreachable) must report "not run" naming the tool and reason, never read as a pass. Carried that into the no-issues summary.
+
+*** 2026-05-22 Fri @ 14:35:16 -0500 Added t-way escalation guidance to pairwise-tests
+
+Added an "Escalating Beyond Pairwise (t-way)" subsection: start with pairwise across the whole space, then escalate specific high-risk clusters to 3-way+ when history, safety, security, or domain coupling says a fault needs more than two interacting factors. Lists escalation triggers and shows the sub-model order syntax (={ A, B, C } @ 3=) vs a blanket =/o:3= bump, stressing targeted not uniform escalation. Cites NIST combinatorial-testing work.
+
+*** 2026-05-22 Fri @ 14:35:16 -0500 Clarified PICT ~ syntax + honest generator-availability path in pairwise-tests
+
+Added a "~ prefix" explanation (PICT marker tagging a value as negative/invalid, not an arithmetic operator; PICT pairs negatives with valid values once and strips the marker before the SUT) and a stop-at-the-model rule: if neither the =pict= binary nor =pypict= is present, produce the model and stop rather than hand-writing a table and passing it off as PICT output.
+
+*** 2026-05-22 Fri @ 14:43:17 -0500 Renamed Metrics → Measures throughout create-v2mom
+
+Full rename across =.claude/commands/create-v2mom.md= (acronym expansions, Phase 7 heading, the "Measures must be measurable" principle, exit criteria, review questions, red flags, examples) to match Salesforce's official term. Kept the "vanity metrics" idiom intact — it's the anti-pattern term, not a section reference.
+
+*** 2026-05-22 Fri @ 14:43:17 -0500 Split strategy from execution in create-v2mom task migration
+
+Rewrote Phase 8 (and tightened Phase 5.5): tasks stay in the backlog grouped by method, and each method gains a one-line link to where its tasks live, instead of transplanting the task tree into the V2MOM. Strategy (V2MOM) and execution (backlog) are now explicitly separate sources of truth, keeping the V2MOM concise.
+
+*** 2026-05-22 Fri @ 14:43:17 -0500 Made create-v2mom obstacles operational (mitigation/owner/cadence)
+
+Phase 6 now captures, per obstacle: name, manifestation, stakes, mitigation, owner, and review cadence — with a worked example per domain (health/finance/software), a "good obstacle" characteristic, a Phase 9 review question, and a red flag for candid-but-not-operational obstacles. An obstacle without a countermove is now flagged as an observation, not a plan.
+
+*** 2026-05-22 Fri @ 14:43:17 -0500 Corrected and narrowed the Meincke citation in prompt-engineering
+
+Fixed the title to "Call Me A Jerk: Persuading AI to Comply with Objectionable Requests" (SSRN abstract_id=5357179) in all three spots (frontmatter, Seven Principles intro, References). Reframed the ~33%→72% result as what it is — a prompt-safety caution that persuasion raises compliance with objectionable requests — explicitly not evidence that persuasion framing improves engineering prompt quality. Kept the seven principles as a tone vocabulary.
+
+*** 2026-05-22 Fri @ 14:43:17 -0500 Added an eval-harness requirement to prompt-engineering critique mode
+
+Added critique step 7 + a checklist line: for fragile or reusable/production prompts, write 3-5 adversarial/edge inputs, run both the old and new prompt against each, and record the behavioral delta. A throwaway prompt can ship on the rewrite alone; a discipline/reused/production one can't. Without examples, "the rewrite is better" is an assertion, not a result.
+
+*** 2026-05-22 Fri @ 14:43:17 -0500 Added mandatory stale-entry + privacy pre-write checks to codify
+
+Added a "Mandatory pre-write checks" block at the top of Phase 3 (Write) in =.claude/commands/codify.md=: a stale-entry scan (update/remove no-longer-true entries in place, don't append contradictions around them) and a privacy/leak check carrying both questions verbatim — "safe if the project were public?" and "belongs in private memory instead?" — routing private content to auto-memory. Gates, not background guidance.
+
+*** 2026-05-22 Fri @ 14:06:41 -0500 Scoped review-code's CI-trust rule to reviewing, not shipping
+
+Expanded the False-Positive Filter bullet in =review-code/SKILL.md=: "trust CI, don't run builds" applies to reading a diff, not producing one. A pre-commit/pre-push flow still owes the local verification =verification.md= requires (run the suite or state "not run because..."). Closes the apparent contradiction with =verification.md= / =finish-branch=.
+
+*** 2026-05-22 Fri @ 14:06:41 -0500 Added private-vs-public CLAUDE.md citation modes to review-code
+
+Expanded the Content scope section in =review-code/SKILL.md= with two modes: a private/internal review cites =CLAUDE.md= directly; a public/team review translates the rule into the engineering reason it encodes and doesn't name the rules file (a teammate can act on the reason, not on a file they can't reach). Same principle =commits.md= states for personal tooling in public artifacts.
+
+*** 2026-05-22 Fri @ 13:48:14 -0500 Relaxed review-code "three strengths" to up-to-three-or-none
+
+Changed all three "three minimum" spots in =review-code/SKILL.md= (Strengths section, Critical Rules DO list, Anti-Patterns) to "up to three specific; say none found on a tiny or weak diff." Reframed the old "No Strengths section" anti-pattern as "Skipping strengths out of laziness" so a substantive diff still demands them while a weak one can honestly report nothing notable. Landed alongside Craig's adjacent edit telling reviewers not to explain why a strength is good (sycophantic padding).
+
+*** 2026-05-22 Fri @ 14:12:24 -0500 Removed review-process language from respond-to-review commit guidance
+
+Replaced the =fix: Address review — [description]= example (and the matching description-line phrasing) in =.claude/commands/respond-to-review.md= with "name the actual fix (=fix: validate export filename=), not the review that prompted it." Killed the non-ASCII dash and the process-in-commit pattern that conflicted with =commits.md=.
+
+*** 2026-05-22 Fri @ 14:12:24 -0500 Made respond-to-review fetch unresolved threads + resolve after verification
+
+Rewrote section 1 (Gather) in =.claude/commands/respond-to-review.md= to pull =reviewThreads= via =gh api graphql= with =isResolved=, skipping already-resolved threads so settled feedback isn't re-processed; top-level conversation comments still come from REST. Added a section-4 step: reply and resolve a thread only after the fix is verified, never before.
+
+*** 2026-05-22 Fri @ 14:12:24 -0500 Verified respond-to-cj-comments no longer embeds an absolute path (moot)
+
+Already resolved by a prior migration: =grep= for =/home/= and =/Users/= in =.claude/commands/respond-to-cj-comments.md= returns nothing. The public-writing section refers to the rules by name, not by local path. No edit needed.
+
+*** 2026-05-22 Fri @ 14:12:24 -0500 Closed respond-to-cj-comments humanizer/emacsclient fallback (largely moot)
+
+Overtaken by two later changes: =/humanizer= was replaced by =/voice personal= (no =/humanizer= invocation remains), and the mandatory =emacsclient= summary-open was replaced by the in-place VERIFY-task pattern (workflow line ~262, Craig's 2026-05-12 standing instruction). Only a stale descriptive phrase remained — tidied "humanizer's signs of AI writing" to "the signs of AI writing." The original fresh-environment-fallback concern no longer applies as written.
+
+*** 2026-05-22 Fri @ 14:51:37 -0500 Fixed finish-branch base-branch detection
+
+Rewrote Phase 2: resolve the base *branch name* in priority order (open PR's =baseRefName=, then =git symbolic-ref --short refs/remotes/origin/HEAD= stripped, then ask), and compute the merge-base *SHA* separately only where a commit range is needed. Made the branch-name-vs-merge-base distinction explicit, since the old command returned a SHA where a branch name was needed.
+
+*** 2026-05-22 Fri @ 14:51:37 -0500 Made finish-branch merge safer + worktree-aware
+
+Added pre-flight checks to Option 1 (Merge Locally): dirty-tree refusal with no auto-stash, protected-branch awareness, upstream-gated =git pull --ff-only=, and merge-commit-vs-rebase as a team-policy choice instead of a hardcoded =--no-ff=. Replaced the fragile =git worktree list | grep <branch>= detection with a =git rev-parse --git-dir= vs =--git-common-dir= comparison plus =git worktree list --porcelain= for the path.
+
+*** 2026-05-22 Fri @ 14:51:37 -0500 Added tool-availability + ceremony-scale paths to start-work
+
+Added a "Tool availability" section (graceful degradation when Linear MCP / =gh= / =/voice= / Playwright are missing — do what's available, surface what isn't, don't block) and a "Ceremony scale" section (trivial / small / standard tiers so a two-line fix skips ticket+branch+gates unless asked). The =humanizer= reference in the original item is moot — the file already uses =/voice= throughout.
+
+*** 2026-05-22 Fri @ 14:51:37 -0500 Resolved start-work claim-before-justify rollback risk
+
+Split the claim by tracker type: personal todo.org claims defer to after the Justify gate (a killed task needs no rollback), while team trackers (Linear/GitHub) still claim first to signal intent but record prior state (status, assignee, label) so the Phase 2 rollback restores exactly it. Updated the per-tracker rollback steps and the matching anti-pattern.
+
+*** 2026-05-22 Fri @ 14:28:41 -0500 Verified add-tests typescript-testing.md reference resolves (moot)
+
+Resolved since the audit: =languages/typescript/claude/rules/typescript-testing.md= now exists, and =add-tests/SKILL.md:68= references it by bare filename, the same way it references =python-testing.md= (both get copied into a project's =.claude/rules/=). The "missing file" premise no longer holds. No edit needed.
+
+*** 2026-05-22 Fri @ 14:28:41 -0500 Added a category-exception protocol to add-tests
+
+Added an exception note to step 7 (proposal) in =add-tests/SKILL.md=: pure adapters, generated code, tiny pass-through wrappers, and framework glue may skip a category that would only re-test the framework, but the skip must be stated and justified in the plan and the behavior covered at integration/E2E level — never a silent omission. Step 12 (write) now points back to "honor documented category exceptions."
+
+*** 2026-05-22 Fri @ 14:25:37 -0500 Added environment + recent-change capture to debug Phase 1
+
+Added a fourth Phase-1 step in =debug/SKILL.md=: record versions, feature-flag/config state, dataset/fixture, seed/clock, concurrency, and recent commits/config-infra changes. Noted that intermittent bugs usually live in environment/state transitions (and "what changed recently" is often the fastest route), while a deterministic local bug only needs a one-liner. Updated the phase's closing recap to include the context.
+
+*** 2026-05-22 Fri @ 14:25:37 -0500 Constrained root-cause-trace defense-in-depth to boundaries
+
+Rewrote step b in =root-cause-trace/SKILL.md=: instead of "add a check at each layer that could have caught it," add one only at a layer that owns a boundary or invariant — ingress/trust, persistence, invariant-owning service, final render. Added the explicit rule that a pass-through function owning neither shouldn't get a duplicate null check (validation spam). Recast the three example layers as the boundary types.
+
+*** 2026-05-22 Fri @ 14:25:37 -0500 Required evidence + counterfactual per why in five-whys
+
+Expanded step 2 in =five-whys/SKILL.md=: each link now owes an evidence field (a log/commit/metric/config you can point to) and a counterfactual check (remove this cause — does the symptom above plausibly not happen?). Framed the counterfactual as the main guard against monocausal storytelling, and updated the worked example to show both fields.
+
+*** 2026-05-22 Fri @ 15:51:59 -0500 Added timebox + fresh-sources rules to brainstorm
+
+Phase 1 gained a "Timebox the dialogue" rule (aim for the one-sentence restatement in ~5-8 questions, then move on and park the rest as open questions). Phase 2 gained "Ground high-stakes claims in fresh sources" (check load-bearing claims about markets/regulations/tools/vendors/APIs against a current source; mark unverified ones as assumptions). The design-doc skeleton gained an "## Assumptions" section that distinguishes researched facts (with source) from assumptions (to confirm before building).
+
+*** 2026-05-22 Fri @ 14:59:32 -0500 Made arch-decide examples timeless + required citations
+
+Dated the MongoDB multi-document-transaction example (scoped to 2024-01) with a backing reference, and added a "Cite, don't assert" Do: every concrete technical claim about a tool/version/platform carries a link, doc, version, or "checked YYYY-MM" date, or gets a domain-neutral placeholder — so unsourced "X can't do Y" doesn't rot into stale fact.
+
+*** 2026-05-22 Fri @ 14:59:32 -0500 Standardized arch-decide ADR statuses + immutability rule
+
+Declared a canonical five-status set (Proposed, Accepted, Rejected, Deprecated, Superseded) with an explicit "no synonyms" line, and spelled out the immutability rule in the Don'ts: an accepted ADR's body is frozen, only status/link metadata changes, a changed decision gets a new superseding ADR and the old one stays as the historical record.
+
+*** 2026-05-22 Fri @ 14:59:32 -0500 Added Trust/Data/Compliance phase to arch-design
+
+Added a new Phase 4 (Trust, Data, and Compliance) before the paradigm shortlist: trust boundaries, data classification, abuse/misuse cases, privacy constraints, compliance evidence, and operational ownership — surfaced early so the architecture is drawn around them, not retrofitted by a downstream =security-check=. Threaded into the workflow list, brief template (new §6), review checklist, and anti-patterns.
+
+*** 2026-05-22 Fri @ 14:59:32 -0500 Split paradigms from tactical patterns in arch-design
+
+Split Phase 5's single mixed table into Step 1 (pick one paradigm: monolith/microservices/layered/event-driven/serverless/pipeline/space-based) and Step 2 (compose tactical patterns: DDD, hexagonal, CQRS, event sourcing — several or none, often per-module), with composition examples and an anti-pattern against treating DDD/CQRS as alternatives to a paradigm. Recommendation + brief now name a paradigm plus composed patterns.
+
+*** 2026-05-22 Fri @ 14:59:32 -0500 Expanded arch-document quality scenarios to the Q42 six-part template
+
+Replaced §10's thin "Under [condition]..." template with the arc42/Q42 six-part structure (source, stimulus, environment, artifact, response, response measure), each glossed, with the cart-checkout example rewritten across all six parts. A one-line prose form stays acceptable once all six parts are recoverable.
+
+*** 2026-05-22 Fri @ 14:59:32 -0500 Added staleness/ownership metadata to arch-document output
+
+Added a per-section metadata block (owner, generated-against SHA + date, review cadence, "stale-when" conditions) as an HTML-comment header plus a visible Doc-status note, with field-fill guidance, and a whole-document Doc Status table replacing the README's "Last Updated" stub. Wired into the review checklist and an "Undated docs" anti-pattern.
+
+*** 2026-05-22 Fri @ 14:59:32 -0500 Added confidence levels to arch-evaluate findings
+
+Added a "Confidence and Provenance" subsection: every framework-agnostic finding carries High/Medium/Low + how it was determined, with a required "Not fully checked because..." note when scale, runtime imports, reflection, or dynamic dispatch cap certainty. Updated the example findings and review checklist; a finding with no note now asserts a full read.
+
+*** 2026-05-22 Fri @ 14:59:32 -0500 Made arch-evaluate report skipped tool checks explicitly
+
+Replaced "skip silently" with explicit reporting: for each detected language whose tool isn't configured or can't run, emit an Info "tool not configured / not run" finding (with an example) so the audit shows what was and wasn't verified. A check that didn't run no longer reads as a pass. Updated workflow step 4 and the review checklist.
+
+*** 2026-05-22 Fri @ 14:51:37 -0500 Added notation/output fallback to c4-analyze + c4-diagram
+
+Both commands now treat C4 as notation-independent: a "Choosing a notation" section (draw.io XML, Structurizr DSL, Mermaid with native C4 types, PlantUML/C4-PlantUML) and a headless fallback that emits a text notation (Mermaid or Structurizr DSL) and skips PNG-export/desktop-open when =drawio= or a GUI is absent, rather than failing. draw.io is now one option, not the only one.
+
+*** 2026-05-22 Fri @ 14:51:37 -0500 Clarified C4 abstraction boundaries in c4-analyze + c4-diagram
+
+Added an "Abstraction boundaries" section to both: a Container is a separately deployable/runnable unit (not synonymous with a Docker container — a SPA or managed DB counts), a Component lives inside one Container and isn't separately deployable. Added a 4e "Verify single abstraction level" check that walks every element and relationship to confirm it stays at the diagram's level, notation-independent.
+
+*** 2026-05-22 Fri @ 15:10:35 -0500 Added "When You Cannot Verify" standard to verification.md
+
+Added a section requiring, when a verification command can't run, a four-part report: command attempted, why it couldn't run, risk left unverified, and the smallest next command for the user. States the principle that a check that didn't run is never reported as a pass — "unable to verify" is a required honest outcome, not silence. Placed after Red Flags.
+
+*** 2026-05-22 Fri @ 15:10:35 -0500 Added property-based + mutation testing escalation to testing.md
+
+Added an "Escalation Beyond Category and Pairwise" section: property-based testing for invariants over a broad input domain (round-trips, idempotence, ordering — Hypothesis/fast-check/proptest) and mutation testing for when high line coverage hides thin assertions (mutmut/cosmic-ray/Stryker). Both framed as escalation paths to reach for on a gap, not gates on every unit.
+
+*** 2026-05-22 Fri @ 15:10:35 -0500 Added a disciplined spike protocol to testing.md
+
+Formalized the existing "I need to spike first" excuse-table row into a "Spike Exception (Disciplined)" subsection under TDD Discipline: TDD stays the default, but a spike is sanctioned when all three hold — timeboxed, spike code not committed, and the first failing test written before productionizing the discovered approach. Built on the existing row rather than contradicting it.
+
+*** 2026-05-22 Fri @ 15:10:35 -0500 Added pre-dispatch availability + cost checks to subagents.md
+
+Added a "Pre-Dispatch Checks" section with two gates: Availability (no Agent capability → do the work in the main thread under the same scope/constraints/output discipline the contract would enforce) and Cost (when writing the full contract costs more than the task, do it inline). Cross-references the existing "Don't Subagent At All" section and "Subagenting trivial work" anti-pattern rather than duplicating.
+
+*** 2026-05-22 Fri @ 15:06:04 -0500 Revised python-testing SQLite guidance toward production-like DBs
+
+Replaced "prefer in-memory SQLite for speed" with: run ORM/query tests against a production-like DB (same engine as prod, often containerized), since SQLite diverges from Postgres/MySQL on query semantics, constraints, transactions, JSON, time zones, and indexes (a test can pass on SQLite and fail in prod). SQLite stays only for pure unit tests with no DB-semantics dependency.
+
+*** 2026-05-22 Fri @ 15:06:04 -0500 Clarified python-testing ORM-mocking boundary
+
+Changed the "never mock" bullet from "ORM queries" to "ORM internals (querysets, sessions, model internals)" and added a paragraph: domain services use real model methods/validation, but a thin orchestration unit can inject a fake at a deliberate data-access port (a repository/interface the code owns). That's still mocking at a boundary, not at ORM internals.
+
+*** 2026-05-22 Fri @ 15:06:04 -0500 Made elisp.md editing advice tool-agnostic
+
+Rephrased the "prefer Write over repeated Edits" bullet around intent: land nontrivial Elisp as one cohesive change rather than dribbling it in over tiny partial edits (which accumulate paren mismatches), and run paren-balance + byte-compile checks immediately after, whatever editing mechanism the environment uses.
+
+*** 2026-05-22 Fri @ 15:06:04 -0500 Added batch-mode + native-comp caveats to elisp-testing.md
+
+Added three sections: Batch-Mode Reproducibility (=emacs --batch= as source of truth, no interactive-session state, no blocking prompts, deterministic), Isolating Emacs State (temp =user-emacs-directory=, explicit load-path, declared deps only, with an unwind-protect sandbox example), and Byte-Compile/Native-Comp Warnings (=byte-compile-error-on-warn=, native-comp gated on =native-comp-available-p= and kept opt-in/version-aware).
+
+*** 2026-05-22 Fri @ 15:16:22 -0500 Synced hooks/README install snippets with the destructive hook (opt-in)
+
+Brought the README's manual-install and settings-JSON snippets in line with the canonical =hooks/settings-snippet.json= (which already wires all three) and the Makefile's opt-in design: added the destructive-bash-confirm.py symlink as an opt-in step, added its settings entry, and reworded the note to say all three are no-op-safe but the destructive gate is opt-in (=make install-hooks= excludes it by default — link manually before relying on the snippet entry).
+
+*** 2026-05-22 Fri @ 15:35:06 -0500 Hooks now scan file-backed commit/PR messages
+
+Added =read_referenced_file()= to =_common.py= (safe local read: missing/oversize/non-UTF-8 → None) and wired it in: =git-commit-confirm.py= =extract_commit_message= now handles =-F=/=--file=/=--file===<path>= (reads + scans the file, falls through to UNPARSEABLE → asks if unreadable), and =gh-pr-create-confirm.py= reads =--body-file= content instead of a placeholder. Attribution scanning now sees the real committed/posted text. Built a pytest harness (=hooks/tests/=, importlib-by-path loader for the hyphen-named hooks) and wired =hooks/tests= into =make test=. 54 hook tests pass; full suite green.
+
+*** 2026-05-22 Fri @ 15:35:06 -0500 Rewrote destructive-bash rm parsing on shlex
+
+=detect_rm_rf= now tokenizes with =shlex.split= instead of a whitespace split, so quoted/spaced paths and combined/separate/reordered flags (=-rf=, =-r -f=, =-fr=, =--recursive=/=--force=) all parse. Fails toward asking — returns a sentinel that still fires the modal — on unbalanced quotes or when a forced recursive rm coexists with a compound/pipeline/substitution/redirect construct. Documented the supported/unsupported shell constructs in the docstrings, and extended the dangerous-path banner to =$HOME=-prefixed and wildcard targets. Covered by 25 new tests. (Pre-existing, out-of-scope: path-prefixed =rm= like =/bin/rm= still isn't matched.)
+** DONE [#B] Add =make remove= for interactive ruleset removal via fzf
+CLOSED: [2026-05-22 Fri]
+Shipped: =scripts/remove.sh= (three modes — =--list=, =--remove-selected= reading stdin, and the default fzf-multi interactive flow) + =make remove= target + =scripts/tests/remove.bats= (5 cases). Lists only symlinks resolving into the repo (foreign links left alone); rm's picked links while leaving repo sources untouched; reports-and-continues on a missing target; quiet no-op on empty selection. shellcheck clean, make test green. Dropped the stale =bridge= entry per the note below.
+
+Add a Makefile target that lists every currently-installed ruleset entry
+and lets me pick one or more to remove via fzf. Granular alternative to
+=make uninstall= (removes everything) and =make uninstall-hooks= (removes
+only hooks).
+
+*** Why this matters
+
+Tearing down a single skill, rule, hook, or config file currently means
+either running =make uninstall= and re-installing what I want to keep,
+or =rm=ing the symlink directly and remembering the exact path. Both are
+friction. An interactive picker lets me filter, multi-select with Tab,
+and confirm with Enter — the typical fzf flow. Costs about 3-5 seconds
+per teardown instead of 15+ seconds of "what's the exact name?".
+
+*** Design
+
+The recipe builds a tab-separated list of every currently-installed item,
+categorized by type, and pipes it to =fzf --multi=. The user filters,
+marks with Tab, and confirms with Enter. The recipe parses the selections
+and =rm=s the matching symlinks.
+
+#+begin_example
+ skill debug
+ rule commits.md
+ hook destructive-bash-confirm.py
+ config settings.json
+ commands commands
+ bridge claude-rules
+#+end_example
+
+Each line is =<kind>\t<name>=. The recipe maps =<kind>= to the right path:
+
+- =skill= → =$(SKILLS_DIR)/<name>=
+- =rule= → =$(RULES_DIR)/<name>=
+- =hook= → =$(HOOKS_DIR)/<name>=
+- =config= → =$(CLAUDE_DIR)/<name>=
+- =commands= → =$(CLAUDE_DIR)/commands=
+- =bridge= → =$(SKILLS_DIR)/claude-rules=
+
+Source files in =rulesets/= stay untouched. =make install= re-creates the
+removed links if needed (the install loop is idempotent).
+
+*** Edge cases
+
+- Esc instead of Enter → empty selection → clean exit, no removal.
+- Filter to nothing then Enter → same as Esc.
+- Selected item already gone → =rm= fails visibly, processing continues
+ on the rest.
+- =fzf= not installed → fail fast with a clear error (matches the pattern
+ used by =install-lang=).
+
+*** Possible extensions
+
+- Parallel =make pick-install= target that lists not-yet-installed items
+ and installs the chosen ones. Symmetric UX, same fzf flow.
+- Confirmation prompt when more than N items selected (defense against
+ accidental select-all).
+- =--source= flag that also runs =git rm= against the rulesets source for
+ the selected item. Probably bad idea — too easy to lose work.
+- The =bridge → $(SKILLS_DIR)/claude-rules= entry above is stale — the
+ bridge symlink got removed in a later commit. Drop that bullet when the
+ recipe lands.
+** DONE [#B] Document the =mcp/= install pipeline in =mcp/README.org=
+CLOSED: [2026-05-22 Fri]
+Wrote =mcp/README.org= covering everything in the "what to cover" list: the file layout (tracked vs gitignored), the secrets-bundle shape (plain =${VAR}= secrets + base64-bundled OAuth artifacts, AES256 symmetric =gpg -c=), the install flow (decrypt → materialize keys/token caches at mode 600 → expand → register unregistered, idempotent), the http/sse-vs-stdio transport split, token rotation when a Google refresh token is revoked, and adding a new server. Grounded in a read of the actual =install.py= + =servers.json=.
+
+=mcp/= has =install.py=, =servers.json=, =secrets.env.gpg=, =gcp-oauth.keys.json= (gitignored, regenerated at install). No README. Coming back to this in three months I'll re-discover how the bundle is structured, what =install.py= does, and how to rotate tokens. Saving that re-discovery is the whole point.
+
+*** What to cover
+
+- Layout: what each file is, which are tracked vs gitignored.
+- Secrets bundle shape: how vars are listed in =secrets.env=, the symmetric-encryption pattern (=gpg -c --cipher-algo AES256=), the base64-bundled OAuth artifacts (=GCP_OAUTH_KEYS_JSON_B64=, =GOOGLE_DOCS_PERSONAL_TOKEN_B64=, =GOOGLE_DOCS_WORK_TOKEN_B64=).
+- Install flow: =make install-mcp= → =install.py= decrypts, writes the keys file and Google Docs token caches at mode 600, expands =${VAR}= in =servers.json=, calls =claude mcp add --scope user= for unregistered servers. Idempotent.
+- Token rotation: when a refresh token gets revoked, the recovery flow (re-auth on one machine, re-bundle, recommit).
+- Adding a new server: edit =servers.json=, add any new =${VAR}= placeholders to the bundle, re-encrypt.
+- The OAuth dance for HTTP-transport servers (linear, notion) versus stdio (google-docs-*) — different paths, different gotchas.
+** DONE [#C] Add =make uninstall-mcp= + =mcp/install.py --check= for symmetry :feature:solo:quick:
+CLOSED: [2026-05-28 Thu]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-05-28
+:END:
+
+Currently the MCP install pipeline only flows one direction. No way to remove rulesets-managed MCP servers in one command. No way to ask "what's the drift between =servers.json= and =claude mcp list=" without eyeballing.
+
+*** =make uninstall-mcp=
+
+Iterate over =servers.json=, run =claude mcp remove <name> -s user= for each. Ignore "not registered" errors. Idempotent.
+
+*** =mcp/install.py --check=
+
+Dry-run mode. Decrypt secrets, but instead of registering, print the drift report:
+
+- Servers in =servers.json= not in =claude mcp list= → =MISSING=
+- Servers in =claude mcp list= not in =servers.json= → =EXTRA=
+- Servers in both → =ok=
+
+Useful for diagnosing connection failures and for the eventual =make doctor= integration.
+** DONE [#C] Update =README.org= with MCP install pipeline section :chore:solo:quick:
+CLOSED: [2026-05-28 Thu]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-05-28
+:END:
+
+=README.org= covers global install, per-project language bundles, and design principles, but doesn't mention =make install-mcp= or the =mcp/= directory. Add a short section after "Per-project language bundles" describing the user-scope MCP install pattern (decrypt → expand → register) and pointing at the eventual =mcp/README.org=.
+** DONE [#C] Consolidate =claude-templates/Makefile= after fold :chore:quick:solo:
+CLOSED: [2026-05-28 Thu]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-05-28
+:END:
+
+Sibling follow-up from the fold child (2026-05-15). After the subtree merge, =rulesets/claude-templates/Makefile= still has its standalone =install= / =uninstall= / =list= / =test-scripts= targets. The =install= target's =bin/ai= logic is now duplicated in =rulesets/Makefile=. Both work; the redundancy is harmless but worth cleaning up.
+
+Options:
+- *Delete* =claude-templates/Makefile= entirely — forces all install through rulesets root. Cleaner.
+- *Strip down* to just =test-scripts= — the one piece not redundant with =rulesets/Makefile=.
+- *Leave it* — slight redundancy, no functional harm.
+
+Triggered by: 2026-05-15 fold session's refactor audit (commit =2d645fc=).
+** DONE [#C] Run =--archive-done= sweep at start of =open-tasks.org= Phase A :chore:quick:solo:
+CLOSED: [2026-05-28 Thu]
+:PROPERTIES:
+:CREATED: [2026-05-28 Thu]
+:LAST_REVIEWED: 2026-05-28
+:END:
+
+From pearl handoff 2026-05-28. =open-tasks.org= Next Mode reads =* Project Open Work= and skips =* Project Resolved= correctly, but a level-2 task that completed during a session sits as =** DONE= under Open Work until something archives it. Between cleanups, a freshly-DONE task can surface as a "what's next" candidate.
+
+Proposed fix: as the first step of =open-tasks.org= Phase A, run =emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done todo.org=, then read =todo.org=. The cleanup tool already exists; this is wiring it into the workflow.
+
+Cost: a few hundred ms at the start of every "what's next" invocation. Win: recommendations never include DONE work.
+
+Optional refinement: gate behind a check for read-only / dry-run mode if that's ever introduced. The default invocation archives.
+** DONE [#C] Triage Codex enhancement backlog :spec:
+CLOSED: [2026-05-28 Thu]
+:PROPERTIES:
+:CREATED: [2026-05-28 Thu]
+:LAST_REVIEWED: 2026-05-28
+:END:
+
+Triaged interactively 2026-05-28. Disposition table for all 14 items lives at [[file:docs/design/2026-05-28-rulesets-enhancement-backlog.org][2026-05-28-rulesets-enhancement-backlog.org]] under "Triage Dispositions": 3 accepted (filed below as TODOs), 3 pilot/scope-limited (filed below), 2 marked as conventions rather than tracked tasks, 6 rejected with rationale. Items #1 and #2 already had homes (#16 and the Phase-1 codex TODO).
+** DONE [#C] Canonical/mirror drift detection via pre-commit hook or =make sync-check= :feature:quick:solo:
+CLOSED: [2026-05-28 Thu]
+:PROPERTIES:
+:CREATED: [2026-05-28 Thu]
+:LAST_REVIEWED: 2026-05-28
+:END:
+
+From the codex enhancement backlog (item #7), reframed: don't dedupe the dual source — the canonical-in-=claude-templates/= + mirror-in-=.ai/= pattern is a feature (other projects rsync from the canonical; the mirror lets rulesets-as-a-project have a working copy). The real pain is sync-discipline overhead — every workflow edit needs both copies updated, and forgetting one leaves the next startup's rsync to surface the drift.
+
+Scope: write a small =scripts/sync-check.sh= (or fold into the existing Makefile) that diffs =claude-templates/.ai/workflows/= against =.ai/workflows/=, exits non-zero on drift. Wire as a pre-commit hook (=githooks/pre-commit= or equivalent) so the discipline is enforced before publish, not at the next startup. =make sync-check= as a manual entry point.
+
+Verification: introduce a deliberate diff, commit, hook should block. Restore parity, hook should pass.
+** DONE [#C] Add =make status= — compose audit + doctor + open-task count :feature:quick:solo:
+CLOSED: [2026-05-28 Thu]
+:PROPERTIES:
+:CREATED: [2026-05-28 Thu]
+:LAST_REVIEWED: 2026-05-28
+:END:
+
+From the codex enhancement backlog (item #12), scope-limited: =make status= only. Reject the rest of #12 (=make sync= duplicates the existing sync flow; =make health= wraps existing checks without adding signal; =make bootstrap-project= duplicates =install-ai= + =install-lang=).
+
+Scope: one Makefile target that prints a compact summary of:
+
+- Install audit state (clean / drift, calling =make audit=).
+- Machine-global doctor state (calling =make doctor=).
+- Open-task count (top-level entries in =todo.org= under =* Rulesets Open Work=).
+- Inbox count (files in =inbox/= excluding =.gitkeep= and =PROCESSED-= prefixes).
+- Git working-tree status (clean / dirty, ahead/behind upstream).
+
+Output should be roughly 10 lines, scannable in one glance. Composes the existing checks; no new logic except the summary formatting.
+** DONE [#C] Iteration-history backfill for spec-review and spec-response :docs:followup:
+CLOSED: [2026-05-28 Thu]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-05-28
+:END:
+Source: org-drill inbox 2026-05-28.
+
+Once the in-flight WIP lands (the requirement that specs carry a bottom =Review and iteration history= section, with iteration / date / contributor / role / what / why / artifacts), backfill the two workflow files themselves using rulesets' session history as evidence.
+
+Files to update:
+- =claude-templates/.ai/workflows/spec-review.org=
+- =claude-templates/.ai/workflows/spec-response.org=
+
+Investigation: search =.ai/sessions/=, =.ai/notes.org=, inbox archive, and git log for mentions of these workflow docs. Identify review/response/design iterations, dates, and contributors (including agents where known: Claude Code, Codex, local models). Distinguish high-confidence history (commits, dated session entries) from inferred (chat-only context). Recommend whether enough evidence exists to populate the section, and draft the entries if so.
+
+Dependency: spec-review.org and spec-response.org have uncommitted edits in flight. Wait for those to land before writing to the files. The read-only research portion (search sessions, identify iterations, draft entries to a scratch file) can run in parallel without conflict.
+** DONE [#B] Startup Phase A rsync propagates dirty rulesets WIP into downstream projects :feature:
+CLOSED: [2026-05-30 Sat]
+:PROPERTIES:
+:CREATED: [2026-05-29 Fri]
+:LAST_REVIEWED: 2026-05-29
+:END:
+Fixed via option 1 (skip-when-dirty), scoped to the synced source paths: startup.org Phase A now guards the protocols/workflows/scripts rsyncs behind a =git status --porcelain= check on =claude-templates/.ai/{protocols.org,workflows/,scripts/}=, skipping the sync when any are dirty. The propagation anomaly (cross-project-broadcast.org / page-signal.org not reaching jr-estate) was a timeline artifact: both files were added in 664bf01 on 2026-05-29, after jr-estate's Phase A rsync had already run — correct behavior, not a bug.
+
+From jr-estate handoff 2026-05-29. When rulesets has uncommitted WIP at the moment a downstream project starts a session, Phase A.0 reports "dirty, skipping pull" and proceeds. Phase A's =rsync -a --delete= then runs against the dirty rulesets working tree and copies the WIP state into the downstream project's =.ai/workflows/= and =.ai/scripts/=. The downstream project's =git status= then shows drift the user did not author. Two bad recovery paths: commit the drift as "chore: sync .ai tooling from templates" (creates fake commit history about template state) or leave it dirty (noisy wrap-ups, pressure to commit anyway).
+
+Three options proposed in the handoff:
+1. *Skip-when-dirty.* Make Phase A's workflows/ and scripts/ rsync no-op when Phase A.0 reports rulesets dirty. Simplest defense.
+2. *Clean-files-only.* Restrict the rsync to files git considers unmodified in rulesets. Untracked files in rulesets do not propagate. Most precise.
+3. *Clean-ref-based.* Cache the last-known-clean state as a git tag or ref and rsync from that ref rather than the working tree. Most decoupled, also the most infrastructure.
+
+Recommendation (mine): option 1. The downstream impact of skipping a sync once is small (the next session with rulesets clean catches up), and the implementation is one =if [ "$dirty" -eq 0 ]= guard around the existing rsync block. Option 2 adds shellout complexity per file; option 3 requires tagging discipline that has no other reason to exist.
+
+The original handoff also noted a related anomaly: even with =--delete=, two files that DO exist in rulesets canonical (=cross-project-broadcast.org=, =page-signal.org=) did NOT propagate to jr-estate. Worth confirming whether that was a transient rsync issue or evidence of a deeper Phase A bug. Could be ordering: those files were added to rulesets AFTER the jr-estate Phase A rsync ran, in which case the behavior is correct and the report is misreading the timeline.
+
+Source: =inbox/2026-05-29-0832-from-jr-estate-investigate-startup-rsync-carried-dirty.org= (processed and deleted).
+** DONE [#B] Codex Phase 1 — AI_AGENT_ID + session-context.d/<id>.org :feature:
+CLOSED: [2026-05-30 Sat]
+:PROPERTIES:
+:CREATED: [2026-05-28 Thu]
+:LAST_REVIEWED: 2026-05-28
+:END:
+Shipped backward-compatibly. New =.ai/scripts/session-context-path= helper resolves the active path from =AI_AGENT_ID=: unset → the legacy =.ai/session-context.org= singleton (one-agent default unchanged, per the spec's compatibility rule), set → =.ai/session-context.d/<sanitized-id>.org=. startup.org's existence check and wrap-it-up.org's rename now resolve through the helper (with a singleton fallback for older checkouts); wrap folds the agent id into the archive name. protocols.org documents the rule. Verified: 5 bats cases + a two-agent simulation showing distinct paths per id. Larger runtime-neutral arc (runtimes/ manifests, launcher refactor) stays parked under the parent spec.
+
+Lifted from the broader codex runtime spec ([[file:docs/design/2026-05-28-generic-agent-runtime-spec.org]]) as the immediate-correctness slice independent of the larger arc. The singleton =.ai/session-context.org= is unsafe under simultaneous agents — two LLMs running in the same project at the same time would overwrite each other's session state.
+
+Scope: introduce an =AI_AGENT_ID= environment variable and split the single =session-context.org= into a per-agent =session-context.d/<id>.org= directory. No other phases of the runtime refactor are in this task — keep the surface small, fix the race, ship.
+
+Touches: =.ai/protocols.org= (rename rule + recovery anchor), =.ai/workflows/startup.org= (Phase A check), wrap-up workflow (rename target), per-project session record discoverability.
+
+Verification: simulate two agents sharing a project (separate AI_AGENT_ID values) and confirm session-context writes land in distinct files without interleaving.
+
+Parent: see [[Generic agent runtime support — Codex spec v0]] above for the larger arc this is sliced from.
+** DONE [#C] Decide on category-3 rule copies in the deepsat tree :chore:quick:solo:
+CLOSED: [2026-05-31 Sun]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-05-28
+:END:
+Diffed 2026-05-31. Both copies (coding-rulesets vendored + orchestration_dashboard_mvp) are byte-identical to each other and stale against canonical: =testing.md= 221 lines behind with 5 lines unique to the copies (older wording or a small team tweak), =verification.md= 40 behind with nothing unique. Same older vendored version in both spots. Left untouched per the A1 decision — team-owned, and canonicalizing would create a cross-repo dependency on the private rulesets (the orchestration_dashboard_mvp pair is team-visible from Vrezh's PR thread). No files modified.
+
+While symlinking personal-project =.claude/rules/= mirrors to the rulesets canonical on 2026-05-07, two locations didn't fit the "personal mirror → symlink" pattern and were left untouched pending judgment:
+
+- =~/projects/work/deepsat/code/coding-rulesets/claude-rules/{testing,verification}.md= — looks like a vendored team-shared copy.
+- =~/projects/work/deepsat/code/orchestration_dashboard_mvp/.claude/rules/{testing,verification}.md= — could be project-specific overrides.
+
+For each: read the file, diff against the rulesets canonical, decide whether it's an intentional diverge (leave alone), stale (sync content), or should canonicalize (replace with symlink and accept the cross-repo dependency). The orchestration_dashboard_mvp pair is the project where Vrezh's PR review surfaced this whole thread, so any decision there has team-visibility implications.
+
+Decision (Craig, 2026-05-31): *leave team-tree copies alone.* Personal rulesets does not reach into team repos — canonicalizing would create a cross-repo dependency on the private rulesets, and the orchestration_dashboard_mvp copy is team-visible. This makes the task solo: diff each copy against canonical, record whether it's identical / drifted / overridden in the disposition, and close as "left alone (team-owned)" without modifying the team-tree files.
+** DONE [#C] Audit language-specific rule files for cross-project duplication :chore:solo:
+CLOSED: [2026-05-31 Sun]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-05-28
+:END:
+Audited 2026-05-31. Findings: in sync with canonical (=languages/<lang>/claude/rules/=) — work =python-testing.md=, deepsat =typescript-testing.md=, =.emacs.d= =elisp-testing.md= + =elisp.md=. Drifted — =gloss= and =chime= (byte-identical to each other): =elisp-testing.md= 44 lines behind (canonical added Batch-Mode Reproducibility + Isolating Emacs State; zero lines unique to the copies), =elisp.md= one line behind (canonical expanded the edit-cohesively guidance). No project-specific additions anywhere — every copy is either current or purely stale.
+
+Disposition: *leave them project-local* (the task's own option). The language-rule copies in code projects are the bundle's deliberate copy-and-sync model, not the symlink pattern the generic rules (commits/testing/verification/subagents) use in personal doc-projects. =sync-language-bundle.sh= auto-fixes drifted bundle rules on each startup, so gloss/chime self-heal the moment those projects next boot — no canonicalize/symlink needed, and symlinking would fight the bundle model. Did not reach into work/deepsat/gloss/chime/.emacs.d from here (cross-project boundary; team copies left alone per the 2026-05-31 category-3 decision).
+
+The four canonical rules (=commits=, =testing=, =verification=, =subagents=) are now symlinked across the five personal-project mirrors as of 2026-05-07. But several language-specific rule files exist in multiple project mirrors and may be duplicated or drifted:
+
+- =python-testing.md= in =~/projects/work/.claude/rules/=
+- =typescript-testing.md= in =~/projects/work/deepsat/code/.claude/rules/=
+- =elisp-testing.md= and =elisp.md= in =~/.emacs.d/=, =~/code/gloss/=, =~/code/chime/=
+
+The Elisp pair is the most suspicious — three repos using essentially the same rules. Audit: diff these across the projects, check for drift, then decide whether to canonicalize them under =~/code/rulesets/claude-rules/languages/<lang>/= and symlink, or leave them as project-local.
+** DONE [#C] Refactor =daily-prep.org= to delegate to =triage-intake.org= for the triage section :chore:solo:
+CLOSED: [2026-05-31 Sun]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-05-28
+:END:
+Collapsed Phase 3's inline source scans (sub-steps 3b email / 3c mark-read / 3d Slack / 3e Linear / 3f PRs / 3g dedup, ~280 lines) into four: 3b runs the triage-intake engine, 3c surfaces today's reactive items as Day's Priorities thin links, 3d re-sorts by urgency, 3e writes the audit footer from the engine's coverage. Source coverage carries via the engine's Phase 0 two-dir glob (general + .ai/project-workflows/ plugins), so the work account's Gmail/Slack/Linear/GHE plugins still get scanned. Adapted the downstream refs (Prep Doc Structure rule, Heads-up FYI source, Recommended Approach Pattern reframed as engine-applied), removed the orphaned Linear-digest note, added a Living Document entry. Verified: workflow-integrity clean (no dangling script refs), sync-check clean, full suite green. daily-prep.org went 825 → 576 lines.
+
+=daily-prep.org= still does its own inline triage (Gmail × 3 accounts, Slack, Linear, GHE PRs, calendars) as part of the full prep flow. =triage-intake.org= is now a source-agnostic engine that loads =triage-intake.<source>.org= plugins (refactored 2026-05-26), so daily-prep could call the engine and consume its synthesis instead of duplicating the source-scan logic. That DRYs up a large workflow and keeps both flows in sync when sources change — a source change now lives in one plugin that both flows pick up.
+
+Scope:
+- Identify the sections in =daily-prep.org= that do the inline triage (the email / Slack / Linear / PR / calendar fan-out, plus the "Sources checked: ..." footer at the top of each generated prep doc).
+- Replace those sections with "run the =triage-intake.org= engine" and adapt the downstream sections (Heads-up, Day's Priorities, Carry-forwards) to read the engine's synthesis output rather than the inline scan results.
+- Verify the generated prep doc still has the same shape (Heads-up + Day's Priorities + Carry-forwards + Sources checked).
+- Reconcile source coverage: daily-prep's inline triage scans work accounts (3 Gmail, Slack, Linear, GHE PRs) that are project-specific plugins under =.ai/project-workflows/=, not general plugins. The delegation must ensure the engine loads those project plugins (Phase 0 globs both dirs) so nothing daily-prep currently scans drops out.
+
+Origin: came up while authoring =triage-intake.org= on 2026-05-11; body refreshed after the engine/plugin refactor on 2026-05-26.
+** DONE [#C] Templatize =make coverage-summary= into the language bundles (Elisp pilot) :feature:solo:
+CLOSED: [2026-05-31 Sun]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-05-28
+:END:
+
+Done 2026-05-31 (Elisp pilot, the scoped milestone): ported the kernel into the elisp bundle as a self-contained =languages/elisp/claude/scripts/coverage-summary.el= (no coverage-core dependency), proven end-to-end against the real dotemacs SimpleCov report (93 tracked, 27 untested modules surfaced, project number 66.4%). The missing-file-as-0% + unit-weighted number is the kernel. Delivery: the script ships under =.claude/scripts/= (gitignored, auto-fixed on drift by =sync-language-bundle.sh=); =languages/elisp/coverage-makefile.txt= holds the project-owned Makefile fragment, seeded at project root by =install-lang.sh= and dropped into =.ai/inbox/= by sync when that convention exists. Tests: 12 ERT (=languages/elisp/tests/=, wired into =make test=), 5 new sync bats, 2 new install-lang bats. The fan-out to Python/Go/TS is the follow-up below.
+
+Borrow dotemacs's =make coverage-summary= into the language bundles. After =make coverage= writes a coverage file, =coverage-summary= prints per-unit covered/total with percentages, a unit-weighted project number, and a list of source files present on disk but missing from the coverage report.
+
+*The kernel — the only part worth building.* Weight the project number by file/module rather than by line, and count a source file absent from the report as 0% instead of omitting it. A module no test imports just doesn't appear in coverage.py or nyc output, so it silently fails to drag the number down. That missing-file detection is the value; everything else (per-file table, total) the built-in reporters already print, so don't reimplement those.
+
+*Scope Elisp-first.* Port the proven dotemacs version into the elisp bundle, prove the pattern end-to-end, then fan out. Don't open all four bundles at once.
+
+*Delivery (settled 2026-05-25).* Two rulesets-owned pieces per language:
+- The summary *script* ships in the bundle under =.claude/= (inside the now-gitignored tooling footprint), copied in on install and auto-fixed on drift by =sync-language-bundle.sh=, never committed by the project.
+- One *text file per language* holding the Makefile fragment (the =coverage-summary= target plus its =coverage= prerequisite) and a block recommending how to set up coverage for that language. The bundle never edits the project's own Makefile.
+ - *New project:* install copies that file in for the project to own.
+ - *Existing project:* sync drops the fragment into the project's =inbox/= rather than touching its Makefile — the project adopts it deliberately.
+
+*Prerequisite caveat.* The summary presumes a coverage harness exists (undercover, coverage.py, nyc, =go cover=). Several bundles may have no =make coverage= yet, so for those this task implies adding the harness first — or the per-language file documents it as a prereq.
+
+Per-language parser (the script is ~40 lines over each tool's output):
+- Elisp: undercover SimpleCov JSON (=.coverage/simplecov.json=) — dotemacs/auto-dim scripts already parse this.
+- Go: =go test -coverprofile=cover.out=; parse =cover.out= (simple text), or lean on =go tool cover -func=.
+- Python: =coverage json= per-file JSON, or lean on =coverage report=.
+- TypeScript/JS: nyc/Istanbul =coverage-final.json= / json-summary.
+
+Reference (dotemacs): =scripts/coverage-summary.el=, =modules/coverage-core.el=, and the =coverage= / =coverage-summary= Makefile targets.
+
+Origin: handoff from the .emacs.d session, 2026-05-25.
+** DONE [#C] Fan out coverage-summary across all language bundles :feature:
+CLOSED: [2026-05-31 Sun]
+:PROPERTIES:
+:CREATED: [2026-05-31 Sun]
+:END:
+
+Done 2026-05-31: coverage-summary now ships in all four bundles. Elisp pilot, then Python, Go, and TypeScript. Each parses its tool's report (SimpleCov / coverage.py JSON / Go cover.out / Istanbul json-summary), counts on-disk source files absent from the report as 0%, and file-weights the project number. The plumbing proved generic: =install-lang.sh= seeds the project-owned =coverage-makefile.txt= and ships the script into the gitignored =.claude/scripts/=; =make test= discovers ERT (=test-*.el=), pytest (=test_*.py=), =go test= (=*_test.go=), and =node --test= (=*.test.js=) under =languages/*/tests/=, each guarded on its toolchain. TypeScript and Go scripts were dogfooded (Go against a live profile, TS against the CLI); Python and TS weren't run against a live coverage tool (coverage.py / nyc not installed) — proven against faithful fixtures matching each tool's stable schema.
+
+Remaining follow-ups (not blockers):
+- Go is a coverage-only slice — =languages/go/= has no rule file, so =sync-language-bundle.sh= can't fingerprint it and won't sync-maintain the script. Build out the real Go bundle (=go.md= / =go-testing.md= + =CLAUDE.md=) to close that.
+- First real adopters of the Python and TS scripts should sanity-check against a live =coverage json= / nyc =coverage-summary.json= run.
+
+Original notes retained below for the next person.
+
+The Elisp pilot proved the pattern; Python and Go followed. The plumbing is generic: =install-lang.sh= seeds the fragment, and =make test= now discovers ERT (=test-*.el=), pytest (=test_*.py=), and =go test= (=*_test.go=) under =languages/*/tests/=. TypeScript is the last one.
+
+- TypeScript/JS: nyc/Istanbul =coverage-final.json= / =coverage-summary.json=. Same kernel: file-weighted project number, on-disk =*.ts=/=*.js= absent from the report counted as 0%. nyc prints its own table, so the script focuses on the missing-file list and the number. Needs a vitest/jest (or =node --test=) discovery path in =make test=, mirroring the go-test block.
+
+Notes for the next person, from the Python + Go runs:
+- Python: parses coverage.py's =files[path].summary.{covered_lines,num_statements}= (stable since coverage 5.x), resolves report paths against the report's parent dir. Proven against a synthetic report, not a live =coverage json= run (coverage.py wasn't installed). Sanity-check against a real one.
+- Go: =languages/go/= is a coverage-only slice with no rule file, so =sync-language-bundle.sh= can't fingerprint it (detection keys on a bundle's own =.claude/rules/*.md=). The script is delivered by =make install-lang LANG=go= but is not sync-maintained until the Go bundle gets a real rule file + =CLAUDE.md=. Building out that bundle is the natural companion task. Also: modern =go test ./...= already lists every module package in the profile at 0%, so the missing-file list is usually empty for in-module code; it earns its keep on build-tagged files and dirs outside =./...=.
+** DONE [#C] Enumerate implementation tasks in =spec-review.org= Phase 6 :feature:solo:
+CLOSED: [2026-05-31 Sun]
+:PROPERTIES:
+:CREATED: [2026-05-28 Thu]
+:LAST_REVIEWED: 2026-05-28
+:END:
+Added a Phase 6 step that lifts the spec's =Implementation phases= into a drop-in =todo.org= block (one =[#B]= per phase + a test-surface entry mirroring =Acceptance criteria=); a spec lacking phase decomposition raises that as a finding instead. Added Exit Criterion 6 and a review-history entry. Pure workflow-doc change.
+
+From pearl handoff 2026-05-28. =spec-review.org= Phase 6 currently says "log deferred work to =todo.org=: v1 implementation = [#B] ... vNext/someday = [#D]." That covers deferred and v1 in passing but doesn't lift the spec's =Implementation phases= section into a drop-in =todo.org= block.
+
+Proposed addition to Phase 6: a structured step that reads the spec's =Implementation phases= section and produces a =[#B] TODO= entry per phase (subject line, tags, one-line body, pointer back to spec), plus a final entry for the test surface (unit / integration / e2e / manual-verify mirroring the spec's =Acceptance criteria= when present). Emit under a new section "Implementation tasks (drop-in for todo.org)" in the review file. Format follows =todo-format.md= (terse heading, body holds context, tags on heading).
+
+Three wins: handoff is one paste not a re-read; forces specs to be implementable in pieces (a spec without a phase decomposition fails this step, surfacing the shape problem); closes the loop on =Acceptance criteria= as manual-verify entries.
+
+If the spec lacks an =Implementation phases= section, the step is the prompt to ask the author to add one before =Ready=.
+** DONE [#C] Add =.aiignore= for agent inventory exclusions :chore:solo:
+CLOSED: [2026-05-31 Sun]
+:PROPERTIES:
+:CREATED: [2026-05-28 Thu]
+:LAST_REVIEWED: 2026-05-28
+:END:
+Shipped a gitignore-syntax =.aiignore= at the rulesets root (deps, build output, language caches, editor cruft, token artifacts, lockfiles-as-agent-read-skip) and documented the convention + defaults + lockfile policy in protocols.org ("Recursive Reads"). Per Craig's scope call (2026-05-31): did NOT wire audit.sh / diff-lang.sh / sync-language-bundle.sh — they do targeted finds over .ai/.claude/bundle dirs, never naive whole-tree walks, so honoring .aiignore there would be dead code. Script-side honoring belongs in a future catalog/inventory tool if one ships; the real consumer today is agent recursive reads (the protocols guidance).
+
+From the codex enhancement backlog (item #8). Filesystem scans by agents and helper scripts pick up =node_modules=, =__pycache__=, =.pytest_cache=, lockfiles, generated OAuth artifacts, and test caches, even when those are gitignored. Token waste during exploration and skewed project summaries.
+
+Scope: add a shared =.aiignore= file (or =rulesets-ignore.json= if a more structured format helps) listing default exclusions. Teach the scripts that walk the project (=audit.sh=, =diff-lang.sh=, =sync-language-bundle.sh=, future =catalog= work if any) to honor it. Document in =protocols.org= so agents know to consult it before naive recursive reads.
+
+Keep the lockfile policy explicit: ignored when a local skill dependency cache, tracked when reproducibility matters.
+** DONE [#C] Workflow test harness — drift + integrity tests :feature:solo:
+CLOSED: [2026-05-31 Sun]
+:PROPERTIES:
+:CREATED: [2026-05-28 Thu]
+:LAST_REVIEWED: 2026-05-28
+:END:
+
+From the codex enhancement backlog (item #10). Startup's drift check catches index-vs-directory mismatches but not deeper integrity: a workflow that references a script that's been renamed, a plugin whose parent engine has been deleted, a required section missing from a newly-added workflow.
+
+Scope: add =scripts/tests/workflow-integrity.bats= (or pytest equivalent) verifying:
+
+- Every =.org= file in =.ai/workflows/= is either indexed in =INDEX.org= or classifiable as a source plugin under an indexed engine.
+- Every indexed workflow file actually exists.
+- Every =file:= or shell-command reference inside a workflow to a script under =.ai/scripts/= or =scripts/= resolves to an existing file.
+- Every source plugin maps to a parent workflow that exists and is indexed.
+- Required sections (Overview, When to Use, the workflow's main phases) are present in each workflow.
+- Workflow trigger phrases are unique enough to route — no two workflows claim the same exact trigger.
+
+Wire into =make test=. Run on the canonical =claude-templates/.ai/workflows/= as the source of truth.
+** DONE [#C] Token-tier pilot on largest workflows :feature:solo:
+CLOSED: [2026-05-31 Sun]
+:PROPERTIES:
+:CREATED: [2026-05-28 Thu]
+:LAST_REVIEWED: 2026-05-28
+:END:
+
+Done 2026-05-31: restructured both =startup.org= and =triage-intake.org= into the four-lane structure (Summary / Execution / Reference / History), preserving every existing instruction. triage-intake's reorder ran through a content-preservation guard (the multiset of content lines is unchanged; only heading depth and lane grouping moved). workflow-integrity, sync-check, and the full test suite pass.
+
+From the codex enhancement backlog (item #5), scope-limited to a pilot rather than a universal template change.
+
+Apply a standardized section structure to the largest workflow files first — =startup.org= and =triage-intake.org= are the prime candidates. Sections:
+
+- *Summary* / *Quick Contract* — one-screen purpose and outputs.
+- *Execution* — the steps an agent must follow.
+- *Reference* — examples, edge cases, rationale, old decisions.
+- *History* / *Design Notes* — durable context not needed every run.
+
+Decision (Craig, 2026-05-31): *approved the four-lane structure (Summary/Execution/Reference/History) and the scope — restructure both =startup.org= and =triage-intake.org= now.* Makes the task solo: apply the lanes to both, preserving every existing instruction (reorganize, don't rewrite), verify the workflows still read coherently and the drift/integrity checks pass.
+
+Teach startup/routing to read =Summary= only at routing time, then =Execution= only for the selected workflow. Other sections become opt-in.
+
+After the pilot, evaluate: did the savings show up in real session token use? Did the structure constrain the workflow expressiveness too much? If yes to savings and no to constraint, expand to the next-largest workflows. If not, document why and stop. Don't templatize universally — shorter workflows don't need tiering.
+** DONE [#B] Add Signal MCP server (rymurr/signal-mcp) :feature:
+CLOSED: [2026-06-02 Tue]
+:PROPERTIES:
+:CREATED: [2026-05-29 Fri]
+:LAST_REVIEWED: 2026-05-29
+:END:
+Done 2026-06-02. Registered signal-cli to the Google Voice pager account, added the signal-mcp entry to servers.json, installed via make install-mcp (claude mcp list shows it connected), and documented the signal-cli + GV dependency in mcp/README.org. The GV-registration dependency this task flagged is resolved. Shipped in cfaff12 (page-signal routing) and this commit (README).
+
+Install [[https://github.com/rymurr/signal-mcp][rymurr/signal-mcp]] so Claude can call =send_message_to_user=, =send_message_to_group=, and =receive_message= natively rather than shelling out to the =page-signal= wrapper. Python, MCP framework, depends on =signal-cli= being configured locally.
+
+Two-way capability is the differentiator over the CLI: =receive_message= lets the agent listen for replies on the phone, enabling page-as-confirm flows, "should I proceed?" loops over Signal, and structured Q&A across devices.
+
+*** Dependency
+
+This depends on the Google Voice account being registered with =signal-cli= first. Sending from Craig's primary number to itself doesn't notify (Signal treats it as one account on linked devices). The MCP server takes =--user-id= at startup, one account per instance, so it has to point at the GV account, with the primary as the per-send recipient.
+
+If GV registration is still pending when this task runs, block here and surface that.
+
+*** Implementation
+
+- =mcp/servers.json= — add =signal-mcp= entry under stdio transport (=command=, =args=, optional =env= for the user-id pointer).
+- =mcp/README.org= — document the signal-cli + GV-registration dependency and the user-id pattern.
+- =mcp/secrets.env.gpg= — only if the MCP server's user-id needs to be encrypted (probably not; the GV number isn't a secret beyond being personal).
+- Verify: =make install-mcp= followed by =make check-mcp= shows =signal-mcp ok=; smoke-test via a Claude tool call sending a message + waiting on =receive_message=.
+
+*** Why this matters
+
+=page-signal= is the fast path (a hook, a script, a make recipe can call it without an MCP round-trip). The MCP server is the smart path. When Claude wants to send and then *react to the reply*, the CLI can't do that — only the MCP server can. The two complement each other; this task adds the second half.
+** DONE [#C] task-review pass at end of task-audit :chore:solo:
+CLOSED: [2026-06-02 Tue]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-06-02
+:END:
+Have the =task-audit= workflow chain a =task-review= pass as its final phase, so a freshly-audited list also gets the lighter staleness/honesty sweep without a second invocation. The legend already notes the division of labor — task-audit assigns and refreshes tags, task-review keeps them honest in passing — so running task-review at the tail of task-audit closes the loop in one pass. Edit =claude-templates/.ai/workflows/task-audit.org= (and the synced mirror) to add the final phase; check whether =open-tasks.org= already invokes task-review so the chaining stays consistent.
+** DONE [#C] lint-followups drift — reconcile-on-write + audit dead-link reaping :feature:solo:
+CLOSED: [2026-06-02 Tue]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-06-02
+:END:
+From an .emacs.d handoff (2026-06-02): running task-audit against a large todo.org proved several =.ai/lint-followups.org= entries stale (four dead-link flags pointed at docs that now exist; three near-duplicate dated lint runs had piled up). Two fixes, scoped separately.
+
+1. =lint-org= workflow/script (the real fix): reconcile-on-write. Before appending a run, drop entries whose finding no longer reproduces (dead link now resolves, flagged block/timestamp now clean) and dedupe against the prior run instead of re-logging. Key entries by content/finding rather than line number, so they survive edits to the target file (line numbers go stale immediately).
+2. =task-audit.org= (small, narrow): in the Phase C link-hygiene step, when fixing/verifying a =file:= link, also reap any matching dead-link entry in the project's lint-followups file so the two artifacts don't drift. Scope explicitly to dead-link entries — do NOT pull general lint cleanup into the audit; that mixes two concerns and slows the audit.
+** DONE [#C] start-work Justify gate: explicit "reasons not to do this" item :feature:quick:solo:
+CLOSED: [2026-06-02 Tue]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-06-02
+:END:
+From a work handoff (2026-06-02, surfaced running /start-work on a clean low-risk refactor). The Phase 2 Justify gate has "Downsides" and "Alternatives considered" but no forced devil's-advocate verdict on "should we even do this?" Add a "top reasons not to do this" item: surface the top three objections if any exist; when none rise to a real objection, state one line instead of manufacturing three (e.g. "Nothing material argues against this; no reason to defer or drop it"). Building the case against the work before committing is cheapest exactly at this gate, which is its purpose. Edit the start-work skill's Justify-gate phase.
+** DONE [#C] start-work Approach gate: spec-needed check :feature:quick:solo:
+CLOSED: [2026-06-02 Tue]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-06-02
+:END:
+From Craig (2026-06-02). The Approach phase should consider whether the work needs a spec when one doesn't already exist. For a big task, this isn't a silent skip — the pre-confirmation summary must explicitly report why a spec isn't needed, so the decision is visible and challengeable at the gate rather than assumed. Small tasks can pass without comment. Edit the start-work skill's Approach-gate phase to add the spec-needed consideration and the big-task report-why-not requirement.
+** DONE [#B] Cross-project pattern catalog :spec:thinking:
+CLOSED: [2026-06-05 Fri]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-06-02
+:END:
+
+From pearl handoffs [[file:docs/design/2026-05-27-pattern-catalog-pearl-notes.org][2026-05-27]] + [[file:docs/design/2026-05-28-pattern-catalog-no-empty-input.org][2026-05-28 follow-up]].
+
+Meta-question: how do good patterns travel from project A to project B? Pearl shipped three worked examples worth capturing — one-prompt picker with typed prefix (pearl-pick-source), magit-transient state buttons, and "no empty input as meaningful" (none-sentinel as first candidate). Each is a small principle with wide surface area; without a catalog, every project re-derives them from scratch.
+
+Open design questions before any implementation:
+- Catalog format — structured (one pattern per file with frontmatter) vs free-form doc
+- Surfacing mechanism — agent-driven (model spots opportunity) vs human-driven (Craig grep-searches)
+- Anti-patterns included or only what worked
+- Intake cadence — every time one lands, or batch review
+- Home — rulesets repo (agent visibility) vs Linear doc vs per-project cross-links
+
+Pearl recommends a one-page spec (problem + design + open questions + acceptance) before implementation. Pearl available to come back for spec-review iterations.
+
+*** 2026-05-28 Thu @ 08:12:55 -0500 Pearl shipped patterns 4-6, filed alongside the prior two
+Three more pearl handoffs landed and were filed during this audit. Filed: [[file:docs/design/2026-05-28-pattern-catalog-prompt-labels-and-defaults.org][prompt-labels-and-defaults]] (patterns 4-5: label-matches-behavior, default-most-common with friction-proportional-to-consequence) and [[file:docs/design/2026-05-28-pattern-catalog-prompt-collapse.org][prompt-collapse]] (pattern 6: collapse N orthogonal prompts into one enriched prompt). The catalog's evidence base is now four pearl notes in =docs/design/= covering six patterns plus the synthesizing principle Pearl articulated — "choices on screen, accurately labeled, ordered by what the user most often wants, friction sized to the cost of being wrong."
+
+*** 2026-06-05 Fri @ 00:47:59 -0500 Spec approved as written — all 5 decisions + 3 open questions accepted
+Craig approved the spec ([[file:docs/design/2026-06-02-pattern-catalog-spec.org][2026-06-02-pattern-catalog-spec.org]]) as written. Confirmed: one file per pattern with frontmatter; home =patterns/= in rulesets; thin =claude-rules/patterns.md= pointer, agent-driven; anti-patterns as a per-pattern field; capture-on-landing/promote-on-review intake. Open questions resolved to the spec's leans: directory name =patterns/=; concrete-now, generalize-on-second-use; manual promote flow first, no =/pattern= skill yet. Built as =.org= files with =#+KEYWORD= frontmatter (Craig's call over the initial =.md= draft); the =claude-rules/patterns.md= pointer stays =.md= since the rules layer and the Makefile glob require it.
+
+*** 2026-06-05 Fri @ 00:47:59 -0500 Built the catalog — 6 seed patterns + pointer + README
+Created =patterns/= with the six seed patterns (one-prompt-picker-typed-prefix, transient-state-buttons, no-empty-input-as-meaningful, label-matches-behavior, default-most-common-friction-proportional, collapse-orthogonal-prompts), each carrying the frontmatter contract (name/principle/problem/tags/source/examples) plus Problem/Do/Anti-pattern/Applicability/Related sections. =patterns/README.org= states the root principle, the frontmatter contract, and the intake cadence. =claude-rules/patterns.md= is the agent-facing pointer, auto-installed via the Makefile RULES glob. Sourced from the four pearl notes in =docs/design/=.
+** CANCELLED [#C] Try Skill Seekers on a real DeepSat docs-briefing need :chore:
+CLOSED: [2026-06-10 Wed]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-05-28
+:END:
+
+=Skill Seekers= ([[https://github.com/yusufkaraaslan/Skill_Seekers]]) is a Python
+CLI + MCP server that ingests 18 source types (docs sites, PDFs, GitHub
+repos, YouTube videos, Confluence, Notion, OpenAPI specs, etc.) and
+exports to 20+ AI targets including Claude skills. MIT licensed, 12.9k
+stars, active as of 2026-04-12.
+
+*Evaluated: 2026-04-19 — not adopted for rulesets.* Generates
+*reference-style* skills (encyclopedic dumps of scraped source material),
+not *operational* skills (opinionated how-we-do-things content). Doesn't
+fit the rulesets curation pattern.
+
+*Next-trigger experiment (this TODO):* the next time a DeepSat task needs
+Claude briefed deeply on a specific library, API, or docs site — try:
+#+begin_src bash
+pip install skill-seekers
+skill-seekers create <url> --target claude
+#+end_src
+Measure output quality vs hand-curated briefing. If usable, consider
+installing as a persistent tool. If output is bloated / under-structured,
+discard and stick with hand briefing.
+
+*Candidate first experiments (pick one from an actual need, don't invent):*
+- A Django ORM reference skill scoped to the version DeepSat pins
+- An OpenAPI-to-skill conversion for a partner-vendor API
+- A React hooks reference skill for the frontend team's current patterns
+- A specific AWS service's docs (e.g. GovCloud-flavored)
+
+*Patterns worth borrowing into rulesets even without adopting the tool:*
+- Enhancement-via-agent pipeline (scrape raw → LLM pass → structured
+ SKILL.md). Applicable if we ever build internal-docs-to-skill tooling.
+- Multi-target export abstraction (one knowledge extraction → many output
+ formats). Clean design for any future multi-AI-tool workflow.
+
+*Concerns to verify on actual use:*
+- =LICENSE= has an unfilled =[Your Name/Username]= placeholder (MIT is
+ unambiguous, but sloppy for a 12k-star project)
+- Default branch is =development=, not =main= — pin with care
+- Heavy commercialization signals (website at skillseekersweb.com,
+ Trendshift promo, branded badges) — license might shift later; watch
+- Companion =skill-seekers-configs= community repo has only 8 stars
+ despite main's 12.9k — ecosystem thinner than headline adoption
+** DONE [#C] Promote meeting-prep to a template workflow :feature:solo:
+CLOSED: [2026-06-10 Wed]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-06-10
+:END:
+meeting-prep lives in the work project's =project-workflows/= and is general-purpose — it builds a per-meeting prep doc — but its body carries project-specific references: =deepsat/assets/= transcript paths, Linear as the tracker, =knowledge.org=. Promoting to =claude-templates= means generalizing those to project-neutral terms (the project's transcript home, the project's tracker), adding it plus its =meeting-prep.pre-wire.org= supporting doc to the =.ai/= mirror and INDEX.org, and a workflow-integrity pass. Once promoted, the daily-prep 5-Day Look-Ahead's conditional "where the project has one" reference can become a direct link.
+
+Out of the 2026-06-10 daily-prep handoff from the work project.
+** DONE [#C] Build Craig's writing voice profile from real corpora :spec:
+CLOSED: [2026-06-10 Wed]
+:PROPERTIES:
+:CREATED: [2026-05-29 Fri]
+:LAST_REVIEWED: 2026-05-29
+:END:
+Shipped across 2026-05-29 → 2026-06-10. =voice/references/voice-profile.org= is the canonical paired file: Phases 1-2 corpora measured (commit bodies 128k words + email/PR/review registers), all 45 patterns carry entries with basis and history, and every reconciliation delta landed in =voice/SKILL.md= (#13/#33 self-discipline reframing, #7 soft flag, new corpus-derived #43-#45). Extension corpora (Slack, long-form, syntactic fragment detection) deliberately not pursued.
+
+Build a grounded profile of Craig's actual writing voice by mining the corpora he's produced over time. The =voice/SKILL.md= patterns today are observation-derived (em-dash zero-tolerance, semicolon → period, contractions kept, sentence-fragment rewrite, felt-experience cut, etc.). Some are spot-on; others are intuition. A real corpus pass would tell us which patterns are genuinely Craig's voice and which were guesses, plus surface idioms, sentence structures, and vocabulary the current ruleset misses.
+
+*** Sources to mine
+
+- *Email* — sent folders across all three accounts (=gmail=, =dmail/DeepSat=, =cmail/Proton=). Filter to Craig-authored (not forwards or replies-just-quoting). Separate work voice (=dmail=) from personal voice (=gmail=, =cmail=) since they're likely distinct registers.
+- *Commit messages* — =git log --author= across his repos. Captures terse-imperative voice.
+- *PR descriptions and review comments* — same corpora. More deliberate prose than commits.
+- *Org files he authored* — =notes.org=, todo bodies he typed, design docs in =docs/design/=, journal entries. Heavier on first-person voice than emails.
+- *Slack/messages* — DeepSat work slack, family group, friends. Casual register.
+- *Long-form artifacts* — résumé, proposals, white papers, blog posts (if any).
+
+Skip session-context files, which are Claude-co-written and would muddy the signal.
+
+*** Output
+
+- =voice/references/voice-profile.org= (or =.md=) — the canonical reference doc:
+ - Vocabulary tendencies (preferred verbs, avoided cliché classes, technical-vs-plain word choice).
+ - Sentence structures (typical length, conjunction patterns, parenthetical use).
+ - Punctuation patterns (em-dash actual frequency, semicolon vs period split, contraction rate).
+ - Register markers (signs of formal vs casual mode, work vs personal).
+ - Idioms and recurring phrasings.
+ - "Anti-patterns" — phrasings Craig consistently avoids that show up in AI-generated prose.
+- Updated =voice/SKILL.md= patterns grounded in evidence rather than intuition. Patterns that the corpus confirms get strengthened; patterns the corpus contradicts get rewritten or removed.
+
+Each finding should cite at least two evidence samples from the corpora so the basis for a rule is reviewable.
+
+*** Approach
+
+Phase 1 (corpus assembly) — pull the relevant slices: sent-mail dumps, =git log --author --no-merges --pretty=format:'%B'=, =gh pr list --author= bodies, org-file extracts. Strip headers, replies-quoted blocks, signatures. Land in =voice/corpus/= (gitignored if the project's =.ai/= is gitignored, tracked if private repo with private remote).
+
+Phase 2 (analysis) — pass over the corpus with focused queries: distribution of em-dashes per 1000 words, semicolon count, contraction frequency by register, sentence-length histogram, top-N adjectives/adverbs, etc. Subagent dispatch fits here.
+
+Phase 3 (draft profile) — write =voice-profile.org= with findings + evidence. Surface contradictions with the current ruleset.
+
+Phase 4 (reconcile with voice/SKILL.md) — present the deltas to Craig. Each delta is one of: confirm existing rule with evidence, strengthen rule, weaken rule, add new pattern, remove unsupported pattern. Apply approved deltas.
+
+*** Privacy
+
+Email and Slack content is private. The corpus must NOT enter any commit unless rulesets stays on the private cjennings.net remote (which it does today). If a future move to a public remote is on the table, the corpus and any direct quotes have to go before that happens. The profile doc itself can stay (it's analysis, not raw content), but cite by pattern not by verbatim quote.
+
+*** Why this matters
+
+The voice skill earns its place when Craig sees the rewrite and recognizes it as his own voice rather than a "clean" AI voice that approximates him. Today the skill catches common AI tells (em-dashes, semicolons, the felt-experience tic), which is useful. Corpus-grounding would make it catch the absence of *Craig-specific positive traits* — the phrasings he actually reaches for — not just the AI traits he doesn't.
+
+Likely improves =/voice personal= output quality on PR bodies, commit messages, and email drafts. Compound interest over the long run.
+** DONE [#C] Wide org-table handling — helper/lint/standard :spec:
+CLOSED: [2026-06-11 Thu]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-06-11
+:END:
+The org-table standard keeps project-doc tables <=120 cols with multi-line wrapped cells and a rule between rows, but nothing enforces it and hand-wrapping a wide cell into multi-row form is tedious and error-prone. Decide among: (a) a helper that auto-wraps a wide table into multi-row cells at a target width, (b) a lint check that flags tables over the width budget, (c) tighten the written standard with a worked before/after example. Likely some combination. A worked before/after example exists in a work-project prep doc (a 6-col table reformatted by hand to a 4-col multi-row-cell version), to be reproduced generically when this lands.
+
+Out of a work-project handoff 2026-06-09.
+
+Resolution 2026-06-11: all three shipped. (c) The standard, generalized from the work project's notes.org local copy, is now claude-rules/org-tables.md (globally loaded; render-width semantics — links measure at their visible label, never split a link) with the worked wrapped-table example. (a) .ai/scripts/wrap-org-table.el reflows tables mechanically: render-width measurement, link-atomic tokenizing, column shrink-to-floor allocation, continuation rows, rules between logical rows; idempotent (rule-delimited continuation groups merge back before re-wrapping); 23 ERT tests. (b) lint-org.el gained an org-table-standard judgment check (width overruns, missing rules; conformant wrapped tables not false-flagged); 5 new ERT tests, 32 total. Verified end-to-end on a demo file: 150-col table reflowed to budget, idempotent second pass, lint clean on the result.
+** DONE [#C] SessionStart-on-clear hook for auto-resume :feature:
+CLOSED: [2026-06-11 Thu]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-06-11
+:END:
+Add a SessionStart hook (matcher: clear) in settings.json that auto-injects "read .ai/session-context.org and resume if present, else run startup.org". Today /flush prompts the user to /clear and the next session relies on the model re-reading session-context; the hook makes resume automatic on /clear. Keep full startup.org for genuine fresh starts (new day, other machine, been away). Likely lands as claude-templates workflow notes plus the hook in settings.json.
+
+The checkpoint+resume halves already shipped as /flush. This is the remaining automation piece. Out of a work-project handoff 2026-06-09 (process tooling, belongs in rulesets not the work project).
+
+Resolution 2026-06-11: the hook itself had already shipped 2026-06-02 (hooks/session-clear-resume.sh + the SessionStart clear entry in the tracked settings.json — this task duplicated it). What was actually broken: make install didn't cover hooks, so the symlink never reached machines that hadn't run make install-hooks by hand, and the hook errored silently on every /clear. Fixed by folding default-hook linking into make install (startup's Phase A.0 now propagates hooks machine-wide), with bats coverage in scripts/tests/install-hooks-link.bats. Both hook branches verified on ratio; the live /clear fire is a one-keystroke manual test.
+*** TODO Manual testing and validation :test:
+**** /clear mid-session resumes from the anchor
+What we're verifying: the SessionStart(clear) hook fires and the fresh context resumes instead of cold-starting.
+- In any project session with a live .ai/session-context.org (this rulesets session qualifies), type /clear
+- Send any short message (the injected context loads but the model waits for your next keystroke)
+Expected: the reply starts with "flushed." on its own line, restates the Active Goal and immediate Next Step, and does NOT run the startup workflow.
+** DONE New personal projects are home regroupings — no mechanism needed
+CLOSED: [2026-06-12 Fri]
+Craig's call (2026-06-12): new personal projects will live in home, and there's no project-creation mechanism to build — he'll be working in home and simply decide to group some things differently. Nothing to do.
+
+Concurrence, verified: no template doc directs new personal work into ~/projects (first-session.org, install-ai.sh, and the README carry no such guidance; the only ~/projects references are discovery-root scans, which home and work still need). The situation as it stands: a new personal "project" is an area dir plus tasks inside home's existing =.ai/= machinery, no bootstrap step; =first-session.org= remains the bootstrap for standalone code projects in ~/code, unchanged and correct; "launch finances"-style trigger phrases for folded names degrade politely to the no-match candidate list, worth work only if real friction shows up.
+** DONE [#C] Build =/update-skills= skill for keeping forks in sync with upstream :feature:
+CLOSED: [2026-06-11 Thu]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-06-10
+:END:
+
+The rulesets repo has a growing set of forks (=arch-decide= from
+wshobson/agents, =playwright-js= from lackeyjb/playwright-skill, =playwright-py=
+from anthropics/skills/webapp-testing). Over time, upstream releases fixes,
+new templates, or scope expansions that we'd want to pull in without losing
+our local modifications. A skill should handle this deliberately rather than
+by manual re-cloning.
+
+Shipped 2026-06-11: [[file:.claude/commands/update-skills.md][/update-skills command]] + [[file:scripts/update-skills.py][helper script]] (17 bats tests) + three bootstrapped manifests under [[file:upstreams/][upstreams/]]. The first real upstream drift will exercise the interactive per-file/per-hunk flow end to end; the merge mechanics are covered by the test suite.
+
+*** 2026-06-11 Thu @ 17:05:28 -0500 Specification written as the shipped artifacts
+The command doc ([[file:.claude/commands/update-skills.md][update-skills.md]]) carries the user-facing spec: discovery, classification statuses, the per-file confirmation and per-hunk conflict flow, mark-synced semantics, and the missing-baseline fallback. The script's module docstring specifies the manifest schema. Two deviations from the 2026-05-16 design, with reasons: manifests live centrally at =upstreams/<name>/= instead of per-skill =.skill-upstream= dotfile dirs (arch-decide became two flat files in =commands/= and can't carry one — a =files= rename map covers it); baselines were seeded from the 2026-06-11 upstream HEADs since the true fork-point commits are unrecoverable, so pre-existing local modifications classify as =local-only= going forward.
+
+*** 2026-05-16 Sat @ 01:14:20 -0500 original goals and decisions
+**** Design decisions (agreed)
+
+- *Upstream tracking:* per-fork manifest =.skill-upstream= (YAML or JSON):
+ - =url= (GitHub URL)
+ - =ref= (branch or tag)
+ - =subpath= (path inside the upstream repo when it's a monorepo)
+ - =last_synced_commit= (updated on successful sync)
+- *Local modifications:* 3-way merge. Requires a pristine baseline snapshot of
+ the upstream-at-time-of-fork. Store under =.skill-upstream/baseline/= or
+ similar; committed to the rulesets repo so the merge base is reproducible.
+- *Apply changes:* skill edits files directly with per-file confirmation.
+- *Conflict policy:* per-hunk prompt inside the skill. When a 3-way merge
+ produces a conflict, the skill walks each conflicting hunk and asks Craig:
+ keep-local / take-upstream / both / skip. Editor-independent; works on
+ machines where Emacs isn't available. Fallback when baseline is missing
+ or corrupt (can't run 3-way merge): write =.local=, =.upstream=,
+ =.baseline= files side-by-side and surface as manual review.
+
+**** V1 Scope
+
+- [ ] Skill at =~/code/rulesets/update-skills/=
+- [ ] Discovery: scan sibling skill dirs for =.skill-upstream= manifests
+- [ ] Helper script (bash or python) to:
+ - Clone each upstream at =ref= shallowly into =/tmp/=
+ - Compare current skill state vs latest upstream vs stored baseline
+ - Classify each file: =unchanged= / =upstream-only= / =local-only= / =both-changed=
+ - For =both-changed=: run =git merge-file --stdout <local> <baseline> <upstream>=;
+ if clean, write result directly; if conflicts, parse the conflict-marker
+ output and feed each hunk into the per-hunk prompt loop
+- [ ] Per-hunk prompt loop:
+ - Show base / local / upstream side-by-side for each conflicting hunk
+ - Ask: keep-local / take-upstream / both (concatenate) / skip (leave marker)
+ - Assemble resolved hunks into the final file content
+- [ ] Per-fork summary output with file-level classification table
+- [ ] Per-file confirmation flow (yes / no / show-diff) BEFORE per-hunk loop
+- [ ] On successful sync: update =last_synced_commit= in the manifest
+- [ ] =--dry-run= to preview without writing
+
+**** V2+ (deferred)
+
+- [ ] Track upstream *releases* (tags) not just branches, so skill can propose
+ "upgrade from v1.2 to v1.3" with release notes pulled in
+- [ ] Generate patch files as an alternative apply method (for users who prefer
+ =git apply= / =patch= over in-place edits)
+- [ ] Non-interactive mode (=--non-interactive= / CI): skip conflict resolution,
+ emit side-by-side files for later manual review
+- [ ] Auto-run on a schedule via Claude Code background agent
+- [ ] Summary of aggregate upstream activity across all forks (which forks have
+ upstream changes waiting, which don't)
+- [ ] Optional editor integration: on machines with Emacs, offer
+ =M-x smerge-ediff= as an alternate path for users who prefer ediff over
+ per-hunk prompts
+
+**** Initial forks to enumerate (for manifest bootstrap)
+
+- [ ] =arch-decide= → =wshobson/agents= :: =plugins/documentation-generation/skills/architecture-decision-records= :: MIT
+- [ ] =playwright-js= → =lackeyjb/playwright-skill= :: =skills/playwright-skill= :: MIT
+- [ ] =playwright-py= → =anthropics/skills= :: =skills/webapp-testing= :: Apache-2.0
+
+**** Open questions
+
+- [ ] What happens when upstream *renames* a file we fork? Skill would see
+ "file gone from upstream, still present locally" — drop, keep, or prompt?
+- [ ] What happens when upstream splits into multiple forks (e.g., a plugin
+ reshuffles its structure)? Probably out of scope for v1; manual migration.
+- [ ] Rate-limit / offline mode: if GitHub is unreachable, should skill fail
+ or degrade gracefully? Likely degrade; print warning per fork.
+** DONE [#C] Monthly session-harvest workflow :feature:
+CLOSED: [2026-06-11 Thu]
+:PROPERTIES:
+:CREATED: [2026-06-11 Thu]
+:LAST_REVIEWED: 2026-06-11
+:END:
+A monthly pass over recent =.ai/sessions/= summaries across projects proposing promotion candidates: patterns for the catalog, durable facts for the KB, rule refinements, workflow learnings. Sibling cadence to the roam-hygiene timer; a workflow run on schedule, not a standing agent. From the 2026-06-11 insights report's "Canonical-Aware Knowledge & Workflow Curator" — the capture/promote machinery exists (pattern catalog, /codify, KB); this adds the mining cadence.
+
+Shipped 2026-06-11 as [[file:.ai/workflows/session-harvest.org][session-harvest.org]] (template + INDEX entry): five phases, four promotion lanes, /codify-grade gates + work-confidentiality scrub, =:LAST_HARVEST:= marker in notes.org, and the KB receipt-line metrics readout for the ~2026-07-10 checkpoint. Window filter reads session-filename date prefixes (mtime proved unreliable in a live test). First run due ~2026-07-11.
+** CANCELLED [#B] todo-cleanup.el per-area Open Work / Resolved pairs :feature:
+CLOSED: [2026-06-11 Thu]
+=--archive-done= assumes exactly one level-1 "Open Work" and one "Resolved" heading per todo.org. Home's consolidated file briefly carried per-area pairs and the pass skipped. Filed from home's 2026-06-11 addendum, then held the same evening when Craig flagged that he expected a single pair.
+
+Cancelled 2026-06-11: Craig confirmed the decision — one todo queue with a single Open Work / Resolved pair. Home reshapes its consolidated file to that form, and the existing single-pair tooling works unmodified. No code change needed.
+** CANCELLED [#D] todo-cleanup =--archive-done= reports 0 moves while moving subtrees :bug:
+CLOSED: [2026-06-12 Fri]
+:PROPERTIES:
+:CREATED: [2026-06-12 Fri]
+:END:
+Observed at the 2026-06-12 wrap: the pass relocated closed subtrees from Open Work to Resolved while printing "todo-cleanup --archive-done: 0 subtree(s) moved".
+
+CANCELLED 2026-06-12 — cannot reproduce. =todo-cleanup.el= is unchanged since the wrap that logged this, and =tc-archived= is incremented inline with each move and read straight in the report, so no move can go uncounted. Running the exact pre-archive state (=b6d286f:todo.org=) through the tool reports the right count (3 moved, all listed). The "0 moved" was a correct second-run report: =open-tasks.org= Phase A runs =--archive-done= after wrap-it-up already archived, so the second pass finds nothing to move and prints 0 next to the first pass's git diff. Not a code defect.
+** DONE [#C] Session title hostname-project, no space :feature:quick:
+CLOSED: [2026-06-13 Sat]
+:PROPERTIES:
+:CREATED: [2026-06-13 Sat]
+:LAST_REVIEWED: 2026-06-13
+:END:
+Routed from the roam global inbox via inbox-zero 2026-06-13. The SessionStart hook (=hooks/session-title.sh=) emitted =<host> <project>= with a space; Craig wanted =<host>-<project>= with a hyphen and no space. Changed the =sessionTitle= join to ="$host-$project"= plus the header comments, and updated the three =session-title-hook.bats= expectations (test-first; 6/6 green).
+** DONE [#B] ~/.dotfiles discovery added to ai launcher; bootstrapped on velox
+CLOSED: [2026-06-20 Sat]
+Craig reported =~/.dotfiles= missing from the launcher picker. Two root causes, both fixed: (1) applied the parked one-liner =maybe_add_candidate "$HOME/.dotfiles"= in =build_candidates()= (=claude-templates/bin/ai=, after the =~/.emacs.d= line); (2) =~/.dotfiles/.ai/= was absent on velox — the 2026-06-16 bootstrap was on another machine and =.ai/= is gitignored, so it never traveled — re-bootstrapped via =install-ai.sh --gitignore ~/.dotfiles=.
+
+Verified end-to-end: =build_candidates()= now lists =~/.dotfiles= (protocols.org guard passes). sync-check clean (bin/ai is single-canonical, no mirror). By-name launch =ai ~/.dotfiles= already worked via single_mode's marker-only check. working/ai-dotfiles-discovery/ staging dir removed.
+** DONE Phase E spec'd — folded into the autonomous-batch spec
+CLOSED: [2026-06-16 Tue]
+:PROPERTIES:
+:CREATED: [2026-06-16 Tue]
+:END:
+Craig's answer (2026-06-16): spec it. Phase E reconciles with the "fix speedrun" proposal into one feature — see [[file:docs/design/2026-06-16-autonomous-batch-execution-spec.org][the autonomous-batch execution spec]]: a dedicated =work-the-backlog.org= holds the execution loop, inbox-zero keeps its A-D routing, and "fix speedrun" is a thin preset over the same loop. The prepared Phase E change stays under [[file:working/inbox-zero-phase-e/]] as a source. Tracked from here under the "fix speedrun" / autonomous-batch task below, where the spec-review VERIFY lives.
+** DONE [#C] Encourage org-roam KB contribution across workflows :feature:
+CLOSED: [2026-06-20 Sat]
+:PROPERTIES:
+:CREATED: [2026-06-16 Tue]
+:END:
+From the roam global inbox (Craig, 2026-06-16). Encourage agents to keep durable, strategic knowledge in the org-roam KB so it compounds into a cross-project asset:
+- Curate a best-practices node (good note-taking + org-roam practices, drawing on established advice) and link it from =startup.org= with encouragement to contribute through the session.
+- Add a reminder at the end of =triage-intake.org= and =inbox-zero.org= to store strategic / durable / useful info in the KB.
+- Add an early =wrap-it-up.org= prompt asking the agent what it learned worth remembering, then to write it to the KB before proceeding.
+Touches four synced template workflows and needs a curation pass on the best-practices content, so it's a design task — not a loop auto-implement. Filed from a =:next:=-tagged roam item; the eligibility tag was dropped on filing because the work needs a design decision (see the loop guardrail). Pairs with [[file:claude-rules/knowledge-base.md]] and the agent-knowledge-base spec.
+
+*** 2026-06-16 Tue @ 00:53:36 -0500 Spec written for review
+Drafted [[file:docs/design/2026-06-16-encourage-kb-contribution-spec.org][the KB-contribution spec]]: four light workflow prompts (startup nudge, triage-intake + inbox-zero end-of-flow reminders, an early wrap-up reflection feeding the existing KB receipt) plus one Craig-authored best-practices node curated from Ahrens / Matuschak / org-roam guidance. Five open sub-decisions filed as decisions-as-TODO in the spec.
+*** 2026-06-20 Sat @ 23:29:10 -0400 Spec ratified + built
+Craig ratified all five decisions (2026-06-20) and added D6 — a read-side startup consult-nudge surfacing project-relevant KB node titles, the counterpart the original write-only design lacked. Built all of it: the best-practices node (=~/org/roam/agents/20260620232112-agent-kb-best-practices.org=), startup's two Phase C nudges (consult + contribute, gated on the roam clone), the conditional capture reminders in triage-intake + inbox-zero, and the early wrap-up reflection feeding the existing receipt. Commits 76e5559 (workflows + spec) and the related lint checker f6dde4e. Trigger for the build: receipt data showed "promoted 0 / consulted no" across recent sessions.
+** DONE [#C] Bash/shell language bundle :feature:
+CLOSED: [2026-06-23 Tue]
+:PROPERTIES:
+:CREATED: [2026-06-23 Tue]
+:END:
+Built =languages/bash/= the same session it was filed: bash.md + bash-testing.md rules, a shellcheck PostToolUse validate hook (covers =.sh=, =.bash=, and extensionless shell scripts by shebang; 8 bats tests), a shellcheck pre-commit githook, settings.json wiring, gitignore-add.txt, and a "Bash/shell project" CLAUDE.md. shfmt left out of the blocking path on purpose (shell has no canonical style). Makefile test target now discovers =languages/*/tests/*.bats=.
+
+No =languages/= bundle fits a shell-heavy project. archangel (437 =.sh= files) and archsetup are bash projects with nothing that matches; installing elisp/python gives them the wrong language rules. Build a =languages/bash/= bundle on the elisp/go pattern: =claude/rules/bash.md= (style — =set -euo pipefail=, quoting, =[[ ]]=, trap/cleanup) + =bash-testing.md= (bats conventions), a PostToolUse validate hook (=shellcheck= on edited =.sh=), a =githooks/pre-commit= running shellcheck on staged shell files, =settings.json= wiring, =gitignore-add.txt=, and its own =CLAUDE.md= headed "Bash/shell project." Urgency dropped 2026-06-23: install-lang now seeds the language-neutral default CLAUDE.md when a bundle ships none, so a bash project no longer gets a mislabeled "Elisp project" header — the bundle is now the accurate-rules win, not a mislabel fix. From archangel 2026-06-23 ([[file:docs/design/2026-06-23-install-lang-claude-md-gap.org][handoff]]).
+** DONE [#B] Consolidate inbox/triage workflows + scheduled inbox check :chore:
+CLOSED: [2026-06-23 Tue]
+:PROPERTIES:
+:CREATED: [2026-06-23 Tue]
+:END:
+Built per the Ready spec: =process-inbox= + =monitor-inbox= + =inbox-zero= merged into one =inbox.org= engine (shared core + process/monitor/roam modes + the interactive =auto inbox zero= =/loop= mode); =triage-intake= and =no-approvals= stay separate. Callers repointed (INDEX, protocols, startup Phase C, wrap-up Step 3), old files deleted, stale-ref grep clean, workflow-integrity + sync-check + full suite green. The fully-unattended =/schedule= cron pass is vNext (see the =[#D]= task above). [[file:docs/specs/inbox-workflow-consolidation-spec.org][spec]].
+** DONE [#C] inbox-zero: delete empty roam entries on triage :feature:
+CLOSED: [2026-06-23 Tue]
+:PROPERTIES:
+:CREATED: [2026-06-23 Tue]
+:END:
+Done in commit 3da2725 (empty-entry sweep folded into Phase D's reconcile, after capture-guard + pull, with the claimed-item removal) and carried into the consolidated =inbox.org= roam mode (Phase B =empty= bucket + Phase D sweep). From the roam inbox 2026-06-23.
+** DONE [#C] Surface cross-project dependencies first in what's-next :feature:spec:
+CLOSED: [2026-06-24 Wed]
+:PROPERTIES:
+:CREATED: [2026-06-24 Wed]
+:END:
+Tasks that depend on another project can sit for ages when the dependency is low-priority or needs its own spec process — e.g. wrap-teardown depends on =.emacs.d= for the =ai-term= companion. Craig's proposal (roam 2026-06-24): (1) an org-tag marking a task as blocked-by / depends-on another project (pick a short tag name); (2) several ways to bind dependencies into the what's-next (=open-tasks.org=) decision tree so blocked-by-dependency tasks surface first; (3) review the what's-next workflow as a whole, since many projects use it.
+
+Built 2026-06-24 (tag name =:blocked:=, Craig's pick): the =:blocked:= tag + =:BLOCKED_BY: <project>: <what>= property convention in =todo-format.md=, and =open-tasks.org= Next Mode now excludes =:blocked:= tasks from the cascade and surfaces them in a dedicated "Blocked on other projects" section with an =inbox-send= nudge offer. Applied live to the wrap-teardown task above. Commits feat(tasks) cross-project-dependency.
+** DONE [#C] Task-audit: consolidate adjacent / related tasks :feature:
+CLOSED: [2026-06-24 Wed]
+:PROPERTIES:
+:CREATED: [2026-06-24 Wed]
+:END:
+The task-audit workflow should also consider combining related tasks when they're adjacent, so a spread-out effort reads as one whole. Craig's example (roam 2026-06-24): the agent-agnostic / agent-source work could collapse into one item, or at least a parent task with the related ones as children.
+
+Built 2026-06-24: =task-audit.org= Phase C.5 reads the open-task set, spots semantic clusters by judgment, and proposes per cluster either a merge (same-work members fold into one) or a parent-with-children grouping (related-but-distinct), applied only on Craig's confirm — broader than Phase C's exact-duplicate fold. Commit feat(task-audit) consolidate.
+** DONE [#B] Anki deck name from #+TITLE :bug:quick:solo:
+CLOSED: [2026-06-24 Wed]
+:PROPERTIES:
+:CREATED: [2026-06-22 Mon]
+:LAST_REVIEWED: 2026-06-24
+:END:
+flashcard-to-anki.py's =default_deck_name= returns =input_path.stem= (the filename), so every deck built through =flashcard-sync= (which passes no =--deck=) is named after the file, not the curated =#+TITLE=. =flashcard-review.org= already documents the intended behavior ("the #+TITLE line drives the Anki deck name"); the script never matched it. Fix: =default_deck_name(input_path, org_text)= scans for a =#+TITLE:= line (case-insensitive, trimmed) and returns it, basename fallback when absent; =main()= passes the already-read =org_text=. Edited script + test ready (validated, 29 pass); the staging =.py= files were removed after the fix landed (see below), rationale kept: [[file:docs/design/2026-06-21-anki-titlefix-proposal.org][proposal]]. Apply to both =.ai/scripts/= and =claude-templates/.ai/scripts/=, sync-check + make test. Migration caveat: deck ID derives from the name, so decks previously built without =--deck= land as new decks on next import (old basename-named decks keep history, delete by hand). Coordinate with "Reconcile flashcard multi-tag tooling into canonical" below — both edit =flashcard-to-anki.py=, build together to avoid conflicting edits. Shared-asset, review-gated. From home 2026-06-21.
+
+Done 2026-06-24 (commit 060a938): applied the pre-staged script + test red-to-green (5 new =#+TITLE= tests, 29 pass total), synced both script dirs, full suite green. The two redundant staging =.py= files removed, the rationale proposal kept.
+** CANCELLED [#C] Morning ops orchestrator pilot — read-only :feature:
+CLOSED: [2026-06-24 Wed 05:46]
+:PROPERTIES:
+:CREATED: [2026-06-11 Thu]
+:LAST_REVIEWED: 2026-06-15
+:END:
+A scheduled headless morning run chaining the existing pieces: startup checks, the triage-intake scan, a system health check — producing the prep doc plus a report and a notify ping, with all remediation propose-only. Staged adoption from the 2026-06-11 insights report's "Self-Healing Daily Ops Orchestrator": read-only first; promote individual routine remediations to auto only after each has a track record. Known blockers to design around: headless MCP auth (interactively-authenticated servers are absent in cron runs) and the consent boundary (triage Phase D, anything destructive).
+
+The triage limb can reuse triage-intake's *auto mode* (added 2026-06-15, see [[file:.ai/workflows/triage-intake.org]]) — its accumulate-don't-mutate sweep is the propose-only behavior this orchestrator wants. Auto mode itself runs in-session (inherited MCP auth); the orchestrator is the durable headless schedule, so the headless-auth blocker above is the part still on this task to solve.
+** DONE [#B] wrap-it-up teardown + "wrap it up and shutdown" :feature:
+CLOSED: [2026-07-01 Wed]
+:PROPERTIES:
+:CREATED: [2026-06-23 Tue]
+:LAST_REVIEWED: 2026-06-24
+:END:
+Two additions to =wrap-it-up.org=, designed by Craig (home, 2026-06-23). Item 1: bare "wrap it up" also tears down the session — kill the =aiv-<proj>= tmux session (takes =claude= with it), kill the vterm buffer, restore geometry. Teardown is the default; "wrap it up with summary" wraps without teardown (keeps the buffer readable). Must fire from a Stop/SessionEnd hook via a sentinel file, decoupled and last, so the valediction flushes before the session dies, and strictly after commit+push is verified. Item 2: "wrap it up and shutdown" → wrap, then a hard blocking gate (abort unless this is the only live =aiv-*= session), then an abort-able 10→1 countdown, then =sudo shutdown now=. Countdown can't run through the Bash tool (stdout buffers — prints all ten at once); needs a detached tty writer or an Emacs =run-at-time= timer. Companion: =cj/ai-term-quit= (and optional =cj/ai-term-live-count=) must live in =.emacs.d/modules/ai-term.el= — route there via inbox-send when building so both sides land together. Open decisions for Craig first: qualifier wording ("with summary" vs "and summarize"), countdown home (tty script vs Emacs timer), session-count mechanism (=tmux ls= / =pgrep claude= / helper). Shared-asset, review-gated. Proposal: [[file:docs/design/2026-06-23-wrap-teardown-shutdown-proposal.org][proposal]]. From home 2026-06-23.
+
+*** 2026-06-23 Tue @ 23:31:59 -0400 Built the rulesets side; companion routed to .emacs.d
+Craig's three decisions (2026-06-23): non-destructive qualifier is *both* "with summary" and "and summarize"; countdown is an Emacs =run-at-time= timer; the gate uses =cj/ai-term-live-count=. Built and pushed: =hooks/ai-wrap-teardown.sh= (Stop hook, sentinel-gated, 8 bats tests green), =hooks/settings-snippet.json= Stop wiring, =wrap-it-up.org= Teardown-mode section + Step 6 + checklist, INDEX trigger update. Architecture: both teardown and shutdown fire from the Stop hook via a basename-keyed sentinel (=/tmp/ai-wrap-teardown-<proj>=, =/tmp/ai-wrap-shutdown-<proj>=) dropped only at the end of Step 6 after commit+push, so the valediction flushes first. No bin script — the gate/countdown/teardown are all =emacsclient= calls into the companion. Companion spec (=cj/ai-term-quit=, =cj/ai-term-live-count=, =cj/ai-term-shutdown-countdown=) routed to .emacs.d via inbox-send. Remaining: .emacs.d lands the three functions, then the manual end-to-end validation below. Task stays DOING until both sides verify.
+
+*** 2026-06-23 Tue @ 23:40 .emacs.d received + filed the companion spec
+Per a roam-inbox FYI from .emacs.d (2026-06-23 23:38): both ai-term handoffs (multi-LLM support + this wrap-teardown companion spec) landed and are filed as .emacs.d tasks. The teardown one is flagged for its own focused session to land alongside the rulesets half. Part (c) is now in progress on the .emacs.d side.
+
+*** 2026-06-24 Wed @ 06:51:13 -0400 Unblocked — .emacs.d companion landed; feature now live
+The three companion functions are in =.emacs.d/modules/ai-term.el= (=cj/ai-term-quit= 1068, =cj/ai-term-live-count= 1087, =cj/ai-term-shutdown-countdown= 1109), matching the contract — double-checked the bodies: quit kills session+buffer+restores layout idempotently, live-count returns the gate integer, shutdown-countdown re-checks the gate (TOCTOU guard), runs an abort-able =run-at-time= countdown (C-g cancels), then a configurable =cj/ai-term-shutdown-command=. 13 ERT tests, headless-verified live (.emacs.d FYI 2026-06-24 06:44). Dropped =:blocked:= / =:BLOCKED_BY:= — the build dependency is resolved; only the manual end-to-end validation below remains. NOTE: with the Stop hook wired and the companion present, the feature is now functional — the next bare "wrap it up" will actually tear the session down. Run the validation below before relying on it.
+
+*** 2026-07-01 Wed @ 21:52:15 -0400 Plumbing pre-flight re-verified; only the eyes-on tests remain
+Fresh pre-flight, all green: Stop hook block in =~/.claude/settings.json= points at =~/.claude/hooks/ai-wrap-teardown.sh=, the symlink resolves to the rulesets canonical, no stale =/tmp/ai-wrap-teardown-*= sentinel, and all three companion functions are live in the daemon (=(t t t)=). Three =aiv-*= sessions live right now (=_emacs_d=, =archsetup=, =rulesets=), so the shutdown-gate refusal test has its multi-session condition available. Everything left in the checklist below needs Craig's eyes on a scratch session: buffer teardown + geometry restore, the qualifier opt-outs, the countdown render + C-g, and the push-failure guard.
+
+*** 2026-07-01 Wed @ 21:59:43 -0400 Manual end-to-end validation passed — all five tests, Craig's live run
+Craig ran the full checklist in a live Emacs/tmux ai-term setup: (1) bare "wrap it up" tore down after the valediction rendered, geometry restored, no lingering sentinel; (2) "with summary" / "and summarize" both wrapped without teardown, buffer stayed readable; (3) "wrap it up and shutdown" with another aiv-* session live refused the shutdown, named the other session, fell back to a normal wrap; (4) as the sole session, the 10→1 echo-area countdown rendered one-per-second, C-g cancelled cleanly, and a full run fired the (stubbed) shutdown command; (5) with the push made to fail, the wrap stopped at the failure and no sentinel was dropped. Works great — feature validated and live. Both sides complete: rulesets Stop hook + wrap-it-up Teardown mode, .emacs.d companion functions.
+** DONE [#C] Guard against hardcoded host identity in synced files :feature:solo:
+CLOSED: [2026-07-02 Thu]
+:PROPERTIES:
+:CREATED: [2026-06-22 Mon]
+:LAST_REVIEWED: 2026-06-24
+:END:
+A =CLAUDE.md= / notes file that asserts mutable environment identity as a fixed fact ("This machine is ratio", a current OS, an IP, "the laptop") is false on every machine the synced/tracked file lands on but one. It bit a real archsetup session: a stale "this machine is ratio" line made the agent reason backwards all session while on velox. Proposal: a claude-rule — don't assert mutable host/env identity as a fixed fact in a tracked/synced project file; derive it at runtime and name the command (=uname -n= for host; the =hostname= binary is often absent). Optionally a codify- or startup-time lint flagging "this machine is <name>" / "the current host is" style claims. Proposal: [[file:docs/design/2026-06-21-host-identity-guard-proposal.org][proposal]]. From archsetup 2026-06-21.
+
+2026-07-02 Thu @ 05:09:58 -0400 — Craig (speedrun pre-flight): rule + startup lint. A new claude-rules file plus a cheap grep probe in startup flagging host-identity claims in CLAUDE.md / notes.org fleet-wide.
+
+Resolution 2026-07-02: claude-rules/host-identity.md written (fixed-identity claims banned in tracked/synced docs, runtime derivation via uname -n, fleet-description carve-out, the archsetup worked failure) and linked machine-wide by make install. startup.org gained Phase A probe 13 (grep for "this machine/host/box is" claims in CLAUDE.md + notes.org, fixture-verified bash+zsh) and the Phase C host-identity flag line. Flags for judgment, never blocks.
+** DONE [#C] No-approvals speedrun — cross-project autonomous-batch mode :feature:spec:
+CLOSED: [2026-07-02 Thu]
+:PROPERTIES:
+:CREATED: [2026-06-15 Mon]
+:LAST_REVIEWED: 2026-06-24
+:SPEC_ID: 90f623cd-fdbe-4f5c-b63d-b2f84d9151cf
+:END:
+A named mode for coding projects: Craig names an ordered task set and says "speedrun" / "no approvals speedrun"; the set is worked autonomously, each task held to the full quality bar (TDD red→green, =/review-code=, =/voice= on the commit) and committed + pushed as its own logical commit, with all needed quick decisions gathered in one pre-flight Q&A (answer or "skip this") and a VERIFY filed for anything underspecified or needing deliberation, plus an end-of-set page listing completed + remaining + skipped tasks. Task size is not a gate — large tasks decompose into per-commit chunks. Surfaced by .emacs.d from a 2026-06-15 theme-studio session where the shape worked. Source proposal: [[file:docs/design/2026-06-15-fix-speedrun-workflow-proposal.org]] (.emacs.d handoff 2026-06-15). Build via =spec-create= when worked; we handle the task in priority order.
+
+Skeptical-review read (open design questions to resolve in the spec, not settled here):
+- *Is it a new workflow or a documented preset?* The proposal frames it as no-approvals + always-push session modes plus an end page. Decide whether it needs its own workflow file or is mostly documentation of a preset over the two existing modes.
+- *Where/how the page fires* — every task vs end-of-set, and via what. The paging surface is in flux (=page-signal= removed 2026-06-12), so reconcile against =notify --persist= or whatever paging stands now.
+- *Auto-pull vs explicit list* — whether the set comes from an explicit ordered list or a tag/priority query.
+- *Guardrails* — must refuse to speedrun tasks needing design decisions or carrying data-loss risk without a checkpoint (the sender's biased-safe unused-tile flag is the worked example).
+
+*** 2026-07-01 Wed @ 22:10:35 -0400 Phase 0 landed — hard tag definitions + review/audit enforcement
+todo-format.md gained the "Hard definitions: :solo: and :quick:" subsection under the scheme header (fixed across projects: :solo: = buildable + agent-verifiable + no deliberation, with one-or-two upfront-answerable quick decisions allowed per the ratified spec; :quick: = ≤30-min effort hint, never a gate). task-review.org: the two tagging sections are now explicitly mandatory ("a review that skips them is incomplete") and gate 3 was realigned from "no upfront decision" to the spec's no-deliberation form — the stricter old wording predated the pre-flight-Q&A decision and would have wrongly excluded quick-question tasks. task-audit.org: the re-assess bullet is marked mandatory and points at the todo-format hard definitions as canonical. Phases 1-6 (work-the-backlog extraction, callers, commit gate, checklist/Q&A/page, metrics, synthesis) remain.
+
+*** 2026-06-16 Tue @ 00:53:36 -0500 Spec written; design questions answered
+Craig's "your call" (2026-06-16) answered in [[id:90f623cd-fdbe-4f5c-b63d-b2f84d9151cf][the autonomous-batch execution spec]], which reconciles this with Phase E into one feature:
+- *Most effective / workflow-vs-preset:* one dedicated =work-the-backlog.org= workflow holds the execution loop; "fix speedrun" is a thin named preset (no-approvals + always-push + end page) feeding it an explicit list, and the inbox-zero loop feeds it a tag query. Pros of the shared workflow: one execution loop to audit, inbox-zero's three callers stay clean, both input shapes reuse one guardrail set. Cons: one more workflow file and a caller-to-workflow indirection. The con list is shorter and lighter than the duplication cost of two separate features, which is why the shared workflow wins. The pros carry the more important entries (single audit surface, clean seam).
+- *Paging:* end-of-set only, via =notify ... --persist= (reconciled past the removed page-signal wrapper).
+- *Auto-pull vs explicit list:* both — explicit list for the preset, tag/priority query for the loop.
+- *Effectiveness measurement (the trial Craig asked for):* the spec designs a per-task JSONL metrics log (=.ai/metrics/work-the-backlog.jsonl=), a corrections-in-next-session signal, and a periodic synthesis step that writes =:agent:metrics:= org-roam articles for later review — the "gather data + create org-roam articles" loop.
+*** 2026-06-29 Mon @ 03:48:09 -0400 Ratified the autonomous-batch execution spec
+Craig ratified all eight decisions in [[id:90f623cd-fdbe-4f5c-b63d-b2f84d9151cf][2026-06-16-autonomous-batch-execution-spec.org]] (revised this session — size gate removed, crisp four-item defer checklist, =:solo:= / =:quick:= definitions + task-review/audit enforcement, speedrun pre-flight Q&A). Spec Status → ready; implementation-ready across Phase 0–6. Decisions grew from six to eight during the revision.
+
+*** 2026-07-02 Thu @ 00:44:59 -0400 spec-response decomposition — :SPEC_ID: bound, spec DOING
+Stamped the spec's UUID on this parent, broke Phases 1-6 into the build tasks below (plus the flip task and a live-trial validation child), and flipped the spec's status heading READY → DOING per the transition-ownership table.
+
+*** 2026-07-02 Thu @ 01:07:29 -0400 Phase 1 landed — execution loop extracted into work-the-backlog.org
+work-the-backlog.org written (canonical + mirror): caller contract (task set + session mode + cap), five-outcome vocabulary, the loop, mechanical eligibility gate (TODO + :solo: per scheme header, safe-by-omission, no-scheme-header → don't run), four-item defer checklist, per-task quality bar, cap/kill-switch semantics, page + metrics stubs pointing at Phases 4-5. inbox.org's auto-mode per-cycle item 3 reverted to routing-only (yes-path execution removed; mode intro + closing line updated to match). INDEX.org entry added. make test green, sync clean; nothing invokes the new workflow yet.
+
+*** 2026-07-02 Thu @ 01:13:33 -0400 Phase 2 landed — both callers wired
+inbox.org auto-mode item 3 regained its "run this batch next?" ask, now chaining into work-the-backlog as an explicit second step after routing (eligibility query + file-only + paging off + cap 1). work-the-backlog.org gained the two caller sections: the auto-loop contract and the no-approvals speedrun preset (seven-step pre-flight → autonomous-commit + always-push + paging-on over an explicit list; finer Q&A mechanics deferred to Phase 4). Speedrun trigger phrases live in the workflow + INDEX; "speedrun" always routes to the preset, with a disambiguation note in no-approvals.org and its INDEX entry. Each caller independently exercisable.
+
+*** 2026-07-02 Thu @ 01:18:07 -0400 Phase 3 landed — waiver-gated commit autonomy
+Pinned the waiver format per D5: two marker lines in .ai/notes.org Workflow State — :COMMIT_AUTONOMY: yes (has the waiver) and :LOOP_MAY_COMMIT: yes (the unattended loop may also commit; requires the first). Absent or non-yes reads as no; the read is a fresh grep each run, never memory. Degrade contract written into work-the-backlog.org (surface in run intro + summary, never honor without the marker, never degrade silently); caller sections + Common Mistakes updated. Stamped rulesets' own :COMMIT_AUTONOMY: yes; :LOOP_MAY_COMMIT: deliberately not granted — Craig's call. .emacs.d holds the waiver too but its notes.org is its own scope; told via inbox-send to stamp its marker.
+
+*** 2026-07-02 Thu @ 01:21:47 -0400 Phase 4 landed — checklist mechanics, pre-flight Q&A contract, page
+The four-item checklist (in since Phase 1) gained its mechanics: a VERIFY-filing subsection (dedup against an existing sibling first — the deferred task stays TODO, so without the check every run re-files; placement/heading/body per todo-format.md) and a quick-question routing subsection (discriminator: one-line factual/preference pick vs tradeoff-weighing; three-plus questions = underspecified = file; item 2 data-loss never routes to Q&A). Preset section gained the batch-ask contract (one message, recommendation-first numbered options per interaction.md, answers recorded as dated lines in the task bodies before the run). Page section finalized (fires once on set-done or cap-hit; notify --persist is the paging surface). Common Mistakes 12-13 added. Checklist only ever reduces what runs; pre-flight fires only under the preset.
+
+*** 2026-07-02 Thu @ 01:24:50 -0400 Phase 5 landed — per-task JSONL metrics log
+Metrics section written into work-the-backlog.org: one record per task at outcome time, appended to the project's .ai/metrics/work-the-backlog.jsonl (git-tracked, append-only, dir+file created on first append). Full field table per the spec (ts, run_id, project, caller, task, outcome, defer_reason, upfront_decision, wall_clock_s, commit_sha, review_findings), outcome slugs mapped to the prose vocabulary, commit_sha flagged as the corrections-signal key (comma-separated when a task decomposed into several commits). Added the sixth outcome the spec's readiness section demanded but the enum missed: failed (tree left working, surfaced, run continues) — wired into the Outcomes vocabulary and loop step 4. A failed append warns in the run summary but never blocks, reorders, or aborts execution.
+
+*** 2026-07-02 Thu @ 01:27:43 -0400 Phase 6 landed — synthesis step to org-roam
+Synthesis section written into work-the-backlog.org (trigger "synthesize backlog metrics", INDEX row added): discover the JSONL union across project roots, classify each project per knowledge-base.md's denylist before reading, exclude work/unknown projects with the refusal contract, compute per-run rollups + trends, compute the corrections signal (later revert/fix commit touching the same files within ~14 days — a flag for human review, not a conviction), write one :agent:metrics: KB node under ~/org/roam/agents/ with [[id:...]] links to prior synthesis nodes, pull-before/commit-push-after. Read-only over the logs plus the single KB write; never mutates JSONL, todo.org, or any tree.
+
+*** 2026-07-02 Thu @ 05:26:07 -0400 Live trial passed — first speedrun ran 3/3, every loop part exercised
+Craig named the ordered set (id-link conversion, host-identity guard, template-sync policy) and said it was the validation run. Pre-flight Q&A fired once (two questions, both answered, answers stamped as dated lines before the run); each task landed as its own reviewed commit under the waiver (78bbaae, b6a977c, ed75d3c); metrics JSONL carries one record per task (run c726f526); the end-of-set page arrived via notify --persist. Nothing needed a VERIFY this run (all three cleared the checklist). Craig's read: granted :LOOP_MAY_COMMIT: on the strength of the run.
+
+*** 2026-07-02 Thu @ 05:26:07 -0400 Flipped the spec DOING → IMPLEMENTED
+All six phases built and the live trial validated. Keyword, dated history line, and Metadata mirror all flipped per the transition-ownership table.
+** DONE [#B] inbox-send filename collision silently overwrote a message :bug:solo:
+CLOSED: [2026-07-02 Thu]
+:PROPERTIES:
+:CREATED: [2026-07-02 Thu]
+:END:
+From archsetup (2026-07-02 0543, found in the wild): two --text sends in the same minute whose text starts with the same phrase derive identical filenames, and the second silently overwrites the first — archsetup lost a message at 05:42 and had to resend. Severity data-loss x rare-edge = P2 = [#B].
+
+Resolution 2026-07-02 (auto-inbox-zero loop, standing yes): uniquify() guard in inbox-send.py — an existing target gets a -2/-3/... stem suffix, both send_text and send_file paths, extension preserved. Four red-first tests reproduce the loss (module-level with a fixed timestamp so the same-minute collision is deterministic, plus a CLI loss-proof check); 30/30 green.
+** DONE [#C] page-me notify styling — all-red too alarming :bug:solo:
+CLOSED: [2026-07-02 Thu]
+:PROPERTIES:
+:CREATED: [2026-07-02 Thu]
+:END:
+From Craig via the roam inbox (2026-07-02, routed by archsetup): the page notify's all-red styling "makes me feel like somehow the system is about to crash" — should be a persistent info-level notification.
+
+Resolution 2026-07-02 (auto-inbox-zero loop, standing yes): pages now use notify info --persist instead of notify alarm — page-me.org (all examples + prose, with Craig's verdict recorded) and work-the-backlog.org's end-of-set page. status-check's success/fail types untouched (job outcomes, not pages).
+** DONE [#C] Template sync with gitignored-only local changes :feature:
+CLOSED: [2026-07-02 Thu]
+From Craig via the roam inbox (2026-07-02, routed by archsetup): downstream projects should still pull template updates when their local changes sit entirely in gitignored files or directories — an inbox drop or a file left to read doesn't affect the templates, yet it currently holds the sync back and projects fall behind. When worked: verify how the sync gate actually detects dirtiness today, then let gitignored-only changes pass it.
+
+2026-07-02 Thu @ 05:09:58 -0400 — Craig (speedrun pre-flight): policy + audit. Scope read found startup's git gates already ignore untracked/ignored files; state the policy in startup.org and audit every dirty-check in the synced workflows to match (monitor-inbox's bare porcelain check is the known offender; tracked-modification blocking stays).
+
+Resolution 2026-07-02: template-freshness policy stated in startup.org Phase A.0 (dirty = tracked modifications only; untracked/gitignored never block pulls, ffs, or monitoring gates; the rsync WIP-guard named as the one deliberate exception — it holds back rulesets' own outbound WIP). Full audit of dirty-checks across synced workflows: startup's two git gates already complied; inbox.org monitor mode was the one offender — its precondition now uses --untracked-files=no with the explicit-staging rationale, and its close-out sweeps tracked changes only. triage-intake auto mode borrows monitor's gates, so it inherits the fix by reference.
+** DONE [#B] Wrap-up inbox/transcript routing to destination projects :feature:spec:
+CLOSED: [2026-07-04 Sat]
+:PROPERTIES:
+:CREATED: [2026-06-13 Sat]
+:LAST_REVIEWED: 2026-06-24
+:END:
+Optional wrap-up step that surfaces filed keepers belonging to another project, recommends a destination, and routes each to that project's =inbox/= via =inbox-send= (the destination's own =process-inbox= files it; transcript filing deferred to vNext). Spec: [[id:00b47414-2213-4a99-be35-48ceb266fc08][wrapup-routing-spec.org]] — Ready, [9/9] decisions. Source proposal: [[file:docs/design/2026-06-13-wrapup-inbox-transcript-routing-proposal.org]].
+
+*** 2026-06-21 Sun @ 02:06:37 -0400 Spec-review + spec-response complete — Ready
+Craig's review challenge reshaped the design from a direct cross-repo =todo.org= move to =inbox-send= delivery into the destination's inbox (safer: reuses the sanctioned cross-project path, gets provenance + per-project filing for free, degrades gracefully where a destination has an =inbox/= but no =todo.org=). D2/D3 superseded; D7 (inbox-send delivery), D8 (=:ROUTE_CANDIDATE:= marker at file time), D9 (local source removal + reject-flow recovery) added. Spec-review file consumed and deleted. Implementation-task breakdown filed below (spec-response Phase 6).
+
+*** 2026-06-24 Wed @ 00:21:20 -0400 Reconcile — marker sub-task repointed at inbox.org
+The 2026-06-23 inbox consolidation (24ca58d) merged =process-inbox= + =monitor-inbox= + =inbox-zero= into one =inbox.org= engine (process/monitor/roam modes) and deleted the three old files. The =:ROUTE_CANDIDATE:= marker sub-task targeted =process-inbox.org='s Phase D — repointed it to =inbox.org= process mode (core §3 "File as TODO"). No build has started, so this is a target-rename only; the spec design is unaffected.
+
+*** 2026-06-28 Sun @ 13:02:42 -0400 Built the recommendation engine + destination discovery
+Added =.ai/scripts/route_recommend.py= (canonical + mirror): pure =recommend(item, projects) → (destination, confidence)= with strong (name/path literal, word-boundary matched, dot-stripped alias aware), weak (distinctive name-token overlap), and none tiers; a multi-way top-tier tie downgrades to weak with a deterministic pick (most overlap, then alphabetical); empty list → none. The CLI (=--item=, =--exclude=) reuses =inbox-send.py='s =discover_projects= via importlib so the candidate set matches inbox-send's project universe. 13 tests (the five spec'd cases + boundary/path/strong-beats-weak + 3 sandboxed CLI integration tests), full =make test= green. Covers spec Phases 1 + 3. Next sub-tasks (=:ROUTE_CANDIDATE:= marker, wrap-up router) call this engine.
+
+*** 2026-07-02 Thu @ 00:36:12 -0400 Phases 2 + 4 + test surface landed — marker, router, route-batch helper
+inbox.org's "File as TODO" disposition now runs route_recommend on each keeper and stamps =:ROUTE_CANDIDATE: <destination>= on strong/weak matches (none stamps nothing; local keepers stay unstamped) — spec Phase 2 / D8. wrap-it-up.org Step 3 gained the optional router directly after the inbox sanity check, with the gate-vs-optional split named in the prose: surface the batch (task / destination / delivery mode / confidence, weak visibly labeled), go/skip, empty set = zero interaction — spec Phase 4 / D7 / D9. The go path is mechanical: new =.ai/scripts/route-batch= (--list read-only, --go extracts the subtree minus the marker with children riding along and headings promoted, delivers via inbox-send for provenance, removes the local subtree only after a successful send; a failed send leaves the task in place and exits non-zero). Test surface: engine unit tests existed (13); route-batch adds a 9-test bats suite (list/backlog-exclusion, empty-set silence, list-modifies-nothing = skip semantics, delivery + provenance + children, local-task survival, drawer-minus-marker, inbox-without-todo.org delivery, empty go, failed-send recovery). cross-project.md notes the router as a sanctioned cross-project write path. make test green, sync clean.
+
+*** 2026-07-04 Sat @ 11:49:59 -0500 Flipped the spec to IMPLEMENTED; parent closed
+The routing build shipped green: route_recommend.py (destination discovery + recommendation, 13 unit tests), the =:ROUTE_CANDIDATE:= marker in inbox process mode, the wrap-it-up router sub-step, and route-batch (9 bats tests). Spec [[id:00b47414-2213-4a99-be35-48ceb266fc08][wrapup-routing]] flipped DOING → IMPLEMENTED. Manual end-to-end validation and the transcript vNext promoted to their own tasks.
+** DONE [#C] Check that memories are sync'd across machines via git :spec:
+CLOSED: [2026-07-04 Sat]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-06-28
+:END:
+v1 implemented end-to-end 2026-06-10 (Phases 0-4, all shipped + pushed) and the agent-KB spec is IMPLEMENTED. Both daily drivers (velox + ratio) carry the =roam.git= clone + sync-timer. Cross-machine sync verified as a full round-trip 2026-07-04: velox created a probe node 2026-07-01 and ratio held it; ratio then deleted it and velox pulled the deletion. Both clones sit at the same HEAD with zero sync-conflict files and both roam-sync timers active. The work/unknown-project write-refusal checks need live sessions in those projects and are tracked as a standalone manual-validation task below.
+*** 2026-05-14 Thu @ 19:14:11 -0500 Investigate current memory storage
+
+Memory files live at
+[[file:/home/cjennings/.claude/projects/-home-cjennings-code-rulesets/memory/][~/.claude/projects/-home-cjennings-code-rulesets/memory/]]
+— four files including =MEMORY.md= and three individual entries
+(=feedback_never_guess.md=, =project_ai_scripts_canonical_source.md=,
+=reference_pdftools_venv.md=). The directory is a plain unmanaged dir
+(no symlink, no enclosing git checkout). Neither
+[[file:/home/cjennings/.claude/][~/.claude/]] itself nor any subtree
+containing the project-memory dirs is tracked in
+[[file:/home/cjennings/code/archsetup/][archsetup]] or
+[[file:/home/cjennings/code/rulesets/][rulesets]]. Without a symlink
+into a stowed or tracked location, memory files don't survive a new
+machine setup or a dotfiles restore.
+
+Proposed setup: stow =~/.claude/projects= →
+=archsetup/dotfiles/common/.claude/projects/= (path doesn't exist yet
+— it's the target location pending VERIFY).
+Create the destination in archsetup, move existing per-project
+=projects/<encoded-cwd>/memory/= dirs there, run =stow= to link, then
+commit + push archsetup. After that, every machine running =stow=
+picks up the same memory tree.
+
+*** 2026-05-23 Sat @ 16:12:48 -0500 Decided: dedicated private repo, not stow
+Worked through dotfiles → rulesets → dedicated repo. Dropped stow/dotfiles (machine config, wrong cadence) and rulesets (it's pulled first in every session, so memory edits would dirty its tree and skip the startup =git pull --ff-only=). Chose a dedicated private repo on cjennings.net: storage is unified there while recall stays per-project (the encoded-cwd subdirs), since pooling recall would hurt relevance and risk work-private facts surfacing in personal-project artifacts.
+
+*** 2026-05-23 Sat @ 16:12:48 -0500 Shipped: claude-memory.git + folded symlinks
+Created bare =git@cjennings.net:claude-memory.git=, cloned to =~/.claude-memory= (later deleted in the reversal below), moved all 7 per-project =memory/= dirs in (54 files; work has 40) and replaced each live =~/.claude/projects/<enc>/memory= with a folded dir-symlink so new memory lands in the clone and a push syncs it. Added =link-claude-memory.sh= (idempotent — recreates the symlinks on a new machine after clone) + README. Private repo, never GitHub (carries work/DeepSat memory). Initial import pushed (=f496370=).
+
+*** 2026-05-24 Sun @ 01:53:35 -0500 Reversed the migration — back to unmanaged per-project memory
+Cancelled the follow-up brainstorm and undid the dedicated-repo migration at Craig's call. Moved all 7 memory dirs back to =~/.claude/projects/<enc>/memory/= (content preserved), deleted the =~/.claude-memory= clone, and deleted the bare =claude-memory.git= on the server. Memory is back to its original at-risk state, so the task reopens at [#C] pending a direction. The brainstorm landed on a two-tier idea for whenever this resumes: promote general lessons into a rulesets-tracked file symlinked into =~/.claude/rules/= (loaded into every project natively, one repo), and keep project-specific memory under each project's own =.ai/memory/= (committed where =.ai/= is tracked, at-risk where it's gitignored). Not implemented.
+
+*** 2026-06-05 Fri @ 05:57:35 -0500 Pivot: adopt the existing org-roam KB as the shared agent substrate
+Pressure-tested the two-tier idea, then Craig redirected: a shared org-roam knowledge base any project can read and write makes this simpler. Ground truth verified: =~/sync/org/roam/= already exists (484 org files, curated since 2023, Syncthing-synced, not git). So cross-machine sync is already solved, and the task stops being "build a memory-sync system" and becomes "point agents at the KB that already syncs." The dedicated-repo and two-tier approaches are both superseded for the storage+sync half.
+
+Wrote a one-page spec: [[id:08a5ec99-9e1e-40e4-8241-e8a41e9de49f][agent-knowledge-base-spec.org]] (originally docs/design/2026-06-05-org-roam-knowledge-base-spec.org; superseded by the 2026-06-10 spec-create rewrite at the new path). Five decisions, mechanics recommended: (1) KB is a queried substrate accessed as files (ripgrep + follow =[[id:]]= by grep), not via the org-roam package; (2) capture in harness memory, promote durable facts into the KB (same cadence as the pattern catalog) — resolves the at-risk problem since the valuable knowledge moves to the synced KB; (3) a =claude-rules/knowledge-base.md= pointer rule carries path/query/write-schema/boundary; (4) write schema = roam-valid node + =:agent:= filetag so agent notes stay distinguishable and index on the next =org-roam-db-sync=. The rules layer (=claude-rules/=, =CLAUDE.md=) is untouched — the KB replaces the memory tier, not the rules tier.
+
+*** 2026-06-10 Wed @ 14:29:20 -0500 Spec ratified — write boundary is option C; rewritten to spec-create format
+Craig answered via cj annotations in the spec (2026-06-10): DECISION 5 is option C (read-shared, write-scoped — work agents never write the KB). Syncthing does replicate ~/sync/ to a work machine and Craig is fine with how C handles it. Node granularity: per-fact nodes. Write review: agent writes land freely in the KB only — explicitly not permission to post to email, Linear, or any public channel without review and consent. The spec was rewritten into the spec-create format at [[id:08a5ec99-9e1e-40e4-8241-e8a41e9de49f][agent-knowledge-base-spec.org]] (old draft removed). Implementation explicitly held pending Craig's go-ahead; one decision still open (D7, next VERIFY).
+
+*** 2026-06-10 Wed @ 14:35:40 -0500 Spec review — not ready
+Review written at docs/agent-knowledge-base-spec-review.org (deleted on disposition completion; content summarized in the spec's Review dispositions). Rubric: =Not ready=. Blockers: resolve D7 (keep vs retire harness memory) and define the executable personal/work/unknown write-boundary classifier plus work-side write/refusal destination. Medium notes: use concrete ripgrep commands that exclude =*.sync-conflict-*= files, and define seed-node approval/rollback.
+
+*** 2026-06-10 Wed @ 14:44:00 -0500 D7 resolved — keep harness memory as the capture layer
+Craig ratified "keep" in chat (2026-06-10). Harness memory stays the ephemeral, auto-recalled capture layer; the KB holds promoted durable facts; Phase 3's wrap-up promotion cadence is mandatory. Spec D7 flipped to accepted; D2 stands as written.
+
+*** 2026-06-10 Wed @ 14:44:00 -0500 Project classification defined — work-root denylist, unknown refuses
+Resolved in the spec-response pass: =knowledge-base.md= carries an explicit work-root denylist (initially =~/projects/work=) as the source of truth. Personal = under a known project parent (=~/code/=, =~/projects/=, =~/.emacs.d=) and not denylisted → KB writes allowed. Work or unknown → no KB write; the agent reports the refusal with a one-line redacted summary of the fact. v1 adds no new work-side store — work projects keep their existing project-tree conventions. See the "Project classification and write routing" section of [[id:08a5ec99-9e1e-40e4-8241-e8a41e9de49f][the spec]]. Denylist completeness is the one open caveat (next VERIFY).
+
+*** 2026-06-10 Wed @ 14:44:00 -0500 Codex review incorporated — spec ready with caveats
+Spec-response pass processed the 2026-06-10 Codex review with D7 = keep as a pre-agreed input. Both blockers cleared (D7 accepted; classification/write-routing section added). Mediums accepted: canonical rg commands with conflict-file exclusion, Phase 2 seed-node approval/rollback mechanics, Makefile no-change note, Testing/Verification section. Three recommendations modified, none rejected — see the spec's Review dispositions. Review file deleted per the workflow. Rubric: ready with caveats (denylist confirmation). Implementation tasks broken out below; implementation itself awaits Craig's go.
+
+*** 2026-06-10 Wed @ 17:29:37 -0500 Work-root denylist confirmed — ~/projects/work only
+Craig confirmed (2026-06-10, in chat): the denylist is just =~/projects/work=. Archangel is not work-scoped. The spec's one caveat clears — status now ready. Phase 1 is unblocked, but implementation still awaits Craig's explicit go.
+
+*** 2026-06-10 Wed @ 17:57:08 -0500 Spec amended — D8 git transport + migration/metrics/docs/maintenance folds
+Craig's five design questions answered and folded into the spec, and D8 ratified (Shape A): the KB moves out of the =~/sync/org= Syncthing share into its own git repo on cjennings.net, with an =agents/= subdirectory for agent writes, a systemd auto-sync timer for Craig's edits, opt-in-by-clone replication (work machine doesn't clone), and the phone staying on the on-demand =~/sync/phone= pattern. Folded in: inclusion criteria + a Phase 1.5 guided memory sweep, a Success metrics section with a 30-day checkpoint, the seed node redefined as the KB's own documentation, and Phase 4 maintenance automation. Phases renumbered 0-4; tasks below updated. Implementation still held.
+
+*** 2026-06-10 Wed @ 18:21:33 -0500 Phase 0 done — roam migrated to git
+Backed up (~/roam-backup-2026-06-10.tar.gz), copied to =~/org/roam=, 63 conflict files deleted (424 org files), git repo with origin =git@cjennings.net:roam.git= (initial commit 515693d), old location replaced with a transition symlink. Emacs =roam-dir= updated in user-constants.el + live-reloaded (db rebuilt, 416 nodes); handoff to .emacs.d for the commit. =roam-sync.sh= (6 bats green) on a 15-min systemd user timer, installed + enabled + round-trip verified. Old-path references repointed (protocols task-list pointer, journal workflow, notes template). archsetup handoff covers dotfiles adoption + other-machine clones. rulesets commit fcf554a.
+
+*** 2026-06-10 Wed @ 18:21:33 -0500 Phase 1 done — knowledge-base.md rule live
+=claude-rules/knowledge-base.md= written (path, git discipline, query commands, agents/ write schema, denylist + refusal contract, inclusion criteria, capture-then-promote). =make install= linked it machine-wide; verified the link, a known-note query, and conflict-glob exclusion with a planted file. Commit d071f1f.
+
+*** 2026-06-10 Wed @ 18:21:33 -0500 Phase 1.5 done — rulesets swept, 10 projects broadcast
+Rulesets' 6 memories classified: 3 promoted as =agents/= nodes (notify-attention pattern, pdftools venv, gpg-agent SSH TTL trap), 2 kept local (rule-encoded in verification.md / interaction.md), 1 kept + de-staled (ai-scripts-canonical updated for the claude-templates subtree fold). Sweep handoff broadcast to the 10 other memory-bearing projects (archsetup, org-drill, pearl, .emacs.d, elibrary, finances, health, home, jr-estate, kit); work skipped by the boundary; the orphaned =linear-emacs= memory dir (project retired, likely pearl's predecessor) noted for Craig.
+
+*** 2026-06-10 Wed @ 18:21:33 -0500 Phase 2 done — seed/doc node written and indexed
+=agents/20260610181640-how-the-agent-knowledge-base-works.org= written: the KB's user-facing guide (what agents do, how it syncs, finding/pruning agent content, the rule pointer). Index verified programmatically: =org-roam-node-from-title-or-alias= resolves it with tags (agent reference); node count 416 → 420. Craig's visual check remains in the manual-testing child.
+
+*** 2026-06-10 Wed @ 18:21:33 -0500 Phase 3 done — wrap-up promotes + records the KB receipt
+wrap-it-up.org Step 1 gains the promotion check (inclusion-criteria bar) and the mandatory "KB: promoted N / consulted yes-no" Summary line; validation checklist enforces it. Mirror synced, integrity OK (44), parse OK. Commit 242b95e.
+
+*** 2026-06-11 Thu @ 19:26:26 -0500 .emacs.d memory sweep complete (first broadcast response)
+First of the 10 broadcast projects to report Phase 1.5 done (handoff 18:23). Inventory 7: promoted 3 to KB (no-make-frame-in-live-daemon, proton-bridge-headless-cert-mismatch, open-images-with-imv — roam commit a915760), kept 3 local at Craig's call (commit-flow-no-approval-gate per-project-scoped; two theme-scoped ones possibly superseded by the palette-columns spec), deleted 1 (superseded by canonical interaction.md rule). 9 projects' sweeps outstanding.
+
+*** 2026-06-12 Fri @ 02:25:12 -0500 Five more sweeps complete via the home folds
+Overnight handoffs from home closed five more broadcast targets, each swept at fold-time triage with Craig's approval: jr-estate 2 promoted (forms name-with-number, PDF-editing tooling split; roam 45d8e6c) / 3 kept with area attribution / 2 deleted as rule-encoded or duplicate; finances 0/1/0 (rosalea-daly contact fact kept local); elibrary 0/0/2, health 0/0/1, kit 1/0/2 (hand-prep-items-to-work-inbox promoted into home's memory; the rest duplicated rules or home memories). Nothing from these five met the KB bar that wasn't already encoded. All folded projects' session archives merged area-prefixed into home's .ai/sessions/, so session-harvest's first run sees them. Home covers its own and remaining areas' sweeps through ongoing discipline; still pending from the broadcast: archsetup and work.
+
+*** 2026-07-04 Sat @ 11:52:00 -0500 Manual validation — checks 1 + 4 verified; refusal checks split out
+Check 1 (seed node in org-roam + rg inventory) verified on velox 2026-07-01 (55 =:agent:= nodes matched the live org-roam DB). Check 4 (cross-machine sync) verified as a full round-trip 2026-07-04: velox pushed a probe 2026-07-01, ratio held it, ratio deleted it, velox pulled the deletion; both clones at the same HEAD, zero sync-conflict files, both timers active. Checks 2 + 3 (work / unknown-project write refusal) need live sessions in those projects — promoted to a standalone manual-validation task.
+
+*** 2026-06-10 Wed @ 18:21:33 -0500 Phase 4 done — monthly hygiene automation live
+=scripts/kb-hygiene.sh= (6 bats green, shellcheck clean, read-only by design) inventories =:agent:= nodes, flags orphans / duplicate titles / conflict files, and writes an org report into the rulesets inbox; =roam-hygiene.timer= (monthly, Persistent) installed + enabled. Live run against the real KB verified (4 agent nodes, 428 files, 0 conflicts). Conditional vNext stays in the spec's scope tiers: a =/promote= command if the wrap-up prompt proves insufficient, an =:agent:inbox:= staging tag if free writes prove too noisy. Commit b014095.
+
+*** 2026-06-30 Tue @ 13:53:34 -0400 ratio roam clone + sync-timer confirmed (cross-machine half done)
+Verified ratio over tailscale ssh: =~/org/roam= is a clone of =git@cjennings.net:roam.git= (HEAD auto-synced 13:11 today), and =roam-sync.timer= is enabled and actively firing (last run 5 min prior, next in 10). Both unit files present. velox was already confirmed, so the one-time clone+timer setup is now done on both daily drivers — the (b) half of this VERIFY's remaining work. Only the manual-validation child (work/unknown-project refusal checks needing Craig's eyes) is left before DONE. Cleared the matching "Current open instance" line in =daily-drivers.md=.
+
+*** 2026-07-01 Wed @ 21:52:15 -0400 Manual-validation checks 1 + 4 (velox half) verified; ratio unreachable
+Check 1 (seed node in org-roam + rg inventory): 55 =:agent:= nodes found by the rg inventory AND 55 nodes under =agents/= indexed in the live org-roam DB (emacsclient query) — match. Check 4 (cross-machine edit): created a temporary probe node =agents/20260701214910-kb-sync-validation-probe.org=, triggered =roam-sync= — committed and pushed to origin within seconds (f0252bb), zero =sync-conflict= files. The ratio half could NOT be verified tonight: =tailscale ping ratio= pongs via DERP, but ssh to 100.71.182.1:22 times out (machine likely suspended). Probe left in place; when ratio is back, confirm with: ssh cjennings@100.71.182.1 'ls ~/org/roam/agents/ | grep kb-sync-validation-probe' — then delete the probe node. Checks 2 + 3 (work/unknown-project refusal) still need live sessions in those projects, and "work machine has no KB clone" needs the work machine named + checked.
+** DONE [#C] Spec storage location + lifecycle-status convention :spec:
+CLOSED: [2026-07-04 Sat]
+:PROPERTIES:
+:CREATED: [2026-06-15 Mon]
+:LAST_REVIEWED: 2026-06-24
+:SPEC_ID: 80b0787b-4a60-4c82-8a16-b383d3e3c8f2
+:END:
+Two coupled documentation conventions for rulesets to adopt, surfaced by .emacs.d while triaging ~28 design docs. Both land in =spec-create= ([[file:.ai/workflows/spec-create.org]]) and likely a new =docs-lifecycle= rule under =claude-rules/=. Source proposal: [[file:docs/design/2026-06-15-spec-storage-lifecycle-proposal.org]] (.emacs.d handoff 2026-06-15).
+
+The two conventions:
+- *Location split* — formal specs live in =docs/specs/=; =docs/design/= keeps working notes, brainstorms, inventories, reviews. A spec is a doc proposing a buildable change with a Decisions section and phases; everything else is a note.
+- *Glanceable lifecycle status* — a spec's state (draft / doing / implemented / superseded / cancelled) is visible without opening the file, plus an authoritative in-file record.
+
+We handle the task in priority order. Mechanism decided 2026-06-28; migrates into the spec when built.
+
+*** Decisions (settled 2026-06-28 — migrate into the spec when built)
+1. *Location split — adopt.* =docs/specs/= for formal specs, =docs/design/= for notes (brainstorms, inventories, reviews). A spec is a doc proposing a buildable change with a Decisions section and phases; everything else is a note. Document in spec-create and the docs-lifecycle rule.
+2. *Status mechanism — org-keyword authoritative.* The spec's =#+TODO:= state on its top heading is authoritative (specs already carry =#+TODO: TODO | DONE SUPERSEDED CANCELLED=), mirrored in a =Status= field in the Metadata table. Drop the filename suffix entirely — it's redundant with the Status field and adds rename churn across a cross-linked, template-synced doc set. (Craig 2026-06-28, choosing org-keyword over his earlier filename-suffix lean.)
+3. *Link safety — adopt =org-id= ([[id:...]]) for cross-doc spec links.* Decouples link stability from the status mechanism; good hygiene regardless.
+4. *Generalize.* Capture the shape (lifecycle state authoritative-in-artifact, formal-vs-notes split, rename-safe links) as a general =docs-lifecycle= convention in =claude-rules/=, with spec-create as the first instance.
+5. *Retrofit existing files across ALL projects* (Craig 2026-06-28). The convention is worthless if legacy docs stay misfiled — every project's existing =docs/design/= pile (the ~28 in .emacs.d that surfaced this) must be sorted: formal specs move to =docs/specs/=, notes stay in =docs/design/=, inbound =file:= links updated. This is a one-time per-project migration that template sync can't perform, so the spec must design the reach mechanism. Proposed shape: a synced classify-and-move helper under =.ai/scripts/= (heuristic: a doc with a Decisions section + phases/Metadata is a spec) that proposes moves for confirmation and relinks, plus a startup nudge gated on a per-project =:LAST_SPEC_SORT:= marker so each project runs it once. Classification is a judgment call — the helper proposes, a human confirms.
+
+Follow-up once built: update spec-create to emit into =docs/specs/= with the org-keyword status; write the =docs-lifecycle= rule; ship the retrofit helper + startup nudge; retrofit rulesets' own =docs/design/= first as the pilot; send a note if .emacs.d should pilot before generalizing.
+
+*** 2026-07-01 Wed @ 22:13:00 -0400 Spec drafted — first resident of docs/specs/, awaiting review
+Wrote [[id:80b0787b-4a60-4c82-8a16-b383d3e3c8f2][the spec]] from the five settled decisions, dogfooding its own conventions: it lives in the new =docs/specs/=, opens with the =* DRAFT Docs lifecycle= status heading (org keyword authoritative, =:ID:= for id-links, dated history in the body), and drops the status filename suffix. It pins the one mechanism the decisions left open — where the keyword lives: a prepended top-level status heading with vocabulary =DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED=, additive and retrofittable, giving both the one-line =rg= board and free org-agenda scanning. Four build phases: rule + template updates → =spec-sort= helper (classify/confirm/move/relink, bats) → rulesets pilot (41 design files, 3 spec-spine candidates, 2 stray root specs) → startup nudge gated on =:LAST_SPEC_SORT:= + .emacs.d note. Status DRAFT until Craig's review flips it READY.
+
+*** 2026-07-01 Wed @ 22:22:34 -0400 Codex spec-review complete — Not ready
+Review findings live in [[id:cc77a7f6-e4c3-488a-ac3b-e739420a5c2b][the spec]]. Four blockers before implementation: the proposed lifecycle =#+TODO:= line drops the =TODO=/=DONE= states needed by Decisions and Review findings cookies; =spec-sort= needs an exact relink/unsupported-residue contract; the sort marker/startup nudge must name the actual =.ai/notes.org= state surface and detection flow; and the stricter =docs/specs/= review precondition must not strand legacy specs before retrofit.
+
+*** 2026-07-01 Wed @ 22:30:06 -0400 Second review merged + responder pass — all nine findings fixed
+A fresh-context Claude reviewer independently rated the draft Not ready, converging on Codex's keyword-vocabulary blocker (adding the cross-sequence uniqueness wrinkle) and contributing five unique findings — the biggest: nobody owned the DOING→IMPLEMENTED flip, the exact mechanism whose failure produced this spec. Craig approved fixing all nine. Fixed in place: two-sequence collision-free keyword header (dogfooded in the spec's own header; org now computes the [5/5] and [9/9] cookies — verified in batch), transition-ownership table incl. spec-response's mandatory flip-to-IMPLEMENTED task + task-audit safety net, single classification predicate (Decisions AND Implementation phases), the -spec.org rename step, the full relink data-safety contract (rewritten roots / report-only surfaces / dry-run default / residue-grep gate), the =.ai/notes.org= marker + Phase A probe + Phase C nudge contract, the legacy-location compatibility rule, the org-id Emacs-resolution prerequisite for .emacs.d, and the three-line transition definition. Ledger + per-finding responses in the spec's Review findings section. Status stays DRAFT pending Craig's READY flip.
+
+*** 2026-07-01 Wed @ 22:41:33 -0400 Codex spec-review rerun — Not ready
+Fresh review after the response/READY flip added five new blocking findings in [[id:cc77a7f6-e4c3-488a-ac3b-e739420a5c2b][the spec]] and demoted the spec back to DRAFT. Remaining blockers: shared helper/workflow edits must name the canonical =claude-templates/.ai/= + mirror sync contract; task-audit needs an explicit spec-to-task binding before it can police =DOING= specs; =spec-sort --apply= needs a failure-safe/rollback contract; the org-id Emacs prerequisite must be executable before link conversion; and lifecycle status confirmation must be evidence-based so the retrofit does not encode stale reality.
+
+*** 2026-07-01 Wed @ 22:46:52 -0400 Second responder pass — all fourteen findings closed
+Fixed Codex's five re-review findings: the canonical-placement contract now opens the retrofit section (helper + tests + workflow edits land in claude-templates first, sync-check --fix propagates, sync-check-clean is an acceptance criterion); spec-response stamps a =:SPEC_ID:= property on the build parent, and task-audit's query checks that parent's keyword — which dissolves the flip-task chicken-and-egg; =--apply= got the fail-safe contract (clean-tree preflight, validate-then-write from a recorded plan, named recovery recipe); id-link conversion is staged (pilot rewrites =file:= links only; =id:= conversion is a follow-up gated on the concrete .emacs.d id-index mechanism — Craig picked this fork); and status confirmation is evidence-based (evidence panel, conservative non-terminal default, terminal states need a stated reason). Also de-cookified bracket tokens in prose that org's cookie updater would mangle. Status stays DRAFT; the READY flip belongs to the reviewers this round — verify pass dispatched.
+
+*** 2026-07-01 Wed @ 23:22:50 -0400 Codex spec-review rerun — Ready
+Codex re-read the revised [[id:80b0787b-4a60-4c82-8a16-b383d3e3c8f2][docs lifecycle spec]] after the second responder pass. All fourteen findings are closed, decisions remain [5/5], and the remaining implementation contracts are concrete enough to build and test. Status flipped to READY in the spec; implementation can proceed.
+
+*** 2026-07-01 Wed @ 23:34:15 -0400 Decomposed into build tasks; spec flipped READY → DOING
+spec-response Phase 6 run: this parent now carries the =:SPEC_ID:= binding (the spec's status-heading UUID), the phase tasks below track the build, and the spec's status heading is DOING. Completeness pass done: all ten acceptance criteria have homes across the phase tasks; vNext (org-agenda view) was already filed as the [#D] task below.
+
+*** 2026-07-01 Wed @ 23:39:10 -0400 Phase 1 landed — docs-lifecycle rule + four spec-workflow updates
+claude-rules/docs-lifecycle.md written and linked machine-wide (make install). Canonical-side updates: spec-create Phase 5 + template (docs/specs/ location, two-sequence keyword header, DRAFT status heading with :ID:, transition mechanics), spec-review (location expectation with the legacy compatibility rule keyed on :LAST_SPEC_SORT:, plus the DRAFT→READY flip — and the demote-back-to-DRAFT path a failed re-review takes), spec-response Phase 6 (owns READY→DOING, stamps :SPEC_ID: on the build parent, always emits the flip-to-IMPLEMENTED task), task-audit Phase B (the :SPEC_ID: reconcile query, checking the parent's keyword rather than counting children). Mirror synced; make test green end to end.
+
+*** 2026-07-01 Wed @ 23:57:44 -0400 Phase 2 landed — spec-sort helper + 30-test bats suite
+Built claude-templates/.ai/scripts/spec-sort (Python, TDD — the 30-test bats suite written red-first in claude-templates/.ai/scripts/tests/spec-sort.bats) covering the full retrofit contract: spine classification with the -spec.org-name-without-spine anomaly case, evidence panel (Status field, cookies, linking todo.org task, dated history, artifact existence) with conservative non-terminal proposals, per-candidate --confirm/--skip gate with --reason required on terminal keywords, clean-worktree preflight (--allow-dirty prints what recovery loses), validate-then-write from a recorded plan file, relink across the rewritten roots (inbound AND the moved doc's own outbound relative links) with report-only for sessions + synced templates (naming the canonical claude-templates file), bare-path mentions blocking until --acknowledge-bare, named recovery on injected mid-apply failure, post-apply residue gate, idempotent :LAST_SPEC_SORT: stamp. Real-data dry run against rulesets' pile matched predictions: 5 candidates, 4 anomalies, 30 notes, 0 bare, 10 report-only (incl. the startup.org synced-template case Codex flagged). make test green; sync-check clean.
+
+*** 2026-07-02 Thu @ 00:18:28 -0400 Phase 3 pilot ran — rulesets' pile sorted, board live
+Craig confirmed all five proposed keywords as-is plus the IMPLEMENTED reason; spec-sort --apply moved the five specs to docs/specs/ (agent-knowledge-base IMPLEMENTED, inbox-workflow-consolidation READY, autonomous-batch-execution READY, encourage-kb-contribution READY, wrapup-routing DOING — joining the docs-lifecycle spec's DOING on the board), rewrote 12 todo.org links plus the moved specs' own outbound links, and stamped :LAST_SPEC_SORT: 2026-07-02. Acceptance verified: status board matches reality, all re-homed specs carry -spec.org, residue zero in the rewritten roots (one acknowledged self bare mention rode along inside inbox-workflow-consolidation-spec), no id: links emitted, make test green. Surfaced and left in place: the four -spec.org-named files in docs/design without a spec spine (generic-agent-runtime, pattern-catalog, daily-prep-template, auto-triage-intake) — notes by predicate, misleading names; rename or leave is a Craig call. Report-only references: 9 frozen session archives + the synced startup.org (canonical edit lands with Phase 4's nudge work).
+
+*** 2026-07-02 Thu @ 00:23:32 -0400 Phase 4 landed — startup nudge live, .emacs.d notified
+Added the spec-sort probe to startup.org Phase A (item 12) and the one-line nudge to Phase C's findings list, canonical-side, mirror synced. One refinement over the spec's sketch: the stray-root check uses find instead of compgen, because compgen is bash-only and zsh aborts on an unmatched glob — the original snippet false-negatived on stray root specs under zsh (spec snippet updated with a note). Fixture-verified in both shells: fires on an unsorted docs/design and on a stray docs/*-spec.org, silent with the marker stamped, silent with no docs at all. Also fixed startup.org's own stale reference to the moved encourage-kb-contribution spec (the pilot's report-only finding). Sent .emacs.d the convention-live note with its ~28-doc pile nudge and the id-index ask (org-id-extra-files enumeration or periodic org-id-update-id-locations, verify by clicking the docs-lifecycle spec's :ID:), asking it to tag the owning task :blocker: since rulesets' id-conversion task waits on it.
+
+*** 2026-07-02 Thu @ 05:12:33 -0400 Converted spec-target file: links to id: form (rulesets)
+All 13 file:docs/specs/ links lived in todo.org (zero in .ai/ or docs/ outside specs). 11 converted straight to [[id:UUID][label]] (bare links labeled with the spec filename); the 2 links carrying a ::*Review findings search target got full fidelity by minting an :ID: on that heading in the docs-lifecycle spec (cc77a7f6-e4c3-488a-ac3b-e739420a5c2b) — the id index scans whole files, so heading-level ids resolve. Residue grep zero; every id verified against its target's :ID:. Gate had cleared earlier tonight via .emacs.d's org-spec-links.el delivery (verified org-id-find on their side); M-x cj/org-id-refresh-spec-locations is the fix if a fresh id doesn't resolve on click.
+
+*** 2026-07-04 Sat @ 11:46:31 -0500 Flipped the spec to IMPLEMENTED
+All four build phases had shipped (docs-lifecycle rule + spec-workflow updates, spec-sort helper + 30-test bats suite, rulesets pilot + status board, startup nudge) plus the conversion of file-style links to id-style links, so the spec's status heading went DOING → IMPLEMENTED with a dated history line and the Metadata mirror, per the transition-ownership table. Parent closed; manual validation promoted to its own task.
diff --git a/claude-rules/commits.md b/claude-rules/commits.md
index a3ec0f2..3283b0a 100644
--- a/claude-rules/commits.md
+++ b/claude-rules/commits.md
@@ -9,6 +9,7 @@ Claude, Claude Code, Anthropic, or any AI tool. Git uses the configured
`user.name` and `user.email` — do not modify git config to attribute
otherwise.
+
## No AI Attribution — Anywhere
Absolutely no AI/LLM/Claude/Anthropic attribution in:
@@ -19,6 +20,7 @@ Absolutely no AI/LLM/Claude/Anthropic attribution in:
- Code comments
- Commit trailers
- Release notes, changelogs, and any public-facing artifact
+- Document author metadata — an org `#+AUTHOR:` line, YAML frontmatter `author:`, a docx or PDF author property, a byline
This means:
@@ -32,133 +34,34 @@ If a tool, template, or default config inserts attribution, remove it. If
settings.json needs it, set `attribution.commit: ""` and `attribution.pr: ""`
to suppress the defaults.
-## Commit Message Format
-
-Commit messages follow the [Conventional Commits](https://www.conventionalcommits.org/) spec.
-
-### Structure
-
- <type>[optional scope]: <description>
-
- [optional body]
-
- [optional footer(s)]
-
-### Types
-
-- `feat:` — new feature (correlates with MINOR in SemVer)
-- `fix:` — bug fix (correlates with PATCH in SemVer)
-- `refactor:` — code restructuring, no behavior change
-- `perf:` — performance improvement
-- `test:` — adding or updating tests
-- `docs:` — documentation only
-- `style:` — formatting, whitespace, missing semicolons (no code-behavior change)
-- `build:` — build system or external dependencies
-- `ci:` — CI configuration and scripts
-- `chore:` — anything else: tooling, meta, housekeeping
-
-The Conventional Commits spec doesn't mandate the type list. Add a new type only when the existing ones genuinely don't fit and the team will agree on what it means.
-
-### Scope
-
-A scope MAY follow the type, in parentheses, naming the affected area of the codebase: `feat(parser): add ability to parse arrays`. Use a single noun.
-
-### Breaking changes
-
-Either append `!` after the type or scope, or include a `BREAKING CHANGE:` footer (uppercase — required). Both at once is fine and adds detail. `!` alone is enough.
-
- feat!: drop support for Node 6
-
- BREAKING CHANGE: uses JavaScript features not available in Node 6.
-
-### Subject line
-
-Imperative mood. ≤72 characters. No trailing period. The full subject is `<type>[scope]: <description>` — the 72-char limit covers the whole thing.
-
-### Body
-
-Optional. Begins one blank line after the subject. Free-form, multiple paragraphs allowed. Don't hard-wrap body lines — write each paragraph and each bullet as a single logical line and let the renderer (GitHub, Linear, `git log`) soft-wrap. Hard wraps shrink the visible render width in web UIs and cause awkward mid-sentence breaks. The same soft-wrap rule applies to PR bodies.
-
-Skip the body when the subject line covers the change.
-
-### Footers
-
-Optional. One blank line after the body. One per line. Format: `Token: value` or `Token #value` — the git trailer convention. The token uses `-` in place of whitespace (e.g. `Reviewed-by`, `Refs`, `Acked-by`). `BREAKING CHANGE:` is the one token allowed to contain a space, and `BREAKING-CHANGE:` is treated as a synonym.
-
-### How to write the message
-
-Write commit messages as if you're explaining the change to someone debugging a failure six months from now. Focus on what changed and why, not the play-by-play of how you typed it. Short imperative summaries like "Validate input before processing" age better than diary-style notes.
-
-The body, when you need it, is where context belongs — the constraint, bug, or tradeoff that forced the change. Over time the body becomes a lightweight decision log, which is more valuable than perfectly formatted messages.
-
-Commit messages describe what changed and why, not the process that produced the change. Don't reference code review, linting, test runs, or other workflow steps in the body (e.g. "from local review," "review surfaced," "flagged by reviewer"). Reviewers and future archaeologists want the what and the why. How you got there belongs in the PR discussion, not the commit.
-
-### Examples
-
-**Subject only:**
+### Generated documents carry the human author only
- docs: correct spelling of CHANGELOG
+Every document an agent writes or updates — a daily prep, a session log, a
+spec, an explainer, a meeting summary — names the human as its sole author.
+`#+AUTHOR: Craig Jennings`, never `Craig Jennings & Claude`, and never an
+agent's name standing alone.
-**With scope:**
+The failure mode is imitation, not intent. No template stamps this line;
+agents copy the author header from whatever sits nearby — yesterday's prep,
+the workflow file they're reading, the doc next to the one they're writing.
+So a single stray `& Claude` propagates through everything generated
+afterward, and no amount of fixing individual files stops it. Fix the files
+an agent reads, and state the rule here.
- feat(lang): add Polish language
+The stakes are highest where the rule is least visible. A private notes repo
+tolerates a co-author line; many employers do not, and their policy is that
+work product carries employee names alone. An `#+AUTHOR:` line survives
+conversion into docx, a published wiki page, or a PDF handed to a customer,
+so a header written in a scratch document three months ago can surface inside
+a deliverable. Write the header correctly at creation.
-**With body and footer:**
+Two things this rule does not ask for. Don't rewrite historical records —
+archived session logs and past dated documents stay as they are, because
+they're a record of what happened rather than a live artifact. And don't
+relabel a document another agent genuinely authored: if Codex wrote it, the
+byline stays Codex. The rule removes false co-authorship, not true
+authorship.
- fix: prevent racing of requests
-
- Introduce a request id and a reference to the latest request. Dismiss incoming responses other than from the latest request.
-
- Remove timeouts which were used to mitigate the racing issue but are obsolete now.
-
- Refs: #123
-
-**Breaking change with `!`:**
-
- feat(api)!: send an email to the customer when a product is shipped
-
-**Breaking change in footer:**
-
- feat: allow provided config object to extend other configs
-
- BREAKING CHANGE: `extends` key in config file is now used for extending other config files.
-
-## Voice and Focus
-
-Applies to commit bodies, PR descriptions, and PR comments (review replies, follow-up notes, thread responses).
-
-**Write as if to a colleague.** The reader is a teammate who'll see this in `git log`, a PR feed, or a Linear thread. "I" is allowed where natural. Don't sound abstract — name the file, the function, the constraint, the symptom. Press-release voice ("This change improves...") and committee voice ("It is recommended that...") both come out. The message has to read like one engineer talking to another, not like a generated artifact.
-
-**No felt-experience narration.** Don't tell the reader how the change will feel or how often you'll use it. Phrases like "I'll feel this every time I commit", "this will be a relief", "I'm excited about" — these read as performance, not communication. State what changed and let the reader decide what to do with it.
-
-**Don't noun-ify verbs.** "The ask", "a learn", "a reveal", "the spend", "a build" — use the real noun: "the request", "the lesson", "the finding", "the budget", "the system". Verb-as-noun reads as corporate-speak and makes the sentence feel performed.
-
-**No sentence fragments in prose.** Every prose sentence needs a subject and a verb. "Two changes." or "Fix incoming." or "Body as decision log." read as bullet-list shorthand even when they're standing alone in a paragraph. Bullets and headings can be fragments — prose sentences cannot.
-
-**"I" is the author, not the user.** First person is for what *I* did or decided in this commit ("I dropped the legacy fallback because..."). It's not for describing how the software or rule behaves for whoever uses it next. "The dialog only opens if I ask" is wrong when the rule is read by someone else — that "I" becomes ambiguous. Use third-person or passive for behavior: "opens on request", "opens when asked", "opens when the user invokes it". Code and systems are the actor; "I" stays for decisions.
-
-**First person where it fits.** When the subject is you or a decision you made, use "I" ("I added X", "I kept the parameter as `Any` because..."). When the subject is a team decision or shared rationale, "we" fits. When another author's prior work is the subject, name them ("Kostya's PR #116 did X"). Third-person constructions like "This PR introduces X" or "This change restores Y" read as press-release self-narration. The commit *is* the change, so don't announce it. Code and systems can stay third-person when they're the actor ("the guard rejects...", "the serializer returns...") — first person is for describing what you did or decided, not for narrating how the code behaves.
-
-**Brief. Terse is preferred.** A one-sentence body beats a paragraph saying the same thing. If the subject line covers it, skip the body entirely. Cut every clause that restates what the diff or the PR card already shows. Length is not a proxy for care. Rhetorical padding ("worth noting", "it's important to understand") always comes out; keep what a reader will actually use.
-
-**Follow-up approvals stay terse.** A re-review that just confirms prior CHANGES_REQUESTED feedback got addressed should be `Approved.` and nothing more. The fixes are visible in the diff and in the prior review thread, so restating them adds noise. The first round of substantive review gets a real comment. Subsequent sign-offs after fixes do not. Counts as a trivial one-liner under the Step 2 exception, so the draft-file flow can be skipped.
-
-**Kind.** PR comments and review replies are directed at a specific person. Acknowledge them when it fits ("thanks for the review") without pouring it on. When you disagree or push back, frame it as your read rather than a correction ("I think...", "my read was...", "did you mean X?"). Leave room for the other person to have seen something you didn't. A polite question beats a defensive explanation. Kindness is free and makes the next review cheaper.
-
-Focus on what was wrong and what was corrected. Not the mechanics.
-Readers skimming `git log` or a PR want the before-state, the
-after-state, and the reason. They don't need a TypeScript-variance
-lesson, a compiler-inference walkthrough, or a trip through an API's
-internals. Keep the "why" to one sentence unless a subtle invariant
-genuinely needs more.
-
-Don't stack technical terms. A sentence that chains three or more type
-signatures, API names, or compiler concepts reads as a jargon wall.
-Break it into shorter sentences and translate to reader-facing
-language. "The mock returns `Promise<Mission>`, so the resolver's
-argument is `Mission`, not `unknown`" beats the full inference chain
-that produces that signature. Keep the terms a reader will grep for,
-drop the ones that name compiler internals.
## Content scope for public artifacts
@@ -185,277 +88,56 @@ Don't write "per `testing.md`, integration tests must hit a real DB" or "the rul
Edge case: when one of these files *is* the change (a commit in the rulesets repo, an edit to a project's `CLAUDE.md`), describe what changed and why without invoking the wider personal-rules layer around it. The commit can absolutely say "tighten testing rule for legacy code". It shouldn't say "per the personal-rules layer this file is loaded into…".
-Different artifact types carry different content. Don't duplicate.
-
-**PR descriptions:** four sections, in order.
-
-1. **Problem** — what's wrong, with enough detail that a teammate can
- recognize the same failure mode in their own work.
-2. **Fix** — what changed.
-3. **Why this fixes it** — causal link, one or two sentences.
-4. **How it was tested** — skip for proposals, specs, or discussions;
- required for shipped fixes.
-
-The PR is the technical artifact. It carries the detail.
-
-If the project's publishing overlay defines a ticket system, see it for
-ticket-body conventions (a ticket body is typically just the Problem and
-Fix, with the causal why and test verification left to the PR).
-
-**PR review comments** are conversational and don't follow this
-structure — they follow the Voice and Focus rules above.
-
-Verbose preambles, motivational language, and context unrelated to the
-problem belong out. Same conciseness pressure as commit-message bodies.
-
-## Review and Publish
-
-Commits and PRs are team-visible, permanent, and hard to amend once shared
-(especially after push or after a reviewer has replied). Before executing
-`git commit` or `gh pr create`, the change must pass a local code review
-*and* the message must be reviewed by the user. The flow has three steps, in
-order.
-
-### Step 0: pre-flight reconcile (mandatory)
-
-Before reviewing the diff, fetch from the remote and reconcile against the
-upstream of the current branch. Reconciliation can change the working state
-when a rebase brings in upstream commits that touch staged files, and that
-would invalidate Step 1's review. Handling drift first means the review and
-the commit message describe the post-reconcile state.
-
-1. Fetch all remotes:
-
- git fetch --all --prune
-
-2. If the current branch has no upstream (new branch, never pushed), skip
- to Step 1 — there's nothing to reconcile against, and the first push
- sets the upstream.
-
-3. Otherwise, check divergence against `@{u}`:
-
- git rev-list --left-right --count @{u}...HEAD
-
- Output is `<behind>\t<ahead>`. Decide based on the pair:
-
- - **0 behind, anything ahead** — no-op. Continue to Step 1.
- - **Behind only, clean tree** — fast-forward: `git merge --ff-only @{u}`.
- - **Behind only, dirty tree** — surface to the user. Don't auto-stash or
- auto-merge. Offer to commit or stash first, or skip the reconcile and
- proceed knowing the push may need attention later.
- - **Diverged (behind AND ahead)** — surface to the user. Ask whether to
- rebase the local commits onto upstream (default for feature branches),
- merge the upstream branch in (rare; preserves both lines), or skip and
- proceed with the divergence. Don't auto-rebase.
-
-4. **PR flow only.** Also fetch the base branch (usually `main`) and check
- whether the feature branch's base is behind. Surface this informationally;
- don't auto-rebase the feature branch without asking. The "X commits
- behind base" badge on the PR is a follow-up decision, not a reason to
- block publish.
-
-The startup workflow's `git fetch --all --prune` doesn't substitute for
-Step 0. Upstream can advance during a long session, especially across
-machines or with teammates pushing in parallel. Run Step 0 every time the
-publish flow starts.
-
-### Step 1: local code review (mandatory)
-
-Run the `review-code` skill against the change:
-
-- Before a commit: `/review-code --staged`
-- Before a PR: `/review-code` (branch diff against `main` merge-base)
-- Before commenting on someone else's PR: `/review-code <PR#>`
-
-Surface **all** findings to the user: Critical, Important, and Minor.
-
-**Default block:** any Critical or Important finding stops the flow. Fix the
-issues and re-run `/review-code` until the diff is clean. Minor findings are
-shown but do not block.
-
-**Override:** the user can bypass the block with an explicit "proceed anyway"
-(or equivalent wording). Without the explicit override, do not proceed to
-Step 2.
-
-The `review-code` skill already has a Phase 0 eligibility gate that handles
-trivial and ineligible diffs (whitespace-only, revert with obvious
-justification, already-reviewed SHA). Trust that gate; there is no "trivial
-enough to skip review" exemption on top of it.
-
-### Step 2: draft, review, publish
-
-**Voice patterns and the approval gate are two independent decisions.** Don't bundle them.
-
-*Voice patterns are always personal for publish artifacts.* Commit messages, PR titles + bodies, and PR review comments all go out under the user's name, so they always run through `/voice personal` (the full pattern walk — general + Craig's-voice + the artifact-mechanics patterns: first-person rewrite, public-artifact scope flag, praise/correction asymmetry, finding stems), regardless of whether `.ai/` is tracked. These three are personal-voice artifacts by definition — the skill's personal mode exists for exactly them. Pattern #39 (public-artifact scope flag) matters *most* on team-visible artifacts, so it must never be skipped on a PR comment or PR body. There is no "general-voice mode" for publish artifacts.
-
-*The approval gate is the only thing `.ai/`-tracking decides.* Before drafting, run this command:
-
-```
-git ls-files :/.ai/ 2>/dev/null | head -1
-```
-
-The `:/` pathspec anchors the search to the repo root, so the command works from any subdirectory. Without it, running from a subdir returns no matches even when `.ai/` is tracked at the repo root, which silently misclassifies the project.
-
-- **No output** — `.ai/` is gitignored, missing, or empty (the user's personal repos). **Gate applies**: write to `/tmp`, run `/voice personal`, print inline, ask approve / request changes / open in editor, then publish only on explicit approval.
-- **Any output** — one or more files under `.ai/` are tracked (a shared / team repo). **Gate skipped for velocity**: write to `/tmp`, run `/voice personal`, print inline, publish immediately.
-
-Either way the draft runs through `/voice personal` first. The subflows below describe the full gated path. For the gate-skipped path, run the same `/voice personal` pass, then collapse the "Ask: approve, request changes, or open in editor" step — the draft prints inline and the publish step runs immediately afterward.
-
-**For commit messages:**
-
-1. Write the proposed message to `/tmp/commit-<short-slug>.md`.
-2. Run `/voice personal` on the file. Always. The skill walks its full pattern list covering signs of AI writing, universal good-writing rules (Strunk & White, Orwell, Plain English, Garner), and Craig's voice patterns (first-person rewrite, semicolons → periods/commas, contractions, sentence-split on conjunctions, felt-experience cut, sentence-fragment rewrite, terse cut for rhetorical padding, no-emphasis-formatting, public-artifact scope flag, praise/correction asymmetry, finding stems). The commit subject line stays imperative per Conventional Commits — `/voice personal` rewrites the body, not the subject. Skip the pass for purely mechanical commits (a chore version bump, a typo fix) where the subject alone carries the message.
-3. Print the final draft inline in the terminal. Every line, exactly as it'll be committed. No truncation, no summary. State that the skill ran (e.g. "/voice personal — full pattern walk"). If pattern #39 (public-artifact scope) flagged anything, surface those warnings; the user resolves them manually.
-4. Ask: approve, request changes, or open in editor. Wait for an explicit answer. Do not open the file in `emacsclient` (or any editor) by default — print first, edit only if asked.
- - **Approve** → commit with `git commit -F /tmp/commit-<short-slug>.md`.
- - **Request changes** → make them, re-run `/voice personal`, re-print inline, ask again.
- - **Open in editor** → only if the user asks. `emacsclient -n /tmp/commit-<short-slug>.md`. After the editor closes, re-read the file, re-print the contents inline, and ask again.
-
-**For PR descriptions:**
-
-1. Write the title as line 1 and the body below it to `/tmp/pr-<slug>.md`. **Title format:** the conventional-commit subject (`refactor: remove dead if-count-is-not-None check in admin`). If the project defines a publishing overlay with a ticket system, follow it for the ticket suffix in the title and the cross-link line in the body (see the overlay).
-2. Run `/voice personal` on the file. The PR title stays imperative per Conventional Commits — `/voice personal` rewrites the body, not the title.
-3. Print the final draft inline in the terminal. Title on line 1, blank line, then body — exactly as it'll be posted. State that the skill ran. Surface any pattern #39 (public-artifact scope) warnings.
-4. Ask: approve, request changes, or open in editor. Wait for an explicit answer. Do not open the file in `emacsclient` (or any editor) by default.
- - **Approve** → continue to step 5.
- - **Request changes** → make them, re-run `/voice personal`, re-print inline, ask again.
- - **Open in editor** → only if the user asks. `emacsclient -n /tmp/pr-<ticket-or-slug>.md`. After the editor closes, re-read the file, re-print inline, ask again.
-5. Split the file on the first blank line and pass the title and body to `gh pr create --title "..." --body "$(tail -n +3 <file>)"` (or a heredoc) so formatting is preserved. Add `--reviewer <user[,user...]>` in the same call when you already know who should review.
-6. Request reviewers on the new PR if you didn't pass `--reviewer` at create time. Use `gh pr edit <N> --add-reviewer <user>`. If the repo has a `CODEOWNERS` file, GitHub auto-suggests based on touched paths. Still issue the explicit request so the reviewer gets notified. Pick reviewers per the team's convention for the area touched (often documented in the per-repo `CLAUDE.md`). For follow-up PRs, consider tagging the parent PR's author if their context would help. PRs without a human reviewer request stall — "checks passed" is not a substitute for review.
-7. **Project publishing overlay (if present).** If the project defines a publishing overlay — a `publishing-<team>.md` rule loaded from its `.claude/rules/` — run its post-create steps now: ticket cross-linking, ticket-state moves, and any other tracker integration it specifies. A project with no overlay skips this; the PR is already open and reviewers are requested, which is the complete universal flow.
-
-**For PR review comments and replies (review verdicts, threaded discussion, follow-up notes on someone else's PR or your own):**
-
-Pick the shape first. Most reviews are Shape 1.
-
-- **Shape 1 — Single review** (verdict + summary body + 0+ inline pins). The default for any post that carries a verdict (`APPROVE`, `REQUEST_CHANGES`, `COMMENT`), even when the verdict has no line-specific findings. One `gh api` call posts the summary, every inline pin, and the verdict together. review notification fires once for `APPROVE` or `REQUEST_CHANGES`.
-- **Shape 2 — Issue-thread comment** (no verdict). General PR discussion, not a review. No inline pins. No review notification.
-- **Shape 3 — Reply on an existing inline thread**. Responding to a specific prior reviewer comment. Threads under that comment. No review notification.
-
-**Inline threshold for Shape 1.** Any finding that names a `path:line` belongs as an inline comment pinned to that line. Cross-cutting observations (verdict rationale, "third PR with the same pattern", overall test-coverage gaps that don't pin to one place) stay in the summary body. There's no "fold one inline into the summary" exception — a single line-specific finding still goes inline.
-
-**Shape 1: Single review (bundled summary + inline)**
-
-1. Identify findings, split into **inline-eligible** (each names a specific `path:line`) and **summary-only** (cross-cutting). Decide the verdict.
-
-2. Write one concatenated draft to `/tmp/pr-<N>-review.md` with explicit separators:
-
- ```
- === SUMMARY ===
- <verdict summary body>
-
- === INLINE path=frontend/src/foo.tsx line=440 ===
- <inline body 1>
-
- === INLINE path=frontend/src/bar.tsx line=137 ===
- <inline body 2>
- ```
-
- The separator format is exactly `=== SUMMARY ===` and `=== INLINE path=<path> line=<n> ===`. The summary block is mandatory even for verdict-only reviews. Inline blocks are zero-or-more.
-
-3. Run `/voice personal` on the file once. The skill walks its full pattern list across every block at the same time. The separators stay intact because they aren't prose.
-
-4. Print the final draft inline in the terminal. Every block, exactly as it'll be posted, with its separator header. State that the skill ran (e.g. "/voice personal — full pattern walk across summary + 3 inline"). Surface any pattern #39 warnings.
-
-5. Ask: approve, request changes, or open in editor. Wait for an explicit answer. Do not open the file in `emacsclient` (or any editor) by default.
- - **Approve** → continue to step 6.
- - **Request changes** → make them, re-run `/voice personal` on the whole file, re-print inline, ask again.
- - **Open in editor** → only if the user asks. `emacsclient -n /tmp/pr-<N>-review.md`. After the editor closes, re-read, re-print inline, ask again.
-
-6. Split the file on the separator lines and post in **a single** `gh api` call:
-
- ```
- gh api repos/<owner>/<repo>/pulls/<N>/reviews \
- --hostname <ghe-host-or-omit> \
- -F event=REQUEST_CHANGES \
- -F body="<summary block>" \
- -F "comments[][path]=<path1>" \
- -F "comments[][line]=<line1>" \
- -F "comments[][body]=<inline 1>" \
- -F "comments[][path]=<path2>" \
- -F "comments[][line]=<line2>" \
- -F "comments[][body]=<inline 2>"
- ```
-
- `event` is one of `APPROVE`, `REQUEST_CHANGES`, `COMMENT`. The `comments[]` array can be empty for verdicts with zero line-specific findings — the call still uses the same endpoint. Pass `--hostname` for non-`github.com` hosts (a project's publishing overlay names its host when it's a GitHub Enterprise instance).
-
-7. Verify the review landed. `gh api repos/<owner>/<repo>/pulls/<N>/reviews --hostname ...` returns the latest review with bundled inlines. Confirm `state` matches the verdict and the inline count matches what was posted.
-
-8. **Project review-notification overlay (if present).** If the project defines a publishing overlay with a review-notification step (e.g. a Slack ping to the PR author), run it now — but only for `APPROVE` and `REQUEST_CHANGES` verdicts. The overlay owns the channel, the message format, the author-mention lookup, and the threading. A project with no overlay skips notification entirely. `COMMENT` verdicts and Shapes 2-3 below never notify, overlay or not.
-
-**Shape 2: Issue-thread comment (no verdict)**
-
-Use when the post is informal discussion that shouldn't appear as a review verdict (e.g. "I'd like to discuss the X approach before you continue").
-
-1. Write the proposed comment to `/tmp/pr-<N>-comment.md`.
-2. Run `/voice personal`.
-3. Print inline, ask approve/changes/edit, gate as in Shape 1 step 5.
-4. Post: `gh pr comment <N> --body-file /tmp/pr-<N>-comment.md`.
-5. Verify: `gh api repos/<owner>/<repo>/issues/<N>/comments`.
-6. No review notification.
+**Tooling-path enumeration is the same leak.** Citing a rule as authority isn't the only way the tooling layer leaks into history. A commit whose *content* must name these paths — a `.gitignore` adding `.claude/`, `CLAUDE.md`, `.ai/` — has unavoidable, correct file content, but its *message prose* must not enumerate them ("chore: ignore .claude tooling, CLAUDE.md, and session files"). On a public or shared-remote repo that enumeration exposes the tooling layer's structure in the log just as a citation would. Name the category instead: "chore: extend gitignore for local tooling and build artifacts". The same holds for any incidental mention, not only `.gitignore` commits. Two exemptions: a commit whose change *is* one of these files (the edge case above), and private single-user repos with no shared remote, where the history is the project and there's no third party to leak to.
-**Shape 3: Reply on an existing inline thread**
-Use when responding to a specific prior reviewer comment.
+## Write in the first person, as Craig
-1. Find the parent comment ID: `gh api repos/<owner>/<repo>/pulls/<N>/comments`.
-2. Write the reply to `/tmp/pr-<N>-reply-<comment-id>.md`.
-3. Run `/voice personal`.
-4. Print inline, ask approve/changes/edit, gate as in Shape 1 step 5.
-5. Post: `gh api repos/<owner>/<repo>/pulls/<N>/comments -F in_reply_to=<comment-id> -F body="$(cat /tmp/pr-<N>-reply-<comment-id>.md)"`.
-6. Verify in the same `comments` list.
-7. No review notification.
+Everything authored in or about this repo is first person: code comments, commit
+messages, PR descriptions, PR review comments, and any note that lands in the
+repo or its history. State a choice as a choice — "I swept the copies rather
+than repairing them, because a drifted copy outranked the global rule" beats
+"the copies are swept rather than repaired."
-**Approve does not authorize a merge.** Reviewing a PR never authorizes merging it. Anything in `## Merge Strategy` below applies only to merges *you* are about to perform on your own branches — and even then, the merge needs its own explicit user confirmation per the rules there. A project's publishing overlay may add a team merge practice (e.g. approve-then-author-merges, where the review notification hands the merge decision to the PR author); that's an overlay concern, not a global one.
+**The "I" is Craig.** He is the author of record on every commit, comment, and
+review in his repos, and these artifacts go out under his name. So the voice is
+his, writing about his own work — not an agent narrating what it did on his
+behalf. Never write the agent into the prose as a separate party: no "Craig
+asked me to", no "I filed this for Craig", no "needs Craig's decision". Where a
+decision is still open, it is *his* open decision, written as "I haven't decided
+whether…" or "this needs a call I haven't made yet."
-**Exception:** trivial one-liners the user dictated verbatim in the
-conversation (e.g. "commit this as `chore: bump version`", "reply just
-'thanks for the review'") can skip the draft-file step in Step 2.
-`/review-code` in Step 1 still runs when it applies; Phase 0 of that skill
-handles trivial diffs, and acknowledgment-only replies don't need it at all.
+The same holds for anyone else's work. Name them ("Kostya's PR #116 did X"),
+because they are a third party. Craig is not.
-**Single-skill gate.** Each of the three subflows above runs `/voice personal` before printing the draft — the full pattern walk covering AI-writing signs, universal good-writing rules, Craig's voice patterns, and the artifact-mechanics patterns (first-person rewrite, public-artifact scope flag, praise/correction asymmetry, finding stems). Publish artifacts (commits, PR titles + bodies, PR review comments) always use personal mode; the `.ai/`-tracking check at the top of Step 2 decides only whether the approval gate fires, not which patterns run. Running the skill is mandatory; the printed draft must have been through it. When the user asks mid-flow for "the voice pass" on an in-progress draft, that means re-run the full pattern walk — not a subset. Always state that the skill ran when announcing the printed draft (e.g. "/voice personal — full pattern walk"). Skipping the pass without flagging it is a defect. The terse/omit-needless-words cut (pattern #38) is the *last* thing the skill does before the draft is printed: read each sentence and cut it in half, keeping only what changes meaning. The draft the user first sees must already be terse — if they have to ask for an Orwell pass after seeing it, the pass was skipped.
+Third-person constructions like "This change introduces X" or "This PR restores
+Y" read as press-release self-narration. The commit is the change, so it does
+not need announcing.
-**If `/voice` is unavailable.** The skill should be installed (it ships with rulesets), but a fresh or partial environment may not have it. Don't let that block the publish, and don't skip the discipline silently. Walk the same patterns inline — they're documented in the skill, and the publish flow already names which ones matter (first-person rewrite, semicolons → periods/commas, contractions, sentence-split, felt-experience cut, fragment rewrite, terse cut, the pattern #39 public-artifact scope flag, plus the AI-writing and good-writing passes). Then state that the skill was unavailable and the pass was applied by hand (e.g. "/voice unavailable — patterns walked inline"). The gate is the pattern walk, not the tooling; the skill is the convenient way to run it, not the only way. Flag the missing skill so it gets installed.
+**The one carve-out: code is the actor when describing behavior.** A comment
+saying *what the code does* stays third person, because the subject genuinely
+is the code and not me — "the sweep only fires when the global rule exists",
+"the guard rejects a malformed payload". First person is for the decision
+behind it, third person for the behavior itself. Both often belong in the same
+comment: what it does, then why I chose it.
-### Hook-level authorization
+This is the rule the publish flow already applied to commit bodies. It lives
+here because code comments get written constantly and the publish skill is not
+loaded then.
-The Step 1 code review plus the Step 2 user approval together constitute the
-authorization gate for the publish action. No separate hook-level approval
-prompt is needed on `git commit`, `gh pr create`, `git push`, or their
-variants once Step 2 has been approved. If a hook is configured, rely on the
-flow above to be the source of truth; do not treat the hook as a second
-independent gate.
+## The publish flow lives in the `publish` skill
-## Merge Strategy
+Everything about *how* a commit, PR, or review comment gets written, reviewed,
+approved, and published — the pre-flight reconcile, the code-review gate, the
+draft/voice/approval gate, conventional-commit format, Voice and Focus, PR
+description structure, the three review shapes, merge strategy, and the
+pre-commit checklist — is in the `publish` skill. Load it before drafting a
+message, not after: the flow gates what gets written.
-- *Squash-merge is the default* for feature branches. It avoids carrying
- WIP and fix-up commits into the target branch history and produces one
- logical change per merge.
-- State the planned merge approach (squash, rebase, or merge commit) and
- the target branch *before* pushing or merging. Wait for explicit user
- confirmation before `git push`, `gh pr merge`, or any equivalent. The
- Review and Publish flow above approves the *content*; merge strategy is
- a separate decision that needs its own confirmation.
-- *Pre-push reconcile.* Right before `git push`, do one more
- `git fetch <remote> <branch>` and verify the local branch is still
- ahead-only against its upstream. If something landed between Step 0 and
- push (review and draft together can take several minutes, and another
- machine or teammate may push in that window), surface and resolve before
- the push command runs. Catching drift here is cheaper than recovering
- from a failed non-fast-forward push under publish-step pressure.
-- Override the squash default only when there's a concrete reason: a
- clean per-commit review history the user has explicitly asked for, a
- multi-commit semantic narrative the team values, etc. Squash is the
- safe default; document why when deviating.
-
-## Before Committing
-
-1. Check author identity: `git log -1 --format='%an <%ae>'` — should be the user.
-2. Scan the message for AI-attribution language (including emojis and footers).
-3. Review the diff — only intended changes staged; no unrelated files.
-4. Confirm staged files belong in the repo: nothing that the project's policy keeps untracked (the personal-tooling set in gitignore-mode projects), and in repos with a canonical/mirror split, the edit is on the canonical side — a mirror-only edit gets reverted by the next sync.
-5. Run tests and linters (see `verification.md`).
+What stays here is what must hold whether or not anything is being published,
+and where a violation is permanent and reaches other people. If the skill fails
+to load you will have to be told the flow, which is recoverable. The rules
+below are not.
## If You Catch Yourself
diff --git a/claude-rules/cross-project.md b/claude-rules/cross-project.md
index ed4a19c..c5de962 100644
--- a/claude-rules/cross-project.md
+++ b/claude-rules/cross-project.md
@@ -35,7 +35,9 @@ Two acceptable outcomes:
```
Output filenames follow `YYYY-MM-DD-HHMM-from-<this-project>-<slug>.<ext>` automatically, so the target's next session sees the source + timestamp at a glance without you having to construct the name. Fall back to `Write`/`Edit` only when the script isn't available (e.g. a freshly-cloned project before the first startup-rsync).
-2. **"Switch projects"** — stop. Let the user reopen Claude in the right cwd.
+
+ The wrap-up cross-project router rides this same sanctioned path: at wrap time, `wrap-it-up.org`'s router step delivers `:ROUTE_CANDIDATE:`-tagged keeper tasks to their home projects' inboxes via `route-batch` → `inbox-send` (never a direct foreign `todo.org` write), and the destination's own inbox processing files each task per its conventions.
+2. **"Switch projects"** — stop. Let the user reopen the agent session in the right cwd.
Don't assume which one was meant. Either guess is wrong half the time and the cost of asking once is one short turn.
@@ -48,6 +50,14 @@ whose canonical home is `~/code/rulesets/`. When work in a downstream project
needs one of these files to change, a local edit alone is a stopgap that the
next sync reverts. The durable change happens only in the rulesets canonical.
+Installed global paths such as `~/.claude/rules/`, `~/.claude/hooks/`,
+`~/.claude/skills/`, and rulesets-owned entries in `~/.local/bin/` are
+symlinks into the canonical repository, not downstream copies. Never edit
+through those paths from another project's session: resolving the symlink
+would dirty rulesets outside its own logging and wrap discipline. The
+`rulesets-write-boundary.py` hook mechanically denies Edit/Write targets whose
+real path lands there and directs the proposal through `inbox-send rulesets`.
+
The process, every time:
1. **Make the change locally** in the downstream project so it's usable
diff --git a/claude-rules/daily-drivers.md b/claude-rules/daily-drivers.md
new file mode 100644
index 0000000..7ee5f06
--- /dev/null
+++ b/claude-rules/daily-drivers.md
@@ -0,0 +1,66 @@
+# Daily-Driver Machines
+
+Applies to: `**/*`
+
+Craig runs exactly two daily-driver machines: **ratio** and **velox**. They are
+kept in sync, and an important change made on one usually needs to reach the
+other.
+
+## The Rule
+
+When you make or notice a change that is **machine-level and important** —
+dotfiles, installed tooling, a synced repo's clone or timer setup, a global
+config, a systemd unit, a credential, a one-time bootstrap step — consider
+whether the *other* daily driver needs the same change, and flag it. Don't
+assume a change made on the current machine is live everywhere.
+
+Both machines are on the same tailnet, so the agent can usually reach the other
+one directly over tailscale ssh — it can sync, verify, or repair the other daily
+driver, not just flag the drift. Reach for that when a change needs to land on
+both boxes now. (This session repaired ratio's dotfiles and verified the fix
+over tailscale; the .emacs.d side has driven ratio the same way — `git fetch` +
+`reset --hard` and an `scp` across.)
+
+When tailscale is down or the other machine is offline, fall back to the
+original discipline: this is a prompt to think, and the point is to surface "the
+other daily driver may need this too" at the moment the change lands, so it
+doesn't silently drift to one box.
+
+## How the sync actually happens
+
+The mechanism depends on what changed:
+
+- **A tracked repo** (rulesets, dotfiles, a project) — the other machine just
+ needs a `git pull` (and, for rulesets, a `make install` to relink anything
+ new). Most changes are this.
+- **Dotfiles** — ride the dotfiles repo; the other machine picks them up on its
+ next stow/pull.
+- **A one-time setup** — a new repo clone, a new systemd timer, a freshly
+ installed tool, a credential — has to be done by hand on each machine. These
+ are the ones that silently drift, because nothing carries them automatically.
+
+When the change is the one-time kind, say so explicitly: name the manual step
+the other machine still needs.
+
+## Reaching the other machine over tailscale
+
+`tailscale status` lists every node with its tailscale IP and online state.
+Connect by tailscale IP (e.g. `100.71.182.1`) or MagicDNS name (e.g.
+`ratio.tailf3bb8c.ts.net`) — both always resolve and connect. A bare hostname
+(`ssh ratio`) works only when MagicDNS is configured on the local machine;
+without it the bare name can fail to resolve, which makes the box look
+unreachable when it isn't. Prefer the IP or the full MagicDNS name when in
+doubt. The first connection from a new address fails host-key verification under
+`BatchMode`; add `-o StrictHostKeyChecking=accept-new` to clear it.
+
+## Knowing which machine you're on
+
+`uname -n` returns the hostname (`ratio` or `velox`). Use it when a reminder is
+machine-specific ("on ratio, you still need to …") so the note is actionable
+rather than abstract — and after an ssh hop, to confirm which machine you landed
+on.
+
+## Current open instance
+
+None. (The org-roam knowledge-base clone + `roam-sync` timer is confirmed on
+both daily drivers, velox and ratio, as of 2026-06-30.)
diff --git a/claude-rules/desktop-capture.md b/claude-rules/desktop-capture.md
new file mode 100644
index 0000000..c4a67f9
--- /dev/null
+++ b/claude-rules/desktop-capture.md
@@ -0,0 +1,60 @@
+# Off-Workspace Windows and Captures
+
+Applies to: `**/*` (any task that opens a window or takes a screenshot on the user's live desktop)
+
+Never open a window or take a screenshot on the user's active workspace. When
+visual verification needs a real window on the user's live desktop, keep it off
+the workspace they're working in. An agent doing its own visual verification
+shouldn't hijack the desktop the user is actively using.
+
+The principle is environment-general: don't commandeer the user's active
+workspace for agent-side visual work. The recipe below is the Hyprland/Wayland
+implementation; other environments implement the same principle with their own
+off-screen mechanism.
+
+## Captures for your own verification
+
+Render and grab the window off the user's physical screen, then tear it down. On
+Hyprland this is a virtual headless output, verified non-disruptive on ratio
+2026-07-06 — the physical monitor stayed on its workspace, focused, throughout:
+
+```sh
+hyprctl output create headless # virtual output on its own workspace
+setsid <app> >/tmp/x.log 2>&1 </dev/null &
+addr=$(hyprctl -j clients | python3 -c 'import json,sys; print(next((c["address"] for c in json.load(sys.stdin) if c.get("class")=="<CLASS>"), ""))')
+hyprctl dispatch movetoworkspacesilent "<ws-on-headless>,address:$addr" # silent = keeps the user's focus
+grim -o HEADLESS-<n> /tmp/shot.png # capture the virtual output only
+pkill -f '<app>$'; hyprctl output remove HEADLESS-<n> # tear down, restore the display
+```
+
+Key constraint: `grim` captures a *visible output*, so a window merely parked on
+another Hyprland workspace can't be screenshotted — it must render on the
+headless (or another real) output. That's why a headless output, not just
+"another workspace," is the tool for self-captures. A nested compositor
+(weston/cage/sway) is the alternative on non-Hyprland Wayland or when a headless
+output isn't available; it needs the compositor installed.
+
+## Showing the user something
+
+Open it on a *separate* real workspace and tell them which one, so it never
+grabs their active workspace. They switch when ready. Craig's viewer preference
+is `imv`; launch it with `gui-open --image <file>` (dotfiles-shipped) rather
+than a bare `&` job or `hyprctl dispatch exec`: it detaches through `systemd-run
+--user` so the agent shell can't reap it, resolves the current Hyprland instance
+after a compositor restart, and confirms the viewer is mapped and visible before
+returning. An HTML render uses `gui-open <file>` (or `--browser`) the same way.
+If `gui-open` isn't on PATH, the machine needs a dotfiles pull.
+
+## Always clean up
+
+Close the window and remove any headless output afterward. Verify the user's
+display is restored: physical monitor back to its workspace, no orphan
+processes.
+
+## Related
+
+- `verification.md` — this is part of how visual verification is done without
+ disrupting the user.
+- `interaction.md` — the broader "don't disrupt the user's active work" concern.
+- `emacs.md` — the screenshot note for Emacs changes uses the same off-screen
+ capture approach.
diff --git a/claude-rules/docs-lifecycle.md b/claude-rules/docs-lifecycle.md
new file mode 100644
index 0000000..3906d86
--- /dev/null
+++ b/claude-rules/docs-lifecycle.md
@@ -0,0 +1,75 @@
+# Docs Lifecycle
+
+Applies to: `**/*` (any project carrying a `docs/` tree)
+
+How formal documents are separated from working notes, and how a document's
+lifecycle state stays visible without opening the file. Specs are the first
+instance of the shape; the pattern is reusable for any growing collection of
+processed artifacts. Full design: the docs-lifecycle spec in rulesets
+`docs/specs/`.
+
+## The shape (reusable)
+
+1. **Separate formal artifacts from working notes by location.** A formal
+ artifact proposes a buildable change and carries the full spine; everything
+ else is a note.
+2. **Lifecycle state lives in the artifact**, on a scannable, greppable
+ carrier — an org TODO keyword on a top-level status heading — with a dated
+ history of every transition.
+3. **Links use rename-safe identifiers** so a move or rename never orphans
+ inbound references.
+4. A collection **earns this treatment when "which of these are live?" starts
+ requiring a file-by-file read.**
+
+## The spec instance
+
+- `docs/specs/` holds formal specs only — a doc with both a `Decisions`
+ section and an `Implementation phases` section (the spec-create spine).
+ `docs/design/` holds everything else: brainstorms, proposals, inventories,
+ research notes, frozen source material. Spec filenames end `-spec.org`
+ (spec-review's precondition keys on it); no status suffixes ever.
+- Every spec opens with a top-level status heading directly after the file
+ header, carrying the lifecycle keyword, an `:ID:` UUID, and dated history
+ lines (newest first). The keyword header is two sequences, and both lines
+ are required:
+
+ #+TODO: TODO | DONE
+ #+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED
+
+ The first drives `Decisions` / `Review findings` tasks and their `[/]`
+ cookies; the second is the lifecycle. They share no keyword — never merge
+ them into one line, and never drop the first (that silently breaks the
+ cookies that gate readiness).
+- **The heading keyword is authoritative.** The Metadata table's `Status`
+ field mirrors it in lowercase; on disagreement the heading wins. A
+ transition is three lines in one file — keyword, history line, mirror — and
+ never a rename or a link edit.
+- **Every flip has a named owner:** spec-create stamps `DRAFT`; spec-review
+ flips `DRAFT` → `READY` on a passing gate; spec-response flips `READY` →
+ `DOING` when it decomposes phases into build tasks (stamping the spec's
+ UUID as a `:SPEC_ID:` property on the build parent, and always emitting a
+ final "flip the spec to IMPLEMENTED" task); task-audit flags any `DOING`
+ spec whose `:SPEC_ID:`-bound parent is closed, archived, or missing.
+ Terminal states (`IMPLEMENTED` / `SUPERSEDED` / `CANCELLED`) always carry a
+ stated reason in the history line.
+- **The status board is one grep:**
+
+ rg -H '^\* (DRAFT|READY|DOING|IMPLEMENTED|SUPERSEDED|CANCELLED) ' docs/specs/
+
+- **Legacy compatibility:** projects that haven't run the one-time `spec-sort`
+ retrofit (no `:LAST_SPEC_SORT:` marker in `.ai/notes.org` Workflow State)
+ keep their legacy spec locations reviewable; the `docs/specs/` requirement
+ hardens only after the sort runs.
+- **Cross-doc links to specs are `file:` links for now.** Specs carry `:ID:`
+ UUIDs, but conversion to `[[id:...]]` is a gated follow-up (the Emacs id
+ index has to know about project docs first) — don't convert links ad hoc.
+
+## Watch for
+
+- Editing the `#+TODO:` header down to one sequence — the `[/]` cookies stop
+ computing and readiness gates go vacuous.
+- Bare `[N/N]` tokens in prose or list items — org's cookie updater rewrites
+ them; spell counts out in words outside real cookie positions.
+- A "done" spec whose keyword still says `DOING` — that's the failure this
+ convention exists to prevent; flip it with a history line rather than
+ leaving it for the audit to catch.
diff --git a/claude-rules/emacs.md b/claude-rules/emacs.md
index 702b40e..846888d 100644
--- a/claude-rules/emacs.md
+++ b/claude-rules/emacs.md
@@ -1,3 +1,8 @@
+---
+paths:
+ - "**/*.el"
+---
+
# Working With Craig's Running Emacs
Applies to: `**/*.el` (and any task that edits Craig's Emacs configuration)
@@ -27,3 +32,27 @@ This re-evaluates the file and redefines its `defun`s live. For straight functio
3. Verify: for visual changes, screenshot and read it (the `screenshot.py` tool under `.ai/scripts/` can capture an app off-screen on a headless output); for behavior, eval or exercise it.
This replaces the quit → relaunch → re-find-and-load-files cycle for most edits. A real restart stays the gold standard for a guaranteed-clean state — anything touching `:config`, load order, or when in doubt.
+
+## Don't edit on disk a file the daemon is capturing into
+
+The reload caveats above are about pushing changes *into* the daemon. The inverse hazard: a tool that edits a file *on disk* while the daemon has an indirect buffer cloned from it. org-capture works through such a buffer, and a disk write (a hand edit, a `git pull` that fast-forwards the file, a `sed`/Write) reverts the base buffer underneath the capture. The capture is left on stale state, can no longer finalize with `C-c C-c`, and a freshly-typed item can be lost or written back against post-edit content. Orphaned `CAPTURE-*` buffers piling up as Craig retries is the visible symptom.
+
+The roam inbox (`~/org/roam/inbox.org`) is the live case — Craig captures into it constantly, and the inbox workflow's roam mode (Phase D) edits it. Before a disk write to a file the daemon may be capturing into, check first: `.ai/scripts/capture-guard <file>` exits non-zero (and names the buffer) when a live capture is cloned from `<file>`, and exits 0 — safe — when there's no capture or no reachable Emacs. Same principle as the reload rule, one layer out: leave the daemon's live buffers authoritative rather than yanking the file from under them.
+
+## SVG Rendering for Emacs App UIs
+
+Consider SVG rendering (`svg.el`) as the default for any Emacs app UI with real visual structure — boards, meters, gauges, status panels (Craig's directive, 2026-07-11). Verified on Emacs 30.2 pgtk via jotto's librsvg spike: `svg.el` cleanly renders rounded rects, linear gradients, fill-opacity, bold/italic/letter-spaced text, stroked shapes, and theme-derived colors.
+
+The payoff is the prototyping pipeline: design the UI as an HTML prototype first (per `ui-prototyping.md` — browser iteration), then port near-1:1 to `svg.el`, since browser SVG and librsvg share primitives.
+
+Constraints to design around:
+
+- **GUI-only.** No tty rendering — ship a second text renderer or accept GUI-only.
+- **The SVG region is an image, not text.** Input stays minibuffer/keymaps; no point-and-click into the drawing.
+- **Theme colors are pulled by hand.** Read them from faces at render time (`face-attribute`); the image inherits nothing.
+- **Whole-image regeneration per state change.** Design the render as a pure state → SVG function so regeneration is cheap to reason about.
+- **Stick to a librsvg-safe subset.** No SMIL, no JS-in-SVG, conservative CSS and filters. `feDropShadow` renders without error but is visually unverified — don't lean on it.
+
+Recommended shape: hybrid layouts — SVG for boards, meters, and visual state; text buffers for prose and transcripts.
+
+Prior art: archsetup's instrument-console widget gallery (`docs/prototypes/2026-07-03-panel-widget-gallery-prototype.html` in archsetup); jotto's spec records the full pipeline and constraint sheet.
diff --git a/claude-rules/host-identity.md b/claude-rules/host-identity.md
new file mode 100644
index 0000000..9f58392
--- /dev/null
+++ b/claude-rules/host-identity.md
@@ -0,0 +1,20 @@
+# Host-Identity Guard
+
+Applies to: `**/*` (any tracked or synced project file)
+
+Never assert mutable environment identity as a fixed fact in a file that git tracks or the template sync distributes. A `CLAUDE.md` or notes file claiming "This machine is ratio", a current OS version, an IP, or "the laptop" lands identical on every machine, so the claim is false everywhere but its origin — and an agent that trusts it reasons backwards the whole session.
+
+## The Rule
+
+- **Don't write fixed identity claims** — "this machine is X", "the current host is X", "we're on the laptop" — in `CLAUDE.md`, `notes.org`, rules files, or any other tracked/synced doc.
+- **Derive identity at runtime and name the command.** The correct phrasing in a doc is an instruction, not a fact: "run `uname -n` to find the hostname." (`uname -n` is the source of truth — the `hostname` binary is often absent, and `uname -r` is the kernel release, not the host.)
+- **Describing the fleet is fine; claiming the current member is not.** "The fleet is ratio (workstation) and velox (laptop)" is a durable fact and belongs in a doc (see `daily-drivers.md`). "This machine is ratio" is a snapshot that rots the moment the file syncs.
+- The same applies to any mutable environment fact: current OS release, current IP, current display topology. State how to derive it, not what it was when the file was written.
+
+## Worked failure
+
+archsetup, 2026-06-21: its `CLAUDE.md` asserted "This machine is **ratio**" as a fixed fact. A session running on velox reasoned from that line all session — skipping velox-only reminders as "not applicable, we're on ratio" — exactly backwards. The fix replaced the claim with the `uname -n` instruction.
+
+## Enforcement
+
+The startup workflow runs a read-only probe that greps `CLAUDE.md` and `.ai/notes.org` for fixed-identity phrasing and surfaces any hit as a startup finding. The probe flags for human judgment; it never blocks. When it fires, replace the claim with the runtime derivation, not a fresher snapshot.
diff --git a/claude-rules/interaction.md b/claude-rules/interaction.md
index 1fd0334..b5798bd 100644
--- a/claude-rules/interaction.md
+++ b/claude-rules/interaction.md
@@ -2,11 +2,50 @@
Applies to: `**/*`
-How Claude communicates with the user during a session — choice prompts, status updates, decision points.
+How the agent reasons with the user and communicates during a session — how interpretations are formed, then how choices, status, and decision points are presented.
+
+## Collaborative Peer Reasoning
+
+Treat the conversation as joint reasoning between peers. The user's words are evidence of intent, not merely a string to execute literally. Use the request, prior context, current state, and likely downstream consequences to form a working interpretation.
+
+### Infer first; clarify at material forks
+
+Infer the intended outcome and proceed when reasonable interpretations lead to the same action. When two plausible interpretations would produce materially different outcomes, strategies, or external effects, state the working interpretation and ask one focused question before crossing that fork. Do not interrupt for reversible implementation details that can be resolved with ordinary judgment.
+
+### Test conclusions before committing to them
+
+Do not promote the first plausible explanation or plan into a conclusion. Check the strongest reasonable alternative, test the assumptions that distinguish it, and weight the tradeoffs that decide between them. Separate verified facts, inferences, and recommendations when the distinction matters. Calibrate confidence instead of projecting certainty.
+
+For a consequential judgment, a useful response shape is: working interpretation, evidence, strongest alternative, recommendation, confidence, and the one clarification that would change the action. Do not force this scaffold onto simple tasks.
+
+### Corrections update the model
+
+Treat a user correction as new evidence that changes the working model. Reconcile its downstream implications immediately: assumptions, source selection, plans, scheduled actions, task state, and conclusions already reached. Do not reduce a substantive correction to a wording change or preserve stale premises silently.
+
+A correction is not automatically true merely because the user made it. If it conflicts with verified evidence, explain the conflict directly and identify what would resolve it. If the correction is supported, change course cleanly without defending the earlier answer.
+
+### Neither sycophantic nor adversarial
+
+Agreement follows evidence and reasoning, not deference. Disagreement serves the shared outcome, not the defense of a prior position. State a contrary view when it changes the decision, give the evidence behind it, and leave room for missing context. Once new evidence resolves the issue, stop arguing the old case.
+
+### Process serves the outcome
+
+Rules and workflows are constraints on the work, not substitutes for judgment. Apply them in service of the user's intended outcome. If a literal workflow interpretation produces a disproportionate, surprising, or strategically different result, surface that implication before proceeding.
+
+Failure signs:
+
+- Executing the narrow literal request while ignoring an evident intended outcome.
+- Asking about a reversible detail while failing to clarify a consequential fork.
+- Presenting one plausible account as settled without testing its strongest alternative.
+- Treating a correction as local wording while leaving downstream assumptions unchanged.
+- Agreeing to preserve rapport or resisting to preserve authority.
+- Letting procedural completeness overwhelm the value of the task.
+
+This is a working collaboration contract, not immutable wording. Refine the section as real sessions expose better distinctions or failure modes; route the feedback to the canonical file rather than accumulating project-local exceptions.
## No Popup Menus for Choices
-When Claude needs the user to pick between options, **do not** use the AskUserQuestion popup. Present the options inline in chat as a numbered list and ask the user to reply with a number.
+When the agent needs the user to pick between options, **do not** use the AskUserQuestion popup. Present the options inline in chat as a numbered list and ask the user to reply with a number.
**Why:** The popup menu UI sits at the bottom of the chat window and obscures the chat content directly above it — exactly the area the user needs to read to make the choice. Inline numbered options keep the question, the surrounding context, and the proposed text all visible in the same scrollback.
@@ -54,6 +93,17 @@ In conversational output to the user, do not use Markdown bold (`**...**`) or in
- Write command names, file paths, key chords, and code identifiers as plain text — `pearl-save-issue` becomes pearl-save-issue, `C-; L s s` becomes C-; L s s.
- Use structure that doesn't invert colors: headers, numbered lists, dashes, parentheses, and double quotes for labels are all fine.
-- Fenced code blocks (triple backtick) are acceptable when the user explicitly wants a block to copy — they don't invert the way inline spans do. Default to plain text otherwise.
+- Fenced code blocks (triple backtick) are not an exception. Craig's direction is zero markup in chat output, always: "always always list it out without markup" (2026-05-30). Fences don't invert the way inline spans do, but they still read as markup he didn't ask for, and the carve-out kept reintroducing them. When he needs something to copy, give it as plain indented text, or write it to a file and name the path.
This governs **chat output**, not the Markdown source of rule files, specs, or docs the user reads in an editor — those keep normal Markdown formatting. The constraint is the terminal rendering of the live conversation.
+
+## Showing Craig Visuals
+
+Craig runs Claude Code inside Emacs EAT (through tmux). EAT renders SendUserFile and inline terminal images as an `[image] path.png` text line — the visual itself never appears. In one session ~20 renders went out that way and Craig approved UI he had never seen (takuzu, 2026-07-11). Never rely on SendUserFile or inline image display to show a visual. SendUserFile stays fine for *delivering* a file; it just doesn't display one.
+
+Two display lanes, by what the visual is for:
+
+- **Durable or interactive visuals** — HTML prototypes, full renders, anything Craig will study or click: open with `gui-open <abs-path>` (dotfiles-shipped; detaches through `systemd-run --user` so the agent shell can't reap it, resolves the Hyprland instance after a compositor restart, and confirms the window is actually visible before returning). It auto-detects image vs HTML; `--image` / `--browser` force. Place it off Craig's active workspace and name it, per `desktop-capture.md`.
+- **Quick inline glances** — a chart, a did-it-render check: sixel in the terminal. Encode with ImageMagick (`magick <img> sixel:<out>`; img2sixel silently emits zero bytes on some builds) and display in a separate tmux window (`tmux new-window -d -n <name>`, then `tmux send-keys -t <name> "clear; cat <out>" Enter`, tell Craig the window name) rather than the Claude Code pane, whose TUI repaints over anything drawn into it. With native tmux sixel active, the image lives in tmux's grid and survives window switches, scrolling, and resizing.
+
+**Capability gate.** Native sixel needs two config pieces: EAT answering the XTWINOPS cell-size query (patched eat.el, owned by .emacs.d) and `terminal-features 'xterm*:sixel'` in tmux.conf (owned by dotfiles). Check before relying on it: `tmux display -p '#{client_cell_width}'` — nonzero means go; 0 means the chain is missing a piece, so fall back to the browser lane. (Diagnosed 2026-07-13: tmux won't transmit sixel until it knows the client's cell pixel size, and stock EAT 0.9.4 silently ignores the CSI 14 t query tmux uses to ask.)
diff --git a/claude-rules/knowledge-base.md b/claude-rules/knowledge-base.md
index 5658498..146a5e4 100644
--- a/claude-rules/knowledge-base.md
+++ b/claude-rules/knowledge-base.md
@@ -2,7 +2,7 @@
Applies to: `**/*`
-Craig's org-roam knowledge base is the shared, cross-project store for durable agent knowledge. It lives at `~/org/roam/` — a git repo (origin `git@cjennings.net:roam.git`), auto-synced on Craig's machines by the `roam-sync` systemd timer. Per-project harness memory stays the fast capture layer; durable facts get promoted here.
+Craig's org-roam knowledge base is the shared, cross-project store for durable agent knowledge. It lives at `~/org/roam/` — a git repo (origin `cjennings@cjennings.net:git/roam.git`), auto-synced on Craig's machines by the `roam-sync` systemd timer. Per-project harness memory stays the fast capture layer; durable facts get promoted here.
## Reading (any project)
@@ -22,13 +22,15 @@ Pull before querying (`git -C ~/org/roam pull --ff-only`); skip silently if offl
Classify the project before any write. The source of truth is the work-root denylist below — never inference from remotes, names, or task content:
- **Work** — project root is, or sits under, a denylisted root. No KB write, ever. Record durable facts per that project's own conventions.
-- **Personal** — project root sits under `~/code/`, `~/projects/`, or `~/.emacs.d` and is not denylisted. KB writes allowed.
+- **Personal** — project root sits under `~/code/`, `~/projects/`, `~/.emacs.d`, or `~/.dotfiles` and is not denylisted. KB writes allowed.
- **Unknown** — anything else. No KB write.
Work-root denylist (confirmed by Craig, 2026-06-10): `~/projects/work`
**Refusal contract** (work and unknown alike): state the classification, name the durable fact in a one-line redacted summary, and say where it was or wasn't written — so Craig can re-route it deliberately instead of losing it silently.
+**Scope of the denylist — durable KB-node writes only.** The work-denylist governs one thing: promoting a durable fact into a new `agents/` node. It is a confidentiality guard so work-confidential material doesn't land in the personal cross-machine store. It is *not* a general "don't touch roam from a work project" boundary. Roam is a *shared resource*, not another project's product scope. Reading it (any project) and *tidying the shared roam inbox* — processing, routing, and filing the capture items in `~/org/roam/inbox.org`, e.g. via inbox-zero — are allowed from any project session, work included; that is housekeeping on a shared resource, not a durable-fact write. Only the durable-node promotion stays work-denylisted. Do not park roam-inbox tidying as a cross-project boundary crossing (a sentry inbox-zero pass did exactly that on 2026-07-19 — the error this note closes).
+
A write is one node per fact, under `agents/`, roam-valid so Craig's org-roam indexes it:
```
@@ -43,7 +45,26 @@ A write is one node per fact, under `agents/`, roam-valid so Craig's org-roam in
<the fact, with [[id:...]] links to related nodes>
```
-Pull before writing, commit and push after (`git -C ~/org/roam add -A && git commit && git push`) — same session discipline as any repo. Never edit Craig's hand-authored nodes; link to them. This write autonomy is scoped to the KB alone — it is not permission to send email, comment on tickets, or post to any public or external channel.
+Pull before writing (`git -C ~/org/roam pull --ff-only`, read-only). Then acquire the roam-write lock, write the node, and trigger roam-sync to commit and push — roam-sync stays the roam repo's only committer (the 2026-06-24 one-git-owner rule). The tree is chronically dirty from live captures, so an agent's own `git add -A && commit` could sweep an in-flight capture into a stray commit; edit-plus-trigger avoids that. Never edit Craig's hand-authored nodes; link to them. This write autonomy is scoped to the KB alone — it is not permission to send email, comment on tickets, or post to any public or external channel.
+
+The write block, with the lock and the trigger:
+
+```sh
+# Acquire the roam-write lock so a concurrent sentry pass or inbox writer can't
+# race this write. Callers pass a name; agent-lock owns the path (tmpfs).
+if [ -x .ai/scripts/agent-lock ]; then
+ if ! .ai/scripts/agent-lock acquire roam-write --wait; then
+ # Busy after the bounded wait — surface and stop, don't write unlocked.
+ echo "roam-write lock held by another writer; try again shortly" >&2
+ exit 1
+ fi
+fi
+# ... write ~/org/roam/agents/<ts>-<slug>.org ...
+systemctl --user start roam-sync.service # roam-sync commits + pushes
+[ -x .ai/scripts/agent-lock ] && .ai/scripts/agent-lock release roam-write
+```
+
+Degrade gracefully when `agent-lock` isn't installed (an older checkout mid-sync): the guard above is skipped and the write proceeds unlocked — today's behavior. Only a *present* helper reporting the lock busy after its bounded wait stops the write; an *absent* helper never blocks it.
## What goes in, what stays out
@@ -53,7 +74,7 @@ Pull before writing, commit and push after (`git -C ~/org/roam add -A && git com
## Capture, then promote
-Harness memory (`~/.claude/projects/<enc>/memory/`) remains the per-project capture layer: fast, automatic, allowed to be at-risk. At wrap-up (or a task audit, or an explicit prompt), promote facts that meet the inclusion bar into the KB as nodes. The wrap-up workflow asks; answer it honestly — promotion discipline is what keeps the capture layer from silting up.
+Harness memory (`~/.claude/projects/<enc>/memory/`) remains the per-project capture layer: fast, automatic, allowed to be at-risk. An agent running without harness memory (a non-Claude runtime) captures into its session log instead and promotes from there — the promotion discipline is the same. At wrap-up (or a task audit, or an explicit prompt), promote facts that meet the inclusion bar into the KB as nodes. The wrap-up workflow asks; answer it honestly — promotion discipline is what keeps the capture layer from silting up.
## Inventory
diff --git a/claude-rules/locating-craig.md b/claude-rules/locating-craig.md
new file mode 100644
index 0000000..c327e3f
--- /dev/null
+++ b/claude-rules/locating-craig.md
@@ -0,0 +1,48 @@
+# Locating Craig
+
+Applies to: `**/*`
+
+When a task needs to know where Craig physically is — a local guide, "what's
+near me", travel logistics, the timezone or weather for his actual spot,
+distance to an appointment — don't ask him. Run the `whereami` command and read
+the result.
+
+## The Rule
+
+`whereami` scans nearby WiFi access points and geolocates the machine it runs
+on. That only tracks *Craig* on the machine that travels with him: **velox**,
+his laptop. On any other machine (ratio, the desktop) it reports that machine's
+fixed location, not Craig's, so the result is only meaningful on velox.
+
+The gate is the machine, stated positively:
+
+1. Check the host with `uname -n`.
+2. **On velox** — run `whereami` whenever location matters, instead of asking.
+ Treat its reverse-geocoded address as Craig's current location.
+3. **Any other host** — don't trust `whereami` as Craig's location. Fall back to
+ asking, or to known context (trip notes, calendar, reminders).
+
+Prefer running it over asking — Craig confirmed he'd rather the agent just
+check. The output gives coordinates, a reverse-geocoded address, and an
+OpenStreetMap link; WiFi-centroid drift of a block or two is normal.
+
+## When whereami can't answer
+
+If `whereami` fails — no WiFi to scan, no network, the geolocation API or its
+BeaconDB fallback unreachable — it can't locate Craig. Fall back to asking or to
+known context, exactly as on a non-velox host. Never fabricate or guess a
+location from a stale reading.
+
+## Keep the location out of shared artifacts
+
+A reverse-geocoded address is personal data. It can drive a task, but it must
+not leak into anything team-visible — commit messages, PR descriptions, tickets,
+or public docs (see the content-scope rule in `commits.md`).
+
+## Why
+
+Craig travels with velox, so on that machine the agent can know his location for
+free rather than interrupting to ask. The command and its design (WiFi BSSID
+scan → Google Geolocation API with a BeaconDB fallback → OSM reverse-geocode)
+were built 2026-06-24 specifically to replace useless cellular-IP lookup that
+reported the carrier gateway instead of the device.
diff --git a/claude-rules/org-tables.md b/claude-rules/org-tables.md
index 1b70085..bc9d27d 100644
--- a/claude-rules/org-tables.md
+++ b/claude-rules/org-tables.md
@@ -1,3 +1,8 @@
+---
+paths:
+ - "**/*.org"
+---
+
# Org Table Standard
Applies to: `**/*.org`
diff --git a/claude-rules/subagents.md b/claude-rules/subagents.md
index 8578dea..e52d906 100644
--- a/claude-rules/subagents.md
+++ b/claude-rules/subagents.md
@@ -34,6 +34,66 @@ This is the same boundary the "Don't Subagent At All" section and the
"Subagenting trivial work" anti-pattern draw; treat it as an explicit gate
at dispatch time.
+Every size-based rule in this file — the cost gate here, "Don't Subagent At
+All", the trivial-work anti-pattern — is subject to the isolation override
+below.
+
+## Isolation Override — When Size Doesn't Gate
+
+Every size heuristic in this file rests on one assumption: that the main
+thread could do the task itself just as well, so the only question is
+whether delegating is worth the overhead. When that assumption fails, the
+heuristics don't apply, and a five-line task can require a subagent that a
+five-hundred-line one wouldn't.
+
+The assumption fails whenever **the main thread is structurally disqualified
+from the task** — not slower at it, disqualified. The test: would the main
+thread's own context make its answer *less* trustworthy? If holding the
+context is what corrupts the judgment, then doing it inline doesn't save the
+overhead, it destroys the result. The isolation *is* the deliverable, and
+"it's only a small diff" is not an argument against it.
+
+**The standing instance is the pre-commit code review** (`publish` skill,
+Step 1). The author cannot review their own change, because a self-review
+checks the diff against the author's own model of it and cannot check the
+model. Errors that survive a self-review are the ones that were never in the
+diff — an inherited scope, an estimated blast radius, a fix correct for the
+case in mind and wrong for the one never considered. So that review is
+dispatched on *every* commit including a one-line one, and the ~10-tool-call
+floor, the single-function rule, and the trivial-work anti-pattern are all
+overridden there by design.
+
+Other cases with the same shape: verifying a claim the main thread already
+committed to in conversation, and any second opinion where the first opinion
+is already in context. If you find yourself reasoning "I already know the
+answer, so a subagent is wasteful," check whether already knowing it is the
+problem.
+
+This override widens *what* gets dispatched. Scope, constraints, and output
+format are still required, and arguably matter more here, since an isolated
+agent can't fall back on shared context to fill a gap.
+
+**Field 2 of the Prompt Contract inverts under this override, and the
+inversion is the whole point.** Normally field 2 says to paste the relevant
+output verbatim and include what you learned in earlier turns. Do that for an
+isolation dispatch and you hand over the very model you spawned the agent to
+escape — a reviewer given your findings reviews your findings. So for an
+isolation dispatch, field 2 is *the artifact under test and the independent
+record of what was asked, and nothing else*: the diff, a one-line claim of
+what it does, and the ticket or plan where one exists. The conversation, the
+rationale, and the dead ends are withheld on purpose.
+
+Keep the requirement source in. A ticket is not your model of the change; it
+was written before the work, usually by someone else, and it is the only
+thing that can contradict your claim about your own diff.
+
+**The output is a judgment, so the review gate resolves differently.** The
+Review-Gate Cadence below says subagent output is a claim to be verified
+before moving on, which is right when the deliverable is *work*. When the
+deliverable is *a judgment about your work*, verifying it against your own
+reading reinstates exactly the bias the dispatch removed. Disagreement goes
+to the user to adjudicate, not back to the author's own judgment.
+
## When to Spawn a Subagent
### Parallel-safe (spawn multiple in parallel)
@@ -63,11 +123,15 @@ at dispatch time.
### Don't Subagent At All
+Unless the Isolation Override applies — these are efficiency rules, and they
+lapse when the main thread's own context is what makes its answer untrustworthy.
+
- **The target is already known** and the work fits in under ~10 tool calls.
- **Single-function logic** — one Read + one Edit is faster than briefing
an agent.
- **You can see the answer from context** — don't spawn a researcher for
- something already on screen.
+ something already on screen. (The inverse of this one is the override's
+ clearest case: when *having* seen it is the disqualification, dispatch.)
## Prompt Contract
@@ -138,7 +202,11 @@ fix), then dispatch the fix with a specific contract.
- **Retrying a failed subagent task in the orchestrator** — pollutes
context. Dispatch a fix agent instead.
- **Subagenting trivial work** — one Read + one Edit doesn't need an
- agent; spawn overhead exceeds benefit.
+ agent; spawn overhead exceeds benefit. Except under the Isolation
+ Override, where a one-line diff still gets its own reviewer.
+- **Reviewing your own change inline** — the mirror-image failure, and the
+ more expensive one. Skipping a dispatch to save overhead on a small diff
+ costs a review that could only have come from outside your context.
- **Skipping review between tasks** — compounding bugs are much harder to
unwind than any single bug.
- **Letting the agent decide scope** — "figure out what needs changing"
diff --git a/claude-rules/testing.md b/claude-rules/testing.md
index b3fa5bf..dd15282 100644
--- a/claude-rules/testing.md
+++ b/claude-rules/testing.md
@@ -16,340 +16,30 @@ TDD is the default workflow for all code, including demos and prototypes. **Writ
Do not skip TDD for demo code. Demos build muscle memory — the habit carries into production.
-### Understand Before You Test
-Before writing tests, invest time in understanding the code:
+## Test Categories — required for all code
-1. **Explore the codebase** — Read the module under test, its callers, and its dependencies. Understand the data flow end to end.
-2. **Identify the root cause** — If fixing a bug, trace the problem to its origin. Don't test (or fix) surface symptoms when the real issue is deeper in the call chain.
-3. **Reason through edge cases** — Consider boundary conditions, error states, concurrent access, and interactions with adjacent modules. Your tests should cover what could actually go wrong, not just the obvious happy path.
+Every unit under test needs all three, not just the happy path:
-### Adding Tests to Existing Untested Code
+1. **Normal** — standard inputs, common workflows, typical volumes.
+2. **Boundary** — zero, one, max, empty vs null, single-element collections, unicode, very long input, timezone and date edges.
+3. **Error** — invalid input, type mismatches, network failure, missing parameters, permission denied, resource exhaustion, malformed data.
-When working in a codebase without tests:
+The negative and boundary cases are the ones that find bugs. A unit with only
+Normal coverage is not tested, it is demonstrated.
-1. Write a **characterization test** that captures current behavior before making changes
-2. Use the characterization test as a safety net while refactoring
-3. Then follow normal TDD for the new change
+## The rest of the standard lives in the `testing-standards` skill
-## Test Categories (Required for All Code)
+Characterization tests for untested code, the per-category detail, combinatorial
+and property-based and mutation testing, organization and the pyramid,
+integration-test rules, naming, the test-quality rules (independence,
+determinism, mocking boundaries, signs of overmocking), the
+refactor-when-tests-are-hard principle, coverage targets, the spike exception,
+and the anti-pattern list are all in the `testing-standards` skill. Load it when
+writing tests.
-Every unit under test requires coverage across three categories:
-
-### 1. Normal Cases (Happy Path)
-- Standard inputs and expected use cases
-- Common workflows and default configurations
-- Typical data volumes
-
-### 2. Boundary Cases
-- Minimum/maximum values (0, 1, -1, MAX_INT)
-- Empty vs null vs undefined (language-appropriate)
-- Single-element collections
-- Unicode and internationalization (emoji, RTL text, combining characters)
-- Very long strings, deeply nested structures
-- Timezone boundaries (midnight, DST transitions)
-- Date edge cases (leap years, month boundaries)
-
-### 3. Error Cases
-- Invalid inputs and type mismatches
-- Network failures and timeouts
-- Missing required parameters
-- Permission denied scenarios
-- Resource exhaustion
-- Malformed data
-
-## Combinatorial Coverage
-
-For functions with 3+ parameters that each take multiple values (feature-flag
-combinations, config matrices, permission/role interactions, multi-field
-form validation, API parameter spaces), the exhaustive test count explodes
-(M^N) while 3-5 ad-hoc cases miss pair interactions. Use **pairwise /
-combinatorial testing** — generate a minimal matrix that hits every 2-way
-combination of parameter values. Empirically catches 60-90% of combinatorial
-bugs with 80-99% fewer tests.
-
-Invoke `/pairwise-tests` on the offending function; continue using `/add-tests`
-and the Normal/Boundary/Error discipline for the rest. The two approaches
-complement: pairwise covers parameter *interactions*; category discipline
-covers each parameter's individual edge space.
-
-Skip pairwise when: the function has 1-2 parameters (just write the cases),
-the context requires *provably* exhaustive coverage (regulated systems — document
-in an ADR), or the testing target is non-parametric (single happy path,
-performance regression, a specific error).
-
-## Escalation Beyond Category and Pairwise
-
-The Normal/Boundary/Error categories and the pairwise matrix are the default
-discipline. Two further techniques escalate beyond them — reach for them when
-the default leaves a gap, not on every unit.
-
-### Property-Based Testing
-
-When an invariant holds across a broad input domain — round-trips
-(`decode(encode(x)) == x`), idempotence (`f(f(x)) == f(x)`), ordering
-invariants (output is always sorted), or any "output always satisfies X" —
-generate inputs and assert the property instead of enumerating cases. The
-generator explores corners you wouldn't think to write by hand, and a
-failing case shrinks to a minimal reproducer. Use the standard tool for the
-language (Hypothesis for Python, fast-check for JS, proptest for Rust).
-State the property as the test name and let the framework supply the inputs.
-
-Reach for this when the behavior is a law over a domain rather than a fixed
-set of examples. Keep category-discipline cases for the specific edges that
-must always hold; the property test covers the space between them.
-
-### Mutation Testing
-
-When line coverage is high but you suspect the assertions are thin — tests
-that execute the code without checking its output, or that pass with a
-function body replaced by a stub — use mutation testing to measure whether
-the suite actually kills injected faults. The tool flips conditionals, swaps
-operators, and deletes statements, then reruns the suite; a surviving mutant
-is a fault the tests didn't catch. Use mutmut or cosmic-ray for Python,
-Stryker for JS. High line coverage with a low mutation score means weak
-assertions, not a tested codebase.
-
-Reach for this on critical logic where coverage looks reassuring but you
-want evidence the tests would fail on a regression. It's a diagnostic, not a
-gate on every change — mutation runs are slow.
-
-## Test Organization
-
-Typical layout:
-
-```
-tests/
- unit/ # One test file per source file
- integration/ # Multi-component workflows
- e2e/ # Full system tests
-```
-
-Per-language files may adjust this (e.g. Elisp collates ERT tests into
-`tests/test-<module>*.el` without subdirectories).
-
-### Testing Pyramid
-
-Rough proportions for most projects:
-- Unit tests: 70-80% (fast, isolated, granular)
-- Integration tests: 15-25% (component interactions, real dependencies)
-- E2E tests: 5-10% (full system, slowest)
-
-Don't duplicate coverage: if unit tests fully exercise a function's logic,
-integration tests should focus on *how* components interact — not repeat the
-function's case coverage.
-
-## Integration Tests
-
-Integration tests exercise multiple components together. Two rules:
-
-**The docstring names every component integrated** and marks which are real vs
-mocked. Integration failures are harder to pinpoint than unit failures;
-enumerating the participants up front tells you where to start looking.
-
-Example:
-
-```
-def test_integration_refund_during_sync_updates_ledger_atomically():
- """Refund processed mid-sync updates order and ledger in one transaction.
-
- Components integrated:
- - OrderService.refund (entry point)
- - PaymentGateway.reverse (MOCKED — returns success)
- - Ledger.credit (real)
- - db.transaction (real)
-
- Validates:
- - Refund rolls back if ledger write fails
- - Both tables updated or neither
- """
-```
-
-**Write an integration test when** multiple components must work together,
-state crosses function boundaries, or edge cases combine. **Don't** when
-single-function behavior suffices, or when mocking would erase the interaction
-you meant to test.
-
-## Naming Convention
-
-- Unit: `test_<module>_<function>_<scenario>_<expected>`
-- Integration: `test_integration_<workflow>_<scenario>_<outcome>`
-
-Examples:
-- `test_cart_apply_discount_expired_coupon_raises_error`
-- `test_integration_order_sync_network_timeout_retries_three_times`
-
-Languages that prefer camelCase, kebab-case, or other conventions keep the
-structure but use their idiom. Consistency within a project matters more than
-the specific case choice.
-
-## Test Quality
-
-### Independence
-- No shared mutable state between tests
-- Each test runs successfully in isolation
-- Explicit setup and teardown
-
-### Determinism
-- Never hardcode dates or times — generate them relative to `now()`
-- No reliance on test execution order
-- No flaky network calls in unit tests
-- Time/clock-mocking helpers must avoid two recurring failure modes:
- - *Infinite recursion.* The helper must not call the primitive it's
- replacing. If the mock for `now()` calls `now()`, the test stack
- overflows. Compute the mock value from a fixed source (a captured
- instant, an injected fake clock).
- - *Scope-shadowing without reach.* A mock that only exists inside
- the test function won't affect production code that reads the
- symbol through its canonical path. Replace the symbol at its
- definition site (monkey-patch the module attribute in Python,
- redefine the global in Lisp, swap the package-level binding in
- Go, replace the named export in JavaScript) — or inject a fake
- via dependency-inversion. Don't lean on scope-shadowing
- primitives (Lisp `let`, Python local rebind, JS shadowed `let`)
- that fence the mock to the test's lexical scope; production code
- won't see them and the test passes against the real clock.
-
-### Performance
-- Unit tests: <100ms each
-- Integration tests: <1s each
-- E2E tests: <10s each
-- Mark slow tests with appropriate decorators/tags
-
-### Mocking Boundaries
-Mock external dependencies at the system boundary:
-- Network calls (HTTP, gRPC, WebSocket)
-- File I/O and cloud storage
-- Time and dates
-- Third-party service clients
-
-Never mock:
-- The code under test
-- Internal domain logic
-- Framework behavior (ORM queries, middleware, hooks, buffer primitives)
-
-### Signs of Overmocking
-
-Ask yourself:
-
-- Would this test still pass if I replaced the function body with `raise NotImplementedError` (or equivalent)? If yes, the mocks are doing the work — you're testing mocks, not code.
-- Is the mock more complex than the function being tested? Smell.
-- Am I mocking internal string / parsing / decoding helpers? Those aren't boundaries — they're the work.
-- Does the test break when I refactor without changing behavior? Good tests survive refactors; overmocked ones couple to implementation.
-
-When tests demand heavy internal mocking, the fix isn't better mocks — it's
-restructuring the code (see *If Tests Are Hard to Write* below).
-
-### Testing Code That Uses Frameworks
-
-When a function mostly delegates to framework or library code, test *your*
-integration logic:
-- ✓ "I call the library with the right arguments in the right context"
-- ✓ "I handle its return value correctly"
-- ✗ "The library works in 50 scenarios" — trust it; it has its own tests
-
-For polyglot behavior (e.g., comment handling across C/Java/Go/JS), test 2-3
-representative modes thoroughly plus a minimal smoke test in the others.
-Exhaustive permutations are diminishing returns.
-
-### Test Real Code, Not Copies
-
-Never inline or copy production code into test files. Always `require`/`import`
-the module under test. Copied code passes even when production breaks — the
-bug hides behind the duplicate.
-
-Mock dependencies at their boundary; exercise the real function body.
-
-### Error Behavior, Not Error Text
-
-Test that errors occur with the right type; don't assert exact wording:
-- ✓ Right exception type (`pytest.raises(ValueError)`, `(should-error ... :type 'user-error)`)
-- ✓ Regex on values the message *must* contain (e.g., the offending filename)
-- ✗ `assert str(e) == "File 'foo' not found"` — breaks when prose changes even though behavior is unchanged
-
-Production code should emit clear, contextual errors. Tests verify the
-behavior (raised, caught, returned nil) and values that must appear — not the
-prose.
-
-## If Tests Are Hard to Write, Refactor the Code
-
-If a test needs extensive mocking of internal helpers, elaborate fixture
-scaffolding, or mocks that recreate the function's own logic, the production
-code needs restructuring — not the test.
-
-Signals:
-- Deep nesting (callbacks inside callbacks)
-- Long functions doing multiple things ("fetch AND parse AND decode AND save")
-- Tests that mock internal string / parsing / I/O helpers
-- Tests that break on refactors with no behavior change
-
-Fix: extract focused helpers (one responsibility each), test each in isolation
-with real inputs, compose them in a thin outer function. Several small unit
-tests plus one composition test beats one monster test behind a wall of mocks.
-
-## Coverage Targets
-
-- Business logic and domain services: **90%+**
-- API endpoints and views: **80%+**
-- UI components: **70%+**
-- Utilities and helpers: **90%+**
-- Overall project minimum: **80%+**
-
-New code must not decrease coverage. PRs that lower coverage require justification.
-
-## TDD Discipline
-
-TDD is non-negotiable. These are the rationalizations agents use to skip it — don't fall for them:
-
-| Excuse | Why It's Wrong |
-|--------|----------------|
-| "This is too simple to need a test" | Simple code breaks too. The test takes 30 seconds. Write it. |
-| "I'll add tests after the implementation" | You won't, and even if you do, they'll test what you wrote rather than what was needed. Test-after validates implementation, not behavior. |
-| "Let me just get it working first" | That's not TDD. If you can't write a failing test, you don't understand the requirement yet. |
-| "This is just a refactor" | Refactors without tests are guesses. Write a characterization test first, then refactor while it stays green. |
-| "I'm only changing one line" | One-line changes cause production outages. Write a test that covers the line you're changing. |
-| "The existing code has no tests" | Start with a characterization test. Don't make the problem worse. |
-| "This is demo/prototype code" | Demos build habits. Untested demo code becomes untested production code. |
-| "I need to spike first" | Spikes are fine — under the protocol below. Throw the spike away, then write the first failing test before productionizing. |
-
-If you catch yourself thinking any of these, stop and write the test.
-
-### The Spike Exception (Disciplined)
-
-TDD stays the default. The one sanctioned way to write code before a test is
-a spike — exploratory code that answers "is this approach even viable?" when
-you can't yet write a meaningful failing test because the shape of the
-solution is unknown. A spike is disciplined only when all three hold:
-
-1. **Timebox it.** Set a limit before starting (an hour, an afternoon) and
- stop when it's up. An open-ended spike is just untested implementation
- wearing a different name.
-2. **Do not commit spike code.** The spike is a learning artifact, not a
- deliverable. It never enters the branch history. Keep it in a scratch
- file or a throwaway worktree.
-3. **Throw the spike away, then start with a failing test.** Once the spike
- has answered the viability question, delete it. Write the first failing
- test against the now-understood behavior, then productionize under normal
- Red/Green/Refactor. The production code is written test-first even though
- the exploration wasn't — you don't promote the spike into production by
- bolting tests on after.
-
-The spike buys understanding, not code. If you find yourself keeping the
-spike because rewriting it feels wasteful, the timebox was too long or the
-problem was tractable enough to TDD from the start.
-
-## Anti-Patterns (Do Not Do)
-
-- Hardcoded dates or timestamps (they rot)
-- Testing implementation details instead of behavior
-- Mocking the thing you're testing
-- Mocking internal helpers (string ops, parsing, decoding) — those are the work
-- Inlining production code into test files — always `require` / `import` the real module
-- Asserting exact error-message text instead of type + key values
-- Shared mutable state between tests
-- Non-deterministic tests (random without seed, network in unit tests)
-- Testing framework behavior instead of your code
-- Ignoring or skipping failing tests without a tracking issue
+What stays here is what has to be true before any code is written, which is when
+no skill has been summoned yet: test first, and cover all three categories.
## Content scope
diff --git a/claude-rules/todo-format.md b/claude-rules/todo-format.md
index b1fb57b..8038b98 100644
--- a/claude-rules/todo-format.md
+++ b/claude-rules/todo-format.md
@@ -1,3 +1,8 @@
+---
+paths:
+ - "**/*.org"
+---
+
# Todo Entry Format
Applies to: `**/*.org` (org-mode todo and inbox files)
@@ -5,6 +10,56 @@ Applies to: `**/*.org` (org-mode todo and inbox files)
How task entries are structured in org-mode todo files (`todo.org`,
`inbox.org`, any GTD-style org file). Same shape across every project.
+## Stamp `:LAST_REVIEWED:` when you create the task, not a cycle later
+
+Every task filed at `**` with a priority cookie carries a `:LAST_REVIEWED:`
+property from the moment it's written:
+
+```
+** TODO [#B] Terse topic phrase :tag:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-23
+:END:
+Body.
+```
+
+Use today's date, from `date +%F`. The org-native `[YYYY-MM-DD Day]` form is
+equally valid; both parse.
+
+**Why.** Writing a task *is* reviewing it. Whoever files it has just written
+the body, chosen the wording, and graded the priority against the scheme — that
+is the same judgment `task-review` applies, made with better context, because
+the reason for the task is still in the room. Leaving the stamp off asserts the
+opposite: `task-review-staleness.sh` sorts a missing property first, as
+never-reviewed, so a task filed today arrives at the top of tomorrow's review
+batch and gets "reviewed" by someone re-deriving what its author knew a day
+earlier. That is ceremony, and ceremony teaches people to click through the
+real thing.
+
+The concrete case: the 2026-07-23 sweep filed eight tasks in one night. Every
+one landed unstamped, and the staleness count went 13 → 22 while the list got
+*more* accurate, not less. The number stopped measuring drift and started
+measuring recent activity.
+
+**This applies to every path that files a task**, not just the inbox: inbox
+filing, triage intake, spec decomposition, task audit, a bug found mid-session,
+a task you write by hand. If you wrote a task body today, stamp it today.
+
+**What it does not do.** The stamp never means "correct forever" — it means
+"a person judged this on that date." A task filed today still enters the
+review rotation on the normal cycle; it just enters it on the *next* cycle
+rather than immediately. And it's a claim about a real event, so don't stamp
+a task you didn't actually consider: a bulk import of someone else's list
+is genuinely unreviewed, and stamping it would convert "nobody has read
+these" into a false "reviewed today."
+
+**Enforcement.** `lint-org.el`'s `task-missing-last-reviewed` checker flags any
+open `**` task with a priority cookie and no stamp, scoped to exactly the
+headings `task-review-staleness.sh` selects, so the checker and the staleness
+count never disagree. It's judgment-only and never auto-fixes: nothing can know
+when an unstamped task was actually last considered, and writing today's date
+onto an old one would destroy the very signal the property carries.
+
## Priority and Tag Scheme Header
Every project's `todo.org` opens with a top-level section named
@@ -24,8 +79,8 @@ guessing:
The section is mandatory. A `todo.org` without it leaves `[#A]` and the tags
undefined, so task-audit can't enforce a vocabulary, task-review can't grade
-against agreed semantics, and process-inbox can't file new tasks correctly
-(its Phase B.1 already checks for this scheme). Each project defines the
+against agreed semantics, and the inbox workflow can't file new tasks correctly
+(its priority-scheme check already gates on this scheme). Each project defines the
scheme its own way; the floor is that priorities and tags are both spelled
out under the header.
@@ -33,6 +88,152 @@ When a project's `todo.org` lacks the section, add it before filing or
grading further tasks — propose the priority semantics and tag set from the
project's existing usage, and confirm with Craig.
+### Hard definitions: `:solo:` and `:quick:` (fixed across projects)
+
+A project's scheme may add or rename its other tags, but these two carry
+fixed definitions everywhere, because autonomous execution
+(work-the-backlog / the no-approvals speedrun) reads `:solo:` as its
+eligibility gate and trusts the author's tag rather than re-deriving
+autonomy at run time.
+
+“Speedrunnable” is shorthand for `:solo:`. The `:quick:` tag plays no part
+in that definition.
+
+- **`:solo:` — autonomy.** The task can be completed *and verified* without
+ Craig's involvement beyond at most one or two quick decisions that can be
+ stated and answered before work starts. No open design question, no
+ "weigh these approaches," no waiting on Craig mid-task. Three gates, all
+ must hold: *buildable* (the agent has the capability and access),
+ *verifiable by the agent* (an objective or local check it can run itself —
+ handing off a residual human-in-the-loop confirmation as a structured
+ manual-testing reminder does not disqualify), and *no deliberation* (a
+ quick, upfront-answerable factual question is allowed — it gets batched
+ into the speedrun's pre-flight Q&A; a genuine design or preference call
+ is not). A wrong `:solo:` is worse than none: it tells Craig he can hand
+ the task off and walk away when he can't.
+- **`:quick:` — effort hint only.** Likely 30 minutes or less from start
+ through verification. Informational, for batching and estimating a run's
+ duration; never an eligibility gate. `:quick:` and `:solo:` are
+ orthogonal — a bounded refactor can be `:solo:` but slow; a five-minute
+ change hinging on a preference call is `:quick:` but not `:solo:`.
+
+Both tags are applied at task creation and **re-checked as a mandatory
+step** in the task-review and task-audit workflows, so the run-time gate
+can trust the tag. A review or audit that skips the `:solo:`/`:quick:`
+assessment is incomplete.
+
+### Making an open-ended task measurable (so it can be `:solo:`)
+
+A task phrased as the *absence* of something — "find bugs until none are
+visible," "refactor until no worthwhile opportunities remain," "clean this
+up until it's good" — cannot be `:solo:`, because it fails the
+*verifiable-by-the-agent* gate. Absence isn't falsifiable: an agent can
+always look once more, so "done" is a judgment call, which is exactly what
+`:solo:` forbids. The fix is not to drop the task but to give it an
+objective completion criterion. Four moves convert a fuzzy goal into a
+measurable one:
+
+1. **Bound the surface.** Enumerate the concrete units the task covers (the
+ N functions, the M files, the named code paths). The done-set is that
+ list, not the platonic set of all possible defects. Every claim is made
+ against the list, so "covered" is checkable where "found everything"
+ isn't.
+2. **Net the behavior.** Bring the surface under characterization tests
+ (Normal/Boundary/Error per unit — see `testing.md`, and the
+ `testing-standards` skill for the characterization recipe) before changing
+ anything. This is the objective floor: writing a characterization test is
+ mechanical (record what the code does, not what it should), so it scales
+ across the surface, and it doubles as the safety net that makes any
+ later refactor falsifiable.
+3. **Disposition every finding.** Run the relevant audits (a fixed
+ footgun/OWASP checklist, `/refactor`, `/review-code`) and give **every**
+ finding a verdict: fixed, filed as its own task, or declined with a
+ one-line reason. "Looked and it's fine" is not a disposition. The
+ measurement is zero undispositioned findings, not zero findings.
+4. **Gate on an objective floor.** Static analysis clean (linter,
+ type-checker, `shellcheck`), the test suite green before and after, and
+ coverage of the enumerated surface (a per-unit test checklist, or a real
+ coverage number where the tooling exists).
+
+The **qualifying answer** is then a dispositioned report — surface split
+covered/uncovered, tests before → after, static-analysis result, the audit
+matrix fully dispositioned, all green — not a claim of perfection. The
+honest limit stays honest ("no visible bugs" means "every enumerated path
+passes its characterization set and clears the audit," never "zero bugs
+exist"), but the criterion is now falsifiable, which is what lets the task
+carry `:solo:`. Write these criteria into the task body at creation or
+review time; a task that can't be given them stays non-`:solo:` until it
+can.
+
+### Bug priority from severity × frequency (mandatory where a codebase exists)
+
+Some projects carry a codebase — source the project maintains under version
+control — even when the project isn't primarily a "code project." home and work
+both have one. Wherever a project has a codebase, a task representing a bug
+*against that codebase* does not get its priority argued: it is **dictated** by
+the severity × frequency matrix below. This is not opt-in. Features keep their
+per-project `[#A]`–`[#D]` roadmap judgment; codebase bugs do not.
+
+Two facts set a bug's priority:
+
+- **Severity** — how bad it is when it occurs (service down / data loss /
+ security or privacy leak at one end; a cosmetic nit at the other).
+- **Frequency** — how often a user hits it (every user every time → rare edge
+ case).
+
+```
+| Frequency / Severity | Critical | Major | Minor | Cosmetic |
+|------------------------+----------+-------+-------+----------|
+| Every user, every time | P1 | P1 | P2 | P3 |
+| Most users, frequently | P1 | P2 | P3 | P4 |
+| Some users, sometimes | P2 | P3 | P3 | P4 |
+| Rare edge case | P2 | P3 | P4 | P4 |
+```
+
+P-level → priority letter (fixed — not a per-project knob):
+
+- P1 → `[#A]`
+- P2 → `[#B]`
+- P3 → `[#C]`
+- P4 → `[#D]`
+
+Release vehicle is illustrative and project-dependent: P1 = current
+release/patch, P2 = next patch, P3 = next major, P4 = backlog. A project with no
+release train maps the letters and skips the vehicle. A "no open `[#A]` bugs"
+release gate therefore means "no open P1."
+
+Each project still defines, in its scheme header, what
+Critical/Major/Minor/Cosmetic and the frequency rows mean *for its own
+codebase* — the bands are concrete per codebase, but the matrix structure and
+the letter mapping are fixed.
+
+**Severity-alone carve-out:** privacy or security leaks, compliance violations,
+and safety issues are graded on severity alone — one occurrence with the right
+consequences is a showstopper no matter how rarely it would be hit.
+
+**Don't double-count rarity.** Grade severity by the rate of harm once the
+failure state is entered, not by how rare it is to enter. Frequency already
+carries the rarity; letting it discount severity too grades the same fact twice,
+and that buries exactly the bugs that compound — the ones where a rare trigger
+produces unbounded harm. A leak that repeats every timeout period until the
+process restarts is Major even when reaching that state is a rare edge case
+("accumulates slowly" describes a bounded trickle, not a fixed-rate leak with no
+workaround). Grade the *being-in-it*, and let the frequency row carry the
+*getting-into-it*.
+
+**Record the grading in the task body.** State the severity band, the frequency
+row, and the arithmetic (e.g. "Major severity × rare edge case = P3 = [#C]"). A
+bare priority cookie can't be argued with; a stated read can be re-checked
+against the source and corrected. This is what lets a misgrade move — a chime
+watchdog bug went [#D] → [#C] an hour after grading precisely because the read
+was written down and re-checked.
+
+**Disagreeing with a grade means fixing an input.** If a letter looks wrong,
+don't override the letter — re-read the severity band and the frequency row
+against the source and correct whichever input is wrong. Overriding the cookie
+directly turns the matrix into a formality and puts you back to grading by
+instinct, which is the thing it exists to replace.
+
## The Rule
A todo entry has two parts:
@@ -117,6 +318,7 @@ A completed sub-task disappears as a task and becomes an in-place event-log entr
2. Generate the timestamp with `date "+%Y-%m-%d %a @ %H:%M:%S %z"`.
3. Reword the original imperative title into the past-tense action that landed. Trim or restate if the original wording doesn't fit the action.
4. Drop the `TODO`/`DOING` keyword, the priority cookie, and the tags. The body stays as the record of what was done (if useful).
+5. Remove any `SCHEDULED:`/`DEADLINE:` planning line. The completion time lives in the heading now, so `CLOSED:` is redundant and an active planning date on a historical log entry is always wrong. Org renders any headline carrying an active `<...>` `SCHEDULED`/`DEADLINE` on the agenda, keyword or not, so a stale one pins the finished entry there as weeks-overdue forever. An interactive close (`org-log-done`) stamps `CLOSED:` but never strips a pre-existing planning line, which is exactly how the stale dates survive.
**Example:**
@@ -126,11 +328,13 @@ becomes
*** 2026-05-15 Fri @ 12:58:08 -0500 Wired yasnippet for universal availability
+**Enforcement.** This is applied at close time by whoever closes the task, but an interactive org close (`org-log-done` flips the keyword to `DONE` and stamps `CLOSED:`) never applies the dated rewrite, so level-3+ closes accumulate as `DONE` keywords. `todo-cleanup.el --convert-subtasks` (run in the `clean-todo` and wrap-up cleanup passes) normalizes them mechanically: it rewrites any level-3+ `DONE`/`CANCELLED`/`FAILED` heading into the dated form above, pulling the timestamp from the `CLOSED` cookie, dropping the whole planning line (`CLOSED`, `SCHEDULED`, and `DEADLINE` together — step 5), and keeping the heading text verbatim (a batch tool can't reliably past-tense a title — polish wording by hand where it matters). `lint-org.el` flags any that slip through: checker `subtask-done-not-dated` for a still-keyworded sub-task, and `dated-log-heading-active-timestamp` for a dated entry that kept an active `SCHEDULED`/`DEADLINE`. So the depth rule holds even when tasks are closed interactively rather than by an agent applying this section.
+
### Why depth-based
The agenda view (`org-agenda`) shows entries at the section + top-task level. Letting `**` tasks stay task-shaped preserves their visibility as "things that recently shipped." Letting `***+` sub-tasks flip to dated entries keeps the agenda from being clogged with a long list of completed sub-tasks at every depth — those become history within their parent instead.
-`VERIFY` is the documented exception: it follows the dated-rewrite rule at **all** depths (including `**`), because a resolved VERIFY is an answered question rather than a finished task. See the VERIFY section below.
+`VERIFY` follows the dated-rewrite rule at `***` and deeper, the same as any sub-task. At `**` it does *not*: a top-level VERIFY completes task-shaped — a `DONE`/`CANCELLED` keyword plus a `CLOSED:` line, exactly like a top-level `TODO`. Dated headers never appear at `**`. Level 2 always carries a terminal keyword; dated headers are a `***`-and-deeper shape only. See the VERIFY section below.
## VERIFY tasks
@@ -191,19 +395,35 @@ The sibling rule is the active force that keeps `todo.org` flat. Without
it, VERIFYs accumulate one level deeper than their trigger every time —
turning a clean parent tree into a long pole of nested sub-headings.
-### Completion — dated rewrite + content replacement
+### Completion — depth decides the heading shape
+
+When a VERIFY resolves, **rewrite the heading and body together**. The body
+replacement is the same at every depth (step 2 below); the heading shape
+depends on the VERIFY's level, mirroring the depth-based rule for ordinary
+tasks — dated entries at `***` and deeper, terminal keyword at `**`.
+
+1. **Replace the heading — by depth.**
+
+ - **At `***` and deeper — dated event-log entry.** Drop the `VERIFY`
+ keyword (and any priority cookie / tags) and replace with a timestamp +
+ short description:
-When a VERIFY resolves, **rewrite the heading and body together** at the
-same depth — regardless of whether the VERIFY is at `**` or `***`:
+ *** 2026-05-15 Fri @ 14:00:00 -0500 <what was answered or done>
-1. **Replace the heading.** Drop the `VERIFY` keyword (and any priority
- cookie / tags) and replace with a timestamp + short description:
+ Generate the timestamp with `date "+%Y-%m-%d %a @ %H:%M:%S %z"`.
+ Remove any `SCHEDULED:`/`DEADLINE:` planning line too, same as the
+ sub-task rule above — a dated event-log entry carries its date in the
+ heading, and an active planning date left on it pins the finished entry
+ to the agenda forever.
- *** 2026-05-15 Fri @ 14:00:00 -0500 <what was answered or done>
+ - **At `**` — terminal keyword, like any top-level task.** Change
+ `VERIFY` to `DONE` (answered / check passed) or `CANCELLED` (abandoned),
+ keep the heading text, priority cookie, and tags, and add a
+ `CLOSED: [YYYY-MM-DD Day]` line. Never a dated heading — a `**` dated
+ header is a defect; repair it to `DONE`/`CANCELLED` + `CLOSED:`.
- Generate the timestamp with `date "+%Y-%m-%d %a @ %H:%M:%S %z"`.
- Match the original depth (a `**` VERIFY becomes `** YYYY-MM-DD ...`;
- a `***` VERIFY becomes `*** YYYY-MM-DD ...`).
+ ** DONE [#B] <original VERIFY topic> :tags:
+ CLOSED: [2026-05-15 Fri]
2. **Replace the body.** Drop the original question/instruction prose and
replace with either:
@@ -213,16 +433,18 @@ same depth — regardless of whether the VERIFY is at `**` or `***`:
instruction or pending-decision marker — what was done, when, where
the artifact lives).
-The completed VERIFY becomes an in-place event log entry. The original
-question is preserved by the dated heading + body shape; anyone scanning
-the agenda or `git log` can see what was asked and what landed.
+Either way the completed VERIFY records what was asked and what landed: at
+`***` and deeper as a dated event-log entry, at `**` as a `DONE`/`CANCELLED`
+task whose body holds the answer. Anyone scanning the agenda or `git log`
+can see both.
-**Note on the top-level case.** Regular `**` DONE tasks stay task-shaped
-with a `DONE` keyword + `CLOSED:` line per *Completion — depth-based*
-above. VERIFYs at `**` are the exception — they convert to dated log
-entries on completion because a resolved VERIFY isn't a "done task," it's
-an answered question. The dated-rewrite rule wins for VERIFYs at all
-depths.
+**Note on the top-level case.** A `**` VERIFY completes exactly like a `**`
+`TODO`: a `DONE`/`CANCELLED` keyword + `CLOSED:` line, with the answer or
+action in the body. The earlier habit of dating a resolved top-level VERIFY
+— treating "answered question, not a finished task" as license for a `**`
+dated header — is retired. It put dated headers at level 2, where the agenda
+truncates them out of a clean keyword scan. Dated rewrite is for `***` and
+deeper only; `**` always carries a terminal keyword.
### Don't leave stale placeholders
@@ -249,3 +471,54 @@ are noise that pollute his `cj:` greps.
** DOING [#A] Kostya's contract :admin:kostya:
*** 2026-05-15 Fri @ 14:00:00 -0500 Kostya basis — part-time, 20 hr/week
Nerses confirmed 5/15 13:30 CDT: Kostya runs at 20 hr/week part-time, mirroring Vrezh's structure. Plugged into Exhibit A § 2 of the contract draft.
+
+## Cross-Project Dependency Tags
+
+A task can be blocked by work that has to happen in a *different project* — a rulesets task that can't finish until `.emacs.d` ships a companion function, say. Left unmarked, two things go wrong: the what's-next workflow keeps recommending the blocked task even though it can't move, and the blocker sits at low priority in the other project, so the dependency stalls silently.
+
+Two plain org tags track it, one on each side, so neither the waiter nor the blocker loses sight of the dependency: `:blocked:` on the task that's waiting, `:blocker:` on the task that owes the work. The cross-project detail — which project, what work — goes in the task *body*, not a property. This applies to *any* project pair; the convention here and the surfacing in `open-tasks.org` live in the shared rule + workflow layer, not in one project.
+
+### `:blocked:` — the waiting side
+
+The task that can't proceed carries `:blocked:`. Its body names the project it's waiting on and what that project owes:
+
+```
+** DOING [#B] Wrap-teardown feature :feature:blocked:
+Blocked on emacsd: needs the ai-term companion functions
+(cj/ai-term-quit, -live-count) before the manual validation can run.
+```
+
+`open-tasks.org` reads the `:blocked:` tag to pull the task out of the "do this next" cascade (it can't be worked) and surface it in a dedicated "Blocked on other projects" section, reading the body for which project to name and nudge.
+
+### Registering with the blocker — the reciprocal handoff (required)
+
+Setting `:blocked:` is not complete until the blocking project knows it's blocking. The moment you mark a task `:blocked:` on another project's work, send that project a dependency handoff:
+
+```
+inbox-send <project> --text "Blocking dependency: <this-project>'s task \"<task>\" is blocked on you — it needs <what>. It stays blocked until this lands. Tag the owning task :blocker: on your side so it surfaces as priority work."
+```
+
+This is what closes the gap: without it, the blocker only learns it's blocking by accident. The handoff lands in `<project>`'s `inbox/` and its normal inbox processing tags the work (below). A `:blocked:` task with no matching reciprocal handoff is half-done — the dependency is invisible to the one project that can clear it. Skip the send only when the blocker demonstrably already tracks the work (e.g. it's the same handoff that spawned the dependency); it dedups against an existing task either way.
+
+### `:blocker:` — the blocking side
+
+When a project processes a blocking-dependency handoff (inbox process mode), it tags the owning task `:blocker:` and names the requesting project in the body:
+
+```
+** TODO [#B] ai-term wrap-teardown companion :feature:blocker:
+Rulesets' wrap-teardown feature is blocked on this — it needs the three
+ai-term functions. Surface first so rulesets unblocks.
+```
+
+The blocking task does *not* carry `:blocked:` — it isn't blocked, it's the blocker. `:blocker:` is a priority signal: `open-tasks.org` surfaces a `:blocker:` task *first*, since clearing it unblocks work in another project, so a dependency that would otherwise stall at low priority gets pulled forward. This is the "surface dependencies first" half of the design.
+
+### Resolving the dependency
+
+When the blocker delivers:
+
+1. The blocking project completes its `:blocker:` task, drops the `:blocker:` tag, and notifies the waiter (`inbox-send <waiter> --text "Delivered: <what> — you're unblocked."`).
+2. The waiting project drops the `:blocked:` tag; the task is workable again. Either side noticing the delivery can lift its own tag — the notification just makes it prompt.
+
+### Not the same as VERIFY
+
+`:blocked:` marks "waiting on another *project's* work"; `VERIFY` marks "waiting on Craig's input." If Craig's input is what's needed, it's a VERIFY, not `:blocked:`. And `:blocker:` only ever sits on the project that *owes* the work, never the one waiting.
diff --git a/claude-rules/triggers.md b/claude-rules/triggers.md
index e45e660..3c4ea6d 100644
--- a/claude-rules/triggers.md
+++ b/claude-rules/triggers.md
@@ -8,21 +8,22 @@ Trigger phrases the user can say from any session to invoke a cross-project acti
Synonyms: "Launch X", "Open project X", "Switch to project X".
-**Action:** run the `ai` script (the Claude Code session launcher, installed at `~/.local/bin/ai`) in single-project mode targeting the named project.
+**Action:** run the `ai` script (the agent session launcher, installed at `~/.local/bin/ai`) in single-project mode targeting the named project.
```
ai <project-path>
```
-The `ai` script handles tmux session creation, window placement, and the per-project Claude opening line — see `~/code/rulesets/claude-templates/bin/ai` for the canonical source.
+The `ai` script handles tmux session creation, window placement, and the per-project agent opening line — see `~/code/rulesets/claude-templates/bin/ai` for the canonical source.
**Resolving X.** Match against project basenames discoverable by `ai` — directories under `~/code/`, `~/projects/`, and `~/.emacs.d` that contain `.ai/protocols.org`.
- Exact basename match (case-insensitive) → invoke `ai <path>` directly.
+- Dot-stripped match → a dotted basename is addressed with its dots removed, so `emacsd` matches `.emacs.d` and `dotfiles` matches `.dotfiles`. Strip dots from both the spoken name and each candidate basename when comparing; an exact match still wins over a dot-stripped one. (`inbox-send` resolves the same way, so the spoken name is consistent across both.)
- No match → list all available basenames, ask which to launch.
- Multiple partial matches (X is a substring of two or more candidates) → list the matching basenames, ask which.
-Do not guess. The cost of asking once is one short turn; launching the wrong project is a wrong-context Claude session that has to be killed and restarted.
+Do not guess. The cost of asking once is one short turn; launching the wrong project is a wrong-context agent session that has to be killed and restarted.
## Why a separate file
diff --git a/claude-rules/ui-prototyping.md b/claude-rules/ui-prototyping.md
new file mode 100644
index 0000000..c453258
--- /dev/null
+++ b/claude-rules/ui-prototyping.md
@@ -0,0 +1,93 @@
+# UI Prototyping for Specs
+
+Applies to: `**/*` (any spec whose deliverable has a non-trivial UI)
+
+How to settle a UI design before committing implementation code: research the
+category, brainstorm the UX in the spec, build a handful of full working
+prototypes, then iterate one to a final. The prototype is the evidence a design
+decision is recorded against. Discovered building archsetup's timer-panel spec,
+promoted here so any UI-bearing spec follows the same shape.
+
+## When this applies — non-trivial UI only
+
+The process fires when a spec's deliverable is a real UI: a panel, a
+multi-control surface, a visual layout with interacting parts. Not a single
+dialog, a CLI flag, or a one-off prompt. The test: if "which of these layouts
+is right?" can't be answered from a sentence, it qualifies.
+
+A spec with no UI, or a trivial one, skips this rule entirely.
+
+## The process
+
+### 1. Research first — during brainstorming, before any prototype
+
+Before building anything, survey how existing and best-in-class tools solve the
+same UX: the category's well-regarded apps, prior art, the conventions users
+already expect. Feed the findings into the spec's Goals and Design so the UX is
+understood before a single prototype exists. Prototyping blind wastes iterations
+re-deriving what a 20-minute survey would have told you. Cite the sources in the
+spec.
+
+### 2. Brainstorm the UX in the spec
+
+Informed by the research, write the goals, the interactions, and the functional
+surface into the spec. This is the "what and why" the prototypes make real.
+
+### 3. Prototype — ~5 distinct directions, then iterate one to final
+
+Build about five genuinely different directions — distinct layouts and
+interaction models, not variations of one — as full working prototypes over one
+shared engine, in the project's design language. Pick a direction, then iterate
+that one across numbered passes to the final. Save each meaningful pass as its
+own numbered prototype so the design history is walkable.
+
+For an Emacs app UI, the production port target is `svg.el` — browser SVG and
+librsvg share primitives, so the winning prototype ports near-1:1. Keep every
+prototype within the librsvg-safe constraint sheet in `emacs.md` (SVG Rendering
+section) so the port stays mechanical.
+
+### 4. Full working prototypes, not mockups
+
+The prototypes must be functional: real state, real controls, real behavior, so
+decisions are made against how it feels to use rather than against a picture. A
+static mockup hides the interaction problems that only surface when you drive it.
+
+### 5. Naming and location
+
+`docs/prototypes/<spec-name>-prototype-<N>.html`, where `<spec-name>` is the
+spec's dated slug with the `-spec` suffix dropped, and `N` is the iteration
+number. For `docs/specs/2026-07-02-timer-panel-spec.org`, the prototypes are
+`docs/prototypes/2026-07-02-timer-panel-prototype-1.html`, `-2.html`, `-3.html`.
+
+### 6. Link from the spec; keep every iteration in history
+
+The spec links the final prototype in its design section, and keeps links to
+every prior iteration in a "Prototype iterations" subsection under the status
+heading — newest last — so the design's evolution is walkable from the spec.
+
+### 7. Decisions get recorded once seen working
+
+A design decision moves into the spec's Decisions only after it has been seen
+working in a prototype. "Resolved live through the prototype iteration" — the
+prototype is the evidence, not an argument on the page.
+
+## How the spec workflows hook in
+
+- **spec-create**: for a non-trivial-UI spec, add the "research → brainstorm →
+ prototype (5 directions → iterate)" step before the design is treated as
+ settled, and require the "Prototype iterations" subsection under the status
+ heading.
+- **spec-review**: for a non-trivial-UI spec, verify the process ran — research
+ cited, final prototype linked, iterations present in history, and each UI
+ design decision backed by a prototype rather than asserted.
+
+Both workflows point here rather than restating the process, so the rule stays
+the single source of truth.
+
+## Why
+
+A UI design argued on paper is a guess. Five working directions surface the
+interaction problems a mockup hides, and iterating one of them to a final makes
+the tradeoffs concrete before any production code is written. Recording each
+decision only once it's been seen working keeps the spec honest: the Decisions
+section documents what the prototype proved, not what the author hoped.
diff --git a/claude-rules/verification.md b/claude-rules/verification.md
index 1bbd8dd..2c2088e 100644
--- a/claude-rules/verification.md
+++ b/claude-rules/verification.md
@@ -13,6 +13,18 @@ This applies to every completion claim:
- "Bug is fixed" → Run the reproduction steps. Confirm the bug is gone.
- "No regressions" → Run the full test suite, not just the tests you added.
+## Green Baseline Before Starting Work
+
+Run the test suite before you start work, not only before you finish. A clean run at the start confirms the tree is in the known-good state you assume it is, so the baseline you build on and measure your changes against is actually green.
+
+If the suite is red before you touch anything, fix or explicitly triage the failure first. A pre-existing failure left in place poisons every later "did I break this?" check: you can't separate your own regressions from the noise, and the end-of-work run stops being readable as pass/fail at a glance. Work that assumes a known-good base may also be built on a broken assumption you never saw.
+
+When a pre-existing failure genuinely can't be fixed before the work begins (out of scope, or it needs a decision), record it as a tracked task with the diagnosis and carry its name forward. The green bar for the rest of the work is then explicitly "only this known failure remains," not a silent tolerance for red.
+
+Two cases are not failures of this rule. A project with no suite has nothing to baseline — note that and proceed. A suite that can't run (no network, a missing dependency, a sandbox limit) is the "When You Cannot Verify" case below, not a work-blocker — record what you couldn't run and proceed with the risk named.
+
+This is the start-of-work counterpart to the Before Committing gate below: one confirms the ground is solid before you build, the other confirms you didn't crack it.
+
## What Fresh Means
- Run the verification command **now**, in the current session
@@ -56,6 +68,8 @@ Do not let an unverifiable check vanish into a confident summary. State it plain
Some checks can only be run by the user: interactive UI a script can't drive, a live external service, visual rendering (colors, layout, faces), a real device, or anything where the verification *is* a human looking at the result. When the gap needs the user's hands or eyes — not just a command they could paste — don't bury the steps in prose. Write them as a structured, runnable checklist in the project's task file.
+When *you* need a window on the user's live desktop to verify something visually — render a panel, grab a screenshot — keep it off the workspace they're actively using. Capture on an off-screen output and tear it down; when showing the user something, open it on a separate workspace and name it. See `desktop-capture.md` for the rule and the Hyprland recipe.
+
Create (or append to) a single parent task named **"Manual testing and validation"** in the project's todo file (`todo.org`, or the project's equivalent). Under it, write **one org sub-header per test**:
- **Title** — descriptive, naming the behavior under test (not "test 1").
@@ -92,7 +106,7 @@ Use this whenever the verification gap from "When You Cannot Verify" above is a
## Before Committing
Before any commit:
-1. Run the test suite — confirm all tests pass
+1. Run the full test suite as its own command, read the result, and commit only when failures are zero — never bundle the run with the commit (e.g. `make test; git commit`), where a red suite can't stop the commit. Run the whole suite, not just the touched file: a change can break a test elsewhere. If the suite can't run, that's "unable to verify" (see When You Cannot Verify above) — surface it, don't commit silently.
2. Run the linter — confirm no new warnings
3. Run the type checker — confirm no new errors
4. Review the diff — confirm only intended changes are staged
diff --git a/claude-rules/working-files.md b/claude-rules/working-files.md
index 9a72702..b915579 100644
--- a/claude-rules/working-files.md
+++ b/claude-rules/working-files.md
@@ -26,6 +26,28 @@ hits a nested path instead of a single canonical name. Always rename the
files individually with a shared prefix so they sort together but live as
flat siblings in `assets/`.
+## `working/` Is Version-Controlled From Creation
+
+`working/` holds the project work currently being developed, so it is
+tracked in git the moment a task subdirectory and its artifacts are created —
+not staged locally and excluded until it graduates. `working/` is the tracked
+home of in-progress work, and it is never added to `.gitignore` (the
+install/sweep tooling deliberately leaves it out).
+
+Filing on completion **reorganizes** durable artifacts into their permanent
+homes; it does not mark the point at which they first become durable. The
+artifacts were durable — and tracked — from creation. Graduation is a move,
+not a promotion from throwaway to keep.
+
+The corollary: genuinely disposable work does not belong in `working/`.
+Ephemeral, single-use, or regenerable artifacts — scratch output, a
+throwaway conversion, intermediate data you will delete — go in a project-root
+`temp/` directory (gitignored) or system `/tmp`, never in `working/`.
+`working/` is for work that will graduate; `temp/` is for work that will be
+thrown away. The install tooling ignores `temp/` in both track and
+gitignore-mode projects, since ephemerality is independent of whether a
+project tracks its `.ai/` tooling.
+
## Directory Layout
<project-root>/
@@ -120,7 +142,7 @@ When the task is marked done:
- *Inbox content* — `inbox/` and `daily-prep/` follow their own
conventions (dated filenames, processed and moved on cadence).
-## Implementation Note for Claude Sessions
+## Implementation Note for Agent Sessions
When the user starts a new task that's going to produce file artifacts:
diff --git a/claude-templates/.ai/notes.org b/claude-templates/.ai/notes.org
index 42ea8df..311a86f 100644
--- a/claude-templates/.ai/notes.org
+++ b/claude-templates/.ai/notes.org
@@ -1,24 +1,24 @@
#+TITLE: Claude Code Notes - [Project Name]
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: [Date]
* About This File
This file contains project-specific information for this project.
-**When to read this:**
+- When to read this:
- At the start of EVERY session (after reading protocols.org)
- When needing project context or history
- When checking reminders or pending decisions
-**What's in this file:**
+- What's in this file:
- Project-specific context and goals
- Pending decisions
- Active reminders
-**Session history is NOT in this file.** Each session's record lives in =.ai/sessions/YYYY-MM-DD-HH-MM-description.org= — one file per session. Catch-up reads the Summary sections of the most recent 5.
+- Session history is NOT in this file. Each session's record lives in =.ai/sessions/YYYY-MM-DD-HH-MM-description.org= — one file per session. Catch-up reads the Summary sections of the most recent 5.
-**For protocols and conventions, see:** [[file:protocols.org][protocols.org]]
+- For protocols and conventions, see [[file:protocols.org][protocols.org]].
* Project-Specific Context
@@ -48,9 +48,9 @@ This section tracks decisions that need Craig's input before work can proceed.
- Include: What needs to be decided, options available, why it matters
- Remove decisions once resolved (the resolution is captured in the Session Log of the session where it was resolved)
-**Example format:**
+- Example format:
#+begin_example
-** Feature Name or Topic
+,** Feature Name or Topic
Craig needs to decide on [specific question].
diff --git a/claude-templates/.ai/protocols.org b/claude-templates/.ai/protocols.org
index 6b1d873..bf9f420 100644
--- a/claude-templates/.ai/protocols.org
+++ b/claude-templates/.ai/protocols.org
@@ -1,5 +1,5 @@
#+TITLE: Claude Code Protocols
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2025-11-05
* About This File
@@ -84,7 +84,7 @@ Do NOT estimate, guess, or rely on memory. Just run the command. It takes one se
Every session pulls rulesets first, then the local project repo. Rulesets carries the canonical behavioral rules and =.ai/= templates (the old =claude-templates= repo is folded in as a subtree at =rulesets/claude-templates/=); the project pull lands commits pushed from other machines or teammates since the last session.
-Resolve any dirty-tree or merge issue at each step before moving on. Both pulls run as =git pull --ff-only= (or =git merge --ff-only= against the fetched upstream), so anything non-trivial — non-fast-forward history, dirty working tree, diverged branches — aborts. Surface the state and stop. Never auto-stash, auto-merge, or auto-rebase; the user resolves the conflict before further work.
+Resolve any sync-blocking tree or merge issue at each step before moving on. The shared =git-worktree-gate sync-safe= policy permits untracked deliveries beneath =inbox/= so receiving a handoff never prevents another project from refreshing rulesets; every staged or tracked change, dirty submodule, Git operation in progress, or untracked path outside =inbox/= blocks. Both pulls run as =git pull --ff-only= (or =git merge --ff-only= against the fetched upstream), so non-fast-forward history and diverged branches also abort. Surface the state and stop. Never auto-stash, auto-merge, or auto-rebase; the user resolves the conflict before further work.
Mechanics live in =startup.org= Phase A.0. The rule lives here because it governs the very first action of every session: load the freshest behavioral rules and templates before anything else runs.
@@ -100,8 +100,14 @@ When two agents share one project at the same time, a single =session-context.or
- =AI_AGENT_ID= unset or empty (the normal one-agent-per-project case): =.ai/session-context.org=, exactly as before.
- =AI_AGENT_ID= set: =.ai/session-context.d/<id>.org= (id sanitized to filename-safe chars). Archived at wrap-up to =.ai/sessions/YYYY-MM-DD-HH-MM-<id>-<description>.org= so concurrent agents don't collide on the archive name either.
+The id must be unique per run, and the spawner makes it so by appending an epoch on the tail. The recommended shape is =host.project.runtime.<epoch>= (e.g. =velox.rulesets.claude.1718400000=); a fresh run of the same logical agent then resolves to a fresh anchor. A bare, reused id (just =codex=) makes the next run resolve to the previous run's leftover =codex.org= and either mistake it for a crashed session to recover or clobber it if both run at once. This happened on 2026-06-13: a =codex= run left =session-context.d/codex.org= behind, and the next session had to clean it up by hand.
+
+The epoch is baked into the id by the spawner, never minted inside =session-context-path=. That resolver is called many times per session (startup existence check, every log write, the wrap-up rename) and must return the same path on every call, so a self-generated =date +%s= would fragment the anchor across calls. The shell that runs each call doesn't carry env between calls either, so the agent can't export the epoch once at startup. The only stable source is the =AI_AGENT_ID= the spawner injects on every call.
+
Resolve the path with =.ai/scripts/session-context-path= rather than hardcoding =.ai/session-context.org=; it prints the right path for the current =AI_AGENT_ID=. Fall back to =.ai/session-context.org= if the script isn't present (older checkouts mid-sync). Everything below — the record/recovery purpose, the update triggers, the startup existence check, the wrap-up rename — operates on that resolved path. The prose says "session-context.org" as the default name; read it as "the resolved active path" when =AI_AGENT_ID= is set.
+A helper instance (a second agent running in this project while a primary session is live) follows a different contract: it skips the pulls and rsync, makes only scoped single-heading edits to shared files, leaves all git mutation to the primary, and wraps up by archiving its own context file without committing. The full rules — read/write tiers, data-integrity, light startup, helper wrap-up — live in [[file:workflows/helper-mode.org][workflows/helper-mode.org]]. A session is a helper only when something routes it there (the =ai --helper= launcher, startup's roster check, or an explicit "you are a helper" instruction); the routing itself ships behind the helper-instance feature gate and isn't live yet.
+
This file serves two purposes with one mechanism:
1. *Crash recovery* — if the session dies mid-work, the live file is all that's left. On 2026-01-22 a session crashed during a 20-minute design discussion and all context was lost because this file wasn't being updated.
2. *Session archive* — at wrap-up the file is renamed into =.ai/sessions/=, becoming the permanent record. No transcription to notes.org; the file IS the record.
@@ -181,6 +187,8 @@ Canonical rule: =~/code/rulesets/claude-rules/cross-project.md=.
Every in-progress task that produces files (drafts, source documents, diagrams, scripts, sub-deliverables) gets a dedicated subdirectory under =<project-root>/working/=, named after the task. All artifacts for that task live in that subdirectory until the task is marked done.
+=working/= is version-controlled from creation — it's the tracked home of in-progress work, never gitignored. Filing on completion *reorganizes* durable artifacts into permanent homes; it doesn't mark when they became durable (they were durable, and tracked, from the start). Genuinely disposable artifacts go in a gitignored =temp/= (or =/tmp=), never =working/=; the install tooling ignores =temp/= in both track and gitignore modes.
+
When the task ships, files are **renamed individually** (standard form: =YYYY-MM-DD-<task-slug>-<descriptor>.<ext>=) and **moved flat** into the appropriate permanent home (typically =assets/= or an area-specific =<area>/assets/=). The working subdirectory is then empty and gets deleted.
***Never rename the directory itself as a substitute for filing.*** The point is to keep =assets/= flat-searchable — a nested =assets/old-tech-deck-2026/slide.png= is harder to find than =assets/2026-05-18-tech-deck-vol2-slide-04-diagram.png=.
@@ -197,7 +205,9 @@ Check =inbox/= at every task boundary (after finishing a unit of work, before re
.ai/scripts/inbox-status -q
#+end_src
-Exit 1 means handoffs are pending — process them per =process-inbox.org=. For each accepted handoff, the act-vs-file rule: *act now* when it's clear, bounded, low-risk, in-scope, and cheaper than deferring — just do it, no asking; *file* otherwise — ask first, with filing as option 1 and "do it now" as option 2; *ask* if unsure. Exception: a proposal to change a shared asset (template workflow, rule, skill, synced script) or a substantive convention never silently acts now — it goes through process-inbox's Skeptical Review and its approval (or park) step. Always reply to a handoff's sender (confirm on accept, the why on reject). Full process, the reply discipline, and the opt-in background-monitor =/loop= recipe live in =monitor-inbox.org=.
+Exit 1 means handoffs are pending — process them per =inbox.org= process mode. For each accepted handoff, the act-vs-file rule: *act now* when it's clear, bounded, low-risk, in-scope, and cheaper than deferring — just do it, no asking; *file* otherwise — ask first, with filing as option 1 and "do it now" as option 2; *ask* if unsure. Exception: a proposal to change a shared asset (template workflow, rule, skill, synced script) or a substantive convention never silently acts now — it goes through the inbox engine's skeptical review and its approval (or park) step. Always reply to a handoff's sender (confirm on accept, the why on reject). Full process, the reply discipline, and the opt-in background-monitor =/loop= recipe live in =inbox.org= monitor mode.
+
+A machine-global =Stop= hook (=inbox-boundary-check.sh=) backs this rule so it isn't prose-only. When handoffs are pending it blocks the turn once and injects the count, so a task boundary can't pass with items unseen. It soft-nudges rather than hard-blocks: on the harness re-entry it steps aside, so a mid-task pause to ask "what's next" is never hijacked into inbox processing. The rule above still governs what to do with the items; the hook only makes sure you look.
** Recursive Reads — Honor =.aiignore=
@@ -236,13 +246,33 @@ Execute the wrap-up workflow (details in Session Protocols section below):
2. Git commit and push all changes
3. Valediction summary
+** "Suspend the session" / "Suspend" / "I need to go" / "Stick a pin in everything"
+
+Execute the suspend workflow ([[file:workflows/suspend.org][suspend.org]]): a capture-only mid-session pause for an abrupt departure. It appends a resume-weighted =SUSPENDED= entry to the Session Log, notes uncommitted work, and LEAVES =.ai/session-context.org= in place so the next startup resumes from it — no archive, no teardown, no valediction. The capture-only counterpart to "wrap it up" (which ends + archives + tears down) and to =/flush= (which prompts =/clear= and resumes the same session). "I need to go" is broad — if it reads as a conversational aside, confirm before suspending.
+
+* Colloquialisms and Expansions
+
+Shorthand phrases Craig uses that expand to a defined action the agent applies without asking. The set is extensible: a project may add its own entries, and new shared shorthands land here.
+
+** "the list": the Before-Close Queue
+
+"Put X on the list" or "add X to the list" appends X to the Before-Close Queue, a FIFO queue of tasks and actions to finish before the session closes. Work it oldest-first at wrap-up, before teardown (=wrap-it-up.org= Step 1 works it before finalizing the Summary), and surface anything unfinished in the valediction rather than dropping it.
+
+The queue lives in the session anchor (=.ai/session-context.org=) under a =* Before-Close Queue= heading. Create the heading on the first "put it on the list" if it's absent, then append one line per item. It's session-scoped: it resets when the anchor is archived at wrap. Anything that must outlive the session is a =todo.org= task instead, not a list item.
+
+** "tell <project> <message>": cross-project handoff
+
+"Tell <project> <message>" drops the message in that project's =inbox/= via =inbox-send= (=python3 .ai/scripts/inbox-send.py <project> --text "<message>"=), the sanctioned cross-project handoff. Never write another project's =todo.org= or =inbox/= directly. Resolve =<project>= the way =inbox-send= does (basename match, dots stripped); if it's ambiguous, ask which project rather than guessing.
+
* User Information
** Calendar Management
Three ways to access Craig's calendars: Google Calendar MCP (preferred, both personal + work accounts), gcalcli (fallback, personal only), Emacs org files (read-only viewer).
-For tool recipes, authentication details, and credentials, see [[file:references/calendar-reference.org][calendar-reference.org]].
+For tool recipes and account details, read the calendar workflows in =.ai/workflows/=: =add-calendar-event.org=, =edit-calendar-event.org=, =delete-calendar-event.org=, =read-calendar-events.org=. They carry the MCP tool names, both account ids, the gcalcli fallback, and the conflict-check discipline.
+
+Credentials are needed only for a re-auth Craig performs himself. The MCP bundle's =mcp/README.org= in the rulesets repo is the authority: =gcp-oauth.keys.json= is gitignored and regenerated at install from a base64 var in the bundle, never committed. Named in prose rather than linked, because that path isn't synced into consuming projects.
** GPG Keys
@@ -346,9 +376,19 @@ Craig runs a pure Wayland setup (Hyprland) and avoids XWayland/Xorg apps.
- Clipboard: Use =wl-copy= and =wl-paste= (NOT =xclip= or =xsel=)
- Window management: Use Hyprland commands (NOT =xkill=, =xdotool=, etc.)
- Prefer Wayland-native tools over X11 equivalents
-- Open URLs in browser: Use =google-chrome-stable "URL" &>/dev/null &=
- - The =&>/dev/null &= is required to detach the process and suppress output
- - Without it, the command may appear to hang or produce no result
+- Open URLs in browser: invoke Chrome directly — never =xdg-open=, which returned success in a home session on 2026-07-26 while no tab appeared.
+
+ Chrome is normally already running, and in that case it hands the URL to the live session and exits immediately (rc 0), printing =Opening in existing browser session.= on *stdout*. So run it in the foreground and read that line as the confirmation the tab actually opened:
+
+ #+begin_src bash
+ google-chrome-stable --new-tab "URL"
+ #+end_src
+
+ Don't redirect stdout away while checking for that line — verified 2026-07-27 on ratio: with =2>/dev/null= the message still appears (it isn't stderr), and with =>/dev/null= it vanishes.
+
+ Several URLs in one invocation open as separate tabs (=google-chrome-stable --new-tab "URL1" "URL2"=). Pass them as separate words or an array — the Bash tool runs zsh, which does not word-split an unquoted =$urls= variable, so a space-joined string arrives as one malformed argument (see the zsh note below).
+
+ *Cold start.* If Chrome is *not* already running, the command becomes the browser process and blocks. Detach that case with =&>/dev/null &=, accepting that the confirmation line is discarded — there is no session to confirm into. Don't apply the detach form unconditionally: it suppresses the very output the warm path is verified by.
*** Shell aliases (=ls= → =exa=)
Craig's shell aliases =ls= to =exa=, which prints nothing to non-TTY pipes (e.g. when capturing =ls= output in a Bash tool call). The result looks like the directory is empty when it isn't.
@@ -357,6 +397,12 @@ Craig's shell aliases =ls= to =exa=, which prints nothing to non-TTY pipes (e.g.
- Applies to =ls -la=, =ls -t=, glob expansions piped through =ls=, and any =ls= invocation whose output gets read programmatically.
- Symptom if forgotten: the Bash tool returns empty output and you mistakenly conclude the directory is empty.
+*** zsh does not word-split unquoted variables
+The Bash tool runs zsh, which (unlike bash) does not split an unquoted =$var= on whitespace. =chrome $urls= passes all the space-joined URLs as one malformed argument.
+
+- Loop over the values, use an array, or force the split with =${=var}=.
+- Symptom if forgotten: a command that "works in bash" gets one garbled argument and fails, often silently (from the takuzu session, 2026-07-11).
+
** Miscellaneous Information
- Craig currently lives in New Orleans, LA
- Craig's phone number: 510-316-9357
@@ -396,6 +442,30 @@ Full usage: =notify --help= or see =~/.local/bin/notify=
- =atq= - list all scheduled alarms
- =atrm [number]= - remove an alarm by its queue number
+** Reaching Craig — the notification vocabulary
+
+Two channels, two trigger words. "page me" is the desktop, "text me" is the phone, "text and page me" is both. Pick by where Craig is, and default to both when a run can't tell. Both work from any agent runtime (nothing here is Claude-specific). The words are what Craig says; a run deciding on its own maps the same way (away run texts, at-desk run pages, unsure does both).
+
+- *"page me" — at his laptop/desktop.* A desktop =notify ... --persist= that reaches him on the machine and stays up until dismissed.
+
+ #+begin_src bash
+ notify info "Title" "Message" --persist
+ #+end_src
+
+- *"text me" — away from his machine.* A Signal push to his phone via =agent-text=:
+
+ #+begin_src bash
+ agent-text "Message for Craig's phone"
+ #+end_src
+
+ =agent-text= (in =~/.local/bin= via the rulesets install) sends from the dedicated Signal identity (+15045173983) to Craig's Signal account UUID, firing a normal mobile push. The account is registered on velox (primary) and ratio (linked device), so either sends directly; a machine without it ssh-relays to velox. Verified end to end 2026-07-13 (velox) and 2026-07-20 (ratio). Never target Craig's phone *number* (it reads as unregistered in Signal's directory); the script targets the UUID.
+
+ Caveats: a relay from a non-linked machine needs velox up on the tailnet, and each device holding the account wants a periodic =receive= (the signal-receive timer handles that). The full runbook lives in rulesets =docs/design/=.
+
+- *"text and page me" — both.* Fire =agent-text= and =notify= together. The phone reaches him now, the desktop note waits for his return. This is the default when a run can't tell whether he's away.
+
+On velox, Claude sessions may also have the *signal-mcp* tool (=send_message_to_user=, same identity), fine to use there, but it exists only in velox's local MCP config, so =agent-text= is the portable habit. The tool was named =agent-page= before 2026-07-20; a deprecated =agent-page= shim still delegates to =agent-text=. Do *not* use the old =page-signal= shell script (removed 2026-06-12).
+
* Session Protocols
** CRITICAL: Git Commit Requirements
@@ -421,7 +491,7 @@ When creating commits:
- Keep messages clear and informative
3. **No Claude-tooling artifacts**: Commit messages describe project changes only — the meta-process of how work got shipped stays out of public git history.
- - **ABSOLUTELY NO** mentions of =notes.org=, =session-context.org=, =.ai/sessions/=, =todo.org=, "session wrap-up", or session timestamps (e.g., "Session YYYY-MM-DD HH:MM → ...")
+ - **ABSOLUTELY NO** mentions of =notes.org=, =session-context.org=, =.ai/= (including =.ai/sessions/=), =.claude/=, =CLAUDE.md=, =todo.org=, "session wrap-up", or session timestamps (e.g., "Session YYYY-MM-DD HH:MM → ..."), except when one of those files is itself the change — then name what changed by category, not the surrounding tooling layer
- Subject lines must NEVER start with =session:= as a conventional-commit type — use =docs:=, =refactor:=, =fix:=, =feat:=, =chore:=, etc. (real change categories)
- When a wrap-up commit bundles many changes from a session, describe what /shipped/ (e.g., =refactor: extract RAID logic + add bats testing infrastructure=), not that a session happened
- Same spirit as the no-Claude-attribution rule: the tooling stays invisible in =git log=
@@ -454,7 +524,7 @@ When Craig says this phrase:
- If exact match found: Read and guide through process
3. **Fuzzy match across both directories:** Ask for clarification
- - Example: User says "empty inbox" but we have "inbox-zero.org"
+ - Example: User says "empty inbox" but we have "inbox.org" (roam mode)
- Ask: "Did you mean the 'inbox zero' workflow, or create new 'empty inbox'?"
4. **No match at all:** Offer to create it
@@ -509,12 +579,13 @@ When monitoring a long-running process (rsync, large downloads, builds, VM tests
** "Wrap it up" / "That's a wrap" / "Let's call it a wrap"
-When Craig says any of these phrases (or variations), execute the wrap-up workflow: [[file:workflows/wrap-it-up.org][wrap-it-up.org]]. Four steps:
+When Craig says any of these phrases (or variations), execute the wrap-up workflow: [[file:workflows/wrap-it-up.org][wrap-it-up.org]]. Five load-bearing steps:
1. *Finalize the Summary* in =.ai/session-context.org= (populate the 5 subsections from the Session Log)
2. *Rename* =.ai/session-context.org= → =.ai/sessions/YYYY-MM-DD-HH-MM-description.org=
3. *Git commit + push* to all remotes (see Git Commit Requirements)
-4. *Valediction* — brief, warm, specific closing
+4. *Certify the clean tree* with =git-worktree-gate certify=. Any remaining staged, unstaged, untracked, submodule, or in-progress-operation state blocks wrap entirely; report each path and the exact decision needed. There is no dirty-file deferral.
+5. *Valediction* — brief, warm, specific closing, reachable only after certification
The absence of =.ai/session-context.org= after wrap-up is the signal that the session ended cleanly. If the file is still there at the next session start, the previous session was interrupted.
@@ -533,6 +604,8 @@ Claude needs to add information to =.ai/notes.org=. For large amounts of informa
**The gitignore set follows that same decision.** A project that gitignores =.ai/= (the code-project case) gitignores the whole personal-tooling set: =.ai/=, =.claude/=, =CLAUDE.md=, =AGENTS.md=. =.claude/= is rulesets-owned — copies of =claude-rules/*.md= plus the language bundle's rules, hooks, and settings — and re-synced from rulesets on every startup, so git isn't how it travels between machines; ignoring it also keeps those private rule copies out of the repo, which ignoring =CLAUDE.md= alone would miss. A track-mode project (personal/doc repos, or a team repo that shares config with teammates who don't run rulesets) tracks the set instead. =install-ai.sh= writes the full set at bootstrap in gitignore mode; =scripts/sweep-gitignore-tooling.sh= backfills it idempotently across existing gitignore-mode projects when the set grows.
+**Public reachability decides harder than project type.** Any repo whose remotes include a non-cjennings.net host gitignores the tooling set, whatever kind of project it is — the only exception is a team repo that deliberately shares the config, decided explicitly, never by default. And a private remote is not proof of privacy: a server-side =post-receive --mirror= hook republishes invisibly from the client (the 2026-06-30 =.emacs.d= exposure rode exactly that — a cjennings.net remote mirroring to public GitHub). The sweep recognizes both the anchored (=/.ai/=) and unanchored (=.ai/=) ignore styles — an anchored-style project used to be misread as track-mode and silently skipped — and warns when tracked tooling can reach a non-cjennings.net remote.
+
**Credential-leak concern: gate it on project type, not on the credential itself.** A tracked secret, token, or credentials doc is only a public-leak risk where the repo can reach a public remote — that is, *code projects pushed to public GitHub*, which is exactly why those gitignore =.ai/= and =.claude/=. For *personal / documentation projects* (the =~/projects/= set: elibrary, home, finances, health, philosophy, etc.), the git remote is a private single-user repo on =cjennings.net=, so tracked credentials inside =.ai/= files are fine — that's the design, the project history IS the project. Do NOT raise a leak warning or suggest gitignoring a secret for these. When the question "is this a leak / should we gitignore this secret?" comes up, decide it on *which kind of project and remote* this is, never on the mere presence of a credential in a tracked file.
**When to break out documents:**
diff --git a/claude-templates/.ai/references/calendar-reference.org b/claude-templates/.ai/references/calendar-reference.org
deleted file mode 100644
index b44c0f1..0000000
--- a/claude-templates/.ai/references/calendar-reference.org
+++ /dev/null
@@ -1,66 +0,0 @@
-#+TITLE: Calendar Reference
-#+AUTHOR: Craig Jennings & Claude
-
-Tool recipes, authentication, and credentials for Craig's calendar
-setup. Three access methods, in order of preference.
-
-* Google Calendar MCP Server (preferred for all calendar operations)
-
-Craig has the =@cocal/google-calendar-mcp= MCP server configured at user scope (=~/.claude.json=). It provides full read/write access to Google Calendar via MCP tools.
-
-Two accounts are authenticated:
-- *personal* — craigmartinjennings@gmail.com (primary: "Craig Google")
-- *work* — craig.jennings@deepsat.com (primary: "Craig Deepsat")
-
-MCP tools available:
-- =list-events=, =search-events=, =get-event= — read events
-- =create-event=, =create-events= — add events
-- =update-event= — modify events
-- =delete-event= — remove events
-- =list-calendars=, =list-colors= — calendar metadata
-- =get-freebusy= — check availability
-- =manage-accounts= — add/remove/list authenticated accounts
-- =respond-to-event= — accept/decline invitations
-- =get-current-time= — current time in any timezone
-
-Use =account_id: "personal"= or =account_id: "work"= to specify which account.
-
-Default calendar for adding events: "Craig Google" (personal account).
-
-Calendar workflows are available alongside this reference: add-calendar-event, edit-calendar-event, delete-calendar-event, read-calendar-events.
-
-If re-authentication is needed:
-- Use the =manage-accounts= MCP tool with =action: "add"= and the account nickname
-- OAuth credentials: =~/projects/homelab/assets/gcp-oauth.keys.json=
-- Google Cloud app is in production mode (tokens don't expire after 7 days)
-- See =~/projects/homelab/.ai/gcalcli-setup.org= for Google Cloud project details
-
-* gcalcli (fallback for personal account only)
-
-Craig has =gcalcli= installed via pipx, authenticated to his personal Google account only.
-
-#+begin_src bash
-gcalcli agenda # upcoming events
-gcalcli calw # weekly view
-gcalcli add --title "..." --when "..." --duration "60" # add event
-gcalcli search "..." # search events
-gcalcli delete "..." # delete event
-#+end_src
-
-Use =--calendar "Craig Google"= when adding events.
-
-gcalcli does NOT have access to the work (DeepSat) calendar. Use the MCP server for work calendar operations.
-
-If gcalcli needs re-authentication, credentials are stored in the homelab project: =~/projects/homelab/assets/gcalcli-client-secret.json.gpg= (GPG encrypted).
-
-* Emacs org files (read-only, for viewing schedules)
-
-Craig's calendars are at: =~/.emacs.d/data/*cal.org= (gcal.org, dcal.org, pcal.org)
-
-These files are **READ-ONLY** — NEVER add anything to them.
-
-Use this to:
-- Check meeting times and schedules
-- Verify when events occurred
-- See what's upcoming
-- Note: only updated periodically when Emacs is running — may be stale
diff --git a/claude-templates/.ai/scripts/agent-lock b/claude-templates/.ai/scripts/agent-lock
new file mode 100755
index 0000000..634412c
--- /dev/null
+++ b/claude-templates/.ai/scripts/agent-lock
@@ -0,0 +1,248 @@
+#!/usr/bin/env bash
+# agent-lock — a mkdir-atomic advisory lock for agent workflows.
+#
+# Why not flock: every Bash call an agent makes is its own short-lived shell,
+# so an flock taken in one /loop turn is gone by the next. This helper persists
+# the lock on disk between calls (an atomic mkdir is the acquire), and a crashed
+# holder's lock self-clears via age-based staleness reclaim instead of wedging
+# every later acquire.
+#
+# Serves both of sentry's locks (the single-runner lock and the roam-write
+# lock); callers pass a name, never a path — the helper owns the path scheme.
+#
+# Usage:
+# agent-lock acquire <name> [--ttl=SECONDS] [--wait[=SECONDS]]
+# Atomic acquire. exit 0 on win (fresh, or reclaimed from a stale holder);
+# exit 1 when a live lock already holds <name> (deferred — a note names the
+# holder on stderr). --wait polls up to SECONDS (default 30) before
+# deferring; without it, acquire is single-shot win-or-lose. --ttl records
+# the staleness horizon in the lock's metadata (default below).
+# agent-lock refresh <name>
+# Heartbeat: re-touch a held lock's mtime so it stays young. A runner
+# refreshes its own lock between passes, so a live run's lock is never older
+# than one pass and the TTL sizes to the longest single pass. exit 1 if the
+# lock is absent (nothing to refresh).
+# agent-lock release <name>
+# Remove the lock. Idempotent: exit 0 even if already free.
+# agent-lock status <name>
+# Print "free" | "held ..." | "stale ..." plus metadata. exit 0 (a query
+# never fails on lock state).
+# agent-lock path <name>
+# Print the resolved lock-directory path without creating it.
+#
+# Lock home (the helper owns this; callers pass names only):
+# $AGENT_LOCK_DIR/<name>/ when AGENT_LOCK_DIR is set (tests / advanced)
+# $XDG_RUNTIME_DIR/agent-locks/<name>/ the tmpfs runtime dir /run/user/<uid>
+# (host-local, out of every repo,
+# cleared on reboot). XDG_RUNTIME_DIR is
+# the standard handle for it and is set
+# in sentry's interactive launch.
+# ${XDG_CACHE_HOME:-~/.cache}/agent-locks/<name>/ fallback where no runtime
+# dir exists (XDG_RUNTIME_DIR unset or
+# unwritable — a headless/container box)
+#
+# tmpfs residence is deliberate: a lock under ~/org/roam would ride roam-sync's
+# `git add -A` to the other machine as a phantom hold. Host-locality is by
+# construction, and reboot clears any lock a crash left behind for free.
+#
+# Staleness is age-based on the metadata file's mtime versus the lock's own
+# recorded TTL. Heartbeat re-touches the mtime; a reclaim is always surfaced,
+# never silent.
+
+set -euo pipefail
+
+DEFAULT_TTL=600 # 10 min: sized to the longest single sentry pass, since a
+ # live runner heartbeats between passes and stays young.
+DEFAULT_WAIT=30 # bounded-wait budget for --wait (capture-guard's shape).
+WAIT_INTERVAL=3 # poll cadence while waiting on a busy lock.
+
+usage() {
+ echo "usage: agent-lock {acquire|refresh|release|status|path} <name> [--ttl=N] [--wait[=N]]" >&2
+ exit 2
+}
+
+# Resolve the base directory that holds all lock dirs, per the home scheme above.
+lock_base() {
+ if [ -n "${AGENT_LOCK_DIR:-}" ]; then
+ printf '%s\n' "$AGENT_LOCK_DIR"
+ elif [ -n "${XDG_RUNTIME_DIR:-}" ] && [ -d "$XDG_RUNTIME_DIR" ] && [ -w "$XDG_RUNTIME_DIR" ]; then
+ printf '%s/agent-locks\n' "$XDG_RUNTIME_DIR"
+ else
+ printf '%s/agent-locks\n' "${XDG_CACHE_HOME:-$HOME/.cache}"
+ fi
+}
+
+# Validate a lock name: non-empty, no path separators (so a name can never
+# escape the base dir).
+valid_name() {
+ case "$1" in
+ ''|*/*|.|..) return 1 ;;
+ *) return 0 ;;
+ esac
+}
+
+lock_dir() { printf '%s/%s\n' "$(lock_base)" "$1"; }
+meta_path() { printf '%s/meta\n' "$(lock_dir "$1")"; }
+
+# Read a key from a lock's metadata file; empty if absent.
+meta_get() {
+ local key="$1" file="$2"
+ [ -f "$file" ] || return 0
+ sed -n "s/^${key}=//p" "$file" | head -n1
+}
+
+# Age of a lock in whole seconds, from the metadata mtime.
+lock_age() {
+ local file="$1" mtime now
+ mtime=$(stat -c %Y "$file" 2>/dev/null) || return 1
+ now=$(date +%s)
+ printf '%s\n' "$((now - mtime))"
+}
+
+# True when a lock dir exists but its age exceeds its recorded TTL.
+is_stale() {
+ local name="$1" file age ttl
+ file="$(meta_path "$name")"
+ [ -f "$file" ] || return 1
+ age="$(lock_age "$file")" || return 1
+ ttl="$(meta_get ttl "$file")"
+ [ -n "$ttl" ] || ttl="$DEFAULT_TTL"
+ [ "$age" -gt "$ttl" ]
+}
+
+# Write the metadata file for a freshly-taken lock.
+write_meta() {
+ local name="$1" ttl="$2" file
+ file="$(meta_path "$name")"
+ {
+ printf 'pid=%s\n' "$$"
+ printf 'host=%s\n' "$(uname -n)"
+ printf 'acquired=%s\n' "$(date +%Y-%m-%dT%H:%M:%S%z)"
+ printf 'ttl=%s\n' "$ttl"
+ } > "$file"
+}
+
+# One-line holder description for surfaced notes.
+holder_desc() {
+ local file="$1"
+ printf "pid=%s host=%s age=%ss ttl=%ss" \
+ "$(meta_get pid "$file")" "$(meta_get host "$file")" \
+ "$(lock_age "$file" 2>/dev/null || echo '?')" "$(meta_get ttl "$file")"
+}
+
+# Attempt a single atomic acquire. exit 0 win, 1 busy (live holder).
+try_acquire() {
+ local name="$1" ttl="$2" dir file
+ dir="$(lock_dir "$name")"
+ file="$(meta_path "$name")"
+ mkdir -p "$(lock_base)"
+
+ if mkdir "$dir" 2>/dev/null; then
+ write_meta "$name" "$ttl"
+ return 0
+ fi
+
+ # Directory exists. Reclaim it if the holder is stale; otherwise it's busy.
+ if is_stale "$name"; then
+ # Claim the stale dir atomically before removing it. `mv` of a directory is
+ # atomic, so when two acquirers both see the lock stale, only one's rename
+ # of $dir succeeds — the other's fails because $dir is already gone, and it
+ # falls through to busy. Never `rm -rf $dir` directly: a plain remove lets
+ # the loser delete the winner's freshly-created lock and double-acquire.
+ local claimed="$dir.stale.$$"
+ if mv "$dir" "$claimed" 2>/dev/null; then
+ echo "agent-lock: reclaimed stale lock '$name' ($(holder_desc "$claimed/meta"))" >&2
+ rm -rf "$claimed"
+ # mkdir stays the sole grant: a concurrent fresh acquirer may win here,
+ # in which case our mkdir fails and we correctly defer to it.
+ if mkdir "$dir" 2>/dev/null; then
+ write_meta "$name" "$ttl"
+ return 0
+ fi
+ fi
+ fi
+ return 1
+}
+
+cmd_acquire() {
+ local name="$1"; shift
+ local ttl="$DEFAULT_TTL" wait_total=0
+ while [ $# -gt 0 ]; do
+ case "$1" in
+ --ttl=*) ttl="${1#--ttl=}" ;;
+ --ttl) shift; ttl="${1:-}" ;;
+ --wait) wait_total="$DEFAULT_WAIT" ;;
+ --wait=*) wait_total="${1#--wait=}" ;;
+ *) usage ;;
+ esac
+ shift
+ done
+ case "$ttl" in ''|*[!0-9]*) usage ;; esac
+ case "$wait_total" in *[!0-9]*) usage ;; esac
+
+ local elapsed=0
+ while :; do
+ if try_acquire "$name" "$ttl"; then
+ exit 0
+ fi
+ if [ "$elapsed" -ge "$wait_total" ]; then
+ echo "agent-lock: '$name' busy ($(holder_desc "$(meta_path "$name")")); deferring" >&2
+ exit 1
+ fi
+ local remaining=$((wait_total - elapsed)) step
+ step=$(( remaining < WAIT_INTERVAL ? remaining : WAIT_INTERVAL ))
+ sleep "$step"
+ elapsed=$((elapsed + step))
+ done
+}
+
+cmd_refresh() {
+ local name="$1" file
+ file="$(meta_path "$name")"
+ [ -f "$file" ] || exit 1
+ # Re-stamp acquired and bump mtime so the age clock restarts.
+ local ttl; ttl="$(meta_get ttl "$file")"; [ -n "$ttl" ] || ttl="$DEFAULT_TTL"
+ write_meta "$name" "$ttl"
+ exit 0
+}
+
+cmd_release() {
+ local name="$1" dir
+ dir="$(lock_dir "$name")"
+ rm -rf "$dir"
+ exit 0
+}
+
+cmd_status() {
+ local name="$1" dir file
+ dir="$(lock_dir "$name")"
+ file="$(meta_path "$name")"
+ if [ ! -d "$dir" ]; then
+ echo "free $name"
+ exit 0
+ fi
+ local state="held"
+ is_stale "$name" && state="stale"
+ echo "$state $name pid=$(meta_get pid "$file") host=$(meta_get host "$file") acquired=$(meta_get acquired "$file") ttl=$(meta_get ttl "$file") age=$(lock_age "$file" 2>/dev/null || echo '?')s"
+ exit 0
+}
+
+cmd_path() {
+ lock_dir "$1"
+ exit 0
+}
+
+[ $# -ge 1 ] || usage
+subcmd="$1"; shift
+[ $# -ge 1 ] || usage
+name="$1"; shift
+valid_name "$name" || usage
+
+case "$subcmd" in
+ acquire) cmd_acquire "$name" "$@" ;;
+ refresh) cmd_refresh "$name" ;;
+ release) cmd_release "$name" ;;
+ status) cmd_status "$name" ;;
+ path) cmd_path "$name" ;;
+ *) usage ;;
+esac
diff --git a/claude-templates/.ai/scripts/agent-roster b/claude-templates/.ai/scripts/agent-roster
new file mode 100755
index 0000000..f32b744
--- /dev/null
+++ b/claude-templates/.ai/scripts/agent-roster
@@ -0,0 +1,84 @@
+#!/usr/bin/env bash
+# agent-roster — list other live Claude agents working in this project.
+#
+# The single source of "who else is live in this project." Both launchers
+# (ai --helper) and the in-session startup check call this rather than
+# reimplementing the scan, so concurrent-agent detection has one definition.
+#
+# Scan (stateless): enumerate running Claude processes (pgrep -x claude), read
+# each one's working directory from /proc/<pid>/cwd, keep those whose cwd is
+# the project root or inside it, and drop the scanner's own process ancestry
+# (walk parent pids from /proc/self up). What remains is the set of *other*
+# live agents in this project.
+#
+# Usage: agent-roster [project-root] (default: $PWD)
+# Output: one "pid<TAB>cwd" line per other agent
+# Exit: 0 = alone (no other agents)
+# 1 = one or more other agents (and printed)
+# 2 = roster unavailable (no /proc; non-Linux or absent)
+#
+# Known limits, accepted for v1: a session not running as a local process on
+# this machine (a cloud session against the same checkout) is invisible, and
+# the match is on process cwd, so an agent started from outside the project
+# tree isn't seen. Both are edge shapes the operator created deliberately.
+#
+# The boundary (pgrep, /proc, self pid) is injectable so the filtering logic
+# is testable without spawning real agents: ROSTER_PGREP, ROSTER_PROC,
+# ROSTER_SELF_PID. Production defaults need no environment.
+set -euo pipefail
+
+PGREP="${ROSTER_PGREP:-pgrep}"
+PROC="${ROSTER_PROC:-/proc}"
+SELF_PID="${ROSTER_SELF_PID:-$$}"
+
+root="${1:-$PWD}"
+root="${root%/}"
+
+# Linux /proc is the substrate. Absent (non-Linux, or unreadable) means the
+# scan can't run; say so explicitly rather than reporting a false "alone".
+if [ ! -d "$PROC" ]; then
+ echo "agent-roster: roster unavailable (no $PROC; non-Linux or absent)" >&2
+ exit 2
+fi
+
+# pgrep is the enumeration boundary. Without it the scan can't run, and the
+# no-match exit code (1) below is indistinguishable from "tool missing" once
+# swallowed, so check up front rather than report a false "alone".
+if ! command -v "$PGREP" >/dev/null 2>&1; then
+ echo "agent-roster: roster unavailable ($PGREP not found)" >&2
+ exit 2
+fi
+
+# Build the scanner's ancestry set: SELF_PID and every parent up to init.
+# A Claude found by pgrep that lands in this set is the current session (or its
+# launcher chain), not another agent.
+ancestry=" "
+pid="$SELF_PID"
+while [ -n "$pid" ] && [ "$pid" != "0" ] && [ "$pid" != "1" ]; do
+ ancestry="${ancestry}${pid} "
+ status="$PROC/$pid/status"
+ [ -r "$status" ] || break
+ pid="$(awk '/^PPid:/{print $2; exit}' "$status")"
+done
+
+found=0
+while read -r candidate; do
+ [ -n "$candidate" ] || continue
+ case "$ancestry" in
+ *" $candidate "*) continue ;; # scanner's own ancestry
+ esac
+ # cwd may be gone if the process exited between pgrep and here; skip it.
+ cwd="$(readlink "$PROC/$candidate/cwd" 2>/dev/null)" || continue
+ [ -n "$cwd" ] || continue
+ # Keep only agents at or inside the project root. The trailing slashes make
+ # the prefix test exact, so /foo/project-other doesn't match /foo/project.
+ case "$cwd/" in
+ "$root"/*) ;;
+ *) continue ;;
+ esac
+ printf '%s\t%s\n' "$candidate" "$cwd"
+ found=1
+done < <("$PGREP" -x claude 2>/dev/null || true)
+
+[ "$found" -eq 1 ] && exit 1
+exit 0
diff --git a/claude-templates/.ai/scripts/apkg-to-orgdrill.py b/claude-templates/.ai/scripts/apkg-to-orgdrill.py
new file mode 100755
index 0000000..79e24a4
--- /dev/null
+++ b/claude-templates/.ai/scripts/apkg-to-orgdrill.py
@@ -0,0 +1,251 @@
+#!/usr/bin/env -S uv run --script
+# /// script
+# requires-python = ">=3.11"
+# dependencies = []
+# ///
+"""Convert an Anki .apkg deck into an org-drill file (inverse of flashcard-to-anki.py).
+
+The flashcard pipeline is otherwise one-directional (org-drill -> apkg).
+Decks curated on the phone, and orphaned apkgs whose .org source was never
+saved, can't get back into the org source of truth. This recovers them.
+
+Reading needs no third-party library: an apkg is a zip holding
+collection.anki2 / .anki21 (an Anki sqlite db) plus a media blob, so stdlib
+zipfile + sqlite3 suffice. genanki is only needed to write apkgs, not read
+them.
+
+Mapping (mirrors flashcard-to-anki.py's parse/build, inverted):
+ - Deck name (from the apkg) -> #+TITLE:
+ - Note Front -> ** <Front> :drill:
+ - Note Back (HTML) -> entry body (<br> -> newlines,
+ &amp;/&lt;/&gt; unescaped,
+ <hr id="answer"> stripped)
+ - Note tag -> * <tag> section grouping
+ (best-effort: the tag is a slug,
+ so it won't round-trip to the exact
+ original section title — a human
+ retitles)
+ - A fresh :ID: UUID per card -> so the output is org-drill-valid
+
+GUIDs in flashcard-to-anki.py are derived from the Front text, not the
+:ID:, so a deck regenerated from recovered org still matches existing phone
+cards by Front. Only Front/Back (Basic) note types convert; other models
+(cloze, etc.) are skipped with a warning rather than silently dropped.
+
+Usage:
+ apkg-to-orgdrill.py <input.apkg> # one <deck-slug>.org per deck in cwd
+ apkg-to-orgdrill.py <input.apkg> --output-dir DIR
+ apkg-to-orgdrill.py <input.apkg> --deck "Name" --output deck.org
+"""
+from __future__ import annotations
+
+import argparse
+import json
+import re
+import sqlite3
+import sys
+import tempfile
+import uuid
+import zipfile
+from collections import OrderedDict
+from dataclasses import dataclass
+from pathlib import Path
+
+# Collection member names Anki uses, newest schema first.
+COLLECTION_NAMES = ("collection.anki21", "collection.anki2")
+
+_BR_RE = re.compile(r"<br\s*/?>", re.IGNORECASE)
+_ANSWER_HR_RE = re.compile(r'<hr id="answer">', re.IGNORECASE)
+_MEDIA_RE = re.compile(r"<img\b|\[sound:|<audio\b|<video\b", re.IGNORECASE)
+
+
+@dataclass
+class Note:
+ deck: str
+ front: str
+ back_html: str
+ tag: str
+
+
+def html_to_org_body(back_html: str) -> list[str]:
+ """Invert flashcard-to-anki.py's back-of-card HTML into org body lines.
+
+ <br> (all spellings) and a stray answer <hr> become line breaks; the
+ entity unescape undoes escape_html, which escaped ``&`` first — so ``&``
+ is unescaped last here, or a literally-escaped ``&lt;`` in the source
+ would wrongly collapse to ``<``.
+ """
+ if not back_html:
+ return []
+ s = _ANSWER_HR_RE.sub("\n", back_html)
+ s = _BR_RE.sub("\n", s)
+ s = s.replace("&lt;", "<").replace("&gt;", ">").replace("&amp;", "&")
+ return s.split("\n")
+
+
+def _slug(title: str) -> str:
+ return re.sub(r"[^a-z0-9]+", "-", title.lower()).strip("-")
+
+
+def _read_collection(db_path: Path) -> list[Note]:
+ con = sqlite3.connect(db_path)
+ try:
+ row = con.execute("SELECT decks, models FROM col LIMIT 1").fetchone()
+ if row is None:
+ raise ValueError("collection has no col row")
+ decks_json, models_json = row
+ decks = {int(k): v["name"] for k, v in json.loads(decks_json).items()}
+ models = {
+ int(k): [f["name"] for f in v["flds"]]
+ for k, v in json.loads(models_json).items()
+ }
+
+ # A note's deck comes from its card; the Default deck (id 1) carries
+ # no cards from this pipeline, so it never shows up here.
+ nid_to_did: dict[int, int] = {}
+ for nid, did in con.execute("SELECT nid, did FROM cards"):
+ nid_to_did.setdefault(nid, did)
+
+ notes: list[Note] = []
+ for nid, mid, flds, tags in con.execute(
+ "SELECT id, mid, flds, tags FROM notes"
+ ):
+ field_names = models.get(mid)
+ if not field_names or "Front" not in field_names or "Back" not in field_names:
+ print(
+ f"apkg-to-orgdrill: skip note {nid} — model is not a Front/Back "
+ f"type (fields={field_names})",
+ file=sys.stderr,
+ )
+ continue
+ fields = flds.split("\x1f")
+ fi, bi = field_names.index("Front"), field_names.index("Back")
+ front = fields[fi] if fi < len(fields) else ""
+ back_html = fields[bi] if bi < len(fields) else ""
+
+ did = nid_to_did.get(nid)
+ if did is None:
+ continue # note with no card — orphan
+ deck = decks.get(did)
+ if deck is None:
+ continue
+
+ tag_list = tags.split()
+ tag = tag_list[0] if tag_list else "drill"
+
+ if _MEDIA_RE.search(back_html):
+ print(
+ f"apkg-to-orgdrill: note {nid} references media; org has no "
+ f"media path (left inline for a human to resolve)",
+ file=sys.stderr,
+ )
+ notes.append(Note(deck=deck, front=front, back_html=back_html, tag=tag))
+ return notes
+ finally:
+ con.close()
+
+
+def read_apkg(path: Path) -> list[Note]:
+ """Read an .apkg and return its Front/Back notes. Raises on a malformed file."""
+ with zipfile.ZipFile(path) as z: # BadZipFile if it isn't a zip
+ names = set(z.namelist())
+ col_name = next((n for n in COLLECTION_NAMES if n in names), None)
+ if col_name is None:
+ raise ValueError(f"{path}: no collection.anki2/.anki21 inside the apkg")
+ with tempfile.TemporaryDirectory() as td:
+ db_path = Path(td) / col_name
+ db_path.write_bytes(z.read(col_name))
+ return _read_collection(db_path)
+
+
+def notes_to_org(notes: list[Note], deck_name: str, *, new_id=None) -> str:
+ """Render one deck's notes as an org-drill file in the house shape."""
+ if new_id is None:
+ new_id = lambda: str(uuid.uuid4()) # noqa: E731
+ groups: "OrderedDict[str, list[Note]]" = OrderedDict()
+ for n in notes:
+ groups.setdefault(n.tag, []).append(n)
+
+ lines: list[str] = [f"#+TITLE: {deck_name}", ""]
+ for tag, group in groups.items():
+ lines.append(f"* {tag}")
+ for n in group:
+ lines.append(f"** {n.front} :drill:")
+ lines.append(":PROPERTIES:")
+ lines.append(f":ID: {new_id()}")
+ lines.append(":END:")
+ lines.extend(html_to_org_body(n.back_html))
+ lines.append("")
+ return "\n".join(lines).rstrip("\n") + "\n"
+
+
+def convert(apkg_path: Path, *, new_id=None) -> "OrderedDict[str, str]":
+ """apkg -> {deck_name: org_text}, one entry per deck that has Front/Back cards."""
+ by_deck: "OrderedDict[str, list[Note]]" = OrderedDict()
+ for n in read_apkg(apkg_path):
+ by_deck.setdefault(n.deck, []).append(n)
+ out: "OrderedDict[str, str]" = OrderedDict()
+ for deck, deck_notes in by_deck.items():
+ out[deck] = notes_to_org(deck_notes, deck, new_id=new_id)
+ return out
+
+
+def main() -> int:
+ parser = argparse.ArgumentParser(
+ description="Convert an Anki .apkg deck into an org-drill file.",
+ )
+ parser.add_argument("input", type=Path, help="Path to the .apkg file.")
+ parser.add_argument("--deck", help="Only convert the deck with this exact name.")
+ parser.add_argument(
+ "--output",
+ type=Path,
+ help="Output .org path. Requires a single deck (use --deck to pick one).",
+ )
+ parser.add_argument(
+ "--output-dir",
+ type=Path,
+ help="Directory for per-deck .org files (default: current directory).",
+ )
+ args = parser.parse_args()
+
+ input_path = args.input.expanduser().resolve()
+ if not input_path.is_file():
+ print(f"error: {input_path} not found", file=sys.stderr)
+ return 1
+
+ by_deck = convert(input_path)
+ if args.deck:
+ by_deck = OrderedDict((k, v) for k, v in by_deck.items() if k == args.deck)
+ if not by_deck:
+ print(f"error: no deck named {args.deck!r} in {input_path}", file=sys.stderr)
+ return 1
+ if not by_deck:
+ print(f"error: no Front/Back cards found in {input_path}", file=sys.stderr)
+ return 1
+
+ if args.output:
+ if len(by_deck) != 1:
+ print(
+ f"error: --output needs a single deck; {input_path} has "
+ f"{len(by_deck)} ({', '.join(by_deck)}). Use --deck or --output-dir.",
+ file=sys.stderr,
+ )
+ return 1
+ out = args.output.expanduser().resolve()
+ out.parent.mkdir(parents=True, exist_ok=True)
+ deck, org = next(iter(by_deck.items()))
+ out.write_text(org, encoding="utf-8")
+ print(f"wrote {out} ({org.count(':drill:')} cards, deck {deck!r})")
+ return 0
+
+ out_dir = (args.output_dir or Path.cwd()).expanduser().resolve()
+ out_dir.mkdir(parents=True, exist_ok=True)
+ for deck, org in by_deck.items():
+ out = out_dir / f"{_slug(deck) or 'deck'}.org"
+ out.write_text(org, encoding="utf-8")
+ print(f"wrote {out} ({org.count(':drill:')} cards, deck {deck!r})")
+ return 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/claude-templates/.ai/scripts/capture-guard b/claude-templates/.ai/scripts/capture-guard
new file mode 100755
index 0000000..6c01f2f
--- /dev/null
+++ b/claude-templates/.ai/scripts/capture-guard
@@ -0,0 +1,91 @@
+#!/usr/bin/env bash
+# capture-guard — detect live org-capture buffers visiting a target file
+# before a workflow edits that file on disk.
+#
+# Editing a file on disk while Emacs has an indirect org-capture buffer
+# cloned from it reverts the base buffer underneath the capture, wedging it:
+# the capture can no longer finalize cleanly with C-c C-c, and a freshly-typed
+# item can be lost or written back against post-edit content. inbox.org
+# roam mode Phase D edits ~/org/roam/inbox.org, the file Craig captures into constantly,
+# so it calls this guard first. See claude-rules/emacs.md.
+#
+# Usage: capture-guard [--wait[=SECONDS]] [TARGET_FILE] (default ~/org/roam/inbox.org)
+#
+# Single-shot (default): check once.
+# exit 0 — safe to edit: no Emacs, daemon unreachable, or no capture buffer
+# visits TARGET_FILE.
+# exit 1 — a live capture buffer visits TARGET_FILE; its name(s) printed to
+# stdout, comma-separated.
+#
+# --wait[=SECONDS]: poll until the capture clears or SECONDS elapse (default
+# 30), re-checking every ~10s. Org captures are usually transient — a few
+# seconds of mid-finalize state — so a short wait clears most false alarms
+# before a caller has to surface or skip. Same exit codes: exit 0 the moment
+# it's clear, exit 1 if still blocked at the deadline (last buffer list on
+# stdout). The common case (nothing capturing) returns instantly without
+# sleeping.
+#
+# Conservative by construction: any uncertainty (no Emacs, query failure)
+# resolves to "safe," so the guard never blocks a workflow that would have
+# been fine. It only stops the one case it can positively confirm.
+
+set -euo pipefail
+
+WAIT_TOTAL=0
+case "${1:-}" in
+ --wait) WAIT_TOTAL=30; shift ;;
+ --wait=*) WAIT_TOTAL="${1#--wait=}"; shift ;;
+esac
+
+TARGET="${1:-$HOME/org/roam/inbox.org}"
+INTERVAL=10
+
+# Names of capture buffers whose base buffer visits TARGET. file-equal-p
+# normalizes symlinks and ./.. so the match survives path spelling; it also
+# returns nil when TARGET doesn't exist, which collapses to "safe" below.
+lisp='(let ((target (expand-file-name "'"$TARGET"'")))
+ (mapconcat (function buffer-name)
+ (seq-filter
+ (lambda (b)
+ (and (string-prefix-p "CAPTURE" (buffer-name b))
+ (let* ((base (or (buffer-base-buffer b) b))
+ (f (buffer-file-name base)))
+ (and f (file-equal-p f target)))))
+ (buffer-list))
+ ","))'
+
+LAST_BUFS=""
+
+# detect — return 0 (safe) or 1 (blocked, name(s) in LAST_BUFS). Any
+# uncertainty resolves to safe, matching the single-shot contract.
+detect() {
+ command -v emacsclient >/dev/null 2>&1 || return 0
+ emacsclient -e t >/dev/null 2>&1 || return 0
+ local bufs
+ bufs="$(emacsclient -e "$lisp" 2>/dev/null)" || return 0
+ bufs="${bufs#\"}"
+ bufs="${bufs%\"}"
+ if [ -n "$bufs" ]; then
+ LAST_BUFS="$bufs"
+ return 1
+ fi
+ return 0
+}
+
+# Poll loop. With WAIT_TOTAL=0 (single-shot) it checks once and falls straight
+# through to the exit-1 branch on a block, never sleeping. Each sleep is capped
+# to the remaining budget so a short --wait never overshoots its deadline.
+elapsed=0
+while :; do
+ if detect; then
+ exit 0
+ fi
+ if [ "$elapsed" -ge "$WAIT_TOTAL" ]; then
+ echo "$LAST_BUFS"
+ exit 1
+ fi
+ remaining=$((WAIT_TOTAL - elapsed))
+ step=$((remaining < INTERVAL ? remaining : INTERVAL))
+ sleep "$step"
+ elapsed=$((elapsed + step))
+done
diff --git a/claude-templates/.ai/scripts/cj-remove-block.py b/claude-templates/.ai/scripts/cj-remove-block.py
index 71c7b3d..d5137a3 100755
--- a/claude-templates/.ai/scripts/cj-remove-block.py
+++ b/claude-templates/.ai/scripts/cj-remove-block.py
@@ -16,8 +16,12 @@ Companion to the /respond-to-cj-comments skill and to cj-scan.py.
from __future__ import annotations
import argparse
+import os
import re
+import shutil
import sys
+import tempfile
+from datetime import datetime
from pathlib import Path
SRC_OPEN_RE = re.compile(r"^\s*#\+begin_src\s+cj:", re.IGNORECASE)
@@ -57,12 +61,83 @@ def looks_like_cj_range(lines: list[str], start: int, end: int) -> tuple[bool, s
f"Line {end} does not look like a #+end_src closing fence "
f"(got: {last[:60]!r})"
)
+
+ # The range must hold exactly ONE block. Checking only the first and last
+ # lines let a drifted range run from one block's opener to a *later* block's
+ # closer: validation passed and the removal silently deleted everything
+ # between, prose and headings included. Drift is the case this check exists
+ # for, so it has to look inside the range, not just at its ends.
+ for offset, line in enumerate(lines[start:end - 1], start=start + 1):
+ if SRC_CLOSE_RE.match(line):
+ return False, (
+ f"Range {start}..{end} covers more than one cj block — "
+ f"a #+end_src appears at line {offset}, before the range ends. "
+ f"Re-scan for current line numbers; removing this range would "
+ f"delete everything between the two blocks."
+ )
+ if SRC_OPEN_RE.match(line):
+ return False, (
+ f"Range {start}..{end} covers more than one cj block — "
+ f"a second #+begin_src cj: appears at line {offset}. "
+ f"Re-scan for current line numbers."
+ )
return True, ""
+def _backup(path: Path) -> Path:
+ """Copy path to /tmp before mutating it, mirroring lint-org.el's convention.
+
+ These are Craig's org files. lint-org.el, the other tool that rewrites them,
+ leaves a /tmp copy before touching anything; this matches it so a bad edit is
+ always recoverable without reaching for git (which only reaches the last
+ commit, losing intra-session work).
+ """
+ stamp = datetime.now().strftime("%Y%m%d-%H%M%S")
+ base = Path(tempfile.gettempdir()) / f"{path.name}.before-cj-remove.{stamp}"
+ # Never overwrite an earlier backup. The skill removes several annotations
+ # in quick succession, so a second-resolution stamp collides and the later
+ # copy would replace the earlier one with already-mutated content — losing
+ # the pre-session original the backup exists to preserve.
+ dest = base
+ n = 2
+ while dest.exists():
+ dest = base.with_name(f"{base.name}-{n}")
+ n += 1
+ shutil.copy2(path, dest)
+ return dest
+
+
+def _atomic_write(path: Path, text: str) -> None:
+ """Write text to path via a temp sibling and os.replace.
+
+ A bare write_text truncates the target on open, so a mid-write failure left
+ the org file truncated with no complete copy on disk. Writing a temp sibling
+ and renaming means the file is either its old content or its new content,
+ never a partial.
+ """
+ # Follow a symlink to the file it names. os.replace would otherwise swap the
+ # symlink itself for a regular file, leaving the real target holding the old
+ # content — the edit silently goes nowhere. Resolving also puts the temp
+ # sibling on the same filesystem as the real file, which os.replace needs.
+ path = path.resolve()
+ fd, tmp = tempfile.mkstemp(dir=path.parent, prefix=f".{path.name}.", suffix=".tmp")
+ os.close(fd)
+ tmp_path = Path(tmp)
+ # Carry the original's permissions across. mkstemp creates 0600, and
+ # defaulting to the umask instead widened a deliberately-restricted file
+ # (a 0600 org file came back 0644).
+ shutil.copymode(path, tmp_path)
+ try:
+ tmp_path.write_text(text, encoding="utf-8")
+ os.replace(tmp_path, path)
+ except BaseException:
+ tmp_path.unlink(missing_ok=True)
+ raise
+
+
def remove_range(path: Path, start: int, end: int) -> None:
"""Read path, validate range looks like cj content, remove the range, write back."""
- text = path.read_text()
+ text = path.read_text(encoding="utf-8")
had_trailing_newline = text.endswith("\n")
lines = text.splitlines(keepends=False)
@@ -77,7 +152,9 @@ def remove_range(path: Path, start: int, end: int) -> None:
new_text += "\n"
elif not new_lines and had_trailing_newline:
new_text = ""
- path.write_text(new_text)
+
+ _backup(path)
+ _atomic_write(path, new_text)
def main() -> int:
diff --git a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-discover b/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-discover
deleted file mode 100755
index 152cf27..0000000
--- a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-discover
+++ /dev/null
@@ -1,230 +0,0 @@
-#!/usr/bin/env python3
-"""Enumerate cross-agent destinations: local projects + tailnet peers.
-
-See cross-agent-discover.md. Local: scan ~/projects/*/.ai/. Peers: read
-peers.toml, SSH-probe each for reachability. --enumerate-remote optionally
-runs `ls -d ~/projects/*/.ai/` over SSH to list remote projects.
-
-Cache results for 5 min at ~/.cache/cross-agent-comms/discovery.json so
-repeated invocations don't re-probe.
-
-HALT: prints a banner; otherwise continues.
-"""
-
-from __future__ import annotations
-
-import argparse
-import datetime as _dt
-import json
-import os
-import subprocess
-import sys
-import time
-import tomllib
-from pathlib import Path
-
-CONFIG_DIR = Path.home() / ".config" / "cross-agent-comms"
-PEERS_TOML = CONFIG_DIR / "peers.toml"
-HALT_FILE = CONFIG_DIR / "HALT"
-CACHE_DIR = Path.home() / ".cache" / "cross-agent-comms"
-CACHE_FILE = CACHE_DIR / "discovery.json"
-CACHE_TTL_SECONDS = 300
-
-EXIT_OK = 0
-EXIT_GENERAL = 1
-EXIT_PEERS_TOML = 1
-
-
-def err(msg: str) -> None:
- print(msg, file=sys.stderr)
-
-
-def render_banner_if_halt() -> None:
- if not HALT_FILE.exists():
- return
- try:
- reason = HALT_FILE.read_text().strip()
- except OSError:
- reason = "(HALT file unreadable; treated as halted)"
- print("⚠ HALT ACTIVE — cross-agent comms paused")
- if reason:
- print(f" reason: {reason}")
- print()
-
-
-def enumerate_local_projects() -> list[str]:
- projects_dir = Path.home() / "projects"
- if not projects_dir.is_dir():
- return []
- found = []
- for child in sorted(projects_dir.iterdir()):
- if child.is_dir() and (child / ".ai").is_dir():
- found.append(child.name)
- return found
-
-
-def load_peers() -> dict:
- if not PEERS_TOML.exists():
- return {"peers": {}}
- try:
- return tomllib.loads(PEERS_TOML.read_text())
- except (tomllib.TOMLDecodeError, OSError) as e:
- err(f"cannot parse peers.toml: {e}")
- sys.exit(EXIT_PEERS_TOML)
-
-
-def probe_peer_reachability(host: str, ssh_user: str | None) -> tuple[bool, str | None]:
- """Run a short SSH probe with BatchMode=yes (no interactive prompt)."""
- target = f"{ssh_user}@{host}" if ssh_user else host
- try:
- result = subprocess.run(
- ["ssh", "-o", "ConnectTimeout=2", "-o", "BatchMode=yes", target, "true"],
- capture_output=True,
- text=True,
- timeout=5,
- )
- except (FileNotFoundError, subprocess.TimeoutExpired):
- return False, "ssh probe failed"
- if result.returncode == 0:
- return True, None
- return False, (result.stderr.strip().splitlines() or [f"exit {result.returncode}"])[-1]
-
-
-def enumerate_remote_projects(host: str, ssh_user: str | None) -> list[str] | None:
- target = f"{ssh_user}@{host}" if ssh_user else host
- try:
- result = subprocess.run(
- [
- "ssh", "-o", "ConnectTimeout=3", "-o", "BatchMode=yes", target,
- "ls -d ~/projects/*/.ai/ 2>/dev/null",
- ],
- capture_output=True,
- text=True,
- timeout=10,
- )
- except (FileNotFoundError, subprocess.TimeoutExpired):
- return None
- if result.returncode != 0:
- return None
- projects = []
- for line in result.stdout.splitlines():
- # Each line looks like /home/<user>/projects/<name>/.ai/
- parts = line.rstrip("/").split("/")
- if len(parts) >= 2 and parts[-1] == ".ai":
- projects.append(parts[-2])
- return projects
-
-
-def read_cache() -> dict | None:
- if not CACHE_FILE.exists():
- return None
- try:
- age = time.time() - CACHE_FILE.stat().st_mtime
- if age > CACHE_TTL_SECONDS:
- return None
- return json.loads(CACHE_FILE.read_text())
- except (OSError, json.JSONDecodeError):
- return None
-
-
-def write_cache(payload: dict) -> None:
- CACHE_DIR.mkdir(parents=True, exist_ok=True)
- CACHE_FILE.write_text(json.dumps(payload, indent=2))
-
-
-def discover(peer_filter: str | None, enumerate_remote: bool) -> dict:
- local = enumerate_local_projects()
- peers_cfg = load_peers().get("peers", {})
-
- peers_out = []
- for name, cfg in sorted(peers_cfg.items()):
- if peer_filter and name != peer_filter:
- continue
- host = cfg.get("host", name)
- ssh_user = cfg.get("ssh_user")
- reachable, error = probe_peer_reachability(host, ssh_user)
- entry = {
- "name": name,
- "host": host,
- "reachable": reachable,
- }
- if not reachable:
- entry["error"] = error
- if enumerate_remote and reachable:
- entry["projects"] = enumerate_remote_projects(host, ssh_user) or []
- peers_out.append(entry)
-
- return {
- "scanned_at": _dt.datetime.now(_dt.timezone.utc).isoformat(),
- "halt_active": HALT_FILE.exists(),
- "local": local,
- "peers": peers_out,
- }
-
-
-def render_table(payload: dict, enumerate_remote: bool) -> None:
- local = payload.get("local", [])
- print(f"Local ({_local_hostname()}):")
- if local:
- wrapped = ", ".join(local)
- print(f" {wrapped} [{len(local)} project{'s' if len(local) != 1 else ''}]")
- else:
- print(" (no projects with .ai/ found)")
- print()
-
- peers = payload.get("peers", [])
- if not peers:
- print("Peers (from peers.toml):")
- print(" (no peers configured)")
- return
-
- print("Peers (from ~/.config/cross-agent-comms/peers.toml):")
- for p in peers:
- marker = "✓ reachable" if p.get("reachable") else f"✗ UNREACHABLE ({p.get('error', 'unknown')})"
- print(f" {p['name']:<16} {p['host']:<24} {marker}")
- if enumerate_remote and p.get("projects"):
- wrapped = ", ".join(p["projects"])
- print(f" projects: {wrapped}")
-
-
-def _local_hostname() -> str:
- import socket
- return socket.gethostname().split(".")[0]
-
-
-def main() -> int:
- parser = argparse.ArgumentParser(description="Discover cross-agent destinations.")
- parser.add_argument("--enumerate-remote", action="store_true",
- help="SSH into each peer and list ~/projects/*/.ai/")
- parser.add_argument("--no-cache", action="store_true", help="Skip cache; force fresh probe")
- parser.add_argument("--peer", help="Limit to a single peer name from peers.toml")
- parser.add_argument("--json", action="store_true", help="Machine-readable output")
- args = parser.parse_args()
-
- render_banner_if_halt()
-
- payload = None
- if not args.no_cache:
- cached = read_cache()
- if cached is not None:
- # Honor --peer filter on cached payload.
- if args.peer:
- cached["peers"] = [p for p in cached.get("peers", []) if p["name"] == args.peer]
- payload = cached
-
- if payload is None:
- payload = discover(args.peer, args.enumerate_remote)
- if not args.no_cache and not args.peer:
- # Only cache full (unfiltered) discoveries.
- write_cache(payload)
-
- if args.json:
- print(json.dumps(payload, indent=2))
- return EXIT_OK
-
- render_table(payload, args.enumerate_remote)
- return EXIT_OK
-
-
-if __name__ == "__main__":
- sys.exit(main())
diff --git a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-discover.md b/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-discover.md
deleted file mode 100644
index 95134bb..0000000
--- a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-discover.md
+++ /dev/null
@@ -1,155 +0,0 @@
-# cross-agent-discover
-
-**Purpose.** Enumerate available cross-agent destinations — local projects on
-this machine and remote projects on tailnet peers. Validates SSH reachability
-for cross-machine destinations before reporting them as usable.
-
-## Usage
-
-```
-cross-agent-discover [--enumerate-remote] [--no-cache] [--peer <name>]
-```
-
-No args required for the common case (local enumeration + peer reachability).
-
-### Flags
-
-| Flag | Default | Purpose |
-|---|---|---|
-| `--enumerate-remote` | off | SSH into each peer and list projects under `~/projects/*/.ai/`. Off by default because SSH adds latency; turn on when you want to see what's available on a remote machine you haven't fully configured. |
-| `--no-cache` | off | Skip the 5-minute cache; force fresh discovery. |
-| `--peer <name>` | (all) | Limit to a single peer from `peers.toml`. |
-| `--json` | off | Machine-readable output. |
-
-## Output
-
-### Default
-
-```
-$ cross-agent-discover
-Local (ratio):
- career, claude-templates, clipper, danneel, documents, elibrary,
- finances, health, homelab, jr-estate, kit, little-elisper,
- philosophy, website [14 projects]
-
-Peers (from ~/.config/cross-agent-comms/peers.toml):
- velox.local reachable (last seen 2 sec ago)
- bastion.local UNREACHABLE (ssh exit 255: connection refused)
-```
-
-### With `--enumerate-remote`
-
-```
-$ cross-agent-discover --enumerate-remote
-Local (ratio):
- ... (as above)
-
-velox.local (reachable):
- career, homelab [2 projects]
-```
-
-## Configuration
-
-Reads `~/.config/cross-agent-comms/peers.toml`:
-
-```toml
-# Each peer is a remote machine reachable via SSH (typically over Tailscale).
-
-[peers.velox]
-host = "velox.local"
-ssh_user = "cjennings"
-
-[peers.bastion]
-host = "bastion.local"
-ssh_user = "cjennings"
-```
-
-Peers entries describe machines, NOT projects. Projects are enumerated
-on-demand under `~/projects/*/.ai/` either locally or via SSH.
-
-## Cache
-
-Successful discovery results are cached at
-`~/.cache/cross-agent-comms/discovery.json` for 5 minutes. Repeated invocations
-within the window read from cache.
-
-`--no-cache` forces a fresh probe. Useful when adding a new peer or after a
-network change.
-
-## SSH reachability check
-
-For each peer, runs:
-
-```
-ssh -o ConnectTimeout=2 -o BatchMode=yes <user>@<host> true
-```
-
-`BatchMode=yes` prevents interactive password prompts — peers that don't have
-key-based auth set up are reported as UNREACHABLE.
-
-If `--enumerate-remote` is set, on success runs:
-
-```
-ssh <user>@<host> 'ls -d ~/projects/*/.ai/ 2>/dev/null'
-```
-
-## Failure modes
-
-| Symptom | Likely cause | Fix |
-|---|---|---|
-| Peer reported UNREACHABLE | Tailscale not connected, SSH key not authorized, host firewalled | `tailscale status`; `ssh -v <peer>` to debug. |
-| Local list is empty | Glob misresolved, or `~/projects/` doesn't exist | Check `ls -d ~/projects/*/.ai/`. |
-| `--enumerate-remote` slow | Cold cache, slow tailnet, many peers | First run is slow, subsequent runs hit cache. Use `--peer <name>` to scope. |
-| Peer unexpectedly missing from output | Not in `peers.toml`, or `peers.toml` malformed | `cat ~/.config/cross-agent-comms/peers.toml` and validate. |
-
-## HALT awareness
-
-Checks `~/.config/cross-agent-comms/HALT` at start. If HALT exists, prints a
-prominent banner before normal output:
-
-```
-$ cross-agent-discover
-⚠ HALT ACTIVE — cross-agent comms paused
- Reason: <reason from HALT file body, if any>
- Resume with: cross-agent-resume
-
-(enumeration continues normally — HALT does not suppress visibility)
-
-Local (ratio):
- career, claude-templates, ...
-
-Peers:
- velox.local reachable
-```
-
-Discover is read-only. Like `cross-agent-status`, it always runs so the user
-keeps visibility into what destinations exist regardless of halt state. The
-banner makes the halt state impossible to miss.
-
-If the HALT file exists but is unreadable, print a warning banner and
-continue.
-
-See `cross-agent-halt.md` for the full halt mechanism.
-
-## Examples
-
-```bash
-# Common: see what's available
-cross-agent-discover
-
-# Force fresh probe after network change
-cross-agent-discover --no-cache
-
-# What's on velox specifically
-cross-agent-discover --peer velox --enumerate-remote
-
-# Pipe to grep
-cross-agent-discover --json | jq '.peers[] | select(.reachable)'
-```
-
-## See also
-
-- `cross-agent-send` — uses `peers.toml` for routing destinations.
-- `cross-agent-status` — local pending messages.
-- `cross-agent-comms.org` — protocol spec, `* Limitations` section
- explains the cross-machine model.
diff --git a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-halt b/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-halt
deleted file mode 100755
index df25115..0000000
--- a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-halt
+++ /dev/null
@@ -1,134 +0,0 @@
-#!/usr/bin/env python3
-"""Failsafe halt for cross-agent comms.
-
-See cross-agent-halt.md. Touches ~/.config/cross-agent-comms/HALT and stops
-the cross-agent-watch systemd user service. With --tailnet, propagates the
-HALT file to every peer in peers.toml via SSH; reports per-peer status with
-non-zero exit on partial halt.
-
-Does NOT pkill in-flight scripts — they detect HALT on next iteration and
-stop themselves.
-"""
-
-from __future__ import annotations
-
-import argparse
-import subprocess
-import sys
-import tomllib
-from pathlib import Path
-
-CONFIG_DIR = Path.home() / ".config" / "cross-agent-comms"
-HALT_FILE = CONFIG_DIR / "HALT"
-PEERS_TOML = CONFIG_DIR / "peers.toml"
-
-EXIT_OK = 0
-EXIT_PARTIAL = 1
-
-
-def err(msg: str) -> None:
- print(msg, file=sys.stderr)
-
-
-def write_halt_file(reason: str) -> None:
- CONFIG_DIR.mkdir(parents=True, exist_ok=True)
- HALT_FILE.write_text((reason + "\n") if reason else "")
-
-
-def stop_watcher_service() -> None:
- """Best-effort stop of the systemd watcher service. Failures are logged but not fatal."""
- try:
- subprocess.run(
- ["systemctl", "--user", "stop", "cross-agent-watch.path"],
- capture_output=True, text=True, timeout=5,
- )
- except (FileNotFoundError, subprocess.TimeoutExpired):
- # Watcher service may not be installed — fine.
- pass
-
-
-def load_peers() -> dict:
- if not PEERS_TOML.exists():
- return {}
- try:
- return tomllib.loads(PEERS_TOML.read_text())
- except (tomllib.TOMLDecodeError, OSError) as e:
- err(f"cannot parse peers.toml: {e}")
- return {}
-
-
-def ssh_touch_halt(host: str, ssh_user: str | None, reason: str) -> tuple[bool, str]:
- target = f"{ssh_user}@{host}" if ssh_user else host
- # Build the remote command. Quote the reason carefully.
- remote_cmd = (
- f"mkdir -p ~/.config/cross-agent-comms && "
- f"printf %s {_sh_quote(reason)} > ~/.config/cross-agent-comms/HALT"
- )
- try:
- result = subprocess.run(
- ["ssh", "-o", "ConnectTimeout=3", "-o", "BatchMode=yes", target, remote_cmd],
- capture_output=True, text=True, timeout=10,
- )
- except (FileNotFoundError, subprocess.TimeoutExpired):
- return False, "ssh unavailable or timed out"
- if result.returncode == 0:
- return True, "HALT file written"
- return False, (result.stderr.strip().splitlines() or [f"exit {result.returncode}"])[-1]
-
-
-def _sh_quote(s: str) -> str:
- return "'" + s.replace("'", "'\"'\"'") + "'"
-
-
-def main() -> int:
- parser = argparse.ArgumentParser(description="Halt all cross-agent comms on this machine (and optionally tailnet).")
- parser.add_argument("reason", nargs="?", default="", help="Optional human-readable reason")
- parser.add_argument("--tailnet", action="store_true",
- help="Propagate HALT to every peer in peers.toml")
- args = parser.parse_args()
-
- # Local halt.
- write_halt_file(args.reason)
- stop_watcher_service()
- print("Halting locally ✓ (HALT file written)")
-
- if not args.tailnet:
- print()
- print(f"Halt active. Remove {HALT_FILE} or run cross-agent-resume to clear.")
- print("Agent polling will stop within ~5 min (one cadence cycle).")
- return EXIT_OK
-
- peers = load_peers().get("peers", {})
- if not peers:
- print()
- print("No peers configured in peers.toml — local-only halt complete.")
- return EXIT_OK
-
- print()
- successes = 1 # local already counted
- failures = []
- for name, cfg in sorted(peers.items()):
- host = cfg.get("host", name)
- ssh_user = cfg.get("ssh_user")
- ok, detail = ssh_touch_halt(host, ssh_user, args.reason)
- marker = "✓" if ok else "✗"
- print(f"Halting {host:<28} {marker} ({detail})")
- if ok:
- successes += 1
- else:
- failures.append(f"{name} ({host}): {detail}")
-
- print()
- total = len(peers) + 1
- if failures:
- print(f"PARTIAL HALT: {successes}/{total} machines halted.")
- for f in failures:
- print(f" - {f}")
- print("Resolve the failures or manually halt each machine.")
- return EXIT_PARTIAL
- print(f"Halt active across {total} machine(s).")
- return EXIT_OK
-
-
-if __name__ == "__main__":
- sys.exit(main())
diff --git a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-halt.md b/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-halt.md
deleted file mode 100644
index b817fbc..0000000
--- a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-halt.md
+++ /dev/null
@@ -1,134 +0,0 @@
-# cross-agent-halt
-
-**Purpose.** Failsafe stop for all cross-agent activity on the local machine
-(or, with `--tailnet`, across all configured peers). Creates the HALT file
-that every component in the protocol checks; within one polling cadence
-(~5 min) all polling, sending, watching, and receiving stops.
-
-This is the user's emergency brake. Use when something is misbehaving and
-visiting individual sessions is too slow.
-
-## Usage
-
-```
-cross-agent-halt [reason] [--tailnet] [--no-stop-watcher]
-```
-
-### Positional argument
-
-| Position | Meaning | Example |
-|---|---|---|
-| 1 | Optional human-readable reason for the halt. Written into the HALT file's body. Helps future-you remember why you stopped things. | `"investigating runaway poll loop, 2026-04-27"` |
-
-### Flags
-
-| Flag | Default | Purpose |
-|---|---|---|
-| `--tailnet` | local only | Propagate halt to every peer in `peers.toml` via SSH over Tailscale. |
-| `--no-stop-watcher` | (stops watcher) | Skip stopping the `cross-agent-watch.path` systemd unit. Useful if the watcher is intentionally separate from comms (rare). |
-
-## Behavior
-
-### Local halt (default)
-
-1. Write the HALT file: `~/.config/cross-agent-comms/HALT`. If a `[reason]` was
- passed, write it as the file's body. Otherwise the file is empty (existence
- alone triggers halt).
-2. Stop the watcher service: `systemctl --user stop cross-agent-watch.path`
- (and the corresponding `.service` if running).
-3. Print a summary:
- ```
- ✓ HALT file written: ~/.config/cross-agent-comms/HALT
- ✓ Watcher service stopped (cross-agent-watch.path)
- - In-flight sends will complete their current rsync step (~seconds), then
- stop. New sends are blocked.
- - Active agent polling sessions stop within one cadence (~5 min).
- - Use `cross-agent-resume` to clear HALT.
- Per-session polling does NOT auto-resume — you re-engage each session by
- telling its agent to resume polling.
- ```
-4. Exit 0.
-
-### Cross-tailnet halt (`--tailnet`)
-
-1. Apply local halt steps 1-2 first.
-2. Read `peers.toml` for the list of remote machines.
-3. For each peer, SSH and write the HALT file:
- ```
- ssh <user>@<host> "echo '<reason>' > ~/.config/cross-agent-comms/HALT && \
- systemctl --user stop cross-agent-watch.path"
- ```
-4. Track per-peer success/failure. Print results:
- ```
- Halting velox.local ✓ (HALT file written)
- Halting bastion.local ✗ (ssh exit 255: no route to host)
- Halting locally ✓ (HALT file written)
-
- PARTIAL HALT: 2/3 machines halted. bastion.local needs manual halt.
- ```
-5. Exit 0 if all peers halted; exit 1 if any peer failed (so scripts can
- detect partial halt). The local halt always succeeds — even on `--tailnet`,
- if remote peers fail, local is still halted.
-
-## What "halt active" means for each component
-
-| Component | Behavior under HALT |
-|---|---|
-| `cross-agent-send` | Refuses to send. Exits 5 with "halt active; remove ~/.config/cross-agent-comms/HALT to resume." Checks HALT at start AND between each retry/rsync step, so an in-flight send completes its current step then stops. |
-| `cross-agent-recv` | Refuses to verify or dedup. Exits 5 with same message. Inbound files are **left in place** — not moved, not rejected — so resume picks them up cleanly via cold-start. |
-| `cross-agent-watch` | Continues running but suppresses notifications. Logs each event with `(suppressed by HALT)` so the operator can see what would have fired. |
-| `cross-agent-status` | Prints prominent `⚠ HALT ACTIVE` banner before normal output. Continues to enumerate (read-only). |
-| `cross-agent-discover` | Same banner. Continues (read-only). |
-| Agent polling loops | Check HALT on every wake. If set: write a final `progress` note to any active conversation ("HALT fired locally; pausing"), surface "(HALT active; cross-agent comms paused)" in every user response, and stop rescheduling. Polling decays naturally within one cadence. |
-| Conversation initiator | Refuses to write sequence 1 of any new conversation. Surfaces refusal to user. |
-| Startup workflow (Phase A) | Checks HALT at session boot. If set, surfaces immediately and skips cross-agent inbox checks. |
-
-## Failure modes
-
-| Symptom | Cause | Fix |
-|---|---|---|
-| `~/.config/cross-agent-comms/HALT` already exists | Halt was already active | OK — running halt again refreshes the reason text. Safe. |
-| `systemctl --user stop` fails | Watcher service not installed, or systemd not available | The HALT file is still written — components that check HALT will still stop. The systemctl failure surfaces as a non-fatal warning. |
-| `--tailnet` halts some peers but not others | One or more peers unreachable | Exit 1 with per-peer status. Manually halt the unreachable peers (visit each machine, `touch ~/.config/cross-agent-comms/HALT`), or fix the network and re-run. |
-| Permission denied writing the HALT file | `~/.config/cross-agent-comms/` doesn't exist or is owned by another user | `mkdir -p ~/.config/cross-agent-comms/`; check ownership. |
-
-## What halt does NOT do
-
-- Does not kill running Claude sessions. Polling stops within ~5 min, but the
- session itself stays alive and can be re-engaged after resume.
-- Does not delete pending messages. Inbound files in `inbox/from-agents/`
- remain; they get processed when polling resumes.
-- Does not abort in-flight rsync push mid-byte. Atomic-write semantics
- guarantee in-flight messages either complete cleanly or leave only `.tmp.*`
- files (which receivers ignore).
-
-## Examples
-
-```bash
-# Quick halt with no reason
-cross-agent-halt
-
-# Halt with a memo
-cross-agent-halt "runaway poll loop in homelab session, debugging"
-
-# Halt all tailnet peers + local
-cross-agent-halt --tailnet "shutting down for system update"
-
-# Halt protocol comms but leave the watcher service running
-cross-agent-halt --no-stop-watcher
-```
-
-## Recovery
-
-Always pair with `cross-agent-resume` when the situation is resolved:
-
-```bash
-cross-agent-resume # local
-cross-agent-resume --tailnet # all peers
-```
-
-## See also
-
-- `cross-agent-resume` — counterpart that clears HALT.
-- `cross-agent-status` — see HALT state at a glance.
-- `cross-agent-comms.org` — protocol spec, `* Halt mechanism` section.
diff --git a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-recv b/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-recv
deleted file mode 100755
index b67533a..0000000
--- a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-recv
+++ /dev/null
@@ -1,250 +0,0 @@
-#!/usr/bin/env python3
-"""Cross-agent message receiver.
-
-See cross-agent-recv.md for the full contract. Reads one message file and
-emits a structured decision the agent acts on:
-
- process | dedup | query | reject
-
-Decision exit codes:
- 0 = process 1 = dedup 2 = query 3 = reject
-
-When HALT is set, the script refuses to verify or dedup and leaves the
-inbound file in place — resume picks it up via cold-start.
-"""
-
-from __future__ import annotations
-
-import argparse
-import hashlib
-import json
-import re
-import shutil
-import subprocess
-import sys
-from pathlib import Path
-
-CONFIG_DIR = Path.home() / ".config" / "cross-agent-comms"
-HALT_FILE = CONFIG_DIR / "HALT"
-EXPECTED_PROTOCOL_VERSION = "5"
-
-REQUIRED_FRONTMATTER = ["TITLE", "CONVERSATION_ID", "MESSAGE_TYPE", "SEQUENCE", "TIMESTAMP", "PROTOCOL_VERSION"]
-VALID_MESSAGE_TYPES = {"request", "progress", "query", "pushback", "complete", "release", "escalate"}
-
-DEC_PROCESS = "process"
-DEC_DEDUP = "dedup"
-DEC_QUERY = "query"
-DEC_REJECT = "reject"
-
-EXIT_FOR_DECISION = {
- DEC_PROCESS: 0,
- DEC_DEDUP: 1,
- DEC_QUERY: 2,
- DEC_REJECT: 3,
-}
-
-EXIT_HALT = 5
-
-
-def err(msg: str) -> None:
- print(msg, file=sys.stderr)
-
-
-def check_halt() -> None:
- if HALT_FILE.exists():
- try:
- reason = HALT_FILE.read_text().strip()
- except OSError:
- err("halt active (HALT file present but unreadable; treated as halted)")
- sys.exit(EXIT_HALT)
- msg = "halt active; leaving inbound message in place (resume will pick up)"
- if reason:
- msg = f"{msg}: {reason}"
- err(msg)
- sys.exit(EXIT_HALT)
-
-
-def parse_frontmatter(path: Path) -> dict[str, str]:
- try:
- text = path.read_text()
- except OSError as e:
- return {"_parse_error": f"cannot read: {e}"}
- fm: dict[str, str] = {}
- for line in text.splitlines():
- line = line.rstrip()
- if not line:
- if fm:
- break
- continue
- m = re.match(r"#\+([A-Z_]+):\s*(.*)", line)
- if m:
- fm[m.group(1)] = m.group(2).strip()
- elif fm:
- break
- return fm
-
-
-def emit_decision(
- decision: str,
- reason: str | None,
- fm: dict[str, str],
- sha256: str | None,
- args: argparse.Namespace,
-) -> int:
- payload = {
- "decision": decision,
- "reason": reason,
- "message_type": fm.get("MESSAGE_TYPE"),
- "conversation_id": fm.get("CONVERSATION_ID"),
- "sequence": fm.get("SEQUENCE"),
- "timestamp": fm.get("TIMESTAMP"),
- "sha256": sha256,
- }
- if args.json:
- print(json.dumps(payload, indent=None if args.compact_json else 2))
- else:
- print(f"decision: {decision}")
- if reason:
- print(f"reason: {reason}")
- for k in ("message_type", "conversation_id", "sequence", "timestamp"):
- v = payload[k]
- if v is not None:
- print(f"{k}: {v}")
- if sha256:
- print(f"sha256: {sha256}")
- return EXIT_FOR_DECISION[decision]
-
-
-def gpg_verify(message_path: Path, sig_path: Path) -> tuple[bool, str]:
- try:
- result = subprocess.run(
- ["gpg", "--verify", str(sig_path), str(message_path)],
- capture_output=True,
- text=True,
- )
- except FileNotFoundError:
- return False, "gpg not installed"
- if result.returncode == 0:
- return True, ""
- return False, result.stderr.strip().splitlines()[-1] if result.stderr.strip() else f"exit {result.returncode}"
-
-
-def sha256_of(path: Path) -> str:
- h = hashlib.sha256()
- with path.open("rb") as f:
- for chunk in iter(lambda: f.read(65536), b""):
- h.update(chunk)
- return h.hexdigest()
-
-
-def find_dedup_match(message_path: Path, fm: dict[str, str], my_hash: str) -> tuple[str, str | None]:
- """Scan the message's directory for same-CONVERSATION_ID/SEQUENCE files.
-
- Returns (decision, reason) — decision is DEC_DEDUP for an exact-hash match,
- or DEC_PROCESS when no match or hash differs (sequence collision is OK).
- """
- parent = message_path.parent
- conv_id = fm["CONVERSATION_ID"]
- sequence = fm["SEQUENCE"]
- for sibling in parent.iterdir():
- if sibling == message_path or not sibling.is_file() or sibling.suffix != ".org":
- continue
- sib_fm = parse_frontmatter(sibling)
- if sib_fm.get("CONVERSATION_ID") != conv_id or sib_fm.get("SEQUENCE") != sequence:
- continue
- # Same conv-id + same sequence — check hash.
- if sha256_of(sibling) == my_hash:
- return DEC_DEDUP, f"identical retry of {sibling.name}"
- return DEC_PROCESS, None
-
-
-def check_requires_tools(fm: dict[str, str]) -> tuple[bool, list[str]]:
- """REQUIRES_TOOLS is a comma-separated list of tool names.
-
- For v5, "tool available" is a heuristic: an executable on PATH whose name
- matches the tool slug. MCP availability is currently out of scope (no
- portable way to query it from a CLI).
- """
- tools_field = fm.get("REQUIRES_TOOLS")
- if not tools_field:
- return True, []
- tools = [t.strip() for t in tools_field.split(",") if t.strip()]
- missing = [t for t in tools if shutil.which(t) is None]
- return len(missing) == 0, missing
-
-
-def main() -> int:
- parser = argparse.ArgumentParser(description="Receive and decide on a cross-agent message.")
- parser.add_argument("message_file", type=Path)
- parser.add_argument("--no-verify", action="store_true", help="Skip GPG verification (testing only)")
- parser.add_argument("--no-dedup", action="store_true", help="Skip SHA-256 dedup against existing files")
- parser.add_argument("--protocol-version", default=EXPECTED_PROTOCOL_VERSION,
- help="Override expected protocol version (default: 5)")
- parser.add_argument("--json", action="store_true", help="Emit JSON output")
- parser.add_argument("--compact-json", action="store_true", help="Compact JSON (no indent)")
- args = parser.parse_args()
-
- check_halt()
-
- if not args.message_file.is_file():
- err(f"message file not found: {args.message_file}")
- return EXIT_FOR_DECISION[DEC_REJECT]
-
- fm = parse_frontmatter(args.message_file)
- if "_parse_error" in fm:
- return emit_decision(DEC_REJECT, fm["_parse_error"], {}, None, args)
-
- # Step 1: frontmatter sanity-check.
- missing = [k for k in REQUIRED_FRONTMATTER if k not in fm]
- if missing:
- return emit_decision(
- DEC_REJECT, f"frontmatter missing required fields: {', '.join(missing)}", fm, None, args
- )
- if fm["MESSAGE_TYPE"] not in VALID_MESSAGE_TYPES:
- return emit_decision(
- DEC_REJECT, f"invalid MESSAGE_TYPE: {fm['MESSAGE_TYPE']!r}", fm, None, args
- )
-
- # Step 2: PROTOCOL_VERSION check.
- if fm["PROTOCOL_VERSION"] != args.protocol_version:
- return emit_decision(
- DEC_QUERY,
- f"PROTOCOL_VERSION mismatch: expected {args.protocol_version}, got {fm['PROTOCOL_VERSION']}",
- fm,
- None,
- args,
- )
-
- # Step 3: GPG verify.
- if not args.no_verify:
- sig_path = args.message_file.with_suffix(args.message_file.suffix + ".asc")
- if not sig_path.is_file():
- return emit_decision(DEC_REJECT, f"signature file missing: {sig_path.name}", fm, None, args)
- ok, gpg_err = gpg_verify(args.message_file, sig_path)
- if not ok:
- return emit_decision(DEC_REJECT, f"gpg verify failed: {gpg_err}", fm, None, args)
-
- # Step 4: SHA-256 dedup.
- my_hash = sha256_of(args.message_file)
- if not args.no_dedup:
- decision, reason = find_dedup_match(args.message_file, fm, my_hash)
- if decision == DEC_DEDUP:
- return emit_decision(DEC_DEDUP, reason, fm, my_hash, args)
-
- # Step 5: REQUIRES_TOOLS check.
- ok, missing_tools = check_requires_tools(fm)
- if not ok:
- return emit_decision(
- DEC_QUERY,
- f"required tools unavailable: {', '.join(missing_tools)}",
- fm,
- my_hash,
- args,
- )
-
- # Step 6: process.
- return emit_decision(DEC_PROCESS, None, fm, my_hash, args)
-
-
-if __name__ == "__main__":
- sys.exit(main())
diff --git a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-recv.md b/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-recv.md
deleted file mode 100644
index 247a27a..0000000
--- a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-recv.md
+++ /dev/null
@@ -1,218 +0,0 @@
-# cross-agent-recv
-
-**Purpose.** The canonical receiver-side processor. Reads a single incoming
-message file and reports a structured decision the agent acts on:
-process / dedup / query / reject.
-
-The script handles only mechanical checks (frontmatter, signature, dedup,
-version, tools). Substance-level decisions like `pushback` ("I disagree with
-this request") happen one layer up — after the agent reads the message body
-the script returns as `process`-able.
-
-This is the read-side counterpart to `cross-agent-send`. Together they are the
-two halves of the per-message contract. The agent's polling loop calls
-`cross-agent-recv` on every new file in `inbox/from-agents/` and dispatches on
-the decision.
-
-Without this script, every receiver implementation re-invents GPG verify +
-frontmatter sanity-check + SHA-256 dedup. With it, behavior is consistent
-across projects.
-
-## Usage
-
-```
-cross-agent-recv <message-file>
-```
-
-Single positional argument: a `.org` file in `inbox/from-agents/`. The matching
-`.asc` signature file must be present alongside it.
-
-### Flags
-
-| Flag | Default | Purpose |
-|---|---|---|
-| `--no-verify` | (verify on) | Skip GPG verification. Testing only. |
-| `--no-dedup` | (dedup on) | Skip SHA-256 dedup against existing files. Testing only. |
-| `--protocol-version <N>` | 5 | Override the expected protocol version. Useful for testing forward-compatibility checks. |
-| `--json` | off | Output decision as JSON for easier parsing by the agent. |
-
-## Behavior
-
-Runs the receiver checks in order. First failure determines the decision.
-
-### Step 1 — Frontmatter sanity-check
-
-Parse the message's org-mode frontmatter. Required fields:
-
-- `#+TITLE`
-- `#+CONVERSATION_ID`
-- `#+MESSAGE_TYPE` (must be one of: `request`, `progress`, `query`, `pushback`,
- `complete`, `release`, `escalate`)
-- `#+SEQUENCE` (integer)
-- `#+TIMESTAMP` (ISO 8601 with explicit offset)
-- `#+PROTOCOL_VERSION` (must match the expected version; default 5)
-
-Any required field missing, malformed, or the protocol version mismatched →
-decision = `reject` (frontmatter) or `query` (version mismatch — see below).
-
-### Step 2 — Protocol-version check
-
-If `PROTOCOL_VERSION` doesn't match the expected:
-
-- Decision = `query`. Action: receiver should write a `query` reply asking the
- sender to upgrade to the expected protocol version.
-
-### Step 3 — Signature verification
-
-Look for `<message-file>.asc` alongside the `.org`. If missing or `gpg
---verify` fails:
-
-- Decision = `reject` (signature). Surface to user; do not act.
-
-The `.asc` file MUST be present when the `.org` is — `cross-agent-send`
-guarantees this with its strict ordering (`.asc` lands first). If the `.asc`
-is missing despite the `.org` being present, the sender violated atomic-write
-ordering or the file was tampered with in transit.
-
-### Step 4 — SHA-256 dedup
-
-Compute SHA-256 of the message file. Scan the same directory for existing
-files matching `CONVERSATION_ID + SEQUENCE`:
-
-- No match → decision = `process` (new message, dispatch by type).
-- Match with **identical** SHA-256 → decision = `dedup` (silent retry; do not
- reprocess).
-- Match with **different** SHA-256 → decision = `process` (sequence collision
- with non-identical content; both are legitimate, ordered by `#+TIMESTAMP`).
-
-### Step 5 — REQUIRES_TOOLS optional check
-
-If the message has a `#+REQUIRES_TOOLS` field, verify each named tool/MCP is
-available in the receiver's environment.
-
-- All available → `process`.
-- One or more missing → decision = `query`. The agent should write a `query`
- reply naming the missing tools, asking the sender to reframe the request to
- avoid them.
-
-### Step 6 — Dispatch decision
-
-If all checks pass, decision = `process` with the parsed `MESSAGE_TYPE` so the
-agent's main loop knows which handler to invoke.
-
-## Output
-
-### Default (human-readable)
-
-```
-$ cross-agent-recv inbox/from-agents/20260427T091015Z-from-homelab-prep-fixup.org
-decision: process
-message_type: request
-conversation_id: prep-fixup
-sequence: 6
-sha256: a1b2c3d4...
-```
-
-### `--json`
-
-```json
-{
- "decision": "process",
- "reason": null,
- "message_type": "request",
- "conversation_id": "prep-fixup",
- "sequence": 6,
- "timestamp": "2026-04-27T04:11:42-05:00",
- "sha256": "a1b2c3d4..."
-}
-```
-
-For decisions other than `process`, `reason` carries a human-readable
-explanation:
-
-```json
-{
- "decision": "query",
- "reason": "PROTOCOL_VERSION mismatch: expected 5, got 4",
- "conversation_id": "prep-fixup",
- "sequence": 6
-}
-```
-
-## Decision exit codes
-
-| Decision | Exit code | Agent action |
-|---|---|---|
-| `process` | 0 | Dispatch to the message-type handler |
-| `dedup` | 1 | Silent — do nothing further |
-| `query` | 2 | Write a `query` reply (see `reason` for what to ask) |
-| `reject` | 3 | Surface to user; do not auto-reply |
-
-The agent reads stdout/JSON to learn the decision; it can also key off exit
-code for simpler bash-style dispatching.
-
-## Failure modes
-
-| Symptom | Cause | Fix |
-|---|---|---|
-| `decision: reject (frontmatter)` | Required field missing or malformed | Open the message; fix or surface to user. The sender should not have produced this file. |
-| `decision: reject (signature)` | `.asc` missing, GPG verify failed, or signer unknown | Check that `.asc` exists alongside `.org`. If yes, run `gpg --verify <msg>.asc <msg>` manually for diagnostic output. |
-| `decision: query (PROTOCOL_VERSION)` | Sender on older/newer protocol | Reply with a `query` asking sender to upgrade. Both sides should align before continuing. |
-| `decision: query (REQUIRES_TOOLS)` | Receiver lacks one of the named tools | Reply with a `query` naming the missing tools; sender should reframe to avoid. |
-| `decision: dedup` | Already-processed identical retry | No action. The script handled it correctly. |
-
-## HALT awareness
-
-Checks `~/.config/cross-agent-comms/HALT` at the start of every invocation. If
-HALT exists, exits with code 5 ("halt active; remove
-~/.config/cross-agent-comms/HALT to resume") without verifying, deduping, or
-returning a decision.
-
-**The inbound file is left in place** — not moved, not rejected, not
-deduped. When HALT clears and polling resumes, the file gets picked up via
-the normal cold-start handling (whichever surfaces first: watcher
-notification, startup workflow check, or the next agent poll). Reversibility
-is preserved.
-
-If the HALT file exists but is unreadable, fail-closed — treat as if HALT is
-set.
-
-See `cross-agent-halt.md` for the full halt mechanism.
-
-## Examples
-
-```bash
-# Basic invocation in an agent's polling loop
-for msg in inbox/from-agents/*.org; do
- decision=$(cross-agent-recv --json "$msg")
- case "$(echo "$decision" | jq -r '.decision')" in
- process) handle_message "$msg" ;;
- dedup) ;; # silent
- query) write_query_reply "$msg" "$decision" ;;
- reject) surface_to_user "$msg" "$decision" ;;
- esac
-done
-
-# Test signature verification only
-cross-agent-recv --no-dedup inbox/from-agents/test-msg.org
-
-# Test against a future protocol version
-cross-agent-recv --protocol-version 6 inbox/from-agents/future-msg.org
-```
-
-## Performance
-
-The script is fast (single SHA-256 compute, single GPG verify, frontmatter
-parse). For typical messages (single-digit KB), runs in well under 100ms.
-Dedup-scan is O(N) over files in the directory; if a project's
-`inbox/from-agents/` accumulates hundreds of files, archive released
-conversations to keep the scan fast.
-
-## See also
-
-- `cross-agent-send` — counterpart writer.
-- `cross-agent-watch` — fires when a new message arrives; agent then calls
- `cross-agent-recv` to process it.
-- `cross-agent-status` — pending-message snapshot (uses similar
- released-vs-unreleased logic, but doesn't process individual messages).
-- `cross-agent-comms.org` — protocol spec, the "what" the script implements.
diff --git a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-resume b/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-resume
deleted file mode 100755
index 1fb83bc..0000000
--- a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-resume
+++ /dev/null
@@ -1,145 +0,0 @@
-#!/usr/bin/env python3
-"""Resume cross-agent comms after a halt.
-
-See cross-agent-resume.md. Removes ~/.config/cross-agent-comms/HALT and
-restarts the cross-agent-watch systemd user service. With --tailnet,
-propagates the removal to every peer in peers.toml via SSH; reports
-per-peer status with non-zero exit on partial resume.
-
-Per the asymmetry rule: clearing HALT does NOT auto-resume agent polling.
-Each session must explicitly re-engage.
-"""
-
-from __future__ import annotations
-
-import argparse
-import subprocess
-import sys
-import tomllib
-from pathlib import Path
-
-CONFIG_DIR = Path.home() / ".config" / "cross-agent-comms"
-HALT_FILE = CONFIG_DIR / "HALT"
-PEERS_TOML = CONFIG_DIR / "peers.toml"
-
-EXIT_OK = 0
-EXIT_PARTIAL = 1
-
-
-def err(msg: str) -> None:
- print(msg, file=sys.stderr)
-
-
-def remove_halt_file() -> bool:
- """Returns True if HALT was removed, False if it didn't exist."""
- if HALT_FILE.exists():
- try:
- HALT_FILE.unlink()
- return True
- except OSError as e:
- err(f"could not remove HALT: {e}")
- return False
- return False
-
-
-def start_watcher_service() -> None:
- """Best-effort start of the systemd watcher path unit."""
- try:
- subprocess.run(
- ["systemctl", "--user", "start", "cross-agent-watch.path"],
- capture_output=True, text=True, timeout=5,
- )
- except (FileNotFoundError, subprocess.TimeoutExpired):
- pass
-
-
-def load_peers() -> dict:
- if not PEERS_TOML.exists():
- return {}
- try:
- return tomllib.loads(PEERS_TOML.read_text())
- except (tomllib.TOMLDecodeError, OSError) as e:
- err(f"cannot parse peers.toml: {e}")
- return {}
-
-
-def ssh_remove_halt(host: str, ssh_user: str | None) -> tuple[bool, str]:
- target = f"{ssh_user}@{host}" if ssh_user else host
- remote_cmd = "rm -f ~/.config/cross-agent-comms/HALT"
- try:
- result = subprocess.run(
- ["ssh", "-o", "ConnectTimeout=3", "-o", "BatchMode=yes", target, remote_cmd],
- capture_output=True, text=True, timeout=10,
- )
- except (FileNotFoundError, subprocess.TimeoutExpired):
- return False, "ssh unavailable or timed out"
- if result.returncode == 0:
- return True, "HALT cleared"
- return False, (result.stderr.strip().splitlines() or [f"exit {result.returncode}"])[-1]
-
-
-def print_re_engage_instructions() -> None:
- print()
- print("Halt cleared. Watcher restarted.")
- print()
- print("Agent polling does NOT auto-resume — per the failsafe asymmetry rule,")
- print("agents stay paused until you explicitly re-engage each session.")
- print("Open the relevant Claude session and tell the agent to resume polling")
- print("for its conversation.")
-
-
-def main() -> int:
- parser = argparse.ArgumentParser(description="Resume cross-agent comms after a halt.")
- parser.add_argument("--tailnet", action="store_true",
- help="Propagate HALT removal to every peer in peers.toml")
- args = parser.parse_args()
-
- removed = remove_halt_file()
- start_watcher_service()
- if removed:
- print("Resuming locally ✓ (HALT cleared)")
- else:
- print("Resuming locally ✓ (no HALT was active)")
-
- if not args.tailnet:
- print_re_engage_instructions()
- return EXIT_OK
-
- peers = load_peers().get("peers", {})
- if not peers:
- print()
- print("No peers configured in peers.toml — local-only resume complete.")
- print_re_engage_instructions()
- return EXIT_OK
-
- print()
- successes = 1
- failures = []
- for name, cfg in sorted(peers.items()):
- host = cfg.get("host", name)
- ssh_user = cfg.get("ssh_user")
- ok, detail = ssh_remove_halt(host, ssh_user)
- marker = "✓" if ok else "✗"
- print(f"Resuming {host:<27} {marker} ({detail})")
- if ok:
- successes += 1
- else:
- failures.append(f"{name} ({host}): {detail}")
-
- print()
- total = len(peers) + 1
- if failures:
- print(f"PARTIAL RESUME: {successes}/{total} machines cleared.")
- for f in failures:
- print(f" - {f}")
- print("Resolve the failures or manually clear HALT on each machine.")
- print_re_engage_instructions()
- return EXIT_PARTIAL
-
- print(f"Resume complete across {total} machine(s).")
- print_re_engage_instructions()
- return EXIT_OK
-
-
-if __name__ == "__main__":
- sys.exit(main())
diff --git a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-resume.md b/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-resume.md
deleted file mode 100644
index 8aa8357..0000000
--- a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-resume.md
+++ /dev/null
@@ -1,117 +0,0 @@
-# cross-agent-resume
-
-**Purpose.** Clear the HALT file and restart the watcher service. Counterpart
-to `cross-agent-halt`. Resuming agent polling is **explicit per-session** —
-this script doesn't auto-revive halted polling loops; you tell each session
-to re-engage.
-
-## Usage
-
-```
-cross-agent-resume [--tailnet]
-```
-
-### Flags
-
-| Flag | Default | Purpose |
-|---|---|---|
-| `--tailnet` | local only | Clear HALT on every peer in `peers.toml` via SSH over Tailscale. |
-
-## Behavior
-
-### Local resume (default)
-
-1. Remove the HALT file: `rm -f ~/.config/cross-agent-comms/HALT`. (Use `-f`
- so a missing file isn't an error — running resume when not halted is safe.)
-2. Restart the watcher service: `systemctl --user start cross-agent-watch.path`.
-3. Print a summary:
- ```
- ✓ HALT file removed
- ✓ Watcher service started (cross-agent-watch.path)
- - cross-agent-send and cross-agent-recv will accept new operations.
- - Inbound messages held during halt will be picked up by the watcher.
- - Agent polling does NOT auto-resume. To re-engage polling in a paused
- session, open that Claude session and tell the agent to resume.
- ```
-4. Exit 0.
-
-### Cross-tailnet resume (`--tailnet`)
-
-1. Apply local resume steps 1-2 first.
-2. Read `peers.toml` for the list of remote machines.
-3. For each peer, SSH:
- ```
- ssh <user>@<host> "rm -f ~/.config/cross-agent-comms/HALT && \
- systemctl --user start cross-agent-watch.path"
- ```
-4. Track per-peer success/failure:
- ```
- Resuming velox.local ✓ (HALT cleared, watcher started)
- Resuming bastion.local ✗ (ssh exit 255: no route to host)
- Resuming locally ✓
-
- PARTIAL RESUME: 2/3 machines resumed. bastion.local still halted.
- ```
-5. Exit 0 if all peers resumed; exit 1 on any failure.
-
-## Why agent polling doesn't auto-resume
-
-Two reasons the asymmetry is deliberate:
-
-1. *Auto-resume could silently invert intentional kills.* If you halted
- because a session was misbehaving, removing HALT shouldn't quietly revive
- that session's polling. You re-engage explicitly so you're aware of which
- sessions came back online.
-
-2. *You may want to inspect before resuming.* After a halt, you might want to
- read pending messages, fix configuration, or kill a particular Claude
- session entirely. Per-session resume forces that pause.
-
-## Re-engaging polling in a Claude session
-
-After `cross-agent-resume`, open the relevant Claude session and say something
-like:
-
-```
-HALT is cleared; resume polling.
-```
-
-The agent will check the HALT file (now absent), re-create its polling
-schedule, and continue the in-flight conversation from wherever it left off.
-The conversation file is intact; the receiver will pick up any new messages
-that arrived during the halt window.
-
-## Failure modes
-
-| Symptom | Cause | Fix |
-|---|---|---|
-| HALT file doesn't exist | Already resumed (or never halted) | OK — `-f` makes this a no-op. |
-| `systemctl --user start` fails | Watcher service not installed | Install per `cross-agent-watch.md`'s systemd recipe. |
-| `--tailnet` resumes some peers but not others | Same as halt: peer unreachable | Per-peer status reported; resolve manually for unreachable peers. |
-| Permission denied removing HALT file | File owned by another user | Check ownership; HALT files should be owned by the running user. |
-
-## Examples
-
-```bash
-# Local resume after a halt
-cross-agent-resume
-
-# Resume all tailnet peers + local
-cross-agent-resume --tailnet
-```
-
-## Recovery flow
-
-After a halt:
-
-1. Investigate whatever caused the halt (runaway loop, bad config, etc.).
-2. Fix the underlying issue.
-3. Run `cross-agent-resume`.
-4. Open each Claude session that was polling and tell its agent to re-engage.
-5. Confirm operation with `cross-agent-status`.
-
-## See also
-
-- `cross-agent-halt` — counterpart that creates the HALT file.
-- `cross-agent-status` — verify HALT cleared and see pending messages.
-- `cross-agent-comms.org` — protocol spec, `* Halt mechanism` section.
diff --git a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-send b/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-send
deleted file mode 100755
index 68c010a..0000000
--- a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-send
+++ /dev/null
@@ -1,356 +0,0 @@
-#!/usr/bin/env python3
-"""Cross-agent message sender.
-
-See cross-agent-send.md for the full contract. Briefly:
-
-- Destination as <machine>.<project>; resolved via peers.toml.
-- Same-machine: cp to receiver's inbox/from-agents/ with atomic rename.
-- Cross-machine: rsync over SSH (typically Tailscale) with retry+backoff.
-- GPG-signs by default; .asc renames before .org so receivers never see
- a .org without its sibling signature.
-- Generates the canonical filename; user's input filename is ignored.
-- Honors the HALT file: refuses to send and exits with code 5 when set.
-"""
-
-from __future__ import annotations
-
-import argparse
-import datetime as _dt
-import json
-import os
-import re
-import shutil
-import socket
-import subprocess
-import sys
-import tempfile
-import time
-import tomllib
-from pathlib import Path
-
-CONFIG_DIR = Path.home() / ".config" / "cross-agent-comms"
-PEERS_TOML = CONFIG_DIR / "peers.toml"
-HALT_FILE = CONFIG_DIR / "HALT"
-STATE_DIR = Path.home() / ".local" / "state" / "cross-agent-comms"
-FAILED_SENDS_DIR = STATE_DIR / "failed-sends"
-
-EXIT_OK = 0
-EXIT_GENERAL = 1
-EXIT_DEST_NOT_FOUND = 2
-EXIT_CROSS_MACHINE_FAILED = 3
-EXIT_FRONTMATTER = 4
-EXIT_HALT = 5
-
-REQUIRED_FRONTMATTER = ["CONVERSATION_ID", "MESSAGE_TYPE", "SEQUENCE", "TIMESTAMP", "PROTOCOL_VERSION"]
-VALID_MESSAGE_TYPES = {"request", "progress", "query", "pushback", "complete", "release", "escalate"}
-
-
-def err(msg: str) -> None:
- print(msg, file=sys.stderr)
-
-
-def check_halt() -> None:
- """Exit with code 5 if HALT file exists."""
- if HALT_FILE.exists():
- try:
- reason = HALT_FILE.read_text().strip()
- except OSError:
- # Fail-closed on unreadable HALT.
- err("halt active (HALT file present but unreadable; treated as halted)")
- err(f"remove {HALT_FILE} to resume")
- sys.exit(EXIT_HALT)
- msg = "halt active"
- if reason:
- msg += f": {reason}"
- err(msg)
- err(f"remove {HALT_FILE} to resume")
- sys.exit(EXIT_HALT)
-
-
-def parse_frontmatter(path: Path) -> dict[str, str]:
- """Extract org-mode #+KEY: value frontmatter from the top of the file."""
- try:
- text = path.read_text()
- except OSError as e:
- err(f"cannot read message file: {e}")
- sys.exit(EXIT_GENERAL)
-
- frontmatter: dict[str, str] = {}
- for line in text.splitlines():
- line = line.rstrip()
- if not line:
- # Blank line ends the frontmatter block.
- if frontmatter:
- break
- continue
- m = re.match(r"#\+([A-Z_]+):\s*(.*)", line)
- if m:
- frontmatter[m.group(1)] = m.group(2).strip()
- else:
- # First non-frontmatter line ends parsing.
- if frontmatter:
- break
- return frontmatter
-
-
-def validate_frontmatter(fm: dict[str, str]) -> None:
- missing = [k for k in REQUIRED_FRONTMATTER if k not in fm]
- if missing:
- err(f"frontmatter missing required fields: {', '.join(missing)}")
- sys.exit(EXIT_FRONTMATTER)
- if fm["MESSAGE_TYPE"] not in VALID_MESSAGE_TYPES:
- err(f"invalid MESSAGE_TYPE: {fm['MESSAGE_TYPE']!r}; expected one of {sorted(VALID_MESSAGE_TYPES)}")
- sys.exit(EXIT_FRONTMATTER)
- try:
- int(fm["SEQUENCE"])
- except ValueError:
- err(f"SEQUENCE must be an integer; got {fm['SEQUENCE']!r}")
- sys.exit(EXIT_FRONTMATTER)
-
-
-def load_peers() -> dict:
- if not PEERS_TOML.exists():
- return {}
- try:
- return tomllib.loads(PEERS_TOML.read_text())
- except (tomllib.TOMLDecodeError, OSError) as e:
- err(f"cannot read {PEERS_TOML}: {e}")
- sys.exit(EXIT_GENERAL)
-
-
-def resolve_destination(dest: str, peers: dict) -> tuple[str, str, str | None, str | None]:
- """Resolve <machine>.<project> to (machine, project, host, ssh_user).
-
- host is None for same-machine destinations.
- """
- if "." not in dest:
- err(f"destination must be <machine>.<project>; got {dest!r}")
- sys.exit(EXIT_DEST_NOT_FOUND)
- machine, project = dest.split(".", 1)
-
- local_hostname = socket.gethostname().split(".")[0]
- is_local = machine == local_hostname or machine == "local"
-
- host = None
- ssh_user = None
- if not is_local:
- peer_cfg = peers.get("peers", {}).get(machine)
- if peer_cfg is None:
- available = list(peers.get("peers", {}).keys())
- err(f"destination not found in peers.toml; available peers: {available or '(none)'}")
- sys.exit(EXIT_DEST_NOT_FOUND)
- host = peer_cfg.get("host", machine)
- ssh_user = peer_cfg.get("ssh_user", os.environ.get("USER"))
-
- return machine, project, host, ssh_user
-
-
-def resolve_inbox_path(project: str, peers: dict) -> str:
- """Inbox path on the receiver. Defaults to ~/projects/<project>/inbox/from-agents."""
- proj_cfg = peers.get("projects", {}).get(project)
- if proj_cfg and "inbox_path" in proj_cfg:
- return os.path.expanduser(proj_cfg["inbox_path"])
- return f"~/projects/{project}/inbox/from-agents"
-
-
-def derive_sender_project() -> str:
- """Walk up from CWD looking for ~/projects/<name>/.
-
- Returns the project name if found; falls back to the basename of CWD.
- """
- cwd = Path.cwd().resolve()
- projects_root = (Path.home() / "projects").resolve()
- try:
- rel = cwd.relative_to(projects_root)
- return rel.parts[0]
- except ValueError:
- return cwd.name
-
-
-def generate_canonical_filename(sender: str, conv_id: str) -> str:
- """YYYYMMDDTHHMMSSZ-from-<sender>-<conv-id>.org"""
- now = _dt.datetime.now(_dt.timezone.utc)
- timestamp = now.strftime("%Y%m%dT%H%M%SZ")
- return f"{timestamp}-from-{sender}-{conv_id}.org"
-
-
-def sign(message_path: Path, sig_path: Path, key: str | None) -> None:
- """gpg --detach-sign --armor --output <sig> [--local-user <key>] <message>"""
- cmd = ["gpg", "--detach-sign", "--armor", "--yes", "--output", str(sig_path)]
- if key:
- cmd.extend(["--local-user", key])
- cmd.append(str(message_path))
- try:
- result = subprocess.run(cmd, capture_output=True, text=True)
- except FileNotFoundError:
- err("gpg not found; install gnupg or use --no-sign for testing")
- sys.exit(EXIT_GENERAL)
- if result.returncode != 0:
- err(f"signing failed: {result.stderr.strip()}")
- sys.exit(EXIT_GENERAL)
-
-
-def same_machine_deliver(message_path: Path, sig_path: Path | None, target_dir: Path, canonical_name: str) -> None:
- """Atomic-write delivery: stage .asc, mv to final, then stage .org, mv to final."""
- target_dir.mkdir(parents=True, exist_ok=True)
- final_msg = target_dir / canonical_name
- final_sig = target_dir / f"{canonical_name}.asc"
-
- if sig_path is not None:
- # Stage .asc first, mv to final, THEN stage .org and mv to final.
- with tempfile.NamedTemporaryFile(
- mode="wb", dir=target_dir, prefix=f".tmp.{canonical_name}.asc.", delete=False
- ) as tmp:
- tmp.write(sig_path.read_bytes())
- tmp_sig_path = Path(tmp.name)
- os.replace(tmp_sig_path, final_sig)
-
- # Re-check HALT between .asc and .org per the layered-checks rule.
- check_halt()
-
- with tempfile.NamedTemporaryFile(
- mode="wb", dir=target_dir, prefix=f".tmp.{canonical_name}.", delete=False
- ) as tmp:
- tmp.write(message_path.read_bytes())
- tmp_msg_path = Path(tmp.name)
- os.replace(tmp_msg_path, final_msg)
-
-
-def cross_machine_deliver(
- message_path: Path,
- sig_path: Path | None,
- canonical_name: str,
- host: str,
- ssh_user: str,
- inbox_path: str,
- retries: int,
-) -> bool:
- """rsync push the .asc first (if signed), re-check HALT, then push the .org.
-
- Returns True on success, False on persistent failure (after retries).
- """
- # Stage local copies with the canonical name so rsync sets the right
- # destination filename.
- with tempfile.TemporaryDirectory(prefix="cross-agent-send-") as staging:
- staging_dir = Path(staging)
- local_msg = staging_dir / canonical_name
- local_msg.write_bytes(message_path.read_bytes())
- local_sig = None
- if sig_path is not None:
- local_sig = staging_dir / f"{canonical_name}.asc"
- local_sig.write_bytes(sig_path.read_bytes())
-
- backoffs = [5, 30, 120]
- # Step 1: push .asc first if signed.
- if local_sig is not None:
- if not _rsync_with_retries(local_sig, host, ssh_user, inbox_path, retries, backoffs):
- return False
-
- # Re-check HALT between .asc and .org per the layered-checks rule.
- check_halt()
-
- # Step 2: push .org.
- if not _rsync_with_retries(local_msg, host, ssh_user, inbox_path, retries, backoffs):
- return False
-
- return True
-
-
-def _rsync_with_retries(
- src: Path, host: str, ssh_user: str, inbox_path: str, retries: int, backoffs: list[int]
-) -> bool:
- target = f"{ssh_user}@{host}:{inbox_path}/"
- last_err = ""
- for attempt in range(retries + 1):
- if attempt > 0:
- check_halt()
- wait = backoffs[min(attempt - 1, len(backoffs) - 1)]
- err(f"rsync attempt {attempt} failed: {last_err}; retrying in {wait}s")
- time.sleep(wait)
- try:
- result = subprocess.run(
- ["rsync", "-a", str(src), target],
- capture_output=True,
- text=True,
- )
- except FileNotFoundError:
- err("rsync not found; install rsync")
- return False
- if result.returncode == 0:
- return True
- last_err = result.stderr.strip() or f"exit {result.returncode}"
- err(f"rsync failed after {retries + 1} attempts: {last_err}")
- return False
-
-
-def write_failed_send_marker(dest: str, message_path: Path, error: str, retry_log: list[str]) -> None:
- FAILED_SENDS_DIR.mkdir(parents=True, exist_ok=True)
- timestamp = _dt.datetime.now(_dt.timezone.utc).strftime("%Y%m%dT%H%M%SZ")
- safe_basename = re.sub(r"[^A-Za-z0-9._-]", "_", message_path.name)
- marker = FAILED_SENDS_DIR / f"{timestamp}-{dest.replace('.', '-')}-{safe_basename}.json"
- marker.write_text(json.dumps(
- {
- "timestamp": timestamp,
- "destination": dest,
- "message_path": str(message_path),
- "error": error,
- "retry_log": retry_log,
- },
- indent=2,
- ))
- err(f"marker written: {marker}")
-
-
-def main() -> int:
- parser = argparse.ArgumentParser(description="Send a cross-agent message.")
- parser.add_argument("destination", help="Destination as <machine>.<project>")
- parser.add_argument("message_file", type=Path, help="Path to the message body file")
- parser.add_argument("--no-sign", action="store_true", help="Skip GPG signing (testing only)")
- parser.add_argument("--retries", type=int, default=3, help="Retry count for cross-machine sends")
- parser.add_argument("--key", help="GPG key id to sign with (default: user's primary)")
- args = parser.parse_args()
-
- check_halt()
-
- if not args.message_file.is_file():
- err(f"message file not found: {args.message_file}")
- return EXIT_GENERAL
-
- fm = parse_frontmatter(args.message_file)
- validate_frontmatter(fm)
-
- peers = load_peers()
- machine, project, host, ssh_user = resolve_destination(args.destination, peers)
- inbox_path = resolve_inbox_path(project, peers)
-
- sender = derive_sender_project()
- canonical_name = generate_canonical_filename(sender, fm["CONVERSATION_ID"])
-
- sig_tmp = None
- if not args.no_sign:
- sig_tmp = args.message_file.with_suffix(args.message_file.suffix + ".asc.tmp")
- sign(args.message_file, sig_tmp, args.key)
-
- try:
- if host is None:
- # Same-machine delivery.
- target_dir = Path(os.path.expanduser(inbox_path))
- same_machine_deliver(args.message_file, sig_tmp, target_dir, canonical_name)
- print(f"sent: {target_dir}/{canonical_name}")
- return EXIT_OK
- else:
- ok = cross_machine_deliver(
- args.message_file, sig_tmp, canonical_name, host, ssh_user, inbox_path, args.retries
- )
- if ok:
- print(f"sent: {ssh_user}@{host}:{inbox_path}/{canonical_name}")
- return EXIT_OK
- write_failed_send_marker(args.destination, args.message_file, "rsync failed after retries", [])
- return EXIT_CROSS_MACHINE_FAILED
- finally:
- if sig_tmp is not None and sig_tmp.exists():
- sig_tmp.unlink()
-
-
-if __name__ == "__main__":
- sys.exit(main())
diff --git a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-send.md b/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-send.md
deleted file mode 100644
index 29bfb24..0000000
--- a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-send.md
+++ /dev/null
@@ -1,199 +0,0 @@
-# cross-agent-send
-
-**Purpose.** Send a cross-agent message file to a specific destination. Handles
-peer-config lookup, GPG signing, atomic write (same-machine) or rsync push
-(cross-machine), retry-with-backoff, and failure surfacing.
-
-This is the canonical writer. The protocol spec defers all writer mechanics to
-this script.
-
-## Usage
-
-```
-cross-agent-send <destination> <message-file> [--no-sign] [--retries N]
-```
-
-### Positional arguments
-
-| Position | Meaning | Example |
-|---|---|---|
-| 1 | Destination as `<machine>.<project>` | `homelab.career`, `velox.career` |
-| 2 | Message file (already-formatted `.org`) | `/tmp/my-message.org` |
-
-### Flags
-
-| Flag | Default | Purpose |
-|---|---|---|
-| `--no-sign` | (signing on) | Skip GPG signing. Use only for testing; receivers reject unsigned messages by default. |
-| `--retries N` | 3 | Override retry count for cross-machine sends. |
-| `--key <key-id>` | (user's primary key) | GPG key to sign with. Resolution order: `--key` flag, `GPG_USER` env, `git config user.signingkey`, then the first secret key in the keyring. |
-
-## Behavior
-
-### Filename generation (script-controlled)
-
-The script generates the canonical destination filename from the message's
-frontmatter and sender context. The user's input filename is ignored — pass any
-path, the script names the destination correctly:
-
-```
-<UTC-now>T<HHMMSS>Z-from-<sender-slug>-<short-conv-id>.org
-```
-
-`<sender-slug>` comes from the sender machine's project name (config or
-hostname-based). `<short-conv-id>` is read from the message's
-`#+CONVERSATION_ID` frontmatter field. UTC timestamp is generated at send time.
-
-The script also performs the **sender-side max-seen scan** before writing: it
-reads the receiver's `from-agents/` directory, finds the highest existing
-sequence in this conversation across both sender prefixes, and (best-effort)
-suggests `max(seen) + 1` for the next sequence. The user/agent is responsible
-for setting `#+SEQUENCE` in the message body; the script only advises.
-
-### Same-machine destinations
-
-Resolved when the destination's machine matches the current hostname (or is
-not in `peers.toml` as a remote). Steps:
-
-1. Parse frontmatter; extract `CONVERSATION_ID` and `TIMESTAMP`. Validate per
- the *Validation before send* section below.
-2. Generate canonical filename per *Filename generation* above.
-3. Sign: `gpg --detach-sign --armor --output <canonical>.asc --local-user <key> <input>`.
-4. Compute target: read `peers.toml` for the project's `inbox_path`. If
- missing, fall back to `~/projects/<project>/inbox/from-agents/`.
-5. **Atomic write with strict ordering** (signature must precede message):
- - Stage `.asc`: write to `<target>/.tmp.XXXXXX-<canonical>.asc`,
- then `mv` to `<target>/<canonical>.asc`.
- - **Then** stage `.org`: write to `<target>/.tmp.XXXXXX-<canonical>`,
- then `mv` to `<target>/<canonical>`.
- - Receivers only act on `.org` files; staging the `.asc` first guarantees
- the signature is present when the receiver opens the message. Out-of-order
- would race: receiver could read the `.org` before the `.asc` lands and
- fail GPG verify even though the sender did everything right.
-6. Exit 0 on success. Exit non-zero if any step fails.
-
-### Cross-machine destinations
-
-Steps:
-
-1. Parse + generate canonical filename, as same-machine steps 1-2.
-2. Sign locally to `<input>.asc` (or a tmp staging file).
-3. rsync push **with the same .asc-first ordering**:
- - `rsync -a <input>.asc <ssh-user>@<host>:<inbox_path>/<canonical>.asc`
- - **Then** `rsync -a <input> <ssh-user>@<host>:<inbox_path>/<canonical>`
- rsync writes to a hidden temp file then renames atomically by default
- (`--inplace` would defeat this; do not pass it).
-4. Retry on failure: 5s, 30s, 120s backoff, then surface error.
-5. On persistent failure: write a marker file to
- `~/.local/state/cross-agent-comms/failed-sends/<timestamp>-<dest>-<canonical>.json`
- containing the destination, message path, error, and retry log. Exit non-zero.
-
-### Validation before send
-
-- Destination resolves via `peers.toml` (or local fallback). If neither, exit
- immediately with `destination not found in peers.toml; available: <list>`.
-- Message file must be readable, non-empty, and have valid org-mode frontmatter
- with **all** of the following required fields:
- - `#+TITLE`
- - `#+CONVERSATION_ID`
- - `#+MESSAGE_TYPE`
- - `#+SEQUENCE`
- - `#+TIMESTAMP`
- - `#+PROTOCOL_VERSION` (must equal `5` for v5)
-
- If any required field is missing or malformed, exit immediately with a parse
- error naming the offending field.
-
-- Optional fields the script recognizes and passes through (no special
- handling beyond preservation):
- - `#+REQUIRES_TOOLS` — comma-separated tool/MCP slugs the receiver needs.
- - `#+RELEASE_STATUS` — valid only on `MESSAGE_TYPE: release`. Values per
- spec: `complete`, `cancelled`, `withdrawn-after-pushback`,
- `abandoned-after-escalation`.
- - `#+WORKFLOW_VERSION` — sender's version of the cross-agent-comms workflow
- file. Currently advisory; receiver may warn on mismatch but does not block.
-
-## Configuration
-
-Reads `~/.config/cross-agent-comms/peers.toml` for peer routing:
-
-```toml
-[peers.velox]
-host = "velox.local"
-ssh_user = "cjennings"
-
-# Optional: per-project inbox-path overrides for non-default layouts.
-[projects.work]
-inbox_path = "~/projects/work/inbox/from-agents"
-
-[projects.homelab]
-inbox_path = "~/projects/homelab/inbox/from-agents"
-```
-
-If a project entry is omitted, defaults to `~/projects/<project>/inbox/from-agents`.
-
-## Failure modes
-
-| Symptom | Cause | Fix |
-|---|---|---|
-| `destination not found in peers.toml` | Misspelled destination, or peer not configured | Run `cross-agent-discover` to see available destinations. |
-| `signing failed: no secret key` | GPG key missing or not in keyring | `gpg --list-secret-keys` to confirm. Override with `--key <id>`. |
-| `signing failed: pinentry timed out` | Headless session, GUI pinentry unavailable | Confirm `pinentry-program` in `gpg-agent.conf` matches available pinentry. Per protocols.org, GUI pinentry works from Claude Code. |
-| `rsync exit 255` | SSH unreachable | `cross-agent-discover --peer <name>` to confirm reachability. |
-| `rsync exit 23` | Permission denied at destination | Check destination directory perms (`chmod 700`) and ownership. |
-| Marker file written to `failed-sends/` | Persistent cross-machine failure | Inspect the marker's `error` field. After fixing, retry: `cross-agent-send <dest> <msg>` (the marker is for visibility; it does not auto-retry). |
-| Receiver complains "unsigned message" | `--no-sign` was used in production | Don't use `--no-sign` outside testing. |
-
-## HALT awareness
-
-Checks `~/.config/cross-agent-comms/HALT` at the start of every send AND
-between the `.asc` and `.org` rsync calls AND between each retry iteration.
-On HALT exists, exits with code 5 ("halt active; remove
-~/.config/cross-agent-comms/HALT to resume") without writing or pushing
-further.
-
-Worst case: one in-flight send completes its current rsync step within a few
-seconds before halt kicks in for the next step. New sends are blocked
-immediately. No `pkill` needed — the per-iteration check stops things
-naturally.
-
-If the HALT file exists but is unreadable (permissions wrong), fail-closed —
-treat as if HALT is set. Safer than fail-open.
-
-See `cross-agent-halt.md` for the full halt mechanism.
-
-## Examples
-
-```bash
-# Same-machine send
-cross-agent-send homelab.career /tmp/my-message.org
-
-# Cross-machine send via Tailscale
-cross-agent-send velox.career /tmp/my-message.org
-
-# Test send without signing (receiver will reject)
-cross-agent-send homelab.career /tmp/test.org --no-sign
-
-# Override retry count for a flaky link
-cross-agent-send velox.career /tmp/my-message.org --retries 10
-
-# After a delivery failure, inspect the marker
-cat ~/.local/state/cross-agent-comms/failed-sends/*.json | jq .
-```
-
-## Exit codes
-
-| Code | Meaning |
-|---|---|
-| 0 | Sent successfully. |
-| 1 | General error (parse failure, signing failure, etc.). |
-| 2 | Destination not found in peers.toml. |
-| 3 | Cross-machine delivery failed after retries. Marker file written. |
-| 4 | Frontmatter validation failed. |
-
-## See also
-
-- `cross-agent-discover` — validate destinations before sending.
-- `cross-agent-watch` — receiver-side notification.
-- `cross-agent-status` — see what's queued.
-- `cross-agent-comms.org` — protocol spec, the "what" the script implements.
diff --git a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-status b/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-status
deleted file mode 100755
index 4eee75b..0000000
--- a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-status
+++ /dev/null
@@ -1,185 +0,0 @@
-#!/usr/bin/env python3
-"""Point-in-time snapshot of pending cross-agent messages across local projects.
-
-See cross-agent-status.md. Pending = messages in inbox/from-agents/ whose
-CONVERSATION_ID has no MESSAGE_TYPE: release at a later #+TIMESTAMP.
-
-HALT: prints a prominent banner before normal output, but continues to enumerate.
-"""
-
-from __future__ import annotations
-
-import argparse
-import glob
-import json
-import os
-import re
-import sys
-from pathlib import Path
-
-CONFIG_DIR = Path.home() / ".config" / "cross-agent-comms"
-HALT_FILE = CONFIG_DIR / "HALT"
-DEFAULT_GLOB = str(Path.home() / "projects" / "*" / "inbox" / "from-agents") + "/"
-
-
-def parse_frontmatter(path: Path) -> dict[str, str]:
- try:
- text = path.read_text()
- except OSError:
- return {}
- fm: dict[str, str] = {}
- for line in text.splitlines():
- line = line.rstrip()
- if not line:
- if fm:
- break
- continue
- m = re.match(r"#\+([A-Z_]+):\s*(.*)", line)
- if m:
- fm[m.group(1)] = m.group(2).strip()
- elif fm:
- break
- return fm
-
-
-def project_name_from_path(path: str) -> str:
- """Walk up from path to find ~/projects/<name>/..."""
- home = str(Path.home())
- parts = Path(path).parts
- for i, part in enumerate(parts):
- if part == "projects" and i + 1 < len(parts) and str(Path(*parts[: i + 1])) == os.path.join(home, "projects"):
- return parts[i + 1]
- # Fallback: dir three levels up from the .org file (project/inbox/from-agents/file.org)
- return Path(path).parent.parent.parent.name
-
-
-def scan_project(inbox_dir: Path) -> tuple[int, str | None, int | None]:
- """Return (pending_count, most_recent_filename_or_None, most_recent_age_seconds_or_None)."""
- if not inbox_dir.is_dir():
- return 0, None, None
-
- # Group .org files by CONVERSATION_ID, also collect release timestamps per conv.
- org_files = sorted(inbox_dir.glob("*.org"))
- if not org_files:
- return 0, None, None
-
- by_conv: dict[str, list[tuple[str, str, Path]]] = {} # conv_id -> [(timestamp, msg_type, path)]
- for f in org_files:
- fm = parse_frontmatter(f)
- conv = fm.get("CONVERSATION_ID")
- ts = fm.get("TIMESTAMP")
- mt = fm.get("MESSAGE_TYPE")
- if not conv or not ts or not mt:
- # Malformed file: count as pending under conv "_unparseable".
- by_conv.setdefault("_unparseable", []).append(("", "request", f))
- continue
- by_conv.setdefault(conv, []).append((ts, mt, f))
-
- pending_files: list[Path] = []
- for conv, entries in by_conv.items():
- entries.sort(key=lambda e: e[0])
- # Find the latest release timestamp.
- release_ts = None
- for ts, mt, _f in entries:
- if mt == "release" and (release_ts is None or ts > release_ts):
- release_ts = ts
- for ts, mt, f in entries:
- if mt == "release":
- continue
- if release_ts is not None and ts <= release_ts:
- continue
- pending_files.append(f)
-
- if not pending_files:
- return 0, None, None
-
- # Most-recent by mtime (proxy for arrival order).
- most_recent = max(pending_files, key=lambda p: p.stat().st_mtime)
- import time
- age = int(time.time() - most_recent.stat().st_mtime)
- return len(pending_files), most_recent.name, age
-
-
-def fmt_age(seconds: int | None) -> str:
- if seconds is None:
- return "—"
- if seconds < 60:
- return f"{seconds}s ago"
- if seconds < 3600:
- return f"{seconds // 60} min ago"
- if seconds < 86400:
- return f"{seconds // 3600} hr ago"
- return f"{seconds // 86400} day(s) ago"
-
-
-def render_banner_if_halt() -> None:
- if not HALT_FILE.exists():
- return
- try:
- reason = HALT_FILE.read_text().strip()
- except OSError:
- reason = "(HALT file unreadable; treated as halted)"
- print("⚠ HALT ACTIVE — cross-agent comms paused")
- if reason:
- print(f" reason: {reason}")
- print(f" clear: rm {HALT_FILE} (or: cross-agent-resume)")
- print()
-
-
-def main() -> int:
- parser = argparse.ArgumentParser(description="Snapshot of pending cross-agent messages across local projects.")
- parser.add_argument("--json", action="store_true", help="Emit JSON output")
- parser.add_argument("--projects-glob", default=DEFAULT_GLOB,
- help=f"Glob for project from-agents dirs (default: {DEFAULT_GLOB})")
- args = parser.parse_args()
-
- render_banner_if_halt()
-
- matched = sorted(glob.glob(args.projects_glob))
- rows = []
- for path in matched:
- inbox = Path(path)
- if not inbox.is_dir():
- continue
- proj = project_name_from_path(path)
- count, most_recent, age = scan_project(inbox)
- rows.append({
- "name": proj,
- "pending_count": count,
- "most_recent": (
- {"filename": most_recent, "age_seconds": age}
- if most_recent else None
- ),
- })
-
- # Sort: pending-first, then alphabetical by name.
- rows.sort(key=lambda r: (-r["pending_count"], r["name"]))
-
- if args.json:
- import datetime as _dt
- payload = {
- "scanned_at": _dt.datetime.now(_dt.timezone.utc).isoformat(),
- "halt_active": HALT_FILE.exists(),
- "projects": rows,
- }
- print(json.dumps(payload, indent=2))
- return 0
-
- if not rows:
- print("No projects with inbox/from-agents/ found — 0 pending.")
- return 0
-
- # Human-readable table.
- name_w = max(len("project"), max(len(r["name"]) for r in rows))
- print(f"{'project':<{name_w}} pending most-recent")
- for r in rows:
- most_recent_str = "—"
- if r["most_recent"]:
- most_recent_str = f"{r['most_recent']['filename']} ({fmt_age(r['most_recent']['age_seconds'])})"
- print(f"{r['name']:<{name_w}} {r['pending_count']:<7} {most_recent_str}")
-
- return 0
-
-
-if __name__ == "__main__":
- sys.exit(main())
diff --git a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-status.md b/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-status.md
deleted file mode 100644
index 070330c..0000000
--- a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-status.md
+++ /dev/null
@@ -1,139 +0,0 @@
-# cross-agent-status
-
-**Purpose.** Point-in-time snapshot of pending cross-agent messages across
-every project on this machine. Run from any terminal. No daemon required.
-
-This is the user-pull layer of the cold-start story — `cross-agent-watch`
-pushes notifications, `cross-agent-status` lets the user query.
-
-## Usage
-
-```
-cross-agent-status [--json] [--projects-glob <glob>]
-```
-
-No args required.
-
-### Flags
-
-| Flag | Default | Purpose |
-|---|---|---|
-| `--json` | off (table) | Output as JSON for scripting. |
-| `--projects-glob <glob>` | `~/projects/*/inbox/from-agents/` | Override which directories to scan. |
-
-## Output
-
-### Default (table)
-
-```
-$ cross-agent-status
-project pending most-recent
-career 0 —
-claude-templates 0 —
-clipper 0 —
-homelab 1 20260427T085611Z-from-career-question.org (3 min ago)
-finances 0 —
-... (other 9 projects)
-```
-
-Sort: pending-first, then alphabetical.
-
-### `--json`
-
-```json
-{
- "scanned_at": "2026-04-27T04:13:00-05:00",
- "projects": [
- {
- "name": "homelab",
- "pending_count": 1,
- "most_recent": {
- "filename": "20260427T085611Z-from-career-question.org",
- "age_seconds": 180
- }
- },
- ...
- ]
-}
-```
-
-## Pending semantics
-
-A message is "pending" if it sits in `inbox/from-agents/` AND no
-`MESSAGE_TYPE: release` exists for the same `CONVERSATION_ID` after it.
-
-Concretely:
-
-1. Scan each project's `inbox/from-agents/` for `.org` files.
-2. Group by `CONVERSATION_ID` from frontmatter.
-3. For each conversation, find the highest-`#+TIMESTAMP` message with
- `MESSAGE_TYPE: release`.
-4. Messages with `#+TIMESTAMP` after that release (or in conversations with no
- release) count as pending.
-
-Files without parseable frontmatter are counted as pending and noted in the
-output (single warning row per project).
-
-## Failure modes
-
-| Symptom | Likely cause | Fix |
-|---|---|---|
-| Project missing from output | Project's `.ai/` directory exists but `inbox/from-agents/` does not | Created lazily on first cross-agent message; `mkdir -p` to surface in output. |
-| All projects show "0 pending" but you know one has messages | Glob misresolved, OR all messages are post-release | `cross-agent-status --projects-glob` with explicit path to confirm. |
-| Warning row "N files unparseable in <project>" | Message file has invalid frontmatter | Open the file, fix or move out. |
-
-## Performance
-
-Scans every `.org` file in every watched directory. For Craig's setup (14
-projects, single-digit messages each), runs in <100ms. If a project
-accumulates hundreds of post-release messages, archive them per the persistence
-guidance in the protocol spec.
-
-## HALT awareness
-
-Checks `~/.config/cross-agent-comms/HALT` at start. If HALT exists, prints a
-prominent banner before normal output:
-
-```
-$ cross-agent-status
-⚠ HALT ACTIVE — cross-agent comms paused
- Reason: investigating runaway poll loop, 2026-04-27
- HALT file: ~/.config/cross-agent-comms/HALT
- Resume with: cross-agent-resume
-
-(snapshot continues normally — HALT does not suppress visibility)
-
-project pending most-recent
-career 0 —
-homelab 1 20260427T085611Z-from-career-question.org (3 min ago)
-...
-```
-
-Status is read-only, so it always runs. The banner ensures the user can't
-miss that halt is active when checking inbox state. Reason text comes from
-the HALT file's body; if empty, omit the reason line.
-
-If the HALT file exists but is unreadable, print a warning banner ("HALT
-file present but unreadable; treat as halted") and continue with normal
-output.
-
-See `cross-agent-halt.md` for the full halt mechanism.
-
-## Examples
-
-```bash
-# Snapshot
-cross-agent-status
-
-# JSON for piping
-cross-agent-status --json | jq '.projects[] | select(.pending_count > 0)'
-
-# Single-project query
-cross-agent-status --projects-glob ~/projects/work/inbox/from-agents/
-```
-
-## See also
-
-- `cross-agent-watch` — push notifications on new arrivals.
-- `cross-agent-discover` — enumerate available agents (cross-machine).
-- `cross-agent-comms.org` — protocol spec.
diff --git a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-watch b/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-watch
deleted file mode 100755
index f50ba26..0000000
--- a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-watch
+++ /dev/null
@@ -1,106 +0,0 @@
-#!/usr/bin/env bash
-# cross-agent-watch — desktop-notify on new cross-agent messages.
-#
-# See cross-agent-watch.md. Watches every ~/projects/*/inbox/from-agents/ by
-# default. inotifywait fires create + moved_to events; .tmp.* files are
-# filtered out. HALT suppresses notifications but the watcher keeps running
-# and logs each event with "(suppressed by HALT)".
-
-set -uo pipefail
-
-# Defaults.
-PROJECTS_GLOB="${HOME}/projects/*/inbox/from-agents/"
-LOG_FILE="${HOME}/.local/state/cross-agent-comms/watch.log"
-HALT_FILE="${HOME}/.config/cross-agent-comms/HALT"
-QUIET=0
-NO_NOTIFY=0
-
-# Arg parsing.
-while [[ $# -gt 0 ]]; do
- case "$1" in
- --projects-glob)
- PROJECTS_GLOB="$2"; shift 2 ;;
- --log)
- LOG_FILE="$2"; shift 2 ;;
- --quiet)
- QUIET=1; shift ;;
- --no-notify)
- NO_NOTIFY=1; shift ;;
- -h|--help)
- cat <<EOF
-Usage: cross-agent-watch [--projects-glob GLOB] [--log PATH] [--quiet] [--no-notify]
-
-Watches inbox/from-agents/ directories for new cross-agent messages and fires
-desktop notifications. See cross-agent-watch.md for details.
-EOF
- exit 0 ;;
- *)
- echo "unknown flag: $1" >&2; exit 1 ;;
- esac
-done
-
-# Resolve glob to a concrete list of directories.
-# shellcheck disable=SC2086
-DIRS=( $PROJECTS_GLOB )
-# Filter out non-existent paths (glob may include literal pattern when no match).
-EXISTING=()
-for d in "${DIRS[@]}"; do
- if [[ -d "$d" ]]; then
- EXISTING+=( "$d" )
- fi
-done
-
-if [[ ${#EXISTING[@]} -eq 0 ]]; then
- echo "cross-agent-watch: glob resolved 0 directories: $PROJECTS_GLOB" >&2
- exit 1
-fi
-
-# Ensure log dir exists.
-mkdir -p "$(dirname "$LOG_FILE")"
-
-[[ $QUIET -eq 0 ]] && echo "cross-agent-watch: watching ${#EXISTING[@]} dir(s); log: $LOG_FILE"
-
-# Helper: project name from path like /home/.../projects/<name>/inbox/from-agents/...
-project_name() {
- local path="$1"
- # Match ~/projects/<name>/...
- if [[ "$path" =~ ${HOME}/projects/([^/]+)/ ]]; then
- echo "${BASH_REMATCH[1]}"
- else
- basename "$(dirname "$(dirname "$path")")"
- fi
-}
-
-# Main loop. inotifywait emits one line per event in the format
-# "<full-path>" because we passed --format '%w%f'.
-inotifywait -m -e create,moved_to --format '%w%f' "${EXISTING[@]}" 2>/dev/null \
- | while IFS= read -r path; do
- filename="$(basename "$path")"
-
- # Filter .tmp.* staging files.
- case "$filename" in
- .tmp.*) continue ;;
- esac
-
- # Filter .asc sidecars — they land first per the atomic-write ordering;
- # the .org event will fire after.
- case "$filename" in
- *.asc) continue ;;
- esac
-
- proj="$(project_name "$path")"
- iso="$(date -u "+%Y-%m-%dT%H:%M:%SZ")"
-
- if [[ -e "$HALT_FILE" ]]; then
- printf '%s\t%s\t%s\t(suppressed by HALT)\n' "$iso" "$proj" "$filename" >> "$LOG_FILE"
- [[ $QUIET -eq 0 ]] && echo "[$iso] $proj: $filename (suppressed by HALT)"
- continue
- fi
-
- printf '%s\t%s\t%s\n' "$iso" "$proj" "$filename" >> "$LOG_FILE"
- [[ $QUIET -eq 0 ]] && echo "[$iso] $proj: $filename"
-
- if [[ $NO_NOTIFY -eq 0 ]]; then
- notify info "Cross-agent message" "${proj}: ${filename}" --persist 2>/dev/null || true
- fi
- done
diff --git a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-watch.md b/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-watch.md
deleted file mode 100644
index 04e8005..0000000
--- a/claude-templates/.ai/scripts/cross-agent-comms/cross-agent-watch.md
+++ /dev/null
@@ -1,130 +0,0 @@
-# cross-agent-watch
-
-**Purpose.** Long-running watcher that fires desktop notifications when new
-cross-agent messages land in any project's `inbox/from-agents/` directory.
-This is the primary cold-start mechanism: messages get noticed even when no
-Claude session is active.
-
-## Usage
-
-```
-cross-agent-watch [--projects-glob <glob>] [--log <path>]
-```
-
-No args required. Defaults:
-
-- Watches `~/projects/*/inbox/from-agents/` (matches every project with the
- cross-agent-comms convention).
-- Logs each event to `~/.local/state/cross-agent-comms/watch.log`.
-
-### Flags
-
-| Flag | Default | Purpose |
-|---|---|---|
-| `--projects-glob <glob>` | `~/projects/*/inbox/from-agents/` | Override which directories to watch. Useful for testing on a single project. |
-| `--log <path>` | `~/.local/state/cross-agent-comms/watch.log` | Override log location. Set to `/dev/null` to disable logging. |
-| `--quiet` | off | Suppress stdout output. Notifications still fire. |
-| `--no-notify` | off | Skip `notify` calls. Useful for testing the watcher loop without spamming notifications. |
-
-## Behavior
-
-1. Resolves the projects-glob to a concrete list of directories at startup.
- New projects added to `~/projects/` after startup are NOT picked up — restart
- the watcher to re-resolve.
-2. Runs `inotifywait -m -e create,moved_to --format '%w%f'` against each
- watched directory.
-3. For each event, calls
- `notify info "Cross-agent message" "<project>: <filename>" --persist`. The
- `--persist` flag keeps the page on screen until dismissed, so an inbound
- message that arrives while Craig is away from the desk isn't missed.
-4. Appends an event line to the log:
- `<ISO-8601-timestamp>\t<project>\t<filename>`.
-
-## Event filtering
-
-- Watches `create` AND `moved_to` events. The `moved_to` part is critical for
- the atomic-write convention (`mktemp` + `mv` produces a `moved_to`, not a
- `create`).
-- Files starting with `.tmp.` are ignored — they're staging files from
- in-progress writes that should never produce a notification.
-
-## Installation
-
-### Option A — tmux pane (personal, easy)
-
-Run in a tmux pane that survives session disconnects:
-
-```
-tmux new -d -s cross-agent-watch 'cross-agent-watch'
-```
-
-### Option B — systemd user service (production)
-
-Provided files:
-
-- `~/.config/systemd/user/cross-agent-watch.service`
-- `~/.config/systemd/user/cross-agent-watch.path`
-
-Enable with:
-
-```
-systemctl --user enable --now cross-agent-watch.path
-```
-
-The path unit triggers the service unit on filesystem changes; the service
-unit re-execs `cross-agent-watch` if it dies. Survives reboot.
-
-## Failure modes
-
-| Symptom | Likely cause | Fix |
-|---|---|---|
-| No notifications fire on new files | inotifywait not running, or glob resolved to zero dirs | Check `cross-agent-watch --projects-glob ... --quiet` exits non-zero immediately. Log shows `"resolved 0 directories"`. |
-| Notifications fire on `.tmp.` files | Filter regression | Verify `inotifywait` events show the `.tmp.` files; if so check this script's filter logic. |
-| Some files missed under rapid bursts | inotify queue overflow | Increase `fs.inotify.max_queued_events` sysctl. Default 16384 is usually fine. |
-| Permission denied on a watched dir | Directory perms wrong | `chmod 700 <dir>` and confirm owner. |
-
-## HALT awareness
-
-Checks `~/.config/cross-agent-comms/HALT` on each iteration (each inotifywait
-event fired). If HALT exists, the watcher continues running but **suppresses
-the `notify` call**. The event is still logged, with `(suppressed by HALT)`
-appended:
-
-```
-2026-04-27T04:42:00-05:00 career 20260427T094200Z-from-homelab-test.org (suppressed by HALT)
-```
-
-Logged-but-suppressed events are useful for the operator to see what would
-have fired during the halt window — helpful for diagnosing whatever caused
-the halt.
-
-When HALT clears, suppression stops; subsequent events fire normally. Backlog
-events that arrived during halt are NOT replayed — they get picked up via
-cold-start handling (status CLI, agent startup check, or the next agent
-poll once polling resumes).
-
-If the HALT file exists but is unreadable, fail-closed (suppress) — safer
-than fail-open.
-
-See `cross-agent-halt.md` for the full halt mechanism.
-
-## Examples
-
-```bash
-# Watch all projects, log everything, fire notifications
-cross-agent-watch
-
-# Test against a single project, no notifications, verbose
-cross-agent-watch \
- --projects-glob "$HOME/projects/work/inbox/from-agents/" \
- --no-notify
-
-# Production-style: quiet stdout, log only
-cross-agent-watch --quiet
-```
-
-## See also
-
-- `cross-agent-status` — point-in-time snapshot of pending messages.
-- `cross-agent-send` — counterpart writer.
-- `cross-agent-comms.org` — protocol spec.
diff --git a/claude-templates/.ai/scripts/flashcard-stats.py b/claude-templates/.ai/scripts/flashcard-stats.py
index 1fa5afb..cb580ac 100755
--- a/claude-templates/.ai/scripts/flashcard-stats.py
+++ b/claude-templates/.ai/scripts/flashcard-stats.py
@@ -35,7 +35,12 @@ import re
import sys
from pathlib import Path
-CARD_RE = re.compile(r"^\*\*\s+(.+?)\s+:drill:\s*$")
+# A card is a level-2 heading whose trailing org tag block includes `drill`.
+# Group 1 is the front, group 2 the tag block — so a curated card multi-tagged
+# :fundamental:drill: still counts (it would silently drop under a :drill:$
+# anchor, undercounting the deck). HEADING_RE bounds a card's body.
+CARD_RE = re.compile(r"^\*\*\s+(.+?)\s+(:[A-Za-z0-9_@#%:]+:)\s*$")
+HEADING_RE = re.compile(r"^\*{1,2}\s")
ANSWER_RE = re.compile(r"^\*\*\*\s+Answer\b")
PROP_START_RE = re.compile(r"^\s*:PROPERTIES:\s*$")
PROP_END_RE = re.compile(r"^\s*:END:\s*$")
@@ -177,7 +182,8 @@ def parse_cards(lines: list[str]) -> tuple[list[dict], int]:
n = len(lines)
while i < n:
m = CARD_RE.match(lines[i])
- if not m:
+ tags = [t for t in m.group(2).split(":") if t] if m else []
+ if not (m and "drill" in tags):
i += 1
continue
heading = m.group(1).strip()
@@ -188,7 +194,7 @@ def parse_cards(lines: list[str]) -> tuple[list[dict], int]:
body_lines: list[str] = []
while i < n:
line = lines[i]
- if line.startswith("* ") or CARD_RE.match(line):
+ if HEADING_RE.match(line):
break
if PROP_START_RE.match(line):
prop_count += 1
diff --git a/claude-templates/.ai/scripts/flashcard-to-anki.py b/claude-templates/.ai/scripts/flashcard-to-anki.py
index 7227683..e369fd8 100755
--- a/claude-templates/.ai/scripts/flashcard-to-anki.py
+++ b/claude-templates/.ai/scripts/flashcard-to-anki.py
@@ -10,12 +10,21 @@
Parses org-drill structure:
- Top-level "* Section" headings become tags on every card under them.
- Each "** Card name :drill:" entry becomes a card. Front = heading
- text (sans :drill: tag). Back = entry body with newlines converted
+ text (sans the tag block). Back = entry body with newlines converted
to <br>.
-
-Deck name defaults to the input basename, case preserved. Deck and model
-IDs are derived from the deck name via stable hash so re-importing the
-same deck updates existing cards instead of duplicating them.
+ - A card may carry a second org tag ("** Card :fundamental:drill:").
+ Any heading whose tag block includes `drill` is a card; the other
+ tags ride along as Anki tags next to the section tag, so a curated
+ subset stays grep-able in the source. --tag-filter <tag> emits only
+ the cards carrying that tag, and a subset deck built that way should
+ pass --guid-salt so its notes get their own GUID space (Anki dedupes
+ on GUID, so without it the subset imports empty against the full deck).
+
+Deck name defaults to the org #+TITLE: (so the phone deck reads as the
+curated title), falling back to the input basename when the source has
+no #+TITLE. Deck and model IDs are derived from the deck name via stable
+hash so re-importing the same deck updates existing cards instead of
+duplicating them.
Output defaults to ~/sync/phone/anki/<input-basename>.apkg. The .apkg is
a mobile-Anki artifact the phone picks up from its sync dir, so it lands
@@ -25,6 +34,8 @@ Usage:
flashcard-to-anki.py <input.org>
flashcard-to-anki.py <input.org> --deck "My Deck Name"
flashcard-to-anki.py <input.org> --output /path/to/deck.apkg
+ flashcard-to-anki.py <input.org> --tag-filter fundamental \
+ --deck "DeepSat Fundamentals" --guid-salt fundamentals
Requires genanki, which uv resolves automatically via the PEP 723
script metadata above. No venv or system install needed.
@@ -45,6 +56,15 @@ import genanki
ID_BASE = 1_500_000_000
ID_RANGE = 500_000_000
+# A card is any level-2 heading whose trailing org tag block includes `drill`.
+# Group 1 is the front text, group 2 the colon-delimited tag block (e.g.
+# ":fundamental:drill:") — so a curated subset can carry a second org tag
+# (:fundamental:) and stay grep-able in the source without dropping from the
+# full deck. HEADING_RE bounds a card's body at the next L1/L2 heading.
+CARD_RE = re.compile(r"^\*\*\s+(.+?)\s+(:[A-Za-z0-9_@#%:]+:)\s*$")
+HEADING_RE = re.compile(r"^\*{1,2}\s")
+SECTION_RE = re.compile(r"^\*\s+(.+?)\s*$")
+
def stable_id(name: str, salt: str) -> int:
"""Derive a deterministic 32-bit id from `name` and a `salt`.
@@ -118,33 +138,40 @@ def strip_org_metadata(body_lines: list[str]) -> list[str]:
return cleaned
-def parse(org_text: str) -> list[tuple[str, str, str]]:
- """Return [(front, back_html, tag), ...] for every :drill: card."""
- cards: list[tuple[str, str, str]] = []
- current_section: str | None = None
+def parse(
+ org_text: str, tag_filter: str | None = None
+) -> list[tuple[str, str, list[str]]]:
+ """Return [(front, back_html, anki_tags), ...] for every :drill: card.
- section_re = re.compile(r"^\*\s+(.+?)\s*$")
- card_re = re.compile(r"^\*\*\s+(.+?)\s+:drill:\s*$")
+ A card is any level-2 heading whose trailing org tag block includes
+ `drill`. Non-drill org tags on the heading (e.g. :fundamental:) ride
+ along as Anki tags next to the section tag, so a curated subset stays
+ grep-able in the source. When `tag_filter` is set, only cards carrying
+ that org tag are returned (the subset-deck path).
+ """
+ cards: list[tuple[str, str, list[str]]] = []
+ current_section: str | None = None
lines = org_text.splitlines()
i = 0
while i < len(lines):
line = lines[i]
- sec = section_re.match(line)
+ sec = SECTION_RE.match(line)
if sec:
current_section = sec.group(1).strip()
i += 1
continue
- card = card_re.match(line)
- if card:
- front = card.group(1).strip()
+ m = CARD_RE.match(line)
+ tags = [t for t in m.group(2).split(":") if t] if m else []
+ if m and "drill" in tags:
+ front = m.group(1).strip()
body_lines: list[str] = []
i += 1
while i < len(lines):
nxt = lines[i]
- if nxt.startswith("* ") or card_re.match(nxt):
+ if HEADING_RE.match(nxt):
break
body_lines.append(nxt)
i += 1
@@ -154,8 +181,17 @@ def parse(org_text: str) -> list[tuple[str, str, str]]:
while body_lines and not body_lines[-1].strip():
body_lines.pop()
back_html = "<br>".join(escape_html(ln) for ln in body_lines)
- tag = section_to_tag(current_section) if current_section else "drill"
- cards.append((front, back_html, tag))
+
+ org_tags = [t for t in tags if t != "drill"]
+ if tag_filter and tag_filter not in org_tags:
+ continue
+ anki_tags: list[str] = []
+ if current_section:
+ anki_tags.append(section_to_tag(current_section))
+ anki_tags.extend(org_tags)
+ if not anki_tags:
+ anki_tags = ["drill"]
+ cards.append((front, back_html, anki_tags))
continue
i += 1
@@ -163,21 +199,46 @@ def parse(org_text: str) -> list[tuple[str, str, str]]:
return cards
-def build(cards: list[tuple[str, str, str]], deck_name: str) -> genanki.Deck:
+def card_guid(front: str, guid_salt: str | None) -> str:
+ """GUID for a card's front. A salt gives a derived subset deck its own
+ GUID space so its notes don't collide with the full deck's (Anki dedupes
+ on GUID, which would otherwise import the subset empty). No salt is the
+ original behavior, so an unsalted deck's GUIDs and SRS state are untouched.
+ """
+ return genanki.guid_for(guid_salt, front) if guid_salt else genanki.guid_for(front)
+
+
+def build(
+ cards: list[tuple[str, str, list[str]]],
+ deck_name: str,
+ guid_salt: str | None = None,
+) -> genanki.Deck:
deck = genanki.Deck(stable_id(deck_name, "deck"), deck_name)
model = make_model(deck_name)
- for front, back, tag in cards:
+ for front, back, tags in cards:
note = genanki.Note(
model=model,
fields=[front, back],
- tags=[tag],
- guid=genanki.guid_for(front),
+ tags=tags,
+ guid=card_guid(front, guid_salt),
)
deck.add_note(note)
return deck
-def default_deck_name(input_path: Path) -> str:
+def default_deck_name(input_path: Path, org_text: str) -> str:
+ """Deck name defaults to the org #+TITLE:, falling back to the basename.
+
+ The #+TITLE drives both the org-drill display in Emacs and the Anki
+ deck name on the phone, so the consumed deck reads as the curated
+ title ("Refutations") rather than the filename slug
+ ("refutation-drill"). Falls back to the input basename (case
+ preserved) when the source has no non-empty #+TITLE line.
+ """
+ for line in org_text.splitlines():
+ m = re.match(r"^#\+TITLE:\s*(.*\S)\s*$", line, re.IGNORECASE)
+ if m:
+ return m.group(1).strip()
return input_path.stem
@@ -197,7 +258,7 @@ def main() -> int:
)
parser.add_argument(
"--deck",
- help="Deck name. Defaults to the input basename.",
+ help="Deck name. Defaults to the org #+TITLE, or the input basename.",
)
parser.add_argument(
"--output",
@@ -205,6 +266,16 @@ def main() -> int:
help="Output .apkg path. Defaults to "
"~/sync/phone/anki/<input-basename>.apkg.",
)
+ parser.add_argument(
+ "--tag-filter",
+ help="Emit only cards carrying this org tag (e.g. --tag-filter "
+ "fundamental for a curated subset deck).",
+ )
+ parser.add_argument(
+ "--guid-salt",
+ help="Salt note GUIDs so a subset deck gets its own GUID space and "
+ "imports non-empty without disturbing the full deck's SRS state.",
+ )
args = parser.parse_args()
input_path: Path = args.input.expanduser().resolve()
@@ -213,16 +284,22 @@ def main() -> int:
return 1
org_text = input_path.read_text(encoding="utf-8")
- deck_name = args.deck or default_deck_name(input_path)
+ deck_name = args.deck or default_deck_name(input_path, org_text)
output_path: Path = (args.output or default_output_path(input_path)).expanduser().resolve()
output_path.parent.mkdir(parents=True, exist_ok=True)
- cards = parse(org_text)
+ cards = parse(org_text, tag_filter=args.tag_filter)
if not cards:
- print(f"error: no :drill: cards found in {input_path}", file=sys.stderr)
+ if args.tag_filter:
+ print(
+ f"error: no :drill: cards tagged :{args.tag_filter}: in {input_path}",
+ file=sys.stderr,
+ )
+ else:
+ print(f"error: no :drill: cards found in {input_path}", file=sys.stderr)
return 1
- deck = build(cards, deck_name)
+ deck = build(cards, deck_name, guid_salt=args.guid_salt)
genanki.Package(deck).write_to_file(str(output_path))
print(f"wrote {output_path} ({len(cards)} cards, deck '{deck_name}')")
return 0
diff --git a/claude-templates/.ai/scripts/inbox-send.py b/claude-templates/.ai/scripts/inbox-send.py
index 5373bd4..663efcb 100755
--- a/claude-templates/.ai/scripts/inbox-send.py
+++ b/claude-templates/.ai/scripts/inbox-send.py
@@ -31,6 +31,7 @@ import os
import re
import shutil
import sys
+import tempfile
from datetime import datetime
from pathlib import Path
@@ -48,7 +49,7 @@ def resolve_roots() -> list[Path]:
config = Path.home() / ".claude" / "inbox-roots.txt"
if config.is_file():
paths: list[Path] = []
- for line in config.read_text().splitlines():
+ for line in config.read_text(encoding="utf-8").splitlines():
line = line.strip()
if line and not line.startswith("#"):
paths.append(Path(line).expanduser())
@@ -69,17 +70,28 @@ def discover_projects(roots: list[Path]) -> list[Path]:
a specific project root (included directly if it qualifies).
"""
projects: list[Path] = []
+ seen: set[Path] = set()
+
+ def _add(p: Path) -> None:
+ # Dedupe on the resolved path: a roots config naming both a parent and
+ # one of its children would otherwise list the child project twice, at
+ # two different indices.
+ key = p.resolve()
+ if key not in seen:
+ seen.add(key)
+ projects.append(p)
+
for root in roots:
if not root.is_dir():
continue
if _is_project(root):
- projects.append(root)
+ _add(root)
continue
for child in sorted(root.iterdir()):
if not child.is_dir():
continue
if _is_project(child):
- projects.append(child)
+ _add(child)
return projects
@@ -136,8 +148,21 @@ def slugify_filename(stem: str, max_length: int = MAX_SLUG_LENGTH) -> str:
return truncated.strip("-._")
+def display_name(path: Path) -> str:
+ """The name a project is referred to by — its basename with dots stripped.
+
+ Dotted directories (`.emacs.d`, `.dotfiles`) are awkward to name in
+ conversation, so they're addressed dot-stripped: `emacsd`, `dotfiles`.
+ """
+ return path.name.replace(".", "")
+
+
def find_target(target_name: str, projects: list[Path]) -> Path | None:
- """Resolve `target_name` against the project list (basename or numeric index)."""
+ """Resolve `target_name` against the project list (basename or numeric index).
+
+ An exact basename match wins. Failing that, a dot-stripped alias matches —
+ so `emacsd` resolves `.emacs.d` and `dotfiles` resolves `.dotfiles`.
+ """
if target_name.isdigit():
idx = int(target_name) - 1
if 0 <= idx < len(projects):
@@ -146,6 +171,10 @@ def find_target(target_name: str, projects: list[Path]) -> Path | None:
for p in projects:
if p.name == target_name:
return p
+ norm = target_name.replace(".", "")
+ for p in projects:
+ if display_name(p) == norm:
+ return p
return None
@@ -160,6 +189,56 @@ def build_text_org(message: str, source_name: str, timestamp: str) -> str:
)
+def uniquify(dest: Path) -> Path:
+ """Return dest, or dest with a -2/-3/... stem suffix when it already exists.
+
+ Two sends in the same minute whose text starts with the same phrase
+ derive identical filenames, and the second silently overwrote the
+ first (a message was lost this way, 2026-07-02). Never overwrite.
+ """
+ if not dest.exists():
+ return dest
+ n = 2
+ while True:
+ candidate = dest.with_name(f"{dest.stem}-{n}{dest.suffix}")
+ if not candidate.exists():
+ return candidate
+ n += 1
+
+
+def _atomic_write(dest: Path, writer) -> None:
+ """Write to a temp file in dest's directory, then rename it into place.
+
+ dest is another project's inbox/, and a direct write truncates the target
+ on open, so any mid-write failure (a full disk, an encoding error, an
+ interrupted process) leaves a zero-byte .org there. inbox-status counts
+ that phantom as a pending handoff and blocks a turn in the receiving
+ project over a file with no content and no sender (2026-07-23). Writing to
+ a temp sibling and os.replace-ing means the inbox only ever sees a complete
+ file. os.replace is atomic within one filesystem, and the temp sits in the
+ same directory as dest, so it is.
+
+ `writer` receives the open temp path and fills it. On any failure the temp
+ is removed and the error re-raised, so a caught error never leaves debris.
+ """
+ fd, tmp = tempfile.mkstemp(
+ dir=dest.parent, prefix=".inbox-send-", suffix=dest.suffix
+ )
+ os.close(fd)
+ tmp_path = Path(tmp)
+ # mkstemp creates the temp 0600; give the delivered file the umask-default
+ # mode the old direct write produced, so inbox files stay readable as before.
+ umask = os.umask(0)
+ os.umask(umask)
+ os.chmod(tmp_path, 0o666 & ~umask)
+ try:
+ writer(tmp_path)
+ os.replace(tmp_path, dest)
+ except BaseException:
+ tmp_path.unlink(missing_ok=True)
+ raise
+
+
def send_text(
target_inbox: Path,
message: str,
@@ -174,8 +253,9 @@ def send_text(
if not slug:
raise ValueError(f"could not derive a slug from text: {message!r}")
filename = f"{now.strftime(TS_FILENAME_FMT)}-from-{source_name}-{slug}.org"
- dest = target_inbox / filename
- dest.write_text(build_text_org(message, source_name, now.strftime(TS_DOC_FMT)))
+ dest = uniquify(target_inbox / filename)
+ body = build_text_org(message, source_name, now.strftime(TS_DOC_FMT))
+ _atomic_write(dest, lambda p: p.write_text(body, encoding="utf-8"))
return dest
@@ -194,8 +274,8 @@ def send_file(
raise ValueError(f"could not derive a slug from file: {src_path}")
ext = src_path.suffix
filename = f"{now.strftime(TS_FILENAME_FMT)}-from-{source_name}-{slug}{ext}"
- dest = target_inbox / filename
- shutil.copy2(src_path, dest)
+ dest = uniquify(target_inbox / filename)
+ _atomic_write(dest, lambda p: shutil.copyfile(src_path, p))
return dest
@@ -206,9 +286,9 @@ def print_project_list(projects: list[Path], current: Path | None) -> None:
print("No projects (.ai/ + inbox/) found under the configured roots.")
return
print(f"Available .ai projects ({len(others)}):")
- width = max(len(p.name) for p in others)
+ width = max(len(display_name(p)) for p in others)
for i, p in enumerate(others, 1):
- print(f" {i}. {p.name:<{width}} {p}")
+ print(f" {i}. {display_name(p):<{width}} {p}")
def main() -> int:
@@ -276,7 +356,10 @@ def main() -> int:
else:
assert args.file is not None
dest = send_file(target_inbox, args.file, source_name, args.name, now)
- except (ValueError, FileNotFoundError) as exc:
+ except (ValueError, OSError) as exc:
+ # OSError covers FileNotFoundError (missing source), PermissionError
+ # (unreadable source), and any atomic-write failure — all should
+ # surface as the clean "inbox-send: <message>" error, never a traceback.
print(f"inbox-send: {exc}", file=sys.stderr)
return 1
diff --git a/claude-templates/.ai/scripts/inbox-status b/claude-templates/.ai/scripts/inbox-status
index b917144..17031af 100755
--- a/claude-templates/.ai/scripts/inbox-status
+++ b/claude-templates/.ai/scripts/inbox-status
@@ -35,6 +35,7 @@ mapfile -t pending < <(find inbox -maxdepth 1 -type f \
! -name '.gitkeep' \
! -name 'lint-followups.org' \
! -name 'PROCESSED-*' \
+ ! -name '.inbox-send-*' \
-printf '%f\n' 2>/dev/null | sort)
n=${#pending[@]}
diff --git a/claude-templates/.ai/scripts/lint-org.el b/claude-templates/.ai/scripts/lint-org.el
index 8f55cc6..33dc52f 100644
--- a/claude-templates/.ai/scripts/lint-org.el
+++ b/claude-templates/.ai/scripts/lint-org.el
@@ -2,16 +2,19 @@
;;
;; Usage:
;; emacs --batch -q -l lint-org.el FILE.org [FILE.org ...]
+;; report only (the default) — categorize without modifying the file.
+;; A linter reports, it doesn't write; mutation requires --fix.
+;; --check is accepted as an explicit alias of this default.
+;;
+;; emacs --batch -q -l lint-org.el --fix FILE.org [FILE.org ...]
;; apply mechanical fixes in place, emit judgment items on stdout for the
;; command layer to walk
;;
-;; emacs --batch -q -l lint-org.el --check FILE.org [FILE.org ...]
-;; report only — categorize without modifying the file
-;;
-;; emacs --batch -q -l lint-org.el --followups-file=PATH FILE.org
+;; emacs --batch -q -l lint-org.el --fix --followups-file=PATH FILE.org
;; apply mechanical fixes; if any judgment items remain, append them to
;; PATH as an org section dated today. Used by wrap-it-up to defer the
;; judgment walk to the next morning's review without blocking the wrap.
+;; (--followups-file only writes in --fix mode.)
;;
;; Mechanical categories (auto-fixed):
;; item-number add [@N] directive to drifted bullets
@@ -29,6 +32,15 @@
;; link-to-local-file broken file: links
;; invalid-fuzzy-link broken *Heading refs
;; suspicious-language-in-src-block unknown source-block language
+;; org-table-standard table wider than budget / missing rules
+;; level-2-dated-header ** dated header instead of a keyword
+;; indented-heading whitespace before stars (demoted to body)
+;; empty-heading bare stars with no title
+;; malformed-priority-cookie [#x]-shaped token org rejected
+;; level2-done-without-closed completed level-2 task with no CLOSED
+;; task-missing-last-reviewed open level-2 task with no :LAST_REVIEWED:
+;; subtask-done-not-dated level-3+ done sub-task still a DONE keyword
+;; dated-log-heading-active-timestamp dated-log heading with a live SCHEDULED/DEADLINE
;; (anything else) surfaced as judgment with checker name
;;
;; Output format on stdout:
@@ -59,9 +71,23 @@
Each plist has :kind (mechanical-fixed | judgment), :line, :checker, :msg.
Mechanical entries from --check mode also carry :preview t.")
(defvar lo-check-only nil
- "Non-nil means run in report-only mode — no buffer writes.")
+ "Non-nil means run in report-only mode — no buffer writes.
+The CLI defaults this to t (a linter reports, it doesn't write);
+`--fix' is what enables writes on a command-line run.")
(defvar lo-current-file nil
"Path of the file currently being processed.")
+
+(defun lo--spec-file-p ()
+ "Non-nil when the current file lives under a docs/specs/ directory.
+The four todo-format-family checkers encode todo.org completion conventions
+and misfire on a spec: a spec's Decisions section legitimately carries a
+level-2 DONE with no CLOSED cookie, and its review-history section carries
+level-2 dated headings. docs/specs/ is the canonical spec home per the
+docs-lifecycle rule, so a path segment match is the scope test. Link,
+table, and structural checks still run on specs — only the todo-format
+family is scoped out."
+ (and lo-current-file
+ (string-match-p "/docs/specs/" (expand-file-name lo-current-file))))
(defvar lo-followups-file nil
"When non-nil, after a non-check run any judgment items are appended to this
path as an org section dated today. The file is created if missing.")
@@ -280,6 +306,52 @@ Craig-specific annotation marker rather than Babel src-block syntax."
(lo--goto-line line)
(looking-at-p "^[ \t]*#\\+begin_src[ \t]+cj:")))
+(defvar-local lo--matched-blocks-cache nil
+ "Cons of (TICK . REGIONS) memoizing `lo--matched-block-regions'.
+TICK is the `buffer-chars-modified-tick' the regions were computed at, so a
+fix applied mid-pass invalidates them.")
+
+(defun lo--matched-block-regions ()
+ "Return ((BEGIN-LINE . END-LINE) ...) for every correctly paired block.
+Scans lines directly rather than asking org, because org's own parser is what
+mis-reads these blocks: a heading-shaped line inside a verbatim body reads as a
+structural break and loses the open block. The scan applies org's real rule —
+once a block is open, only its own `#+end_TYPE' closes it, so a nested
+`#+begin_' or a foreign `#+end_' in the body is just text."
+ (let ((tick (buffer-chars-modified-tick)))
+ (if (eql (car lo--matched-blocks-cache) tick)
+ (cdr lo--matched-blocks-cache)
+ (let ((case-fold-search t)
+ (regions nil) (open-type nil) (open-line nil) (line 0))
+ (save-excursion
+ (goto-char (point-min))
+ (while (not (eobp))
+ (setq line (1+ line))
+ (let ((text (buffer-substring-no-properties
+ (line-beginning-position) (line-end-position))))
+ (cond
+ (open-type
+ (when (string-match
+ (format "\\`[ \t]*#\\+end_%s[ \t]*\\'"
+ (regexp-quote open-type))
+ text)
+ (push (cons open-line line) regions)
+ (setq open-type nil open-line nil)))
+ ((string-match "\\`[ \t]*#\\+begin_\\([^ \t\n]+\\)" text)
+ (setq open-type (match-string 1 text)
+ open-line line))))
+ (forward-line 1)))
+ (setq lo--matched-blocks-cache (cons tick (nreverse regions)))
+ (cdr lo--matched-blocks-cache)))))
+
+(defun lo--in-matched-block-p (line)
+ "Non-nil when LINE sits within a correctly paired block, delimiters included.
+org-lint reports `invalid-block' at the delimiter lines themselves, so the
+range has to be inclusive for the suppression to reach them."
+ (cl-some (lambda (region)
+ (and (>= line (car region)) (<= line (cdr region))))
+ (lo--matched-block-regions)))
+
(defun lo--handle-item (item)
(let ((name (lo--checker-name item))
(line (lo--line item))
@@ -292,6 +364,13 @@ Craig-specific annotation marker rather than Babel src-block syntax."
wrong-header-argument))
(lo--cj-comment-block-opener-p line))
nil)
+ ;; `invalid-block' on a block that is in fact correctly paired — the
+ ;; checker is org-lint's own, so this filters its output rather than
+ ;; fixing a local checker. A genuinely unterminated block isn't in any
+ ;; matched region, so it still reports.
+ ((and (eq name 'invalid-block)
+ (lo--in-matched-block-p line))
+ nil)
((eq name 'item-number)
(lo--apply-or-preview name line msg #'lo-fix-item-number))
((eq name 'missing-language-in-src-block)
@@ -348,24 +427,266 @@ logical row, matching wrap-org-table.el's grouping."
(defun lo--check-tables ()
"Scan the current buffer for org tables violating the table standard.
-Emits one judgment item per violating table."
+Emits one judgment item per violating table. Pipe-led lines inside
+#+begin_/#+end_ blocks are content (ASCII art, shell pipes), not tables,
+and are skipped — the same block rule `wot-process-file' applies."
(save-excursion
(goto-char (point-min))
- (while (re-search-forward "^[ \t]*|" nil t)
- (let ((start-line (line-number-at-pos))
- (lines nil))
- (beginning-of-line)
- (while (and (not (eobp)) (looking-at "[ \t]*|"))
- (push (buffer-substring-no-properties (line-beginning-position)
- (line-end-position))
- lines)
+ (let ((in-block nil)) ; the open block's type, e.g. "example" — nil outside
+ (while (not (eobp))
+ (cond
+ ;; Type-matched close only: literal #+end_src quoted inside an
+ ;; example block must not clear the flag (see wot-process-file).
+ ((and (not in-block)
+ (looking-at "^[ \t]*#\\+begin_\\([^ \t\n]+\\)"))
+ (setq in-block (downcase (match-string 1)))
+ (forward-line 1))
+ ((and in-block
+ (looking-at-p (format "^[ \t]*#\\+end_%s\\([ \t]\\|$\\)"
+ (regexp-quote in-block))))
+ (setq in-block nil)
(forward-line 1))
- (let ((violations (lo--table-violations (nreverse lines))))
- (when violations
- (lo--emit-judgment
- 'org-table-standard start-line
- (format "table violates the org-table standard: %s — wrap-org-table.el reflows it"
- (string-join violations "; ")))))))))
+ ((and (not in-block) (looking-at-p "^[ \t]*|"))
+ (let ((start-line (line-number-at-pos))
+ (lines nil))
+ (while (and (not (eobp)) (looking-at "[ \t]*|"))
+ (push (buffer-substring-no-properties (line-beginning-position)
+ (line-end-position))
+ lines)
+ (forward-line 1))
+ (let ((violations (lo--table-violations (nreverse lines))))
+ (when violations
+ (lo--emit-judgment
+ 'org-table-standard start-line
+ (format "table violates the org-table standard: %s — wrap-org-table.el reflows it"
+ (string-join violations "; ")))))))
+ (t (forward-line 1)))))))
+
+;;; ---------------------------------------------------------------------------
+;;; level-2 dated-header check (claude-rules/todo-format.md)
+;;
+;; A completed task or resolved VERIFY at level 2 must carry a terminal
+;; keyword (DONE/CANCELLED + CLOSED:), never a dated heading. A `** <date>'
+;; header has no keyword, so todo-cleanup's --archive-done can never archive
+;; it (it accumulates in Open Work forever) and task-review drops it from
+;; selection. Judgment-only, never auto-fixed: the repair needs a
+;; DONE-vs-CANCELLED call and the original heading text, which is a judgment
+;; the sweep can't make. Targets todo/task files; a dated-log-format org
+;; file using `** <date>' headings intentionally will false-positive here, in
+;; which case the human dismisses the judgment item.
+
+(defun lo--check-level2-dated-headers ()
+ "Flag level-2 headings whose text begins with a YYYY-MM-DD date.
+Emits one judgment item per offending heading (checker
+`level-2-dated-header')."
+ (save-excursion
+ (goto-char (point-min))
+ (while (re-search-forward
+ "^\\*\\* \\([0-9]\\{4\\}-[0-9]\\{2\\}-[0-9]\\{2\\}\\)" nil t)
+ (lo--emit-judgment
+ 'level-2-dated-header (line-number-at-pos)
+ "level-2 dated header is a completion defect (todo-format.md): a ** task or VERIFY closes with DONE/CANCELLED + CLOSED:, not a dated heading — convert it so --archive-done can archive it"))))
+
+;;; ---------------------------------------------------------------------------
+;;; structural heading checks (mistakes org-lint does not cover)
+;;
+;; org-lint validates links, drawers, blocks, and babel — but not heading
+;; well-formedness. These four catch hand-edit defects it misses, all
+;; judgment-only (each repair is a human call) and regex-based (no dependence on
+;; which TODO keywords the batch Emacs happens to recognize):
+;;
+;; indented-heading leading whitespace before two-or-more stars; org
+;; demotes it to body text, so the task vanishes from
+;; the agenda and never archives. The worst case — an
+;; invisible task — and silent. Single `*' is left
+;; alone (a valid indented plain-list bullet).
+;; empty-heading a line of bare stars with no title.
+;; malformed-priority-cookie a `[#x]'-shaped token org rejected (lowercase,
+;; multi-char, non-letter) sitting where a cookie
+;; would be.
+;; level2-done-without-closed a level-2 DONE/CANCELLED with no CLOSED line —
+;; directly relevant to todo-cleanup's aging step,
+;; which archives an undated completed task at once.
+
+(defconst lo-done-keywords '("DONE" "CANCELLED")
+ "Heading keywords treated as completed for `lo--check-level2-done-without-closed'.")
+
+(defun lo--check-indented-headings ()
+ "Flag lines that are whitespace + two-or-more stars + space outside any block.
+Org parses a heading only at column 0, so leading whitespace silently demotes a
+would-be heading to body text. Two-or-more stars is required: an indented
+single `*' is a valid plain-list bullet, not a lost heading, so flagging it
+false-positives on legitimate lists; `**'+ is never a bullet, so an indented one
+is unambiguously a demoted level-2+ heading turned invisible. Lines inside
+`#+begin_/#+end_' blocks are skipped — indented asterisks there are legitimate
+content."
+ (save-excursion
+ (goto-char (point-min))
+ (let ((in-block nil))
+ (while (not (eobp))
+ (cond
+ ((looking-at-p "^[ \t]*#\\+begin_") (setq in-block t))
+ ((looking-at-p "^[ \t]*#\\+end_") (setq in-block nil))
+ ((and (not in-block) (looking-at-p "^[ \t]+\\*\\*+[ \t]"))
+ (lo--emit-judgment
+ 'indented-heading (line-number-at-pos)
+ "indented heading: leading whitespace before the stars demotes this to body text — org won't treat it as a heading (it vanishes from the agenda and never archives); dedent to column 0")))
+ (forward-line 1)))))
+
+(defun lo--check-empty-headings ()
+ "Flag headings that are bare stars with no title text.
+A line of nothing but stars is an empty heading — a stray heading-star carrying
+no content."
+ (save-excursion
+ (goto-char (point-min))
+ (while (re-search-forward "^\\*+[ \t]*$" nil t)
+ (lo--emit-judgment
+ 'empty-heading (line-number-at-pos)
+ "empty heading: a line of stars with no title — delete it or give it a title"))))
+
+(defun lo--check-malformed-priority-cookies ()
+ "Flag a heading whose first cookie-shaped token is not a valid priority.
+A valid cookie is a single uppercase letter in `[#A]' form. Verbatim-wrapped
+cookies (`=[#D]=' quoted in a dated-log title) are skipped. Only the first
+token on the line is checked, so a real cookie earlier on the line means a
+later `[#x]' in the title is left alone."
+ (save-excursion
+ (goto-char (point-min))
+ ;; Case-sensitive: a cookie is uppercase only, and case-fold-search defaults
+ ;; to t (which would accept [#a] as valid).
+ (let ((case-fold-search nil))
+ (while (re-search-forward "^\\*+ " nil t)
+ (let ((eol (line-end-position)) (hline (line-number-at-pos)))
+ (when (re-search-forward "\\[#\\([^]]*\\)\\]" eol t)
+ (let ((inner (match-string 1))
+ (before (char-before (match-beginning 0)))
+ (after (char-after (match-end 0))))
+ (unless (or (eq before ?=) (eq after ?=)
+ (string-match-p "\\`[A-Z]\\'" inner))
+ (lo--emit-judgment
+ 'malformed-priority-cookie hline
+ (format "malformed priority cookie [#%s] — a cookie is a single uppercase letter ([#A]) right after the keyword; fix or remove it"
+ inner)))))
+ (goto-char eol))))))
+
+(defun lo--check-level2-done-without-closed ()
+ "Flag a level-2 DONE/CANCELLED heading with no CLOSED line in its own entry.
+todo-cleanup's `--archive-done' aging step archives a completed task with no
+parseable CLOSED date immediately, so an undated completed task silently leaves
+the live file on the next `task-sorted'."
+ (save-excursion
+ (goto-char (point-min))
+ ;; Case-sensitive: DONE/CANCELLED are uppercase keywords, not the words
+ ;; "done"/"cancelled" in a heading title (case-fold-search defaults to t).
+ (let ((case-fold-search nil)
+ (re (format "^\\*\\* \\(%s\\) "
+ (mapconcat #'regexp-quote lo-done-keywords "\\|"))))
+ (while (re-search-forward re nil t)
+ (let ((hline (line-number-at-pos))
+ (entry-end (save-excursion (outline-next-heading) (point))))
+ (save-excursion
+ (forward-line 1)
+ (unless (re-search-forward "^[ \t]*CLOSED:[ \t]*\\[" entry-end t)
+ (lo--emit-judgment
+ 'level2-done-without-closed hline
+ "level-2 DONE/CANCELLED has no CLOSED date — add CLOSED: [YYYY-MM-DD Day]; task-sorted's aging step archives an undated completed task immediately"))))))))
+
+;;; ---------------------------------------------------------------------------
+;;; task-missing-last-reviewed check (claude-rules/todo-format.md)
+;;
+;; A task is stamped `:LAST_REVIEWED:' when it is *created*, not a review cycle
+;; later. An agent filing a task has just written its body and graded its
+;; priority, which is a review by any honest reading — so a fresh task that
+;; carries no stamp reads as "never reviewed" and lands at the top of the next
+;; staleness batch, where re-reviewing it is pure ceremony. Every task filed
+;; during the 2026-07-23 sweep hit exactly that, which is what prompted the rule.
+;;
+;; Judgment-only, deliberately. The stamp's whole value is that its date is
+;; true, and nothing here can know when an unstamped task was actually last
+;; looked at. Auto-stamping today's date would convert a "nobody has reviewed
+;; this" signal into a false "reviewed today" one — worse than the gap it
+;; closes. Flag it; a human or the filing workflow supplies the honest date.
+;;
+;; Scope matches `task-review-staleness.sh' exactly (level-2, open keyword,
+;; priority cookie), so the checker and the staleness count never disagree
+;; about which headings are in the review pool.
+
+(defun lo--check-task-missing-last-reviewed ()
+ "Flag an open level-2 task with a priority cookie and no `:LAST_REVIEWED:'."
+ (save-excursion
+ (goto-char (point-min))
+ (let ((case-fold-search nil))
+ (while (re-search-forward "^\\*\\* \\(TODO\\|DOING\\|VERIFY\\) \\[#[A-D]\\]" nil t)
+ (let ((hline (line-number-at-pos))
+ (entry-end (save-excursion (outline-next-heading) (point))))
+ (save-excursion
+ (forward-line 1)
+ (unless (re-search-forward "^[ \t]*:LAST_REVIEWED:[ \t]*[[0-9]"
+ entry-end t)
+ (lo--emit-judgment
+ 'task-missing-last-reviewed hline
+ "task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed"))))))))
+
+;;; ---------------------------------------------------------------------------
+;;; level-3+ dated-header check (claude-rules/todo-format.md)
+;;
+;; The inverse of the level-2 check above. A completed sub-task — a heading at
+;; level 3 or deeper, under a parent task — becomes a dated event-log entry, not
+;; a DONE keyword, so the parent's subtree grows a chronological history instead
+;; of a long tail of nested DONE lines. An interactive org close
+;; (`org-log-done' → DONE + CLOSED) leaves the keyword in place, and
+;; `--archive-done' only touches level 2, so these accumulate. Flag them for
+;; conversion. Judgment-only and regex-based (independent of which TODO keywords
+;; the batch Emacs recognizes); todo-cleanup.el --convert-subtasks does the fix.
+
+(defun lo--check-subtask-done-not-dated ()
+ "Flag level-3+ headings carrying a done keyword (DONE/CANCELLED/FAILED).
+Emits one judgment item per offending heading (checker
+`subtask-done-not-dated')."
+ (save-excursion
+ (goto-char (point-min))
+ ;; Case-sensitive: the keywords are uppercase, not the words in a title.
+ (let ((case-fold-search nil))
+ (while (re-search-forward
+ "^\\*\\{3,\\} \\(DONE\\|CANCELLED\\|FAILED\\) " nil t)
+ (lo--emit-judgment
+ 'subtask-done-not-dated (line-number-at-pos)
+ "level-3+ done sub-task should be a dated event-log entry (todo-format.md): run todo-cleanup.el --convert-subtasks to rewrite it")))))
+
+;;; ---------------------------------------------------------------------------
+;;; dated-log heading with a stale active planning timestamp (todo-format.md)
+;;
+;; The mechanical backstop for the planning-line-strip rule. A dated event-log
+;; heading (`<stars> YYYY-MM-DD Day @ ...', no TODO keyword) records completed
+;; work — its date lives in the heading. An active `<...>' SCHEDULED or DEADLINE
+;; left on it pins the entry to the agenda forever: org renders any headline with
+;; an active planning timestamp, keyword or not, so a stale SCHEDULED shows as
+;; weeks-overdue long after the work is done. Invisible to a keyword scan (no
+;; TODO) and it survives --archive-done, so nothing else catches it. The
+;; completion rewrite and todo-cleanup --convert-subtasks now strip the planning
+;; line; this flags any that slipped through before that landed, the same way
+;; subtask-done-not-dated backstops the depth rule. Judgment-only.
+
+(defun lo--check-dated-log-active-timestamp ()
+ "Flag a dated event-log heading that still carries an active SCHEDULED/DEADLINE.
+The heading matches `<stars> YYYY-MM-DD Day @ ...' with no TODO keyword; an
+active `<...>' planning timestamp in its entry is the defect. An inactive
+`[...]' timestamp is ignored (org doesn't render it on the agenda). Emits one
+judgment item per offending heading (checker `dated-log-heading-active-timestamp')."
+ (save-excursion
+ (goto-char (point-min))
+ (let ((case-fold-search nil))
+ (while (re-search-forward
+ "^\\*+ [0-9]\\{4\\}-[0-9]\\{2\\}-[0-9]\\{2\\} [A-Za-z]+ @ " nil t)
+ (let ((hline (line-number-at-pos))
+ (entry-end (save-excursion (outline-next-heading) (point))))
+ (save-excursion
+ (forward-line 1)
+ (when (re-search-forward
+ "^[ \t]*\\(?:SCHEDULED\\|DEADLINE\\):[ \t]*<" entry-end t)
+ (lo--emit-judgment
+ 'dated-log-heading-active-timestamp hline
+ "dated-log heading carries an active SCHEDULED/DEADLINE — org renders any active planning timestamp (keyword or not), so it stays on the agenda as weeks-overdue; delete the planning line (todo-format.md)"))))))))
;;; ---------------------------------------------------------------------------
;;; File processing
@@ -401,6 +722,22 @@ left unmodified and mechanical entries are recorded with :preview t."
;; After org-lint items: the custom table-standard scan. Runs on the
;; post-fix buffer; judgment-only, so order doesn't perturb fixes.
(lo--check-tables)
+ ;; Structural heading defects org-lint doesn't cover. These run on
+ ;; every org file, specs included.
+ (lo--check-indented-headings)
+ (lo--check-empty-headings)
+ (lo--check-malformed-priority-cookies)
+ ;; The todo-format family encodes todo.org completion conventions and
+ ;; misfires on a spec (a Decisions section's undated DONE, a
+ ;; review-history dated heading, a phases task with no LAST_REVIEWED).
+ ;; Scope them out of docs/specs/; link, table, and structural checks
+ ;; above still run there.
+ (unless (lo--spec-file-p)
+ (lo--check-level2-dated-headers)
+ (lo--check-level2-done-without-closed)
+ (lo--check-task-missing-last-reviewed)
+ (lo--check-subtask-done-not-dated)
+ (lo--check-dated-log-active-timestamp))
(when (and (not lo-check-only) (buffer-modified-p))
(save-buffer)))
(with-current-buffer buf (set-buffer-modified-p nil))
@@ -507,6 +844,13 @@ After printing, also append judgments to `lo-followups-file' when set."
;;; CLI
(defun lo-main ()
+ ;; Report-only is the CLI default; --fix is the only way a command-line run
+ ;; writes to disk. The old mutate-by-default reformatted five files in one
+ ;; pass before anyone confirmed anything (work project, 2026-07-09).
+ (setq lo-check-only t)
+ (when (member "--fix" command-line-args-left)
+ (setq lo-check-only nil)
+ (setq command-line-args-left (delete "--fix" command-line-args-left)))
(when (member "--check" command-line-args-left)
(setq lo-check-only t)
(setq command-line-args-left (delete "--check" command-line-args-left)))
@@ -518,7 +862,7 @@ After printing, also append judgments to `lo-followups-file' when set."
(setq command-line-args-left (delete followups command-line-args-left))))
(if (null command-line-args-left)
(progn
- (princ "Usage: emacs --batch -q -l lint-org.el [--check] [--followups-file=PATH] FILE.org ...\n")
+ (princ "Usage: emacs --batch -q -l lint-org.el [--fix] [--check] [--followups-file=PATH] FILE.org ...\n")
(kill-emacs 1))
(let ((files command-line-args-left))
(setq command-line-args-left nil)
@@ -537,7 +881,7 @@ this file without firing the CLI dispatch — under `ert-run-tests-batch-and-exi
the trailing args are things like `-f ert-run-tests-batch-and-exit'."
(and command-line-args-left
(cl-every (lambda (a)
- (cond ((member a '("--check")) t)
+ (cond ((member a '("--check" "--fix")) t)
((string-prefix-p "--followups-file=" a) t)
((string-prefix-p "-" a) nil)
(t (file-readable-p a))))
diff --git a/claude-templates/.ai/scripts/route-batch b/claude-templates/.ai/scripts/route-batch
new file mode 100755
index 0000000..8f27d19
--- /dev/null
+++ b/claude-templates/.ai/scripts/route-batch
@@ -0,0 +1,175 @@
+#!/usr/bin/env python3
+"""route-batch — the wrap-up router's mechanical go path.
+
+The wrap-up cross-project router (wrap-it-up.org Step 3; wrapup-routing spec
+D7/D8/D9) surfaces the local tasks that inbox process mode stamped with
+:ROUTE_CANDIDATE: <destination> at file time, and on "go" delivers each to its
+destination project's inbox. This script does the mechanical half so the
+subtree surgery is deterministic:
+
+ route-batch --list [--todo todo.org]
+ One "<destination>\t<heading>" line per :ROUTE_CANDIDATE:-tagged task.
+ Silent with exit 0 when there are no candidates (the workflow's
+ empty-set-equals-zero-interaction rule). Read-only.
+
+ route-batch --go [--todo todo.org]
+ For each candidate, bottom-up: extract the task's whole subtree
+ (children ride along), drop the :ROUTE_CANDIDATE: line (and the
+ property drawer if that leaves it empty), promote the subtree so its
+ top heading is level 1, write it to a temp file, and deliver it via
+ the sibling inbox-send.py to the destination's inbox/ (one file per
+ task, from-<source> provenance stamped by inbox-send). Only after a
+ successful send is the subtree removed from the local todo.org — a
+ failed send leaves that task in place, is reported, and the run exits
+ non-zero after attempting the rest.
+
+The candidate set is exactly the tagged tasks — never the standing backlog.
+Discovery, roots, and the source-project name all come from inbox-send.py
+(INBOX_SEND_ROOTS sandboxes it in tests). The reject-from-another-project
+flow in inbox process mode is the mis-route recovery; that path is why
+removing the local source after a successful send is safe.
+"""
+
+import argparse
+import os
+import re
+import subprocess
+import sys
+import tempfile
+from pathlib import Path
+
+HEADING_RE = re.compile(r"^(\*+)\s+(.*)$")
+MARKER_RE = re.compile(r"^\s*:ROUTE_CANDIDATE:\s+(\S+)\s*$")
+
+
+def find_candidates(lines):
+ """[(heading_idx, end_idx, marker_idx, destination, heading_text)] —
+ end_idx is one past the subtree's last line."""
+ candidates = []
+ for i, line in enumerate(lines):
+ m = MARKER_RE.match(line)
+ if not m:
+ continue
+ head_idx = None
+ for j in range(i, -1, -1):
+ hm = HEADING_RE.match(lines[j])
+ if hm:
+ head_idx = j
+ level = len(hm.group(1))
+ heading = hm.group(2)
+ break
+ if head_idx is None:
+ continue
+ end = len(lines)
+ for k in range(head_idx + 1, len(lines)):
+ km = HEADING_RE.match(lines[k])
+ if km and len(km.group(1)) <= level:
+ end = k
+ break
+ candidates.append((head_idx, end, i, m.group(1), heading))
+ return candidates
+
+
+def extract_handoff(lines, head_idx, end):
+ """The subtree as handoff text: every :ROUTE_CANDIDATE: line dropped
+ (a marker is meaningless at the destination), empty drawers pruned,
+ headings promoted so the task is level 1."""
+ sub = [l for l in lines[head_idx:end] if not MARKER_RE.match(l)]
+
+ pruned = []
+ i = 0
+ while i < len(sub):
+ if sub[i].strip() == ":PROPERTIES:" and i + 1 < len(sub) and sub[i + 1].strip() == ":END:":
+ i += 2
+ continue
+ pruned.append(sub[i])
+ i += 1
+
+ shift = len(HEADING_RE.match(pruned[0]).group(1)) - 1
+ if shift > 0:
+ pruned = [l[shift:] if HEADING_RE.match(l) else l for l in pruned]
+ return "\n".join(pruned).rstrip() + "\n"
+
+
+def send(destination, handoff_text, slug):
+ inbox_send = Path(__file__).with_name("inbox-send.py")
+ with tempfile.NamedTemporaryFile(
+ "w", suffix=".org", prefix=f"route-{slug}-", delete=False, encoding="utf-8"
+ ) as tf:
+ tf.write(handoff_text)
+ tmp = tf.name
+ try:
+ result = subprocess.run(
+ [sys.executable, str(inbox_send), destination, "--file", tmp],
+ capture_output=True, text=True,
+ )
+ return result.returncode == 0, (result.stderr or result.stdout).strip()
+ finally:
+ os.unlink(tmp)
+
+
+def main():
+ ap = argparse.ArgumentParser(prog="route-batch")
+ mode = ap.add_mutually_exclusive_group(required=True)
+ mode.add_argument("--list", action="store_true", dest="list_mode")
+ mode.add_argument("--go", action="store_true")
+ ap.add_argument("--todo", default="todo.org")
+ args = ap.parse_args()
+
+ todo_path = Path(args.todo)
+ if not todo_path.is_file():
+ return 0 # no todo file, no candidates
+ lines = todo_path.read_text(encoding="utf-8").splitlines()
+ candidates = find_candidates(lines)
+
+ # Two markers in one task's drawer are one candidate, not two: same span +
+ # same destination dedupes. Everything else that overlaps — a tagged child
+ # inside a tagged parent, one task tagged for two destinations — is a
+ # conflict: routing either span would silently take the other (or, with a
+ # stale end index, a bystander task) along. Conflicts are left in place
+ # and reported; the human untangles which project the pieces belong to.
+ deduped = []
+ for cand in candidates:
+ if not any(c[0] == cand[0] and c[1] == cand[1] and c[3] == cand[3] for c in deduped):
+ deduped.append(cand)
+ conflicted = set()
+ for a in deduped:
+ for b in deduped:
+ if a is not b and a[0] <= b[0] and b[1] <= a[1]:
+ conflicted.add(a)
+ conflicted.add(b)
+ routable = [c for c in deduped if c not in conflicted]
+
+ if not deduped:
+ return 0
+
+ if args.list_mode:
+ for _h, _e, _m, dest, heading in deduped:
+ flag = "\tCONFLICT (overlapping candidates — resolve by hand)" if (_h, _e, _m, dest, heading) in conflicted else ""
+ print(f"{dest}\t{heading}{flag}")
+ return 0
+
+ failures = 0
+ for _h, _e, _m, dest, heading in sorted(conflicted):
+ failures += 1
+ print(f"CONFLICT: {dest}\t{heading}\t(overlapping candidate subtrees — left in place, resolve by hand)")
+
+ # Bottom-up so earlier indices stay valid as subtrees are removed; the
+ # file is rewritten after every successful send so a crash mid-run never
+ # leaves an already-sent task still present locally.
+ for head_idx, end, _marker_idx, dest, heading in sorted(routable, reverse=True):
+ handoff = extract_handoff(lines, head_idx, end)
+ slug = re.sub(r"[^a-z0-9]+", "-", heading.lower()).strip("-")[:40] or "task"
+ ok, detail = send(dest, handoff, slug)
+ if ok:
+ del lines[head_idx:end]
+ todo_path.write_text("\n".join(lines).rstrip("\n") + "\n", encoding="utf-8")
+ print(f"routed: {dest}\t{heading}")
+ else:
+ failures += 1
+ print(f"FAILED: {dest}\t{heading}\t({detail})")
+ return 1 if failures else 0
+
+
+if __name__ == "__main__":
+ sys.exit(main())
diff --git a/claude-templates/.ai/scripts/route_recommend.py b/claude-templates/.ai/scripts/route_recommend.py
new file mode 100644
index 0000000..12ab132
--- /dev/null
+++ b/claude-templates/.ai/scripts/route_recommend.py
@@ -0,0 +1,145 @@
+#!/usr/bin/env python3
+"""Wrap-up routing recommendation engine.
+
+Given an inbox keeper's text and a list of candidate project names, infer which
+project the item belongs to, with a confidence tier:
+
+ strong a project's name (or its dot-stripped form, or a path containing it)
+ appears literally in the item
+ weak a distinctive name token overlaps, but the full name doesn't
+ none no overlap; the item stays put
+
+A multi-way tie at the top tier is ambiguous, so it downgrades to weak with a
+deterministic pick (most token overlap, then alphabetical). An empty candidate
+list yields none.
+
+The pure core is `recommend(item, projects) -> (destination, confidence)` — the
+shape the wrap-up router (Phase 4) and the process-inbox marker (Phase 2) both
+call. The CLI wires it to inbox-send.py's `discover_projects` so the candidate
+set is the same project universe inbox-send already knows.
+
+CLI:
+ route_recommend.py --item "<text>" [--exclude <current-project>]
+prints "<destination>\\t<confidence>" on a match, or "none".
+"""
+
+import argparse
+import importlib.util
+import re
+import sys
+from pathlib import Path
+
+# A distinctive-enough token for weak matching; shorter tokens (of, to, id) are
+# too noisy to route on.
+MIN_WEAK_TOKEN = 4
+
+_TOKEN_RE = re.compile(r"[a-z0-9]+")
+
+
+def _tokens(text: str) -> set[str]:
+ return set(_TOKEN_RE.findall(text.lower()))
+
+
+def _name_variants(name: str) -> set[str]:
+ """A project name and its dot-stripped alias (.emacs.d -> emacsd)."""
+ return {v for v in (name.lower(), name.replace(".", "").lower()) if v}
+
+
+def _literal_present(name: str, item_lower: str) -> bool:
+ """True if a name variant appears in the item on word-ish boundaries.
+
+ Boundaries keep 'home' from matching inside 'homeowner' while still
+ matching it inside a path ('~/code/home/...') or a hyphenated name.
+ """
+ for variant in _name_variants(name):
+ if re.search(r"(?<![a-z0-9])" + re.escape(variant) + r"(?![a-z0-9])", item_lower):
+ return True
+ return False
+
+
+def _tiebreak(candidates: list[str], item_tokens: set[str]) -> str:
+ """Most token overlap first, then alphabetical — deterministic."""
+ return sorted(candidates, key=lambda p: (-len(_tokens(p) & item_tokens), p))[0]
+
+
+def recommend(item: str, projects: list[str]) -> tuple[str | None, str]:
+ """Infer the destination project for `item` from `projects`.
+
+ Returns (destination, confidence). confidence is "strong" / "weak" / "none";
+ destination is None exactly when confidence is "none".
+ """
+ if not projects:
+ return (None, "none")
+
+ # Collapse identical names first. Projects are addressed by bare basename, so
+ # two projects sharing one across roots (~/code/notes, ~/projects/notes) arrive
+ # twice; both literal-match, and the tie test below then read that as ambiguity
+ # and downgraded a correct strong match to weak. Deduping here rather than in
+ # discover_destination_names protects every caller of the pure core, not just
+ # the CLI path. Order-preserving, and it collapses only identical names — two
+ # *different* projects matching is real ambiguity and still downgrades.
+ projects = list(dict.fromkeys(projects))
+
+ item_lower = item.lower()
+ item_tokens = _tokens(item)
+
+ strong: list[str] = []
+ weak: list[str] = []
+ for project in projects:
+ if _literal_present(project, item_lower):
+ strong.append(project)
+ continue
+ name_tokens = {t for t in _tokens(project) if len(t) >= MIN_WEAK_TOKEN}
+ if name_tokens & item_tokens:
+ weak.append(project)
+
+ if len(strong) == 1:
+ return (strong[0], "strong")
+ if len(strong) > 1:
+ return (_tiebreak(strong, item_tokens), "weak")
+ if len(weak) == 1:
+ return (weak[0], "weak")
+ if len(weak) > 1:
+ return (_tiebreak(weak, item_tokens), "weak")
+ return (None, "none")
+
+
+def _load_inbox_send():
+ """Load the sibling kebab-named inbox-send.py as a module for its discovery."""
+ path = Path(__file__).with_name("inbox-send.py")
+ spec = importlib.util.spec_from_file_location("inbox_send", path)
+ if spec is None or spec.loader is None:
+ raise ImportError(f"cannot load {path}")
+ module = importlib.util.module_from_spec(spec)
+ spec.loader.exec_module(module)
+ return module
+
+
+def discover_destination_names(exclude: str | None = None) -> list[str]:
+ """The candidate project names, reusing inbox-send's discovery.
+
+ `exclude` drops the current project (matched by exact name or dot-stripped
+ alias) so the engine never recommends routing an item to where it already is.
+ """
+ mod = _load_inbox_send()
+ names = [p.name for p in mod.discover_projects(mod.resolve_roots())]
+ if exclude:
+ drop = _name_variants(exclude)
+ names = [n for n in names if not (_name_variants(n) & drop)]
+ return names
+
+
+def main() -> int:
+ parser = argparse.ArgumentParser(description="Recommend a routing destination for an inbox keeper.")
+ parser.add_argument("--item", required=True, help="the keeper's text")
+ parser.add_argument("--exclude", help="current project to exclude from candidates")
+ args = parser.parse_args()
+
+ projects = discover_destination_names(exclude=args.exclude)
+ destination, confidence = recommend(args.item, projects)
+ print("none" if destination is None else f"{destination}\t{confidence}")
+ return 0
+
+
+if __name__ == "__main__":
+ sys.exit(main())
diff --git a/claude-templates/.ai/scripts/self-inject.sh b/claude-templates/.ai/scripts/self-inject.sh
new file mode 100755
index 0000000..e7340c1
--- /dev/null
+++ b/claude-templates/.ai/scripts/self-inject.sh
@@ -0,0 +1,68 @@
+#!/bin/sh
+# self-inject.sh — type text into the tmux pane running this agent session.
+#
+# The building block for AUTO-FLUSH: an agent checkpoints its session-context,
+# then has tmux type "/clear" and a resume prompt at its own idle prompt, so a
+# session flushes with no human at the keyboard.
+#
+# Usage:
+# self-inject.sh -t %PANE <delay> <text> [<delay2> <text2> ...]
+# self-inject.sh <delay> <text> [...] # derive pane from ancestry
+# self-inject.sh [-t %PANE] # no pairs: report the pane
+#
+# Each pair: sleep <delay> seconds, then type <text> literally and press Enter.
+#
+# TWO HARD-WON GOTCHAS (2026-07-02, archsetup session):
+# 1. A detached child (setsid/nohup/&) of an agent tool call DIES when the
+# tool call ends — the harness cleans up the process group. The arm step
+# must run under the tmux SERVER instead:
+# tmux run-shell -b "self-inject.sh -t %1 25 '/clear' 15 'go — resume...'"
+# 2. Under tmux run-shell the process is a child of the tmux server, so
+# ancestry-based pane detection CANNOT work there. Derive the pane FIRST,
+# synchronously from the agent's own shell (no -t), then pass it
+# explicitly with -t when arming.
+#
+# Collision hazard: if the user happens to be typing when the send fires, the
+# injected text merges into their input line (a real /clear became "/clearto"
+# mid-word). Auto-flush is for sessions running unattended; warn the user to
+# keep hands off for the armed window if they're present.
+
+PANE=""
+if [ "$1" = "-t" ]; then
+ PANE=$2; shift 2
+fi
+
+ppid_of() {
+ # /proc/<pid>/stat: pid (comm) state ppid ... — comm may contain spaces,
+ # so take the 2nd field after the LAST ')'.
+ stat=$(cat "/proc/$1/stat" 2>/dev/null) || return 1
+ # shellcheck disable=SC2086 # word-splitting the stat tail is the point
+ set -- ${stat##*) }
+ echo "$2"
+}
+
+find_pane() {
+ anc=" "
+ pid=$$
+ while [ -n "$pid" ] && [ "$pid" -gt 1 ] 2>/dev/null; do
+ anc="$anc$pid "
+ pid=$(ppid_of "$pid") || break
+ done
+ tmux list-panes -a -F "#{pane_pid} #{pane_id}" 2>/dev/null | \
+ while read -r ppid pane; do
+ case "$anc" in *" $ppid "*) echo "$pane"; break;; esac
+ done
+}
+
+[ -n "$PANE" ] || PANE=$(find_pane)
+[ -n "$PANE" ] || { echo "self-inject: no owning pane found (pass -t %PANE)" >&2; exit 1; }
+
+# With no delay/text pairs, just report the pane (the derive-first step).
+[ $# -ge 2 ] || { echo "$PANE"; exit 0; }
+
+while [ $# -ge 2 ]; do
+ sleep "$1"
+ tmux send-keys -t "$PANE" -l "$2"
+ tmux send-keys -t "$PANE" Enter
+ shift 2
+done
diff --git a/claude-templates/.ai/scripts/session-context-path b/claude-templates/.ai/scripts/session-context-path
index 8cc56f6..670a610 100755
--- a/claude-templates/.ai/scripts/session-context-path
+++ b/claude-templates/.ai/scripts/session-context-path
@@ -10,6 +10,14 @@
# instead of clobbering the singleton. The id is sanitized to filename-safe
# characters so a stray value can't escape the .d/ directory.
#
+# The id must be unique per run; the spawner appends an epoch on the tail
+# (recommended shape host.project.runtime.<epoch>) so a re-run of the same
+# logical agent gets a fresh anchor instead of resolving to a prior run's
+# leftover. The epoch is never minted here: this resolver is called many times
+# per session and must return the same path each call, so it can't generate a
+# new value. See protocols.org "Agent-scoped path". A bare, reused id (just
+# "codex") is the bug that motivated this note.
+#
# Workflows call this to resolve the path; both startup (existence check) and
# wrap-up (rename source) read/write through it. Callers should fall back to
# .ai/session-context.org if this script isn't present yet (older checkouts
diff --git a/claude-templates/.ai/scripts/spec-sort b/claude-templates/.ai/scripts/spec-sort
new file mode 100755
index 0000000..ebfef82
--- /dev/null
+++ b/claude-templates/.ai/scripts/spec-sort
@@ -0,0 +1,715 @@
+#!/usr/bin/env python3
+"""spec-sort — one-time docs-pile retrofit for the docs-lifecycle convention.
+
+Classifies every docs/**/*.org outside docs/specs/ by one predicate: a doc
+carrying BOTH a "Decisions" heading AND an "Implementation phases" heading is
+a spec candidate; everything else is a note. For each candidate it shows an
+evidence panel (Status field, decision/finding cookies, the linking todo.org
+task, recent dated history, cheap existence checks on phase-named artifacts)
+and proposes a lifecycle keyword the evidence supports — conservative
+non-terminal (DRAFT) when inconclusive. The helper proposes; a human confirms
+every move.
+
+Dry-run report is the default. --apply executes under the fail-safe contract:
+
+ - Clean-worktree preflight: refuses on a dirty git tree (exit 2) unless
+ --allow-dirty, which prints exactly what recovery loses.
+ - Every candidate must be addressed with --confirm REL=KEYWORD or
+ --skip REL; terminal keywords (IMPLEMENTED SUPERSEDED CANCELLED) also
+ need --reason REL=TEXT, recorded in the status-history line.
+ - The full move + relink plan is computed and validated first (every
+ destination free, every link resolvable), written to a plan file, and
+ only then executed from that recorded plan.
+ - Bare-path mentions of a moving doc inside the rewritten roots are
+ reported, never rewritten; they block --apply until --acknowledge-bare
+ explicitly waives them.
+ - Mid-apply failure stops the run, names what was and wasn't applied, and
+ prints the git-restore recovery recipe (plus deletion of newly created
+ destination copies, which git restore can't remove).
+ - After a successful apply, a residue scan across the rewritten roots must
+ find no link still resolving to an old path, or spec-sort exits non-zero
+ naming the residue.
+
+Per move: rename to carry the -spec.org suffix, prepend the status heading
+(:ID: UUID + dated history line), rewrite the keyword header to the
+two-sequence form, mirror the keyword into the Metadata Status field, and
+recompute every affected file: link (inbound links to the moved doc AND the
+moved doc's own outbound relative links). Rewritten roots: todo.org,
+.ai/notes.org, docs/**, .ai/project-workflows/, .ai/project-scripts/.
+Reported-never-rewritten: .ai/sessions/ (frozen history) and synced template
+paths (.ai/workflows/, .ai/scripts/, .ai/protocols.org — the report names
+the canonical claude-templates file instead).
+
+Finally stamps :LAST_SPEC_SORT: YYYY-MM-DD in .ai/notes.org's
+* Workflow State section (created idempotently), which permanently clears
+the startup nudge. A run with zero candidates still stamps.
+
+Exit codes: 0 done (or clean report), 1 blocked (confirm gate, validation,
+bare mentions, residue, mid-apply failure), 2 usage / preflight refusal.
+
+Test hook: SPEC_SORT_INJECT_FAIL_AFTER=N aborts the apply after N write
+operations, exercising the recovery path in the bats suite.
+"""
+
+import argparse
+import json
+import os
+import re
+import subprocess
+import sys
+import tempfile
+import uuid
+from datetime import datetime
+
+LIFECYCLE = ("DRAFT", "READY", "DOING", "IMPLEMENTED", "SUPERSEDED", "CANCELLED")
+TERMINAL = {"IMPLEMENTED", "SUPERSEDED", "CANCELLED"}
+TODO_HEADER = [
+ "#+TODO: TODO | DONE",
+ "#+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED",
+]
+
+# Project-owned surfaces whose file: links get rewritten.
+REWRITE_ROOTS = ("todo.org", ".ai/notes.org", "docs", ".ai/project-workflows", ".ai/project-scripts")
+# Frozen or synced surfaces: occurrences are reported, never rewritten.
+REPORT_ROOTS = (".ai/sessions", ".ai/workflows", ".ai/scripts", ".ai/protocols.org")
+# Synced template paths map to their canonical rulesets file for the report.
+SYNCED_PREFIX = (".ai/workflows", ".ai/scripts", ".ai/protocols.org")
+
+LINK_RE = re.compile(r"\[\[file:([^\]\[]+)\](?:\[([^\]\[]*)\])?\]")
+HEADING_RE = re.compile(r"^(\*+)\s+(.*)$")
+COOKIE_RE = re.compile(r"\[\d+/\d+\]")
+DATED_RE = re.compile(r"\b\d{4}-\d{2}-\d{2}\b")
+
+
+def read_text(path):
+ try:
+ with open(path, encoding="utf-8") as f:
+ return f.read()
+ except (UnicodeDecodeError, OSError):
+ return None
+
+
+def heading_text(line):
+ """Heading text with the org keyword and priority cookie stripped."""
+ m = HEADING_RE.match(line)
+ if not m:
+ return None
+ text = re.sub(r"^[A-Z]+\s+", "", m.group(2))
+ text = re.sub(r"^\[#[A-Z]\]\s+", "", text)
+ return text.strip()
+
+
+def has_spine(content):
+ """The classification predicate: Decisions AND Implementation phases."""
+ dec = imp = False
+ for line in content.splitlines():
+ t = heading_text(line)
+ if t is None:
+ continue
+ tl = t.lower()
+ if tl.startswith("decisions"):
+ dec = True
+ elif tl.startswith("implementation phases"):
+ imp = True
+ return dec and imp
+
+
+def walk_files(root, rel_base):
+ """Yield project-relative paths of files under rel_base (file or dir)."""
+ abs_base = os.path.join(root, rel_base)
+ if os.path.isfile(abs_base):
+ yield rel_base
+ return
+ for dirpath, dirs, files in os.walk(abs_base):
+ dirs.sort()
+ for name in sorted(files):
+ yield os.path.relpath(os.path.join(dirpath, name), root)
+
+
+def classify(root):
+ """Split docs/**/*.org outside docs/specs/ into candidates / anomalies / notes."""
+ candidates, anomalies, notes = [], [], []
+ docs = os.path.join(root, "docs")
+ if not os.path.isdir(docs):
+ return candidates, anomalies, notes
+ for rel in walk_files(root, "docs"):
+ if not rel.endswith(".org"):
+ continue
+ parts = rel.split(os.sep)
+ if len(parts) > 1 and parts[1] == "specs":
+ continue
+ content = read_text(os.path.join(root, rel))
+ if content is None:
+ continue
+ if has_spine(content):
+ candidates.append(rel)
+ elif os.path.basename(rel).endswith("-spec.org"):
+ anomalies.append(rel)
+ else:
+ notes.append(rel)
+ return candidates, anomalies, notes
+
+
+def dest_for(rel):
+ base = os.path.basename(rel)
+ if not base.endswith("-spec.org"):
+ base = base[: -len(".org")] + "-spec.org"
+ return os.path.join("docs", "specs", base)
+
+
+# ---- Evidence panel ---------------------------------------------------
+
+
+def todo_task_for(root, rel):
+ """Heading of the first todo.org task whose subtree mentions the doc."""
+ content = read_text(os.path.join(root, "todo.org"))
+ if content is None:
+ return None
+ lines = content.splitlines()
+ basename = os.path.basename(rel)
+ for i, line in enumerate(lines):
+ if basename in line or rel in line:
+ for j in range(i, -1, -1):
+ if HEADING_RE.match(lines[j]):
+ return lines[j].lstrip("* ").strip()
+ return None
+ return None
+
+
+def gather_evidence(root, rel, content):
+ ev = {}
+ m = re.search(r"^\|\s*Status\s*\|\s*([^|]*)\|", content, re.MULTILINE | re.IGNORECASE)
+ ev["status"] = m.group(1).strip() if m else None
+
+ cookies = []
+ for line in content.splitlines():
+ t = heading_text(line)
+ if t and COOKIE_RE.search(t) and (
+ t.lower().startswith("decisions") or t.lower().startswith("review findings")
+ ):
+ cookies.append(t)
+ ev["cookies"] = cookies
+
+ ev["todo"] = todo_task_for(root, rel)
+ kw = None
+ if ev["todo"]:
+ m = re.match(r"([A-Z]+)\s", ev["todo"])
+ kw = m.group(1) if m else None
+ ev["todo_keyword"] = kw
+
+ dated = [ln.strip() for ln in content.splitlines() if DATED_RE.search(ln)]
+ ev["history"] = dated[-1][:100] if dated else None
+
+ # Cheap artifact check: =path= tokens inside the Implementation phases section.
+ artifacts, exists = [], 0
+ section = re.split(r"^\*+\s+.*implementation phases.*$", content, maxsplit=1, flags=re.MULTILINE | re.IGNORECASE)
+ if len(section) > 1:
+ for tok in re.findall(r"=([^=\s]+)=", section[1]):
+ if "/" in tok:
+ artifacts.append(tok)
+ if os.path.exists(os.path.join(root, tok)):
+ exists += 1
+ ev["artifacts"] = (exists, artifacts)
+ return ev
+
+
+def propose_keyword(ev):
+ s = (ev["status"] or "").lower()
+ words = set(re.findall(r"[a-z]+", s))
+ if words & {"implemented", "shipped", "complete", "completed", "done"}:
+ return "IMPLEMENTED"
+ if words & {"superseded"}:
+ return "SUPERSEDED"
+ if words & {"cancelled", "canceled", "dead", "abandoned"}:
+ return "CANCELLED"
+ if words & {"doing", "implementing"} or "in progress" in s or "in-progress" in s:
+ return "DOING"
+ if ev["todo_keyword"] == "DOING":
+ return "DOING"
+ if words & {"ready", "approved", "accepted"}:
+ return "READY"
+ return "DRAFT" # conservative non-terminal default
+
+
+# ---- Link scanning ----------------------------------------------------
+
+
+def rewrite_files(root):
+ """Project-relative *.org files under the rewritten roots."""
+ seen = []
+ for base in REWRITE_ROOTS:
+ if not os.path.exists(os.path.join(root, base)):
+ continue
+ for rel in walk_files(root, base):
+ if rel.endswith(".org") and rel not in seen:
+ seen.append(rel)
+ return seen
+
+
+def resolve_target(root, linker_rel, raw_target, moved):
+ """Resolve a file: link target to a project-relative path (org semantics
+ first — relative to the linking file's directory — then project-root
+ anchoring as a fallback for root-anchored links)."""
+ if raw_target.startswith(("/", "~", "http:", "https:")):
+ return None
+ rel_a = os.path.normpath(os.path.join(os.path.dirname(linker_rel), raw_target))
+ if rel_a in moved or os.path.exists(os.path.join(root, rel_a)):
+ return rel_a
+ rel_b = os.path.normpath(raw_target)
+ if rel_b in moved or os.path.exists(os.path.join(root, rel_b)):
+ return rel_b
+ return rel_a
+
+
+def plan_link_edits(root, moved):
+ """Compute every link rewrite: inbound links to moved docs and moved
+ docs' own outbound relative links. Returns ({linker_rel: [(old, new)]},
+ [ambiguity descriptions]) — a link whose file-relative and root-anchored
+ readings are both live and disagree about a moving doc blocks validation
+ rather than being rewritten against a guess."""
+ edits = {}
+ ambiguous = []
+ for linker in rewrite_files(root):
+ content = read_text(os.path.join(root, linker))
+ if content is None:
+ continue
+ linker_post = moved.get(linker, linker)
+ for m in LINK_RE.finditer(content):
+ raw = m.group(1)
+ desc = m.group(2)
+ target_path, sep, anchor = raw.partition("::")
+ target = resolve_target(root, linker, target_path, moved)
+ if target is None:
+ continue
+ rel_a = os.path.normpath(os.path.join(os.path.dirname(linker), target_path))
+ rel_b = os.path.normpath(target_path)
+ if rel_a != rel_b:
+ live_a = rel_a in moved or os.path.exists(os.path.join(root, rel_a))
+ live_b = rel_b in moved or os.path.exists(os.path.join(root, rel_b))
+ if live_a and live_b and (rel_a in moved or rel_b in moved):
+ ambiguous.append(
+ "%s: [[file:%s]] reads as %s (file-relative) or %s (root-anchored) "
+ "and a moving doc is involved — resolve the link by hand" % (linker, raw, rel_a, rel_b))
+ continue
+ if target not in moved and linker not in moved:
+ continue
+ if target not in moved and not os.path.exists(os.path.join(root, target)):
+ continue # already broken before this run; not ours to guess
+ target_post = moved.get(target, target)
+ new_path = os.path.relpath(target_post, os.path.dirname(linker_post) or ".")
+ new_raw = new_path + (sep + anchor if sep else "")
+ if new_raw == raw:
+ continue
+ new_link = "[[file:%s]%s]" % (new_raw, "[%s]" % desc if desc is not None else "")
+ if m.group(0) != new_link:
+ edits.setdefault(linker, []).append((m.group(0), new_link))
+ return edits, ambiguous
+
+
+def scan_bare_mentions(root, moved):
+ """Bare-path mentions of moving docs in the rewritten roots — text
+ occurrences outside any [[...]] link. Reported, never rewritten."""
+ found = []
+ for base in REWRITE_ROOTS:
+ if not os.path.exists(os.path.join(root, base)):
+ continue
+ for rel in walk_files(root, base):
+ content = read_text(os.path.join(root, rel))
+ if content is None:
+ continue
+ for i, line in enumerate(content.splitlines(), 1):
+ stripped = re.sub(r"\[\[[^\]]*\](?:\[[^\]]*\])?\]", "", line)
+ for src in moved:
+ if src in stripped:
+ found.append((rel, i, src))
+ return found
+
+
+def scan_report_only(root, moved):
+ """Occurrences of moving docs in frozen/synced surfaces."""
+ reports = []
+ for base in REPORT_ROOTS:
+ if not os.path.exists(os.path.join(root, base)):
+ continue
+ for rel in walk_files(root, base):
+ content = read_text(os.path.join(root, rel))
+ if content is None:
+ continue
+ for src in moved:
+ if src in content:
+ if rel.startswith(SYNCED_PREFIX):
+ note = ("synced template, not rewritten — a local edit is reverted by the "
+ "next sync; edit the canonical claude-templates/%s instead" % rel)
+ else:
+ note = "frozen history; not rewritten"
+ reports.append((rel, src, note))
+ return reports
+
+
+# ---- Content transforms -----------------------------------------------
+
+
+def transform_spec(content, keyword, reason, title, doc_id, link_edits):
+ """Apply the retrofit rewrite to a moving spec's content: two-sequence
+ keyword header, prepended status heading, Status-field mirror, and the
+ doc's own link edits."""
+ for old, new in link_edits:
+ content = content.replace(old, new)
+ lines = content.splitlines()
+
+ todo_idx = None
+ kept = []
+ for line in lines:
+ if line.startswith("#+TODO:"):
+ if todo_idx is None:
+ todo_idx = len(kept)
+ continue
+ kept.append(line)
+ lines = kept
+ if todo_idx is None:
+ todo_idx = 0
+ while todo_idx < len(lines) and lines[todo_idx].startswith("#+"):
+ todo_idx += 1
+ lines[todo_idx:todo_idx] = TODO_HEADER
+
+ head_end = 0
+ while head_end < len(lines) and (lines[head_end].startswith("#+") or not lines[head_end].strip()):
+ head_end += 1
+ ts = datetime.now().astimezone().strftime("%Y-%m-%d %a @ %H:%M:%S %z")
+ provenance = "reason: %s" % reason if reason else "evidence-based, human-confirmed"
+ block = [
+ "* %s %s" % (keyword, title),
+ ":PROPERTIES:",
+ ":ID: %s" % doc_id,
+ ":END:",
+ "- %s — retrofitted by spec-sort; status set to %s (%s)" % (ts, keyword, provenance),
+ "",
+ ]
+ lines[head_end:head_end] = block
+
+ out = []
+ mirrored = False
+ for line in lines:
+ m = re.match(r"^(\|\s*Status\s*\|)([^|]*)(\|.*)$", line, re.IGNORECASE)
+ if m and not mirrored:
+ value = " %s" % keyword.lower()
+ width = len(m.group(2))
+ line = m.group(1) + (value.ljust(width) if len(value) <= width else value + " ") + m.group(3)
+ mirrored = True
+ out.append(line)
+ return "\n".join(out) + "\n"
+
+
+def title_for(content, rel):
+ m = re.search(r"^#\+TITLE:\s*(.+)$", content, re.MULTILINE | re.IGNORECASE)
+ if m:
+ return m.group(1).strip()
+ base = os.path.basename(rel)[: -len(".org")]
+ return base[: -len("-spec")] if base.endswith("-spec") else base
+
+
+# ---- Marker ------------------------------------------------------------
+
+
+def stamp_marker(root, date):
+ path = os.path.join(root, ".ai", "notes.org")
+ os.makedirs(os.path.dirname(path), exist_ok=True)
+ content = read_text(path) or ""
+ line = ":LAST_SPEC_SORT: %s" % date
+ if ":LAST_SPEC_SORT:" in content:
+ content = re.sub(r":LAST_SPEC_SORT:.*", line, content, count=1)
+ elif re.search(r"^\* Workflow State\s*$", content, re.MULTILINE):
+ content = re.sub(r"(^\* Workflow State\s*$)", r"\1\n" + line, content, count=1, flags=re.MULTILINE)
+ else:
+ if content and not content.endswith("\n"):
+ content += "\n"
+ content += "\n* Workflow State\n\n%s\n" % line
+ with open(path, "w", encoding="utf-8") as f:
+ f.write(content)
+
+
+# ---- Apply -------------------------------------------------------------
+
+
+class ApplyFailure(Exception):
+ """Mid-apply failure: args are (applied_labels, remaining_ops, cause)."""
+
+
+def apply_plan(root, plan, fail_after):
+ """Execute the recorded plan. Returns the applied-op labels; raises
+ ApplyFailure mid-way on a write error or when the test hook fires."""
+ ops = []
+ for mv in plan["moves"]:
+ ops.append(("move", mv))
+ for linker, edits in plan["link_edits"].items():
+ if linker in {mv["src"] for mv in plan["moves"]}:
+ continue # a moving doc's own edits ride along in its transform
+ ops.append(("relink", (linker, edits)))
+
+ applied = []
+ specs_dir = os.path.join(root, "docs", "specs")
+ if plan["moves"] and not os.path.isdir(specs_dir):
+ os.makedirs(specs_dir)
+ plan["created_dirs"].append(os.path.join("docs", "specs"))
+
+ for n, (kind, payload) in enumerate(ops, 1):
+ if fail_after and n > fail_after:
+ raise ApplyFailure(applied, ops[n - 1:], "injected test failure")
+ try:
+ if kind == "move":
+ mv = payload
+ content = read_text(os.path.join(root, mv["src"]))
+ new = transform_spec(content, mv["keyword"], mv["reason"], mv["title"], mv["id"],
+ plan["link_edits"].get(mv["src"], []))
+ with open(os.path.join(root, mv["dest"]), "w", encoding="utf-8") as f:
+ f.write(new)
+ os.remove(os.path.join(root, mv["src"]))
+ applied.append("move %s -> %s" % (mv["src"], mv["dest"]))
+ else:
+ linker, edits = payload
+ path = os.path.join(root, linker)
+ content = read_text(path)
+ for old, new in edits:
+ content = content.replace(old, new)
+ with open(path, "w", encoding="utf-8") as f:
+ f.write(content)
+ applied.append("relink %s (%d link%s)" % (linker, len(edits), "s" if len(edits) != 1 else ""))
+ except OSError as exc:
+ raise ApplyFailure(applied, ops[n - 1:], str(exc))
+ return applied
+
+
+def residue_check(root, plan):
+ """Post-apply: no link in the rewritten roots may still resolve to an
+ old path; bare mentions beyond the acknowledged set fail too."""
+ moved = {mv["src"]: mv["dest"] for mv in plan["moves"]}
+ residue = []
+ for linker in rewrite_files(root):
+ content = read_text(os.path.join(root, linker))
+ if content is None:
+ continue
+ for m in LINK_RE.finditer(content):
+ target_path = m.group(1).partition("::")[0]
+ target = resolve_target(root, linker, target_path, {})
+ if target in moved:
+ residue.append("%s: link still resolves to %s" % (linker, target))
+ # Acknowledged mentions were recorded pre-apply; a mention inside a moved
+ # doc now lives at the doc's destination, so map the file side through the
+ # moves before comparing.
+ acknowledged = {(moved.get(f, f), src) for f, _ln, src in plan["bare"]}
+ for f, ln, src in scan_bare_mentions(root, moved):
+ if (f, src) not in acknowledged:
+ residue.append("%s:%d: bare mention of %s" % (f, ln, src))
+ return residue
+
+
+def print_recovery(plan, applied, not_applied):
+ print("FAILURE — the apply did not complete.")
+ print(" applied:")
+ for a in applied or ["(nothing)"]:
+ print(" %s" % a)
+ print(" not applied:")
+ for kind, payload in not_applied:
+ if kind == "move":
+ print(" move %s -> %s" % (payload["src"], payload["dest"]))
+ else:
+ print(" relink %s" % payload[0])
+ print("RECOVERY — restore the pre-run state (safe: preflight required a clean tree):")
+ touched = [mv["src"] for mv in plan["moves"]] + [l for l in plan["link_edits"] if l not in {mv["src"] for mv in plan["moves"]}]
+ print(" git restore -- %s" % " ".join(touched))
+ created = [mv["dest"] for mv in plan["moves"]]
+ print(" rm -f -- %s # git restore can't remove the created copies" % " ".join(created))
+ for d in plan.get("created_dirs", []):
+ print(" rmdir --ignore-fail-on-non-empty -- %s" % d)
+
+
+# ---- Main ---------------------------------------------------------------
+
+
+def parse_kv(pairs, label):
+ out = {}
+ for item in pairs or []:
+ if "=" not in item:
+ sys.exit("spec-sort: %s expects REL=VALUE, got %r" % (label, item))
+ k, v = item.split("=", 1)
+ out[os.path.normpath(k)] = v
+ return out
+
+
+def main():
+ ap = argparse.ArgumentParser(prog="spec-sort", add_help=True)
+ ap.add_argument("--project-root", default=".")
+ ap.add_argument("--apply", action="store_true")
+ ap.add_argument("--allow-dirty", action="store_true")
+ ap.add_argument("--acknowledge-bare", action="store_true")
+ ap.add_argument("--confirm", action="append", metavar="REL=KEYWORD")
+ ap.add_argument("--reason", action="append", metavar="REL=TEXT")
+ ap.add_argument("--skip", action="append", metavar="REL")
+ ap.add_argument("--plan-file")
+ args = ap.parse_args()
+
+ root = os.path.abspath(args.project_root)
+ confirms = parse_kv(args.confirm, "--confirm")
+ reasons = parse_kv(args.reason, "--reason")
+ skips = {os.path.normpath(s) for s in (args.skip or [])}
+
+ candidates, anomalies, notes = classify(root)
+ if not candidates and not anomalies and not notes and not os.path.isdir(os.path.join(root, "docs")):
+ return 0 # no docs pile at all — silent no-op
+
+ for named in list(confirms) + list(skips) + list(reasons):
+ if named not in candidates:
+ print("spec-sort: %s is not a spec candidate" % named)
+ return 1
+ for rel, kw in confirms.items():
+ if kw not in LIFECYCLE:
+ print("spec-sort: %r is not a lifecycle keyword (%s)" % (kw, " ".join(LIFECYCLE)))
+ return 1
+
+ # ---- Build the plan (shared by report and apply) ----
+ moves = []
+ for rel in candidates:
+ if rel in skips:
+ continue
+ if args.apply and rel not in confirms:
+ continue # gate failure reported below
+ content = read_text(os.path.join(root, rel))
+ moves.append({
+ "src": rel,
+ "dest": dest_for(rel),
+ "keyword": confirms.get(rel, None),
+ "reason": reasons.get(rel),
+ "title": title_for(content, rel),
+ "id": str(uuid.uuid4()),
+ })
+ moved_map = {mv["src"]: mv["dest"] for mv in moves}
+ link_edits, ambiguous = plan_link_edits(root, moved_map)
+ bare = scan_bare_mentions(root, moved_map)
+ reports = scan_report_only(root, moved_map)
+
+ # ---- Report ----
+ for rel in candidates:
+ content = read_text(os.path.join(root, rel))
+ ev = gather_evidence(root, rel, content)
+ proposed = propose_keyword(ev)
+ print("CANDIDATE %s -> %s" % (rel, dest_for(rel)))
+ suffix = " (terminal — requires --reason to apply)" if proposed in TERMINAL else ""
+ print(" proposed keyword: %s%s" % (proposed, suffix))
+ print(" evidence:")
+ print(" status field: %s" % (ev["status"] or "(none)"))
+ print(" cookies: %s" % ("; ".join(ev["cookies"]) or "(none)"))
+ print(" todo.org: %s" % (ev["todo"] or "(no linking task)"))
+ print(" history: %s" % (ev["history"] or "(none)"))
+ n_exist, artifacts = ev["artifacts"]
+ if artifacts:
+ print(" artifacts: %d/%d named paths exist (%s)" % (n_exist, len(artifacts), ", ".join(artifacts)))
+ else:
+ print(" artifacts: (none named)")
+ for rel in anomalies:
+ print("ANOMALY %s: named -spec.org but lacks the spec spine (Decisions + Implementation phases); surfaced, not moved" % rel)
+ for rel in notes:
+ print("NOTE %s" % rel)
+ for linker, edits in sorted(link_edits.items()):
+ for old, new in edits:
+ print("RELINK %s: %s -> %s" % (linker, old, new))
+ for a in ambiguous:
+ print("AMBIGUOUS %s" % a)
+ for f, ln, src in bare:
+ print("BARE-PATH %s:%d: %s (reported for manual handling, never rewritten)" % (f, ln, src))
+ for rel, src, note in reports:
+ print("REPORT %s: reference to %s (%s)" % (rel, src, note))
+
+ if not args.apply:
+ if candidates or anomalies or notes:
+ print("DRY RUN — no changes written. Pass --apply with per-candidate --confirm/--skip to execute.")
+ return 0
+
+ # ---- Apply: preflight ----
+ try:
+ porcelain = subprocess.run(
+ ["git", "status", "--porcelain"], cwd=root,
+ capture_output=True, text=True, check=True,
+ ).stdout
+ except (subprocess.CalledProcessError, FileNotFoundError):
+ print("spec-sort: --apply needs a git worktree (recovery depends on git restore)")
+ return 2
+ if porcelain.strip():
+ dirty = [ln[3:] for ln in porcelain.splitlines()]
+ if not args.allow_dirty:
+ print("spec-sort: refusing --apply on a dirty worktree (%d path%s). Commit or stash first, or pass --allow-dirty."
+ % (len(dirty), "s" if len(dirty) != 1 else ""))
+ return 2
+ print("WARNING --allow-dirty: recovery via git restore would also revert your pre-existing uncommitted changes:")
+ for p in dirty:
+ print(" %s" % p)
+
+ # ---- Apply: confirm gate ----
+ unaddressed = [rel for rel in candidates if rel not in confirms and rel not in skips]
+ if unaddressed:
+ print("spec-sort: unconfirmed candidate(s) — pass --confirm REL=KEYWORD or --skip REL for each:")
+ for rel in unaddressed:
+ print(" %s" % rel)
+ return 1
+ for mv in moves:
+ if mv["keyword"] in TERMINAL and not mv["reason"]:
+ print("spec-sort: %s -> %s is a terminal state and requires an explicit --reason %s=TEXT"
+ % (mv["src"], mv["keyword"], mv["src"]))
+ return 1
+
+ # ---- Apply: validation ----
+ problems = []
+ dests = {}
+ for mv in moves:
+ if os.path.exists(os.path.join(root, mv["dest"])):
+ problems.append("%s: destination exists (%s)" % (mv["src"], mv["dest"]))
+ if mv["dest"] in dests:
+ problems.append("%s and %s: destination exists twice (%s)" % (mv["src"], dests[mv["dest"]], mv["dest"]))
+ dests[mv["dest"]] = mv["src"]
+ for a in ambiguous:
+ problems.append("ambiguous link: %s" % a)
+ if bare and not args.acknowledge_bare:
+ problems.append("bare-path mention(s) listed above need manual handling — re-run with --acknowledge-bare to proceed without rewriting them")
+ if problems:
+ print("spec-sort: validation blocked — nothing written:")
+ for p in problems:
+ print(" %s" % p)
+ return 1
+
+ # ---- Apply: record the plan, then execute from it ----
+ today = datetime.now().astimezone().strftime("%Y-%m-%d")
+ plan = {
+ "root": root, "date": today, "moves": moves,
+ "link_edits": link_edits, "bare": bare,
+ "reports": [list(r) for r in reports], "created_dirs": [],
+ }
+ plan_path = args.plan_file or os.path.join(
+ tempfile.gettempdir(), "spec-sort-plan-%s.json" % os.path.basename(root))
+ with open(plan_path, "w", encoding="utf-8") as f:
+ json.dump(plan, f, indent=2)
+ print("plan written: %s" % plan_path)
+
+ fail_after = int(os.environ.get("SPEC_SORT_INJECT_FAIL_AFTER", "0") or 0)
+ try:
+ applied = apply_plan(root, plan, fail_after)
+ except ApplyFailure as exc:
+ print("write failed: %s" % exc.args[2])
+ print_recovery(plan, exc.args[0], exc.args[1])
+ return 1
+
+ residue = residue_check(root, plan)
+ if residue:
+ print("spec-sort: residue after apply — old paths still referenced:")
+ for r in residue:
+ print(" %s" % r)
+ print_recovery(plan, applied, [])
+ return 1
+
+ stamp_marker(root, today)
+ for a in applied:
+ print("applied: %s" % a)
+ print("spec-sort: done — %d spec(s) sorted, :LAST_SPEC_SORT: %s stamped" % (len(moves), today))
+ return 0
+
+
+if __name__ == "__main__":
+ sys.exit(main())
diff --git a/claude-templates/.ai/scripts/task-review-staleness.sh b/claude-templates/.ai/scripts/task-review-staleness.sh
index ed43712..50e0257 100755
--- a/claude-templates/.ai/scripts/task-review-staleness.sh
+++ b/claude-templates/.ai/scripts/task-review-staleness.sh
@@ -19,9 +19,14 @@
# deeper headings, and cookie-less headings are not review units.
#
# A task is stale (count mode), and sorts oldest (list mode), when its
-# :LAST_REVIEWED: property is missing or unparseable (NIL sorts first), or
-# when its age strictly exceeds the threshold (age > N days; age == N is
-# still fresh).
+# :LAST_REVIEWED: property is missing (NIL sorts first) or when its age
+# strictly exceeds the threshold (age > N days; age == N is still fresh).
+#
+# :LAST_REVIEWED: accepts a bare date (2026-07-09) or an org-native
+# timestamp ([2026-07-09 Thu] or <2026-07-09 Thu ...>); both normalize to
+# the ISO date. A value that is present but parses to neither is a data
+# error: the script warns loudly to stderr (file:line:value) and leaves it
+# out of the stale count rather than silently treating it as never-reviewed.
set -euo pipefail
@@ -46,7 +51,7 @@ num="$2"
# only the property drawer between a qualifying heading and the next heading
# is scanned.
extract_tasks() {
- awk '
+ awk -v fname="$todo_file" '
function flush() {
if (in_task) printf "%s\t%s\t%s\n", hline, (have_lr ? lr : "NONE"), heading
}
@@ -65,7 +70,21 @@ extract_tasks() {
v = $0
sub(/^[ \t]*:LAST_REVIEWED:[ \t]*/, "", v)
sub(/[ \t]*$/, "", v)
- lr = v; have_lr = 1
+ raw = v
+ # Accept org-native stamps: strip a leading [ or < (inactive/active
+ # timestamp bracket) and take the leading ISO date, so [2026-07-09 Thu]
+ # and 2026-07-09 both normalize to 2026-07-09.
+ sub(/^[[<]/, "", v)
+ if (match(v, /^[0-9][0-9][0-9][0-9]-[0-9][0-9]-[0-9][0-9]/)) {
+ lr = substr(v, RSTART, RLENGTH); have_lr = 1
+ } else {
+ # Present but unparseable — a data error, not "never reviewed". Warn
+ # loudly (file:line:value) instead of silently folding it into the
+ # stale count, where a re-review in the same bad format would never
+ # drop the number and nothing would explain why.
+ printf "task-review-staleness: %s:%d: unparseable :LAST_REVIEWED: %s (expected YYYY-MM-DD or [YYYY-MM-DD Day])\n", fname, NR, raw > "/dev/stderr"
+ lr = "INVALID"; have_lr = 1
+ }
next
}
END { flush() }
@@ -98,9 +117,16 @@ while IFS=$'\t' read -r hline value heading; do
continue
fi
- # Unparseable date → treat as NIL (stale).
+ # Malformed stamp: extract_tasks already warned. A data error is not
+ # "never reviewed", so it stays out of the stale count.
+ if [ "$value" = "INVALID" ]; then
+ continue
+ fi
+
+ # A shape-valid but impossible date (e.g. 2026-13-45) slips past the awk
+ # regex; warn and skip rather than silently counting it.
if ! rev_epoch=$(date -d "$value" +%s 2>/dev/null); then
- count=$((count + 1))
+ echo "task-review-staleness: $todo_file: unparseable :LAST_REVIEWED: $value" >&2
continue
fi
diff --git a/claude-templates/.ai/scripts/tests/agent-lock.bats b/claude-templates/.ai/scripts/tests/agent-lock.bats
new file mode 100644
index 0000000..dbcffe1
--- /dev/null
+++ b/claude-templates/.ai/scripts/tests/agent-lock.bats
@@ -0,0 +1,214 @@
+#!/usr/bin/env bats
+#
+# Tests for claude-templates/.ai/scripts/agent-lock — a mkdir-atomic advisory
+# lock helper for agent workflows (sentry's single-runner and roam-write
+# locks). flock can't span an agent's tool calls: every Bash call is its own
+# short-lived shell, so a flock dies with the call that took it. This helper
+# persists the lock on disk between calls and self-clears after a crash via
+# age-based staleness reclaim.
+#
+# Contract under test:
+# agent-lock acquire <name> [--ttl=SECONDS] [--wait[=SECONDS]]
+# exit 0 → acquired (fresh, or reclaimed from a stale prior holder).
+# exit 1 → busy: a live lock holds <name>; deferred (note on stderr).
+# exit 2 → usage error (bad/absent name, unknown subcommand).
+# agent-lock refresh <name> → re-touch a held lock (heartbeat); exit 1 if absent.
+# agent-lock release <name> → remove the lock; idempotent (exit 0 if already free).
+# agent-lock status <name> → print free|held|stale + metadata; exit 0 (query).
+# agent-lock path <name> → print the resolved lock dir path; does not create it.
+#
+# Staleness is age-based on the metadata file's mtime versus the lock's own
+# recorded TTL, so a crashed holder's lock expires instead of wedging every
+# later acquire. Heartbeat (refresh) re-touches the mtime, keeping a live
+# holder's lock young. Every reclaim surfaces a note (never silent).
+#
+# Lock home: /run/user/<uid>/agent-locks/<name>/ (tmpfs: host-local, out of
+# every repo, cleared on reboot), with ~/.cache/agent-locks/ as the fallback
+# where no runtime dir exists. AGENT_LOCK_DIR overrides the base for tests and
+# advanced callers; the helper otherwise owns the path scheme and callers pass
+# only names.
+#
+# Strategy: AGENT_LOCK_DIR points every lock at a temp base, so tests never
+# touch a real runtime dir. Staleness is exercised by aging the metadata
+# file's mtime with `touch` rather than sleeping.
+
+SCRIPT="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/agent-lock"
+BASH_BIN="$(command -v bash)"
+
+setup() {
+ TEST_DIR="$(mktemp -d -t agent-lock-bats.XXXXXX)"
+ LOCK_BASE="$TEST_DIR/locks"
+}
+
+teardown() {
+ rm -rf "$TEST_DIR"
+}
+
+lock() {
+ run env AGENT_LOCK_DIR="$LOCK_BASE" "$BASH_BIN" "$SCRIPT" "$@"
+}
+
+# meta-file path for a lock name, for direct inspection / aging.
+meta_of() {
+ printf '%s/%s/meta\n' "$LOCK_BASE" "$1"
+}
+
+# ---- acquire: fresh win + metadata --------------------------------------
+
+@test "acquire: fresh name wins (exit 0) and writes pid/host/timestamp/ttl" {
+ lock acquire job
+ [ "$status" -eq 0 ]
+ local meta; meta="$(meta_of job)"
+ [ -f "$meta" ]
+ grep -q "^pid=$$\|^pid=[0-9][0-9]*$" "$meta"
+ grep -q "^host=$(uname -n)$" "$meta"
+ grep -qE "^acquired=[0-9]{4}-[0-9]{2}-[0-9]{2}T" "$meta"
+ grep -qE "^ttl=[0-9]+$" "$meta"
+}
+
+@test "acquire: honors an explicit --ttl in the metadata" {
+ lock acquire job --ttl=45
+ [ "$status" -eq 0 ]
+ grep -q "^ttl=45$" "$(meta_of job)"
+}
+
+# ---- acquire: contention (one winner) -----------------------------------
+
+@test "acquire: a second acquire of a live lock defers (exit 1, note)" {
+ lock acquire job
+ [ "$status" -eq 0 ]
+ lock acquire job
+ [ "$status" -eq 1 ]
+ [[ "$output" == *job* ]]
+}
+
+@test "acquire: two racing acquires yield exactly one winner" {
+ # Fire both without releasing; exactly one mkdir wins.
+ env AGENT_LOCK_DIR="$LOCK_BASE" "$BASH_BIN" "$SCRIPT" acquire race & p1=$!
+ env AGENT_LOCK_DIR="$LOCK_BASE" "$BASH_BIN" "$SCRIPT" acquire race & p2=$!
+ local r1=0 r2=0
+ wait $p1 || r1=$?
+ wait $p2 || r2=$?
+ # One exits 0 (won), one exits 1 (deferred).
+ [ "$((r1 + r2))" -eq 1 ]
+}
+
+# ---- release: frees the lock --------------------------------------------
+
+@test "release: frees a held lock so the next acquire wins" {
+ lock acquire job
+ [ "$status" -eq 0 ]
+ lock release job
+ [ "$status" -eq 0 ]
+ [ ! -d "$LOCK_BASE/job" ]
+ lock acquire job
+ [ "$status" -eq 0 ]
+}
+
+@test "release: is idempotent on an already-free lock (exit 0)" {
+ lock release never-held
+ [ "$status" -eq 0 ]
+}
+
+# ---- staleness reclaim (surfaced, never silent) -------------------------
+
+@test "acquire: reclaims a stale lock and surfaces the reclaim note" {
+ lock acquire job --ttl=1
+ [ "$status" -eq 0 ]
+ # Age the metadata mtime well past the 1s TTL.
+ touch -d '1 hour ago' "$(meta_of job)"
+ lock acquire job --ttl=1
+ [ "$status" -eq 0 ]
+ [[ "$output" == *reclaim* ]]
+ [[ "$output" == *job* ]]
+ # The reclaim installed fresh metadata (young again), not the aged holder's.
+ lock status job
+ [[ "$output" == *held* ]]
+ [[ "$output" != *stale* ]]
+}
+
+@test "acquire: a lock inside its TTL is not stale (stays deferred)" {
+ lock acquire job --ttl=3600
+ [ "$status" -eq 0 ]
+ lock acquire job --ttl=3600
+ [ "$status" -eq 1 ]
+}
+
+# ---- heartbeat (refresh keeps a live lock young) ------------------------
+
+@test "refresh: re-touches a held lock so it is no longer stale" {
+ lock acquire job --ttl=1
+ [ "$status" -eq 0 ]
+ touch -d '1 hour ago' "$(meta_of job)"
+ lock status job
+ [[ "$output" == *stale* ]]
+ lock refresh job
+ [ "$status" -eq 0 ]
+ lock status job
+ [[ "$output" == *held* ]]
+ [[ "$output" != *stale* ]]
+}
+
+@test "refresh: an absent lock cannot be refreshed (exit 1)" {
+ lock refresh nothing
+ [ "$status" -eq 1 ]
+}
+
+# ---- status query -------------------------------------------------------
+
+@test "status: reports free for an unheld lock (exit 0)" {
+ lock status job
+ [ "$status" -eq 0 ]
+ [[ "$output" == *free* ]]
+}
+
+@test "status: reports held with metadata for a live lock" {
+ lock acquire job --ttl=3600
+ lock status job
+ [ "$status" -eq 0 ]
+ [[ "$output" == *held* ]]
+ [[ "$output" == *"host=$(uname -n)"* ]]
+}
+
+# ---- path resolution: runtime dir home with cache fallback --------------
+
+@test "path: resolves under AGENT_LOCK_DIR when set" {
+ lock path job
+ [ "$status" -eq 0 ]
+ [ "$output" = "$LOCK_BASE/job" ]
+ [ ! -d "$LOCK_BASE/job" ] # path does not create the lock
+}
+
+@test "path: prefers the runtime dir home when no override is set" {
+ local rt="$TEST_DIR/run"
+ mkdir -p "$rt"
+ run env -u AGENT_LOCK_DIR XDG_RUNTIME_DIR="$rt" "$BASH_BIN" "$SCRIPT" path job
+ [ "$status" -eq 0 ]
+ [ "$output" = "$rt/agent-locks/job" ]
+}
+
+@test "path: falls back to the cache home when no runtime dir exists" {
+ local home="$TEST_DIR/home"
+ mkdir -p "$home"
+ run env -u AGENT_LOCK_DIR -u XDG_RUNTIME_DIR -u XDG_CACHE_HOME \
+ HOME="$home" "$BASH_BIN" "$SCRIPT" path job
+ [ "$status" -eq 0 ]
+ [ "$output" = "$home/.cache/agent-locks/job" ]
+}
+
+# ---- usage errors -------------------------------------------------------
+
+@test "usage: a missing name is a usage error (exit 2)" {
+ lock acquire
+ [ "$status" -eq 2 ]
+}
+
+@test "usage: a name with a slash is rejected (exit 2)" {
+ lock acquire bad/name
+ [ "$status" -eq 2 ]
+}
+
+@test "usage: an unknown subcommand is a usage error (exit 2)" {
+ lock frobnicate job
+ [ "$status" -eq 2 ]
+}
diff --git a/claude-templates/.ai/scripts/tests/agent-roster.bats b/claude-templates/.ai/scripts/tests/agent-roster.bats
new file mode 100644
index 0000000..939a7df
--- /dev/null
+++ b/claude-templates/.ai/scripts/tests/agent-roster.bats
@@ -0,0 +1,141 @@
+#!/usr/bin/env bats
+# Tests for agent-roster: report other live Claude agents in a project.
+#
+# pgrep and /proc are the system boundary, so the test injects both and runs
+# the real include/exclude logic against fixtures — no Claude processes are
+# spawned. Injection points:
+# ROSTER_PGREP command standing in for pgrep (a stub printing $FAKE_PIDS)
+# ROSTER_PROC proc dir (a fixture of <pid>/cwd symlinks + <pid>/status)
+# ROSTER_SELF_PID the scanner's own pid, so the ancestry walk is testable
+#
+# Fixture process tree: pid 1000 (the scanner) is a child of 999 (the current
+# session's claude), which is a child of init (1). So 999 must always be
+# excluded as scanner ancestry; other claude pids are judged by cwd.
+
+setup() {
+ SCRIPT_DIR="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)"
+ ROSTER="$SCRIPT_DIR/agent-roster"
+
+ PROC="$BATS_TEST_TMPDIR/proc"
+ ROOT="$BATS_TEST_TMPDIR/project"
+ mkdir -p "$PROC" "$ROOT/sub" "$BATS_TEST_TMPDIR/elsewhere"
+
+ PGREP_STUB="$BATS_TEST_TMPDIR/pgrep"
+ cat >"$PGREP_STUB" <<'EOF'
+#!/usr/bin/env bash
+printf '%s\n' $FAKE_PIDS
+EOF
+ chmod +x "$PGREP_STUB"
+
+ SELF=1000
+ # scanner (1000) <- session claude (999) <- init (1)
+ mkproc 1000 "$ROOT" 999
+ mkproc 999 "$ROOT" 1
+}
+
+# mkproc PID CWD PPID — register a fake process in the fixture proc dir.
+mkproc() {
+ mkdir -p "$PROC/$1"
+ ln -sf "$2" "$PROC/$1/cwd"
+ printf 'PPid:\t%s\n' "$3" >"$PROC/$1/status"
+}
+
+run_roster() {
+ ROSTER_PGREP="$PGREP_STUB" ROSTER_PROC="$PROC" ROSTER_SELF_PID="$SELF" \
+ FAKE_PIDS="$FAKE_PIDS" run "$ROSTER" "$ROOT"
+}
+
+@test "agent-roster: alone (only the session's own claude) exits 0, no output" {
+ FAKE_PIDS="999"
+ run_roster
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "agent-roster: one other agent in-project is printed, exit 1" {
+ mkproc 2000 "$ROOT" 1
+ FAKE_PIDS="999 2000"
+ run_roster
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"2000"* ]]
+ [[ "$output" == *"$ROOT"* ]]
+}
+
+@test "agent-roster: two other agents both printed, exit 1" {
+ mkproc 2000 "$ROOT" 1
+ mkproc 2001 "$ROOT/sub" 1
+ FAKE_PIDS="999 2000 2001"
+ run_roster
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"2000"* ]]
+ [[ "$output" == *"2001"* ]]
+ [ "${#lines[@]}" -eq 2 ]
+}
+
+@test "agent-roster: the scanner's session-claude ancestor is excluded even with matching cwd" {
+ # 999 has cwd == ROOT but is scanner ancestry; must not appear.
+ FAKE_PIDS="999"
+ run_roster
+ [ "$status" -eq 0 ]
+ [[ "$output" != *"999"* ]]
+}
+
+@test "agent-roster: cwd outside the project root is excluded" {
+ mkproc 3000 "$BATS_TEST_TMPDIR/elsewhere" 1
+ FAKE_PIDS="999 3000"
+ run_roster
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "agent-roster: cwd in a subdirectory of root is included" {
+ mkproc 2002 "$ROOT/sub" 1
+ FAKE_PIDS="999 2002"
+ run_roster
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"2002"* ]]
+}
+
+@test "agent-roster: a sibling path sharing a prefix is not a false match" {
+ # ROOT is .../project; .../project-other must not count as inside it.
+ mkdir -p "$BATS_TEST_TMPDIR/project-other"
+ mkproc 3100 "$BATS_TEST_TMPDIR/project-other" 1
+ FAKE_PIDS="999 3100"
+ run_roster
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "agent-roster: a pid that vanished between pgrep and the proc read is skipped" {
+ # 4000 has no fixture dir, simulating a process gone by readlink time.
+ FAKE_PIDS="999 4000"
+ run_roster
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "agent-roster: missing proc reports unavailable on stderr, exit 2, never silent-alone" {
+ ROSTER_PGREP="$PGREP_STUB" ROSTER_PROC="$BATS_TEST_TMPDIR/nonexistent" \
+ ROSTER_SELF_PID="$SELF" FAKE_PIDS="999 2000" run "$ROSTER" "$ROOT"
+ [ "$status" -eq 2 ]
+ [[ "$output" == *"roster unavailable"* ]]
+}
+
+@test "agent-roster: a missing pgrep reports unavailable, exit 2, never silent-alone" {
+ # If pgrep itself is absent, the scan can't run; reporting "alone" would be a
+ # false negative the "never silent-alone" invariant forbids.
+ mkproc 2000 "$ROOT" 1
+ ROSTER_PGREP="$BATS_TEST_TMPDIR/no-such-pgrep" ROSTER_PROC="$PROC" \
+ ROSTER_SELF_PID="$SELF" FAKE_PIDS="999 2000" run "$ROSTER" "$ROOT"
+ [ "$status" -eq 2 ]
+ [[ "$output" == *"roster unavailable"* ]]
+}
+
+@test "agent-roster: defaults project root to PWD when no argument is given" {
+ mkproc 2000 "$ROOT" 1
+ FAKE_PIDS="999 2000"
+ ROSTER_PGREP="$PGREP_STUB" ROSTER_PROC="$PROC" ROSTER_SELF_PID="$SELF" \
+ FAKE_PIDS="$FAKE_PIDS" run env -C "$ROOT" "$ROSTER"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"2000"* ]]
+}
diff --git a/claude-templates/.ai/scripts/tests/capture-guard.bats b/claude-templates/.ai/scripts/tests/capture-guard.bats
new file mode 100644
index 0000000..31632a4
--- /dev/null
+++ b/claude-templates/.ai/scripts/tests/capture-guard.bats
@@ -0,0 +1,130 @@
+#!/usr/bin/env bats
+#
+# Tests for claude-templates/.ai/scripts/capture-guard — detects live
+# org-capture buffers visiting a target file before a workflow edits that
+# file on disk (the roam inbox, in inbox.org roam mode Phase D). Editing the file
+# underneath an indirect org-capture buffer wedges the capture (see emacs.md).
+#
+# Contract under test:
+# capture-guard [TARGET_FILE] (default TARGET_FILE = ~/org/roam/inbox.org)
+# exit 0 → safe to edit: emacsclient absent, daemon unreachable, or no
+# capture buffer visits TARGET_FILE.
+# exit 1 → a live capture buffer visits TARGET_FILE; its name(s) printed.
+#
+# Strategy: the emacsclient boundary is mocked with a PATH stub. The stub
+# answers the reachability probe (`-e t`) per STUB_REACHABLE and returns a
+# canned, real-emacsclient-shaped result (quoted string) for the buffer query
+# per STUB_BUFS. The script's own quote-stripping and exit logic is the code
+# under test; the file-equal-p precision is real-Emacs behavior we trust.
+
+SCRIPT="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/capture-guard"
+BASH_BIN="$(command -v bash)"
+
+setup() {
+ TEST_DIR="$(mktemp -d -t capture-guard-bats.XXXXXX)"
+ STUB_DIR="$TEST_DIR/bin"
+ mkdir -p "$STUB_DIR"
+
+ cat > "$STUB_DIR/emacsclient" <<'STUB'
+#!/usr/bin/env bash
+# Mock emacsclient. `-e t` is the reachability probe; anything else is the
+# buffer query, answered with the real-emacsclient-shaped quoted string.
+expr="$2"
+if [ "$expr" = "t" ]; then
+ [ "${STUB_REACHABLE:-1}" = "1" ] && { echo t; exit 0; }
+ exit 1
+fi
+printf '%s\n' "${STUB_BUFS:-\"\"}"
+exit 0
+STUB
+ chmod +x "$STUB_DIR/emacsclient"
+
+ EMPTY_DIR="$TEST_DIR/empty"
+ mkdir -p "$EMPTY_DIR"
+}
+
+teardown() {
+ rm -rf "$TEST_DIR"
+}
+
+# ---- Safe-to-edit (exit 0) cases ------------------------------------
+
+@test "capture-guard: emacsclient absent is safe (exit 0, no output)" {
+ run env PATH="$EMPTY_DIR" "$BASH_BIN" "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "capture-guard: daemon unreachable is safe (exit 0)" {
+ run env PATH="$STUB_DIR:$PATH" STUB_REACHABLE=0 "$BASH_BIN" "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "capture-guard: reachable with no capture buffers is safe (exit 0)" {
+ run env PATH="$STUB_DIR:$PATH" STUB_REACHABLE=1 STUB_BUFS='""' "$BASH_BIN" "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+# ---- Blocked (exit 1) cases -----------------------------------------
+
+@test "capture-guard: one live capture buffer blocks (exit 1, name printed)" {
+ run env PATH="$STUB_DIR:$PATH" STUB_REACHABLE=1 STUB_BUFS='"CAPTURE-inbox.org"' \
+ "$BASH_BIN" "$SCRIPT"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"CAPTURE-inbox.org"* ]]
+}
+
+@test "capture-guard: multiple live capture buffers all reported (exit 1)" {
+ run env PATH="$STUB_DIR:$PATH" STUB_REACHABLE=1 \
+ STUB_BUFS='"CAPTURE-inbox.org,CAPTURE-2-inbox.org"' \
+ "$BASH_BIN" "$SCRIPT"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"CAPTURE-inbox.org"* ]]
+ [[ "$output" == *"CAPTURE-2-inbox.org"* ]]
+}
+
+@test "capture-guard: blocked output does not contain stray surrounding quotes" {
+ run env PATH="$STUB_DIR:$PATH" STUB_REACHABLE=1 STUB_BUFS='"CAPTURE-inbox.org"' \
+ "$BASH_BIN" "$SCRIPT"
+ [ "$status" -eq 1 ]
+ [[ "$output" != \"* ]]
+ [[ "$output" != *\" ]]
+}
+
+# ---- Argument handling ----------------------------------------------
+
+@test "capture-guard: accepts an explicit target-file argument" {
+ run env PATH="$STUB_DIR:$PATH" STUB_REACHABLE=1 STUB_BUFS='""' \
+ "$BASH_BIN" "$SCRIPT" "$TEST_DIR/some-other-inbox.org"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+# ---- --wait poll mode -----------------------------------------------
+
+@test "capture-guard --wait: returns 0 instantly when already safe (no sleep)" {
+ SECONDS=0
+ run env PATH="$STUB_DIR:$PATH" STUB_REACHABLE=1 STUB_BUFS='""' \
+ "$BASH_BIN" "$SCRIPT" --wait
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+ [ "$SECONDS" -lt 2 ] # didn't poll-sleep
+}
+
+@test "capture-guard --wait=1: times out to exit 1 when persistently blocked" {
+ # Stub always reports the buffer, so it never clears — the short budget
+ # forces a timeout. Capped sleep keeps this near 1s.
+ run env PATH="$STUB_DIR:$PATH" STUB_REACHABLE=1 STUB_BUFS='"CAPTURE-inbox.org"' \
+ "$BASH_BIN" "$SCRIPT" --wait=1
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"CAPTURE-inbox.org"* ]]
+}
+
+@test "capture-guard --wait=N accepts a target after the flag" {
+ run env PATH="$STUB_DIR:$PATH" STUB_REACHABLE=1 STUB_BUFS='""' \
+ "$BASH_BIN" "$SCRIPT" --wait=1 "$TEST_DIR/some-other-inbox.org"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
diff --git a/claude-templates/.ai/scripts/tests/flashcard-sync.bats b/claude-templates/.ai/scripts/tests/flashcard-sync.bats
index 608a280..e6ffc21 100644
--- a/claude-templates/.ai/scripts/tests/flashcard-sync.bats
+++ b/claude-templates/.ai/scripts/tests/flashcard-sync.bats
@@ -6,6 +6,7 @@
setup() {
SCRIPT_DIR="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)"
SYNC="$SCRIPT_DIR/flashcard-sync"
+ STATS="$SCRIPT_DIR/flashcard-stats.py"
TMP="$(mktemp -d)"
}
@@ -36,3 +37,27 @@ EOF
[ "$status" -eq 1 ]
[ ! -f "$HOME/sync/phone/anki/dirty.apkg" ]
}
+
+@test "flashcard-stats: a multi-tagged :fundamental:drill: card still counts" {
+ # Regression guard: a curated card carrying a second org tag must not drop
+ # from the count. A :drill:$ anchor would have counted only one card here.
+ cat > "$TMP/multitag.org" <<'EOF'
+#+TITLE: Multitag Test
+
+* Orbital Regimes
+** What is LEO? :fundamental:drill:
+:PROPERTIES:
+:ID: c1
+:END:
+Low Earth Orbit is the region below about 2000 kilometers.
+** What is GEO? :drill:
+:PROPERTIES:
+:ID: c2
+:END:
+Geostationary orbit sits at roughly 35786 kilometers of altitude.
+EOF
+ run python3 "$STATS" "$TMP/multitag.org"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"Cards: 2"* ]]
+ [[ "$output" == *clean* ]]
+}
diff --git a/claude-templates/.ai/scripts/tests/inbox-status.bats b/claude-templates/.ai/scripts/tests/inbox-status.bats
index bc8a734..27a497e 100644
--- a/claude-templates/.ai/scripts/tests/inbox-status.bats
+++ b/claude-templates/.ai/scripts/tests/inbox-status.bats
@@ -45,6 +45,18 @@ teardown() {
[[ "$output" == *"0 pending"* ]]
}
+@test "inbox-status: ignores an in-flight .inbox-send-* temp file" {
+ mkdir "$TMP/inbox"
+ # inbox-send writes to a .inbox-send-* temp then renames it into place;
+ # during that window the temp must not read as a pending handoff, or a
+ # concurrent boundary check blocks on a file that's about to become real.
+ touch "$TMP/inbox/.inbox-send-abc123.org"
+ cd "$TMP"
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"0 pending"* ]]
+}
+
@test "inbox-status: -q suppresses the per-item lines" {
mkdir "$TMP/inbox"
echo body > "$TMP/inbox/handoff.org"
diff --git a/claude-templates/.ai/scripts/tests/lint-org-cli.bats b/claude-templates/.ai/scripts/tests/lint-org-cli.bats
index d457696..b9faef6 100644
--- a/claude-templates/.ai/scripts/tests/lint-org-cli.bats
+++ b/claude-templates/.ai/scripts/tests/lint-org-cli.bats
@@ -20,6 +20,24 @@ teardown() {
[[ "$output" == *"lint-org: file="* ]]
}
+@test "lint-org.el default invocation is report-only — file untouched" {
+ # bare #+begin_src is a mechanical fix (→ #+begin_example) that the old
+ # default applied on disk; a linter reports, it doesn't write
+ printf '* H\n\n#+begin_src\nx\n#+end_src\n' > "$TMPFILE"
+ before="$(cat "$TMPFILE")"
+ run emacs --batch -q -l "$SCRIPTS_DIR/lint-org.el" "$TMPFILE"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"would-fix"* ]]
+ [ "$(cat "$TMPFILE")" = "$before" ]
+}
+
+@test "lint-org.el --fix applies mechanical fixes on disk" {
+ printf '* H\n\n#+begin_src\nx\n#+end_src\n' > "$TMPFILE"
+ run emacs --batch -q -l "$SCRIPTS_DIR/lint-org.el" --fix "$TMPFILE"
+ [ "$status" -eq 0 ]
+ grep -q '#+begin_example' "$TMPFILE"
+}
+
@test "wrap-org-table.el loads and runs without -L on the load path" {
run emacs --batch -q -l "$SCRIPTS_DIR/wrap-org-table.el" --width=120 "$TMPFILE"
[ "$status" -eq 0 ]
diff --git a/claude-templates/.ai/scripts/tests/route-batch.bats b/claude-templates/.ai/scripts/tests/route-batch.bats
new file mode 100644
index 0000000..84ded5f
--- /dev/null
+++ b/claude-templates/.ai/scripts/tests/route-batch.bats
@@ -0,0 +1,202 @@
+#!/usr/bin/env bats
+#
+# Tests for claude-templates/.ai/scripts/route-batch — the wrap-up router's
+# mechanical go path (wrapup-routing spec, Phase 4 / D7 / D9).
+#
+# Contract under test:
+# route-batch --list one "<destination>\t<heading>" line per task
+# carrying :ROUTE_CANDIDATE:; silent when none;
+# never modifies anything
+# route-batch --go per candidate: write the subtree (minus the
+# :ROUTE_CANDIDATE: line) as a one-task handoff,
+# deliver via inbox-send to the destination's
+# inbox/, then remove the subtree from the local
+# todo.org. Send failure leaves the task in
+# place and exits non-zero. Empty set: no-op.
+#
+# Strategy: fixture roots under $TEST_DIR hold a source project and two
+# destination projects; INBOX_SEND_ROOTS sandboxes inbox-send's discovery to
+# them (the same hook inbox-send's own tests use).
+
+SCRIPT="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/route-batch"
+
+setup() {
+ TEST_DIR="$(mktemp -d -t route-batch-bats.XXXXXX)"
+ ROOTS="$TEST_DIR/roots"
+ SRC="$ROOTS/srcproj"
+ mkdir -p "$SRC/.ai" "$SRC/inbox" \
+ "$ROOTS/alpha/.ai" "$ROOTS/alpha/inbox" \
+ "$ROOTS/beta/.ai" "$ROOTS/beta/inbox"
+ touch "$ROOTS/alpha/todo.org" # alpha has a todo.org; beta deliberately not
+
+ cat > "$SRC/todo.org" <<'EOF'
+* Srcproj Open Work
+** TODO [#B] Alpha-bound task :feature:
+:PROPERTIES:
+:ROUTE_CANDIDATE: alpha
+:END:
+Body line about the alpha work.
+*** TODO Sub-task that rides along
+** TODO [#C] Purely local task
+Local body stays put.
+** TODO [#C] Beta-bound task :quick:
+:PROPERTIES:
+:CREATED: [2026-07-01 Tue]
+:ROUTE_CANDIDATE: beta
+:END:
+Beta body.
+EOF
+
+ export INBOX_SEND_ROOTS="$ROOTS"
+ cd "$SRC"
+}
+
+teardown() {
+ rm -rf "$TEST_DIR"
+}
+
+# ---- --list ------------------------------------------------------------
+
+@test "route-batch --list: one destination+heading line per candidate, backlog excluded" {
+ run "$SCRIPT" --list
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"alpha"*"Alpha-bound task"* ]]
+ [[ "$output" == *"beta"*"Beta-bound task"* ]]
+ [[ "$output" != *"Purely local task"* ]]
+}
+
+@test "route-batch --list: empty candidate set is silent (exit 0)" {
+ sed -i '/:ROUTE_CANDIDATE:/d' todo.org
+ run "$SCRIPT" --list
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "route-batch --list: modifies nothing (skip leaves all in place)" {
+ before="$(cat todo.org)"
+ run "$SCRIPT" --list
+ [ "$status" -eq 0 ]
+ [ "$(cat todo.org)" = "$before" ]
+ [ -z "$(ls "$ROOTS/alpha/inbox" "$ROOTS/beta/inbox" 2>/dev/null | grep -v ':')" ]
+}
+
+# ---- --go --------------------------------------------------------------
+
+@test "route-batch --go: delivers each candidate to its destination inbox with provenance" {
+ run "$SCRIPT" --go
+ [ "$status" -eq 0 ]
+ alpha_file=$(find "$ROOTS/alpha/inbox" -name '*from-srcproj*' -type f)
+ beta_file=$(find "$ROOTS/beta/inbox" -name '*from-srcproj*' -type f)
+ [ -n "$alpha_file" ]
+ [ -n "$beta_file" ]
+ grep -q 'Alpha-bound task' "$alpha_file"
+ grep -q 'Sub-task that rides along' "$alpha_file" # children ride along
+ grep -q 'Beta-bound task' "$beta_file"
+ ! grep -q ':ROUTE_CANDIDATE:' "$alpha_file"
+ ! grep -q ':ROUTE_CANDIDATE:' "$beta_file"
+}
+
+@test "route-batch --go: removes routed subtrees from todo.org, leaves local tasks" {
+ run "$SCRIPT" --go
+ [ "$status" -eq 0 ]
+ ! grep -q 'Alpha-bound task' todo.org
+ ! grep -q 'Sub-task that rides along' todo.org
+ ! grep -q 'Beta-bound task' todo.org
+ grep -q 'Purely local task' todo.org
+ grep -q 'Local body stays put' todo.org
+}
+
+@test "route-batch --go: a kept property drawer survives minus the marker" {
+ run "$SCRIPT" --go
+ [ "$status" -eq 0 ]
+ beta_file=$(find "$ROOTS/beta/inbox" -name '*from-srcproj*' -type f)
+ grep -q ':CREATED: \[2026-07-01 Tue\]' "$beta_file"
+}
+
+@test "route-batch --go: destination with inbox/ but no todo.org still delivers" {
+ run "$SCRIPT" --go
+ [ "$status" -eq 0 ]
+ [ ! -f "$ROOTS/beta/todo.org" ]
+ [ -n "$(find "$ROOTS/beta/inbox" -name '*from-srcproj*' -type f)" ]
+}
+
+@test "route-batch --go: empty candidate set is a silent no-op (exit 0)" {
+ sed -i '/:ROUTE_CANDIDATE:/d' todo.org
+ before="$(cat todo.org)"
+ run "$SCRIPT" --go
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+ [ "$(cat todo.org)" = "$before" ]
+}
+
+@test "route-batch --go: a failed send leaves that task in place, marker intact, and exits non-zero" {
+ sed -i 's/:ROUTE_CANDIDATE: beta/:ROUTE_CANDIDATE: ghost/' todo.org
+ run "$SCRIPT" --go
+ [ "$status" -ne 0 ]
+ grep -q 'Beta-bound task' todo.org # failed route stays local
+ grep -q ':ROUTE_CANDIDATE: ghost' todo.org # marker survives so it resurfaces next wrap
+ ! grep -q 'Alpha-bound task' todo.org # the good route still landed
+ [ -n "$(find "$ROOTS/alpha/inbox" -name '*from-srcproj*' -type f)" ]
+}
+
+@test "route-batch --go: handoff headings are promoted to top level" {
+ run "$SCRIPT" --go
+ [ "$status" -eq 0 ]
+ alpha_file=$(find "$ROOTS/alpha/inbox" -name '*from-srcproj*' -type f)
+ grep -q '^\* TODO \[#B\] Alpha-bound task' "$alpha_file"
+ grep -q '^\*\* TODO Sub-task that rides along' "$alpha_file"
+}
+
+@test "route-batch --go: a drawer emptied by the marker strip is pruned from the handoff" {
+ run "$SCRIPT" --go
+ [ "$status" -eq 0 ]
+ alpha_file=$(find "$ROOTS/alpha/inbox" -name '*from-srcproj*' -type f)
+ ! grep -q ':PROPERTIES:' "$alpha_file"
+}
+
+# ---- Overlapping candidates (nested marker data-loss regression) --------
+
+@test "route-batch --go: nested candidates conflict — both stay, bystander survives, exit non-zero" {
+ cat > todo.org <<'EOF'
+* Srcproj Open Work
+** TODO [#B] Parent bound for alpha
+:PROPERTIES:
+:ROUTE_CANDIDATE: alpha
+:END:
+Parent body.
+*** TODO Child bound for beta
+:PROPERTIES:
+:ROUTE_CANDIDATE: beta
+:END:
+Child body.
+** TODO [#C] Innocent bystander task
+Bystander body.
+EOF
+ run "$SCRIPT" --go
+ [ "$status" -ne 0 ]
+ [[ "$output" == *"CONFLICT"* ]]
+ grep -q 'Parent bound for alpha' todo.org
+ grep -q 'Child bound for beta' todo.org
+ grep -q 'Innocent bystander task' todo.org
+ grep -q 'Bystander body' todo.org
+ [ -z "$(find "$ROOTS/alpha/inbox" "$ROOTS/beta/inbox" -name '*from-srcproj*' -type f)" ]
+}
+
+@test "route-batch: duplicate identical markers in one drawer dedupe to a single route" {
+ cat > todo.org <<'EOF'
+* Srcproj Open Work
+** TODO [#B] Double-tagged for alpha
+:PROPERTIES:
+:ROUTE_CANDIDATE: alpha
+:ROUTE_CANDIDATE: alpha
+:END:
+Body.
+EOF
+ run "$SCRIPT" --list
+ [ "$status" -eq 0 ]
+ [ "$(echo "$output" | grep -c 'Double-tagged')" -eq 1 ]
+ [[ "$output" != *"CONFLICT"* ]]
+ run "$SCRIPT" --go
+ [ "$status" -eq 0 ]
+ [ "$(find "$ROOTS/alpha/inbox" -name '*from-srcproj*' -type f | wc -l)" -eq 1 ]
+}
diff --git a/claude-templates/.ai/scripts/tests/self-inject.bats b/claude-templates/.ai/scripts/tests/self-inject.bats
new file mode 100644
index 0000000..482f61d
--- /dev/null
+++ b/claude-templates/.ai/scripts/tests/self-inject.bats
@@ -0,0 +1,78 @@
+#!/usr/bin/env bats
+# Tests for self-inject.sh — tmux is the external boundary, stubbed with a
+# recording fake so no real server is needed.
+
+setup() {
+ SCRIPT="$BATS_TEST_DIRNAME/../self-inject.sh"
+ STUB_DIR="$BATS_TEST_TMPDIR/bin"
+ LOG="$BATS_TEST_TMPDIR/tmux.log"
+ mkdir -p "$STUB_DIR"
+}
+
+# A tmux stub that records every invocation and answers list-panes from
+# $STUB_PANES (empty by default, so pane derivation fails unless a test
+# provides ancestry-matching output).
+make_stub() {
+ cat > "$STUB_DIR/tmux" <<'EOF'
+#!/bin/sh
+echo "$@" >> "$LOG"
+case "$1" in
+ list-panes) printf '%s\n' "$STUB_PANES" ;;
+esac
+EOF
+ chmod +x "$STUB_DIR/tmux"
+}
+
+@test "self-inject: -t pane with no pairs echoes the pane and exits 0" {
+ make_stub
+ run env PATH="$STUB_DIR:$PATH" LOG="$LOG" STUB_PANES="" sh "$SCRIPT" -t %42
+ [ "$status" -eq 0 ]
+ [ "$output" = "%42" ]
+ # Pane was supplied, nothing sent: tmux must not have been called.
+ [ ! -e "$LOG" ]
+}
+
+@test "self-inject: no pane derivable and no -t exits 1 with an error" {
+ make_stub
+ run env PATH="$STUB_DIR:$PATH" LOG="$LOG" STUB_PANES="" sh "$SCRIPT" 0 "hello"
+ [ "$status" -eq 1 ]
+ case "$output" in *"no owning pane"*) : ;; *) false ;; esac
+}
+
+@test "self-inject: derives the pane from process ancestry via list-panes" {
+ make_stub
+ # The stub reports the bats test process itself as a pane's pane_pid;
+ # the script runs as our child, so that pid is in its ancestry.
+ run env PATH="$STUB_DIR:$PATH" LOG="$LOG" STUB_PANES="$$ %7" sh "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [ "$output" = "%7" ]
+}
+
+@test "self-inject: one delay/text pair sends literal text then Enter" {
+ make_stub
+ run env PATH="$STUB_DIR:$PATH" LOG="$LOG" STUB_PANES="" sh "$SCRIPT" -t %3 0 "/clear"
+ [ "$status" -eq 0 ]
+ run cat "$LOG"
+ [ "${lines[0]}" = "send-keys -t %3 -l /clear" ]
+ [ "${lines[1]}" = "send-keys -t %3 Enter" ]
+}
+
+@test "self-inject: multiple pairs send in order" {
+ make_stub
+ run env PATH="$STUB_DIR:$PATH" LOG="$LOG" STUB_PANES="" \
+ sh "$SCRIPT" -t %3 0 "/clear" 0 "go — resume"
+ [ "$status" -eq 0 ]
+ run cat "$LOG"
+ [ "${lines[0]}" = "send-keys -t %3 -l /clear" ]
+ [ "${lines[1]}" = "send-keys -t %3 Enter" ]
+ [ "${lines[2]}" = "send-keys -t %3 -l go — resume" ]
+ [ "${lines[3]}" = "send-keys -t %3 Enter" ]
+}
+
+@test "self-inject: dangling odd argument after pairs is ignored" {
+ make_stub
+ run env PATH="$STUB_DIR:$PATH" LOG="$LOG" STUB_PANES="" sh "$SCRIPT" -t %3 0 "one" 99
+ [ "$status" -eq 0 ]
+ run cat "$LOG"
+ [ "${#lines[@]}" -eq 2 ]
+}
diff --git a/claude-templates/.ai/scripts/tests/spec-sort.bats b/claude-templates/.ai/scripts/tests/spec-sort.bats
new file mode 100644
index 0000000..583e458
--- /dev/null
+++ b/claude-templates/.ai/scripts/tests/spec-sort.bats
@@ -0,0 +1,453 @@
+#!/usr/bin/env bats
+#
+# Tests for claude-templates/.ai/scripts/spec-sort — the one-time docs-pile
+# retrofit from the docs-lifecycle spec: classify docs/**/*.org outside
+# docs/specs/ (spec candidate iff it carries BOTH a Decisions heading AND an
+# Implementation phases heading), show an evidence panel, and on --apply
+# move + rename confirmed candidates to docs/specs/*-spec.org, prepend the
+# status heading (:ID:, dated history line), rewrite the keyword header to
+# the two-sequence form, relink file: links across the rewritten roots,
+# stamp :LAST_SPEC_SORT: in .ai/notes.org.
+#
+# Contract under test (docs/specs/2026-07-01-docs-lifecycle-spec.org,
+# "The retrofit"):
+# - dry-run report is the default; --apply writes
+# - --apply refuses on a dirty worktree (exit 2) unless --allow-dirty
+# - every candidate needs --confirm REL=KEYWORD or --skip REL (exit 1
+# otherwise); terminal keywords need --reason REL=TEXT
+# - plan validated before the first write; destination collisions block
+# - bare-path mentions in rewritten roots block --apply until
+# --acknowledge-bare waives them (reported, never rewritten)
+# - mid-apply failure names applied/not-applied + git restore recovery
+# - idempotent: a sorted project yields no candidates, no changes
+#
+# Strategy: each test builds a throwaway git project fixture and runs the
+# real script against it. Mid-apply failure is forced via the test-only
+# SPEC_SORT_INJECT_FAIL_AFTER env hook.
+
+SCRIPT="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/spec-sort"
+
+setup() {
+ TEST_DIR="$(mktemp -d -t spec-sort-bats.XXXXXX)"
+ PROJ="$TEST_DIR/proj"
+ mkdir -p "$PROJ"
+}
+
+teardown() {
+ rm -rf "$TEST_DIR"
+}
+
+# Standard fixture: one spec candidate, one note, a stray root spec with a
+# spine, an anomaly (-spec.org name, no spine), inbound links from todo.org,
+# a sibling note, a session archive (report-only surface), and .ai/notes.org
+# with a Workflow State section.
+make_project() {
+ cd "$PROJ"
+ git init -q
+ git config user.email test@test
+ git config user.name test
+ mkdir -p docs/design .ai/sessions
+
+ cat > docs/design/widget.org <<'EOF'
+#+TITLE: Widget Feature
+#+DATE: 2026-05-01
+#+TODO: DRAFT REVIEW | SHIPPED
+
+* Metadata
+| Status | draft |
+| Owner | Craig |
+
+* Summary
+The widget feature. See [[file:scratch-note.org][the note]].
+
+* Decisions [1/2]
+** DONE Pick the widget shape
+** TODO Pick the color
+
+* Implementation phases
+** Phase 1 — build =src/widget.py=
+EOF
+
+ cat > docs/design/scratch-note.org <<'EOF'
+#+TITLE: Scratch Note
+
+* Metadata
+| Status | n/a |
+
+* Thoughts
+See [[file:widget.org][the widget spec]].
+EOF
+
+ cat > docs/rooty-spec.org <<'EOF'
+#+TITLE: Rooty
+
+* Decisions
+** DONE Only decision
+
+* Implementation phases
+** Phase 1 — nothing
+EOF
+
+ cat > docs/lonely-spec.org <<'EOF'
+#+TITLE: Lonely
+Just prose, no spine.
+EOF
+
+ cat > todo.org <<'EOF'
+* Open Work
+** DOING [#B] Widget feature
+Spec: [[file:docs/design/widget.org][widget spec]].
+Summary anchor: [[file:docs/design/widget.org::*Summary][the summary]].
+EOF
+
+ cat > .ai/notes.org <<'EOF'
+* Active Reminders
+
+* Workflow State
+:LAST_AUDIT: 2026-06-28
+EOF
+
+ cat > .ai/sessions/2026-06-01-old.org <<'EOF'
+Old log: [[file:../../docs/design/widget.org][widget]]
+EOF
+
+ git add -A
+ git commit -qm init
+}
+
+# Confirm flags that satisfy the gate for the standard fixture's candidates.
+CONFIRM_ALL=(--confirm docs/design/widget.org=DRAFT --confirm docs/rooty-spec.org=DRAFT)
+
+# ---- Classification (dry-run) ----------------------------------------
+
+@test "spec-sort: dry-run classifies the spine-carrying doc as a candidate" {
+ make_project
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"CANDIDATE docs/design/widget.org -> docs/specs/widget-spec.org"* ]]
+}
+
+@test "spec-sort: a Metadata table alone does not qualify — note stays a note" {
+ make_project
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"NOTE docs/design/scratch-note.org"* ]]
+ [[ "$output" != *"CANDIDATE docs/design/scratch-note.org"* ]]
+}
+
+@test "spec-sort: stray root spec with a spine is a candidate, suffix not doubled" {
+ make_project
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"CANDIDATE docs/rooty-spec.org -> docs/specs/rooty-spec.org"* ]]
+ [[ "$output" != *"rooty-spec-spec.org"* ]]
+}
+
+@test "spec-sort: -spec.org name without a spine is an anomaly, never auto-moved" {
+ make_project
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"ANOMALY docs/lonely-spec.org"* ]]
+ [[ "$output" != *"CANDIDATE docs/lonely-spec.org"* ]]
+}
+
+@test "spec-sort: docs/specs/ contents are excluded from classification" {
+ make_project
+ mkdir -p docs/specs
+ cp docs/design/widget.org docs/specs/sorted-spec.org
+ git add -A && git commit -qm more
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" != *"CANDIDATE docs/specs/sorted-spec.org"* ]]
+}
+
+@test "spec-sort: no docs/ directory is a silent no-op" {
+ cd "$PROJ"
+ git init -q
+ git config user.email test@test
+ git config user.name test
+ echo x > README.md
+ git add -A && git commit -qm init
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+# ---- Evidence panel ---------------------------------------------------
+
+@test "spec-sort: evidence panel shows status field, cookies, and todo.org task" {
+ make_project
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"status field: draft"* ]]
+ [[ "$output" == *"Decisions [1/2]"* ]]
+ [[ "$output" == *"todo.org:"*"DOING"*"Widget feature"* ]]
+}
+
+@test "spec-sort: keyword proposal follows the evidence — DOING from the linked DOING task" {
+ make_project
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ # status field says draft, but the linking todo.org task is DOING — the
+ # panel proposes the state the strongest evidence supports
+ [[ "$output" == *"proposed keyword: DOING"* ]]
+}
+
+@test "spec-sort: an 'incomplete' status field never proposes the terminal IMPLEMENTED" {
+ make_project
+ sed -i 's/| Status | draft |/| Status | incomplete |/' docs/design/widget.org
+ git add -A && git commit -qm status
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" != *"proposed keyword: IMPLEMENTED"* ]]
+}
+
+# ---- Confirm gate -----------------------------------------------------
+
+@test "spec-sort --apply: refuses when a candidate is neither confirmed nor skipped" {
+ make_project
+ run "$SCRIPT" --apply --confirm docs/design/widget.org=DRAFT
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"unconfirmed"* ]]
+ [[ "$output" == *"docs/rooty-spec.org"* ]]
+ [ -f docs/design/widget.org ] # nothing moved
+}
+
+@test "spec-sort --apply: a terminal keyword without --reason refuses" {
+ make_project
+ run "$SCRIPT" --apply --confirm docs/design/widget.org=IMPLEMENTED --skip docs/rooty-spec.org
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"--reason"* ]]
+ [ -f docs/design/widget.org ]
+}
+
+@test "spec-sort --apply: a terminal keyword with --reason records it in the history line" {
+ make_project
+ run "$SCRIPT" --apply --confirm docs/design/widget.org=IMPLEMENTED \
+ --reason "docs/design/widget.org=shipped in v2, confirmed against src" \
+ --skip docs/rooty-spec.org
+ [ "$status" -eq 0 ]
+ grep -q '^\* IMPLEMENTED Widget Feature' docs/specs/widget-spec.org
+ grep -q 'shipped in v2, confirmed against src' docs/specs/widget-spec.org
+}
+
+@test "spec-sort --apply: --skip leaves the candidate in place and still stamps the marker" {
+ make_project
+ run "$SCRIPT" --apply --skip docs/design/widget.org --skip docs/rooty-spec.org
+ [ "$status" -eq 0 ]
+ [ -f docs/design/widget.org ]
+ grep -q ':LAST_SPEC_SORT:' .ai/notes.org
+}
+
+# ---- Preflight --------------------------------------------------------
+
+@test "spec-sort --apply: refuses on a dirty worktree (exit 2)" {
+ make_project
+ echo "drift" >> todo.org
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 2 ]
+ [[ "$output" == *"dirty"* ]]
+ [ -f docs/design/widget.org ]
+}
+
+@test "spec-sort --apply --allow-dirty: proceeds and names what recovery loses" {
+ make_project
+ echo "drift" >> todo.org
+ git add todo.org && git commit -qm drift # keep the link intact; dirty a different file
+ echo "scratch" > untracked-note.txt
+ echo "local edit" >> .ai/notes.org
+ run "$SCRIPT" --apply --allow-dirty "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"pre-existing"* ]]
+ [[ "$output" == *".ai/notes.org"* ]]
+ [ -f docs/specs/widget-spec.org ]
+}
+
+# ---- Move + rename + rewrite ------------------------------------------
+
+@test "spec-sort --apply: moves, renames to -spec.org, prepends status heading with :ID: and history" {
+ make_project
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ [ -f docs/specs/widget-spec.org ]
+ [ ! -f docs/design/widget.org ]
+ grep -q '^\* DRAFT Widget Feature' docs/specs/widget-spec.org
+ grep -q ':ID:' docs/specs/widget-spec.org
+ grep -q 'retrofitted by spec-sort' docs/specs/widget-spec.org
+}
+
+@test "spec-sort --apply: keyword header rewritten to the two-sequence form" {
+ make_project
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ grep -q '^#+TODO: TODO | DONE$' docs/specs/widget-spec.org
+ grep -q '^#+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED$' docs/specs/widget-spec.org
+ ! grep -q 'DRAFT REVIEW | SHIPPED' docs/specs/widget-spec.org
+}
+
+@test "spec-sort --apply: Metadata Status field mirrors the confirmed keyword in lowercase" {
+ make_project
+ run "$SCRIPT" --apply --confirm docs/design/widget.org=READY --skip docs/rooty-spec.org
+ [ "$status" -eq 0 ]
+ grep -q '^\* READY Widget Feature' docs/specs/widget-spec.org
+ grep -Eq '^\| Status[[:space:]]*\|[[:space:]]*ready' docs/specs/widget-spec.org
+}
+
+# ---- Relink -----------------------------------------------------------
+
+@test "spec-sort --apply: rewrites the todo.org link, preserving the description" {
+ make_project
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ grep -q '\[\[file:docs/specs/widget-spec.org\]\[widget spec\]\]' todo.org
+ ! grep -q 'docs/design/widget.org' todo.org
+}
+
+@test "spec-sort --apply: preserves a ::anchor suffix through the rewrite" {
+ make_project
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ grep -q '\[\[file:docs/specs/widget-spec.org::\*Summary\]\[the summary\]\]' todo.org
+}
+
+@test "spec-sort --apply: recomputes a sibling note's relative link to the moved spec" {
+ make_project
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ grep -q '\[\[file:../specs/widget-spec.org\]\[the widget spec\]\]' docs/design/scratch-note.org
+}
+
+@test "spec-sort --apply: recomputes the moved spec's own outbound link to an unmoved note" {
+ make_project
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ grep -q '\[\[file:../design/scratch-note.org\]\[the note\]\]' docs/specs/widget-spec.org
+}
+
+@test "spec-sort: session archives are reported, never rewritten" {
+ make_project
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"REPORT .ai/sessions/2026-06-01-old.org"* ]]
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ grep -q 'docs/design/widget.org' .ai/sessions/2026-06-01-old.org
+}
+
+@test "spec-sort: a synced template path report names the canonical rulesets file" {
+ make_project
+ mkdir -p .ai/workflows
+ echo 'See [[file:../../docs/design/widget.org][widget]]' > .ai/workflows/startup.org
+ git add -A && git commit -qm wf
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"REPORT .ai/workflows/startup.org"* ]]
+ [[ "$output" == *"claude-templates/.ai/workflows/startup.org"* ]]
+}
+
+# ---- Bare-path mentions -----------------------------------------------
+
+@test "spec-sort --apply: a bare-path mention in a rewritten root blocks until acknowledged" {
+ make_project
+ echo "raw mention: docs/design/widget.org needs review" >> todo.org
+ git add -A && git commit -qm bare
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"BARE"* ]]
+ [ -f docs/design/widget.org ] # nothing moved
+ run "$SCRIPT" --apply --acknowledge-bare "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ grep -q 'raw mention: docs/design/widget.org' todo.org # reported, never rewritten
+}
+
+@test "spec-sort --apply: a moving doc's bare mention of its own old path is acknowledgeable, not post-apply residue" {
+ make_project
+ echo "History: docs/design/widget.org was drafted in May." >> docs/design/widget.org
+ git add -A && git commit -qm selfmention
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"BARE"* ]]
+ run "$SCRIPT" --apply --acknowledge-bare "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ] # the acknowledged mention rides along to docs/specs/; not residue
+ grep -q ':LAST_SPEC_SORT:' .ai/notes.org
+}
+
+# ---- Plan validation ---------------------------------------------------
+
+@test "spec-sort --apply: a destination collision blocks validation, nothing moved" {
+ make_project
+ mkdir -p docs/specs
+ echo "occupied" > docs/specs/widget-spec.org
+ git add -A && git commit -qm occupy
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"destination exists"* ]]
+ [ -f docs/design/widget.org ]
+ [ "$(cat docs/specs/widget-spec.org)" = "occupied" ]
+}
+
+@test "spec-sort --apply: writes the plan file before executing" {
+ make_project
+ run "$SCRIPT" --apply --plan-file "$TEST_DIR/plan.json" "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ [ -f "$TEST_DIR/plan.json" ]
+ grep -q 'widget-spec.org' "$TEST_DIR/plan.json"
+}
+
+# ---- Mid-apply failure recovery ----------------------------------------
+
+@test "spec-sort --apply: forced mid-apply failure yields named recovery, not a half-migrated shrug" {
+ make_project
+ run env SPEC_SORT_INJECT_FAIL_AFTER=1 "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"RECOVERY"* ]]
+ [[ "$output" == *"git restore"* ]]
+ [[ "$output" == *"applied"* ]]
+ [[ "$output" == *"not applied"* ]]
+ ! grep -q ':LAST_SPEC_SORT:' .ai/notes.org # no stamp on a failed apply
+}
+
+# ---- Idempotence + marker ----------------------------------------------
+
+@test "spec-sort --apply: stamps :LAST_SPEC_SORT: in the Workflow State section" {
+ make_project
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ grep -q ':LAST_SPEC_SORT: ' .ai/notes.org
+ # lands inside the Workflow State section, alongside the existing marker
+ awk '/^\* Workflow State/{ws=1} ws && /:LAST_SPEC_SORT:/{found=1} END{exit !found}' .ai/notes.org
+}
+
+@test "spec-sort --apply: creates the Workflow State section when notes.org lacks it" {
+ make_project
+ printf '* Active Reminders\n' > .ai/notes.org
+ git add -A && git commit -qm notes
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ grep -q '^\* Workflow State' .ai/notes.org
+ grep -q ':LAST_SPEC_SORT: ' .ai/notes.org
+}
+
+@test "spec-sort --apply: zero candidates still stamps the marker (clears the nudge)" {
+ make_project
+ rm docs/design/widget.org docs/rooty-spec.org docs/lonely-spec.org
+ git add -A && git commit -qm notes-only
+ run "$SCRIPT" --apply
+ [ "$status" -eq 0 ]
+ grep -q ':LAST_SPEC_SORT:' .ai/notes.org
+}
+
+@test "spec-sort: a second run after a successful apply finds nothing to do" {
+ make_project
+ run "$SCRIPT" --apply "${CONFIRM_ALL[@]}"
+ [ "$status" -eq 0 ]
+ git add -A && git commit -qm sorted
+ run "$SCRIPT"
+ [ "$status" -eq 0 ]
+ [[ "$output" != *"CANDIDATE"* ]]
+ run "$SCRIPT" --apply
+ [ "$status" -eq 0 ]
+ run git status --porcelain
+ # only the re-stamped marker (same date) may differ — tree stays clean
+ [ -z "$(git status --porcelain -- docs todo.org)" ]
+}
diff --git a/claude-templates/.ai/scripts/tests/task-review-staleness.bats b/claude-templates/.ai/scripts/tests/task-review-staleness.bats
index 488b023..79aad79 100644
--- a/claude-templates/.ai/scripts/tests/task-review-staleness.bats
+++ b/claude-templates/.ai/scripts/tests/task-review-staleness.bats
@@ -49,6 +49,16 @@ task_unreviewed() {
printf '** %s [#%s] %s\nBody.\n\n' "$keyword" "$prio" "$title" >> "$TODO"
}
+# Emit a qualifying task whose LAST_REVIEWED is an org-native inactive
+# timestamp — [YYYY-MM-DD Day] — matching the CREATED:/CLOSED: cookies that
+# sit in the same drawer. The date is derived from an ISO date via `date`.
+task_reviewed_org() {
+ local keyword="$1" prio="$2" title="$3" isodate="$4"
+ local org="[$(date -d "$isodate" '+%F %a')]"
+ printf '** %s [#%s] %s\n:PROPERTIES:\n:LAST_REVIEWED: %s\n:END:\nBody.\n\n' \
+ "$keyword" "$prio" "$title" "$org" >> "$TODO"
+}
+
# ---- Normal cases ----------------------------------------------------
@test "staleness: empty file reports zero" {
@@ -85,6 +95,20 @@ task_unreviewed() {
[ "$output" = "2" ]
}
+@test "staleness: org-native bracketed LAST_REVIEWED parses — recent is fresh" {
+ task_reviewed_org TODO A "Reviewed five days ago, org stamp" "$D5"
+ run bash "$SCRIPT" "$TODO" 30
+ [ "$status" -eq 0 ]
+ [ "$output" = "0" ]
+}
+
+@test "staleness: org-native bracketed LAST_REVIEWED parses — old is stale" {
+ task_reviewed_org TODO A "Reviewed forty days ago, org stamp" "$D40"
+ run bash "$SCRIPT" "$TODO" 30
+ [ "$status" -eq 0 ]
+ [ "$output" = "1" ]
+}
+
# ---- Boundary cases --------------------------------------------------
@test "staleness: age exactly equal to threshold is fresh" {
@@ -136,9 +160,23 @@ task_unreviewed() {
[ "$output" = "0" ]
}
-@test "staleness: malformed LAST_REVIEWED is treated as stale" {
+@test "staleness: malformed LAST_REVIEWED warns to stderr and is not counted" {
task_reviewed TODO A "Bad date" "not-a-date"
- run bash "$SCRIPT" "$TODO" 30
+ # stdout carries only the count — the malformed stamp is not folded in.
+ run bash -c "bash '$SCRIPT' '$TODO' 30 2>/dev/null"
+ [ "$status" -eq 0 ]
+ [ "$output" = "0" ]
+ # stderr carries the loud warning naming the offending value.
+ run bash -c "bash '$SCRIPT' '$TODO' 30 2>&1 1>/dev/null"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"not-a-date"* ]]
+ [[ "$output" == *"LAST_REVIEWED"* ]]
+}
+
+@test "staleness: malformed stamp is excluded while real stale tasks still count" {
+ task_reviewed TODO A "Real stale" "$D40"
+ task_reviewed TODO B "Broken stamp" "garbage"
+ run bash -c "bash '$SCRIPT' '$TODO' 30 2>/dev/null"
[ "$status" -eq 0 ]
[ "$output" = "1" ]
}
@@ -161,6 +199,17 @@ task_unreviewed() {
[[ "${lines[2]}" == *"Reviewed recently"* ]]
}
+@test "staleness --list: org-native bracketed stamp sorts by its real date" {
+ task_reviewed TODO A "Bare recent" "$D5"
+ task_reviewed_org TODO B "Org-stamped old" "$D40"
+ run bash "$SCRIPT" --list "$TODO" 10
+ [ "$status" -eq 0 ]
+ # The org-bracketed old stamp must sort ahead of the bare recent one —
+ # proof it parsed to a real date rather than falling to 0000-00-00.
+ [[ "${lines[0]}" == *"Org-stamped old"* ]]
+ [[ "${lines[1]}" == *"Bare recent"* ]]
+}
+
@test "staleness --list: takes only the requested count" {
task_unreviewed TODO A "First"
task_reviewed TODO B "Second" "$D40"
diff --git a/claude-templates/.ai/scripts/tests/test-lint-org.el b/claude-templates/.ai/scripts/tests/test-lint-org.el
index 3a83602..ceee209 100644
--- a/claude-templates/.ai/scripts/tests/test-lint-org.el
+++ b/claude-templates/.ai/scripts/tests/test-lint-org.el
@@ -193,6 +193,65 @@ real suspicious-language warning here
#+end_src
")
+;; invalid-block, false-positive case — a correctly paired example block whose
+;; body holds a heading-shaped line. org's parser reads the `** ' inside the
+;; verbatim body as a structural break, loses the open block, and flags BOTH
+;; delimiters as "Possible incomplete block".
+(defconst lo-test--verbatim-heading-block "\
+* Heading
+
+#+begin_example
+** Feature Name or Topic
+Body line.
+#+end_example
+
+Trailing prose.
+")
+
+;; invalid-block, literal-delimiter case — a paired src block whose body holds
+;; a literal `#+end_example' plus a heading-shaped line. Only `#+end_src'
+;; closes a src block, so all three findings here are false.
+(defconst lo-test--literal-end-in-src "\
+* Heading
+
+#+begin_src text
+#+end_example
+** heading shaped
+#+end_src
+")
+
+;; invalid-block, uppercase-delimiter case — org accepts #+BEGIN_/#+END_ in
+;; either case, and the pre-fix script flagged both delimiters here too.
+(defconst lo-test--uppercase-verbatim-block "\
+* Heading
+
+#+BEGIN_EXAMPLE
+** heading shaped
+#+END_EXAMPLE
+")
+
+;; invalid-block, genuine case — a block that really is never closed. The
+;; suppression must not reach this one.
+(defconst lo-test--unterminated-block "\
+* Heading
+
+#+begin_example
+truly unterminated block body
+")
+
+;; A genuinely unterminated block *after* a correctly paired one — verifies the
+;; suppression is scoped per block rather than per file.
+(defconst lo-test--paired-then-unterminated "\
+* Heading
+
+#+begin_example
+** heading shaped
+#+end_example
+
+#+begin_example
+never closed
+")
+
;; Mixed fixture — each category once.
(defconst lo-test--mixed "\
* Mixed
@@ -392,6 +451,55 @@ suspicious-language judgment."
(should (= 1 suspicious))))
;;; ---------------------------------------------------------------------------
+;;; invalid-block — false positives on correctly paired verbatim blocks
+
+(ert-deftest lo-verbatim-heading-block-emits-no-invalid-block ()
+ "Normal: a paired example block containing a heading-shaped body line emits
+no invalid-block judgment. Both delimiters are flagged by org-lint because the
+parser treats the `** ' inside the verbatim body as a structural break."
+ (let* ((out (lo-test--run lo-test--verbatim-heading-block))
+ (res (plist-get out :result))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ ;; File untouched, no fixes applied — suppression only, never a rewrite.
+ (should (equal lo-test--verbatim-heading-block res))
+ (should (= 0 (plist-get out :fixes)))
+ (should-not (member 'invalid-block (lo-test--checkers judgments)))))
+
+(ert-deftest lo-literal-end-delimiter-in-src-emits-no-invalid-block ()
+ "Boundary: a paired src block whose body holds a literal `#+end_example' and
+a heading-shaped line emits no invalid-block judgment. Only `#+end_src' closes
+a src block, so the interior delimiter is body text."
+ (let* ((out (lo-test--run lo-test--literal-end-in-src))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'invalid-block (lo-test--checkers judgments)))))
+
+(ert-deftest lo-uppercase-verbatim-block-emits-no-invalid-block ()
+ "Boundary: block delimiters are case-insensitive in org, so an uppercase
+`#+BEGIN_EXAMPLE' pair is suppressed the same as a lowercase one."
+ (let* ((out (lo-test--run lo-test--uppercase-verbatim-block))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'invalid-block (lo-test--checkers judgments)))))
+
+(ert-deftest lo-unterminated-block-still-emits-invalid-block ()
+ "Error: a block that is never closed still emits its invalid-block judgment.
+This is the finding the checker exists for — the suppression must not mask it."
+ (let* ((out (lo-test--run lo-test--unterminated-block))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should (member 'invalid-block (lo-test--checkers judgments)))))
+
+(ert-deftest lo-invalid-block-suppression-is-scoped-per-block ()
+ "Boundary: a paired block and an unterminated block in the same file — the
+paired one is suppressed and the unterminated one still reports. Exactly one
+invalid-block judgment, and it points at the unterminated opener (line 7)."
+ (let* ((out (lo-test--run lo-test--paired-then-unterminated))
+ (judgments (lo-test--judgments (plist-get out :issues)))
+ (invalid (cl-remove-if-not
+ (lambda (i) (eq (plist-get i :checker) 'invalid-block))
+ judgments)))
+ (should (= 1 (length invalid)))
+ (should (= 7 (plist-get (car invalid) :line)))))
+
+;;; ---------------------------------------------------------------------------
;;; --check mode
(ert-deftest lo-check-mode-does-not-modify-file ()
@@ -620,6 +728,29 @@ followups file on the next run."
;;; ---------------------------------------------------------------------------
;;; org-table-standard check (width budget + rules between rows)
+(ert-deftest lo-table-inside-example-block-not-flagged ()
+ "Pipe-led ASCII art inside an example block is not a table; no judgment."
+ (let* ((run (lo-test--run
+ "* H\n\n#+begin_example\n| client |----->| server |\n| box | | box |\n#+end_example\n"
+ 1 t))
+ (judgments (lo-test--judgments (plist-get run :issues))))
+ (should-not (memq 'org-table-standard (lo-test--checkers judgments)))))
+
+(ert-deftest lo-table-inside-src-block-not-flagged ()
+ "Shell pipes inside a src block are not a table; no judgment."
+ (let* ((run (lo-test--run
+ "* H\n\n#+begin_src sh\n| sort\n| uniq -c\n#+end_src\n" 1 t))
+ (judgments (lo-test--judgments (plist-get run :issues))))
+ (should-not (memq 'org-table-standard (lo-test--checkers judgments)))))
+
+(ert-deftest lo-real-table-after-block-still-flagged ()
+ "Block safety must not mask a genuine violation later in the file."
+ (let* ((run (lo-test--run
+ "* H\n\n#+begin_example\n| art |\n#+end_example\n\n| a | b |\n| 1 | 2 |\n"
+ 1 t))
+ (judgments (lo-test--judgments (plist-get run :issues))))
+ (should (memq 'org-table-standard (lo-test--checkers judgments)))))
+
(ert-deftest lo-table-over-budget-emits-judgment ()
"A table line rendering wider than 120 surfaces as an org-table-standard judgment."
(let* ((wide (make-string 130 ?x))
@@ -659,5 +790,311 @@ missing-rules violation."
(judgments (lo-test--judgments (plist-get run :issues))))
(should-not (memq 'org-table-standard (lo-test--checkers judgments)))))
+;;; ---------------------------------------------------------------------------
+;;; level-2 dated-header check (claude-rules/todo-format.md)
+
+(ert-deftest lo-level2-dated-header-is-judgment ()
+ "A level-2 heading beginning with a YYYY-MM-DD date is flagged."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** 2026-06-20 Sat @ 10:00:00 -0500 Something resolved\nBody.\n"))
+ (res (plist-get out :result))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should (= 0 (plist-get out :fixes))) ; judgment-only, never auto-fixed
+ (should (member 'level-2-dated-header (lo-test--checkers judgments)))))
+
+(ert-deftest lo-level2-done-task-not-flagged ()
+ "A level-2 task closed with a terminal keyword + CLOSED: is fine."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** DONE [#B] Something resolved\nCLOSED: [2026-06-20 Sat]\nBody.\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'level-2-dated-header (lo-test--checkers judgments)))))
+
+(ert-deftest lo-level3-dated-entry-not-flagged ()
+ "A dated event-log entry at level 3 is the correct sub-task shape, not a defect."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** TODO [#B] Parent task\n*** 2026-06-20 Sat @ 10:00:00 -0500 sub-entry landed\nBody.\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'level-2-dated-header (lo-test--checkers judgments)))))
+
+;;; subtask-done-not-dated check (the inverse: level-3+ done keyword)
+
+(ert-deftest lo-subtask-done-not-dated-flags-level3 ()
+ "A level-3 DONE sub-task still carrying the keyword is flagged for conversion."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** TODO [#B] Parent\n*** DONE [#C] Sub-task done\nCLOSED: [2026-06-20 Sat 10:00]\nBody.\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should (= 0 (plist-get out :fixes))) ; judgment-only, never auto-fixed
+ (should (member 'subtask-done-not-dated (lo-test--checkers judgments)))))
+
+(ert-deftest lo-subtask-done-not-dated-flags-level4-cancelled ()
+ "A level-4 CANCELLED sub-task is flagged too."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** PROJECT [#B] Parent\n*** TODO Mid\n**** CANCELLED Deep abandoned\nCLOSED: [2026-06-20 Sat]\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should (member 'subtask-done-not-dated (lo-test--checkers judgments)))))
+
+(ert-deftest lo-subtask-done-not-dated-ignores-level2 ()
+ "A level-2 DONE task is a top-level task, not a sub-task — this checker skips it."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** DONE [#B] Top-level\nCLOSED: [2026-06-20 Sat]\nBody.\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'subtask-done-not-dated (lo-test--checkers judgments)))))
+
+(ert-deftest lo-subtask-done-not-dated-ignores-dated-and-lowercase ()
+ "An already-dated level-3 entry, and the word done in a title, are not flagged."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0400 landed\n*** TODO wrap the done cleanup\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'subtask-done-not-dated (lo-test--checkers judgments)))))
+
+;;; dated-log-heading-active-timestamp check (stale SCHEDULED/DEADLINE on a
+;;; completed dated-log entry — the home 2026-07-17 agenda-pollution bug)
+
+(ert-deftest lo-dated-log-active-scheduled-is-flagged ()
+ "A dated-log entry still carrying an active SCHEDULED is flagged: org renders
+it on the agenda forever despite the missing keyword."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 trip booked\nSCHEDULED: <2026-06-18 Thu>\nBody.\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should (= 0 (plist-get out :fixes))) ; judgment-only, never auto-fixed
+ (should (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments)))))
+
+(ert-deftest lo-dated-log-active-deadline-is-flagged ()
+ "An active DEADLINE on a dated-log entry is flagged too."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 shipped\nDEADLINE: <2026-06-25 Thu>\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments)))))
+
+(ert-deftest lo-dated-log-clean-entry-not-flagged ()
+ "A dated-log entry with no active planning timestamp is correct — not flagged."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 done cleanly\nBody only.\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments)))))
+
+(ert-deftest lo-dated-log-inactive-timestamp-not-flagged ()
+ "An inactive [..] timestamp doesn't render on the agenda, so it isn't flagged —
+only active <..> planning timestamps are the defect."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 recorded\nSCHEDULED: [2026-06-18 Thu]\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments)))))
+
+(ert-deftest lo-dated-log-active-scheduled-on-live-todo-not-flagged ()
+ "A live TODO (keyword present) that legitimately carries an active SCHEDULED is
+not a dated-log heading, so this checker leaves it alone."
+ (let* ((out (lo-test--run
+ "* Open Work\n\n** TODO [#B] Parent\n*** TODO [#C] real upcoming task\nSCHEDULED: <2026-06-18 Thu>\n"))
+ (judgments (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments)))))
+
+;;; ---------------------------------------------------------------------------
+;;; structural heading checks (org-lint gaps)
+
+(defun lo-test--checker-lines (issues checker)
+ "Lines of judgment ISSUES whose :checker is CHECKER, document order."
+ (mapcar (lambda (i) (plist-get i :line))
+ (cl-remove-if-not
+ (lambda (i) (and (eq (plist-get i :kind) 'judgment)
+ (eq (plist-get i :checker) checker)))
+ (reverse issues))))
+
+(ert-deftest lo-indented-heading-flags-leading-whitespace ()
+ "Error: a heading indented off column 0 is flagged (org demotes it to body)."
+ (let* ((out (lo-test--run "* Open\n ** TODO indented and lost\n** TODO fine\n"))
+ (j (lo-test--judgments (plist-get out :issues))))
+ (should (member 'indented-heading (lo-test--checkers j)))
+ (should (= 1 (length (lo-test--checker-lines (plist-get out :issues)
+ 'indented-heading))))))
+
+(ert-deftest lo-indented-heading-skips-stars-inside-blocks ()
+ "Boundary: indented stars inside a #+begin_/#+end_ block are legitimate content."
+ (let* ((out (lo-test--run "* Open\n#+begin_example\n ** not a heading\n#+end_example\n"))
+ (j (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'indented-heading (lo-test--checkers j)))))
+
+(ert-deftest lo-indented-heading-skips-single-star-list-bullets ()
+ "Normal: an indented single `*' is a valid plain-list bullet, not a demoted
+heading, so it is not flagged — only two-or-more indented stars are."
+ (let* ((out (lo-test--run "* Open\nintro line\n * first bullet\n * second bullet\n * nested bullet\n"))
+ (j (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'indented-heading (lo-test--checkers j)))))
+
+(ert-deftest lo-empty-heading-flags-bare-stars ()
+ "Error: a line of bare stars with no title is flagged."
+ (let* ((out (lo-test--run "* Open\n** \n** TODO real\n"))
+ (j (lo-test--judgments (plist-get out :issues))))
+ (should (member 'empty-heading (lo-test--checkers j)))))
+
+(ert-deftest lo-malformed-priority-flags-lowercase-and-skips-valid ()
+ "Error + Normal: a lowercase/oversized cookie flags; a valid [#B] stays silent."
+ (let* ((bad (lo-test--run "* Open\n** TODO [#a] lowercase cookie\n** TODO [#BB] oversized\n"))
+ (ok (lo-test--run "* Open\n** TODO [#B] valid cookie\n"))
+ (jo (lo-test--judgments (plist-get ok :issues))))
+ (should (= 2 (length (lo-test--checker-lines (plist-get bad :issues)
+ 'malformed-priority-cookie))))
+ (should-not (member 'malformed-priority-cookie (lo-test--checkers jo)))))
+
+(ert-deftest lo-malformed-priority-skips-verbatim-cookie-in-title ()
+ "Boundary: a dated-log title quoting =[#D]= verbatim is not a real cookie."
+ (let* ((out (lo-test--run "* Open\n** TODO [#B] parent\n*** 2026-05-14 reprioritized =[#D]= -> =[#B]=\n"))
+ (j (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'malformed-priority-cookie (lo-test--checkers j)))))
+
+(ert-deftest lo-done-without-closed-flags-undated-level2 ()
+ "Error: a level-2 DONE with no CLOSED line is flagged; a dated one is not."
+ (let* ((bad (lo-test--run "* Resolved\n** DONE undated finished\nbody\n"))
+ (jb (lo-test--judgments (plist-get bad :issues)))
+ (ok (lo-test--run "* Resolved\n** DONE dated\nCLOSED: [2026-06-29 Mon]\n"))
+ (jo (lo-test--judgments (plist-get ok :issues))))
+ (should (member 'level2-done-without-closed (lo-test--checkers jb)))
+ (should-not (member 'level2-done-without-closed (lo-test--checkers jo)))))
+
+(ert-deftest lo-done-without-closed-ignores-deeper-levels ()
+ "Boundary: a level-3 DONE (a dated-log sub-entry) need not carry CLOSED."
+ (let* ((out (lo-test--run "* Resolved\n** DONE parent\nCLOSED: [2026-06-29 Mon]\n*** DONE nested no-closed\n"))
+ (j (lo-test--judgments (plist-get out :issues))))
+ (should-not (member 'level2-done-without-closed (lo-test--checkers j)))))
+
+(ert-deftest lo-structural-checks-silent-on-clean-file ()
+ "Normal: a well-formed file trips none of the four structural checkers."
+ (let* ((out (lo-test--run "* Open Work\n** TODO [#A] a task :tag:\n** DOING [#B] another\n* Resolved\n** DONE [#C] done\nCLOSED: [2026-06-29 Mon]\n"))
+ (checkers (lo-test--checkers (lo-test--judgments (plist-get out :issues)))))
+ (dolist (c '(indented-heading empty-heading malformed-priority-cookie
+ level2-done-without-closed))
+ (should-not (member c checkers)))))
+
(provide 'test-lint-org)
;;; test-lint-org.el ends here
+
+;;; ---------------------------------------------------------------------------
+;;; task-missing-last-reviewed (claude-rules/todo-format.md)
+
+(ert-deftest lo-task-without-last-reviewed-is-judgment ()
+ "An open level-2 task with no :LAST_REVIEWED: is flagged."
+ (let* ((out (lo-test--run "* Open Work\n** TODO [#B] A task :feature:\nBody.\n"))
+ (js (lo-test--judgments (plist-get out :issues))))
+ (should (memq 'task-missing-last-reviewed (lo-test--checkers js)))))
+
+(ert-deftest lo-task-with-last-reviewed-is-clean ()
+ "A task carrying the property is not flagged."
+ (let* ((out (lo-test--run (concat "* Open Work\n** TODO [#B] A task :feature:\n"
+ ":PROPERTIES:\n:LAST_REVIEWED: 2026-07-23\n:END:\n"
+ "Body.\n")))
+ (js (lo-test--judgments (plist-get out :issues))))
+ (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js)))))
+
+(ert-deftest lo-task-last-reviewed-accepts-org-timestamp ()
+ "The org-native [YYYY-MM-DD Day] form counts, matching the staleness script."
+ (let* ((out (lo-test--run (concat "* Open Work\n** TODO [#B] A task :feature:\n"
+ ":PROPERTIES:\n:LAST_REVIEWED: [2026-07-23 Thu]\n:END:\n")))
+ (js (lo-test--judgments (plist-get out :issues))))
+ (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js)))))
+
+(ert-deftest lo-done-task-without-last-reviewed-is-clean ()
+ "Completed tasks leave the review pool, so they are never flagged."
+ (let* ((out (lo-test--run (concat "* Open Work\n** DONE [#B] A task :feature:\n"
+ "CLOSED: [2026-07-23 Thu]\n")))
+ (js (lo-test--judgments (plist-get out :issues))))
+ (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js)))))
+
+(ert-deftest lo-subtask-without-last-reviewed-is-clean ()
+ "Only level-2 tasks are in the review pool; deeper headings are not."
+ (let* ((out (lo-test--run (concat "* Open Work\n** TODO [#B] Parent :feature:\n"
+ ":PROPERTIES:\n:LAST_REVIEWED: 2026-07-23\n:END:\n"
+ "*** TODO A sub-task\n")))
+ (js (lo-test--judgments (plist-get out :issues))))
+ (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js)))))
+
+(ert-deftest lo-cookieless-task-without-last-reviewed-is-clean ()
+ "The staleness script selects on a priority cookie, so match that scope."
+ (let* ((out (lo-test--run "* Open Work\n** TODO Manual testing and validation\n"))
+ (js (lo-test--judgments (plist-get out :issues))))
+ (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js)))))
+
+(ert-deftest lo-verify-task-without-last-reviewed-is-judgment ()
+ "VERIFY is in the review pool too."
+ (let* ((out (lo-test--run "* Open Work\n** VERIFY [#B] Waiting on Craig\n"))
+ (js (lo-test--judgments (plist-get out :issues))))
+ (should (memq 'task-missing-last-reviewed (lo-test--checkers js)))))
+
+;;; ---------------------------------------------------------------------------
+;;; todo-format checkers skip docs/specs/ files (claude-rules/todo-format.md)
+;;
+;; The four todo-format-family checkers encode todo.org completion conventions.
+;; A spec legitimately uses ** DONE <decision> with no CLOSED cookie and
+;; ** <dated> — <who> review-history headings, so those checkers misfire on
+;; every spec. They must skip any file under a docs/specs/ path segment.
+
+(defun lo-test--run-at (relpath content)
+ "Write CONTENT to <tmpdir>/RELPATH, run lint on it, return :issues.
+RELPATH is a relative path (may contain slashes) so a docs/specs/ segment
+can be exercised — the checkers key on the file's path, not just its name."
+ (let* ((root (make-temp-file "lo-test-root-" t))
+ (file (expand-file-name relpath root)))
+ (make-directory (file-name-directory file) t)
+ (unwind-protect
+ (progn
+ (with-temp-file file (insert content))
+ (lo-test--reset)
+ (lo-process-file file)
+ (prog1 (list :issues lo-issues)
+ (lo-test--drop-buffer file)))
+ (delete-directory root t))))
+
+(defconst lo-test--spec-decisions
+ "* Decisions [1/1]\n** DONE Some decision\n- Context: x\n"
+ "A spec Decisions section: a level-2 DONE with no CLOSED cookie.")
+
+(defconst lo-test--spec-history
+ "* Review history\n** 2026-07-14 Tue @ 02:03:28 -0500 — Claude — responder\n- What: x\n"
+ "A spec review-history section: a level-2 dated header.")
+
+(ert-deftest lo-todo-checkers-fire-on-a-normal-org-file ()
+ "Baseline: the checkers DO fire on a non-spec path (the bug is scope, not silence)."
+ (let* ((out (lo-test--run-at "todo.org" lo-test--spec-decisions))
+ (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues)))))
+ (should (memq 'level2-done-without-closed cs))))
+
+(ert-deftest lo-level2-done-without-closed-skips-specs ()
+ (let* ((out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" lo-test--spec-decisions))
+ (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues)))))
+ (should-not (memq 'level2-done-without-closed cs))))
+
+(ert-deftest lo-level2-dated-header-skips-specs ()
+ (let* ((out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" lo-test--spec-history))
+ (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues)))))
+ (should-not (memq 'level-2-dated-header cs))))
+
+(ert-deftest lo-dated-log-active-timestamp-skips-specs ()
+ (let* ((c "* History\n** 2026-07-14 Tue @ 02:03:28 -0500 — did a thing\nSCHEDULED: <2026-07-20 Mon>\n")
+ (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c))
+ (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues)))))
+ (should-not (memq 'dated-log-heading-active-timestamp cs))))
+
+(ert-deftest lo-subtask-done-not-dated-skips-specs ()
+ (let* ((c "* Work\n** TODO Parent\n*** DONE A sub-decision\n")
+ (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c))
+ (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues)))))
+ (should-not (memq 'subtask-done-not-dated cs))))
+
+(ert-deftest lo-link-checks-still-fire-on-specs ()
+ "Only the todo-format family is scoped out; a broken link in a spec still flags."
+ (let* ((c "* X\n[[file:does-not-exist-xyz.org][link]]\n")
+ (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c))
+ (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues)))))
+ (should (memq 'link-to-local-file cs))))
+
+(ert-deftest lo-task-missing-last-reviewed-skips-specs ()
+ "The fifth todo-format checker (added 2026-07-23) skips specs too — a spec's
+phases section may carry ** TODO [#x] items that aren't backlog tasks."
+ (let* ((c "* Implementation phases\n** TODO [#B] Phase one\nBody.\n")
+ (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c))
+ (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues)))))
+ (should-not (memq 'task-missing-last-reviewed cs)))
+ ;; And still fires on a normal file.
+ (let* ((c "* Work\n** TODO [#B] Real backlog task\nBody.\n")
+ (out (lo-test--run-at "todo.org" c))
+ (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues)))))
+ (should (memq 'task-missing-last-reviewed cs))))
diff --git a/claude-templates/.ai/scripts/tests/test-todo-cleanup.el b/claude-templates/.ai/scripts/tests/test-todo-cleanup.el
index ad9260b..1e964b3 100644
--- a/claude-templates/.ai/scripts/tests/test-todo-cleanup.el
+++ b/claude-templates/.ai/scripts/tests/test-todo-cleanup.el
@@ -30,16 +30,22 @@
;;; Harness
(defun tc-test--reset (&optional check)
- (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-issues nil
+ (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-archived-to-file 0 tc-issues nil
+ tc-sealed 0 tc-seal nil tc-convert-subtasks nil
tc-check-only (and check t)
tc-archive-done t tc-sync-child-priority nil
- tc-current-file nil))
+ tc-current-file nil
+ ;; Aging step OFF by default so the in-file-move tests are unaffected by
+ ;; the wall clock; the aging harness re-enables it with fixed params.
+ tc-archive-retain-days nil tc-archive-reference-date nil tc-archive-file nil))
(defun tc-test--reset-sync (&optional check)
- (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-issues nil
+ (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-archived-to-file 0 tc-issues nil
+ tc-sealed 0 tc-seal nil
tc-check-only (and check t)
tc-archive-done nil tc-sync-child-priority t
- tc-current-file nil))
+ tc-current-file nil
+ tc-archive-retain-days nil tc-archive-reference-date nil tc-archive-file nil))
(defun tc-test--drop-buffer (file)
(let ((buf (find-buffer-visiting file)))
@@ -355,6 +361,207 @@ from the heading line through (not including) the next level-1 heading or EOF."
(should (tc-test--has (plist-get out :report) "skipped"))))
;;; ---------------------------------------------------------------------------
+;;; --archive-done file-aging: keep last week in-file, move older to task-archive
+
+(defun tc-test--age (content &optional opts)
+ "Run `--archive-done' with the file-aging step enabled.
+OPTS is a plist: :retain (days; default 7, may be nil to disable), :ref
+\(YEAR MONTH DAY reference date), :runs (default 1), :check. Writes CONTENT to a
+temp todo file and points `tc-archive-file' at a not-yet-existing temp archive.
+Returns a plist: :result (todo contents), :archive (archive-file contents or
+nil), :archived (in-file move count), :to-file (aged count), :issues — all from
+the last run."
+ (let* ((retain (if (plist-member opts :retain) (plist-get opts :retain) 7))
+ (ref (plist-get opts :ref))
+ (runs (or (plist-get opts :runs) 1))
+ (check (plist-get opts :check))
+ (todo (make-temp-file "tc-age-todo-" nil ".org"))
+ (adir (make-temp-file "tc-age-arch-" t))
+ (afile (expand-file-name "task-archive.org" adir))
+ last)
+ (unwind-protect
+ (progn
+ (with-temp-file todo (insert content))
+ (dotimes (_ runs)
+ (tc-test--reset check)
+ (setq tc-archive-retain-days retain
+ tc-archive-reference-date ref
+ tc-archive-file afile)
+ (tc-process-file todo)
+ (setq last (list :archived tc-archived :to-file tc-archived-to-file
+ :issues tc-issues))
+ (tc-test--drop-buffer todo))
+ (append
+ last
+ (list :result (with-temp-buffer (insert-file-contents todo) (buffer-string))
+ :archive (and (file-readable-p afile)
+ (with-temp-buffer (insert-file-contents afile)
+ (buffer-string))))))
+ (tc-test--drop-buffer todo)
+ (delete-file todo)
+ (delete-directory adir t))))
+
+;; Reference "today" for these fixtures is 2026-06-29; with retain 7 the cutoff
+;; is 2026-06-22, so a task closed on or after 2026-06-22 stays in-file.
+(defconst tc-test--age-resolved "\
+* Age Open Work
+** TODO [#A] still open
+* Age Resolved
+** DONE [#B] recent within window
+CLOSED: [2026-06-25 Thu]
+recent body
+** DONE [#C] old beyond window
+CLOSED: [2026-05-01 Fri]
+old body line
+** CANCELLED [#C] old cancelled too
+CLOSED: [2026-04-15 Wed]
+** DONE [#B] exactly at cutoff stays
+CLOSED: [2026-06-22 Sun]
+** DONE [#C] undated no-date archived
+no closed date in this body
+")
+
+(defconst tc-test--age-straggler "\
+* Age Open Work
+** TODO [#A] still open
+** DONE [#C] old straggler
+CLOSED: [2026-03-01 Sun]
+straggler body
+* Age Resolved
+** DONE [#B] recent stays
+CLOSED: [2026-06-26 Fri]
+")
+
+(ert-deftest tc-age-moves-old-and-undated-resolved ()
+ "Normal: closed-beyond-window AND undated subtrees leave the file; only those
+closed within the window (cutoff inclusive) stay."
+ (let* ((out (tc-test--age tc-test--age-resolved '(:ref (2026 6 29))))
+ (resolved (tc-test--section (plist-get out :result) "Age Resolved"))
+ (arch (plist-get out :archive)))
+ (should (= 3 (plist-get out :to-file)))
+ (should-not (tc-test--has resolved "old beyond window"))
+ (should-not (tc-test--has resolved "old cancelled too"))
+ (should-not (tc-test--has resolved "undated no-date archived"))
+ (should (tc-test--has resolved "recent within window"))
+ (should (tc-test--has resolved "exactly at cutoff stays"))
+ (should arch)
+ (should (tc-test--has arch "Resolved (archived)"))
+ (should (tc-test--has arch "old beyond window"))
+ (should (tc-test--has arch "old body line"))
+ (should (tc-test--has arch "old cancelled too"))
+ (should (tc-test--has arch "undated no-date archived"))
+ (should-not (tc-test--has arch "recent within window"))))
+
+(ert-deftest tc-age-disabled-when-retain-nil ()
+ "Boundary: nil retain disables the aging step entirely (legacy behavior)."
+ (let ((out (tc-test--age tc-test--age-resolved '(:retain nil :ref (2026 6 29)))))
+ (should (= 0 (plist-get out :to-file)))
+ (should (equal tc-test--age-resolved (plist-get out :result)))
+ (should-not (plist-get out :archive))))
+
+(ert-deftest tc-age-is-idempotent ()
+ "Boundary: a second run finds nothing new to age; the todo file is stable."
+ (let ((once (tc-test--age tc-test--age-resolved '(:ref (2026 6 29) :runs 1)))
+ (twice (tc-test--age tc-test--age-resolved '(:ref (2026 6 29) :runs 2))))
+ (should (equal (plist-get once :result) (plist-get twice :result)))
+ (should (= 0 (plist-get twice :to-file)))))
+
+(ert-deftest tc-age-check-mode-previews-without-writing ()
+ "Boundary: --check reports the aged count but writes neither file."
+ (let ((out (tc-test--age tc-test--age-resolved '(:ref (2026 6 29) :check t))))
+ (should (= 3 (plist-get out :to-file)))
+ (should (equal tc-test--age-resolved (plist-get out :result)))
+ (should-not (plist-get out :archive))))
+
+(ert-deftest tc-age-straggler-moves-through-to-archive ()
+ "Normal: an old-dated DONE in Open Work moves to Resolved then ages out in one run."
+ (let* ((out (tc-test--age tc-test--age-straggler '(:ref (2026 6 29))))
+ (open (tc-test--section (plist-get out :result) "Age Open Work"))
+ (resolved (tc-test--section (plist-get out :result) "Age Resolved"))
+ (arch (plist-get out :archive)))
+ (should-not (tc-test--has open "old straggler"))
+ (should-not (tc-test--has resolved "old straggler"))
+ (should (tc-test--has arch "old straggler"))
+ (should (tc-test--has arch "straggler body"))
+ (should (tc-test--has resolved "recent stays"))
+ (should (= 1 (plist-get out :archived)))
+ (should (= 1 (plist-get out :to-file)))))
+
+(ert-deftest tc-age-append-preserves-existing-archive ()
+ "Error/edge: appending to a populated archive keeps prior entries and one scaffold."
+ (let* ((adir (make-temp-file "tc-arch-" t))
+ (afile (expand-file-name "task-archive.org" adir)))
+ (unwind-protect
+ (progn
+ (tc--append-subtrees-to-archive-file afile (list "** DONE one\n"))
+ (tc--append-subtrees-to-archive-file afile (list "** DONE two\n"))
+ (let ((content (with-temp-buffer (insert-file-contents afile)
+ (buffer-string)))
+ (n 0) (start 0))
+ (should (tc-test--has content "** DONE one"))
+ (should (tc-test--has content "** DONE two"))
+ (should (tc-test--before-p content "** DONE one" "** DONE two"))
+ (while (string-match "\\* Resolved (archived)" content start)
+ (setq n (1+ n) start (match-end 0)))
+ (should (= 1 n))))
+ (delete-directory adir t))))
+
+;;; ---------------------------------------------------------------------------
+;;; --archive-done aging: the archive follows the todo file's gitignore status
+
+(defun tc-test--age-in-git-repo (gitignore-todo)
+ "Init a temp git repo, write todo.org with an old Resolved entry, optionally
+gitignore todo.org, then run `--archive-done' aging with the DEFAULT archive path
+(archive/task-archive.org beside the todo file). Return a plist: :gitignore (final
+.gitignore contents or nil), :archive-ignored (whether git ignores the archive),
+:archive-exists."
+ (let* ((root (make-temp-file "tc-git-" t))
+ ;; Private backup dir: this helper writes a file literally named
+ ;; todo.org and runs a real (non-check) pass, so without this its
+ ;; backup lands in the shared temp dir under the exact production
+ ;; name and is indistinguishable from a real one.
+ (temporary-file-directory
+ (file-name-as-directory (make-temp-file "tc-git-bk-" t)))
+ (todo (expand-file-name "todo.org" root))
+ (archive (expand-file-name "archive/task-archive.org" root))
+ (gi (expand-file-name ".gitignore" root)))
+ (unwind-protect
+ (let ((default-directory root))
+ (call-process "git" nil nil nil "init" "-q")
+ (with-temp-file todo (insert tc-test--age-resolved))
+ (when gitignore-todo (with-temp-file gi (insert "/todo.org\n")))
+ (tc-test--reset nil)
+ (setq tc-archive-retain-days 7
+ tc-archive-reference-date '(2026 6 29)
+ tc-archive-file nil) ; default path, beside the todo file
+ (tc-process-file todo)
+ (tc-test--drop-buffer todo)
+ (list :gitignore (and (file-readable-p gi)
+ (with-temp-buffer (insert-file-contents gi)
+ (buffer-string)))
+ :archive-ignored
+ (eq 0 (call-process "git" nil nil nil "check-ignore" "-q" archive))
+ :archive-exists (file-readable-p archive)))
+ (delete-directory root t)
+ (delete-directory temporary-file-directory t))))
+
+(ert-deftest tc-age-self-protect-gitignores-archive-when-todo-ignored ()
+ "When the todo file is gitignored, the aged-out archive is added to .gitignore
+so it inherits the same privacy."
+ (let ((out (tc-test--age-in-git-repo t)))
+ (should (plist-get out :archive-exists))
+ (should (string-match-p "task-archive" (or (plist-get out :gitignore) "")))
+ (should (plist-get out :archive-ignored))))
+
+(ert-deftest tc-age-self-protect-leaves-tracked-todo-archive-tracked ()
+ "When the todo file is tracked, the archive is not gitignored — no .gitignore
+entry is added for it."
+ (let ((out (tc-test--age-in-git-repo nil)))
+ (should (plist-get out :archive-exists))
+ (should-not (plist-get out :archive-ignored))
+ (should-not (string-match-p "task-archive" (or (plist-get out :gitignore) "")))))
+
+;;; ---------------------------------------------------------------------------
;;; Realistic synthetic sample (committed under fixtures/)
(defun tc-test--sample-file ()
@@ -380,6 +587,95 @@ from the heading line through (not including) the next level-1 heading or EOF."
(should (> (plist-get out :archived) 0)))))
;;; ---------------------------------------------------------------------------
+;;; --archive-done retention default
+
+(ert-deftest tc-archive-retain-default-is-one-month ()
+ "The shipped retention default is one month (31 days), not the legacy 7.
+The defvar initializes from this defconst; the live var itself is mutated by
+other tests, so the immutable defconst is the stable contract to pin."
+ (should (= 31 tc-archive-retain-days-default)))
+
+;;; ---------------------------------------------------------------------------
+;;; --seal: rename the working archive to resolved-YYYY-MM-DD.org
+
+(defun tc-test--seal (&optional opts)
+ "Run `--seal' against a temp todo file with a temp archive dir.
+OPTS is a plist: :archive-content (seed task-archive.org with this; nil = no
+working archive), :ref (YEAR MONTH DAY seal date; default (2026 7 18)),
+:check, :presealed (also create resolved-<ref>.org first, to test collision).
+Returns a plist: :sealed count, :issues, :working-exists, :sealed-exists,
+:sealed-name, :report."
+ (let* ((ref (or (plist-get opts :ref) '(2026 7 18)))
+ (check (plist-get opts :check))
+ (archive-content (plist-get opts :archive-content))
+ (todo (make-temp-file "tc-seal-todo-" nil ".org"))
+ (adir (make-temp-file "tc-seal-arch-" t))
+ (afile (expand-file-name "task-archive.org" adir))
+ (sealed-name (format "resolved-%04d-%02d-%02d.org"
+ (nth 0 ref) (nth 1 ref) (nth 2 ref)))
+ (sealed (expand-file-name sealed-name adir)))
+ (unwind-protect
+ (progn
+ (with-temp-file todo (insert "* Open Work\n** TODO [#A] live\n"))
+ (when archive-content (with-temp-file afile (insert archive-content)))
+ (when (plist-get opts :presealed)
+ (with-temp-file sealed (insert "pre-existing seal\n")))
+ (tc-test--reset check)
+ ;; Set every mode flag explicitly: tc-test--reset leaves
+ ;; tc-convert-subtasks untouched, so a convert test running earlier in
+ ;; the suite would otherwise still own the dispatch and run convert.
+ (setq tc-archive-done nil tc-sync-child-priority nil
+ tc-convert-subtasks nil tc-seal t tc-sealed 0
+ tc-archive-reference-date ref
+ tc-archive-file afile)
+ (let ((report (with-output-to-string (tc-process-file todo) (tc-emit-report))))
+ (tc-test--drop-buffer todo)
+ (list :sealed tc-sealed
+ :issues tc-issues
+ :working-exists (file-readable-p afile)
+ :sealed-exists (file-readable-p sealed)
+ :sealed-name sealed-name
+ :report report)))
+ (tc-test--drop-buffer todo)
+ (delete-file todo)
+ (delete-directory adir t))))
+
+(ert-deftest tc-seal-renames-working-archive-to-dated-file ()
+ "Normal: --seal renames task-archive.org to resolved-<seal-date>.org."
+ (let ((out (tc-test--seal '(:archive-content "* Resolved (archived)\n** DONE old\n"
+ :ref (2026 7 18)))))
+ (should (= 1 (plist-get out :sealed)))
+ (should-not (plist-get out :working-exists))
+ (should (plist-get out :sealed-exists))
+ (should (equal "resolved-2026-07-18.org" (plist-get out :sealed-name)))
+ (should (tc-test--has (plist-get out :report) "sealed task-archive.org → resolved-2026-07-18.org"))))
+
+(ert-deftest tc-seal-nothing-to-seal-is-a-reported-noop ()
+ "Boundary: no working archive present — reported no-op, nothing created."
+ (let ((out (tc-test--seal '(:ref (2026 7 18)))))
+ (should (= 0 (plist-get out :sealed)))
+ (should-not (plist-get out :sealed-exists))
+ (should (tc-test--has (plist-get out :report) "no working archive to seal"))))
+
+(ert-deftest tc-seal-check-mode-previews-without-renaming ()
+ "Boundary: --check reports the seal but leaves the working archive in place."
+ (let ((out (tc-test--seal '(:archive-content "* Resolved (archived)\n"
+ :ref (2026 7 18) :check t))))
+ (should (= 1 (plist-get out :sealed)))
+ (should (plist-get out :working-exists))
+ (should-not (plist-get out :sealed-exists))
+ (should (tc-test--has (plist-get out :report) "would seal"))))
+
+(ert-deftest tc-seal-refuses-to-clobber-existing-sealed-file ()
+ "Error: resolved-<today>.org already exists — refuse, leave both files intact."
+ (let ((out (tc-test--seal '(:archive-content "* Resolved (archived)\n"
+ :ref (2026 7 18) :presealed t))))
+ (should (= 0 (plist-get out :sealed)))
+ (should (plist-get out :working-exists))
+ (should (plist-get out :sealed-exists))
+ (should (tc-test--has (plist-get out :report) "already exists"))))
+
+;;; ---------------------------------------------------------------------------
;;; Sync-child-priority harness + fixtures
(defun tc-test--sync (content &optional runs check)
@@ -570,5 +866,311 @@ in ISSUES, in document order."
(should (= 2 (plist-get once :bumped)))
(should (= 2 (plist-get twice :bumped)))))
+;;; ---------------------------------------------------------------------------
+;;; --convert-subtasks harness + tests
+
+(defun tc-test--reset-convert (&optional check)
+ (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-converted 0 tc-archived-to-file 0
+ tc-issues nil tc-sealed 0 tc-seal nil
+ tc-check-only (and check t)
+ tc-archive-done nil tc-sync-child-priority nil tc-convert-subtasks t
+ tc-current-file nil
+ tc-archive-retain-days nil tc-archive-reference-date nil tc-archive-file nil))
+
+(defun tc-test--convert (content &optional runs check)
+ "Write CONTENT to a temp .org file, run `--convert-subtasks' RUNS times (default 1).
+Return a plist: :result final file contents, :converted count from the last run,
+:issues from the last run. CHECK non-nil ⇒ --check (preview, no writes)."
+ (let ((file (make-temp-file "tc-test-" nil ".org"))
+ last-converted last-issues)
+ (unwind-protect
+ (progn
+ (with-temp-file file (insert content))
+ (dotimes (_ (or runs 1))
+ (tc-test--reset-convert check)
+ (tc-process-file file)
+ (setq last-converted tc-converted last-issues tc-issues)
+ (tc-test--drop-buffer file))
+ (list :result (with-temp-buffer (insert-file-contents file)
+ (buffer-string))
+ :converted last-converted
+ :issues last-issues))
+ (tc-test--drop-buffer file)
+ (delete-file file))))
+
+;; The UTC offset in a converted header is the test machine's local offset for
+;; that date, so assertions match it as `[-+]NNNN' rather than a fixed value —
+;; the mode's job is to emit a well-formed offset, not to run in one timezone.
+
+(defconst tc-test--convert-timed
+ "* Project Open Work
+** TODO [#B] Parent task
+*** DONE [#C] F12 opens the terminal :feature:quick:
+CLOSED: [2026-06-27 Sat 12:50]
+Verified live: docks, toggles, colors clean.
+")
+
+(ert-deftest tc-convert-timed-subtask-normal ()
+ "Normal: a timed CLOSED close becomes a dated header, keyword/priority/tags/CLOSED gone."
+ (let* ((out (tc-test--convert tc-test--convert-timed))
+ (res (plist-get out :result)))
+ (should (= 1 (plist-get out :converted)))
+ (should (string-match-p
+ "^\\*\\*\\* 2026-06-27 Sat @ 12:50:00 [-+][0-9]\\{4\\} F12 opens the terminal$"
+ res))
+ (should-not (string-match-p "CLOSED:" res))
+ (should-not (string-match-p "DONE" res))
+ (should (string-match-p "Verified live: docks, toggles, colors clean\\." res))
+ (should (string-match-p "^\\*\\* TODO \\[#B\\] Parent task$" res))))
+
+(defconst tc-test--convert-dateonly
+ "* Project Open Work
+** PROJECT [#B] Parent
+**** DONE [#B] Write full spec :refactor:
+CLOSED: [2026-05-04 Mon]
+Body.
+")
+
+(ert-deftest tc-convert-dateonly-boundary-midnight ()
+ "Boundary: a date-only CLOSED (no time) yields 00:00:00, at level 4."
+ (let ((res (plist-get (tc-test--convert tc-test--convert-dateonly) :result)))
+ (should (string-match-p
+ "^\\*\\*\\*\\* 2026-05-04 Mon @ 00:00:00 [-+][0-9]\\{4\\} Write full spec$"
+ res))
+ (should-not (string-match-p "CLOSED:" res))))
+
+(defconst tc-test--convert-level2
+ "* Project Open Work
+** DONE [#B] Top-level task
+CLOSED: [2026-06-01 Mon 09:00]
+Body.
+")
+
+(ert-deftest tc-convert-leaves-level-2-alone-boundary ()
+ "Boundary: a level-2 DONE task is a top-level task, not a sub-task — untouched."
+ (let ((out (tc-test--convert tc-test--convert-level2)))
+ (should (= 0 (plist-get out :converted)))
+ (should (equal tc-test--convert-level2 (plist-get out :result)))))
+
+(ert-deftest tc-convert-idempotent-boundary ()
+ "Boundary: a second run over an already-dated entry converts nothing new."
+ (let ((once (tc-test--convert tc-test--convert-timed 1))
+ (twice (tc-test--convert tc-test--convert-timed 2)))
+ (should (equal (plist-get once :result) (plist-get twice :result)))
+ (should (= 0 (plist-get twice :converted)))))
+
+(defconst tc-test--convert-nested
+ "* Project Open Work
+** TODO [#B] Parent
+*** DONE Outer sub :feature:
+CLOSED: [2026-06-10 Wed 08:15]
+**** DONE Inner sub
+CLOSED: [2026-06-09 Tue 07:00]
+Inner body.
+")
+
+(ert-deftest tc-convert-nested-done-subtasks-boundary ()
+ "Boundary: a done sub-task nested under a done sub-task — both convert."
+ (let* ((out (tc-test--convert tc-test--convert-nested))
+ (res (plist-get out :result)))
+ (should (= 2 (plist-get out :converted)))
+ (should (string-match-p
+ "^\\*\\*\\* 2026-06-10 Wed @ 08:15:00 [-+][0-9]\\{4\\} Outer sub$" res))
+ (should (string-match-p
+ "^\\*\\*\\*\\* 2026-06-09 Tue @ 07:00:00 [-+][0-9]\\{4\\} Inner sub$" res))
+ (should-not (string-match-p "CLOSED:" res))))
+
+(defconst tc-test--convert-cancelled
+ "* Project Open Work
+** TODO [#B] Parent
+*** CANCELLED [#C] Abandoned idea :feature:
+CLOSED: [2026-06-15 Mon 10:00]
+")
+
+(ert-deftest tc-convert-cancelled-subtask-boundary ()
+ "Boundary: a CANCELLED sub-task converts too (terminal state)."
+ (let ((res (plist-get (tc-test--convert tc-test--convert-cancelled) :result)))
+ (should (string-match-p
+ "^\\*\\*\\* 2026-06-15 Mon @ 10:00:00 [-+][0-9]\\{4\\} Abandoned idea$" res))
+ (should-not (string-match-p "CANCELLED" res))))
+
+(defconst tc-test--convert-noclosed
+ "* Project Open Work
+** TODO [#B] Parent
+*** DONE Orphan with no closed date
+Body only.
+")
+
+(ert-deftest tc-convert-skips-subtask-without-closed-error ()
+ "Error: a done sub-task with no parseable CLOSED is flagged and left unchanged."
+ (let ((out (tc-test--convert tc-test--convert-noclosed)))
+ (should (= 0 (plist-get out :converted)))
+ (should (equal tc-test--convert-noclosed (plist-get out :result)))
+ (should (cl-some (lambda (i) (eq (plist-get i :kind) 'convert-skip))
+ (plist-get out :issues)))))
+
+(ert-deftest tc-convert-check-mode-previews-without-writing ()
+ "Check mode reports the conversion but writes nothing."
+ (let ((out (tc-test--convert tc-test--convert-timed 1 t)))
+ (should (= 1 (plist-get out :converted)))
+ (should (equal tc-test--convert-timed (plist-get out :result)))
+ (should (cl-some (lambda (i) (eq (plist-get i :kind) 'convert-would))
+ (plist-get out :issues)))))
+
+(defconst tc-test--convert-closed-with-deadline
+ "* Project Open Work
+** TODO [#B] Parent task
+*** DONE [#C] Ship the panel :feature:
+CLOSED: [2026-06-27 Sat 12:50] DEADLINE: <2026-06-30 Tue>
+Body line.
+")
+
+(ert-deftest tc-convert-strips-deadline-sharing-the-planning-line-boundary ()
+ "Boundary: a DEADLINE sharing the CLOSED planning line goes too — a dated-log
+entry carries no active planning timestamp (todo-format.md). Body survives."
+ (let* ((out (tc-test--convert tc-test--convert-closed-with-deadline))
+ (res (plist-get out :result)))
+ (should (= 1 (plist-get out :converted)))
+ (should (string-match-p
+ "^\\*\\*\\* 2026-06-27 Sat @ 12:50:00 [-+][0-9]\\{4\\} Ship the panel$"
+ res))
+ (should-not (string-match-p "CLOSED:" res))
+ (should-not (string-match-p "DEADLINE:" res))
+ (should (string-match-p "^Body line\\.$" res))))
+
+(defconst tc-test--convert-closed-and-scheduled-separate-lines
+ "* Project Open Work
+** TODO [#B] Parent task
+*** DONE [#C] Book the venue :feature:
+CLOSED: [2026-06-27 Sat 12:50]
+SCHEDULED: <2026-06-20 Sat>
+Body line.
+")
+
+(ert-deftest tc-convert-strips-scheduled-on-its-own-line ()
+ "Normal (the home bug): a SCHEDULED planning line on its own — the completion
+rewrite dropped keyword/priority/tags but left the SCHEDULED, pinning the dated
+entry to the agenda as weeks-overdue. Both planning lines go; body survives."
+ (let* ((out (tc-test--convert tc-test--convert-closed-and-scheduled-separate-lines))
+ (res (plist-get out :result)))
+ (should (= 1 (plist-get out :converted)))
+ (should (string-match-p
+ "^\\*\\*\\* 2026-06-27 Sat @ 12:50:00 [-+][0-9]\\{4\\} Book the venue$"
+ res))
+ (should-not (string-match-p "CLOSED:" res))
+ (should-not (string-match-p "SCHEDULED:" res))
+ (should (string-match-p "^Body line\\.$" res))))
+
+(defconst tc-test--convert-scheduled-in-body-prose
+ "* Project Open Work
+** TODO [#B] Parent task
+*** DONE [#C] Note the mechanism :feature:
+CLOSED: [2026-06-27 Sat 12:50]
+An active SCHEDULED: <2026-06-20 Sat> in prose must survive.
+")
+
+(ert-deftest tc-convert-leaves-planning-shaped-body-prose-alone ()
+ "Boundary: a planning-shaped token inside body prose (not a canonical planning
+line) is left untouched — the strip stops at the first non-planning line."
+ (let* ((out (tc-test--convert tc-test--convert-scheduled-in-body-prose))
+ (res (plist-get out :result)))
+ (should (= 1 (plist-get out :converted)))
+ (should-not (string-match-p "CLOSED:" res))
+ (should (string-match-p "An active SCHEDULED: <2026-06-20 Sat> in prose must survive\\." res))))
+
(provide 'test-todo-cleanup)
;;; test-todo-cleanup.el ends here
+
+;;; ---------------------------------------------------------------------------
+;;; Backup before mutating (parity with lint-org.el / wrap-org-table.el)
+;;
+;; todo-cleanup rewrites todo.org in place and left no copy behind, while both
+;; sibling org-mutators back up to /tmp first. It is also the one that runs most
+;; often (every wrap, every sentry cycle). Emacs's own backup does not fire under
+;; --batch -q, so there was genuinely no undo short of git.
+
+(ert-deftest tc-backup-written-before-a-real-mutation ()
+ "A real (non-check) run leaves a copy holding the pre-edit content.
+
+`temporary-file-directory' is rebound to a private dir for the duration: the
+backup name derives from the *file's* basename, and the real todo.org shares
+that basename, so a live sentry run writing /tmp/todo.org.before-todo-cleanup.*
+would otherwise be indistinguishable from this test's own artifact. The first
+version of this test globbed the shared /tmp and passed only until a real run
+created one (2026-07-24)."
+ (let* ((dir (make-temp-file "tc-backup-" t))
+ (bdir (file-name-as-directory (make-temp-file "tc-bk-" t)))
+ (file (expand-file-name "todo.org" dir))
+ (before "* P Open Work\n** TODO [#B] parent\n*** DONE a subtask\nCLOSED: [2026-07-01 Tue]\n"))
+ (unwind-protect
+ (progn
+ (with-temp-file file (insert before))
+ (let ((tc-check-only nil)
+ (tc-convert-subtasks t)
+ (temporary-file-directory bdir))
+ (tc-process-file file))
+ (let ((backups (file-expand-wildcards
+ (concat bdir "todo.org.before-todo-cleanup.*"))))
+ (should backups)
+ (should (string-match-p
+ "a subtask"
+ (with-temp-buffer (insert-file-contents (car backups))
+ (buffer-string))))))
+ (delete-directory dir t)
+ (delete-directory bdir t))))
+
+(ert-deftest tc-no-backup-in-check-mode ()
+ "--check writes nothing, so it must not leave a backup either.
+Uses a private `temporary-file-directory' for the same isolation reason."
+ (let* ((dir (make-temp-file "tc-backup-" t))
+ (bdir (file-name-as-directory (make-temp-file "tc-bk-" t)))
+ (file (expand-file-name "todo.org" dir)))
+ (unwind-protect
+ (progn
+ (with-temp-file file
+ (insert "* P Open Work\n** TODO [#B] parent\n*** DONE sub\nCLOSED: [2026-07-01 Tue]\n"))
+ (let ((tc-check-only t)
+ (tc-convert-subtasks t)
+ (temporary-file-directory bdir))
+ (tc-process-file file))
+ (should-not (file-expand-wildcards
+ (concat bdir "todo.org.before-todo-cleanup.*"))))
+ (delete-directory dir t)
+ (delete-directory bdir t))))
+
+(ert-deftest tc-backup-never-overwrites-an-earlier-one ()
+ "Two invocations in the same second must not collapse to one backup.
+
+open-tasks.org runs --convert-subtasks then --archive-done back to back, each
+a sub-second batch run. With a second-resolution stamp and copy-file's
+OK-IF-ALREADY-EXISTS, the second invocation overwrote the first's backup with
+already-mutated content, so the true pre-session original was unrecoverable —
+the exact state the backup exists to preserve (found 2026-07-24 in review)."
+ (let* ((dir (make-temp-file "tc-collide-" t))
+ (bdir (file-name-as-directory (make-temp-file "tc-cbk-" t)))
+ (file (expand-file-name "todo.org" dir))
+ (original (concat "* P Open Work\n** TODO [#B] parent\n*** DONE sub\n"
+ "CLOSED: [2026-07-01 Tue]\n"
+ "* P Resolved\n** DONE [#C] old\nCLOSED: [2025-01-01 Wed]\n")))
+ (unwind-protect
+ (progn
+ (with-temp-file file (insert original))
+ ;; Two back-to-back invocations, as the shipped workflow does.
+ (let ((temporary-file-directory bdir))
+ (let ((tc-check-only nil) (tc-convert-subtasks t))
+ (tc-process-file file))
+ (let ((tc-check-only nil) (tc-convert-subtasks nil) (tc-archive-done t)
+ (tc-archive-retain-days nil))
+ (tc-process-file file)))
+ (let ((backups (file-expand-wildcards
+ (concat bdir "todo.org.before-todo-cleanup.*"))))
+ ;; Both invocations kept their own backup.
+ (should (= (length backups) 2))
+ ;; And one of them still holds the true original.
+ (should (cl-some (lambda (b)
+ (string= original
+ (with-temp-buffer (insert-file-contents b)
+ (buffer-string))))
+ backups))))
+ (delete-directory dir t)
+ (delete-directory bdir t))))
diff --git a/claude-templates/.ai/scripts/tests/test-wrap-org-table.el b/claude-templates/.ai/scripts/tests/test-wrap-org-table.el
index 8d1ecb6..0b3b375 100644
--- a/claude-templates/.ai/scripts/tests/test-wrap-org-table.el
+++ b/claude-templates/.ai/scripts/tests/test-wrap-org-table.el
@@ -186,3 +186,45 @@
(should (string-match-p "Prose before\\." content))
(should (string-match-p "Prose after\\." content))))
(delete-file file))))
+
+;;; ---------------------------------------------------------------------------
+;;; block safety — pipe lines inside #+begin_/#+end_ blocks are never tables
+
+(defconst wot-test--block-content
+ "#+begin_example
+| client |----->| server |
+| box | | box |
+#+end_example
+"
+ "An example block whose ASCII-art lines start with pipes.")
+
+(defun wot-test--process-content (content budget)
+ "Write CONTENT to a temp file, run `wot-process-file' at BUDGET, return result."
+ (let ((file (make-temp-file "wot-test" nil ".org")))
+ (unwind-protect
+ (progn
+ (with-temp-file file (insert content))
+ (wot-process-file file budget)
+ (with-temp-buffer (insert-file-contents file) (buffer-string)))
+ (delete-file file))))
+
+(ert-deftest wot-process-file-leaves-example-block-byte-identical ()
+ (let ((content (concat "* Diagram\n\n" wot-test--block-content)))
+ (should (equal (wot-test--process-content content 120) content))))
+
+(ert-deftest wot-process-file-reformats-table-but-not-block ()
+ (let* ((content (concat "* Doc\n\n" wot-test--block-content "\n"
+ wot-test--wide-input))
+ (result (wot-test--process-content content 40)))
+ (should (string-match-p (regexp-quote wot-test--block-content) result))
+ (should (string-match-p (regexp-quote wot-test--wide-expected) result))))
+
+(ert-deftest wot-process-file-skips-pipes-in-src-block ()
+ (let ((content "* Pipeline\n\n#+begin_src sh\n| sort\n| uniq -c\n#+end_src\n"))
+ (should (equal (wot-test--process-content content 120) content))))
+
+(ert-deftest wot-process-file-literal-inner-end-marker-stays-in-block ()
+ "A literal #+end_src quoted inside an example block must not close it."
+ (let ((content (concat "* Doc\n\n#+begin_example\n#+begin_src sh\nx\n"
+ "#+end_src\n| art |----| art |\n#+end_example\n")))
+ (should (equal (wot-test--process-content content 120) content))))
diff --git a/claude-templates/.ai/scripts/tests/test_apkg_to_orgdrill.py b/claude-templates/.ai/scripts/tests/test_apkg_to_orgdrill.py
new file mode 100644
index 0000000..6a95ea4
--- /dev/null
+++ b/claude-templates/.ai/scripts/tests/test_apkg_to_orgdrill.py
@@ -0,0 +1,301 @@
+"""Tests for apkg-to-orgdrill.py — the inverse of flashcard-to-anki.py.
+
+The converter reads an Anki .apkg (a zip holding collection.anki2 / .anki21
+sqlite) and emits an org-drill .org in the house canonical shape. It is
+stdlib-only (zipfile + sqlite3), so it imports directly — no genanki stub.
+
+The apkg schema these tests build by hand mirrors what genanki actually
+writes, confirmed against a real apkg generated from flashcard-to-anki.py:
+ - col.decks : JSON {did: {"name": ...}}, always including id-1 "Default"
+ - col.models : JSON {mid: {"name": ..., "flds": [{"name": "Front"}, ...]}}
+ - notes.flds : fields joined by \x1f; tags space-padded (" tag ")
+ - cards : nid -> did (the Default deck carries no cards)
+
+The round-trip test closes the loop through flashcard-to-anki.py's own
+parse(): original org -> forward parse tuples -> apkg fixture -> converter
+-> recovered org -> forward parse -> assert the (front, back, tag) tuples
+match. Only the apkg materialization is hand-built (the genanki boundary);
+everything else is the real code on both sides.
+"""
+from __future__ import annotations
+
+import importlib.util
+import json
+import sqlite3
+import sys
+import types
+import zipfile
+from pathlib import Path
+
+import pytest
+
+SCRIPTS = Path(__file__).resolve().parents[1]
+CONVERTER = SCRIPTS / "apkg-to-orgdrill.py"
+FORWARD = SCRIPTS / "flashcard-to-anki.py"
+
+
+def _load(path: Path, name: str, stub_genanki: bool = False):
+ if stub_genanki:
+ sys.modules.setdefault("genanki", types.ModuleType("genanki"))
+ spec = importlib.util.spec_from_file_location(name, path)
+ assert spec and spec.loader
+ module = importlib.util.module_from_spec(spec)
+ # Register before exec: @dataclass resolves cls.__module__ via sys.modules
+ # (Python 3.14), which is None for an unregistered importlib module.
+ sys.modules[name] = module
+ spec.loader.exec_module(module)
+ return module
+
+
+@pytest.fixture(scope="module")
+def conv():
+ return _load(CONVERTER, "apkg_to_orgdrill")
+
+
+@pytest.fixture(scope="module")
+def forward():
+ return _load(FORWARD, "flashcard_to_anki", stub_genanki=True)
+
+
+# --- fixture builder: write a genanki-shaped apkg by hand ------------------
+
+def _make_apkg(
+ path: Path,
+ decks: dict[int, str],
+ models: dict[int, list[str]],
+ notes: list[tuple[int, int, list[str], str]], # (nid, mid, fields, tag)
+ cards: list[tuple[int, int]], # (nid, did)
+ *,
+ media: str = "{}",
+) -> None:
+ """Materialize a minimal apkg matching genanki's collection.anki2 shape."""
+ col_dir = path.parent / f"{path.stem}-build"
+ col_dir.mkdir(parents=True, exist_ok=True)
+ db = col_dir / "collection.anki2"
+ if db.exists():
+ db.unlink()
+ con = sqlite3.connect(db)
+ con.execute("CREATE TABLE col (id INTEGER, decks TEXT, models TEXT)")
+ decks_json = {"1": {"name": "Default"}}
+ decks_json.update({str(did): {"name": name} for did, name in decks.items()})
+ models_json = {
+ str(mid): {"name": f"{decks.get(list(decks)[0], 'M')} model",
+ "flds": [{"name": n, "ord": i} for i, n in enumerate(flds)]}
+ for mid, flds in models.items()
+ }
+ con.execute("INSERT INTO col (id, decks, models) VALUES (1, ?, ?)",
+ (json.dumps(decks_json), json.dumps(models_json)))
+ con.execute("CREATE TABLE notes (id INTEGER, mid INTEGER, flds TEXT, tags TEXT)")
+ for nid, mid, fields, tag in notes:
+ con.execute("INSERT INTO notes (id, mid, flds, tags) VALUES (?, ?, ?, ?)",
+ (nid, mid, "\x1f".join(fields), f" {tag} " if tag else " "))
+ con.execute("CREATE TABLE cards (id INTEGER, nid INTEGER, did INTEGER)")
+ for i, (nid, did) in enumerate(cards):
+ con.execute("INSERT INTO cards (id, nid, did) VALUES (?, ?, ?)", (1000 + i, nid, did))
+ con.commit()
+ con.close()
+ with zipfile.ZipFile(path, "w") as z:
+ z.write(db, "collection.anki2")
+ z.writestr("media", media)
+
+
+# --- html_to_org_body ------------------------------------------------------
+
+def test_html_to_org_splits_br_into_lines(conv):
+ assert conv.html_to_org_body("one<br>two<br>three") == ["one", "two", "three"]
+
+
+def test_html_to_org_handles_br_variants(conv):
+ assert conv.html_to_org_body("a<br/>b<br />c<BR>d") == ["a", "b", "c", "d"]
+
+
+def test_html_to_org_unescapes_entities_amp_last(conv):
+ # Inverts escape_html (which escapes & first): &lt; &gt; &amp; -> < > &.
+ assert conv.html_to_org_body("x &lt;tag&gt; &amp; y") == ["x <tag> & y"]
+
+
+def test_html_to_org_preserves_a_literal_escaped_entity(conv):
+ # Forward-escaping the literal "&lt;" yields "&amp;lt;"; the inverse must
+ # recover "&lt;", not "<".
+ assert conv.html_to_org_body("&amp;lt;") == ["&lt;"]
+
+
+def test_html_to_org_strips_answer_hr(conv):
+ assert conv.html_to_org_body('front<hr id="answer">back') == ["front", "back"]
+
+
+def test_html_to_org_empty_back_is_empty(conv):
+ assert conv.html_to_org_body("") == []
+
+
+# --- read_apkg -------------------------------------------------------------
+
+def test_read_apkg_single_deck_recovers_front_back_tag_deck(conv, tmp_path):
+ apkg = tmp_path / "d.apkg"
+ _make_apkg(
+ apkg,
+ decks={20: "My Deck"},
+ models={9: ["Front", "Back"]},
+ notes=[(100, 9, ["Q1?", "A1.<br>line2"], "sec-one")],
+ cards=[(100, 20)],
+ )
+ recovered = conv.read_apkg(apkg)
+ assert len(recovered) == 1
+ note = recovered[0]
+ assert note.deck == "My Deck"
+ assert note.front == "Q1?"
+ assert note.back_html == "A1.<br>line2"
+ assert note.tag == "sec-one"
+
+
+def test_read_apkg_multiple_decks_grouped(conv, tmp_path):
+ apkg = tmp_path / "multi.apkg"
+ _make_apkg(
+ apkg,
+ decks={20: "Deck A", 21: "Deck B"},
+ models={9: ["Front", "Back"]},
+ notes=[(100, 9, ["QA?", "AA"], "ta"), (101, 9, ["QB?", "AB"], "tb")],
+ cards=[(100, 20), (101, 21)],
+ )
+ decks = {n.deck for n in conv.read_apkg(apkg)}
+ assert decks == {"Deck A", "Deck B"}
+
+
+def test_read_apkg_skips_default_deck_without_cards(conv, tmp_path):
+ apkg = tmp_path / "def.apkg"
+ _make_apkg(
+ apkg,
+ decks={20: "Real Deck"},
+ models={9: ["Front", "Back"]},
+ notes=[(100, 9, ["Q?", "A"], "t")],
+ cards=[(100, 20)],
+ )
+ assert {n.deck for n in conv.read_apkg(apkg)} == {"Real Deck"}
+
+
+def test_read_apkg_warns_and_skips_non_basic_model(conv, tmp_path, capsys):
+ apkg = tmp_path / "cloze.apkg"
+ _make_apkg(
+ apkg,
+ decks={20: "Cloze Deck"},
+ models={9: ["Text", "Extra"]}, # not Front/Back
+ notes=[(100, 9, ["some {{c1::text}}", "extra"], "t")],
+ cards=[(100, 20)],
+ )
+ recovered = conv.read_apkg(apkg)
+ assert recovered == []
+ assert "skip" in capsys.readouterr().err.lower()
+
+
+def test_read_apkg_reads_anki21_collection_name(conv, tmp_path):
+ # A .anki21 collection filename must be read the same as .anki2.
+ apkg = tmp_path / "new.apkg"
+ _make_apkg(
+ apkg,
+ decks={20: "Deck"},
+ models={9: ["Front", "Back"]},
+ notes=[(100, 9, ["Q?", "A"], "t")],
+ cards=[(100, 20)],
+ )
+ # Rewrite the zip renaming the collection member to .anki21.
+ with zipfile.ZipFile(apkg) as z:
+ data = z.read("collection.anki2")
+ media = z.read("media")
+ with zipfile.ZipFile(apkg, "w") as z:
+ z.writestr("collection.anki21", data)
+ z.writestr("media", media)
+ assert conv.read_apkg(apkg)[0].front == "Q?"
+
+
+def test_read_apkg_flags_media_reference(conv, tmp_path, capsys):
+ apkg = tmp_path / "media.apkg"
+ _make_apkg(
+ apkg,
+ decks={20: "Deck"},
+ models={9: ["Front", "Back"]},
+ notes=[(100, 9, ["Q?", 'see <img src="x.png">'], "t")],
+ cards=[(100, 20)],
+ )
+ conv.read_apkg(apkg)
+ assert "media" in capsys.readouterr().err.lower()
+
+
+# --- notes_to_org ----------------------------------------------------------
+
+def test_notes_to_org_emits_canonical_shape(conv):
+ Note = conv.Note
+ notes = [
+ Note(deck="My Deck", front="Q1?", back_html="A1.", tag="alpha"),
+ Note(deck="My Deck", front="Q2?", back_html="A2.", tag="alpha"),
+ ]
+ ids = iter(["id-1", "id-2"])
+ org = conv.notes_to_org(notes, "My Deck", new_id=lambda: next(ids))
+ assert "#+TITLE: My Deck" in org
+ assert "* alpha" in org
+ assert "** Q1? :drill:" in org
+ assert ":ID: id-1" in org
+ assert ":ID: id-2" in org
+ assert org.count("* alpha") == 1 # both cards share one section
+
+
+def test_notes_to_org_distinct_tags_get_distinct_sections(conv):
+ Note = conv.Note
+ notes = [
+ Note(deck="D", front="Qa?", back_html="a", tag="alpha"),
+ Note(deck="D", front="Qb?", back_html="b", tag="beta"),
+ ]
+ org = conv.notes_to_org(notes, "D", new_id=lambda: "x")
+ assert "* alpha" in org and "* beta" in org
+
+
+# --- round-trip through the real forward parse() ---------------------------
+
+def test_round_trip_matches_forward_parse_tuples(conv, forward, tmp_path):
+ original = (
+ "#+TITLE: RT Deck\n"
+ "\n"
+ "* First Section\n"
+ "** What is 2+2? :drill:\n"
+ ":PROPERTIES:\n:ID: aaaa\n:END:\n"
+ "Four.\n"
+ "Second line with <angle> & amp.\n"
+ "\n"
+ "* Second Section\n"
+ "** Capital of France? :drill:\n"
+ "Paris.\n"
+ )
+ tuples = forward.parse(original) # [(front, back_html, anki_tags), ...]
+ assert len(tuples) == 2
+
+ apkg = tmp_path / "rt.apkg"
+ _make_apkg(
+ apkg,
+ decks={20: "RT Deck"},
+ models={9: ["Front", "Back"]},
+ # anki_tags is a list; the apkg tags field is space-joined.
+ notes=[(100 + i, 9, [f, b], " ".join(tags))
+ for i, (f, b, tags) in enumerate(tuples)],
+ cards=[(100 + i, 20) for i in range(len(tuples))],
+ )
+
+ by_deck = conv.convert(apkg)
+ assert set(by_deck) == {"RT Deck"}
+ recovered_tuples = forward.parse(by_deck["RT Deck"])
+ assert recovered_tuples == tuples
+
+
+# --- errors ----------------------------------------------------------------
+
+def test_read_apkg_missing_collection_errors(conv, tmp_path):
+ bad = tmp_path / "bad.apkg"
+ with zipfile.ZipFile(bad, "w") as z:
+ z.writestr("media", "{}")
+ with pytest.raises(Exception):
+ conv.read_apkg(bad)
+
+
+def test_read_apkg_not_a_zip_errors(conv, tmp_path):
+ notzip = tmp_path / "plain.apkg"
+ notzip.write_text("not a zip")
+ with pytest.raises(Exception):
+ conv.read_apkg(notzip)
diff --git a/claude-templates/.ai/scripts/tests/test_cj_remove_block.py b/claude-templates/.ai/scripts/tests/test_cj_remove_block.py
index 2c8dade..3cdee46 100644
--- a/claude-templates/.ai/scripts/tests/test_cj_remove_block.py
+++ b/claude-templates/.ai/scripts/tests/test_cj_remove_block.py
@@ -14,6 +14,34 @@ import pytest
SCRIPT = Path(__file__).parent.parent / "cj-remove-block.py"
+@pytest.fixture(autouse=True)
+def isolated_tmpdir(tmp_path, monkeypatch):
+ """Give every test in this module a private TMPDIR.
+
+ The script backs up to the system temp dir under a name derived from the
+ edited file's BASENAME. The real todo.org shares that basename, so any test
+ operating on a fixture named todo.org writes something indistinguishable
+ from a production backup — and an earlier version of this file globbed the
+ shared /tmp and unlinked every match, so a routine `make test` destroyed
+ Craig's real backups (found in review, 2026-07-24).
+
+ Isolating at module scope rather than per-test is deliberate: the same bug
+ was fixed once in the elisp sibling and left here, so relying on each new
+ test to remember is exactly how it recurred. Autouse makes it structural.
+ """
+ d = tmp_path / "_tmpdir"
+ d.mkdir()
+ # TMPDIR covers subprocess invocations of the script.
+ monkeypatch.setenv("TMPDIR", str(d))
+ # tempfile.gettempdir() caches its answer on first call, so a test that
+ # loads the module in-process would keep writing to the real /tmp no matter
+ # what TMPDIR says. Override the cache too — this is the gap that made the
+ # env-var-only version still leak one backup per suite run.
+ import tempfile as _tempfile
+ monkeypatch.setattr(_tempfile, "tempdir", str(d))
+ return d
+
+
@pytest.fixture
def run_remove(tmp_path):
"""Write content to a temp org file, run cj-remove-block, return new contents."""
@@ -155,3 +183,142 @@ class TestCjRemoveBlockSafety:
err, post_content = run_remove_expecting_failure(original, start=4, end=2)
assert err.returncode != 0
assert post_content == original
+
+
+class TestMultiBlockRangeRefused:
+ """The validation exists to catch a drifted range, but it only checked the
+ first and last lines of that range. A span from one block's opening fence to
+ a LATER block's closing fence passed, and the removal silently deleted every
+ line between — real prose, headings, whole tasks — with a zero exit. Drift is
+ the skill's normal operating mode (respond-to-cj-comments edits the file as it
+ processes, and a file under cj review usually holds several blocks), so this
+ is the exact scenario the check was written for. Reproduced 2026-07-24."""
+
+ TWO_BLOCKS = (
+ "* Alpha\n"
+ "#+begin_src cj:\n"
+ "note A\n"
+ "#+end_src\n"
+ "KEEP THIS LINE\n"
+ "* Beta\n"
+ "#+begin_src cj:\n"
+ "note B\n"
+ "#+end_src\n"
+ )
+
+ def test_range_spanning_two_blocks_is_refused(self, run_remove_expecting_failure):
+ # Lines 2..9: block one's opener through block two's closer.
+ err, content = run_remove_expecting_failure(self.TWO_BLOCKS, 2, 9)
+ assert err.returncode == 1
+ assert "KEEP THIS LINE" in content, "content between the blocks was destroyed"
+ assert "* Beta" in content, "a heading between the blocks was destroyed"
+
+ def test_refusal_names_the_reason(self, run_remove_expecting_failure):
+ err, _ = run_remove_expecting_failure(self.TWO_BLOCKS, 2, 9)
+ assert "more than one" in err.stderr.decode().lower()
+
+ def test_a_correct_single_block_range_still_removes(self, run_remove):
+ # The fix must not over-tighten: the legitimate range still works.
+ out = run_remove(self.TWO_BLOCKS, 2, 4)
+ assert "note A" not in out
+ assert "KEEP THIS LINE" in out
+ assert "note B" in out, "the second block must be untouched"
+
+ def test_a_nested_end_src_inside_the_range_is_refused(self, run_remove_expecting_failure):
+ # Any #+end_src before the final line means the range covers >1 block.
+ content = (
+ "#+begin_src cj:\n"
+ "a\n"
+ "#+end_src\n"
+ "middle\n"
+ "#+begin_src cj:\n"
+ "b\n"
+ "#+end_src\n"
+ )
+ err, after = run_remove_expecting_failure(content, 1, 7)
+ assert err.returncode == 1
+ assert "middle" in after
+
+
+class TestSafeMutation:
+ """The script rewrites Craig's org files (todo.org, notes.org). It wrote with
+ a bare write_text, which truncates the target on open, and took no backup —
+ so a mid-write failure left the file truncated with no copy to recover from.
+ lint-org.el, the other tool that mutates these files, backs up to a temp dir
+ first. Match that, and make the write atomic.
+
+ Every test here redirects TMPDIR to a private directory. The backup name
+ derives from the file's basename, and the real todo.org shares it, so a test
+ globbing the shared temp dir cannot tell its own artifact from a genuine
+ backup — and an earlier version of this class globbed /tmp and unlinked every
+ match, so a routine `make test` destroyed real backups (found in review,
+ 2026-07-24). Never glob or delete across the shared temp dir."""
+
+ ONE_BLOCK = "* T\n#+begin_src cj:\nnote\n#+end_src\nkeep\n"
+
+ def test_a_backup_is_written_before_mutating(self, tmp_path):
+ import subprocess, glob, os
+ bdir = tmp_path / "bk"
+ bdir.mkdir()
+ f = tmp_path / "todo.org"
+ f.write_text(self.ONE_BLOCK)
+ subprocess.run(
+ ["python3", str(SCRIPT), "--file", str(f), "--start", "2", "--end", "4"],
+ check=True, capture_output=True,
+ env={**os.environ, "TMPDIR": str(bdir)},
+ )
+ backups = glob.glob(str(bdir / "todo.org.before-cj-remove.*"))
+ assert backups, "no backup was written before mutating the org file"
+ assert "note" in Path(max(backups)).read_text()
+
+ def test_no_partial_file_when_the_write_fails(self, tmp_path, monkeypatch):
+ import importlib.util
+ spec = importlib.util.spec_from_file_location("crb", SCRIPT)
+ mod = importlib.util.module_from_spec(spec)
+ spec.loader.exec_module(mod)
+ bdir = tmp_path / "bk"
+ bdir.mkdir()
+ monkeypatch.setenv("TMPDIR", str(bdir))
+ f = tmp_path / "todo.org"
+ f.write_text(self.ONE_BLOCK)
+ def boom(*a, **k):
+ raise OSError("disk full")
+ monkeypatch.setattr(mod.os, "replace", boom)
+ with pytest.raises(OSError):
+ mod.remove_range(f, 2, 4)
+ # The original survives intact — no truncation, no partial.
+ assert f.read_text() == self.ONE_BLOCK
+
+
+class TestBackupNeverOverwrites:
+ """Same defect class as todo-cleanup's, and more reachable here: the
+ respond-to-cj-comments skill removes several annotations in quick
+ succession, so a second-resolution stamp collides and the later backup
+ overwrote the earlier one with already-mutated content."""
+
+ TWO_BLOCKS = (
+ "* A\n#+begin_src cj:\nfirst\n#+end_src\n"
+ "* B\n#+begin_src cj:\nsecond\n#+end_src\n"
+ )
+
+ def test_consecutive_removals_each_keep_a_backup(self, tmp_path, monkeypatch):
+ import subprocess, glob
+ bdir = tmp_path / "bk"
+ bdir.mkdir()
+ monkeypatch.setenv("TMPDIR", str(bdir))
+ f = tmp_path / "todo.org"
+ f.write_text(self.TWO_BLOCKS)
+ original = f.read_text()
+ # Remove the second block, then the first — back to back, same second.
+ subprocess.run(["python3", str(SCRIPT), "--file", str(f),
+ "--start", "6", "--end", "8"],
+ check=True, capture_output=True,
+ env={**__import__("os").environ, "TMPDIR": str(bdir)})
+ subprocess.run(["python3", str(SCRIPT), "--file", str(f),
+ "--start", "2", "--end", "4"],
+ check=True, capture_output=True,
+ env={**__import__("os").environ, "TMPDIR": str(bdir)})
+ backups = glob.glob(str(bdir / "todo.org.before-cj-remove.*"))
+ assert len(backups) == 2, f"expected 2 backups, got {len(backups)}"
+ contents = [Path(b).read_text() for b in backups]
+ assert original in contents, "no backup holds the true original"
diff --git a/claude-templates/.ai/scripts/tests/test_cross_agent_discover.py b/claude-templates/.ai/scripts/tests/test_cross_agent_discover.py
deleted file mode 100644
index f0d2bb7..0000000
--- a/claude-templates/.ai/scripts/tests/test_cross_agent_discover.py
+++ /dev/null
@@ -1,204 +0,0 @@
-"""Tests for cross-agent-discover (TDD: tests written before implementation)."""
-
-from __future__ import annotations
-
-import json
-import os
-import subprocess
-import textwrap
-from pathlib import Path
-
-import pytest
-
-SCRIPT = Path(__file__).resolve().parent.parent / "cross-agent-comms" / "cross-agent-discover"
-
-
-def _run(args: list[str], env: dict | None = None) -> subprocess.CompletedProcess:
- return subprocess.run([str(SCRIPT), *args], capture_output=True, text=True, env=env)
-
-
-@pytest.fixture
-def fake_home(tmp_path, monkeypatch):
- home = tmp_path / "home"
- home.mkdir()
- monkeypatch.setenv("HOME", str(home))
- return home
-
-
-def _make_project(home: Path, name: str) -> Path:
- proj = home / "projects" / name
- (proj / ".ai").mkdir(parents=True)
- return proj
-
-
-def _write_peers_toml(home: Path, content: str) -> Path:
- cfg = home / ".config" / "cross-agent-comms"
- cfg.mkdir(parents=True, exist_ok=True)
- peers = cfg / "peers.toml"
- peers.write_text(content)
- return peers
-
-
-def test_discover_help(fake_home):
- result = _run(["--help"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0
- assert "discover" in result.stdout.lower() or "enumerate" in result.stdout.lower()
-
-
-def test_discover_local_only_no_projects(fake_home):
- """Empty home → reports zero local projects, zero peers."""
- result = _run(["--no-cache"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0
- # No crash; mentions local somehow.
- assert "local" in result.stdout.lower() or "0 project" in result.stdout.lower()
-
-
-def test_discover_lists_local_projects(fake_home):
- _make_project(fake_home, "homelab")
- _make_project(fake_home, "career")
- _make_project(fake_home, "claude-templates")
- result = _run(["--no-cache"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0
- assert "homelab" in result.stdout
- assert "career" in result.stdout
- assert "claude-templates" in result.stdout
-
-
-def test_discover_excludes_dirs_without_ai_subdir(fake_home):
- """Directories under ~/projects/ that lack .ai/ are NOT projects."""
- _make_project(fake_home, "real-project")
- (fake_home / "projects" / "not-a-project").mkdir(parents=True)
- result = _run(["--no-cache"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0
- assert "real-project" in result.stdout
- assert "not-a-project" not in result.stdout
-
-
-def test_discover_no_peers_toml_just_local(fake_home):
- _make_project(fake_home, "homelab")
- result = _run(["--no-cache"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0
- # No peers section since no toml.
- assert "homelab" in result.stdout
-
-
-def test_discover_lists_peers_from_toml(fake_home):
- _write_peers_toml(fake_home, textwrap.dedent("""\
- [peers.velox]
- host = "velox"
- ssh_user = "cjennings"
-
- [peers.bastion]
- host = "bastion.local"
- ssh_user = "cjennings"
- """))
- _make_project(fake_home, "homelab")
- result = _run(["--no-cache"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0
- assert "velox" in result.stdout
- assert "bastion" in result.stdout
-
-
-def test_discover_malformed_peers_toml_errors_clearly(fake_home):
- _write_peers_toml(fake_home, "not valid toml at all = = =")
- result = _run(["--no-cache"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode != 0
- assert "peers.toml" in result.stderr or "TOML" in result.stderr or "parse" in result.stderr.lower()
-
-
-def test_discover_json_output_schema(fake_home):
- _make_project(fake_home, "homelab")
- _make_project(fake_home, "career")
- _write_peers_toml(fake_home, textwrap.dedent("""\
- [peers.velox]
- host = "velox"
- """))
- result = _run(["--json", "--no-cache"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0
- payload = json.loads(result.stdout)
- assert "local" in payload
- assert "peers" in payload
- assert isinstance(payload["local"], list)
- assert isinstance(payload["peers"], list)
- assert "homelab" in payload["local"]
- assert "career" in payload["local"]
- velox = next((p for p in payload["peers"] if p["name"] == "velox"), None)
- assert velox is not None
- # Reachability is a key — value depends on actual SSH state.
- assert "reachable" in velox
-
-
-def test_discover_peer_scope(fake_home):
- _write_peers_toml(fake_home, textwrap.dedent("""\
- [peers.velox]
- host = "velox"
-
- [peers.bastion]
- host = "bastion.local"
- """))
- result = _run(["--peer", "velox", "--no-cache", "--json"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0
- payload = json.loads(result.stdout)
- peer_names = [p["name"] for p in payload["peers"]]
- assert "velox" in peer_names
- assert "bastion" not in peer_names
-
-
-def test_discover_unreachable_peer_marked(fake_home):
- """A peer with a definitely-unreachable host gets reachable=False."""
- _write_peers_toml(fake_home, textwrap.dedent("""\
- [peers.bogus]
- host = "definitely-not-a-real-host.invalid"
- ssh_user = "nobody"
- """))
- result = _run(["--no-cache", "--json"], env={**os.environ, "HOME": str(fake_home)}, )
- assert result.returncode == 0
- payload = json.loads(result.stdout)
- bogus = next((p for p in payload["peers"] if p["name"] == "bogus"), None)
- assert bogus is not None
- assert bogus["reachable"] is False
-
-
-def test_discover_cache_hit_within_window(fake_home):
- """Second invocation within 5 min reads cache (skip the SSH probe)."""
- _make_project(fake_home, "homelab")
- # First call populates cache.
- result1 = _run(["--json"], env={**os.environ, "HOME": str(fake_home)})
- assert result1.returncode == 0
- cache = fake_home / ".cache" / "cross-agent-comms" / "discovery.json"
- assert cache.exists()
- # Tamper with the cache to a marker only the cache path can produce.
- payload = json.loads(cache.read_text())
- payload["_test_marker"] = True
- cache.write_text(json.dumps(payload))
- # Second call (no --no-cache) should return the tampered payload.
- result2 = _run(["--json"], env={**os.environ, "HOME": str(fake_home)})
- assert result2.returncode == 0
- payload2 = json.loads(result2.stdout)
- assert payload2.get("_test_marker") is True
-
-
-def test_discover_no_cache_flag_bypasses(fake_home):
- """--no-cache ignores even a fresh cache."""
- _make_project(fake_home, "homelab")
- cache_dir = fake_home / ".cache" / "cross-agent-comms"
- cache_dir.mkdir(parents=True)
- cache_dir.joinpath("discovery.json").write_text(json.dumps({
- "_test_marker": True, "local": [], "peers": []
- }))
- result = _run(["--no-cache", "--json"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0
- payload = json.loads(result.stdout)
- # Cache marker should NOT appear in fresh result.
- assert payload.get("_test_marker") is None or payload.get("_test_marker") is False
- assert "homelab" in payload["local"]
-
-
-def test_discover_halt_shows_banner(fake_home):
- halt = fake_home / ".config" / "cross-agent-comms" / "HALT"
- halt.parent.mkdir(parents=True)
- halt.write_text("halted")
- _make_project(fake_home, "homelab")
- result = _run(["--no-cache"], env={**os.environ, "HOME": str(fake_home)})
- assert result.returncode == 0 # discover continues to print under HALT
- assert "HALT" in result.stdout
diff --git a/claude-templates/.ai/scripts/tests/test_cross_agent_halt.py b/claude-templates/.ai/scripts/tests/test_cross_agent_halt.py
deleted file mode 100644
index f8bf0b3..0000000
--- a/claude-templates/.ai/scripts/tests/test_cross_agent_halt.py
+++ /dev/null
@@ -1,204 +0,0 @@
-"""Tests for cross-agent-halt and cross-agent-resume (TDD)."""
-
-from __future__ import annotations
-
-import os
-import subprocess
-import textwrap
-from pathlib import Path
-
-import pytest
-
-HALT_SCRIPT = Path(__file__).resolve().parent.parent / "cross-agent-comms" / "cross-agent-halt"
-RESUME_SCRIPT = Path(__file__).resolve().parent.parent / "cross-agent-comms" / "cross-agent-resume"
-
-
-def _run(script: Path, args: list[str], env: dict | None = None) -> subprocess.CompletedProcess:
- return subprocess.run([str(script), *args], capture_output=True, text=True, env=env)
-
-
-@pytest.fixture
-def isolated_env(tmp_path, monkeypatch):
- """Isolated HOME + a fake systemctl that records calls without acting."""
- fake_home = tmp_path / "home"
- fake_home.mkdir()
- fake_bin = tmp_path / "bin"
- fake_bin.mkdir()
- # Fake systemctl: no-op, exit 0.
- fake_systemctl = fake_bin / "systemctl"
- fake_systemctl.write_text("#!/usr/bin/env bash\nexit 0\n")
- fake_systemctl.chmod(0o755)
- # Fake ssh: succeed only for known-good host.
- fake_ssh = fake_bin / "ssh"
- fake_ssh.write_text(textwrap.dedent("""\
- #!/usr/bin/env bash
- # Find the destination arg (skip flags).
- target=""
- for arg in "$@"; do
- case "$arg" in
- -*|*=*) ;;
- *@*|localhost|*.local|*.invalid) target="$arg"; break ;;
- *) target="$arg"; break ;;
- esac
- done
- case "$target" in
- *invalid*|*unreachable*) exit 255 ;;
- *) exit 0 ;;
- esac
- """))
- fake_ssh.chmod(0o755)
-
- monkeypatch.setenv("HOME", str(fake_home))
- # Prepend our fake bin so systemctl + ssh are intercepted, but keep real /bin etc.
- monkeypatch.setenv("PATH", f"{fake_bin}:{os.environ.get('PATH', '')}")
- return fake_home
-
-
-# ---- cross-agent-halt ----
-
-
-def test_halt_help(isolated_env):
- result = _run(HALT_SCRIPT, ["--help"], env={**os.environ, "HOME": str(isolated_env),
- "PATH": os.environ["PATH"]})
- assert result.returncode == 0
- assert "halt" in result.stdout.lower()
-
-
-def test_halt_creates_halt_file(isolated_env):
- halt_file = isolated_env / ".config" / "cross-agent-comms" / "HALT"
- assert not halt_file.exists()
- result = _run(HALT_SCRIPT, [], env={**os.environ, "HOME": str(isolated_env),
- "PATH": os.environ["PATH"]})
- assert result.returncode == 0
- assert halt_file.exists()
-
-
-def test_halt_with_reason_writes_body(isolated_env):
- result = _run(HALT_SCRIPT, ["pausing for incident review"],
- env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert result.returncode == 0
- halt_file = isolated_env / ".config" / "cross-agent-comms" / "HALT"
- assert halt_file.exists()
- assert "pausing for incident review" in halt_file.read_text()
-
-
-def test_halt_idempotent(isolated_env):
- """Running halt twice doesn't error."""
- halt_file = isolated_env / ".config" / "cross-agent-comms" / "HALT"
- r1 = _run(HALT_SCRIPT, [], env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert r1.returncode == 0
- assert halt_file.exists()
- r2 = _run(HALT_SCRIPT, [], env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert r2.returncode == 0
- assert halt_file.exists()
-
-
-def test_halt_does_not_pkill(isolated_env):
- """Per design: halt does NOT call pkill. Verify by checking no pkill process gets launched."""
- # Replace pkill in PATH with something that fails loudly so we'd see if halt invoked it.
- fake_bin = isolated_env.parent / "bin"
- pkill = fake_bin / "pkill"
- pkill.write_text("#!/usr/bin/env bash\necho 'PKILL CALLED' >&2\nexit 99\n")
- pkill.chmod(0o755)
- result = _run(HALT_SCRIPT, [], env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert result.returncode == 0
- assert "PKILL CALLED" not in result.stderr
-
-
-def test_halt_tailnet_reports_per_peer(isolated_env):
- """--tailnet iterates peers.toml and reports per-peer status."""
- cfg = isolated_env / ".config" / "cross-agent-comms"
- cfg.mkdir(parents=True)
- (cfg / "peers.toml").write_text(textwrap.dedent("""\
- [peers.velox]
- host = "velox"
- ssh_user = "cjennings"
-
- [peers.bogus]
- host = "definitely-unreachable.invalid"
- ssh_user = "cjennings"
- """))
- result = _run(HALT_SCRIPT, ["--tailnet"],
- env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- # Partial halt → exit 1.
- assert result.returncode == 1
- assert "velox" in result.stdout
- assert "bogus" in result.stdout
- # ✓ marker for velox, ✗ for bogus.
- assert "✓" in result.stdout
- assert "✗" in result.stdout
- assert "PARTIAL" in result.stdout or "partial" in result.stdout.lower()
-
-
-def test_halt_tailnet_all_reachable_exits_zero(isolated_env):
- cfg = isolated_env / ".config" / "cross-agent-comms"
- cfg.mkdir(parents=True)
- (cfg / "peers.toml").write_text(textwrap.dedent("""\
- [peers.velox]
- host = "velox"
- ssh_user = "cjennings"
- """))
- result = _run(HALT_SCRIPT, ["--tailnet"],
- env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert result.returncode == 0
- assert "velox" in result.stdout
-
-
-# ---- cross-agent-resume ----
-
-
-def test_resume_help(isolated_env):
- result = _run(RESUME_SCRIPT, ["--help"],
- env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert result.returncode == 0
- assert "resume" in result.stdout.lower()
-
-
-def test_resume_removes_halt_file(isolated_env):
- halt_file = isolated_env / ".config" / "cross-agent-comms" / "HALT"
- halt_file.parent.mkdir(parents=True)
- halt_file.write_text("halted")
- assert halt_file.exists()
- result = _run(RESUME_SCRIPT, [],
- env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert result.returncode == 0
- assert not halt_file.exists()
-
-
-def test_resume_when_no_halt_active_succeeds(isolated_env):
- """No HALT to clear is not an error."""
- result = _run(RESUME_SCRIPT, [],
- env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert result.returncode == 0
-
-
-def test_resume_prints_per_session_instructions(isolated_env):
- """Resume must surface that polling does NOT auto-resume."""
- halt_file = isolated_env / ".config" / "cross-agent-comms" / "HALT"
- halt_file.parent.mkdir(parents=True)
- halt_file.write_text("halted")
- result = _run(RESUME_SCRIPT, [],
- env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert result.returncode == 0
- out = result.stdout.lower()
- assert "polling" in out
- assert "auto" in out or "explicit" in out or "session" in out
-
-
-def test_resume_tailnet_partial_failure_exit_1(isolated_env):
- cfg = isolated_env / ".config" / "cross-agent-comms"
- cfg.mkdir(parents=True)
- (cfg / "peers.toml").write_text(textwrap.dedent("""\
- [peers.velox]
- host = "velox"
-
- [peers.bogus]
- host = "unreachable-host.invalid"
- """))
- halt_file = cfg / "HALT"
- halt_file.write_text("halted")
- result = _run(RESUME_SCRIPT, ["--tailnet"],
- env={**os.environ, "HOME": str(isolated_env), "PATH": os.environ["PATH"]})
- assert result.returncode == 1
- assert "velox" in result.stdout
- assert "bogus" in result.stdout
diff --git a/claude-templates/.ai/scripts/tests/test_cross_agent_recv.py b/claude-templates/.ai/scripts/tests/test_cross_agent_recv.py
deleted file mode 100644
index 27c53a5..0000000
--- a/claude-templates/.ai/scripts/tests/test_cross_agent_recv.py
+++ /dev/null
@@ -1,176 +0,0 @@
-"""Tests for cross-agent-recv."""
-
-from __future__ import annotations
-
-import json
-import os
-import subprocess
-from pathlib import Path
-
-import pytest
-
-SCRIPT = Path(__file__).resolve().parent.parent / "cross-agent-comms" / "cross-agent-recv"
-
-
-def _make_message(path: Path, *, conv_id: str = "test-conv", seq: int = 1, msg_type: str = "request",
- proto_version: str = "5", title: str = "Test", requires_tools: str | None = None,
- body: str = "Body.\n") -> Path:
- fm_lines = [
- f"#+TITLE: {title}",
- f"#+CONVERSATION_ID: {conv_id}",
- f"#+MESSAGE_TYPE: {msg_type}",
- f"#+SEQUENCE: {seq}",
- "#+TIMESTAMP: 2026-04-27T05:00:00-05:00",
- f"#+PROTOCOL_VERSION: {proto_version}",
- ]
- if requires_tools:
- fm_lines.append(f"#+REQUIRES_TOOLS: {requires_tools}")
- path.write_text("\n".join(fm_lines) + "\n\n" + body)
- return path
-
-
-def _run(args: list[str], env: dict | None = None) -> subprocess.CompletedProcess:
- return subprocess.run([str(SCRIPT), *args], capture_output=True, text=True, env=env)
-
-
-@pytest.fixture
-def isolated_env(tmp_path, monkeypatch):
- fake_home = tmp_path / "home"
- fake_home.mkdir()
- monkeypatch.setenv("HOME", str(fake_home))
- return fake_home
-
-
-def test_recv_help(isolated_env):
- result = _run(["--help"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 0
- assert "Receive and decide" in result.stdout
-
-
-def test_recv_missing_file_rejects(isolated_env, tmp_path):
- result = _run([str(tmp_path / "nope.org")], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 3 # reject
-
-
-def test_recv_malformed_frontmatter_rejects(isolated_env, tmp_path):
- bad = tmp_path / "bad.org"
- bad.write_text("not org-mode at all\n")
- result = _run([str(bad), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 3
- assert "decision: reject" in result.stdout
-
-
-def test_recv_missing_required_field_rejects(isolated_env, tmp_path):
- msg = tmp_path / "msg.org"
- # Missing PROTOCOL_VERSION among others.
- msg.write_text("#+TITLE: x\n#+CONVERSATION_ID: c\n\nBody.\n")
- result = _run([str(msg), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 3
- assert "missing required" in result.stdout
-
-
-def test_recv_protocol_version_mismatch_query(isolated_env, tmp_path):
- msg = _make_message(tmp_path / "msg.org", proto_version="4")
- result = _run([str(msg), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 2 # query
- assert "PROTOCOL_VERSION mismatch" in result.stdout
-
-
-def test_recv_invalid_message_type_rejects(isolated_env, tmp_path):
- msg = _make_message(tmp_path / "msg.org", msg_type="banana")
- result = _run([str(msg), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 3
- assert "invalid MESSAGE_TYPE" in result.stdout
-
-
-def test_recv_missing_signature_rejects(isolated_env, tmp_path):
- """When verify is on, a missing .asc sibling rejects."""
- msg = _make_message(tmp_path / "msg.org")
- # No .asc sidecar.
- result = _run([str(msg)], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 3
- assert "signature file missing" in result.stdout
-
-
-def test_recv_valid_processes(isolated_env, tmp_path):
- """A valid message with --no-verify and no dedup match → process."""
- msg = _make_message(tmp_path / "msg.org")
- result = _run([str(msg), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 0 # process
- assert "decision: process" in result.stdout
- assert "sha256:" in result.stdout
-
-
-def test_recv_dedup_against_identical_existing(isolated_env, tmp_path):
- """Same content + same SEQUENCE in same dir → dedup."""
- inbox = tmp_path / "inbox"
- inbox.mkdir()
- first = _make_message(inbox / "20260427T100000Z-from-x-c.org", conv_id="c", seq=5)
- # Second message with same content — name differs (canonical-style would have different timestamp).
- second = _make_message(inbox / "20260427T100100Z-from-x-c.org", conv_id="c", seq=5)
- # Bodies must be byte-identical for hash equality.
- second.write_bytes(first.read_bytes())
- result = _run([str(second), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 1 # dedup
- assert "decision: dedup" in result.stdout
-
-
-def test_recv_collision_with_different_content_processes(isolated_env, tmp_path):
- """Same SEQUENCE + same CONVERSATION_ID but different content → process both."""
- inbox = tmp_path / "inbox"
- inbox.mkdir()
- _make_message(inbox / "20260427T100000Z-from-x-c.org", conv_id="c", seq=5, body="First body.\n")
- second = _make_message(inbox / "20260427T100100Z-from-x-c.org", conv_id="c", seq=5, body="Different body.\n")
- result = _run([str(second), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 0 # process
- assert "decision: process" in result.stdout
-
-
-def test_recv_requires_tools_missing_query(isolated_env, tmp_path):
- """REQUIRES_TOOLS naming a definitely-missing binary → query."""
- msg = _make_message(tmp_path / "msg.org", requires_tools="definitely-not-installed-xyzzy-9000")
- result = _run([str(msg), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 2 # query
- assert "required tools unavailable" in result.stdout
-
-
-def test_recv_requires_tools_present_processes(isolated_env, tmp_path):
- """REQUIRES_TOOLS naming a real binary → process."""
- msg = _make_message(tmp_path / "msg.org", requires_tools="ls,cat")
- result = _run([str(msg), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 0
- assert "decision: process" in result.stdout
-
-
-def test_recv_json_output(isolated_env, tmp_path):
- msg = _make_message(tmp_path / "msg.org")
- result = _run([str(msg), "--no-verify", "--json"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 0
- payload = json.loads(result.stdout)
- assert payload["decision"] == "process"
- assert payload["message_type"] == "request"
- assert payload["conversation_id"] == "test-conv"
-
-
-def test_recv_halt_blocks(isolated_env, tmp_path):
- halt = isolated_env / ".config" / "cross-agent-comms" / "HALT"
- halt.parent.mkdir(parents=True)
- halt.write_text("halted\n")
- msg = _make_message(tmp_path / "msg.org")
- result = _run([str(msg), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 5
- assert "halt active" in result.stderr.lower()
-
-
-def test_recv_halt_leaves_message_in_place(isolated_env, tmp_path):
- """Per spec: under HALT, recv must NOT move/dedup/reject — leave file in place."""
- halt = isolated_env / ".config" / "cross-agent-comms" / "HALT"
- halt.parent.mkdir(parents=True)
- halt.write_text("halted\n")
- msg = _make_message(tmp_path / "msg.org")
- pre_content = msg.read_text()
- result = _run([str(msg), "--no-verify"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 5
- # File still exists with same content.
- assert msg.exists()
- assert msg.read_text() == pre_content
diff --git a/claude-templates/.ai/scripts/tests/test_cross_agent_send.py b/claude-templates/.ai/scripts/tests/test_cross_agent_send.py
deleted file mode 100644
index f716e95..0000000
--- a/claude-templates/.ai/scripts/tests/test_cross_agent_send.py
+++ /dev/null
@@ -1,210 +0,0 @@
-"""Tests for cross-agent-send.
-
-Subprocess-based: treat the script as a black-box CLI and assert on its
-exit codes, stdout, and the files it produces.
-"""
-
-from __future__ import annotations
-
-import os
-import subprocess
-import textwrap
-from pathlib import Path
-
-import pytest
-
-SCRIPT = Path(__file__).resolve().parent.parent / "cross-agent-comms" / "cross-agent-send"
-
-
-def _make_message(tmp_path: Path, conv_id: str = "test-conv", seq: int = 1, msg_type: str = "request",
- proto_version: str = "5") -> Path:
- msg = tmp_path / "msg.org"
- msg.write_text(textwrap.dedent(f"""\
- #+TITLE: Test message
- #+CONVERSATION_ID: {conv_id}
- #+MESSAGE_TYPE: {msg_type}
- #+SEQUENCE: {seq}
- #+TIMESTAMP: 2026-04-27T05:00:00-05:00
- #+PROTOCOL_VERSION: {proto_version}
-
- Body.
- """))
- return msg
-
-
-def _run(args: list[str], env: dict | None = None, cwd: Path | None = None) -> subprocess.CompletedProcess:
- return subprocess.run(
- [str(SCRIPT), *args],
- capture_output=True,
- text=True,
- env=env,
- cwd=cwd,
- )
-
-
-@pytest.fixture
-def isolated_env(tmp_path, monkeypatch):
- """Redirect HOME so peers.toml, HALT, marker files are scoped to the test."""
- fake_home = tmp_path / "home"
- fake_home.mkdir()
- monkeypatch.setenv("HOME", str(fake_home))
- # Pre-create projects/ so derive_sender_project has somewhere to look.
- (fake_home / "projects" / "homelab").mkdir(parents=True)
- return fake_home
-
-
-def test_send_help(isolated_env):
- """--help works without side effects."""
- result = _run(["--help"], env={**os.environ, "HOME": str(isolated_env)})
- assert result.returncode == 0
- assert "Send a cross-agent message" in result.stdout
-
-
-def test_send_missing_message_file(isolated_env):
- """Nonexistent message file returns general error."""
- import socket
- machine = socket.gethostname().split(".")[0]
- result = _run(
- [f"{machine}.homelab", str(isolated_env / "nonexistent.org")],
- env={**os.environ, "HOME": str(isolated_env)},
- )
- assert result.returncode == 1
- assert "not found" in result.stderr.lower()
-
-
-def test_send_invalid_destination_format(isolated_env, tmp_path):
- """Destination without . returns dest-not-found exit code."""
- msg = _make_message(tmp_path)
- result = _run(
- ["bogus", str(msg)],
- env={**os.environ, "HOME": str(isolated_env)},
- )
- assert result.returncode == 2
- assert "<machine>.<project>" in result.stderr or "destination" in result.stderr.lower()
-
-
-def test_send_dest_not_in_peers(isolated_env, tmp_path):
- """Cross-machine destination with no peers.toml entry exits 2."""
- msg = _make_message(tmp_path)
- result = _run(
- ["unknownmachine.homelab", str(msg)],
- env={**os.environ, "HOME": str(isolated_env)},
- )
- assert result.returncode == 2
- assert "not found in peers" in result.stderr
-
-
-def test_send_frontmatter_missing_required(isolated_env, tmp_path):
- """Message missing required fields exits 4."""
- bad = tmp_path / "bad.org"
- bad.write_text("#+TITLE: nope\n\nBody.\n")
- import socket
- machine = socket.gethostname().split(".")[0]
- result = _run(
- [f"{machine}.homelab", str(bad)],
- env={**os.environ, "HOME": str(isolated_env)},
- )
- assert result.returncode == 4
- assert "missing required fields" in result.stderr
-
-
-def test_send_invalid_message_type(isolated_env, tmp_path):
- """Unknown MESSAGE_TYPE exits 4."""
- msg = _make_message(tmp_path, msg_type="frobnicate")
- import socket
- machine = socket.gethostname().split(".")[0]
- result = _run(
- [f"{machine}.homelab", str(msg)],
- env={**os.environ, "HOME": str(isolated_env)},
- )
- assert result.returncode == 4
- assert "MESSAGE_TYPE" in result.stderr
-
-
-def test_send_halt_blocks(isolated_env, tmp_path):
- """When HALT exists, send refuses with exit 5."""
- halt = isolated_env / ".config" / "cross-agent-comms" / "HALT"
- halt.parent.mkdir(parents=True)
- halt.write_text("test halt\n")
- msg = _make_message(tmp_path)
- import socket
- machine = socket.gethostname().split(".")[0]
- result = _run(
- [f"{machine}.homelab", str(msg)],
- env={**os.environ, "HOME": str(isolated_env)},
- )
- assert result.returncode == 5
- assert "halt active" in result.stderr.lower()
-
-
-def test_send_same_machine_no_sign_delivers(isolated_env, tmp_path):
- """Same-machine delivery with --no-sign produces a canonically named file."""
- msg = _make_message(tmp_path, conv_id="my-conv")
- import socket
- machine = socket.gethostname().split(".")[0]
- # Sender is derived from CWD walking up to ~/projects/<name>/
- cwd = isolated_env / "projects" / "homelab"
- result = _run(
- [f"{machine}.homelab", str(msg), "--no-sign"],
- env={**os.environ, "HOME": str(isolated_env)},
- cwd=cwd,
- )
- assert result.returncode == 0, f"stderr={result.stderr}"
- inbox = isolated_env / "projects" / "homelab" / "inbox" / "from-agents"
- files = list(inbox.glob("*-from-homelab-my-conv.org"))
- assert len(files) == 1
- # No sig file with --no-sign.
- assert not list(inbox.glob("*.asc"))
- # Canonical filename pattern.
- assert files[0].name.startswith("2026") and files[0].name.endswith("-from-homelab-my-conv.org")
-
-
-def test_send_same_machine_signed_writes_asc(isolated_env, tmp_path):
- """Signed delivery writes both .org and .asc."""
- msg = _make_message(tmp_path, conv_id="signed-conv")
- import socket
- machine = socket.gethostname().split(".")[0]
- cwd = isolated_env / "projects" / "homelab"
- # Use the real GPG keyring (not isolating GPG — Craig's existing keys are fine for tests).
- real_env = {**os.environ, "HOME": str(isolated_env), "GNUPGHOME": str(Path.home() / ".gnupg")}
- result = _run(
- [f"{machine}.homelab", str(msg)],
- env=real_env,
- cwd=cwd,
- )
- if result.returncode != 0:
- pytest.skip(f"GPG signing unavailable in this environment: {result.stderr}")
- inbox = isolated_env / "projects" / "homelab" / "inbox" / "from-agents"
- org_files = list(inbox.glob("*-from-homelab-signed-conv.org"))
- asc_files = list(inbox.glob("*-from-homelab-signed-conv.org.asc"))
- assert len(org_files) == 1
- assert len(asc_files) == 1
-
-
-def test_send_filename_ignores_input_basename(isolated_env, tmp_path):
- """User's input filename is ignored; canonical filename is generated."""
- weird = tmp_path / "weird-user-name.org"
- weird.write_text(textwrap.dedent("""\
- #+TITLE: Title
- #+CONVERSATION_ID: ignored-input
- #+MESSAGE_TYPE: request
- #+SEQUENCE: 1
- #+TIMESTAMP: 2026-04-27T05:00:00-05:00
- #+PROTOCOL_VERSION: 5
-
- Body.
- """))
- import socket
- machine = socket.gethostname().split(".")[0]
- cwd = isolated_env / "projects" / "homelab"
- result = _run(
- [f"{machine}.homelab", str(weird), "--no-sign"],
- env={**os.environ, "HOME": str(isolated_env)},
- cwd=cwd,
- )
- assert result.returncode == 0
- inbox = isolated_env / "projects" / "homelab" / "inbox" / "from-agents"
- # No file named after the user's input.
- assert not (inbox / "weird-user-name.org").exists()
- # Canonical naming used.
- assert list(inbox.glob("*-from-homelab-ignored-input.org"))
diff --git a/claude-templates/.ai/scripts/tests/test_cross_agent_status.py b/claude-templates/.ai/scripts/tests/test_cross_agent_status.py
deleted file mode 100644
index bb5b8ba..0000000
--- a/claude-templates/.ai/scripts/tests/test_cross_agent_status.py
+++ /dev/null
@@ -1,165 +0,0 @@
-"""Tests for cross-agent-status (TDD: tests written before implementation)."""
-
-from __future__ import annotations
-
-import json
-import os
-import subprocess
-import textwrap
-from pathlib import Path
-
-import pytest
-
-SCRIPT = Path(__file__).resolve().parent.parent / "cross-agent-comms" / "cross-agent-status"
-
-
-def _make_msg(path: Path, *, conv_id: str, seq: int, msg_type: str = "request",
- proto_version: str = "5", timestamp: str = "2026-04-27T05:00:00-05:00") -> Path:
- path.parent.mkdir(parents=True, exist_ok=True)
- path.write_text(textwrap.dedent(f"""\
- #+TITLE: T
- #+CONVERSATION_ID: {conv_id}
- #+MESSAGE_TYPE: {msg_type}
- #+SEQUENCE: {seq}
- #+TIMESTAMP: {timestamp}
- #+PROTOCOL_VERSION: {proto_version}
-
- Body.
- """))
- return path
-
-
-def _run(args: list[str], env: dict | None = None) -> subprocess.CompletedProcess:
- return subprocess.run([str(SCRIPT), *args], capture_output=True, text=True, env=env)
-
-
-@pytest.fixture
-def fake_projects(tmp_path, monkeypatch):
- """Create a fake ~/projects/<name>/inbox/from-agents/ tree under tmp_path."""
- home = tmp_path / "home"
- home.mkdir()
- monkeypatch.setenv("HOME", str(home))
- return home
-
-
-def test_status_help(fake_projects):
- result = _run(["--help"], env={**os.environ, "HOME": str(fake_projects)})
- assert result.returncode == 0
- assert "snapshot" in result.stdout.lower() or "pending" in result.stdout.lower()
-
-
-def test_status_no_projects_clean_output(fake_projects):
- result = _run([], env={**os.environ, "HOME": str(fake_projects)})
- assert result.returncode == 0
- # Empty machine prints either header-only table or "no projects" — accept either.
- # No crash, no pending claims.
- assert "pending" in result.stdout.lower() or result.stdout.strip() == ""
-
-
-def test_status_one_pending_shows_up(fake_projects):
- inbox = fake_projects / "projects" / "homelab" / "inbox" / "from-agents"
- _make_msg(inbox / "20260427T100000Z-from-career-fixup.org", conv_id="fixup", seq=1)
- result = _run([], env={**os.environ, "HOME": str(fake_projects)})
- assert result.returncode == 0
- assert "homelab" in result.stdout
- assert "1" in result.stdout # pending count
- assert "20260427T100000Z-from-career-fixup.org" in result.stdout
-
-
-def test_status_released_conversation_zero_pending(fake_projects):
- """A conversation with a release message in it counts as 0 pending."""
- inbox = fake_projects / "projects" / "homelab" / "inbox" / "from-agents"
- _make_msg(inbox / "20260427T100000Z-from-career-done.org", conv_id="done", seq=1)
- _make_msg(inbox / "20260427T100100Z-from-homelab-done.org", conv_id="done", seq=2, msg_type="release")
- result = _run([], env={**os.environ, "HOME": str(fake_projects)})
- assert result.returncode == 0
- # Check the homelab row shows 0 pending.
- lines = [ln for ln in result.stdout.splitlines() if "homelab" in ln]
- # At least one homelab line should show 0 pending or "—".
- assert any("0" in ln or "—" in ln for ln in lines)
-
-
-def test_status_partial_release(fake_projects):
- """Conversation with release + a later message → that later message counts as pending."""
- inbox = fake_projects / "projects" / "homelab" / "inbox" / "from-agents"
- _make_msg(inbox / "20260427T100000Z-from-career-x.org", conv_id="x", seq=1,
- timestamp="2026-04-27T05:00:00-05:00")
- _make_msg(inbox / "20260427T100100Z-from-homelab-x.org", conv_id="x", seq=2, msg_type="release",
- timestamp="2026-04-27T05:01:00-05:00")
- # New message AFTER release: starts a fresh thread that's pending.
- _make_msg(inbox / "20260427T200000Z-from-career-x.org", conv_id="x", seq=3,
- timestamp="2026-04-27T15:00:00-05:00")
- result = _run([], env={**os.environ, "HOME": str(fake_projects)})
- assert result.returncode == 0
- homelab_line = next(ln for ln in result.stdout.splitlines() if "homelab" in ln)
- assert "1" in homelab_line # the post-release message is pending
-
-
-def test_status_multiple_projects(fake_projects):
- inbox_a = fake_projects / "projects" / "homelab" / "inbox" / "from-agents"
- inbox_b = fake_projects / "projects" / "career" / "inbox" / "from-agents"
- _make_msg(inbox_a / "20260427T100000Z-from-x-a.org", conv_id="a", seq=1)
- _make_msg(inbox_b / "20260427T100100Z-from-x-b.org", conv_id="b", seq=1)
- _make_msg(inbox_b / "20260427T100200Z-from-x-c.org", conv_id="c", seq=1)
- result = _run([], env={**os.environ, "HOME": str(fake_projects)})
- assert result.returncode == 0
- # career has 2 pending, homelab has 1.
- career_line = next(ln for ln in result.stdout.splitlines() if "career" in ln)
- homelab_line = next(ln for ln in result.stdout.splitlines() if "homelab" in ln)
- assert "2" in career_line
- assert "1" in homelab_line
-
-
-def test_status_json_output(fake_projects):
- inbox = fake_projects / "projects" / "homelab" / "inbox" / "from-agents"
- _make_msg(inbox / "20260427T100000Z-from-career-test.org", conv_id="test", seq=1)
- result = _run(["--json"], env={**os.environ, "HOME": str(fake_projects)})
- assert result.returncode == 0
- payload = json.loads(result.stdout)
- assert "projects" in payload
- assert isinstance(payload["projects"], list)
- homelab = next((p for p in payload["projects"] if p["name"] == "homelab"), None)
- assert homelab is not None
- assert homelab["pending_count"] == 1
-
-
-def test_status_sort_pending_first(fake_projects):
- """Projects with pending messages sort before projects with 0."""
- (fake_projects / "projects" / "alpha" / "inbox" / "from-agents").mkdir(parents=True)
- inbox_zeta = fake_projects / "projects" / "zeta" / "inbox" / "from-agents"
- _make_msg(inbox_zeta / "20260427T100000Z-from-x-z.org", conv_id="z", seq=1)
- result = _run([], env={**os.environ, "HOME": str(fake_projects)})
- assert result.returncode == 0
- lines = result.stdout.splitlines()
- zeta_idx = next(i for i, ln in enumerate(lines) if "zeta" in ln)
- alpha_idx = next(i for i, ln in enumerate(lines) if "alpha" in ln)
- assert zeta_idx < alpha_idx, "pending project should sort before zero-pending project"
-
-
-def test_status_halt_shows_banner(fake_projects):
- halt = fake_projects / ".config" / "cross-agent-comms" / "HALT"
- halt.parent.mkdir(parents=True)
- halt.write_text("halted for test")
- inbox = fake_projects / "projects" / "homelab" / "inbox" / "from-agents"
- _make_msg(inbox / "20260427T100000Z-from-x-x.org", conv_id="x", seq=1)
- result = _run([], env={**os.environ, "HOME": str(fake_projects)})
- assert result.returncode == 0 # status continues to print under HALT
- assert "HALT" in result.stdout
- # Banner should mention the reason.
- assert "halted for test" in result.stdout
-
-
-def test_status_projects_glob_override(fake_projects):
- inbox = fake_projects / "projects" / "homelab" / "inbox" / "from-agents"
- _make_msg(inbox / "20260427T100000Z-from-x-a.org", conv_id="a", seq=1)
- other_inbox = fake_projects / "projects" / "career" / "inbox" / "from-agents"
- _make_msg(other_inbox / "20260427T100100Z-from-x-b.org", conv_id="b", seq=1)
- # Glob limits to homelab only.
- result = _run(
- ["--projects-glob", str(fake_projects / "projects" / "homelab" / "inbox" / "from-agents") + "/"],
- env={**os.environ, "HOME": str(fake_projects)},
- )
- assert result.returncode == 0
- assert "homelab" in result.stdout
- # career not in scope.
- assert "career" not in result.stdout
diff --git a/claude-templates/.ai/scripts/tests/test_cross_agent_watch.py b/claude-templates/.ai/scripts/tests/test_cross_agent_watch.py
deleted file mode 100644
index 417cc19..0000000
--- a/claude-templates/.ai/scripts/tests/test_cross_agent_watch.py
+++ /dev/null
@@ -1,155 +0,0 @@
-"""Tests for cross-agent-watch.
-
-Black-box: spawn the script, drop files into a watched dir, read the log.
-Tests use --no-notify to avoid firing real desktop notifications.
-"""
-
-from __future__ import annotations
-
-import os
-import subprocess
-import time
-from pathlib import Path
-
-import pytest
-
-SCRIPT = Path(__file__).resolve().parent.parent / "cross-agent-comms" / "cross-agent-watch"
-
-
-def _spawn(watched_dir: Path, log_path: Path, env: dict) -> subprocess.Popen:
- return subprocess.Popen(
- [
- str(SCRIPT),
- "--projects-glob", str(watched_dir) + "/",
- "--log", str(log_path),
- "--no-notify",
- "--quiet",
- ],
- stdout=subprocess.DEVNULL,
- stderr=subprocess.PIPE,
- env=env,
- )
-
-
-def _wait_for_log_lines(log_path: Path, expected: int, timeout: float = 5.0) -> list[str]:
- deadline = time.time() + timeout
- while time.time() < deadline:
- if log_path.exists():
- lines = [ln for ln in log_path.read_text().splitlines() if ln]
- if len(lines) >= expected:
- return lines
- time.sleep(0.1)
- if log_path.exists():
- return [ln for ln in log_path.read_text().splitlines() if ln]
- return []
-
-
-@pytest.fixture
-def isolated_env(tmp_path, monkeypatch):
- fake_home = tmp_path / "home"
- fake_home.mkdir()
- monkeypatch.setenv("HOME", str(fake_home))
- return fake_home
-
-
-def test_watch_help(isolated_env):
- result = subprocess.run(
- [str(SCRIPT), "--help"],
- capture_output=True, text=True,
- env={**os.environ, "HOME": str(isolated_env)},
- )
- assert result.returncode == 0
- assert "Usage:" in result.stdout
-
-
-def test_watch_empty_glob_exits_nonzero(isolated_env):
- """Glob resolving to zero dirs should exit non-zero with a clear message."""
- result = subprocess.run(
- [str(SCRIPT), "--projects-glob", "/nonexistent/path/*/foo/", "--no-notify", "--quiet"],
- capture_output=True, text=True,
- env={**os.environ, "HOME": str(isolated_env)},
- timeout=3,
- )
- assert result.returncode != 0
- assert "0 directories" in result.stderr
-
-
-def test_watch_logs_org_file_create(isolated_env, tmp_path):
- watched = tmp_path / "watched"
- watched.mkdir()
- log = tmp_path / "watch.log"
- proc = _spawn(watched, log, {**os.environ, "HOME": str(isolated_env)})
- try:
- # Give inotifywait a moment to attach.
- time.sleep(0.3)
- (watched / "test-msg.org").write_text("hello")
- lines = _wait_for_log_lines(log, expected=1, timeout=3.0)
- assert len(lines) >= 1
- assert "test-msg.org" in lines[-1]
- finally:
- proc.terminate()
- proc.wait(timeout=2)
-
-
-def test_watch_filters_tmp_files(isolated_env, tmp_path):
- """Files starting with .tmp. must NOT trigger log entries."""
- watched = tmp_path / "watched"
- watched.mkdir()
- log = tmp_path / "watch.log"
- proc = _spawn(watched, log, {**os.environ, "HOME": str(isolated_env)})
- try:
- time.sleep(0.3)
- (watched / ".tmp.staging-file.org").write_text("hello")
- # Wait briefly to confirm nothing logs.
- time.sleep(0.5)
- if log.exists():
- content = log.read_text()
- assert ".tmp.staging-file" not in content
- # Then drop a real file to confirm watcher is alive.
- (watched / "real.org").write_text("real")
- lines = _wait_for_log_lines(log, expected=1, timeout=3.0)
- assert any("real.org" in ln for ln in lines)
- finally:
- proc.terminate()
- proc.wait(timeout=2)
-
-
-def test_watch_filters_asc_sidecars(isolated_env, tmp_path):
- """Only .org events fire; .asc sidecars are silent."""
- watched = tmp_path / "watched"
- watched.mkdir()
- log = tmp_path / "watch.log"
- proc = _spawn(watched, log, {**os.environ, "HOME": str(isolated_env)})
- try:
- time.sleep(0.3)
- (watched / "msg.org.asc").write_text("sig")
- time.sleep(0.5)
- if log.exists():
- assert "msg.org.asc" not in log.read_text()
- # .org event still works.
- (watched / "msg.org").write_text("body")
- lines = _wait_for_log_lines(log, expected=1, timeout=3.0)
- assert any(ln.endswith("msg.org") for ln in lines)
- finally:
- proc.terminate()
- proc.wait(timeout=2)
-
-
-def test_watch_halt_suppresses_but_logs(isolated_env, tmp_path):
- """When HALT is set, watcher logs the event with (suppressed by HALT) marker."""
- halt = isolated_env / ".config" / "cross-agent-comms" / "HALT"
- halt.parent.mkdir(parents=True)
- halt.write_text("halted")
- watched = tmp_path / "watched"
- watched.mkdir()
- log = tmp_path / "watch.log"
- proc = _spawn(watched, log, {**os.environ, "HOME": str(isolated_env)})
- try:
- time.sleep(0.3)
- (watched / "halted-event.org").write_text("body")
- lines = _wait_for_log_lines(log, expected=1, timeout=3.0)
- assert len(lines) >= 1
- assert "suppressed by HALT" in lines[-1]
- finally:
- proc.terminate()
- proc.wait(timeout=2)
diff --git a/claude-templates/.ai/scripts/tests/test_flashcard_stats.py b/claude-templates/.ai/scripts/tests/test_flashcard_stats.py
index 606f7c1..46deccc 100644
--- a/claude-templates/.ai/scripts/tests/test_flashcard_stats.py
+++ b/claude-templates/.ai/scripts/tests/test_flashcard_stats.py
@@ -217,6 +217,31 @@ def test_parse_cards_captures_body_without_drawer_planning_or_answer_header(stat
assert c["body"] == "the real answer"
+def test_parse_cards_counts_a_multitag_heading_as_a_card(stats):
+ """A card multi-tagged :fundamental:drill: still counts; the front is clean."""
+ text = "* Sec\n** Q multi? :fundamental:drill:\nthe answer\n"
+ cards, _ = stats.parse_cards(text.splitlines())
+ assert len(cards) == 1
+ assert cards[0]["heading"] == "Q multi?"
+ assert cards[0]["body"] == "the answer"
+
+
+def test_parse_cards_ignores_a_tagged_heading_without_drill(stats):
+ """A tagged heading missing :drill: is not a drill card."""
+ text = "* Sec\n** Just a note :note:\nbody\n"
+ cards, _ = stats.parse_cards(text.splitlines())
+ assert cards == []
+
+
+def test_parse_cards_body_stops_at_next_multitag_card(stats):
+ """The body scan ends at the next L2 card even when it is multi-tagged."""
+ text = "** Q1? :a:drill:\nbody1\n** Q2? :drill:\nbody2\n"
+ cards, _ = stats.parse_cards(text.splitlines())
+ assert len(cards) == 2
+ assert cards[0]["body"] == "body1"
+ assert cards[1]["body"] == "body2"
+
+
def test_find_duplicate_fronts_matches_normalized_headings(stats):
cards = [
{"heading": "What is LEO?"},
diff --git a/claude-templates/.ai/scripts/tests/test_flashcard_to_anki.py b/claude-templates/.ai/scripts/tests/test_flashcard_to_anki.py
index 058b0cd..fa38b64 100644
--- a/claude-templates/.ai/scripts/tests/test_flashcard_to_anki.py
+++ b/claude-templates/.ai/scripts/tests/test_flashcard_to_anki.py
@@ -34,14 +34,33 @@ def test_default_output_path_targets_phone_anki_dir(drill):
assert result == Path.home() / "sync" / "phone" / "anki" / "health-drill.apkg"
-def test_default_deck_name_is_raw_basename(drill):
- """Deck name is the input basename with case preserved; #+TITLE is ignored."""
- assert drill.default_deck_name(Path("/x/deepsat.org")) == "deepsat"
+def test_default_deck_name_uses_org_title(drill):
+ """The #+TITLE drives the Anki deck name, not the filename slug."""
+ org = "#+TITLE: Refutations\n* Section\n** Q? :drill:\na\n"
+ assert drill.default_deck_name(Path("/x/refutation-drill.org"), org) == "Refutations"
-def test_default_deck_name_keeps_hyphens(drill):
- """A hyphenated basename is kept verbatim rather than title-cased."""
- assert drill.default_deck_name(Path("/x/health-drill.org")) == "health-drill"
+def test_default_deck_name_title_is_trimmed(drill):
+ """Surrounding whitespace on the #+TITLE value is stripped."""
+ org = "#+TITLE: DeepSat Flashcards \n"
+ assert drill.default_deck_name(Path("/x/deepsat.org"), org) == "DeepSat Flashcards"
+
+
+def test_default_deck_name_title_match_is_case_insensitive(drill):
+ """A lowercase #+title: keyword is still recognized."""
+ org = "#+title: Health Flashcards\n"
+ assert drill.default_deck_name(Path("/x/health-drill.org"), org) == "Health Flashcards"
+
+
+def test_default_deck_name_falls_back_to_basename_without_title(drill):
+ """No #+TITLE line falls back to the input basename, case preserved."""
+ org = "* Section\n** Q? :drill:\na\n"
+ assert drill.default_deck_name(Path("/x/deepsat.org"), org) == "deepsat"
+
+
+def test_default_deck_name_blank_title_falls_back_to_basename(drill):
+ """An empty #+TITLE value is ignored in favour of the basename."""
+ assert drill.default_deck_name(Path("/x/health-drill.org"), "#+TITLE: \n") == "health-drill"
# --- section_to_tag (pure) ---
@@ -139,17 +158,18 @@ Geostationary Earth Orbit.
def test_parse_returns_front_back_tag_per_card(drill):
cards = drill.parse(SECTIONED)
assert len(cards) == 2
- assert cards[0] == ("What is LEO?", "Low Earth Orbit.", "orbital-regimes")
+ # The section becomes the sole Anki tag (as a one-element list).
+ assert cards[0] == ("What is LEO?", "Low Earth Orbit.", ["orbital-regimes"])
assert cards[1][0] == "What is GEO?"
def test_parse_card_without_a_section_gets_the_drill_tag(drill):
- assert drill.parse("** Lone card? :drill:\nbody\n") == [("Lone card?", "body", "drill")]
+ assert drill.parse("** Lone card? :drill:\nbody\n") == [("Lone card?", "body", ["drill"])]
def test_parse_strips_properties_drawer_from_back(drill):
text = "** Q? :drill:\n:PROPERTIES:\n:ID: abc\n:END:\nThe answer.\n"
- assert drill.parse(text) == [("Q?", "The answer.", "drill")]
+ assert drill.parse(text) == [("Q?", "The answer.", ["drill"])]
def test_parse_trims_leading_and_trailing_blank_body_lines(drill):
@@ -159,7 +179,59 @@ def test_parse_trims_leading_and_trailing_blank_body_lines(drill):
def test_parse_card_with_only_a_drawer_has_empty_back(drill):
text = "** Q? :drill:\n:PROPERTIES:\n:ID: x\n:END:\n"
- assert drill.parse(text) == [("Q?", "", "drill")]
+ assert drill.parse(text) == [("Q?", "", ["drill"])]
+
+
+# --- multi-tag headings, --tag-filter, --guid-salt -------------------------
+
+MULTITAG = """* Fundamentals
+** What is LEO? :fundamental:drill:
+Low Earth Orbit.
+** What is GEO? :drill:
+Geostationary Earth Orbit.
+"""
+
+
+def test_parse_multitag_heading_is_a_card_when_drill_is_present(drill):
+ """A heading with a second org tag still parses when drill is among them."""
+ cards = drill.parse(MULTITAG)
+ assert len(cards) == 2
+ assert cards[0][0] == "What is LEO?"
+
+
+def test_parse_multitag_tags_ride_along_next_to_the_section_tag(drill):
+ """Non-drill org tags become Anki tags alongside the section tag."""
+ cards = drill.parse(MULTITAG)
+ assert cards[0][2] == ["fundamentals", "fundamental"] # section slug + org tag
+ assert cards[1][2] == ["fundamentals"] # drill-only -> section only
+
+
+def test_parse_heading_without_drill_tag_is_not_a_card(drill):
+ """A tagged heading missing :drill: is not a card (e.g. :note:)."""
+ assert drill.parse("* S\n** Just a note :note:\nbody\n") == []
+
+
+def test_parse_tag_filter_returns_only_cards_with_that_org_tag(drill):
+ """--tag-filter narrows to cards carrying the given org tag."""
+ cards = drill.parse(MULTITAG, tag_filter="fundamental")
+ assert len(cards) == 1
+ assert cards[0][0] == "What is LEO?"
+
+
+def test_parse_body_bounded_by_any_l1_or_l2_heading(drill):
+ """A card body stops at the next L1/L2 heading, multi-tagged or not."""
+ text = "** Q1? :a:drill:\nbody1\n** Q2? :drill:\nbody2\n"
+ cards = drill.parse(text)
+ assert cards[0][1] == "body1"
+ assert cards[1][1] == "body2"
+
+
+def test_card_guid_salt_changes_the_guid(drill, monkeypatch):
+ """--guid-salt gives a subset deck its own GUID space; no salt is unchanged."""
+ monkeypatch.setattr(drill.genanki, "guid_for", lambda *a: ":".join(a), raising=False)
+ assert drill.card_guid("front", None) == "front"
+ assert drill.card_guid("front", "fundamentals") == "fundamentals:front"
+ assert drill.card_guid("front", None) != drill.card_guid("front", "fundamentals")
def test_parse_joins_multiline_body_with_br(drill):
diff --git a/claude-templates/.ai/scripts/tests/test_inbox_send.py b/claude-templates/.ai/scripts/tests/test_inbox_send.py
index a0094dc..9b0a8c6 100644
--- a/claude-templates/.ai/scripts/tests/test_inbox_send.py
+++ b/claude-templates/.ai/scripts/tests/test_inbox_send.py
@@ -97,6 +97,52 @@ class TestInboxSendDiscovery:
result = run_script(["--list"], roots=[tmp_path / "does-not-exist"])
assert result.returncode == 0
+ def test_inbox_send_list_displays_dot_stripped_name(self, project_root, run_script, tmp_path):
+ """Dotted project basenames display dot-stripped (.emacs.d → emacsd)."""
+ project_root(".emacs.d")
+ result = run_script(["--list"], roots=[tmp_path / "projects"])
+ assert "emacsd" in result.stdout
+
+
+class TestInboxSendDotAlias:
+ """A dotted project basename resolves both verbatim and dot-stripped."""
+
+ def test_resolves_by_dot_stripped_alias(self, project_root, run_script, tmp_path):
+ """'emacsd' delivers to the .emacs.d project."""
+ project_root(".emacs.d")
+ cwd = project_root("source")
+ run_script(
+ ["emacsd", "--text", "hi"],
+ cwd=cwd, roots=[tmp_path / "projects"],
+ )
+ files = list((tmp_path / "projects" / ".emacs.d" / "inbox").iterdir())
+ assert len(files) == 1
+
+ def test_resolves_by_exact_dotted_name_still(self, project_root, run_script, tmp_path):
+ """Backward-compat: the verbatim '.emacs.d' target still resolves."""
+ project_root(".emacs.d")
+ cwd = project_root("source")
+ run_script(
+ [".emacs.d", "--text", "hi"],
+ cwd=cwd, roots=[tmp_path / "projects"],
+ )
+ files = list((tmp_path / "projects" / ".emacs.d" / "inbox").iterdir())
+ assert len(files) == 1
+
+ def test_exact_match_wins_over_alias(self, project_root, run_script, tmp_path):
+ """An exact basename match is preferred over a dot-stripped collision."""
+ project_root("emacsd") # exact
+ project_root(".emacs.d") # would also normalize to 'emacsd'
+ cwd = project_root("source")
+ run_script(
+ ["emacsd", "--text", "hi"],
+ cwd=cwd, roots=[tmp_path / "projects"],
+ )
+ exact = list((tmp_path / "projects" / "emacsd" / "inbox").iterdir())
+ dotted = list((tmp_path / "projects" / ".emacs.d" / "inbox").iterdir())
+ assert len(exact) == 1
+ assert dotted == []
+
# ----------------------------------------------------------------------
# Slug derivation from text and from filenames
@@ -355,3 +401,192 @@ class TestInboxSendErrors:
assert result.returncode != 0
files = list((tmp_path / "projects" / "target" / "inbox").iterdir())
assert files == []
+
+
+# ----------------------------------------------------------------------
+# Filename collisions (two sends deriving the same name must not overwrite)
+# ----------------------------------------------------------------------
+
+def _load_module():
+ import importlib.util
+ spec = importlib.util.spec_from_file_location("inbox_send", SCRIPT)
+ mod = importlib.util.module_from_spec(spec)
+ spec.loader.exec_module(mod)
+ return mod
+
+
+class TestFilenameCollisions:
+ """Two sends in the same minute with the same leading phrase derived
+ identical filenames and the second silently overwrote the first
+ (a message was lost this way, 2026-07-02)."""
+
+ def test_send_text_same_minute_same_phrase_keeps_both(self, tmp_path):
+ from datetime import datetime
+ mod = _load_module()
+ inbox = tmp_path / "inbox"
+ inbox.mkdir()
+ now = datetime(2026, 7, 2, 5, 42, 0)
+ prefix = "identical leading phrase long enough to fill the whole slug budget entirely"
+ first = mod.send_text(inbox, prefix + " tail one", "archsetup", None, now)
+ second = mod.send_text(inbox, prefix + " tail two", "archsetup", None, now)
+ assert first != second
+ assert first.exists() and second.exists()
+ assert first.name != second.name
+ assert "tail one" in first.read_text()
+ assert "tail two" in second.read_text()
+
+ def test_send_text_collision_suffix_increments(self, tmp_path):
+ from datetime import datetime
+ mod = _load_module()
+ inbox = tmp_path / "inbox"
+ inbox.mkdir()
+ now = datetime(2026, 7, 2, 5, 42, 0)
+ paths = [mod.send_text(inbox, "same lead phrase differs later A", "src", "fixed-slug", now)
+ for _ in range(3)]
+ names = [p.name for p in paths]
+ assert names[0].endswith("fixed-slug.org")
+ assert names[1].endswith("fixed-slug-2.org")
+ assert names[2].endswith("fixed-slug-3.org")
+
+ def test_send_file_collision_preserves_extension(self, tmp_path):
+ from datetime import datetime
+ mod = _load_module()
+ inbox = tmp_path / "inbox"
+ inbox.mkdir()
+ src = tmp_path / "note.org"
+ src.write_text("body one")
+ now = datetime(2026, 7, 2, 5, 42, 0)
+ first = mod.send_file(inbox, src, "src", None, now)
+ src.write_text("body two")
+ second = mod.send_file(inbox, src, "src", None, now)
+ assert second.name.endswith("note-2.org")
+ assert first.read_text() == "body one"
+ assert second.read_text() == "body two"
+
+ def test_cli_two_rapid_sends_lose_nothing(self, project_root, run_script, tmp_path):
+ project_root("sender")
+ target = project_root("receiver")
+ roots = [tmp_path / "projects"]
+ prefix = "identical leading phrase long enough to fill the whole slug budget entirely"
+ run_script(["receiver", "--text", prefix + " message one"],
+ cwd=tmp_path / "projects" / "sender", roots=roots)
+ run_script(["receiver", "--text", prefix + " message two"],
+ cwd=tmp_path / "projects" / "sender", roots=roots)
+ files = list((target / "inbox").iterdir())
+ assert len(files) == 2
+ bodies = "".join(f.read_text() for f in files)
+ assert "message one" in bodies and "message two" in bodies
+
+
+class TestAtomicWrite:
+ """A send wrote straight to the destination path in another project's
+ inbox/, and write_text truncates on open, so any mid-write failure left a
+ zero-byte .org there. inbox-status counts that phantom as a pending
+ handoff, blocking a turn in the receiving project over a file with no
+ content (2026-07-23). The write must be atomic: the inbox sees a complete
+ file or nothing."""
+
+ def test_send_text_writes_utf8(self, tmp_path):
+ from datetime import datetime
+ mod = _load_module()
+ inbox = tmp_path / "inbox"
+ inbox.mkdir()
+ now = datetime(2026, 7, 23, 4, 36, 0)
+ # An em dash and an accented char — both non-ASCII.
+ dest = mod.send_text(inbox, "accent café and dash — here", "src", None, now)
+ # Reading as utf-8 must round-trip; a locale-encoded write would raise
+ # under a C locale, and reading back proves the bytes are utf-8.
+ assert "—" in dest.read_text(encoding="utf-8")
+
+ def test_send_text_no_partial_on_write_failure(self, tmp_path, monkeypatch):
+ from datetime import datetime
+ mod = _load_module()
+ inbox = tmp_path / "inbox"
+ inbox.mkdir()
+ now = datetime(2026, 7, 23, 4, 36, 0)
+ # Force the atomic finalize to fail after the temp file is written.
+ def boom(*a, **k):
+ raise OSError("disk full")
+ monkeypatch.setattr(mod.os, "replace", boom)
+ with pytest.raises(OSError):
+ mod.send_text(inbox, "a message that should never half-land", "src", None, now)
+ # No phantom, no leftover temp: the inbox is empty.
+ assert list(inbox.iterdir()) == []
+
+ def test_send_text_leaves_no_temp_on_success(self, tmp_path):
+ from datetime import datetime
+ mod = _load_module()
+ inbox = tmp_path / "inbox"
+ inbox.mkdir()
+ now = datetime(2026, 7, 23, 4, 36, 0)
+ dest = mod.send_text(inbox, "clean send", "src", None, now)
+ assert list(inbox.iterdir()) == [dest]
+
+ def test_send_file_no_partial_on_write_failure(self, tmp_path, monkeypatch):
+ from datetime import datetime
+ mod = _load_module()
+ inbox = tmp_path / "inbox"
+ inbox.mkdir()
+ src = tmp_path / "note.org"
+ src.write_text("body")
+ now = datetime(2026, 7, 23, 4, 36, 0)
+ def boom(*a, **k):
+ raise OSError("disk full")
+ monkeypatch.setattr(mod.os, "replace", boom)
+ with pytest.raises(OSError):
+ mod.send_file(inbox, src, "src", None, now)
+ assert list(inbox.iterdir()) == []
+
+ def test_send_file_leaves_no_temp_on_success(self, tmp_path):
+ from datetime import datetime
+ mod = _load_module()
+ inbox = tmp_path / "inbox"
+ inbox.mkdir()
+ src = tmp_path / "note.org"
+ src.write_text("payload")
+ now = datetime(2026, 7, 23, 4, 36, 0)
+ dest = mod.send_file(inbox, src, "src", None, now)
+ assert list(inbox.iterdir()) == [dest]
+ assert dest.read_text() == "payload"
+
+
+class TestSmallerDefects:
+ """Two low-severity defects found reading inbox-send during the 2026-07-23
+ sweep: an unreadable source raised an uncaught traceback instead of the
+ clean error every other failure path produces, and a roots config naming
+ both a parent and one of its children listed the same project twice."""
+
+ def test_unreadable_source_gives_clean_error_not_traceback(
+ self, project_root, run_script, tmp_path
+ ):
+ project_root("sender")
+ project_root("receiver")
+ roots = [tmp_path / "projects"]
+ src = tmp_path / "secret.bin"
+ src.write_text("x")
+ src.chmod(0o000)
+ try:
+ result = run_script(
+ ["receiver", "--file", str(src)],
+ cwd=tmp_path / "projects" / "sender",
+ roots=roots,
+ expect_failure=True,
+ )
+ finally:
+ src.chmod(0o644)
+ assert result.returncode == 1
+ # The clean "inbox-send: <message>" shape, not a Python traceback.
+ assert result.stderr.startswith("inbox-send:")
+ assert "Traceback" not in result.stderr
+
+ def test_discover_projects_dedupes_parent_and_child_root(self, tmp_path):
+ mod = _load_module()
+ # A project directory, reachable both as a child of its parent root and
+ # as a root in its own right.
+ parent = tmp_path / "projects"
+ proj = parent / "app"
+ (proj / ".ai").mkdir(parents=True)
+ (proj / "inbox").mkdir()
+ found = mod.discover_projects([parent, proj])
+ resolved = [p.resolve() for p in found]
+ assert resolved.count(proj.resolve()) == 1
diff --git a/claude-templates/.ai/scripts/tests/test_route_recommend.py b/claude-templates/.ai/scripts/tests/test_route_recommend.py
new file mode 100644
index 0000000..2ec900a
--- /dev/null
+++ b/claude-templates/.ai/scripts/tests/test_route_recommend.py
@@ -0,0 +1,152 @@
+"""Tests for route_recommend.py — the wrap-up routing recommendation engine.
+
+The core is a pure function recommend(item, projects) -> (destination, confidence):
+- strong: a project's name (or its dot-stripped form) appears literally in the item
+- weak: a distinctive name token overlaps, but the full name doesn't
+- none: no overlap; the item stays put (destination is None)
+
+A multi-way tie at the top tier downgrades to weak with a deterministic pick.
+An empty project list yields none.
+
+The CLI wires this to inbox-send.py's discover_projects (sandboxed here via the
+INBOX_SEND_ROOTS env var, the same hook inbox-send's own tests use).
+"""
+
+import subprocess
+import sys
+from pathlib import Path
+
+SCRIPTS = Path(__file__).parent.parent
+SCRIPT = SCRIPTS / "route_recommend.py"
+sys.path.insert(0, str(SCRIPTS))
+
+import route_recommend as rr # noqa: E402
+
+
+# --- pure function: the five spec'd cases -----------------------------------
+
+def test_strong_match_named_literally():
+ dest, conf = rr.recommend("fix the rulesets refactor command", ["rulesets", "home", "work"])
+ assert (dest, conf) == ("rulesets", "strong")
+
+
+def test_strong_match_via_dot_stripped_name():
+ # ".emacs.d" addressed as "emacsd" in the item is still a literal hit.
+ dest, conf = rr.recommend("update the emacsd ai-term module", [".emacs.d", "rulesets"])
+ assert (dest, conf) == (".emacs.d", "strong")
+
+
+def test_strong_match_dotted_name_verbatim():
+ dest, conf = rr.recommend("patch .emacs.d startup", [".emacs.d", "rulesets"])
+ assert (dest, conf) == (".emacs.d", "strong")
+
+
+def test_weak_match_topic_token_only():
+ # "wttrin" is a token of "emacs-wttrin" but the full name isn't present.
+ dest, conf = rr.recommend("the wttrin weather bug", ["emacs-wttrin", "rulesets"])
+ assert (dest, conf) == ("emacs-wttrin", "weak")
+
+
+def test_no_match_stays_put():
+ dest, conf = rr.recommend("calibrate the telescope mount", ["rulesets", "deepsat"])
+ assert dest is None
+ assert conf == "none"
+
+
+def test_two_project_strong_tie_downgrades_to_weak():
+ # Both named literally → ambiguous → weak, deterministic tie-break (alphabetical).
+ dest, conf = rr.recommend("sync rulesets and home configs", ["rulesets", "home", "work"])
+ assert conf == "weak"
+ assert dest == "home" # tie-break: most-overlap then alphabetical
+
+
+def test_empty_project_list_is_none():
+ assert rr.recommend("anything at all", []) == (None, "none")
+
+
+# --- boundary / robustness --------------------------------------------------
+
+def test_literal_name_requires_word_boundary():
+ # "home" must not match inside "homeowner".
+ dest, conf = rr.recommend("the homeowner association meeting", ["home", "rulesets"])
+ assert dest is None and conf == "none"
+
+
+def test_path_mention_counts_as_literal():
+ dest, conf = rr.recommend("edit ~/code/rulesets/Makefile", ["rulesets", "home"])
+ assert (dest, conf) == ("rulesets", "strong")
+
+
+def test_strong_beats_weak_when_both_present():
+ # "rulesets" named literally (strong) outranks an emacs-wttrin token hit (weak).
+ dest, conf = rr.recommend("the wttrin fix belongs in rulesets", ["rulesets", "emacs-wttrin"])
+ assert (dest, conf) == ("rulesets", "strong")
+
+
+# --- CLI + discovery reuse (sandboxed roots) --------------------------------
+
+def _run(args, roots, item):
+ import os
+ env = {"PATH": os.environ.get("PATH", ""), "HOME": os.environ.get("HOME", "/tmp"),
+ "INBOX_SEND_ROOTS": ":".join(str(r) for r in roots)}
+ return subprocess.run([sys.executable, str(SCRIPT), "--item", item, *args],
+ capture_output=True, text=True, env=env)
+
+
+def _mk_project(tmp_path, name):
+ proj = tmp_path / "projects" / name
+ (proj / ".ai").mkdir(parents=True, exist_ok=True)
+ (proj / "inbox").mkdir(exist_ok=True)
+ return proj
+
+
+def test_cli_discovers_and_recommends(tmp_path):
+ _mk_project(tmp_path, "foo")
+ _mk_project(tmp_path, "bar")
+ r = _run([], roots=[tmp_path / "projects"], item="fix the foo widget")
+ assert r.returncode == 0
+ assert r.stdout.strip() == "foo\tstrong"
+
+
+def test_cli_no_match_prints_none(tmp_path):
+ _mk_project(tmp_path, "foo")
+ r = _run([], roots=[tmp_path / "projects"], item="unrelated grocery list")
+ assert r.returncode == 0
+ assert r.stdout.strip() == "none"
+
+
+def test_cli_exclude_drops_current_project(tmp_path):
+ _mk_project(tmp_path, "foo")
+ _mk_project(tmp_path, "bar")
+ # Item names foo, but foo is excluded as the current project → no other match.
+ r = _run(["--exclude", "foo"], roots=[tmp_path / "projects"], item="fix the foo widget")
+ assert r.returncode == 0
+ assert r.stdout.strip() == "none"
+
+
+# ----------------------------------------------------------------------
+# Duplicate candidate names
+#
+# Projects are collapsed to bare basenames, so two projects sharing a basename
+# across roots (~/code/notes and ~/projects/notes) appear twice in the candidate
+# list. Both literal-match, recommend read len(strong) > 1 as an ambiguous tie,
+# and a correct strong match was downgraded to weak. Latent when discovered
+# 2026-07-24 (27 projects, 27 distinct basenames) but real.
+# ----------------------------------------------------------------------
+
+def test_duplicate_candidate_name_keeps_strong_confidence():
+ assert rr.recommend("fix the notes thing", ["notes", "other"]) == ("notes", "strong")
+ # The same name twice must not read as a tie.
+ assert rr.recommend("fix the notes thing", ["notes", "notes", "other"]) == ("notes", "strong")
+
+
+def test_genuine_ambiguity_still_downgrades():
+ # Two DIFFERENT projects both matching is a real tie and stays weak — the
+ # dedupe must collapse identical names only, never real ambiguity.
+ dest, conf = rr.recommend("notes and other both", ["notes", "other"])
+ assert conf == "weak"
+
+
+def test_duplicates_do_not_change_the_chosen_destination():
+ dest, _ = rr.recommend("fix the notes thing", ["notes", "notes"])
+ assert dest == "notes"
diff --git a/claude-templates/.ai/scripts/tests/test_upcoming_birthdays.py b/claude-templates/.ai/scripts/tests/test_upcoming_birthdays.py
new file mode 100644
index 0000000..1e15183
--- /dev/null
+++ b/claude-templates/.ai/scripts/tests/test_upcoming_birthdays.py
@@ -0,0 +1,168 @@
+"""Tests for upcoming_birthdays.py — the daily-prep upcoming-birthdays block.
+
+Pure core:
+ parse_birthdays(text) -> [Birthday(name, month, day, year|None), ...]
+ upcoming(birthdays, today, window=30) -> [Upcoming(name, date, days_away, age|None), ...]
+ format_block(items, window, callout_days=7) -> str
+
+Birth year 1900 is the placeholder org-contacts uses when the real year is
+unknown; those entries carry year=None and render date-only (no age).
+"""
+
+import datetime as dt
+import subprocess
+import sys
+from pathlib import Path
+
+SCRIPTS = Path(__file__).parent.parent
+SCRIPT = SCRIPTS / "upcoming_birthdays.py"
+sys.path.insert(0, str(SCRIPTS))
+
+import upcoming_birthdays as ub # noqa: E402
+
+
+# --- parse_birthdays --------------------------------------------------------
+
+def test_parse_reads_name_month_day_year():
+ text = "** Jane Doe\n:PROPERTIES:\n:BIRTHDAY: 1970-08-05\n:END:\n"
+ bdays = ub.parse_birthdays(text)
+ assert bdays == [ub.Birthday("Jane Doe", 8, 5, 1970)]
+
+
+def test_parse_placeholder_year_1900_becomes_none():
+ text = "** John Smith\n:PROPERTIES:\n:BIRTHDAY: 1900-07-14\n:END:\n"
+ bdays = ub.parse_birthdays(text)
+ assert bdays == [ub.Birthday("John Smith", 7, 14, None)]
+
+
+def test_parse_skips_contacts_without_birthday():
+ text = (
+ "** No Birthday\n:PROPERTIES:\n:PHONE: 555\n:END:\n"
+ "** Has Birthday\n:PROPERTIES:\n:BIRTHDAY: 1990-03-02\n:END:\n"
+ )
+ bdays = ub.parse_birthdays(text)
+ assert [b.name for b in bdays] == ["Has Birthday"]
+
+
+def test_parse_strips_heading_stars_and_tags():
+ text = "*** Bob Jones :friend:\n:PROPERTIES:\n:BIRTHDAY: 1980-01-01\n:END:\n"
+ bdays = ub.parse_birthdays(text)
+ assert bdays[0].name == "Bob Jones"
+
+
+def test_parse_ignores_malformed_birthday_lines():
+ text = "** Bad Date\n:PROPERTIES:\n:BIRTHDAY: not-a-date\n:END:\n"
+ assert ub.parse_birthdays(text) == []
+
+
+# --- upcoming ---------------------------------------------------------------
+
+TODAY = dt.date(2026, 7, 18)
+
+
+def test_upcoming_birthday_today_is_zero_days():
+ bdays = [ub.Birthday("Today Person", 7, 18, 1990)]
+ got = ub.upcoming(bdays, TODAY)
+ assert got[0].days_away == 0
+ assert got[0].date == dt.date(2026, 7, 18)
+
+
+def test_upcoming_includes_within_window():
+ bdays = [ub.Birthday("Soon", 7, 23, 1990)]
+ got = ub.upcoming(bdays, TODAY, window=30)
+ assert got[0].days_away == 5
+
+
+def test_upcoming_excludes_beyond_window():
+ bdays = [ub.Birthday("Far", 9, 1, 1990)] # 45 days out
+ assert ub.upcoming(bdays, TODAY, window=30) == []
+
+
+def test_upcoming_boundary_day_30_included_day_31_excluded():
+ on = [ub.Birthday("On", 8, 17, 1990)] # exactly 30 days
+ off = [ub.Birthday("Off", 8, 18, 1990)] # 31 days
+ assert ub.upcoming(on, TODAY, window=30)[0].days_away == 30
+ assert ub.upcoming(off, TODAY, window=30) == []
+
+
+def test_upcoming_uses_next_year_when_this_years_passed():
+ # today is 2026-07-18; a Jan 5 birthday recurs on 2027-01-05
+ today = dt.date(2026, 12, 27)
+ bdays = [ub.Birthday("New Year", 1, 5, 1990)]
+ got = ub.upcoming(bdays, today, window=30)
+ assert got[0].date == dt.date(2027, 1, 5)
+ assert got[0].days_away == 9
+
+
+def test_upcoming_age_is_occurrence_year_minus_birth_year():
+ bdays = [ub.Birthday("Ager", 7, 23, 1970)]
+ got = ub.upcoming(bdays, TODAY)
+ assert got[0].age == 56 # 2026 - 1970
+
+
+def test_upcoming_age_none_for_placeholder():
+ bdays = [ub.Birthday("Placeholder", 7, 23, None)]
+ got = ub.upcoming(bdays, TODAY)
+ assert got[0].age is None
+
+
+def test_upcoming_sorted_by_days_away():
+ bdays = [
+ ub.Birthday("Later", 8, 10, 1990),
+ ub.Birthday("Sooner", 7, 20, 1990),
+ ]
+ got = ub.upcoming(bdays, TODAY)
+ assert [u.name for u in got] == ["Sooner", "Later"]
+
+
+def test_upcoming_leap_day_maps_to_feb_28_in_non_leap_year():
+ today = dt.date(2027, 2, 1) # 2027 is not a leap year
+ bdays = [ub.Birthday("Leapling", 2, 29, 2000)]
+ got = ub.upcoming(bdays, today, window=30)
+ assert got[0].date == dt.date(2027, 2, 28)
+
+
+# --- format_block -----------------------------------------------------------
+
+def test_format_block_empty_reports_none():
+ out = ub.format_block([], window=30)
+ assert "No birthdays" in out
+
+
+def test_format_block_callout_marks_within_seven_days():
+ items = [ub.Upcoming("Soon", dt.date(2026, 7, 22), 4, 40)]
+ out = ub.format_block(items, window=30, callout_days=7)
+ assert "⚠" in out
+ assert "Soon" in out
+ assert "40" in out # age shown
+
+
+def test_format_block_beyond_callout_is_not_flagged():
+ items = [ub.Upcoming("Later", dt.date(2026, 8, 10), 23, 30)]
+ out = ub.format_block(items, window=30, callout_days=7)
+ assert "⚠" not in out
+
+
+def test_format_block_placeholder_shows_date_only_no_age():
+ items = [ub.Upcoming("NoYear", dt.date(2026, 7, 25), 7, None)]
+ out = ub.format_block(items, window=30)
+ assert "NoYear" in out
+ assert "turns" not in out
+
+
+# --- CLI --------------------------------------------------------------------
+
+def test_cli_runs_against_a_fixture_file(tmp_path):
+ contacts = tmp_path / "contacts.org"
+ contacts.write_text(
+ "** Alice\n:PROPERTIES:\n:BIRTHDAY: 1990-07-20\n:END:\n"
+ "** Bob\n:PROPERTIES:\n:BIRTHDAY: 1900-12-01\n:END:\n"
+ )
+ res = subprocess.run(
+ [sys.executable, str(SCRIPT), "--file", str(contacts),
+ "--today", "2026-07-18", "--window", "30"],
+ capture_output=True, text=True,
+ )
+ assert res.returncode == 0
+ assert "Alice" in res.stdout
+ assert "Bob" not in res.stdout # Dec 1 is outside the 30-day window
diff --git a/claude-templates/.ai/scripts/todo-cleanup.el b/claude-templates/.ai/scripts/todo-cleanup.el
index 6b3081a..516e9b1 100644
--- a/claude-templates/.ai/scripts/todo-cleanup.el
+++ b/claude-templates/.ai/scripts/todo-cleanup.el
@@ -5,10 +5,14 @@
;; emacs --batch -q -l todo-cleanup.el --check todo.org # hygiene report only
;; emacs --batch -q -l todo-cleanup.el --archive-done todo.org # archive completed subtrees
;; emacs --batch -q -l todo-cleanup.el --archive-done --check todo.org # preview the archive
+;; emacs --batch -q -l todo-cleanup.el --seal todo.org # seal the working archive to resolved-YYYY-MM-DD.org
+;; emacs --batch -q -l todo-cleanup.el --seal --check todo.org # preview the seal
+;; emacs --batch -q -l todo-cleanup.el --convert-subtasks todo.org # dated-rewrite done level-3+ sub-tasks
+;; emacs --batch -q -l todo-cleanup.el --convert-subtasks --check todo.org # preview the conversion
;; emacs --batch -q -l todo-cleanup.el --sync-child-priority todo.org # bump children whose priority drifted below the parent's
;; emacs --batch -q -l todo-cleanup.el --check-child-priority todo.org # preview the sync (same as --sync-child-priority --check)
;;
-;; Three independent modes:
+;; Four independent modes:
;;
;; * Default (hygiene). Designed for the wrap-it-up workflow: cheap, idempotent,
;; safe to run every session.
@@ -25,14 +29,60 @@
;; line isn't in canonical position. Reports these for manual fix; doesn't
;; auto-rewrite (preserving real state-log history is judgement work).
;;
-;; * --archive-done (opt-in). Moves every level-2 subtree whose TODO state is
-;; DONE or CANCELLED out of the "Open Work" section and into the "Resolved"
-;; section of the same file, subtree intact. The sections are matched by a
-;; unique level-1 heading containing "Open Work" (case-insensitive) and one
-;; containing "Resolved"; if either is missing or ambiguous, the file is
-;; skipped with a message. Only direct level-2 children move — a DONE entry
-;; nested under an open parent stays put. Archiving is consequential, so it's
-;; never run by default; it does *not* also run the hygiene passes.
+;; * --archive-done (opt-in). Two steps, in order:
+;;
+;; 1. Moves every level-2 subtree whose TODO state is DONE or CANCELLED out of
+;; the "Open Work" section and into the "Resolved" section of the same
+;; file, subtree intact. The sections are matched by a unique level-1
+;; heading containing "Open Work" (case-insensitive) and one containing
+;; "Resolved"; if either is missing or ambiguous, the file is skipped with
+;; a message. Only direct level-2 children move — a DONE entry nested under
+;; an open parent stays put.
+;;
+;; 2. Ages the "Resolved" section: a level-2 DONE/CANCELLED subtree is moved
+;; out to `tc-archive-file' (default `archive/task-archive.org' beside the
+;; todo file) when its CLOSED date is older than `tc-archive-retain-days'
+;; (default 31 — one month) OR its CLOSED date can't be parsed. The last
+;; month of closed tasks stays browsable in the file itself; older ones age
+;; out. The unparseable-CLOSED case archives too, deliberately: a
+;; keyword-complete task with no readable close date is cruft, not live
+;; work. Set `tc-archive-retain-days' to nil to disable this step (legacy
+;; in-file-only behavior). The aging date is `tc-archive-reference-date'
+;; when set (tests), otherwise the real current date. The archive inherits
+;; the todo file's gitignore status: when the todo file is gitignored, the
+;; archive path is added to .gitignore before the first write, so private
+;; task history never lands in a tracked path (see
+;; `tc--ensure-archive-gitignored').
+;;
+;; Archiving is consequential, so it's never run by default; it does *not*
+;; also run the hygiene passes.
+;;
+;; * --seal (opt-in). Renames the working archive file (`tc-archive-file',
+;; default `archive/task-archive.org') to `resolved-YYYY-MM-DD.org' beside it,
+;; dated by the seal run, and leaves the next `--archive-done' to recreate a
+;; fresh working file. The dated file means "everything sealed as of that
+;; date" — not a calendar quarter — so a task closed late in a quarter and
+;; archived after the boundary is never mislabeled; cadence (e.g. quarterly)
+;; becomes independent of correctness and any slip is harmless. The sealed
+;; file inherits the todo file's gitignore status the same way the working
+;; archive does. A no-op (reported) when there's no working archive to seal;
+;; refuses to clobber an existing `resolved-<today>.org'. Honors `--check'.
+;; The seal date is `tc-archive-reference-date' when set (tests), otherwise the
+;; real current date.
+;;
+;; * --convert-subtasks (opt-in). Rewrites every level-3-and-deeper heading whose
+;; TODO state is DONE/CANCELLED/FAILED into a dated event-log entry
+;; (`<stars> YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <text>'), dropping the keyword,
+;; priority cookie, and tags, and removing the now-redundant CLOSED line. The
+;; date and time come from that entry's own CLOSED cookie; a date-only close
+;; yields 00:00:00, and the UTC offset is computed DST-aware for that date.
+;; This enforces the todo-format depth rule that interactive closes
+;; (`org-log-done' → DONE + CLOSED) and `--archive-done' (level-2 only) leave
+;; unapplied. The heading text is preserved verbatim — a batch tool can't
+;; past-tense an imperative title reliably. Idempotent (an already-dated
+;; heading has no done keyword); a done sub-task with no parseable CLOSED date
+;; is flagged and left alone, never stamped with a fabricated date. Like
+;; --archive-done it does not also run the hygiene passes.
;;
;; * --sync-child-priority (opt-in). Walks every heading with a priority cookie
;; ([#A]-[#D]) and, for each of its direct child headings whose own priority
@@ -50,18 +100,41 @@
;; --check-child-priority is the report-only alias for --sync-child-priority
;; --check.
+;; Before any modification a backup is copied to
+;; /tmp/<basename>.before-todo-cleanup.<YYYYMMDD-HHMMSS>
+;; matching lint-org.el and wrap-org-table.el. Skipped under --check, which
+;; writes nothing.
+;;
+
(require 'org)
(require 'cl-lib)
+(require 'calendar)
(setq org-todo-keywords
- '((sequence "TODO" "DOING" "WAITING" "NEXT" "|" "DONE" "CANCELLED")))
+ '((sequence "TODO" "DOING" "WAITING" "NEXT" "|" "DONE" "CANCELLED" "FAILED")))
(defconst tc-done-states '("DONE" "CANCELLED")
"TODO keywords that mark an entry as completed for `--archive-done'.")
+(defconst tc--convert-done-states '("DONE" "CANCELLED" "FAILED")
+ "TODO keywords whose level-3-and-deeper entries `--convert-subtasks' rewrites
+to dated event-log entries. Broader than `tc-done-states' because a FAILED
+sub-task is terminal too and belongs in the parent's dated history.")
+
(defconst tc--priority-cookie-regexp "\\[#\\([A-Z]\\)\\]"
"Regexp matching an org priority cookie. Match group 1 is the letter.")
+(defconst tc--planning-cookie-regexp
+ "\\(?:CLOSED\\|DEADLINE\\|SCHEDULED\\):[ \t]*[[<][^]>\n]*[]>]"
+ "One org planning cookie: a CLOSED/DEADLINE/SCHEDULED keyword followed by a
+bracketed (inactive) or angled (active) timestamp.")
+
+(defconst tc--planning-line-regexp
+ (concat "\\`[ \t]*\\(?:" tc--planning-cookie-regexp "[ \t]*\\)+\\'")
+ "A whole org planning line: nothing but planning cookies and whitespace.
+Anchored to a single line's contents so a line mixing a cookie with real body
+text is never matched.")
+
(defconst tc-no-sync-tag "no-sync"
"Org tag that opts a heading and all its descendants out of
`--sync-child-priority'. Inherits down: a tag on an ancestor counts for
@@ -70,11 +143,40 @@ every heading below it.")
(defvar tc-fixes 0)
(defvar tc-archived 0)
(defvar tc-bumped 0)
+(defvar tc-converted 0)
(defvar tc-issues nil)
+(defvar tc-sealed 0)
(defvar tc-check-only nil)
(defvar tc-archive-done nil)
(defvar tc-sync-child-priority nil)
+(defvar tc-convert-subtasks nil)
+(defvar tc-seal nil)
(defvar tc-current-file nil)
+(defvar tc-current-dir nil)
+(defvar tc-archived-to-file 0)
+
+(defconst tc-archive-retain-days-default 31
+ "Default retention window (days) for the `--archive-done' file-aging step —
+one month. A closed Resolved subtree stays in-file for this long before it ages
+out to `tc-archive-file'; the last month of resolved work stays browsable in the
+todo file itself. Named so the \"one month\" contract is explicit and testable.")
+
+(defvar tc-archive-retain-days tc-archive-retain-days-default
+ "Retention window for the `--archive-done' file-aging step. A closed Resolved
+subtree whose CLOSED date is within this many days of the reference date stays
+in the in-file Resolved section; an older one is moved out to `tc-archive-file'.
+A subtree with no parseable CLOSED date is aged out too (a keyword-complete task
+with no readable close date is cruft, not live work). nil disables the aging
+step entirely, leaving the legacy in-file-only behavior. Defaults to
+`tc-archive-retain-days-default' (one month).")
+
+(defvar tc-archive-reference-date nil
+ "(YEAR MONTH DAY) treated as \"today\" when aging Resolved subtrees out to a
+file; nil means the real current date. Set in tests for determinism.")
+
+(defvar tc-archive-file nil
+ "Destination file for aged-out Resolved subtrees; nil means
+`archive/task-archive.org' beside the todo file being processed.")
;;; ---------------------------------------------------------------------------
;;; Hygiene mode
@@ -224,7 +326,8 @@ are reported but not performed."
:line (line-number-at-pos)
:heading (org-get-heading t t t t))
tc-issues)
- (cl-incf tc-archived))))
+ (cl-incf tc-archived)))
+ (tc-archive-old-resolved-to-file))
(t
(catch 'done
(while t
@@ -252,7 +355,216 @@ are reported but not performed."
(cl-incf tc-archived)
(push (list :kind 'archive-moved :file tc-current-file
:line line :heading heading)
- tc-issues)))))))))
+ tc-issues)))))
+ (tc-archive-old-resolved-to-file)))))
+
+;;; ---------------------------------------------------------------------------
+;;; --archive-done: age old Resolved subtrees out to a file
+
+(defconst tc-archive-file-scaffold
+ "#+TITLE: Task Archive\n#+FILETAGS: :archive:\n\n* Resolved (archived)\n"
+ "Initial content written to a fresh `tc-archive-file'. Aged subtrees are
+appended as level-2 children under the level-1 heading.")
+
+(defun tc--reference-absolute ()
+ "Absolute (Gregorian serial) day number of the aging reference date —
+`tc-archive-reference-date' when set, otherwise the real current date."
+ (if tc-archive-reference-date
+ (pcase-let ((`(,y ,m ,d) tc-archive-reference-date))
+ (calendar-absolute-from-gregorian (list m d y)))
+ (pcase-let ((`(,m ,d ,y) (calendar-current-date)))
+ (calendar-absolute-from-gregorian (list m d y)))))
+
+(defun tc--closed-absolute-in-region (beg end)
+ "Absolute day number of the first CLOSED: [YYYY-MM-DD ...] line in BEG..END,
+or nil when the region carries no parseable CLOSED date. The task's own CLOSED
+line sits in canonical position directly under the heading, so the first match
+in the subtree is the task's close."
+ (save-excursion
+ (goto-char beg)
+ (when (re-search-forward
+ "CLOSED:[ \t]*\\[\\([0-9][0-9][0-9][0-9]\\)-\\([0-9][0-9]\\)-\\([0-9][0-9]\\)"
+ end t)
+ (calendar-absolute-from-gregorian
+ (list (string-to-number (match-string 2))
+ (string-to-number (match-string 3))
+ (string-to-number (match-string 1)))))))
+
+(defun tc--archive-file-path ()
+ "Resolve the destination file for aged-out subtrees: `tc-archive-file' if set,
+else `archive/task-archive.org' beside the todo file being processed."
+ (or tc-archive-file
+ (and tc-current-dir
+ (expand-file-name "archive/task-archive.org" tc-current-dir))))
+
+(defun tc--git-ignored-p (path)
+ "Non-nil when PATH is gitignored (git check-ignore exits 0). nil on any git
+error or when git is unavailable."
+ (let ((default-directory (or tc-current-dir default-directory)))
+ (eq 0 (ignore-errors
+ (call-process "git" nil nil nil "check-ignore" "-q"
+ (expand-file-name path))))))
+
+(defun tc--ensure-archive-gitignored (archive-path)
+ "Keep the aged-out archive as private as the todo file it derives from. When the
+todo file being processed is gitignored but ARCHIVE-PATH is not, append a
+root-relative ignore entry for ARCHIVE-PATH to the project's .gitignore. No-op
+when the todo file is tracked, the archive is already ignored, or there is no git
+work tree — so track-mode projects (todo file tracked) leave the archive tracked
+too. This is what makes the aging step safe to ship to gitignore-mode projects,
+where todo.org is private: the archive inherits that privacy instead of leaking
+previously-ignored task history into a tracked path."
+ (when (and tc-current-file tc-current-dir)
+ (let* ((todo (expand-file-name tc-current-file tc-current-dir))
+ (default-directory tc-current-dir)
+ (root (with-temp-buffer
+ (when (eq 0 (ignore-errors
+ (call-process "git" nil (current-buffer) nil
+ "rev-parse" "--show-toplevel")))
+ (string-trim (buffer-string))))))
+ (when (and root (> (length root) 0) (file-directory-p root)
+ (tc--git-ignored-p todo)
+ (not (tc--git-ignored-p archive-path)))
+ (let ((entry (concat "/" (file-relative-name
+ (expand-file-name archive-path) root)))
+ (gi (expand-file-name ".gitignore" root)))
+ (with-temp-buffer
+ (when (file-readable-p gi) (insert-file-contents gi))
+ (unless (save-excursion
+ (goto-char (point-min))
+ (re-search-forward (concat "^" (regexp-quote entry) "$") nil t))
+ (goto-char (point-max))
+ (unless (bolp) (insert "\n"))
+ (insert "\n# Claude Code: task archive (follows todo file privacy)\n"
+ entry "\n")
+ (write-region (point-min) (point-max) gi nil 'silent))))))))
+
+(defun tc--append-subtrees-to-archive-file (path texts)
+ "Append TEXTS (subtree strings) under the level-1 heading in PATH, creating the
+file with `tc-archive-file-scaffold' and the parent directory when absent.
+Ensures the archive inherits the todo file's gitignore status first."
+ (when (and path texts)
+ (tc--ensure-archive-gitignored path)
+ (let ((dir (file-name-directory path)))
+ (when (and dir (not (file-directory-p dir)))
+ (make-directory dir t)))
+ (with-temp-buffer
+ (when (file-readable-p path)
+ (insert-file-contents path))
+ (when (= (point-min) (point-max))
+ (insert tc-archive-file-scaffold))
+ ;; Guarantee a level-1 heading to append under (older files might lack one).
+ (goto-char (point-min))
+ (unless (re-search-forward "^\\* " nil t)
+ (goto-char (point-max))
+ (unless (bolp) (insert "\n"))
+ (insert "* Resolved (archived)\n"))
+ (goto-char (point-max))
+ (unless (bolp) (insert "\n"))
+ (dolist (text texts)
+ (insert text)
+ (unless (bolp) (insert "\n")))
+ (write-region (point-min) (point-max) path nil 'silent))))
+
+(defun tc-archive-old-resolved-to-file ()
+ "Move level-2 DONE/CANCELLED subtrees in the \"Resolved\" section whose CLOSED
+date predates the `tc-archive-retain-days' window out to `tc--archive-file-path'.
+Only subtrees closed within the window stay; older ones, and those with no
+parseable CLOSED date, are moved out. A nil `tc-archive-retain-days' disables the
+step. Honors `tc-check-only' (report only)."
+ (when tc-archive-retain-days
+ (let ((res (tc--find-section "resolved")))
+ (when (integerp res)
+ (let* ((cutoff (- (tc--reference-absolute) tc-archive-retain-days))
+ (moves nil))
+ (dolist (pos (tc--done-level-2-children res))
+ (save-excursion
+ (goto-char pos)
+ (let* ((region (tc--subtree-region))
+ (beg (car region))
+ (end (cdr region))
+ (closed (tc--closed-absolute-in-region beg end)))
+ ;; Archive anything not provably within the window: closed
+ ;; before the cutoff, or with no parseable CLOSED date at all.
+ (when (or (null closed) (< closed cutoff))
+ (push (list :beg beg :end end
+ :heading (org-get-heading t t t t)
+ :line (line-number-at-pos beg))
+ moves)))))
+ (setq moves (nreverse moves)) ; document order
+ (cond
+ ((null moves) nil)
+ (tc-check-only
+ (dolist (m moves)
+ (cl-incf tc-archived-to-file)
+ (push (list :kind 'archive-file-would :file tc-current-file
+ :line (plist-get m :line) :heading (plist-get m :heading))
+ tc-issues)))
+ (t
+ ;; Capture text before any deletion (positions are still valid), then
+ ;; delete bottom-up so earlier subtree positions stay correct.
+ (let ((texts (mapcar
+ (lambda (m)
+ (concat (string-trim-right
+ (buffer-substring-no-properties
+ (plist-get m :beg) (plist-get m :end))
+ "[ \t\n]+")
+ "\n"))
+ moves)))
+ (dolist (m (sort (copy-sequence moves)
+ (lambda (a b) (> (plist-get a :beg) (plist-get b :beg)))))
+ (delete-region (plist-get m :beg) (plist-get m :end)))
+ (tc--append-subtrees-to-archive-file (tc--archive-file-path) texts)
+ (dolist (m moves)
+ (cl-incf tc-archived-to-file)
+ (push (list :kind 'archive-file-moved :file tc-current-file
+ :line (plist-get m :line) :heading (plist-get m :heading))
+ tc-issues))))))))))
+
+;;; ---------------------------------------------------------------------------
+;;; --seal mode: rename the working archive to a dated resolved-YYYY-MM-DD.org
+
+(defun tc--seal-date-string ()
+ "YYYY-MM-DD for the seal — `tc-archive-reference-date' when set (tests),
+otherwise the real current date."
+ (if tc-archive-reference-date
+ (pcase-let ((`(,y ,m ,d) tc-archive-reference-date))
+ (format "%04d-%02d-%02d" y m d))
+ (format-time-string "%Y-%m-%d")))
+
+(defun tc-seal-archive-file ()
+ "Rename the working archive file to `resolved-YYYY-MM-DD.org' beside it.
+The next `--archive-done' run recreates a fresh working file. No-op (reported)
+when there is no working archive to seal; refuses to clobber an existing
+`resolved-<today>.org'. Ensures the sealed file inherits the todo file's
+gitignore status. Honors `tc-check-only'."
+ (let ((path (tc--archive-file-path)))
+ (cond
+ ((or (null path) (not (file-readable-p path)))
+ (push (list :kind 'seal-nothing :file tc-current-file) tc-issues))
+ (t
+ (let* ((dir (file-name-directory path))
+ (sealed (expand-file-name
+ (format "resolved-%s.org" (tc--seal-date-string)) dir)))
+ (cond
+ ((file-exists-p sealed)
+ (push (list :kind 'seal-collision :file tc-current-file
+ :detail (file-name-nondirectory sealed))
+ tc-issues))
+ (tc-check-only
+ (cl-incf tc-sealed)
+ (push (list :kind 'seal-would :file tc-current-file
+ :detail (file-name-nondirectory sealed))
+ tc-issues))
+ (t
+ ;; Ignore the sealed name before the rename so its history stays as
+ ;; private as the working archive it derives from.
+ (tc--ensure-archive-gitignored sealed)
+ (rename-file path sealed)
+ (cl-incf tc-sealed)
+ (push (list :kind 'seal-done :file tc-current-file
+ :detail (file-name-nondirectory sealed))
+ tc-issues))))))))
;;; ---------------------------------------------------------------------------
;;; --sync-child-priority mode
@@ -377,10 +689,186 @@ before their descendants — a [#A] → [#B] → [#D] chain collapses in one pas
(org-map-entries #'tc-sync-child-priority-at-heading nil 'file))
;;; ---------------------------------------------------------------------------
+;;; --convert-subtasks mode
+;;
+;; A sub-task (a heading at level 3 or deeper, i.e. under a parent task) that is
+;; marked DONE/CANCELLED/FAILED should become a dated event-log entry per the
+;; todo-format depth rule: drop the keyword, priority cookie, and tags, and
+;; rewrite the heading to `<stars> YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <text>' so the
+;; parent's subtree grows a chronological history instead of a long tail of
+;; nested DONE lines. Nothing enforced this before: `org-log-done' just flips an
+;; interactive close to DONE + CLOSED, and `--archive-done' only touches level 2.
+;; So level-3+ closes piled up as DONE keywords. This mode converts them
+;; mechanically, pulling the timestamp from each entry's own CLOSED cookie. The
+;; heading text is kept verbatim (a batch tool can't reliably past-tense an
+;; imperative title, and guessing prose in the task file is worse than leaving it
+;; as written). Idempotent: an already-dated heading has no done keyword, so it
+;; is skipped. A done sub-task with no parseable CLOSED cookie can't be dated, so
+;; it is flagged and left alone rather than stamped with a fabricated date.
+;;
+;; The planning line goes entirely. A dated-log entry carries its date in the
+;; heading, so CLOSED is redundant and an active DEADLINE/SCHEDULED is wrong: org
+;; renders any headline with an active planning timestamp — keyword or not — so a
+;; SCHEDULED left on a dated-log heading pins it to the agenda as weeks-overdue
+;; long after the work is done. The conversion deletes the whole planning line,
+;; not just the CLOSED cookie (todo-format.md; lint checker
+;; `dated-log-heading-active-timestamp' backstops any that slip through).
+
+(defun tc--closed-parts-in-entry ()
+ "Return a plist (:year :month :day :dow :hour :minute) from the CLOSED cookie
+of the entry at point, or nil when the entry has no parseable CLOSED line.
+:hour and :minute are nil when the cookie carries only a date. The CLOSED line
+sits in canonical position directly under the heading, so the first match within
+the entry is the task's own close."
+ (save-excursion
+ (org-back-to-heading t)
+ (let ((end (save-excursion
+ (or (outline-next-heading) (goto-char (point-max)))
+ (point))))
+ (when (re-search-forward
+ (concat "CLOSED:[ \t]*\\[\\([0-9]\\{4\\}\\)-\\([0-9]\\{2\\}\\)-\\([0-9]\\{2\\}\\)"
+ "[ \t]+\\([A-Za-z]+\\)"
+ "\\(?:[ \t]+\\([0-9]\\{2\\}\\):\\([0-9]\\{2\\}\\)\\)?\\]")
+ end t)
+ (list :year (match-string 1) :month (match-string 2) :day (match-string 3)
+ :dow (match-string 4)
+ :hour (match-string 5) :minute (match-string 6))))))
+
+(defun tc--tz-offset-string (year month day hour minute)
+ "Return the local UTC offset (e.g. \"-0500\") for the given wall-clock instant.
+DST-aware: `encode-time' with an unknown-DST field lets the system pick the
+correct offset for that date, so a summer close reads -0400 and a winter one
+-0500 without hardcoding either."
+ (format-time-string
+ "%z" (encode-time (list 0 minute hour day month year nil -1 nil))))
+
+(defun tc--dated-header-line (level parts title)
+ "Build the dated event-log heading string from LEVEL, CLOSED PARTS, and TITLE.
+Missing time in PARTS defaults to 00:00:00 (the close logged only a date)."
+ (let* ((year (plist-get parts :year))
+ (month (plist-get parts :month))
+ (day (plist-get parts :day))
+ (dow (plist-get parts :dow))
+ (hh (or (plist-get parts :hour) "00"))
+ (mm (or (plist-get parts :minute) "00"))
+ (tz (tc--tz-offset-string (string-to-number year)
+ (string-to-number month)
+ (string-to-number day)
+ (string-to-number hh)
+ (string-to-number mm))))
+ (format "%s %s-%s-%s %s @ %s:%s:00 %s %s"
+ (make-string level ?*) year month day dow hh mm tz title)))
+
+(defun tc--convert-collect-targets ()
+ "Markers at every heading at level >= 3 whose TODO state is a done state.
+Collected up front so the rewrite loop can edit the buffer without disturbing an
+in-progress `org-map-entries' walk; markers track their headings across edits."
+ (let (targets)
+ (org-map-entries
+ (lambda ()
+ (when (and (>= (org-current-level) 3)
+ (member (org-get-todo-state) tc--convert-done-states))
+ (push (copy-marker (point)) targets)))
+ nil 'file)
+ (nreverse targets)))
+
+(defun tc--strip-planning-lines-in-entry ()
+ "Delete the canonical planning line(s) directly under the heading at point.
+A planning line is one composed solely of CLOSED/DEADLINE/SCHEDULED cookies and
+whitespace. Walks the lines immediately after the heading and stops at the first
+non-planning line, so a planning-shaped line deeper in the body (e.g. in a code
+block) is never touched. Returns the count of lines removed."
+ (save-excursion
+ (org-back-to-heading t)
+ (forward-line 1)
+ (let ((removed 0) (continue t))
+ (while (and continue (not (eobp)))
+ (let ((line (buffer-substring-no-properties
+ (line-beginning-position) (line-end-position))))
+ (if (string-match-p tc--planning-line-regexp line)
+ (progn
+ (delete-region (line-beginning-position)
+ (min (1+ (line-end-position)) (point-max)))
+ (cl-incf removed))
+ (setq continue nil))))
+ removed)))
+
+(defun tc--convert-one-subtask (marker)
+ "Convert the done sub-task heading at MARKER to a dated event-log entry.
+Under `tc-check-only' the conversion is reported but not performed."
+ (goto-char marker)
+ (org-back-to-heading t)
+ (let* ((level (org-current-level))
+ (title (org-get-heading t t t t))
+ (line (line-number-at-pos))
+ (parts (tc--closed-parts-in-entry)))
+ (cond
+ ((null parts)
+ (push (list :kind 'convert-skip :file tc-current-file
+ :line line :heading title
+ :detail "no CLOSED date to derive the timestamp")
+ tc-issues))
+ (t
+ (let ((new (tc--dated-header-line level parts title)))
+ (cl-incf tc-converted)
+ (if tc-check-only
+ (push (list :kind 'convert-would :file tc-current-file
+ :line line :heading title :new new)
+ tc-issues)
+ ;; Replace the heading line, then drop the whole planning line. The
+ ;; date now lives in the header, so CLOSED is redundant and an active
+ ;; DEADLINE/SCHEDULED would wrongly pin this completed entry to the
+ ;; agenda (todo-format.md). Both go, not just the CLOSED cookie.
+ (delete-region (line-beginning-position) (line-end-position))
+ (insert new)
+ (tc--strip-planning-lines-in-entry)
+ (push (list :kind 'convert-done :file tc-current-file
+ :line line :heading title :new new)
+ tc-issues)))))))
+
+(defun tc-convert-subtasks-in-file ()
+ "Rewrite every level-3-and-deeper DONE/CANCELLED/FAILED heading to a dated
+event-log entry, pulling the timestamp from its CLOSED cookie. Honors
+`tc-check-only'."
+ (let ((targets (tc--convert-collect-targets)))
+ (dolist (m targets)
+ (tc--convert-one-subtask m)
+ (set-marker m nil))))
+
+;;; ---------------------------------------------------------------------------
;;; Driver + reporting
+(defun tc--backup (file)
+ "Copy FILE to /tmp before any modification. Skipped in --check mode.
+
+Matches `lint-org.el' and `wrap-org-table.el', the other tools that rewrite
+these org files. todo-cleanup runs the most often of the three (every wrap,
+every sentry cycle), and Emacs's own backup does not fire under --batch -q, so
+without this a mechanical rewrite has no undo short of git — which recovers
+only to the last commit and loses intra-session work."
+ (let* ((base (format "%s%s.before-todo-cleanup.%s"
+ temporary-file-directory
+ (file-name-nondirectory file)
+ (format-time-string "%Y%m%d-%H%M%S")))
+ (backup base)
+ (n 2))
+ ;; Never overwrite an earlier backup. A second-resolution stamp collides
+ ;; when two invocations run back to back, which the shipped workflow does
+ ;; (open-tasks.org runs --convert-subtasks then --archive-done, each a
+ ;; sub-second batch run). Overwriting there replaces the true pre-session
+ ;; original with already-mutated content — losing exactly what the backup
+ ;; exists to preserve. Suffix instead, so every invocation keeps its own.
+ (while (file-exists-p backup)
+ (setq backup (format "%s-%d" base n))
+ (setq n (1+ n)))
+ (copy-file file backup nil)
+ backup))
+
(defun tc-process-file (file)
(setq tc-current-file (file-name-nondirectory file))
+ (setq tc-current-dir (file-name-directory (expand-file-name file)))
+ (unless tc-check-only
+ (tc--backup file))
(with-current-buffer (find-file-noselect file)
(org-mode)
(cond
@@ -388,6 +876,10 @@ before their descendants — a [#A] → [#B] → [#D] chain collapses in one pas
(tc-archive-done-in-file))
(tc-sync-child-priority
(tc-sync-child-priority-in-file))
+ (tc-convert-subtasks
+ (tc-convert-subtasks-in-file))
+ (tc-seal
+ (tc-seal-archive-file))
(t
;; Pass 1: auto-fix bogus state logs (or report under --check).
(org-map-entries #'tc-fix-bogus-state-log-in-entry nil 'file)
@@ -420,6 +912,21 @@ before their descendants — a [#A] → [#B] → [#D] chain collapses in one pas
(plist-get i :file)
(plist-get i :line)
(if tc-check-only "would move" "moved")
+ (plist-get i :heading)))))))
+ ;; Aged-out subtrees: only reported when some moved (or would). Additive to
+ ;; the in-file report above, and absent when the aging step is disabled.
+ (when (> tc-archived-to-file 0)
+ (princ (format "todo-cleanup --archive-done: %d aged subtree(s) %s task-archive.org%s\n"
+ tc-archived-to-file
+ (if tc-check-only "would move to" "moved to")
+ (if tc-check-only " — CHECK MODE (no writes)" "")))
+ (dolist (i (reverse tc-issues))
+ (pcase (plist-get i :kind)
+ ((or 'archive-file-moved 'archive-file-would)
+ (princ (format " %s:%d: %s %s\n"
+ (plist-get i :file)
+ (plist-get i :line)
+ (if tc-check-only "would archive" "archived")
(plist-get i :heading)))))))))
(defun tc--emit-hygiene-report ()
@@ -467,9 +974,50 @@ before their descendants — a [#A] → [#B] → [#D] chain collapses in one pas
(plist-get i :child-heading)
(plist-get i :parent-heading)))))))
+(defun tc--emit-convert-report ()
+ ;; Silent on a real-mode no-op (nothing to convert and nothing skipped), for
+ ;; the same reason as the archive report: the wrap runs cleanup passes more
+ ;; than once, and a vocal \"0 converted\" reads as noise. Check mode always
+ ;; reports (the preview is what the caller asked for), and a skip always
+ ;; reports (a done sub-task with no CLOSED date is a real condition to see).
+ (let ((has-skip (cl-some (lambda (i) (eq (plist-get i :kind) 'convert-skip))
+ tc-issues)))
+ (when (or tc-check-only (> tc-converted 0) has-skip)
+ (princ (format "todo-cleanup --convert-subtasks: %d sub-task(s) %s%s\n"
+ tc-converted
+ (if tc-check-only "would convert" "converted")
+ (if tc-check-only " — CHECK MODE (no writes)" "")))
+ (dolist (i (reverse tc-issues))
+ (pcase (plist-get i :kind)
+ ((or 'convert-done 'convert-would)
+ (princ (format " %s:%d: %s\n → %s\n"
+ (plist-get i :file) (plist-get i :line)
+ (plist-get i :heading) (plist-get i :new))))
+ ('convert-skip
+ (princ (format " skipped %s:%d: %s — %s\n"
+ (plist-get i :file) (plist-get i :line)
+ (plist-get i :heading) (plist-get i :detail)))))))))
+
+(defun tc--emit-seal-report ()
+ (dolist (i (reverse tc-issues))
+ (pcase (plist-get i :kind)
+ ('seal-done
+ (princ (format "todo-cleanup --seal: sealed task-archive.org → %s\n"
+ (plist-get i :detail))))
+ ('seal-would
+ (princ (format "todo-cleanup --seal: would seal task-archive.org → %s — CHECK MODE (no writes)\n"
+ (plist-get i :detail))))
+ ('seal-collision
+ (princ (format "todo-cleanup --seal: %s already exists — not sealing (already sealed today?)\n"
+ (plist-get i :detail))))
+ ('seal-nothing
+ (princ "todo-cleanup --seal: no working archive to seal\n")))))
+
(defun tc-emit-report ()
(cond (tc-archive-done (tc--emit-archive-report))
(tc-sync-child-priority (tc--emit-sync-report))
+ (tc-convert-subtasks (tc--emit-convert-report))
+ (tc-seal (tc--emit-seal-report))
(t (tc--emit-hygiene-report))))
(defun tc-main ()
@@ -484,6 +1032,12 @@ before their descendants — a [#A] → [#B] → [#D] chain collapses in one pas
(when (member "--sync-child-priority" command-line-args-left)
(setq tc-sync-child-priority t)
(setq command-line-args-left (delete "--sync-child-priority" command-line-args-left)))
+ (when (member "--convert-subtasks" command-line-args-left)
+ (setq tc-convert-subtasks t)
+ (setq command-line-args-left (delete "--convert-subtasks" command-line-args-left)))
+ (when (member "--seal" command-line-args-left)
+ (setq tc-seal t)
+ (setq command-line-args-left (delete "--seal" command-line-args-left)))
;; --check-child-priority is the report-only alias for
;; `--sync-child-priority --check'.
(when (member "--check-child-priority" command-line-args-left)
@@ -491,7 +1045,7 @@ before their descendants — a [#A] → [#B] → [#D] chain collapses in one pas
(setq command-line-args-left (delete "--check-child-priority" command-line-args-left)))
(if (null command-line-args-left)
(progn
- (princ "Usage: emacs --batch -q -l todo-cleanup.el [--check] [--archive-done | --sync-child-priority | --check-child-priority] FILE...\n")
+ (princ "Usage: emacs --batch -q -l todo-cleanup.el [--check] [--archive-done | --seal | --convert-subtasks | --sync-child-priority | --check-child-priority] FILE...\n")
(kill-emacs 1))
(let ((files command-line-args-left))
(setq command-line-args-left nil)
@@ -510,6 +1064,8 @@ ert-run-tests-batch-and-exit'."
(cl-every (lambda (a)
(cond ((member a '("--check"
"--archive-done"
+ "--seal"
+ "--convert-subtasks"
"--sync-child-priority"
"--check-child-priority"))
t)
diff --git a/claude-templates/.ai/scripts/upcoming_birthdays.py b/claude-templates/.ai/scripts/upcoming_birthdays.py
new file mode 100755
index 0000000..d3f30c0
--- /dev/null
+++ b/claude-templates/.ai/scripts/upcoming_birthdays.py
@@ -0,0 +1,182 @@
+#!/usr/bin/env python3
+"""Upcoming-birthdays block for daily prep.
+
+Reads an org-contacts file for ``:BIRTHDAY: YYYY-MM-DD`` properties, finds the
+ones whose next occurrence falls within a window (default 30 days) from today,
+and prints a daily-prep block: name, date, days-away, and — when the birth year
+is real — the age the person is turning. Many contacts use ``1900`` as a
+placeholder year when the real one is unknown; those render date-only, no age.
+Anything within the callout window (default 7 days) is flagged so a gift or
+plan gets prompted.
+
+Pure core:
+ parse_birthdays(text) -> [Birthday(name, month, day, year|None), ...]
+ upcoming(birthdays, today, window=30) -> [Upcoming(name, date, days_away, age|None), ...]
+ format_block(items, window, callout_days=7) -> str
+
+CLI:
+ upcoming_birthdays.py [--file PATH] [--today YYYY-MM-DD] [--window N] [--callout N]
+prints the block on stdout. Default --file is ~/sync/org/contacts.org.
+"""
+
+import argparse
+import datetime as dt
+import re
+import sys
+from dataclasses import dataclass
+from pathlib import Path
+
+PLACEHOLDER_YEAR = 1900
+DEFAULT_CONTACTS = Path.home() / "sync" / "org" / "contacts.org"
+
+_HEADING_RE = re.compile(r"^\*+\s+(.*?)\s*$")
+_TAGS_RE = re.compile(r"\s+:[A-Za-z0-9_@#%:]+:$")
+_BIRTHDAY_RE = re.compile(r"^\s*:BIRTHDAY:\s*(\d{4})-(\d{2})-(\d{2})\s*$")
+
+
+@dataclass(frozen=True)
+class Birthday:
+ name: str
+ month: int
+ day: int
+ year: int | None # None when the source used the 1900 placeholder
+
+
+@dataclass(frozen=True)
+class Upcoming:
+ name: str
+ date: dt.date
+ days_away: int
+ age: int | None # None when the birth year is unknown
+
+
+def _clean_name(heading_text: str) -> str:
+ """Strip a trailing org tag cluster from a heading's text."""
+ return _TAGS_RE.sub("", heading_text).strip()
+
+
+def parse_birthdays(text: str) -> list[Birthday]:
+ """Extract (name, month, day, year|None) for every contact with a BIRTHDAY.
+
+ The name is the nearest preceding org heading. A birth year of 1900 is the
+ placeholder org-contacts uses for an unknown year and is returned as None.
+ Malformed birthday lines are ignored.
+ """
+ birthdays: list[Birthday] = []
+ current_name: str | None = None
+ for line in text.splitlines():
+ heading = _HEADING_RE.match(line)
+ if heading:
+ current_name = _clean_name(heading.group(1))
+ continue
+ bday = _BIRTHDAY_RE.match(line)
+ if bday and current_name:
+ year, month, day = (int(g) for g in bday.groups())
+ # Guard against a nonsense month/day that regex width still admits.
+ try:
+ dt.date(2000, month, day)
+ except ValueError:
+ continue
+ birthdays.append(
+ Birthday(
+ current_name,
+ month,
+ day,
+ None if year == PLACEHOLDER_YEAR else year,
+ )
+ )
+ return birthdays
+
+
+def _next_occurrence(month: int, day: int, today: dt.date) -> dt.date:
+ """First date on/after ``today`` landing on this month/day.
+
+ A Feb 29 birthday maps to Feb 28 in a non-leap year.
+ """
+ def on(year: int) -> dt.date:
+ try:
+ return dt.date(year, month, day)
+ except ValueError:
+ # Only Feb 29 can fail here; fall back to Feb 28.
+ return dt.date(year, 2, 28)
+
+ candidate = on(today.year)
+ if candidate < today:
+ candidate = on(today.year + 1)
+ return candidate
+
+
+def upcoming(
+ birthdays: list[Birthday], today: dt.date, window: int = 30
+) -> list[Upcoming]:
+ """Birthdays whose next occurrence is within ``window`` days, soonest first."""
+ items: list[Upcoming] = []
+ for b in birthdays:
+ occ = _next_occurrence(b.month, b.day, today)
+ days = (occ - today).days
+ if 0 <= days <= window:
+ age = None if b.year is None else occ.year - b.year
+ items.append(Upcoming(b.name, occ, days, age))
+ items.sort(key=lambda u: (u.days_away, u.name))
+ return items
+
+
+def _days_phrase(days: int) -> str:
+ if days == 0:
+ return "today"
+ if days == 1:
+ return "tomorrow"
+ return f"in {days} days"
+
+
+def format_block(items: list[Upcoming], window: int, callout_days: int = 7) -> str:
+ """Render the daily-prep block. Callout entries (within ``callout_days``) are
+ flagged with a marker and a plan-a-gift nudge."""
+ if not items:
+ return f"No birthdays in the next {window} days."
+
+ lines = [f"Upcoming birthdays (next {window} days):"]
+ for u in items:
+ callout = u.days_away <= callout_days
+ marker = "⚠" if callout else "·"
+ date_str = u.date.strftime("%a %b %d")
+ piece = f" {marker} {date_str} — {u.name}"
+ if u.age is not None:
+ piece += f" turns {u.age}"
+ piece += f" ({_days_phrase(u.days_away)})"
+ if callout:
+ piece += " — plan a gift/card"
+ lines.append(piece)
+ return "\n".join(lines)
+
+
+def _parse_today(value: str | None) -> dt.date:
+ if value is None:
+ return dt.date.today()
+ return dt.date.fromisoformat(value)
+
+
+def main(argv: list[str] | None = None) -> int:
+ parser = argparse.ArgumentParser(description="Print the upcoming-birthdays daily-prep block.")
+ parser.add_argument("--file", type=Path, default=DEFAULT_CONTACTS,
+ help="org-contacts file (default: ~/sync/org/contacts.org)")
+ parser.add_argument("--today", default=None,
+ help="override today's date (YYYY-MM-DD), for testing")
+ parser.add_argument("--window", type=int, default=30,
+ help="look-ahead window in days (default: 30)")
+ parser.add_argument("--callout", type=int, default=7,
+ help="flag birthdays within this many days (default: 7)")
+ args = parser.parse_args(argv)
+
+ if not args.file.exists():
+ print(f"contacts file not found: {args.file}", file=sys.stderr)
+ return 1
+
+ text = args.file.read_text(encoding="utf-8")
+ items = upcoming(parse_birthdays(text), _parse_today(args.today), args.window)
+ print(format_block(items, args.window, args.callout))
+ return 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/claude-templates/.ai/scripts/wrap-org-table.el b/claude-templates/.ai/scripts/wrap-org-table.el
index ddbea65..173e44d 100644
--- a/claude-templates/.ai/scripts/wrap-org-table.el
+++ b/claude-templates/.ai/scripts/wrap-org-table.el
@@ -228,22 +228,40 @@ continuation lines merge back into their logical row before re-wrapping."
;;; file layer
(defun wot-process-file (file &optional budget)
- "Reformat every org table in FILE in place to BUDGET width."
+ "Reformat every org table in FILE in place to BUDGET width.
+Pipe-led lines inside #+begin_/#+end_ blocks (example, src, quote, …) are
+content, not tables — ASCII art in an example block once got mangled into a
+bordered table — so block regions are skipped verbatim."
(with-temp-buffer
(insert-file-contents file)
(goto-char (point-min))
- (while (re-search-forward "^[ \t]*|" nil t)
- (let ((start (line-beginning-position)))
- (while (and (not (eobp))
- (save-excursion (beginning-of-line)
- (looking-at "[ \t]*|")))
+ (let ((in-block nil)) ; the open block's type, e.g. "example" — nil outside
+ (while (not (eobp))
+ (cond
+ ;; Only the matching #+end_<type> closes a block: an example block
+ ;; often quotes literal #+begin_src/#+end_src lines, and a boolean
+ ;; flag would let that inner literal end-marker re-expose the rest
+ ;; of the block to reformatting.
+ ((and (not in-block)
+ (looking-at "^[ \t]*#\\+begin_\\([^ \t\n]+\\)"))
+ (setq in-block (downcase (match-string 1)))
(forward-line 1))
- (let* ((end (point))
- (table (buffer-substring-no-properties start end))
- (reformatted (wot-reformat-table-string table budget)))
- (delete-region start end)
- (goto-char start)
- (insert reformatted))))
+ ((and in-block
+ (looking-at-p (format "^[ \t]*#\\+end_%s\\([ \t]\\|$\\)"
+ (regexp-quote in-block))))
+ (setq in-block nil)
+ (forward-line 1))
+ ((and (not in-block) (looking-at-p "^[ \t]*|"))
+ (let ((start (point)))
+ (while (and (not (eobp)) (looking-at-p "^[ \t]*|"))
+ (forward-line 1))
+ (let* ((end (point))
+ (table (buffer-substring-no-properties start end))
+ (reformatted (wot-reformat-table-string table budget)))
+ (delete-region start end)
+ (goto-char start)
+ (insert reformatted))))
+ (t (forward-line 1)))))
(write-region (point-min) (point-max) file)))
;;; ---------------------------------------------------------------------------
@@ -289,7 +307,21 @@ so the ERT suite can `require' this file without firing the CLI dispatch."
(t (file-readable-p a))))
command-line-args-left)))
-(when (and noninteractive (wot--cli-invocation-p))
+(defun wot--entry-script-p ()
+ "Non-nil when wrap-org-table.el itself was named on the command line.
+lint-org.el `require's this file, and a load-triggered dispatch would run
+the table reformatter over lint-org's file arguments — that's how a lint
+invocation once reformatted the files it was only supposed to report on.
+Only dispatch when a -l/--load argument names this very file."
+ (and load-file-name
+ (cl-loop for (flag arg) on command-line-args
+ thereis (and (member flag '("-l" "--load"))
+ (stringp arg)
+ (file-exists-p arg)
+ (string= (file-truename (expand-file-name arg))
+ (file-truename load-file-name))))))
+
+(when (and noninteractive (wot--entry-script-p) (wot--cli-invocation-p))
(wot-main))
(provide 'wrap-org-table)
diff --git a/claude-templates/.ai/workflows/INDEX.org b/claude-templates/.ai/workflows/INDEX.org
index 17963ef..f18d953 100644
--- a/claude-templates/.ai/workflows/INDEX.org
+++ b/claude-templates/.ai/workflows/INDEX.org
@@ -1,5 +1,5 @@
#+TITLE: Workflow Index
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-04-25
* Purpose
@@ -15,10 +15,15 @@ This index must list every =.org= file in =.ai/workflows/= except this one and e
** Session lifecycle
- =startup.org= — runs automatically at session start. No manual trigger.
+- =helper-mode.org= — role contract for a helper instance (a second Claude in the same project as a live primary). No manual trigger; the spawn paths route to it, "you are a helper" is the manual fallback.
- =first-session.org= — initialize =.ai/= for a brand-new project.
- Triggers: "this is a new project", "let's set this project up". Auto-runs if =.ai/sessions/= is empty.
-- =wrap-it-up.org= — end-of-session: write summary, archive, commit, push.
+- =wrap-it-up.org= — end-of-session: write summary, archive, commit, push, then a phrase-dependent Step 6 teardown. Bare "wrap it up" tears the session down (kills the ai-term buffer + =aiv-<project>= tmux session via a =Stop=-hook sentinel, after the valediction flushes); a "with summary" / "and summarize" wrap keeps the buffer; "and shutdown" gates on being the only live ai-term session, then powers the machine off via an abort-able Emacs countdown.
- Triggers: "wrap it up", "that's a wrap", "let's call it a wrap"
+ - No-teardown triggers: "wrap it up with summary", "wrap it up and summarize"
+ - Shutdown trigger: "wrap it up and shutdown"
+- =suspend.org= — capture-only mid-session pause for an abrupt departure: append a resume-weighted =SUSPENDED= entry to the Session Log, note uncommitted work, and LEAVE =.ai/session-context.org= in place so the next startup resumes from it. The capture-only counterpart to =wrap-it-up= (which archives + tears down) and to =flush= (=/flush=, which prompts =/clear= and resumes the same session). Provides only the capture half; startup's interrupted-session path is the resume half.
+ - Triggers: "suspend the session", "suspend", "I need to go", "stick a pin in everything"
- =retrospective.org= — post-mortem after a tough session.
- Triggers: "let's do a retrospective", "retrospective time"
@@ -43,12 +48,19 @@ This index must list every =.org= file in =.ai/workflows/= except this one and e
- Triggers: "let's do a journal entry", "create a journal entry"
- =clean-todo.org= — tidy =todo.org=: hygiene pass + =--archive-done=, then summarize. Wrap-up does this automatically; this is the manual entry point.
- Triggers: "clean up todo.org", "clean-todo", "tidy the todo file", "archive the done items in todo.org", "run the todo cleanup"
-- =process-inbox.org= — evaluate each inbox item against a three-question value gate (advances an existing TODO / improves the project / serves the mission), then implement, fold, file, defer, or reject per source (Craig / project handoff / script). Auto-invoked by startup when inbox is non-empty. Source-aware rejection flow: handoff rejections write a response back via =inbox-send= naming the failed gate question and any reconsideration condition.
- - Triggers: "process inbox", "process the inbox", "handle the inbox", "what's in inbox", "what's in the inbox", "let's clear the inbox", "let's process the inbox items"
-- =monitor-inbox.org= — the cadence + act-vs-file + reply layer over process-inbox: check =inbox-status= at every task boundary, decide act-now (just do it) vs file (ask, file = option 1), and confirm back to handoff senders. Includes the opt-in background-monitor =/loop= recipe.
- - Triggers: "monitor the inbox", "watch the inbox", "respond to the handoffs", "handle the handoffs"
-- =inbox-zero.org= — route the *global roam inbox* (=~/org/roam/inbox.org=) to owning projects by =<project>:= heading prefix. Distinct from =process-inbox.org= (the project's own =inbox/= dir). The current session claims only its own prefixed items, files them into =todo.org=, removes them from the shared inbox, and leaves foreign/unowned items. Every scan reports the total item count plus how many appear related to this project. v1 is single-destination (prefix-claim only); domain-aware whole-inbox routing is deferred. Called read-only from startup (count + offer) and as a wrap-up Step 3 sub-step.
- - Triggers: "inbox zero", "empty the inbox", "process the roam inbox", "triage my roam inbox"
+- =inbox.org= — one engine for the project's inbox surfaces, with the shared value gate / skeptical review / disposition ladder / reply discipline / capture-guard / priority-scheme check in one place, plus thin per-surface modes. *Process mode* evaluates each project-local =inbox/= item against the three-question value gate, then implements / folds / files / defers / rejects per source (auto-invoked by startup when inbox is non-empty). *Monitor mode* runs process mode now then loops it every 15 min, gating on a clean tree + green suite and adding the act-vs-file + no-approvals-execute + reply discipline. *Roam mode* routes the global roam inbox (=~/org/roam/inbox.org=) to owning projects by =<project>:= prefix (read-only nudge at startup, sweep at wrap-up). *Auto inbox zero* runs roam mode on an interactive =/loop= at a Craig-chosen interval. Distinct from =triage-intake.org= (external accounts), which stays separate.
+ - Process-mode triggers: "process inbox", "process the inbox", "handle the inbox", "what's in inbox", "what's in the inbox", "let's clear the inbox", "let's process the inbox items"
+ - Monitor-mode triggers: "monitor the inbox", "watch the inbox", "respond to the handoffs", "handle the handoffs"
+ - Roam-mode triggers: "inbox zero", "empty the inbox", "process the roam inbox", "triage my roam inbox"
+ - Auto-mode trigger: "auto inbox zero" (match before "inbox zero")
+
+- =work-the-backlog.org= — the autonomous task-execution loop, the single home for working a batch of marked tasks unattended: takes an ordered task set (explicit list or tag query) + session mode (=file-only= default / =autonomous-commit= + paging) + a hard run cap; each candidate passes the mechanical eligibility gate (status =TODO= + =:solo:= per the project's scheme header) and the four-item defer checklist, then is implemented to the full quality bar (TDD, =/review-code=, =/voice=) as its own logical commits. Fed by the inbox auto-loop's chain step (yes-gated, file-only, cap 1) and the no-approvals speedrun preset (pre-flight Q&A → autonomous-commit + always-push + end-of-set page over an explicit ordered list).
+ - Speedrun triggers: "speedrun", "no approvals speedrun", "speedrun these: <task set>" — any phrase containing "speedrun" routes here (the preset), never to =no-approvals.org=
+ - Manual triggers: "work the backlog", "work the backlog with <task set>" (file-only defaults)
+ - Synthesis trigger: "synthesize backlog metrics" — read the per-project metrics logs, compute trends + the corrections signal, write one =:agent:metrics:= KB node (personal projects only)
+- =sentry.org= — the overnight hygiene supervisor: an interval loop (default hourly) that walks a fixed pass list (roam pull, inbox zero, triage, todo cleanup, task audit, working-files hygiene, spec board, link integrity, git health, prep freshness), commits each writing pass to a throwaway =sentry/<date>-<host>= branch (never pushed), and parks every judgment call and destructive action in a morning-approval queue. Gated on =:COMMIT_AUTONOMY: yes= plus interactive entry gates (clean tree, green suite) with Craig present. Locks via =agent-lock=; morning teardown (review, squash-merge, delete) is Craig's, never automated.
+ - Triggers: "start sentry", "run sentry", "arm sentry", "sentry mode", "start sentry every <interval>"
+ - Stop trigger: "stop sentry", "stand down sentry", "sentry off"
** Calendar
@@ -80,11 +92,18 @@ This index must list every =.org= file in =.ai/workflows/= except this one and e
- =spec-create.org= — author a design/feature spec before non-trivial work (more than ~6 hours, or real trade-offs): a when-to-spec gate, then problem-first framing, design + alternatives + inline mini-ADR decisions, implementation phases + acceptance criteria + a readiness-dimensions menu, a terseness pass, and a hand-off self-check against the review rubric. The *author* side that starts the trio; its output feeds =spec-review.org=.
- Triggers: "let's write a spec", "spec this out", "create a spec for X", "spec-create workflow"
-- =spec-review.org= — review a design/feature spec for implementation-readiness: run the readiness gate, read the code first, evaluate across dimensions, assign a rubric (Ready / Ready-with-caveats / Not-ready / Needs-research), and write a =<spec>-review.org= file when not ready. The *reviewer* side; its output feeds =spec-response.org=.
+- =spec-review.org= — review a design/feature spec for implementation-readiness: run the readiness gate, read the code first, evaluate across dimensions, assign a rubric (Ready / Ready-with-caveats / Not-ready / Needs-research), and record findings as =TODO= tasks in the spec's =* Review findings= section when not ready. The *reviewer* side; its output feeds =spec-response.org=.
- Triggers: "review the spec", "is this spec implementation-ready?", "spec-review workflow", "review the design"
-- =spec-response.org= — fold an external spec review back in: decide accept / modify / reject for every recommendation, weave accepts into the spec body, document modifies and rejects in a "Review dispositions" section, reconcile cross-spec tensions, iterate to implementation-ready. The *author* side; consumes the =<spec>-review.org= file =spec-review.org= produces.
+- =spec-response.org= — fold a spec review back in: decide accept / modify / reject for every finding, weave accepts into the spec body, complete each finding task in place (the reason recorded on modifies and rejects), reconcile cross-spec tensions, iterate to implementation-ready. The *author* side; consumes the =* Review findings= =spec-review.org= produces.
- Triggers: "respond to the review", "process the spec reviews", "spec-response workflow", "fold in the review"
+** Code quality
+
+- =code-quality.org= — one trigger that sequences every behavior-preserving quality pass over a scope of existing code: =/refactor= (complexity, duplication, dead-code, simplification) then =readability-audit= (comments, headers, names, organization), then surfaces the =:refactor:= tasks readability filed and any deferred =/refactor= findings. A thin orchestrator — each pass keeps its own gate. Excludes =/simplify= (that's for the current diff, not existing code).
+ - Triggers: "code quality sweep", "quality sweep", "run every quality pass on <scope>", "give me every pass on <scope>"
+- =readability-audit.org= — make code readable to a future maintainer: audit file-top commentary, inline comments (why-not-what), names (intention-revealing), and organization (co-location / stepdown / cohesion). The cheap comment- and name-only fixes (dimensions A/B/C) land inline, verified by a green suite; the structural findings (dimension D — split a module, rename a public symbol) are *filed* as =:refactor:= tasks, not done here. Language-agnostic. Feeds =/refactor= (which executes the filed structural work); distinct from =/refactor='s metric scans and =/simplify='s diff cleanup.
+ - Triggers: "let's run the readability-audit workflow", "audit the comments and commentary in <area>", "clean up the structure/organization of <module>", "readability audit"
+
** Tools and meta
- =process-meeting-transcript.org= — record → transcript → labeled archive.
@@ -94,8 +113,8 @@ This index must list every =.org= file in =.ai/workflows/= except this one and e
- Situational triggers: "broadcast the <event> to all projects", "broadcast that <situation>", "let every project know I'll be away ..."
- =flashcard-review.org= — review an org-drill flashcard file, restructure cards to question-form headings (no answer hints), audit content accuracy against project source-of-truth via subagent, rewrite source preserving SRS state, regenerate the Anki =.apkg= to =~/sync/phone/anki/=. Person cards use "Who is X? Tell me about their Y."; talking-points cards stay as-is. Script behavior: =flashcard-to-anki.py= strips =:PROPERTIES:= drawers + =SCHEDULED:= / =DEADLINE:= planning lines from Anki output.
- Triggers: "review the flashcards", "update the flashcards", "review the drill deck", "update the drill deck", "refresh the Anki cards", "let's run the flashcard-review workflow"
-- =page-me.org= — set a timed notification.
- - Triggers: anything containing the word "page" used as a verb ("page me", "page me in 10 minutes", "page me at 3pm")
+- =page-me.org= — set a timed notification. "page me" desktop =notify=, "text me" phone via =agent-text=, "text and page me" both.
+ - Triggers: anything containing the word "page" used as a verb ("page me", "page me in 10 minutes", "page me at 3pm", "page my phone")
- =status-check.org= — proactive long-running-job updates.
- Triggers: "keep me posted on this", "provide status checks on this job", "let me know when it's done", "monitor this for me". Auto: any job estimated 10+ min.
- =create-workflow.org= — define a new workflow.
@@ -106,7 +125,7 @@ This index must list every =.org= file in =.ai/workflows/= except this one and e
- Triggers: "session harvest", "harvest the sessions", "let's run the session-harvest workflow", "monthly harvest", "mine the sessions"
- =no-approvals.org= — drop the interaction-level approval gates for a pre-agreed batch while keeping engineering-discipline gates (=/review-code=, =/voice personal=, tests, session-log updates, subagent reviews, destructive-action consent). Mode stays on until Craig turns it off, a real question arises, the queue empties, or the conversation switches topics.
- Triggers: "no-approvals mode", "no approvals", "no-approval", "no need for approval gates", "stop asking, just keep going", "I'll check back in when you're done or stuck", "do all =<selector>= with no-approval"
-- =cross-agent-comms.org= — protocol for cross-project agent coordination via =inbox/from-agents/= (file-based IPC, GPG-signed, supports cross-machine over Tailscale). Auto: when =cross-agent-watch= detects a new inbound message, or when an agent decides to initiate a cross-project conversation. Operational scripts (=cross-agent-send=, =-recv=, =-watch=, =-status=, =-discover=, =-halt=, =-resume=) and their READMEs live at =.ai/scripts/cross-agent-comms/=.
+ - Exception: any phrase containing "speedrun" routes to =work-the-backlog.org='s no-approvals speedrun preset instead
* Living Document
diff --git a/claude-templates/.ai/workflows/add-calendar-event.org b/claude-templates/.ai/workflows/add-calendar-event.org
index 2650fb7..5dd6c42 100644
--- a/claude-templates/.ai/workflows/add-calendar-event.org
+++ b/claude-templates/.ai/workflows/add-calendar-event.org
@@ -1,5 +1,5 @@
#+TITLE: Add Calendar Event Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-01
* Overview
diff --git a/claude-templates/.ai/workflows/broadcast.org b/claude-templates/.ai/workflows/broadcast.org
index 1be07d2..cc14f00 100644
--- a/claude-templates/.ai/workflows/broadcast.org
+++ b/claude-templates/.ai/workflows/broadcast.org
@@ -1,5 +1,5 @@
#+TITLE: Broadcast Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-29
* Overview
@@ -159,11 +159,11 @@ broadcast, not a task and not tailored to this project.
- Ask Craig any follow-up questions then — this message is deliberately general.
#+end_example
-The "For the receiving agent" block is fixed text — it travels with every situational broadcast so the message is self-describing. A receiving project's =process-inbox= reads it and acts on those instructions without needing any special-casing; the value gate accepts it as situational awareness that improves how the project works.
+The "For the receiving agent" block is fixed text — it travels with every situational broadcast so the message is self-describing. A receiving project's =inbox.org= process mode reads it and acts on those instructions without needing any special-casing; the value gate accepts it as situational awareness that improves how the project works.
** Receiving behavior (what a project does with an incoming situational broadcast)
-When =process-inbox= encounters a =Broadcast:= item, the disposition is *record-and-hold*, not file-as-task:
+When =inbox.org= process mode encounters a =Broadcast:= item, the disposition is *record-and-hold*, not file-as-task:
1. Add a dated entry to =notes.org= Active Reminders capturing the situation and its end date (if any).
2. If the event bears on an open task, note the connection in that task's body.
diff --git a/claude-templates/.ai/workflows/clean-todo.org b/claude-templates/.ai/workflows/clean-todo.org
index dd33056..48d3084 100644
--- a/claude-templates/.ai/workflows/clean-todo.org
+++ b/claude-templates/.ai/workflows/clean-todo.org
@@ -1,5 +1,5 @@
#+TITLE: Clean-Todo Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-11
* Overview
@@ -27,7 +27,17 @@ Deletes bogus =- State "X" from "X" [date]= log lines (state didn't actually cha
To preview without writing, run =--check= first: =emacs --batch -q -l .ai/scripts/todo-cleanup.el --check todo.org=.
-** Step 2: Archive completed work
+** Step 2: Convert done sub-tasks to dated entries
+
+#+begin_src bash
+emacs --batch -q -l .ai/scripts/todo-cleanup.el --convert-subtasks todo.org
+#+end_src
+
+Rewrites every heading at level 3 or deeper whose TODO state is DONE/CANCELLED/FAILED into a dated event-log entry (=<stars> YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <text>=), dropping the keyword, priority cookie, and tags, and removing the =CLOSED:= line. Enforces the depth rule that a completed sub-task becomes dated history — a shape interactive org closes and =--archive-done= (level-2 only) leave unapplied. Timestamp comes from each entry's =CLOSED= cookie; heading text kept verbatim; idempotent; a done sub-task with no parseable =CLOSED= is flagged and left alone. Run before archiving so a parent's sub-tasks are already dated when it moves. Capture the output.
+
+To preview without writing: =emacs --batch -q -l .ai/scripts/todo-cleanup.el --convert-subtasks --check todo.org=.
+
+** Step 3: Archive completed work
#+begin_src bash
emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done todo.org
@@ -37,10 +47,11 @@ Moves every level-2 subtree whose TODO state is DONE or CANCELLED out of the "Op
To preview the moves without writing: =emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done --check todo.org=.
-** Step 3: Summarize
+** Step 4: Summarize
-Report to Craig from the two captured outputs:
+Report to Craig from the three captured outputs:
- Hygiene: how many bogus state-log lines were deleted; any orphan-planning warnings (file:line + heading), or "none".
+- Convert: how many done sub-tasks were rewritten to dated entries (heading + line), any flagged for no =CLOSED= date, or "nothing to convert".
- Archive: how many subtrees moved and which (heading + line), or "nothing to move" / the skip reason if a section was missing or ambiguous.
- If the file changed, note that =todo.org= now has an uncommitted edit — review =git diff -- todo.org= and commit it (in this repo's commit style) if it looks right. If nothing changed, say so and stop.
@@ -49,7 +60,7 @@ Don't auto-commit. The summary is the review point; Craig decides whether the di
* Principles
- *Both passes apply, not just preview.* The workflow is invoked because cleanup is wanted. Use the =--check= variants only when Craig asks for a dry run.
-- *Two passes, two invocations.* =--archive-done= is its own mode and does not run the hygiene pass; run both.
+- *Separate modes, separate invocations.* =--convert-subtasks=, =--archive-done=, and the hygiene pass are each their own mode and don't run the others; run all three.
- *Never auto-commit todo.org.* Surface the diff and let Craig commit it. The cleanup is a working-tree change, fully reversible until committed.
- *Trust the script.* It's fast and idempotent; if there's nothing to do, it reports zero and exits clean. No pre-checks.
diff --git a/claude-templates/.ai/workflows/code-quality.org b/claude-templates/.ai/workflows/code-quality.org
new file mode 100644
index 0000000..3c4ed8f
--- /dev/null
+++ b/claude-templates/.ai/workflows/code-quality.org
@@ -0,0 +1,90 @@
+#+TITLE: Code-Quality Sweep Workflow
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-06-28
+
+* Overview
+
+One trigger that runs every behavior-preserving quality pass over a scope of
+*existing* code, in order, then surfaces what got filed for later. It's a thin
+orchestrator — each pass keeps its own discipline and its own confirm gate; this
+workflow only sequences them and collects the residue.
+
+*Behavior-preserving rests on a test net.* The passes below claim to preserve
+behavior, but a refactor on untested code is a guess, not a preservation. Where
+the scope has no tests, bring it under a characterization net first
+(Normal/Boundary/Error per unit, per the =testing-standards= skill's "Adding Tests to Existing
+Untested Code") — that net is what turns "behavior-preserving" from an assertion
+into something the green suite actually verifies across each pass.
+
+The passes it chains:
+
+1. =/refactor= — structural and logic cleanup on measurable metrics (complexity,
+ duplication, dead-code) plus the simplification lens.
+2. =readability-audit= ([[file:readability-audit.org][readability-audit.org]]) — prose and human-reader clarity
+ (comments, file headers, names, organization).
+
+It deliberately does *not* run =/simplify=: that works the current uncommitted
+diff, not existing committed code, so it belongs to the moment you've just made a
+change, not to a sweep of code already in the tree (see "The /simplify boundary"
+below).
+
+* When to Use This Workflow
+
+- "code quality sweep" / "quality sweep"
+- "run every quality pass on <scope>" / "full quality pass on <scope>"
+- "give me every pass on <file/module/tree>"
+
+Do NOT use it for:
+- *In-flight diff cleanup* — that's =/simplify= on the change you just made.
+- *Bug hunting* — these passes are behavior-preserving; for defects use =debug=
+ or =/review-code=.
+- *Performing the structural refactors it files* — those become =:refactor:=
+ tasks; work them later via =/refactor rename= / =/refactor simplification= or
+ =/start-work=.
+
+* Steps
+
+** 1. Scope
+
+Pick the target: one file, a named module set, or the whole tree (honor
+=.aiignore=). The same scope is passed to both passes so they cover the same
+code.
+
+** 2. /refactor <scope>
+
+Run =/refactor= on the scope. Its default full scan covers complexity,
+duplication, dead-code, and simplification. It presents findings and applies
+only what's approved (its own gate) — structure and logic first, so the
+readability pass audits the cleaned-up code.
+
+** 3. readability-audit on <scope>
+
+Run the readability-audit workflow on the same scope. Its cheap comment- and
+name-only fixes (dimensions A/B/C) land inline and are verified by a green
+suite; its structural findings (dimension D — split a module, rename a public
+symbol) are *filed* as =:refactor:= tasks rather than done here.
+
+** 4. Surface the residue
+
+Collect and report what the sweep left behind for later work:
+
+- The =:refactor:= tasks readability-audit filed (the structural backlog).
+- Any =/refactor= findings deferred rather than applied in step 2.
+
+That residue is the "do this next" list the sweep produces; it's not a failure
+to finish, it's the structural work that needs its own design and test pass.
+
+* The /simplify boundary
+
+=/simplify= and this sweep don't overlap: =/simplify= cleans the *current diff*
+and applies its fixes directly, so reach for it right after making a change,
+before committing. This sweep works *existing committed code* and runs the
+scan-and-present passes. One trigger can't sensibly do both — a diff you're
+holding and a tree you're auditing are different inputs.
+
+* Verification
+
+Each pass owns its verification (=/refactor= runs the suite after applying;
+readability-audit verifies inline fixes against a green suite). The umbrella
+adds nothing beyond sequencing, so when both passes report green, the sweep is
+clean — confirm that before reporting done rather than assuming it.
diff --git a/claude-templates/.ai/workflows/create-workflow.org b/claude-templates/.ai/workflows/create-workflow.org
index 6060df1..393fce5 100644
--- a/claude-templates/.ai/workflows/create-workflow.org
+++ b/claude-templates/.ai/workflows/create-workflow.org
@@ -195,7 +195,7 @@ After the Q&A, ask together:
Decide on a name for this workflow.
*Naming convention:* Action-oriented (verb form)
-- Examples: "refactor", "inbox-zero", "create-workflow", "review-code"
+- Examples: "refactor", "clean-todo", "create-workflow", "review-code"
- Why: Shorter, natural when saying "let's do a [name] workflow"
- Filename: =.ai/workflows/[name].org=
@@ -240,10 +240,10 @@ Update =notes.org=:
Example entry:
#+begin_src org
-,** inbox-zero
-File: =.ai/workflows/inbox-zero.org=
+,** journal-entry
+File: =.ai/workflows/journal-entry.org=
-Workflow for processing inbox to zero:
+Workflow for capturing a daily journal entry:
1. [Brief workflow summary]
2. [Key steps]
diff --git a/claude-templates/.ai/workflows/cross-agent-comms.org b/claude-templates/.ai/workflows/cross-agent-comms.org
deleted file mode 100644
index 430b4b0..0000000
--- a/claude-templates/.ai/workflows/cross-agent-comms.org
+++ /dev/null
@@ -1,334 +0,0 @@
-#+TITLE: Cross-Agent Communication Workflow (v5)
-#+AUTHOR: Craig Jennings & Claude (homelab + career sessions)
-#+DATE: 2026-04-27
-#+VERSION: 5
-
-* Status
-
-Draft. Iterating between the homelab and career sessions through a multi-round design discussion. Awaiting Craig's review for promotion to =~/code/rulesets/claude-templates/.ai/workflows/=.
-
-v5 changes from v4:
-- *Script absorption.* Seven operational scripts (=cross-agent-send=, =cross-agent-recv=, =cross-agent-watch=, =cross-agent-status=, =cross-agent-discover=, =cross-agent-halt=, =cross-agent-resume=) now own most implementation detail. Their READMEs are the operational source of truth. The spec stays declarative.
-- *Failsafe halt.* Layered HALT-file mechanism stops all cross-agent activity on a machine within ~5 min, without visiting individual sessions or restarting Claude Code. =cross-agent-halt= and =cross-agent-resume= are the convenience entry points; every other component checks the HALT file independently.
-- *Identity.* Messages are GPG-signed by sender and verified by receiver. Combined with POSIX permissions on =from-agents/= and Tailscale-level network auth, identity becomes a three-layer story.
-- *Atomic writes.* Writers MUST use temp-file + rename. =cross-agent-send= handles this; the spec just states the contract.
-- *Dedup.* Sequence-collision dedup is now binary SHA-256 equality, not a fuzzy ">90% match" threshold.
-- *Cold-start handling.* Layered: =cross-agent-watch= (push notifications via =inotifywait=) is the primary mechanism; startup-workflow check and user-direct-injection are coverage layers.
-- *Spec stays roughly the same length but does more protocol work.* Operational detail (rsync retry numbers, inotifywait recipes, peers.toml schema, GPG flags, dedup mechanics) moved to the script READMEs. The spec adds new protocol elements (identity layer, atomic-writes contract, SHA-256 dedup, =escalate= type, =RELEASE_STATUS= values, =REQUIRES_TOOLS= optional field) in the freed space. Total documentation surface (spec + seven READMEs ≈ 1000 lines) is larger than v4's 259 lines, but the spec and the READMEs serve different audiences — protocol-thinkers and CLI-users — and a reader of just the spec can comprehend the protocol without consulting any README.
-
-* When to use
-
-When two Claude sessions in different projects (same machine or different machines on the same Tailscale tailnet) need to coordinate on a shared task that one session can't complete alone — typically because one has tooling, context, or MCP access the other doesn't.
-
-Examples that fit:
-- Session A asks session B to apply a workflow patch in B's project, then verify it.
-- Session A runs a long task and needs session B to monitor results in B's domain.
-- Two sessions co-design a workflow.
-
-Examples that don't fit:
-- A simple file handoff that doesn't require iteration.
-- A task one session can do alone.
-- Cross-tailnet or cross-organization. The protocol is local-tailnet-scoped.
-
-* Protocol
-
-** File location
-
-Each project has =inbox/from-agents/= as its agent-comms mailbox. Create the directory if it doesn't exist; set permissions =chmod 700= and ownership to the user.
-
-- Sender writes to receiver's =inbox/from-agents/=.
-- Receiver polls (or watches) =inbox/from-agents/=, *not* the parent =inbox/=.
-- The parent =inbox/= stays reserved for human-triage items.
-- Out-of-band artifacts (PDFs, datasets) live at =inbox/from-agents/artifacts/=. Reference by relative path in the message body.
-
-The user does NOT write directly to =from-agents/=. To inject input into a running conversation, the user tells one of the agents in that agent's session; the agent writes the input as a normal message attributed to the user.
-
-** File naming
-
-=YYYYMMDDTHHMMSSZ-from-<sender>-<short-conv-id>.org=
-
-- Timestamp is UTC ISO 8601 compact. The trailing =Z= is mandatory.
-- =from-<sender>= prefix.
-- =<short-conv-id>= is a stable kebab-case slug across the back-and-forth. Reusable across time; ordering relies on filename timestamps.
-
-Frontmatter =#+TIMESTAMP= carries the same instant in local time with explicit offset. The two MUST refer to the same instant.
-
-The implementation (=cross-agent-send=) generates the canonical filename from the message's frontmatter (=CONVERSATION_ID=, current UTC time) and the sender's project context. Senders supply only the message body file; the script handles naming. Senders MUST NOT pre-name files in this format and pass them through; the script overwrites with its own canonical name to ensure consistency and enable the sender-side max-seen sequence-collision-reduction scan.
-
-GPG signatures live in a sibling file =YYYYMMDDTHHMMSSZ-from-<sender>-<short-conv-id>.org.asc=. Receivers verify before processing. See =* Writes are atomic= for the two-file delivery ordering rule.
-
-** Frontmatter
-
-Required:
-
-#+begin_example
-#+TITLE: <human-readable subject>
-#+CONVERSATION_ID: <stable across the thread>
-#+MESSAGE_TYPE: <see types below>
-#+SEQUENCE: <integer hint>
-#+TIMESTAMP: <ISO 8601 with explicit offset>
-#+PROTOCOL_VERSION: 5
-#+end_example
-
-Optional:
-
-#+begin_example
-#+REQUIRES_TOOLS: <comma-separated tool/MCP slugs, e.g. gmail-mcp, slack-mcp>
-#+RELEASE_STATUS: <see release-statuses; valid only on MESSAGE_TYPE: release>
-#+WORKFLOW_VERSION: <sender's version of cross-agent-comms.org; informational only in v5 — no enforcement>
-#+end_example
-
-Receiver sanity-checks frontmatter before acting. Missing or malformed frontmatter → surface to user, don't proceed. Mismatched =PROTOCOL_VERSION= → receiver writes a =query= asking the originator to upgrade.
-
-** Identity
-
-Messages are GPG-signed by the sender. Receivers verify the detached signature before processing the message body.
-
-The implementation (=cross-agent-send=) signs automatically with the sender's configured key (the user's primary GPG key by default; configurable via =--key= flag or environment). Receivers verify automatically against the keys in their GPG keyring.
-
-Identity is a three-layer story:
-
-1. *Tailscale layer.* Only tailnet members can reach the rsync-over-SSH endpoint at all.
-2. *POSIX layer.* =chmod 700= on =from-agents/= means only processes running as the directory's owner can write.
-3. *GPG layer.* Sender's signature on each message proves the message originated from a process holding the key.
-
-Three independent layers. Per-user GPG (using existing keys) gives a correctness check more than a security boundary — unsigned messages are almost certainly bugs, not attackers. That's still load-bearing.
-
-** Writes are atomic
-
-Writers MUST use a temp-file + rename pattern (=mktemp= + =mv= within the same filesystem) so receivers never see partial files. The implementation script (=cross-agent-send=) handles this.
-
-Receivers ignore =.tmp.*= files, processing only the final renamed name.
-
-*Two-file ordering.* When a message has a sibling GPG signature file (=.org.asc=), the writer MUST rename the =.asc= to its final name *before* renaming the =.org=. Two =mv= operations are not atomic together — without this ordering, a receiver could read the =.org= in the window between the two renames and fail GPG verify because the =.asc= hasn't landed yet. The rule: receiver only acts on =.org= files, and a =.org= without a corresponding =.asc= means the signature is genuinely missing (not still in flight).
-
-** Sequence numbering
-
-=#+SEQUENCE= is a *hint*, not a strict counter. Canonical order is =#+TIMESTAMP=. Sequences may collide under rapid back-and-forth (both sides write what they think is sequence N near-simultaneously). Treat collision as a normal protocol event.
-
-*Receiver-side dedup rule.* When a new file shares =CONVERSATION_ID= + =SEQUENCE= with an already-processed message, compare SHA-256 hashes. Identical hashes → silent dedup, treat as a retry. Different hashes → process both, ordered by =#+TIMESTAMP=.
-
-*Sender-side collision-reduction (best-effort).* Before picking sequence, scan the receiver's =from-agents/= for the highest existing sequence in this conversation across both sender prefixes. Use =max(seen) + 1=.
-
-** Message types
-
-- *request* — a side asks for work, input, or a decision. Sequence 1 is always =request=.
-- *progress* — work-in-progress checkpoint. "Here's where I am, no action needed from you, more coming." Originator's poll loop should NOT page the user on progress messages.
-- *query* — either side asks a clarifying question that blocks further work. Originator's poll loop SHOULD surface this immediately. Originator answers and work continues.
-- *pushback* — receiver formally disagrees with the request and has *not* started the work. Carries reasoning. Distinct from =query= because the originator's response path differs.
-- *complete* — receiver signals the requested work is done. Triggers verification.
-- *release* — terminal type. Originator writes after verifying =complete=. Carries =RELEASE_STATUS= to disambiguate the closure mode.
-- *escalate* — punts the conversation to the user for adjudication. Both sides pause polling on =escalate=; the user resolves.
-
-Reply expectation is implied by type: =request=, =query=, =pushback=, =escalate= expect a reply; =progress=, =complete=, =release= don't.
-
-** Conversation lifecycle
-
-A conversation is a directed loop between an originator (issued sequence 1) and a receiver:
-
-1. Originator writes =request= (sequence 1). Begins polling for replies.
-2. *Optional acknowledgment.* Receiver may write a =progress= at sequence 2 to acknowledge receipt and set expectations. Required if work will take >5 minutes (so the originator's poll loop doesn't waste wakes).
-3. *Optional echo-back.* For ambiguous or large requests, receiver writes a =progress= that restates work items and announces "starting now unless you push back within N minutes."
-4. Receiver works. May write =progress= updates. =query= mid-work if blocked. =pushback= if the request is wrong.
-5. Receiver writes =complete=. Begins polling for =release=.
-6. Originator reads, *verifies the deliverable directly*. For subjective deliverables, verification is the originator's editorial accept.
-7. If verified: =release= with =RELEASE_STATUS: complete=. If problems: new =request= (next sequence number).
-8. Receiver sees =release=, stops polling.
-
-The verification step is load-bearing. =complete= is a *claim*; =release= is *verification*.
-
-** Pushback path
-
-On receiving a =pushback=, the originator chooses:
-
-1. *Revise* — new =request= with adjusted scope.
-2. *Insist* — new =request= addressing the pushback's reasoning, standing by direction.
-3. *Withdraw* — =release= with =RELEASE_STATUS: withdrawn-after-pushback=.
-
-*Deadlock cap.* After two pushback-insist exchanges, the next message MUST be =MESSAGE_TYPE: escalate=. Both agents pause polling; the user resolves.
-
-** =RELEASE_STATUS= values
-
-| Status | Meaning |
-|---+---|
-| =complete= | Goal achieved, originator verified |
-| =cancelled= | Originator changed their mind mid-conversation |
-| =withdrawn-after-pushback= | Originator chose option 3 on receiver's =pushback= |
-| =abandoned-after-escalation= | User adjudicated and chose to close the conversation |
-| =abandoned-after-timeout= | Receiver auto-closed after originator never returned to verify |
-
-** Async fallback
-
-If the originator session ends between =request= and =complete=, the receiver's =complete= goes unverified. Receiver behavior:
-
-- Polls for =release= up to ~24 hours of cycles (implementation default).
-- After timeout, writes a final =progress= message ("treating as terminal-without-verification; originator never returned to release") and stops polling. Receiver does NOT write =release= itself — that would contradict the lifecycle rule that =release= is the originator's terminal action.
-- Next time the originator project starts, the unreleased =complete= is surfaced as a startup item. The user can issue a late =release= (with whichever =RELEASE_STATUS= fits) or open a fresh conversation to revisit. =RELEASE_STATUS: abandoned-after-timeout= is used at that point if the user wants to formally close the orphaned thread.
-
-** Escalation
-
-A side writes =escalate= when:
-- Pushback-insist deadlock cap reached.
-- Conversation has stalled (no productive movement in N exchanges).
-- A reply-expecting message has gone unanswered past timeout.
-
-Body summarizes both sides' positions in 60 seconds of reading. Both agents pause polling; the user resolves.
-
-* Implementation notes
-
-This sub-section describes how to operate the protocol. Operational detail lives in the seven scripts' READMEs.
-
-** Recommended scripts
-
-| Script | Replaces user action | README |
-|---+---+---|
-| =cross-agent-send <dest> <msg>= | Filename generation, GPG sign, atomic write, peer lookup, rsync push, retry+backoff, failure surfacing — seven mechanical sender-side steps. Frontmatter and message body are still author-supplied. | =cross-agent-send.md= |
-| =cross-agent-recv <msg>= | Frontmatter sanity-check, =PROTOCOL_VERSION= verify, GPG verify, SHA-256 dedup, =REQUIRES_TOOLS= check — five mechanical receiver-side steps. Output is a structured decision (=process= / =dedup= / =query= / =reject=) the agent acts on. | =cross-agent-recv.md= |
-| =cross-agent-watch= | Manually checking inboxes; "did I get a message?" | =cross-agent-watch.md= |
-| =cross-agent-status= | Walking each project to count pending messages | =cross-agent-status.md= |
-| =cross-agent-discover= | Remembering project topology and reachability | =cross-agent-discover.md= |
-| =cross-agent-halt [reason] [--tailnet]= | Visiting each session to stop polling, restarting Claude Code, or hand-killing processes when comms go runaway. =--tailnet= propagates HALT to all peers. | =cross-agent-halt.md= |
-| =cross-agent-resume [--tailnet]= | Manually clearing the HALT state and restarting the watcher. Per-session polling does NOT auto-resume — the user re-engages each session explicitly. | =cross-agent-resume.md= |
-
-The scripts are tools the user runs from any terminal. They do not depend on agent context — =cross-agent-status= run from a fresh shell works.
-
-A reader can comprehend this protocol from this spec alone. Script READMEs add operational detail that makes the protocol practical to use, but understanding the protocol's semantics requires only this document.
-
-** Polling
-
-Default cadence: 270 seconds (≈4.5 min). Sits just under the 5-minute prompt-cache TTL.
-
-If a side needs to slow down (heads-down work, idle wait), it writes a =progress= message saying so in prose. The other side adapts. There are no named polling modes.
-
-After ~12 empty polls in a row, the poll loop surfaces the silence to the user.
-
-A future runtime with native filesystem-event support could replace polling for active sessions; =cross-agent-watch= already provides event-driven notifications outside active sessions.
-
-** User multi-tasking
-
-- *Deferral.* If the user's last message in the agent's session was less than 60 seconds ago AND a poll fires, queue the inbox check until either the user sends another message OR 5 minutes pass without further input.
-- *Surfacing.* On the next user-facing response: "While we were working on X, a cross-agent message landed from <project>. It's a =<type>= — want me to handle it now or after we finish?"
-- *Mid-question.* Answer the user first.
-- *Project switch.* If the user moves to the receiver project mid-conversation, the receiver agent surfaces the in-flight thread on first user prompt.
-- *Conversation state.* Always include in any response that mentions a cross-agent thread: "<conv-id> at sequence N, awaiting <event>."
-
-** Failure modes
-
-The seven scripts surface most failures with concrete error messages. Spec-level failure modes:
-
-- *Malformed frontmatter on a received file.* Surface to user; do not act.
-- *Mismatched =PROTOCOL_VERSION=.* Receiver writes =query= asking originator to upgrade.
-- *Missing or invalid GPG signature.* Receiver surfaces "unsigned/unverified message"; refuses to act.
-- *Sequence collision* with non-matching SHA-256. Process both, ordered by timestamp.
-- *Required tool unavailable.* Receiver checks =REQUIRES_TOOLS= during frontmatter-sanity-check (before any work begins). On a missing tool, receiver writes =query= asking the originator to reframe the request to avoid the unavailable tool. Originator may revise (new =request=) or withdraw (=release= with =RELEASE_STATUS: cancelled=). =query= is the right type rather than =pushback= because missing-tool is a capability gap, not disagreement.
-- *Runaway resource usage.* User invokes =cross-agent-halt= globally (or =cross-agent-halt --tailnet= for cross-machine). HALT file stops all components within one polling cycle (~5 min). See =* Halt mechanism= for the layered checks.
-- *User halts mid-conversation.* Both sides write a final =progress= note ("HALT fired; pausing"); polling stops within one cadence; conversations resume on explicit per-session re-engage after HALT clears.
-- *HALT file accidentally created* (typo, errant =touch=). =cross-agent-status= prominently flags HALT active; user clears with =cross-agent-resume=. Cost: no messages send during the typo window.
-- *HALT file unreadable* (perms wrong, partial write). Each component fails-closed (treats as halted) and reports "HALT file present but unreadable; treat as halted." Safer than fail-open.
-
-Operational failures (rsync push fails, watcher dies, peer unreachable) live in the script READMEs' failure-mode tables.
-
-* Halt mechanism
-
-A failsafe to stop all cross-agent activity on a machine without visiting individual sessions or restarting Claude Code. Designed for the runaway-polling case: an agent has spun up conversations with N other agents, polling is eating CPU, and the user needs to stop everything *now*.
-
-** The HALT file
-
-Path: =~/.config/cross-agent-comms/HALT=.
-
-Existence triggers halt across all components on the machine. The file's body may carry an optional human-readable reason (reviewed by the user later when deciding to resume).
-
-User commands:
-
-#+begin_example
-$ touch ~/.config/cross-agent-comms/HALT # halt
-$ rm ~/.config/cross-agent-comms/HALT # resume
-#+end_example
-
-Or via convenience scripts (=cross-agent-halt= / =cross-agent-resume=) that also handle the watcher service and cross-machine propagation.
-
-** Layered checks (the failsafe property)
-
-Every component MUST check the HALT file. The "any one component stops the system independently" property is what makes this failsafe — the system doesn't depend on a single point doing the right thing.
-
-| Component | Check timing | Behavior on HALT |
-|---+---+---|
-| =cross-agent-send= | At start of send + between =.asc= and =.org= rsync + between retry iterations | Refuse to start new send; complete current step then exit. Worst case: one in-flight send finishes within a few seconds. |
-| =cross-agent-recv= | Before any verify or dedup | Leave inbound message in place — do NOT dedup, reject, or move. Resume picks it up via cold-start handling. |
-| =cross-agent-watch= | At iteration start | Suppress notifications; log only. Continues running, no-op until HALT clears. |
-| =cross-agent-status= | At start | Print prominent "⚠ HALT ACTIVE" banner before normal output. Read-only, continues. |
-| =cross-agent-discover= | At start | Print HALT banner; continue read-only enumeration. |
-| Agent polling loop | First action on every wake | Write a final =progress= note to any active conversation ("HALT fired; pausing"), do NOT reschedule, surface "halt active" to user. Polling decays within one cadence (~5 min). |
-| Agent user-facing responses | Every response while HALT is set | Append "(HALT active; cross-agent comms paused)" to the response. On HALT clear, the next response says "(HALT cleared; cross-agent comms ready to resume — say so to re-engage polling)." Persistent, not just first-response — keeps awareness alive. |
-| Conversation initiator | Before writing sequence 1 of any new conversation | Refuse and surface to user. |
-| Startup workflow | Phase A on session start | If HALT exists, surface immediately and skip cross-agent inbox checks. |
-
-The agent polling-loop check is the load-bearing one for "stops eating CPU." Wake-ups already scheduled fire, but each wake on-HALT is a no-op + reschedule-prevention. Within one polling cadence (~5 min) all polling stops.
-
-*Fail-closed on unreadable HALT.* If the HALT file exists but is unreadable (wrong permissions, partial write), components MUST treat as halted. Safer than fail-open.
-
-** Resume asymmetry (deliberate)
-
-Halt is automatic everywhere. Resume requires explicit user intent per-session.
-
-When the user removes HALT (or runs =cross-agent-resume=), components stop refusing to act, but agent polling does NOT auto-resume. The user must open each session and tell that agent to resume polling for its conversations.
-
-The asymmetry exists because:
-
-1. Auto-resume could silently invert intentional kills. If the user halted because a session was misbehaving, removing HALT shouldn't quietly revive it.
-2. Per-session resume forces the user to look at each session and confirm the situation is resolved before re-engaging.
-
-** Cross-machine halt
-
-=cross-agent-halt --tailnet= iterates =peers.toml= and SSH-touches HALT on each peer. Same shape for resume.
-
-Reports per-peer status with non-zero exit on partial halt:
-
-#+begin_example
-$ cross-agent-halt --tailnet
-Halting velox.local ✓ (HALT file written)
-Halting bastion.local ✗ (ssh exit 255: no route to host)
-Halting locally ✓ (HALT file written)
-
-PARTIAL HALT: 2/3 machines halted. bastion.local needs manual halt.
-Exit 1.
-#+end_example
-
-Scripting can detect partial halt via the exit code. Same pattern for =--tailnet= on resume.
-
-* Limitations
-
-- *Local-tailnet only.* Filesystem IPC + rsync over SSH. Cross-tailnet or cross-organization is out of scope.
-- *Identity has three layers (Tailscale + POSIX + GPG)* but no message-content encryption. Confidentiality is not the goal; signing is correctness, not secrecy.
-- *Single-receiver per conversation.* Fan-out to multiple receivers requires manually orchestrating multiple parallel conversations.
-- *Polling is best-effort.* A wake may be delayed by an in-flight tool call until the runtime is idle. =cross-agent-watch= mitigates by offering event-driven notifications.
-- *Project-extension drift.* If two projects' =.ai/project-workflows/= modify shared workflow definitions in incompatible ways, cross-agent assumptions can diverge silently. The optional =#+WORKFLOW_VERSION= advisory field is informational only in v5 — no implementation reads or acts on it. A future version may add enforcement on mismatch (e.g. receiver writes =query= asking which side is stale). Today, alignment is verified manually before high-stakes conversations.
-
-* Persistence after release
-
-Conversation files persist by default. The conversation log is the audit trail.
-
-Manual archival is fine if the inbox grows unmanageable. Suggested cadence: once the conversation has been =release='d AND the work it produced has shipped, archive both projects' message files into =.ai/sessions/cross-agent/= as a flat directory — no per-conversation subdirectories. Rename each archived file to lead with the conversation-id so messages from the same conversation cluster on =ls=: =<conv-id>-<TIMESTAMP>-from-<sender>.org= (and the matching =.asc= sibling, if present). Inbox filenames lead with the timestamp because chronological arrival is what matters in =from-agents/=; archives invert that because grouping by conversation is what matters when reading history. Keep the =.asc= signatures alongside the =.org= files in archive — they're small and document the GPG verification chain.
-
-Old messages don't affect protocol behavior (=cross-agent-status='s pending semantics correctly ignore released messages) but the =from-agents/= directory grows indefinitely without manual archival. =cross-agent-status= performance degrades noticeably when a project's =from-agents/= exceeds a few hundred files. =cross-agent-init= (deferred to v6) would include an archival sub-command.
-
-* Open questions
-
-- *=cross-agent-init= and =cross-agent-compose= helper scripts.* =-init= would be one-command project bootstrap (creates =inbox/from-agents/= with =chmod 700=, installs the =cross-agent-watch= systemd path unit, validates peer config, runs a discovery probe). =-compose= would be interactive frontmatter authoring (prompts for required fields, produces a draft message file). Both deferred to v6. Current onboarding requires manual =mkdir= + systemd setup per =cross-agent-watch.md='s install recipe; current message authoring requires writing the file by hand or via a small in-agent template.
-- *Hard conversation timeout.* The async-fallback timeout is implementation-default ~24 hours. Right number depends on use case; tighten as patterns emerge.
-- *=paused= polling state.* Today there's no clean signal for "pause without ending." Add when first user complaint surfaces.
-- *Multi-LLM context.* If we ever bring in a non-Claude agent, the protocol's natural-language framing may need formalization.
-
-* Examples
-
-** =prep-fixup= conversation (2026-04-26 → 2026-04-27)
-
-Eleven exchanges between homelab and career produced the v4 spec by iterative critique-and-simplification. Three real-time sequence collisions during the conversation drove the sequence-as-hint rule that landed in v4 and persists in v5.
-
-Files at =~/projects/{homelab,career}/inbox/from-agents/= named =*-prep-fixup.org=. Worth re-reading when designing future cross-agent flows.
-
-** =comms-cold-start-discovery= conversation (2026-04-27)
-
-The follow-up that produced this v5 spec. Cold-start, watcher tooling, agent discovery, GPG identity, sha256 dedup, atomic writes, POSIX perms, script absorption, and process-vs-text simplification. Tonight's first cold-start in real time (career session went dormant after =prep-fixup= release; Craig's user-injection re-engaged it) is the worked demonstration of the v5 user-injection rule.
-
-Files at =~/projects/{homelab,career}/inbox/from-agents/= named =*-comms-cold-start-discovery.org=.
diff --git a/claude-templates/.ai/workflows/daily-prep.org b/claude-templates/.ai/workflows/daily-prep.org
index b6989e7..3f21214 100644
--- a/claude-templates/.ai/workflows/daily-prep.org
+++ b/claude-templates/.ai/workflows/daily-prep.org
@@ -1,5 +1,5 @@
#+TITLE: Daily Prep Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-06-11
* Overview
@@ -74,7 +74,7 @@ The separate =* Standup Briefs= and =* Upcoming Deadlines= sections are *retired
Items that frame the day. Four standing items are *always* present, plus the look-ahead:
1. *Meeting-density framing.* One line on the day's shape, e.g. "Meeting-dense morning: 09:00 team discussion, 10:00 standup, 11:00 general standup. Real focus time only opens at noon."
-2. *Calendar events from BOTH calendars* (work + personal): birthdays, holidays, anniversaries, vacations, trips, big events. "Your trip begins Friday." "It's <person>'s birthday today."
+2. *Calendar events from BOTH calendars* (work + personal): birthdays, holidays, anniversaries, vacations, trips, big events. "Your trip begins Friday." "It's <person>'s birthday today." Fold in the =upcoming_birthdays.py= block from Phase A (source 8) for contact birthdays the calendar doesn't carry; keep its callout entries (within 7 days) so a gift or plan gets prompted.
3. *Reminders due, imminently due, or past trigger* — from notes.org Active Reminders plus scheduled/deadline tasks. A reminder tied to a rescheduled meeting reports against the new date.
4. *Requested metrics* — a slot for any metric Craig has asked to track in the daily prep. Render the slot only when a metric is active; none are active by default. (Metric design is a separate discussion — don't invent metrics.)
5. *5-Day Look-Ahead* — one day per line, format =Fri 12:= / =Mon 15:=, including clear days marked =clear=. Built by Phase 1's forward scan with the invite quick-read and decline gate applied.
@@ -192,6 +192,7 @@ Pull every source in a *single batch of parallel tool calls*:
5. Pull the project tracker's view of Craig's plate (assigned issues, items in review, blocked items) where the project has one.
6. List + read the *previous* prep doc. Glob =daily-prep/*-daily-prep.org=, sort by date, take the file *before* the one the root =daily-prep.org= symlink resolves to. If the symlink doesn't resolve yet, take the most recent file. The standup lookback anchors on this file's date.
7. Read the most recent =.ai/sessions/= summary (for the standup brief's lookback).
+8. Run =.ai/scripts/upcoming_birthdays.py= (reads =~/sync/org/contacts.org=). Its block feeds the Heads-Up birthday line — contacts carry birthdays the calendar doesn't, and anything inside 7 days comes back flagged so a gift/plan gets prompted.
This fetch *is* the live calendar read for build time. In Update mode, re-run the calendar fetches — never reuse the build-time snapshot.
@@ -239,7 +240,7 @@ Assemble priorities from both sources; no mid-flow confirmation (the gate handle
*** Sub-step 3b: Triage external sources (delegate to triage-intake.org)
-Don't scan email / Slack / tracker / PRs inline. Run the =triage-intake.org= engine (if the mode's freshness check already ran one this hour, use its synthesis). It classifies everything new against the four-bucket model, writes every Action item into =todo.org= as its own =:quick:reactive:= task, and executes the routine actions on confirmation. Source coverage comes from its Phase 0 plugin load — both =.ai/workflows/triage-intake.*.org= (general) and =.ai/project-workflows/triage-intake.*.org= (project-specific). A missing source is a missing plugin, not a daily-prep regression.
+Don't scan email / Slack / tracker / PRs inline. Run the =triage-intake.org= engine (if the mode's freshness check already ran one this hour, use its synthesis). It surfaces the three-section digest (==TASKS== / ==FYI== / ==MISC==) and then closes by default per its Phase D — filing every TASKS item into =todo.org= as its own =:quick:reactive:= task and running the routine mail hygiene without itemized confirmation. Source coverage comes from its Phase 0 plugin load — both =.ai/workflows/triage-intake.*.org= (general) and =.ai/project-workflows/triage-intake.*.org= (project-specific). A missing source is a missing plugin, not a daily-prep regression.
*** Sub-step 3c: Build the entries
diff --git a/claude-templates/.ai/workflows/delete-calendar-event.org b/claude-templates/.ai/workflows/delete-calendar-event.org
index 5bb92a1..7de0086 100644
--- a/claude-templates/.ai/workflows/delete-calendar-event.org
+++ b/claude-templates/.ai/workflows/delete-calendar-event.org
@@ -1,5 +1,5 @@
#+TITLE: Delete Calendar Event Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-01
* Overview
diff --git a/claude-templates/.ai/workflows/edit-calendar-event.org b/claude-templates/.ai/workflows/edit-calendar-event.org
index 662f0b4..27a9dd3 100644
--- a/claude-templates/.ai/workflows/edit-calendar-event.org
+++ b/claude-templates/.ai/workflows/edit-calendar-event.org
@@ -1,5 +1,5 @@
#+TITLE: Edit Calendar Event Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-01
* Overview
diff --git a/claude-templates/.ai/workflows/email-assembly.org b/claude-templates/.ai/workflows/email-assembly.org
index 003459c..699dbc0 100644
--- a/claude-templates/.ai/workflows/email-assembly.org
+++ b/claude-templates/.ai/workflows/email-assembly.org
@@ -1,5 +1,5 @@
#+TITLE: Email Assembly Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-01-29
* Overview
diff --git a/claude-templates/.ai/workflows/extract-email.org b/claude-templates/.ai/workflows/extract-email.org
index 3a70bea..c68bafe 100644
--- a/claude-templates/.ai/workflows/extract-email.org
+++ b/claude-templates/.ai/workflows/extract-email.org
@@ -1,5 +1,5 @@
#+TITLE: Extract Email Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-06
* Overview
diff --git a/claude-templates/.ai/workflows/find-email.org b/claude-templates/.ai/workflows/find-email.org
index 0ef9615..d71ed3e 100644
--- a/claude-templates/.ai/workflows/find-email.org
+++ b/claude-templates/.ai/workflows/find-email.org
@@ -1,5 +1,5 @@
#+TITLE: Find Email Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-01
* Overview
diff --git a/claude-templates/.ai/workflows/first-session.org b/claude-templates/.ai/workflows/first-session.org
index 60118a2..147026f 100644
--- a/claude-templates/.ai/workflows/first-session.org
+++ b/claude-templates/.ai/workflows/first-session.org
@@ -1,5 +1,5 @@
#+TITLE: First Session Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
Run this workflow on the first Claude Code session for a new
project. It establishes the git/.ai policy, orients Claude to the
diff --git a/claude-templates/.ai/workflows/flashcard-review.org b/claude-templates/.ai/workflows/flashcard-review.org
index 31027b3..09af348 100644
--- a/claude-templates/.ai/workflows/flashcard-review.org
+++ b/claude-templates/.ai/workflows/flashcard-review.org
@@ -1,5 +1,5 @@
#+TITLE: Drill Deck Review Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-30
* Overview
diff --git a/claude-templates/.ai/workflows/helper-mode.org b/claude-templates/.ai/workflows/helper-mode.org
new file mode 100644
index 0000000..a6acfa7
--- /dev/null
+++ b/claude-templates/.ai/workflows/helper-mode.org
@@ -0,0 +1,101 @@
+#+TITLE: Helper Mode Workflow
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-06-15
+
+* Overview
+
+The role contract for a *helper instance*: a second Claude session running in the same project as a live primary session, spawned to look something up or make a scoped task update while the primary keeps working. This file is the single canonical home of the helper rules. [[file:../protocols.org][protocols.org]] carries a one-paragraph pointer here; the spawn paths and the wrap-up reference this file rather than restating it.
+
+A helper is not a subagent. When the work fits a dispatched subagent (the Agent tool) — a bounded lookup or analysis the primary folds back into its own context — use that instead; no second session exists and none of this applies. A helper is for interactive, long-lived parallel work Craig drives himself in a second terminal.
+
+The governing fact behind every rule below: the session-context split isolates each agent's /session state/ (=.ai/session-context.d/<id>.org=), but everything else in the project — =todo.org=, =.ai/notes.org=, =inbox/=, docs, the git index — is shared mutable state. Two Edit-tool writers on one org file lose updates silently: both read, both write, last write wins. The helper's job is to stay useful without ever being the second writer that clobbers the primary.
+
+* When to Use This Workflow
+
+No operator trigger phrase. A helper reaches this contract one of three ways:
+
+- The =ai --helper= launcher routes here after the roster confirms a live agent (the deterministic path).
+- Startup's roster check finds the session is not alone and routes here instead of running normal startup (the safety net for a raw =claude= launch).
+- An explicit "you are a helper, follow helper-mode.org" instruction (the manual fallback).
+
+If none of those applies — the roster shows the session is alone — this is a primary session. Run normal [[file:startup.org][startup.org]], not this.
+
+* Identity
+
+A helper is =helper-<rand4>= (four random hex/alphanumeric characters, e.g. =helper-a83f=).
+
+- If the launcher exported =AI_AGENT_ID=, use it. Otherwise self-assign =helper-<rand4>= now.
+- Record the chosen id as the *first line* of the session-context file, so it survives across tool calls (shell state does not). The id lives in the file, never in ambient shell state.
+- The active session-context path is =.ai/session-context.d/<id>.org=. Resolve it by prefixing the id explicitly wherever a script consumes =AI_AGENT_ID=:
+
+ #+begin_src bash
+ AI_AGENT_ID=<id> .ai/scripts/session-context-path
+ #+end_src
+
+ The id must be unique per run; =helper-<rand4>= and the launcher's epoch-tailed id both satisfy that (see protocols.org "Agent-scoped path").
+
+* Read/Write Contract
+
+Four tiers, by how much coordination the write needs.
+
+** Always safe
+
+Any read, anywhere in the project. Writes to the helper's own =.ai/session-context.d/<id>.org=.
+
+** Safe by discipline — scoped task updates (the case Craig named)
+
+Scoped writes to shared org files (=todo.org=, =.ai/notes.org=) are allowed under four rules, together:
+
+1. Re-read the file region immediately before each edit.
+2. Anchor the edit on a unique heading.
+3. Scope each edit to that single heading's subtree.
+4. Never reflow, restructure, or sweep the file.
+
+Appending a new =**= task at a section's end and editing one existing task's body or state both qualify. The race window collapses to seconds, and a collision corrupts one heading rather than the whole file. Log the intended edit first (see Write-ahead journal below).
+
+** Primary-only — never as a helper
+
+- File-wide passes: =todo-cleanup.el=, =lint-org.el=, =wrap-org-table.el=, archive sweeps.
+- Inbox processing and template sync.
+- ALL git mutation: commit, push, pull, stash. Two committers in one worktree contend on the index lock and interleave staging.
+- Startup's Phase A.0 pulls and the =.ai/= rsync. The primary already did them, and a concurrent pull-under-edit is exactly the race the startup guards exist to prevent.
+- Memory writes. =MEMORY.md= is a shared read-modify-write index with no heading anchors, so it has the lost-update shape of =todo.org= with none of the scoped-edit protection.
+
+The git ban is concurrency-scoped. /Helper wrap-up/ below lifts it for exactly one case: an orphaned helper that finds itself alone.
+
+** Escalation
+
+Anything the contract blocks gets reported to Craig, or — for a cross-project handoff — routed through =inbox-send= to the owning project's =inbox/=. The helper leaves its tree changes for the primary's next commit, or describes them in a note to Craig.
+
+* Data-Integrity Rules
+
+The scoped-edit discipline covers helper-vs-primary edits on /different/ headings. Four loss windows remain. They matter doubly in a consolidated project where one =todo.org= carries every task and corruption has maximal blast radius.
+
+1. *Primary file-wide passes vs a live helper.* A whole-file rewrite run while a helper is mid-edit clobbers the helper's just-written change. Enforced primary-side (the live-helper gate before any hygiene pass), but the helper's part is to keep its own session-context file current so the primary's gate can see it is live.
+2. *A new primary starting while a helper runs.* The helper's uncommitted edits make the tree dirty; Phase A.0 already skips pulls on a dirty tree, and startup surfaces live =session-context.d/= files so the new primary knows /why/ the tree is dirty rather than treating it as mess to resolve.
+3. *Write-ahead edit journal.* Before applying any shared-file edit, log it — file, heading, one-line intent — to the helper's own session-context file. This tightens the Session Log discipline to log-before-write for shared files specifically, so after a crash the journal shows which edits landed.
+4. *The memory dir.* Helpers don't write memory at all. Candidate memories go into the helper's session log; the primary (or its wrap-up promotion check) writes them.
+
+One collision nit: =inbox-send= filenames carry minute-resolution timestamps, so a helper and primary sending to the same target in the same minute with the same slug would collide. Helper-originated sends include the agent id in the slug.
+
+* Light Startup
+
+A helper does not run normal startup. It runs a light version:
+
+1. Self-assign or adopt the identity (above) and create =.ai/session-context.d/<id>.org= with the id on the first line.
+2. Read [[file:../protocols.org][protocols.org]] and this file. Read =.ai/notes.org= for project context if useful.
+3. Do NOT run Phase A.0 pulls, =make install=, or the =.ai/= rsync — the primary owns those.
+4. Do NOT process the inbox — primary-only.
+5. Begin the work Craig spawned the helper for.
+
+* Helper Wrap-Up
+
+When the helper's work is done:
+
+1. Re-run the roster (=.ai/scripts/agent-roster=) to learn whether a primary is still live.
+2. *Primary still live (the normal case):* finalize the Summary in the helper's own =.ai/session-context.d/<id>.org=, archive it to =.ai/sessions/YYYY-MM-DD-HH-MM-<id>-<description>.org=, and stop. Do NOT commit, push, or run hygiene — the primary's next commit picks up the archived file and any scoped edits the helper left in the tree.
+3. *Orphaned helper (roster shows the helper is now alone):* the primary already exited, so the helper assumes full closing duties — the git ban lifts because the concurrency that justified it is gone. Commit and push the tree (including the helper's own edits, which would otherwise strand as a dirty tree), per the normal wrap-up flow in [[file:wrap-it-up.org][wrap-it-up.org]].
+
+* Status
+
+Phase 1.5 of the generic-agent-runtime spec. This contract is the canonical home; the spawn paths (=ai --helper=, startup's roster branch) and the [[file:wrap-it-up.org][wrap-it-up.org]] helper branch route here. Those wiring pieces ship behind the spec's bats-then-drills-then-pilot gate and are not yet live; until then, the manual "you are a helper" instruction is how a session adopts this contract.
diff --git a/claude-templates/.ai/workflows/inbox-zero.org b/claude-templates/.ai/workflows/inbox-zero.org
deleted file mode 100644
index aa7c273..0000000
--- a/claude-templates/.ai/workflows/inbox-zero.org
+++ /dev/null
@@ -1,97 +0,0 @@
-#+TITLE: Inbox Zero Workflow
-#+AUTHOR: Craig Jennings & Claude
-#+DATE: 2026-06-13
-
-* Overview
-
-The roam global inbox (=~/org/roam/inbox.org=) is Craig's cross-project GTD capture: one shared file every project can see. This workflow routes each inbox item to the project that owns it. The current session claims only the items belonging to THIS project, files them into the project's =todo.org=, and removes them from the shared inbox. Everything it doesn't own stays. The aspiration is inbox zero: over time, every item lands in its owning project.
-
-This is NOT =process-inbox.org=. That workflow handles the project's own =inbox/= directory (handoffs from other projects, scripts, Craig). This one handles the single global =roam/inbox.org= and the cross-project routing a shared file creates.
-
-This is also distinct from the wrap-up inbox/transcript routing feature (which moves session-filed keepers between projects). This routes the shared roam capture file by ownership prefix.
-
-** Scope: single-destination (v1)
-
-This version routes each item to its one owning project, identified by an explicit =<project>:= heading prefix. The multi-project domain-aware mode, which would guess the owner of every unprefixed item and empty the whole inbox in one run, is deferred (see "Deferred: domain-aware routing" at the end). v1 claims only what's prefixed for the current project, surfaces the rest, and never guesses.
-
-** Three callers
-
-Reused from three callers so the steps live in one place:
-- *Startup* (read-only nudge) — count the items, identify which appear related to this project, surface both numbers, offer processing as one of the startup options. Never auto-files.
-- *Wrap-up* (Step 3 sub-step) — sweep items that belong here before the cleanup scripts, so imported tasks lint and ride the wrap commit.
-- *On demand* — "inbox zero", "empty the inbox", "process the roam inbox", "triage my roam inbox".
-
-Each project touches the roam inbox at least twice a session this way: once at startup, once at wrap-up.
-
-* The ownership rule (the coordination primitive)
-
-The inbox is shared, so the workflow must never let two projects fight over an item or let one project grab another's. Ownership is by explicit prefix:
-
-- =<project>: ...= heading → owned by that project. The current project claims only items prefixed with its own identifier.
-- Prefixed for *another* project → leave untouched (cross-project boundary, =protocols.org=).
-- *No prefix* → unowned. Never auto-claim. Surface as candidates a human can claim or prefix.
-
-The prefix partition is what makes concurrent triage across projects safe: each project only ever removes its own items, so two sessions editing the inbox touch disjoint lines.
-
-** Resolving this project's identifier (v1)
-
-Use the project root basename plus its common aliases (=.emacs.d= ↔ =emacs=, and the obvious ones: =rulesets=, =work=, =home=). A project may override the inferred set with an =:INBOX_PREFIX:= line in =notes.org='s *Workflow State* section when the basename is fragile (a dot in the name, an alias the inference misses). The explicit override is optional in v1; the durable multi-project resolution is part of the deferred domain-aware mode.
-
-* Phase A — Identify, count, and match
-
-1. Resolve the current project's identifier and aliases (above).
-2. Read =~/org/roam/inbox.org=. If absent, silent no-op (the file lives only on machines with the roam clone).
-3. Bucket every item under the inbox heading:
- - *claimed* — prefixed for this project
- - *foreign* — prefixed for another project → leave
- - *unowned* — no project prefix
-4. *Summarize the scan* (Craig's requirement, every scan): report the total item count in the inbox, then the count that appears related to this project. "Appears related" is the union of claimed items (exact prefix) and any unowned item whose topic plainly concerns this project's domain (a content judgment, surfaced as a candidate, never auto-claimed). Foreign-prefixed items are not "related" — they belong to their owner.
-5. If both claimed and related-unowned are empty, report the total and stop (the common case for most wraps).
-
-* Phase B — File each claimed item into todo.org
-
-Apply =process-inbox.org='s discipline against the project's =todo.org=; don't reinvent it:
-
-1. *Status check first.* Already done, or already a task in =todo.org=? → drop it, or fold into the existing task (dated sub-entry per =todo-format.md=). Don't duplicate.
-2. *Rewrite* to terse-heading + body per =todo-format.md=.
-3. *Priority + tags from THIS project's scheme* — the legend at the top of its =todo.org=, tags from that scheme's allowed set only. The project expresses someday-maybe with =[#D]=; there's no special someday-maybe routing.
-4. *File* under the project's Open Work section.
-
-* Phase C — Reconcile the shared inbox
-
-The roam inbox lives in a git repo (=~/org/roam=, auto-synced by the =roam-sync= timer). Edit it carefully:
-
-1. *Pull first* (=git -C ~/org/roam pull --ff-only=). If it can't fast-forward (dirty tree, divergence), surface and stop. Don't auto-stash, auto-merge, or force. Resolve before removing items.
-2. *Remove only the claimed items.* Never touch foreign or unowned items.
-3. *Commit the roam repo as its own commit* (separate from any project wrap commit): =chore(inbox): route <project> tasks to <project>/todo.org=. Push, or leave for the =roam-sync= timer. Surface a blocked push; don't force.
-
-* Phase D — Surface
-
-Report: moved (with their new priorities and tags), folded, dropped-as-done. Then the residue: foreign items (left for their owners, count only) and unowned items (count plus the headings that appear related to this project, for manual claim or prefix). Same "summarize what we kept" shape.
-
-* Skip conditions
-
-- No =~/org/roam/inbox.org= → silent no-op.
-- No claimed and no related-unowned items → report the total, stop.
-- Roam pull blocked → surface, stop before editing.
-
-* Caller integration
-
-** Startup (read-only nudge)
-
-Phase A of =startup.org= reads =~/org/roam/inbox.org= and produces the scan summary; Phase C surfaces one line: "Roam inbox: N items total, M appear related to this project — say 'inbox zero' to file them." Offered as one of the priority options. Startup never auto-files; it counts and offers.
-
-** Wrap-up (Step 3 sub-step)
-
-A sub-step at the start of wrap-up Step 3 (before the cleanup scripts, so imported tasks get linted and ride the wrap commit) delegates here for the claimed set. Skip-fast when nothing matches.
-
-* Deferred: domain-aware routing (future work, multi-project)
-
-v1 handles the single-destination case via the prefix rule. The multi-project parts are deferred until the need is real:
-
-1. *Domain-aware empty-it-all mode.* If rulesets held a description of each project's domain, one run could guess the owner of every item (prefixed or not) and empty the whole inbox at once, delivering each item to its owning project's =inbox/= via =inbox-send= (where that project's =process-inbox= gate still decides whether to file it). This turns "inbox zero" from a per-project aspiration into a single command. Open: where the domain map lives (central registry vs each project's =notes.org=), how confident a guess must be before auto-routing vs surfacing, and whether a low-confidence item stays put.
-2. *Explicit per-project =:INBOX_PREFIX:= as the durable resolver*, replacing basename inference.
-3. *Unowned-item lifecycle* once domain-aware routing exists (no item stays unrouted indefinitely).
-4. *Concurrent push contention* on the shared roam repo: the pull-before-edit + ff-only + surface-on-conflict floor may want a retry-once-after-pull.
-
-Take these up when the single-destination version is in use and the multi-project pain is concrete.
diff --git a/claude-templates/.ai/workflows/inbox.org b/claude-templates/.ai/workflows/inbox.org
new file mode 100644
index 0000000..6faa20f
--- /dev/null
+++ b/claude-templates/.ai/workflows/inbox.org
@@ -0,0 +1,524 @@
+#+TITLE: Inbox Workflow (Engine)
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-06-23
+
+* Overview
+
+One engine for the project's inbox surfaces. Inbox items are *ideas to evaluate*, not orders to execute — each is a proposal that earns a place in =todo.org= or git history only when it passes the value gate. The engine holds the shared disposition machinery once: the three-question value gate, the skeptical review, the disposition ladder, the reply-to-sender discipline, the capture-guard before a roam write, and the priority-scheme check. Each *mode* is a thin section that names which surface it reads, how it enters and exits, and which core steps it runs.
+
+Two surfaces feed a project, and there is a recurring check over the second:
+
+1. *Project-local =inbox/= dir* — handoffs from other projects (via =inbox-send=), from scripts, and from Craig (typed directives saved as files). Handled by *process mode*; watched on a cadence by *monitor mode*.
+2. *Global roam inbox* (=~/org/roam/inbox.org=) — Craig's cross-project GTD capture, one shared file every project can see. Handled by *roam mode*, which claims only the items this project owns. *Auto inbox zero* runs roam mode on a recurring interactive loop.
+
+A *third* surface — external accounts (email / calendar / PRs) — is a different domain and stays in its own engine: =triage-intake.org= and its source plugins are *not* part of this engine. "Deal with my inbox dirs" is here; "what's new across my accounts" is there.
+
+*Two altitudes.* For the user, the trigger phrase picks the mode and the phrases are unchanged (see When to Use). For the implementer, this is one file: the core sections are written once, and each mode references them by name ("run the value gate (core §1) on each item") rather than restating them.
+
+* When to Use This Workflow
+
+The trigger phrase selects the mode. Every phrase below still works; it now routes to a mode of this engine.
+
+** Process mode — the local =inbox/= dir
+
+- "process inbox" / "process the inbox"
+- "handle the inbox"
+- "what's in inbox" / "what's in the inbox"
+- "let's clear the inbox" / "let's process the inbox items"
+
+Auto-invocation: startup Phase C delegates here when the local inbox is non-empty — don't ask, just run it.
+
+** Monitor mode — process mode on a cadence
+
+- "monitor the inbox" / "watch the inbox" — *the defined meaning:* one process pass now, then loop every 15 minutes (see the Monitor mode Cadence section). The phrase *is* the loop, not an opt-in extra.
+- "respond to the handoffs" / "handle the handoffs" — a single pass now, no loop.
+
+Ambient (always on, even with no loop running): the =inbox-status= task-boundary check (Monitor mode).
+
+** Roam mode — the global roam inbox
+
+- "inbox zero" / "empty the inbox" / "process the roam inbox" / "triage my roam inbox"
+
+Called read-only from startup (count + offer) and as a wrap-up Step 3 sub-step.
+
+** Auto inbox zero — recurring interactive roam check
+
+- "auto inbox zero"
+
+Match this before "inbox zero" — the auto phrase contains the roam phrase as a substring, so the longer match wins. Starts a recurring =/loop=-driven roam-mode pass; see the Auto inbox zero mode.
+
+** Boundary
+
+Do *not* invoke this engine for an inbox item that is clearly out-of-scope for the project — that is a cross-project routing problem, handled per the cross-project boundary rule in =protocols.org=. And do not invoke it for external-account triage ("what's new in email/cal/PRs") — that is =triage-intake.org=.
+
+* Core §1 — The value gate
+
+Every inbox item (local or roam) passes through three questions. One *yes* is enough to accept.
+
+1. *Does it advance an existing TODO?* Look up by topic in =todo.org='s open work. If the item extends a filed task, fold it in. If it implements a filed task, do the work.
+2. *Does it improve how the project works?* Architecture cleanup, workflow refinement, tooling, rule hygiene, drift detection — anything that makes the project itself more effective.
+3. *Does it serve the project's stated mission?* Read =notes.org= *Project-Specific Context* if the mission isn't obvious from the working directory and current task. The item should advance that mission, not orbit it.
+
+Three *no*s means reject. The rejection isn't lazy — an idea that doesn't help any current task, doesn't improve the system, and doesn't serve the mission is genuine noise, and accepting it inflates =todo.org= without payoff.
+
+* Core §2 — The skeptical review
+
+The value gate decides whether an item is worth taking. This review decides whether what it proposes is *right*, *complete*, and *as simple as it should be*. Run it on every task and file that arrives — not only shared-asset change proposals. Pure FYIs and replies that ask for nothing skip it.
+
+Approach the file with curiosity and skepticism. Work through, in writing — the core pass on every item:
+
+1. Is the request actually right — does it do what it claims, and is the claim correct for this project?
+2. Is it complete, or does it leave a gap — an unhandled case, a missing step, an untested path?
+3. Should it be simpler?
+4. Can it be enhanced to be more effective than as proposed?
+5. Does it conflict with any existing instruction — workflows, skills, rules, protocols, CLAUDE.md?
+
+When the item proposes a change to *shared assets* — template workflows, rules, skills, scripts, anything synced to consuming projects — or to a substantive convention, add the cross-project battery. It arrived from one project's context; you're evaluating it for all of them:
+
+6. Does this make sense for *all* consuming projects, or just the sender's situation?
+7. How does it change a common activity Craig performs — better, worse, or differently than the sender assumed?
+8. Plus at least three more questions specific to this change — what breaks for artifacts already using the old shape, what tooling interacts with it, what's underspecified, what the sender's worked example doesn't exercise.
+
+Output: a short summary of the thinking and a recommendation (do it / do it with named changes / file / reject). For shared-asset and convention changes the recommendation is surfaced to Craig for approval before applying; for ordinary tasks and files it feeds the act-vs-file and no-approvals-execute decision (Monitor mode).
+
+** In a no-approvals session: shared-asset changes defer and stage
+
+Shared-asset and convention changes still don't self-apply when Craig has put the session in no-approvals mode — they need his decision, so they fail the *solo* test in Monitor mode's executing-in-no-approvals criteria. Ordinary tasks and files that pass the review and are quick + solo execute under that criteria instead; this defer-and-stage path is for the shared-asset and convention changes that don't qualify. Run the review, prepare the edits in =working/<task-slug>/= (a patch file or the worked-out diff), file a =[#B]= VERIFY carrying the decision package, and reply to the sender that it's parked. The sender's local stopgap (per =cross-project.md='s propagation process) means the delay costs nothing — the canonical update is about durability, not speed.
+
+Wording-only fixes — no consuming project acts differently — may proceed even then, logged in the session log.
+
+The VERIFY shape (top-level, =[#B]= so startup's A/B surfacing catches it; no =SCHEDULED= unless the proposal names a real deadline):
+
+#+begin_example
+** VERIFY [#B] Parked: <proposal topic> (from <sender>)
+What arrived: <one line — what the handoff proposes>.
+Recommendation: <accept as-is / accept with changes / reject> — <2-3 line
+skeptical-review summary: what's right, what to change, what was checked>.
+Prepared diff: [[file:working/<slug>/proposed.diff]] — apply is mechanical on
+your go.
+Say "approve the parked <topic>" (or adjust / reject) and it gets applied.
+#+end_example
+
+The full question-battery answers live in the session log and the =working/= dir, not the task body — the body carries the conclusion, with the trail one link away.
+
+* Core §3 — The disposition ladder
+
+Every item that clears the value gate gets one disposition. The first six are the per-item outcomes; *park* is the no-approvals shared-asset path from core §2.
+
+** Implement now
+Small, scoped, clear, no design call required. The work is the disposition. Do the work, commit per the project's commit flow, delete the inbox file. The commit message references the inbox item by filename so the provenance lands in =git log=.
+
+** Fold into existing TODO
+The item extends a task already filed. Update the parent TODO's body with a dated reconciliation sub-entry per =todo-format.md= (=*** YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <what landed>=). Move substantive content to =docs/design/<date>-<topic>.<ext>= if it's worth keeping; reference from the TODO body. Delete the inbox file.
+
+** File as TODO
+Substantive but waits, or needs design/triage before implementation. Add the TODO under =* <Project> Open Work= with priority + tags per the priority-scheme check (core §6). Body summarizes the proposal and links the inbox content if it's been moved to =docs/design/=. Delete the inbox file (or move it to =docs/design/= first if the content survives).
+
+*Route-candidate marking (feeds the wrap-up router).* After filing, check whether the keeper's inferred home is a different project:
+
+#+begin_src bash
+python3 .ai/scripts/route_recommend.py --item "<the keeper's heading + body text>" --exclude "$(basename "$PWD")"
+#+end_src
+
+On a =<destination>\tstrong= or =<destination>\tweak= result, stamp the new TODO's property drawer with =:ROUTE_CANDIDATE: <destination>= (create the drawer if the task has none). A =none= result stamps nothing, and a local keeper stays unstamped. The marker is the wrap-up router's entire candidate set — =wrap-it-up.org= Step 3 surfaces exactly the =:ROUTE_CANDIDATE:=-tagged tasks and offers to deliver each to its destination's inbox, never scanning the standing backlog. Stamping is cheap and reversible (the router's skip leaves the task in place; a wrong marker is one property line to delete), so prefer stamping on any plausible match — the human reviews the batch at wrap time.
+
+*Blocking-dependency handoff.* A special shape: another project sends a note that *this* project's work is blocking one of theirs ("your task X is blocked on us — we need Y"). File or link the owning task, tag it =:blocker:=, and name the requesting project in the body (see the cross-project dependency convention in =todo-format.md=). The =:blocker:= tag makes =open-tasks.org= surface that task *first*, since clearing it unblocks the other project. Dedup against an existing task rather than filing a duplicate. When the work later lands, drop =:blocker:= and notify the waiting project (=inbox-send <their-project> --text "Delivered: <what> — you're unblocked."=) so it can lift its own =:blocked:=.
+
+** Defer
+Rename in place to =inbox/PROCESSED-<original-filename>= and add a brief comment line at the top: =# Deferred YYYY-MM-DD: <condition>=. Don't accumulate deferred items indefinitely — sweep them on a future process pass when the condition is met or the deferral has aged out.
+
+** Reject — by source
+- *From Craig* — push back honestly in chat. State why you won't implement; offer the conditions under which you would, if any. The inbox file stays until Craig confirms — override re-enters as accept, acknowledgment deletes the file. Don't theatre the pushback: if you don't genuinely think Craig is wrong, just do the work.
+- *From another project (handoff)* — write a response file at =/tmp/inbox-response-<topic>.org=: a heading naming the original handoff and date, one paragraph on the rejection rationale (*which* value-gate question failed and why), one paragraph on the condition under which you'd reconsider (or "never, this misreads the project's mission" if that's the truth). Deliver via =inbox-send <sender> --file /tmp/inbox-response-<topic>.org=. Delete the local inbox file after the response lands. Silent rejection on a handoff trains the sender to escalate around the channel — always close the loop.
+- *From a script or automated system* — just delete. No notification.
+
+** Park (skeptical review in a no-approvals session)
+Move the proposal file into =working/<task-slug>/= alongside the prepared diff, file the =[#B]= VERIFY per core §2, reply to the sender that it's parked for Craig's review, and delete the inbox file. On Craig's approval the apply is mechanical: apply the prepared edits, run the normal verify-and-publish flow, close the parked =**= VERIFY per =todo-format.md= (a top-level VERIFY resolves to =DONE= + =CLOSED:=, not a dated header), and send the acceptance reply. On rejection, the reject-from-another-project flow above runs unchanged.
+
+* Core §4 — Reply-to-sender discipline
+
+A handoff came from another project's agent (or the user). Close the loop:
+
+- *Accepted and acted on* — send a confirmation to the sender via =inbox-send <sender> --text "..."=, naming what landed and the commit, so they're not left guessing (they can't see this project's git log). =inbox-send= excludes the current project as a target, so a self-sourced item is handled in-session, not sent.
+- *Accepted and filed* — a short confirmation that it's filed and where, so the sender knows it wasn't dropped.
+- *Rejected* — always state the why (which value-gate question failed), per the reject-by-source ladder (core §3).
+
+Cross-project boundary: never act on a file under another project's =.ai/= scope from here — route it back as a handoff (see =cross-project.md=).
+
+* Core §5 — Capture-guard before a roam write
+
+Before *any* read-modify-write of =~/org/roam/inbox.org=, run the capture-guard. This runs first because the Phase D edit rewrites the file on disk, and editing underneath a live capture wedges it just as a stray hand edit would.
+
+*Wait-and-retry, not bounce.* Use the poll mode so a *transient* capture clears itself instead of immediately kicking the work back to the caller:
+
+#+begin_src bash
+.ai/scripts/capture-guard --wait "$HOME/org/roam/inbox.org"
+#+end_src
+
+An org capture is usually only a few seconds of mid-finalize state, so =--wait= (default 30s, re-checking every ~10s) returns the instant it clears and reports blocked only if it's *still* open at the deadline. The common case — nothing capturing — returns instantly without sleeping. This is the fix for the guard bouncing a caller over a capture that would have cleared on its own a moment later. (The bare single-shot form — no =--wait= — stays available for a caller that genuinely must not block.)
+
+- *Exit 0* → no live capture, or it cleared during the wait (or no reachable Emacs). Proceed with the edit.
+- *Exit 1* → an indirect org-capture buffer is *still* cloned from the roam inbox after the wait (the script prints the offending buffer name). Editing underneath it would leave the capture pointing at stale state and unable to finalize with =C-c C-c= (see =emacs.md=). Only now does the per-caller fallback fire:
+ - *On-demand / interactive run* → stop and surface: "You have a live org-capture session open against the roam inbox (=<buffer>=) — finalize it (=C-c C-c=) or abort it (=C-c C-k=) and I'll continue." Re-run the guard and resume once it returns clean.
+ - *Auto inbox zero (=/loop=) cycle* → don't surface or wait further; defer the roam reconcile to the next cycle, which is itself the retry at loop cadence. The items were already filed in Phase C, so the next cycle's Phase C status-check drops the duplicates and its Phase D removes them. Note one line: "roam reconcile deferred — a capture is still open; next cycle catches it."
+ - *Wrap-up sub-step* → don't block the wrap. Skip the roam reconcile for this run and surface one line: "Skipped roam-inbox reconcile — a live org-capture is open against it; claimed items stay and get caught next run." The items were already filed into =todo.org= in roam mode Phase C, so the next roam run's Phase C status-check drops the duplicates and its Phase D removes them — the skip self-heals.
+
+*The roam-write lock (around the Phase D edit).* Capture-guard protects against a live *human* capture; the roam-write lock protects against a concurrent *agent* writer (a sentry inbox pass, a KB promotion) editing =~/org/roam= at the same time. Acquire it after the capture-guard clears and release it after the edit-and-trigger, so the two guards nest — capture-guard underneath, the agent lock around the write:
+
+#+begin_src bash
+if [ -x .ai/scripts/agent-lock ]; then
+ .ai/scripts/agent-lock acquire roam-write --wait || { echo "roam-write busy; deferring roam reconcile" >&2; exit 1; }
+fi
+# capture-guard (above), then the Phase D read-modify-write of ~/org/roam/inbox.org,
+# then trigger the sync — roam-sync stays the only committer:
+systemctl --user start roam-sync.service
+[ -x .ai/scripts/agent-lock ] && .ai/scripts/agent-lock release roam-write
+#+end_src
+
+Degrade gracefully when =agent-lock= isn't installed (an older checkout mid-sync): the write proceeds unlocked, today's behavior. A *present* helper reporting the lock busy after its bounded wait defers the roam reconcile (the auto-loop and wrap-up paths already defer-and-retry per the fallback list above); an *absent* helper never blocks it.
+
+* Core §6 — Priority-scheme check
+
+This gates filing whenever there are accept-and-file items. Check whether =todo.org= has a top-of-file priority scheme (an explicit legend defining =[#A]= through =[#D]= semantics and mandatory/optional tag conventions — a =* <Project> Priority Scheme= section or similar).
+
+- *Scheme present* — file new TODOs per the scheme. Every TODO gets a priority cookie matching the legend's rules, the mandatory type tag, and any applicable effort/autonomy tags.
+- *Scheme absent* — surface one sentence: "This project has no priority scheme. We should adopt one before filing the new TODOs from this inbox pass — want me to propose one based on the rulesets scheme?" If Craig says yes, do that first (the =/research-priority-scheme= research subagent pattern in rulesets is the reference). If Craig says no, file the TODOs without grading but flag in the commit message that they're un-prioritized pending a scheme.
+
+The point is to avoid adding ungraded =TODO= entries to a project that's never agreed on what =[#A]= means.
+
+* Mode: process
+
+Reads the project-local =inbox/= dir. Entry: a trigger phrase, or startup Phase C on a non-empty inbox. Exit: inbox empty (excluding =.gitkeep= and intentional =PROCESSED-*=), session log updated, =:LAST_INBOX_PROCESS:= stamped.
+
+** Phase A — Inventory (one parallel batch)
+
+Issue these reads in one parallel batch:
+
+1. List =inbox/= excluding =.gitkeep= and =PROCESSED-*= prefixes (use =\ls -la inbox/= per the protocols.org exa-alias note).
+2. Read =notes.org= *Project-Specific Context* if mission isn't already loaded in the session.
+3. Read =todo.org='s top-of-file priority scheme if present.
+
+For each inbox file, parse the filename for sender. Two common patterns:
+
+- =YYYY-MM-DD-HHMM-from-<sender>-<topic>.<ext>= — from another project via =inbox-send=.
+- =<topic>.org= — typically from Craig directly, or from a script.
+
+Note the file type. =.eml= files need the extract script (not raw =Read=):
+
+#+begin_src bash
+# View mode
+python3 .ai/scripts/eml-view-and-extract-attachments.py inbox/<file>.eml
+
+# Pipeline mode (extract attachments to a directory)
+python3 .ai/scripts/eml-view-and-extract-attachments.py inbox/<file>.eml --output-dir assets/<target>/
+#+end_src
+
+Everything else, read directly.
+
+** Phase B — Evaluate each item
+
+For each inbox file:
+
+1. *Read it.* Full read for substantive proposals (org files with TODO entries, design notes, multi-section docs); skim short FYIs and one-liner asks.
+2. *Identify the shape.* Instruction, question, proposal, FYI, or handoff — shapes guide disposition.
+3. *Apply the value gate* (core §1). One yes → candidate accept. Three nos → candidate reject.
+4. *Run the skeptical review* (core §2) on the item before classifying — the core pass on every accepted task and file, plus the cross-project battery when it proposes a shared-asset or convention change. Its summary + recommendation rides along to Phase C; in a no-approvals session it gates whether the item self-applies (quick + solo + agreed, per Monitor mode) or, for shared-asset and convention changes, defers and stages.
+5. *Within accept, classify* by the disposition ladder (core §3): implement now / fold into existing TODO / file as TODO.
+6. *Within reject, classify by source* (core §3): from Craig / from another project / from a script.
+
+** Phase B.1 — Priority-scheme check
+
+Run core §6. This gates Phase C filing when there are accept-and-file items.
+
+** Phase C — Surface dispositions
+
+Numbered options inline per =interaction.md= (no popup). Recommendation at item 1.
+
+Batch trivial items (one-line rejections of script noise, obvious file-as-TODO accepts where the scheme is already settled) into a single confirm-all prompt. Walk substantive items one at a time so the decision is visible.
+
+Per-item template:
+
+#+begin_example
+<filename> from <sender>: <one-line summary>
+Value-gate read: <yes/no on each of the three questions, one phrase each>
+Disposition recommendation: <implement / fold into <TODO> / file [#X] :tags: / reject>
+
+1. <recommendation as item 1>
+2. <alternative>
+3. Defer — leave in inbox under PROCESSED-<topic>.<ext> until <condition>
+4. Something else
+#+end_example
+
+For items that went through the skeptical review, the surfaced disposition includes its summary + recommendation, and approval here is what authorizes the apply. In a no-approvals session those items are reported as parked (the =[#B]= VERIFY) rather than surfaced for live approval.
+
+For pure FYIs that need no action, surface as a single line and recommend delete-with-acknowledgment.
+
+** Phase D — Apply
+
+Apply each disposition per the ladder (core §3). The flow is autonomous past Craig's Phase C approval.
+
+** Phase E — Close out
+
+Verify =inbox/= is empty (excluding =.gitkeep= and any intentional =PROCESSED-*= files). Run =\ls -la inbox/= and confirm.
+
+Update the session log per =protocols.org= with one short paragraph: count processed, count accepted (implement/fold/file split), count rejected (Craig/handoff/script split), and the commit SHA if a commit landed.
+
+Stamp =:LAST_INBOX_PROCESS:= in =notes.org='s *Workflow State* section if it exists, so future workflows that gate on freshness can read it. Same format as =:LAST_AUDIT:= (=YYYY-MM-DD=).
+
+* Mode: monitor
+
+Process mode on a cadence. This is the *when, how-often, and act-vs-file* layer; the per-item disposition mechanics are the core sections, run via process mode — not restated here. Monitor decides *that* an item gets handled and *how I respond*; the core decides *what disposition* each item gets.
+
+The gap it closes: handoffs that arrive mid-session used to sit unseen until the user asked or the next startup ran. A handoff the sender can't see being handled trains them to escalate around the inbox channel.
+
+** Preconditions — before starting
+
+Never begin monitoring on a dirty worktree or a failing test suite. A dirty tree means the auto-commit at the end of an executed item sweeps up unrelated changes; a red suite means you can't tell whether the monitor broke something. At the start:
+
+1. =git status --porcelain --untracked-files=no= is empty (no tracked modifications). Untracked and gitignored files never block — an inbox drop is exactly what this mode processes, and a scratch file is none of its business (the template-freshness policy in =startup.org= Phase A.0). The tracked-only gate is safe because the per-item commit stages its files explicitly (=commits.md=: only intended changes staged) — never =git add -A=, which would sweep untracked files and is the failure this gate guards against.
+2. A full test run is all green (=make test= here, or the project's full-suite command).
+
+If *dirty*: offer to commit the pending changes in discrete, logical batches before starting. If *red*: offer to investigate the failures first. Surface the blocker with inline numbered options per =interaction.md= and wait — monitoring does not start until the tree is clean and the suite is green.
+
+** Cadence — how often to check
+
+*"Monitor the inbox" = run now, then loop every 15 minutes.* Do one process pass over any pending handoffs immediately, then start the loop:
+
+#+begin_src
+/loop 15m check the inbox with inbox-status and run inbox.org process mode over any pending handoffs
+#+end_src
+
+Each firing runs the cheap =inbox-status= check first and only does a full process pass when items are pending. The loop is the monitoring; it runs until Craig stops it or the session ends. Honor the Preconditions gate before the first pass and the Close-out gate when the loop stops.
+
+*Ambient task-boundary check (always on, even without a loop).* After finishing a unit of work, before reporting back or asking "what's next," run the cheap status check:
+
+#+begin_src bash
+.ai/scripts/inbox-status -q
+#+end_src
+
+Exit 1 means handoffs are pending — list them (drop =-q=) and run process mode. Exit 0 means clean; say nothing. This is one =find=; it costs nothing to run often, and it's the fix for handoffs piling up unseen during long sessions.
+
+*Startup and wrap-up already cover their ends.* Startup Phase C processes a non-empty inbox; the wrap-up sanity check refuses to wrap with unprocessed handoffs. The task-boundary cadence fills the middle.
+
+*Mid-task arrivals.* If a handoff lands while you're mid-task and it's urgent (blocks the current work, or is time-sensitive), surface it right away. Otherwise batch it to the next task boundary so the current work isn't thrashed.
+
+** The act-vs-file decision
+
+Every accepted handoff (one that clears the value gate) is then either acted on now or filed as a task.
+
+*Act immediately — and just do it, no asking — when all of these hold:*
+- *Clear* — the action is unambiguous; no design decision or option-choice is needed.
+- *Bounded* — small, finishable this session, ideally a tight file set.
+- *Low-risk and verifiable now* — not a risky change to load-bearing infra (or trivially revertible), and testable/lintable this session.
+- *In-scope and safe* — within this project, not destructive or outward-facing without confirmation, not across a project boundary.
+- *Cheaper than deferring* — doing it now costs less than filing plus re-triaging later.
+
+When you decide to act, queue the work and do it. Don't ask first.
+
+*Exception:* a proposal to change a shared asset (template workflow, rule, skill, synced script) or a substantive convention never qualifies for silent act-now, however clear and bounded it looks — it routes through the skeptical review (core §2), which carries its own approval (or, in a no-approvals session, park) step.
+
+*File a task when any of these hold:*
+- It needs a judgment call, a design decision, or an option the user would pick.
+- It's large, multi-session, or sprawls across many files.
+- It's blocked (a dependency, an external thing, the user is away).
+- It's risky enough to want the user's eyes before it lands.
+- It's off the session's active goal and acting now would derail it (file and keep going, unless it's urgent).
+
+When you decide to file, *ask first* — inline numbered options per =interaction.md=, with *filing as option 1 (the recommendation)* and *"do it now" as option 2*:
+
+#+begin_example
+<handoff> wants <X>. My read: file it (needs <reason>).
+
+1. File as a TODO ([#?] :tags:) — Recommended
+2. Do it now instead
+3. Something else
+
+Pick a number.
+#+end_example
+
+*Always ask if you're unsure* which side of the line an item falls on. Decisiveness on clear act-now items is the point of the rule; the ask is for genuine ambiguity and for filing.
+
+** Executing in no-approvals mode
+
+When Craig has put the session in no-approvals mode, an accepted item may be implemented automatically — but only when all three of these hold:
+
+1. *Agreed* — you've run the value gate and the full skeptical review and concluded the change should be done, not merely that it's harmless.
+2. *Quick* — the whole implementation, including verification, is under ~15 minutes.
+3. *Solo* — you can carry it end to end without a decision from Craig. Manual verification you perform yourself is fine; needing Craig to choose an option, approve a design, or resolve an ambiguity is not.
+
+All three → implement it, verify, then commit and push at the end of that item (the Step 0 reconcile and pre-push check from the =publish= skill still run). Miss any one and it doesn't self-apply: a shared-asset or convention change needs Craig's decision, so it fails *solo* and routes to the defer-and-stage park (core §2 / core §3); an oversized item fails *quick* and gets filed.
+
+** Replying to handoffs
+
+Close the loop per the reply-to-sender discipline (core §4): confirm what landed (accepted-and-acted), confirm where it's filed (accepted-and-filed), or state the why (rejected).
+
+** The inbox-status script
+
+=.ai/scripts/inbox-status= lists unprocessed handoffs and exits nonzero when any are pending. Exclusions match the wrap-up sanity check (=.gitkeep=, =lint-followups.org=, =PROCESSED-*=). Exit 0 = clean, 1 = pending, 2 = no inbox/ or bad usage. Use =-q= for the count-only form the cadence check calls.
+
+** Close out — before finishing
+
+End the way it started: clean worktree, green suite. Before stopping the loop or reporting the pass done:
+
+1. Commit or revert every tracked modification left in the worktree — no tracked change remains uncommitted. Untracked files (unprocessed inbox drops, scratch) are not the monitor's to sweep.
+2. Run the full test suite once more and confirm all green.
+
+If either can't be satisfied — a half-done item, a failure introduced during the pass — surface it rather than leaving it. The next monitor run assumes a clean, green starting state (the Preconditions gate).
+
+* Mode: roam
+
+Reads the *global roam inbox* (=~/org/roam/inbox.org=), Craig's cross-project GTD capture: one shared file every project can see. This mode routes each roam item to the project that owns it. The current session claims only the items belonging to THIS project, files them into the project's =todo.org=, and removes them from the shared inbox. Everything it doesn't own stays.
+
+*Allowed from any project, work included.* Tidying the shared roam inbox is housekeeping on a shared resource, not a cross-project boundary crossing and not a durable KB-node write, so the =knowledge-base.md= work-denylist doesn't gate it (a sentry inbox-zero pass mis-parked the whole inbox as a boundary crossing from the work project on 2026-07-19 — the error this note closes). Reading roam and tidying its inbox are fine from work; only promoting a durable =agents/= node stays work-denylisted.
+
+The aspiration is inbox zero: after this mode runs, the current project's local handoff inbox has been processed (Phase A delegates to process mode) and the shared roam inbox no longer contains items explicitly owned by this project.
+
+This is distinct from the wrap-up inbox/transcript routing feature (which moves session-filed keepers between projects). This routes the shared roam capture file by ownership prefix.
+
+** Scope: single-destination (v1)
+
+Routes each item to its one owning project, identified by an explicit =<project>:= heading prefix. The multi-project domain-aware mode (guess the owner of every unprefixed item and empty the whole inbox in one run) is deferred — see "Deferred: domain-aware routing" at the end. v1 claims only what's prefixed for the current project, surfaces the rest, and never guesses.
+
+** Callers
+
+The steps live here so three callers reuse them:
+- *Startup* (read-only nudge) — count the items, identify which appear related to this project, surface both numbers, offer processing as one of the startup options. Never auto-files.
+- *Wrap-up* (Step 3 sub-step) — sweep items that belong here before the cleanup scripts, so imported tasks lint and ride the wrap commit.
+- *On demand* — the roam-mode trigger phrases.
+
+Each project touches the roam inbox at least twice a session this way: once at startup, once at wrap-up.
+
+** The ownership rule (the coordination primitive)
+
+The inbox is shared, so the mode must never let two projects fight over an item or let one grab another's. Ownership is by explicit prefix:
+
+- =<project>: ...= heading → owned by that project. The current project claims only items prefixed with its own identifier.
+- Prefixed for *another* project → leave untouched (cross-project boundary, =protocols.org=).
+- *No prefix* → unowned. Never auto-claim. Surface as candidates a human can claim or prefix.
+
+The prefix partition is what makes concurrent triage across projects safe: each project only ever removes its own items, so two sessions editing the inbox touch disjoint lines.
+
+*Resolving this project's identifier (v1).* Use the project root basename plus its common aliases (=.emacs.d= ↔ =emacs=, and the obvious ones: =rulesets=, =work=, =home=). A project may override the inferred set with an =:INBOX_PREFIX:= line in =notes.org='s *Workflow State* section when the basename is fragile. The explicit override is optional in v1; the durable multi-project resolution is part of the deferred domain-aware mode.
+
+** Phase A — Process the project-local inbox
+
+1. Check the project-local =inbox/= with =.ai/scripts/inbox-status -q=.
+2. If pending handoffs exist, run *process mode* before touching the roam inbox. Project handoffs are already addressed to this project, so they are higher-confidence and cheaper to clear than shared roam captures.
+3. If =inbox-status= reports no =inbox/= directory, note it and continue to the roam inbox. Some projects only participate in the shared roam capture flow.
+4. If process mode cannot finish because it needs Craig's decision, stop after surfacing that decision. Do not remove roam items in the same run; the project still does not have a clean inbox.
+
+** Phase B — Identify, count, and match roam items
+
+1. Resolve the current project's identifier and aliases (above).
+2. Read =~/org/roam/inbox.org=. If absent, silent no-op (the file lives only on machines with the roam clone).
+3. Bucket every item under the inbox heading:
+ - *claimed* — prefixed for this project
+ - *foreign* — prefixed for another project → leave
+ - *unowned* — no project prefix
+ - *empty* — a heading with no title and no body: just stars, optionally a =TODO=/keyword, and whitespace (e.g. =** =, =** TODO =, =*** TODO =). These are aborted or accidental captures, owned by nobody, and safe to delete regardless of project. A heading with any title text or any body content is never empty.
+4. *Summarize the scan* (Craig's requirement, every scan): report the total item count in the inbox, then the count that appears related to this project. "Appears related" is the union of claimed items (exact prefix) and any unowned item whose topic plainly concerns this project's domain (a content judgment, surfaced as a candidate, never auto-claimed). Foreign-prefixed items are not "related" — they belong to their owner. Note the empty count separately.
+5. If claimed, related-unowned, *and* empty are all absent, report the total and stop (the common case for most wraps). Empty entries on their own are enough to enter Phase D — the cleanup runs even when this project owns nothing else, since empties belong to nobody and removing them is what "check the inbox" should always do.
+
+** Phase C — File each claimed roam item into todo.org
+
+Apply the core disposition discipline against the project's =todo.org=; don't reinvent it:
+
+1. *Status check first.* Already done, or already a task in =todo.org=? → drop it, or fold into the existing task (dated sub-entry per =todo-format.md=). Don't duplicate.
+2. *Rewrite* to terse-heading + body per =todo-format.md=.
+3. *Priority + tags from THIS project's scheme* (core §6) — the legend at the top of its =todo.org=, tags from that scheme's allowed set only. The project expresses someday-maybe with =[#D]=; there's no special someday-maybe routing.
+4. *File* under the project's Open Work section.
+
+** Phase D — Reconcile the shared roam inbox
+
+The roam inbox lives in a git repo (=~/org/roam=, auto-synced every 15 minutes by the =roam-sync= timer). Craig captures into it constantly, so its working tree is dirty most of the time — which is exactly why this mode never runs =git pull= itself. A pull on a dirty tree fails, and that would block triage on nearly every run. Instead, edit the file and hand the git work to =roam-sync=, which already commits-first-then-rebases and so handles the dirty tree correctly.
+
+1. *Guard against a live org-capture session* — run the capture-guard in poll mode (=capture-guard --wait=, core §5) before the edit, so a transient capture clears itself rather than bouncing the run. On a still-blocked exit 1 the caller-specific fallback (interactive stop-and-surface / auto-loop defer-to-next-cycle / wrap-up skip-and-self-heal) is in core §5.
+2. *Remove the claimed items and the empty entries* from the working-tree file. Never touch foreign or unowned (titled) items. Empty entries (Phase B's =empty= bucket) are removed on every triage regardless of who would own a titled version, since an aborted capture belongs to nobody. The claimed-item removal and the empty sweep happen in the same edit.
+3. *Hand the commit + push to =roam-sync=.* Don't =git pull=, =git commit=, or =git push= here. Trigger the sync so the edit lands promptly rather than waiting up to 15 minutes for the next timer tick:
+
+ #+begin_src bash
+ systemctl --user start roam-sync.service
+ #+end_src
+
+ =roam-sync= commits the edit (under its generic auto-sync message), rebases onto the remote, and pushes. The removal is safe to land without a pull-first because only this project ever touches =<project>:=-prefixed lines (the ownership partition), so =roam-sync='s rebase can't conflict on the edit. Provenance for the routed tasks lives in the project's =todo.org= and session log, not the roam commit message. If =systemctl= isn't available, leave the edit for the next timer tick — it still lands.
+
+ Don't pull or stash the roam tree to "clean" it first: that fights =roam-sync= for ownership of the repo's git state. The edit-then-sync handoff is the whole point.
+
+** Phase E — Surface
+
+Report: local project inbox disposition first (processed count and whether it is clear), then roam disposition: moved (with their new priorities and tags), folded, dropped-as-done, and empty entries swept (count). Then the residue: foreign items (left for their owners, count only) and unowned items (count plus the headings that appear related to this project, for manual claim or prefix). Same "summarize what we kept" shape.
+
+If triaging this batch surfaced a durable, cross-project fact (a reference pointer worth keeping, a pattern worth recording), consider writing it to the agent KB as one =:agent:= node (see =knowledge-base.md=; personal projects only). Skip silently when nothing durable came up — never pad an empty run with a KB line.
+
+** Skip conditions
+
+- No project-local =inbox/= and no =~/org/roam/inbox.org= → silent no-op.
+- Project-local =inbox/= exists but has no pending handoffs → continue to roam scan.
+- No =~/org/roam/inbox.org= after the local inbox check → report the local inbox disposition and stop.
+- No claimed, no related-unowned, and no empty roam entries → report the total, stop.
+- Live org-capture against the roam inbox (capture-guard exit 1) → surface (interactive) or skip-and-self-heal (wrap-up), per core §5.
+
+** Caller integration
+
+*Startup (read-only nudge).* Startup already checks the project-local =inbox/= via =inbox-status= and processes it through process mode when needed. It also reads =~/org/roam/inbox.org= and produces the roam scan summary; one line surfaces: "Roam inbox: N items total, M appear related to this project (K empty entries to sweep) — say 'inbox zero' to file them." Offered as one of the priority options. The empty count rides along so a clean-up-only run still gets offered. Startup never auto-files or auto-sweeps roam items; it counts and offers (the read-only nudge never edits, so empties are reported, not removed, until a real triage runs).
+
+*Wrap-up (Step 3 sub-step).* A sub-step at the start of wrap-up Step 3 (before the cleanup scripts, so imported tasks get linted and ride the wrap commit) delegates here for the claimed set. Skip-fast when nothing matches.
+
+** Deferred: domain-aware routing (future work, multi-project)
+
+v1 handles the single-destination case via the prefix rule. The multi-project parts are deferred until the need is real:
+
+1. *Domain-aware empty-it-all mode.* If rulesets held a description of each project's domain, one run could guess the owner of every item (prefixed or not) and empty the whole inbox at once, delivering each item to its owning project's =inbox/= via =inbox-send= (where that project's process-mode gate still decides whether to file it). Open: where the domain map lives, how confident a guess must be before auto-routing vs surfacing, and whether a low-confidence item stays put.
+2. *Explicit per-project =:INBOX_PREFIX:= as the durable resolver*, replacing basename inference.
+3. *Unowned-item lifecycle* once domain-aware routing exists (no item stays unrouted indefinitely).
+4. *Concurrent push contention* on the shared roam repo: triage now hands its commit + push to =roam-sync=, which already aborts-and-surfaces on a rebase conflict. If multi-machine contention ever makes that abort frequent, =roam-sync= may want a retry-once-after-rebase.
+
+Take these up when the single-destination version is in use and the multi-project pain is concrete.
+
+* Mode: auto inbox zero
+
+A recurring, *interactive* roam check. Trigger phrase: "auto inbox zero" (match before "inbox zero" — the longer phrase wins). On invocation, *ask Craig for the interval* (e.g. 30 min, 2 hours), then drive the loop with =/loop <interval>= running roam mode. It is in-session and interactive by design — each cycle reports what it found and filed.
+
+** Per cycle
+
+1. Run roam mode's scan (Phase A local check + Phase B roam scan), read-only — no =git pull=. The capture-guard still gates any write: use =capture-guard --wait= (core §5) so a transient capture clears itself; if it's still open after the wait, *defer this cycle's roam reconcile to the next cycle* rather than surfacing — the loop cadence is the retry, and the filed items get swept next time. The rare write hands its git to =roam-sync= (roam Phase D).
+2. *Nothing found* → no inbox summary. One heartbeat line: =inbox zero at HH:MM: nothing= (HH:MM local, from =date=) — the silent-until-signal policy, see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=. Nothing else. Keeping a quiet inbox quiet is the whole point.
+3. *Items found* → summarize the found items, file them as tasks (roam Phase C), and *append them to a displayed queue* — the harness task list, via =TaskCreate= — so the queue accumulates across cycles. Then ask: "run this batch next?"
+ - *Yes* → chain into =work-the-backlog.org= as an explicit second step after routing completes: pass it the eligibility query over the queued items (status =TODO= + =:solo:= per the scheme header, priority-ordered), =file-only= mode, paging off, cap 1. The highest-priority eligible candidate runs; the rest wait for the next tick or a later yes.
+ - *No* → they stay queued for a later go.
+ This mode never implements anything itself — routing ends here, and the execution loop lives in =work-the-backlog.org=, its one home.
+4. *Cross-cycle dedup.* Subsequent cycles add only *newly-found* items to the same displayed queue, never re-surfacing what's already there. Dedup against the queue (the =TaskCreate= list), not against what's already been implemented — a find that was queued-but-not-yet-run must not reappear, and one already filed into =todo.org= is dropped by roam Phase C's status check.
+
+A find is always surfaced and filed; execution happens only through the =work-the-backlog.org= chain and waits for Craig's yes. A quiet inbox produces only the =inbox zero at HH:MM: nothing= heartbeat. =auto inbox zero= is inherently in-session because its chain step waits for that yes.
+
+** Fully-unattended pass (=/schedule=) — vNext, not v1
+
+A fully-unattended cron pass (firing while Craig is away) is a *different contract* and is deferred. It can't wait for a yes, so it has to decide up front whether it may mutate =todo.org= and the roam inbox or stays read-only, how a find reaches Craig asynchronously, how dedup state survives across runs that don't share a session, and what session/auth context a cron run carries.
+
+The =/schedule= recipe, once that contract is designed, would look like:
+
+#+begin_src
+/schedule <cron-expression> run inbox.org roam mode read-only, and <surface-mechanism> any finds
+#+end_src
+
+v1 ships only the interactive =/loop= shape above; the unattended contract is logged to =todo.org= for its own design pass. Don't invent the unattended behavior here — route a request for it to that task.
+
+* Common Mistakes
+
+1. *Treating items as orders.* Inbox content is a proposal. The value gate is the rule. Implementing every item without evaluation inflates =todo.org= and trains senders to keep sending noise.
+2. *Filing without applying the value gate.* "File as TODO" is not a default — it's the disposition for proposals that pass the gate but wait. A reject is also a valid answer.
+3. *Filing raw TODOs when the project has a priority scheme.* Core §6 is mandatory when the scheme exists. An un-graded TODO in a project with a legend is a defect.
+4. *Silently deleting a project handoff.* Send a response naming which value-gate question failed. Silent rejection trains the sender to escalate to Craig instead of through the inbox channel.
+5. *Pushing back on a Craig directive only to immediately implement it anyway.* If you genuinely think Craig is wrong, say so and wait. If you don't, just do the work — don't theatre the pushback.
+6. *Skipping the implement-vs-fold-vs-file classification.* Defaulting every accept to "file as TODO" turns the inbox into a queue that flows into =todo.org= without filtering.
+7. *Not propagating value-gate failure to the response.* When you reject a handoff, name *which* gate question failed so the sender can recalibrate, not just resend.
+8. *Forgetting to delete the inbox file after acting.* The local inbox should be empty when process mode ends. Files left behind become noise on the next startup.
+9. *Applying a shared-asset change proposal without the skeptical review.* The value gate alone asks whether to take the change, never whether the change is right, complete, or as simple as it should be. (Worked example: the 2026-06-12 spec-decisions handoff was applied as-is and the after-the-fact review surfaced a lost state, a vacuous gate pass, and an enhancement — all catchable up front.)
+10. *Editing the roam inbox without the capture-guard.* A disk write under a live org-capture wedges the capture (core §5). Guard first, every roam write.
+11. *Auto inbox zero re-surfacing queued items.* The loop must dedup against the displayed queue, not just against what's been implemented — or every cycle re-lists the same un-run finds.
+
+* Living Document
+
+Refine the value gate's three questions if the project's mission sharpens. Tune the per-source rejection-response template if =inbox-send= response loops surface a pattern. Tune the monitor cadence if task-boundary checking proves too frequent or too sparse. Capture the auto-loop interval that worked once the pattern recurs.
+
+If a mode wants real depth — enough that it bloats the core — it can become an =inbox.<mode>.org= plugin under this engine's namespace (the pattern =triage-intake= uses) rather than swelling this file. The principle that inbox items are *ideas to evaluate* is the part that doesn't change.
diff --git a/claude-templates/.ai/workflows/journal-entry.org b/claude-templates/.ai/workflows/journal-entry.org
index 3f476a7..c70dfe8 100644
--- a/claude-templates/.ai/workflows/journal-entry.org
+++ b/claude-templates/.ai/workflows/journal-entry.org
@@ -1,5 +1,5 @@
#+TITLE: Journal Entry Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2025-11-07
* Overview
diff --git a/claude-templates/.ai/workflows/meeting-prep.org b/claude-templates/.ai/workflows/meeting-prep.org
index 162ae30..563328b 100644
--- a/claude-templates/.ai/workflows/meeting-prep.org
+++ b/claude-templates/.ai/workflows/meeting-prep.org
@@ -1,5 +1,5 @@
#+TITLE: Meeting-Prep Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-06-10
* Overview
diff --git a/claude-templates/.ai/workflows/meeting-prep.pre-wire.org b/claude-templates/.ai/workflows/meeting-prep.pre-wire.org
index 6a156c0..3e27c2a 100644
--- a/claude-templates/.ai/workflows/meeting-prep.pre-wire.org
+++ b/claude-templates/.ai/workflows/meeting-prep.pre-wire.org
@@ -1,5 +1,5 @@
#+TITLE: Meeting-Prep — Pre-Wire Method (supporting doc)
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-06-10
Supporting document for the [[file:meeting-prep.org][meeting-prep workflow]]'s Phase 3.5. The workflow carries the condensed, in-flow version of pre-wiring; this file is the full Manager Tools method, kept beside the workflow (same name + =.pre-wire= suffix) so it travels with the workflow. Source casts: "How to Prewire a Meeting" (2007) and "Peer Prewire" (2015).
diff --git a/claude-templates/.ai/workflows/monitor-inbox.org b/claude-templates/.ai/workflows/monitor-inbox.org
deleted file mode 100644
index 1639f3b..0000000
--- a/claude-templates/.ai/workflows/monitor-inbox.org
+++ /dev/null
@@ -1,94 +0,0 @@
-#+TITLE: Monitor Inbox Workflow
-#+AUTHOR: Craig Jennings & Claude
-#+DATE: 2026-05-31
-
-* Overview
-
-Keep the project's =inbox/= responsive: notice handoffs on a cadence, triage each one, decide whether to act now or file it, and reply to the sender. This workflow is the /when, how-often, and act-vs-file/ layer. The per-item disposition mechanics — the value gate, the implement/fold/file classification, the per-source rejection flow — live in [[file:process-inbox.org][process-inbox.org]] and are not duplicated here. Think of it as: monitor-inbox decides /that/ an item gets handled and /how I respond/; process-inbox decides /what disposition/ each item gets.
-
-The gap this closes: handoffs that arrive mid-session used to sit unseen until the user asked or the next startup ran. A handoff the sender can't see being handled trains them to escalate around the inbox channel.
-
-* When to Use This Workflow
-
-Trigger phrases:
-
-- "monitor the inbox" / "watch the inbox"
-- "respond to the handoffs" / "handle the handoffs"
-
-Cadence auto-trigger (the main mechanism — see Cadence below): check at every task boundary during a session, not only when asked.
-
-* Cadence — how often to check
-
-*Default: check at every task boundary.* After finishing a unit of work, before reporting back or asking "what's next," run the cheap status check:
-
-#+begin_src bash
-.ai/scripts/inbox-status -q
-#+end_src
-
-Exit 1 means handoffs are pending — list them (drop =-q=) and process per process-inbox.org. Exit 0 means clean; say nothing. This is one =find=; it costs nothing to run often, and it's the fix for handoffs piling up unseen during long sessions.
-
-*Startup and wrap-up already cover their ends.* Startup Phase C processes a non-empty inbox; the wrap-up sanity check refuses to wrap with unprocessed handoffs. The task-boundary cadence fills the middle.
-
-*Mid-task arrivals.* If a handoff lands while you're mid-task and it's urgent (blocks the current work, or is time-sensitive), surface it right away. Otherwise batch it to the next task boundary so the current work isn't thrashed.
-
-*Unattended / background monitoring (opt-in).* When the user is working elsewhere and wants rulesets handoffs handled without being present, run a polling loop:
-
-#+begin_src
-/loop 15m check the inbox with inbox-status and process any handoffs per process-inbox.org
-#+end_src
-
-This is opt-in, not the default — continuous polling has a cost, and most handoffs aren't urgent. Pick an interval matched to how fast handoffs actually arrive (a burst of cross-project work warrants a tighter loop; a quiet day warrants none).
-
-* The act-vs-file decision
-
-Every accepted handoff (one that clears process-inbox's value gate) is then either acted on now or filed as a task. The rule, and how to surface it:
-
-*Act immediately — and just do it, no asking — when all of these hold:*
-- *Clear* — the action is unambiguous; no design decision or option-choice is needed.
-- *Bounded* — small, finishable this session, ideally a tight file set.
-- *Low-risk and verifiable now* — not a risky change to load-bearing infra (or trivially revertible), and testable/lintable this session.
-- *In-scope and safe* — within this project, not destructive or outward-facing without confirmation, not across a project boundary.
-- *Cheaper than deferring* — doing it now costs less than filing plus re-triaging later.
-
-When you decide to act, queue the work and do it. Don't ask first.
-
-*Exception:* a proposal to change a shared asset (template workflow, rule, skill, synced script) or a substantive convention never qualifies for silent act-now, however clear and bounded it looks — it routes through process-inbox's *Skeptical Review*, which carries its own approval (or, in a no-approvals session, park) step.
-
-*File a task when any of these hold:*
-- It needs a judgment call, a design decision, or an option the user would pick.
-- It's large, multi-session, or sprawls across many files.
-- It's blocked (a dependency, an external thing, the user is away).
-- It's risky enough to want the user's eyes before it lands.
-- It's off the session's active goal and acting now would derail it (file and keep going, unless it's urgent).
-
-When you decide to file, *ask first* — inline numbered options per =interaction.md=, with *filing as option 1 (the recommendation)* and *"do it now" as option 2*:
-
-#+begin_example
-<handoff> wants <X>. My read: file it (needs <reason>).
-
-1. File as a TODO ([#?] :tags:) — Recommended
-2. Do it now instead
-3. Something else
-
-Pick a number.
-#+end_example
-
-*Always ask if you're unsure* which side of the line an item falls on. Decisiveness on clear act-now items is the point of the rule; the ask is for genuine ambiguity and for filing.
-
-* Replying to handoffs
-
-A handoff came from another project's agent (or the user). Close the loop:
-
-- *Accepted and acted on* — send a confirmation to the sender via =inbox-send <sender> --text "..."=, naming what landed and the commit, so they're not left guessing (they can't see this project's git log). =inbox-send= excludes the current project as a target, so a self-sourced item is handled in-session, not sent.
-- *Accepted and filed* — a short confirmation that it's filed and where, so the sender knows it wasn't dropped.
-- *Rejected* — always state the why (which value-gate question failed), per process-inbox's per-source rejection flow.
-
-Cross-project boundary: never act on a file under another project's =.ai/= scope from here — route it back as a handoff (see =cross-project.md=).
-
-* The inbox-status script
-
-=.ai/scripts/inbox-status= lists unprocessed handoffs and exits nonzero when any are pending. Exclusions match the wrap-up sanity check (=.gitkeep=, =lint-followups.org=, =PROCESSED-*=). Exit 0 = clean, 1 = pending, 2 = no inbox/ or bad usage. Use =-q= for the count-only form the cadence check calls.
-
-* Living Document
-
-Tune the cadence if task-boundary checking proves too frequent or too sparse in practice. Refine the act-vs-file criteria as edge cases recur. If the background-monitor loop becomes a common pattern, capture the interval that worked. The decision rule itself — act-now is silent, filing asks with file-as-option-1, ambiguity asks — is the stable core (set by Craig, 2026-05-30).
diff --git a/claude-templates/.ai/workflows/no-approvals.org b/claude-templates/.ai/workflows/no-approvals.org
index 1efce82..6b5c7fa 100644
--- a/claude-templates/.ai/workflows/no-approvals.org
+++ b/claude-templates/.ai/workflows/no-approvals.org
@@ -1,5 +1,5 @@
#+TITLE: No-Approvals Mode
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-28
* Overview
@@ -22,6 +22,8 @@ Craig activates the mode with any of:
- Queuing several tasks in =todo.org= followed by any phrase above
- Any equivalent phrasing that signals he doesn't want to be re-asked between items
+*Not this mode:* any phrase containing "speedrun" ("speedrun", "no approvals speedrun") routes to =work-the-backlog.org='s no-approvals speedrun preset — an autonomous batch over an explicit ordered task set, with a pre-flight Q&A, autonomous commits, always-push, and an end-of-set page. This mode is the general interaction-gate suspension for whatever work is already underway; the speedrun is the dedicated backlog-batch workflow.
+
Mode resets when:
- Craig says approvals are back on
@@ -33,7 +35,7 @@ Mode resets when:
The interaction gates that step the workflow back to Craig for an "OK to proceed?" check:
-- The commit-message gate in =commits.md= Step 2. Print the final commit message inline, then commit immediately. No "approve / request changes / open in editor" prompt.
+- The commit-message gate in the =publish= skill, Step 2. Print the final commit message inline, then commit immediately. No "approve / request changes / open in editor" prompt.
- The PR-description gate. Print the final body, then create the PR.
- The PR-review-reply gate. Print the final reply, then post.
- "Ready to start?" / "Plan looks like X, proceed?" gates before implementation work begins.
@@ -44,10 +46,10 @@ The interaction gates that step the workflow back to Craig for an "OK to proceed
The engineering-discipline gates protect quality, not Craig's interaction time. They remain in force:
-- =/review-code= against the staged diff before every commit. Critical and Important findings still block. Minor findings still surface. No "proceed anyway" override unless Craig has given it explicitly for this batch.
+- =/review-code= against the staged diff before every commit, dispatched as an isolated adversarial reviewer per the =publish= skill's Step 1 — no-approvals removes *interaction* gates, never the isolation. Critical and Important findings still block, and the re-review loop still runs to approval. Minor findings still surface. No "proceed anyway" override unless Craig has given it explicitly for this batch. If the review can't reach approval — three rounds, a recurring finding, or a =Needs Discussion= verdict — that is a genuine question: park the item per step 4 and move to the next one rather than committing past a standing finding.
- =/voice personal= on every publish artifact (commit messages, PR titles + bodies, PR review comments). The full pattern walk happens. The printed result just doesn't wait for approval.
- The full test suite + lint + compile before commit (per =verification.md=).
-- Fetch-and-reconcile in =commits.md= Step 0.
+- Fetch-and-reconcile in the =publish= skill, Step 0.
- Session Log updates per =protocols.org=. Every state-mutating turn writes to =.ai/session-context.org= before the closing message. The log is the crash-recovery anchor while Craig is away. Missing entries lose work.
- Subagent review-gate cadence (=subagents.md=). Review each subagent's output before the next dispatch.
- Destructive or irreversible operations per =CLAUDE.md='s "Executing actions with care": force-push, =rm -rf=, dropping a column, dropping a branch, package removal. These need explicit consent regardless of mode. No-approvals is for *interaction* gates, not destructive-action consent.
@@ -68,7 +70,7 @@ For each item:
- Do the work.
- Update the Session Log per the rules in =protocols.org=.
-- Before any commit: run =/review-code= against the staged diff. Surface Critical and Important findings inline; fix them and re-review until clean. Minor findings show but don't block.
+- Before any commit: dispatch the isolated adversarial reviewer per the =publish= skill's Step 1 — never review your own staged diff inline. Surface Critical and Important findings; fix them and send the updated diff back to the *same* reviewer until it approves. Minor findings show but don't block and never earn another round. If the review can't reach approval — three rounds, a finding that recurs after being reported fixed, or a =Needs Discussion= verdict — park the item per step 4 with the standing findings and move on; don't commit past a blocking finding.
- Draft the commit message. Run =/voice personal= (the skill, or walk the patterns inline if unavailable). Print the final message inline before committing so the log shows it.
- Commit and push.
- One-line status between items ("Task X done, on to Y.") so Craig knows what's happening when he checks back in.
diff --git a/claude-templates/.ai/workflows/open-tasks.org b/claude-templates/.ai/workflows/open-tasks.org
index fe782d6..205d95c 100644
--- a/claude-templates/.ai/workflows/open-tasks.org
+++ b/claude-templates/.ai/workflows/open-tasks.org
@@ -1,5 +1,5 @@
#+TITLE: Open Tasks Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-04-25
* Overview
@@ -23,15 +23,16 @@ Don't route "task review" / "review tasks" here — those trigger the hygiene ha
* Phase A: Data Gathering (both modes)
-** Phase A pre-step — archive any freshly-DONE tasks
+** Phase A pre-step — normalize freshly-closed tasks
-Before reading =todo.org=, run the cleanup script's archive-done sweep so completed level-2 subtrees move from =* $Project Open Work= to =* $Project Resolved=:
+Before reading =todo.org=, run two cleanup sweeps so the read reflects current state. First convert any done sub-tasks to dated entries, then archive completed level-2 subtrees from =* $Project Open Work= to =* $Project Resolved=:
#+begin_src bash
+emacs --batch -q -l .ai/scripts/todo-cleanup.el --convert-subtasks todo.org
emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done todo.org
#+end_src
-Costs a few hundred milliseconds. Without it, a task that completed earlier in the session sits as =** DONE= under Open Work until the next =clean-todo= or wrap-up pass, and Next Mode would surface it as a "what's next" candidate. The sweep makes Phase A's read of =todo.org= reflect current state.
+Costs a few hundred milliseconds. Without the archive sweep, a task that completed earlier in the session sits as =** DONE= under Open Work until the next =clean-todo= or wrap-up pass, and Next Mode would surface it as a "what's next" candidate. The convert sweep runs first so a completed parent's sub-tasks are already dated when it archives; it also keeps interactive level-3 closes from lingering as DONE keywords. Together they make Phase A's read of =todo.org= reflect current state.
Skip the sweep if the workflow is invoked in an explicit read-only or dry-run context. Default is to run it.
@@ -176,6 +177,10 @@ Next Mode answers two questions in one output: "what matters most right now?" (t
Apply the prioritization cascade in order. Stop at the first matching step. This is the importance/urgency answer.
+*Exclude blocked tasks.* A task tagged =:blocked:= has an unmet cross-project dependency (its body names the project and the work owed, per =todo-format.md=). It can't be worked until that other project delivers, so it is *never* the cascade recommendation — skip it at every cascade step below. Blocked tasks are surfaced on their own in Step 3 so the stalled dependency stays visible instead of silently dropping out of view.
+
+*Surface blocking tasks first.* The mirror of the above: a task tagged =:blocker:= is holding up work in *another* project (its body names which project and what's owed, per =todo-format.md=). Clearing it unblocks that project, so it carries borrowed urgency — surface it at the *top* of the cascade recommendation regardless of its own priority cookie, ahead of the normal In-Progress / deadline / priority order. When several =:blocker:= tasks exist, lead with the one blocking the most, or the longest. This is the "do the thing that unblocks someone else first" rule; a =:blocker:= task left at its own low priority is exactly how a cross-project dependency stalls.
+
**** 1. In-Progress Tasks
- Look for tasks marked =DOING= or partially complete.
- *If found:* Recommend that task (always finish what's started).
@@ -228,11 +233,22 @@ Within each row, pick a single task per the same-level tie-breakers above (block
The friction filter is the override path. When the cascade winner is partially blocked, hardware-dependent, or simply too large for the user's current state, one of the friction rows is what they pick instead.
+*** Step 3 — Blocked-on-other-projects surface
+
+Independently of the cascade and the friction filter, collect every open task tagged =:blocked:=. These are tasks this project can't advance until another project delivers; surfacing them keeps a cross-project dependency from rotting at low priority on the other side — the exact failure the tag exists to prevent (a blocked task whose blocker is a =[#D]= in another project sits forever otherwise).
+
+For each blocked task, read its body for the blocking project and what's owed, and present one line: the task, the blocking project, and what that project owes. Then offer — per blocked task — to nudge the blocker: an =inbox-send <project> --text= note naming what's needed and why it's blocking, so the dependency gets attention in the project that owns it. Don't send without the user's go.
+
+If no =:blocked:= tasks exist, omit this surface entirely (the common case).
+
*** Output Format
-Pair the cascade recommendation with the friction block beneath it. Recommendation-at-item-1 convention applies to the friction rows — quick+solo first, since it's the strongest low-friction pick.
+Pair the cascade recommendation with the friction block beneath it, and the blocked-on-other-projects surface (Step 3) beneath that when any blocked task exists. Recommendation-at-item-1 convention applies to the friction rows — quick+solo first, since it's the strongest low-friction pick.
#+begin_example
+Unblocks other projects (do these first):
+- ai-term wrap-teardown companion — :blocker:, unblocks rulesets (the three ai-term functions)
+
Cascade recommendation (importance/urgency):
- Fix org-noter reliability — [#A], Method 1, 8/18 complete, blocks daily reading/annotation
@@ -240,17 +256,25 @@ If you want lower friction instead:
1. Quick + solo: Bump linter config — [#C] :quick:solo:, ~15 min
2. Quick: Confirm new dirvish setup — [#B] :quick:, needs your eye
3. Solo: Refactor config-utilities — [#B] :solo:, bounded but multi-hour
+
+Blocked on other projects (can't advance until the blocker delivers):
+- Wrap-teardown feature — blocked by emacsd: ai-term companion functions — nudge?
#+end_example
+The =:blocker:= surface sits at the very top — clearing one of those is the highest-leverage thing on the list, since it frees work in another project. Omit it when no =:blocker:= task exists (the common case).
+
Include for each row:
- Task name / description.
- Priority + tag cluster.
- One-line reasoning. For the cascade row, name which cascade step matched. For friction rows, an effort hint when one is obvious.
- Progress indicator (for V2MOM-structured todos) on the cascade row only.
+- For a =:blocker:= row: the project it unblocks and what's owed (from the task body).
+- For a blocked row: the blocking project and what it owes (from the task body), plus the nudge offer.
**** Edge cases
- *Empty friction block.* If no =:quick:= or =:solo:= tagged tasks exist in the open set, omit the friction block entirely. Present only the cascade recommendation.
+- *No =:blocker:= tasks.* Omit the "Unblocks other projects" surface entirely (the common case) — show it only when a task carries the =:blocker:= tag.
- *Dedupe.* If the cascade recommendation IS the same task as one of the friction rows (e.g. it's =:quick:solo:= and also won the cascade), show it once at the top with both labels. Don't list it twice.
- *Decline behavior.* If the user declines the cascade recommendation, drop straight to the friction block as the natural next prompt. Do not fall through to lower-cascade-tier tasks; the friction filter IS the override.
diff --git a/claude-templates/.ai/workflows/page-me.org b/claude-templates/.ai/workflows/page-me.org
index 607ed51..7a3b792 100644
--- a/claude-templates/.ai/workflows/page-me.org
+++ b/claude-templates/.ai/workflows/page-me.org
@@ -1,21 +1,29 @@
#+TITLE: Page Me Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-01-31
#+UPDATED: 2026-02-27
* Overview
-This workflow enables Claude to set timers and alarms that reliably notify Craig, even if the terminal session ends or is accidentally closed. Notifications are distinctive (audible + visual with alarm icon) and persist until manually dismissed.
+This workflow enables Claude to set timers and alarms that reliably notify Craig, even if the terminal session ends or is accidentally closed. Notifications are distinctive (audible + visual with the blue info icon) and persist until manually dismissed.
-Uses the =notify= command (alarm type) for consistent notifications across all AI workflows.
+Uses the =notify= command (info type) for consistent notifications across all AI workflows. Info-level on purpose: the earlier alarm styling read as all-red urgency, and Craig's verdict was that a page "should be a persistent info notification" — noticeable, never crash-scary (2026-07-02).
* Trigger Phrase
Craig says *"page me"* (or variations like "page me in 10 minutes", "page me at 3pm").
-The word "page" is the trigger for this workflow. It means: set a timed notification.
+The word "page" is the trigger for this workflow. It means: set a timed notification on the *desktop* channel (=notify=).
-Previously called "set-alarm" -- renamed to "page-me" for a distinctive, short trigger phrase that won't collide with common words like "remind" or "alert."
+Two sibling triggers pick a different channel; the timed =at= machinery below is identical for all three, only the fired command changes:
+
+- *"page me"* — desktop =notify= (this workflow's default).
+- *"text me"* — a Signal push to Craig's phone via =agent-text= (the away channel).
+- *"text and page me"* — both, for when he might be either place.
+
+Scope the triggers to the reflexive "me": "page me" and "text me", not a bare "page" or "text" in prose. The full channel vocabulary lives in protocols.org "Reaching Craig".
+
+"page" was chosen (renamed from the old "set-alarm") for a distinctive, short trigger that won't collide with common words like "remind" or "alert".
* Problem We're Solving
@@ -63,8 +71,8 @@ Craig tells Claude when and why:
Claude schedules the alarm using the =at= daemon with =notify=:
#+begin_src bash
-echo "notify alarm 'Page' 'Time to call the dentist' --persist" | at 3:30pm
-echo "notify alarm 'Page' 'Meeting starts' --persist" | at now + 45 minutes
+echo "notify info 'Page' 'Time to call the dentist' --persist" | at 3:30pm
+echo "notify info 'Page' 'Meeting starts' --persist" | at now + 45 minutes
#+end_src
The =at= daemon:
@@ -89,30 +97,43 @@ Craig dismisses the notification and acts on it.
** Setting Alarms
-Use the =at= daemon to schedule a =notify alarm= command:
+Use the =at= daemon to schedule a =notify info= command:
#+begin_src bash
# Schedule for specific time
-echo "notify alarm 'Page' 'Meeting starts' --persist" | at 3:30pm
+echo "notify info 'Page' 'Meeting starts' --persist" | at 3:30pm
# Schedule for relative time
-echo "notify alarm 'Page' 'Check the build' --persist" | at now + 30 minutes
+echo "notify info 'Page' 'Check the build' --persist" | at now + 30 minutes
# Schedule for tomorrow
-echo "notify alarm 'Page' 'Call the dentist' --persist" | at 3:30pm tomorrow
+echo "notify info 'Page' 'Call the dentist' --persist" | at 3:30pm tomorrow
#+end_src
** Notification System
-Uses the =notify= command with the =alarm= type. The =notify= command provides 8 notification types with matching icons and sounds.
+Uses the =notify= command with the =info= type. The =notify= command provides 8 notification types with matching icons and sounds.
#+begin_src bash
-# Immediate alarm notification (for testing)
-notify alarm "Page" "Your message here" --persist
+# Immediate page notification (for testing)
+notify info "Page" "Your message here" --persist
#+end_src
The =--persist= flag keeps the notification on screen until manually dismissed. All page-me notifications should use =--persist= by default.
+** Texting Craig's phone (the "text me" channel)
+
+The timed =notify= alarm above is the desktop channel. When Craig says "text me" (or a run expects him away from the machine), use =agent-text= instead, a Signal push to his phone from any machine or agent runtime:
+
+#+begin_src bash
+agent-text "Build finished, ready for your eyes"
+
+# Timed phone message: same at-daemon pattern, different channel
+echo "agent-text 'Meeting starts in 5'" | at 3:25pm
+#+end_src
+
+Channel selection and the mechanics live in protocols.org "Reaching Craig". On "text and page me", fire both: the desktop notification persists for whenever he returns, the phone push reaches him now.
+
** Managing Alarms
#+begin_src bash
@@ -139,10 +160,10 @@ The alarm must fire. Use the =at= daemon which is designed for exactly this purp
Simple invocation - Claude runs one command. No complex setup required per alarm.
** Fail Audibly
-If the alarm fails to schedule, report the error clearly. Don't fail silently.
+If the page fails to schedule, report the error clearly. Don't fail silently.
** Testable
-The =notify alarm= command can be called directly to verify notifications work without waiting for a timer.
+The =notify info= command can be called directly to verify notifications work without waiting for a timer.
** Non-Alarming
Use normal urgency, not critical. The notification should be noticeable but not imply something has gone horribly wrong.
diff --git a/claude-templates/.ai/workflows/process-inbox.org b/claude-templates/.ai/workflows/process-inbox.org
deleted file mode 100644
index 86df4c2..0000000
--- a/claude-templates/.ai/workflows/process-inbox.org
+++ /dev/null
@@ -1,215 +0,0 @@
-#+TITLE: Process Inbox Workflow
-#+AUTHOR: Craig Jennings & Claude
-#+DATE: 2026-05-28
-
-* Overview
-
-Inbox items are *ideas to evaluate*, not orders to execute. They arrive from Craig (typed directives saved as files), from other projects (handoffs via =inbox-send=), and from scripts/automated systems. Each is a proposal. An item earns a place in =todo.org= or git history only when it passes the value gate: it advances an existing task, improves how the project works, or serves the project's stated mission.
-
-The workflow is the disposition discipline. Read each item, evaluate honestly, apply the decision, then notify the sender if it was a project handoff and you're rejecting. Silent rejection on a handoff is worse than no reply.
-
-* When to Use This Workflow
-
-User triggers:
-
-- "process inbox" / "process the inbox"
-- "handle the inbox"
-- "what's in inbox" / "what's in the inbox"
-- "let's clear the inbox" / "let's process the inbox items"
-
-Auto-invocation:
-
-- Startup =Phase C step 2= delegates here when the inbox is non-empty. Don't ask Craig — just run it.
-
-Do *not* invoke this for inbox items that are clearly out-of-scope for the project — those are cross-project routing problems, handled per the cross-project boundary rule in =protocols.org=.
-
-* The Value Gate
-
-Every inbox item passes through three questions. One *yes* is enough to accept.
-
-1. *Does it advance an existing TODO?* Look up by topic in =todo.org='s open work. If the item extends a filed task, fold it in. If it implements a filed task, do the work.
-2. *Does it improve how the project works?* Architecture cleanup, workflow refinement, tooling, rule hygiene, drift detection — anything that makes the project itself more effective.
-3. *Does it serve the project's stated mission?* Read =notes.org= *Project-Specific Context* if the mission isn't obvious from the working directory and current task. The item should advance that mission, not orbit it.
-
-Three *no*s means reject. The rejection isn't lazy — an idea that doesn't help any current task, doesn't improve the system, and doesn't serve the mission is genuine noise, and accepting it inflates =todo.org= without payoff.
-
-* The Skeptical Review (change proposals)
-
-The value gate decides whether an item is worth taking. This review decides whether the proposed change is *right*. It applies to any item proposing a change to shared assets — template workflows, rules, skills, scripts, anything synced to consuming projects — and to any substantive convention change. FYIs, replies, and routine task filings skip it.
-
-Approach the file with curiosity and skepticism. It arrived from one project's context; you're evaluating it for all of them. Work through, in writing:
-
-1. Does this make sense for *all* consuming projects, or just the sender's situation?
-2. Does it conflict with any existing instruction — workflows, skills, rules, protocols, CLAUDE.md?
-3. How does it change a common activity Craig performs — better, worse, or differently than the sender assumed?
-4. Can it be enhanced to be more effective than as proposed?
-5. Should it be simpler?
-6. Plus at least three more questions specific to this change — e.g. what breaks for artifacts already using the old shape, what tooling interacts with it, what's underspecified, what does the sender's worked example not exercise?
-
-Output: a short summary of the thinking and a recommendation (accept as-is / accept with named changes / reject), surfaced to Craig for approval before applying.
-
-** In a no-approvals session: defer and stage
-
-Behavior-changing proposals still don't self-apply when Craig has put the session in no-approvals mode. Run the review, prepare the edits in =working/<task-slug>/= (a patch file or the worked-out diff), file a =[#B]= VERIFY carrying the decision package, and reply to the sender that it's parked. The sender's local stopgap (per =cross-project.md='s propagation process) means the delay costs nothing — the canonical update is about durability, not speed.
-
-Wording-only fixes — no consuming project acts differently — may proceed even then, logged in the session log.
-
-The VERIFY shape (top-level, =[#B]= so startup's A/B surfacing catches it; no =SCHEDULED= unless the proposal names a real deadline):
-
-#+begin_example
-** VERIFY [#B] Parked: <proposal topic> (from <sender>)
-What arrived: <one line — what the handoff proposes>.
-Recommendation: <accept as-is / accept with changes / reject> — <2-3 line
-skeptical-review summary: what's right, what to change, what was checked>.
-Prepared diff: [[file:working/<slug>/proposed.diff]] — apply is mechanical on
-your go.
-Say "approve the parked <topic>" (or adjust / reject) and it gets applied.
-#+end_example
-
-The full question-battery answers live in the session log and the =working/= dir, not the task body — the body carries the conclusion, with the trail one link away.
-
-* Phase A — Inventory (one parallel batch)
-
-Issue these reads in one parallel batch:
-
-1. List =inbox/= excluding =.gitkeep= and =PROCESSED-*= prefixes (use =\ls -la inbox/= per the protocols.org exa-alias note).
-2. Read =notes.org= *Project-Specific Context* if mission isn't already loaded in the session.
-3. Read =todo.org='s top-of-file priority scheme if present (look for a =* Priority and Tag Scheme= section or similar between the intro and the first =* <Project> Open Work= header).
-
-For each inbox file, parse the filename for sender. Two common patterns:
-
-- =YYYY-MM-DD-HHMM-from-<sender>-<topic>.<ext>= — from another project via =inbox-send=.
-- =<topic>.org= — typically from Craig directly, or from a script.
-
-Note the file type. =.eml= files need the extract script (not raw =Read=):
-
-#+begin_src bash
-# View mode
-python3 .ai/scripts/eml-view-and-extract-attachments.py inbox/<file>.eml
-
-# Pipeline mode (extract attachments to a directory)
-python3 .ai/scripts/eml-view-and-extract-attachments.py inbox/<file>.eml --output-dir assets/<target>/
-#+end_src
-
-Everything else, read directly.
-
-* Phase B — Evaluate each item
-
-For each inbox file:
-
-1. *Read it.* For substantive proposals (org files with TODO entries, design notes, multi-section docs), the full read is the right move. For short FYIs and one-liner asks, skim.
-2. *Identify the shape.* Is it an instruction, a question, a proposal, an FYI, or a handoff? Shapes guide disposition.
-3. *Apply the value gate.* Three questions above. One yes → candidate accept. Three nos → candidate reject.
-4. *Run the Skeptical Review* (section above) on any accepted item that proposes a shared-asset or convention change, before classifying. Its summary + recommendation rides along to Phase C; in a no-approvals session its defer-and-stage path replaces implement-now for behavior-changing proposals.
-5. *Within accept, classify:*
- - *Implement now* — small, scoped, clear, no design call required. The work is the disposition.
- - *Fold into existing TODO* — the item extends a task already filed; update the TODO body and link the inbox content if substantive.
- - *File as TODO* — substantive but waits, or needs design/triage before implementation.
-6. *Within reject, classify by source:*
- - *From Craig* — push back honestly in chat. State why you won't implement. Offer the conditions under which you would, if any. Wait for Craig to override or accept.
- - *From another project* — write a response file naming the rejection rationale and (optionally) the condition under which you'd reconsider. Deliver via =inbox-send <sender> --file <response>= per the cross-project handoff convention.
- - *From a script or automated system* — just delete; no notification needed.
-
-* Phase B.1 — Priority-scheme check
-
-This gates Phase C filing when there are accept-and-file items.
-
-Check whether =todo.org= has a top-of-file priority scheme (an explicit legend defining =[#A]= through =[#D]= semantics and mandatory/optional tag conventions).
-
-- *Scheme present* — file new TODOs per the scheme. Every TODO gets a priority cookie matching the legend's rules, the mandatory type tag, and any applicable effort/autonomy tags.
-- *Scheme absent* — surface one sentence: "This project has no priority scheme. We should adopt one before filing the new TODOs from this inbox pass — want me to propose one based on the rulesets scheme?" If Craig says yes, do that first (the =/research-priority-scheme= research subagent pattern in rulesets is the reference). If Craig says no, file the TODOs without grading but flag in the commit message that they're un-prioritized pending a scheme.
-
-The point is to avoid adding ungraded =TODO= entries to a project that's never agreed on what =[#A]= means.
-
-* Phase C — Surface dispositions
-
-Numbered options inline per =interaction.md= (no popup). Recommendation at item 1.
-
-Batch trivial items (one-line rejections of script noise, obvious file-as-TODO accepts where the scheme is already settled) into a single confirm-all prompt. Walk substantive items one at a time so the decision is visible.
-
-Per-item template:
-
-#+begin_example
-<filename> from <sender>: <one-line summary>
-Value-gate read: <yes/no on each of the three questions, one phrase each>
-Disposition recommendation: <implement / fold into <TODO> / file [#X] :tags: / reject>
-
-1. <recommendation as item 1>
-2. <alternative>
-3. Defer — leave in inbox under PROCESSED-<topic>.<ext> until <condition>
-4. Something else
-#+end_example
-
-For items that went through the Skeptical Review, the surfaced disposition includes its summary + recommendation, and approval here is what authorizes the apply. In a no-approvals session those items are reported as parked (the =[#B]= VERIFY) rather than surfaced for live approval.
-
-For pure FYIs that need no action, surface as a single line and recommend delete-with-acknowledgment.
-
-* Phase D — Apply
-
-Apply each disposition. The flow is autonomous past Craig's Phase C approval.
-
-** Implement-now
-
-Do the work. Commit per the project's commit flow. Delete the inbox file. The commit message references the inbox item by filename so the provenance lands in =git log=.
-
-** Fold into existing TODO
-
-Update the parent TODO's body with a dated reconciliation sub-entry per =todo-format.md= (=*** YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <what landed>=). Move substantive content to =docs/design/<date>-<topic>.<ext>= if it's worth keeping; reference from the TODO body. Delete the inbox file.
-
-** File as TODO
-
-Add the TODO under =* <Project> Open Work= with priority + tags per Phase B.1. Body summarizes the proposal and links the inbox content if it's been moved to =docs/design/=. Delete the inbox file (or move it to =docs/design/= first if the content survives).
-
-** Reject from Craig
-
-State the rejection in chat clearly: what you won't implement, why, and the conditions (if any) under which you would. Wait for Craig's override or acknowledgment. The inbox file stays until Craig confirms — if he overrides, re-enter Phase D as accept; if he acknowledges the rejection, delete the file.
-
-** Reject from another project (handoff)
-
-Write the response file at =/tmp/inbox-response-<topic>.org=. Contents:
-
-- Heading naming the original handoff and date
-- One paragraph: the rejection rationale (which value-gate question failed and why)
-- One paragraph: the condition under which you'd reconsider, if such a condition exists. If the answer is "never, this misreads the project's mission," say so directly.
-
-Deliver via =inbox-send <sender> --file /tmp/inbox-response-<topic>.org=. The =inbox-send= script (per =cross-project.md=) handles the from-prefix, date stamp, and target inbox path.
-
-Delete the local inbox file after the response lands in the sender's inbox.
-
-** Reject from script or automated system
-
-Just delete. No notification.
-
-** Defer
-
-Rename in place to =inbox/PROCESSED-<original-filename>= and add a brief comment line at the top: =# Deferred YYYY-MM-DD: <condition>=. Don't accumulate deferred items indefinitely — sweep them on a future =process-inbox= run when the condition is met or the deferral has aged out.
-
-** Park (Skeptical Review in a no-approvals session)
-
-Move the proposal file into =working/<task-slug>/= alongside the prepared diff, file the =[#B]= VERIFY per the Skeptical Review section, reply to the sender that it's parked for Craig's review, and delete the inbox file. On Craig's approval the apply is mechanical: apply the prepared edits, run the normal verify-and-publish flow, rewrite the VERIFY to a dated log entry per =todo-format.md=, and send the sender the acceptance reply. On rejection, the reject-from-another-project flow above runs unchanged.
-
-* Phase E — Close out
-
-Verify =inbox/= is empty (excluding =.gitkeep= and any intentional =PROCESSED-*= files). Run =\ls -la inbox/= and confirm.
-
-Update the session log per =protocols.org= with one short paragraph summarizing this pass: count processed, count accepted (implement/fold/file split), count rejected (Craig/handoff/script split), and the commit SHA if a commit landed.
-
-Stamp =:LAST_INBOX_PROCESS:= in =notes.org='s *Workflow State* section if it exists, so future workflows that gate on freshness can read it. Same format as =:LAST_AUDIT:= (=YYYY-MM-DD=).
-
-* Common Mistakes
-
-1. *Treating items as orders.* Inbox content is a proposal. The value gate is the rule. Implementing every item without evaluation inflates =todo.org= and trains senders to keep sending noise.
-2. *Filing without applying the value gate.* "File as TODO" is not a default — it's the disposition for proposals that pass the gate but wait. A reject is also a valid file-as-TODO answer to nothing.
-3. *Filing raw TODOs when the project has a priority scheme.* Phase B.1 is mandatory when the scheme exists. An un-graded TODO in a project with a legend is a defect.
-4. *Silently deleting a project handoff.* Send a response. The sender's next session sees the response in their inbox and learns the rejection rationale. Silent rejection trains the sender to escalate to Craig instead of through the inbox channel.
-5. *Pushing back on a Craig directive only to immediately implement it anyway.* If you genuinely think Craig is wrong, say so and wait for his call. If you don't, just do the work — don't theatre the pushback.
-6. *Skipping the implement-vs-fold-vs-file classification.* Defaulting every accept to "file as TODO" turns the inbox into a queue that flows into =todo.org= without filtering. Small, scoped, clear items get implemented now; substantive proposals get filed; extensions to existing work get folded.
-7. *Not propagating value-gate failure to the response.* When you reject a handoff, the response should name *which* gate question failed (advances no current task / doesn't improve the project / doesn't serve the mission) so the sender can recalibrate, not just resend.
-8. *Forgetting to delete the inbox file after acting.* The inbox should be empty when this workflow ends. Files left behind become noise on the next startup.
-9. *Applying a shared-asset change proposal without the Skeptical Review.* The value gate alone asks whether to take the change, never whether the change is right, complete, or as simple as it should be. A proposal that's clear and bounded can still carry a design gap — the review is where that surfaces, before the change syncs to every consuming project. (Worked example: the 2026-06-12 spec-decisions handoff was applied as-is and the after-the-fact review surfaced a lost state, a vacuous gate pass, and an enhancement — all catchable up front.)
-
-* Living Document
-
-Refine the value gate's three questions if the project's mission sharpens. Tune the per-source rejection-response template if =inbox-send= response loops surface a pattern. Add new auto-classification shortcuts if certain item shapes (e.g. routine FYIs from a script) become common.
-
-The workflow is shaped by use. The principle that inbox items are *ideas to evaluate* is the part that doesn't change.
diff --git a/claude-templates/.ai/workflows/process-meeting-transcript.org b/claude-templates/.ai/workflows/process-meeting-transcript.org
index 4dd340f..d0806ad 100644
--- a/claude-templates/.ai/workflows/process-meeting-transcript.org
+++ b/claude-templates/.ai/workflows/process-meeting-transcript.org
@@ -1,5 +1,5 @@
#+TITLE: Process Meeting Transcript Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-03
* Overview
@@ -10,16 +10,16 @@ This workflow defines the process for processing meeting recordings from start t
Trigger this workflow when:
- Craig says "process the transcript" or "process the recording" or similar
-- New recording files (.mkv) appear in ~/sync/recordings/ after meetings
+- New recording files (.mkv, .m4a, or .flac) appear in ~/sync/recordings/ after meetings
- Craig wants to process meeting recordings into labeled transcripts
* Prerequisites
-- Recording file(s) exist in ~/sync/recordings/ (*.mkv)
+- Recording file(s) exist in ~/sync/recordings/ (*.mkv, *.m4a, or *.flac)
- Calendar files available at ~/.emacs.d/data/*cal.org for meeting titles
- AssemblyAI transcription script at ~/.emacs.d/scripts/assemblyai-transcribe
- AssemblyAI API key stored in ~/.authinfo.gpg (machine api.assemblyai.com)
-- ffmpeg available for audio extraction
+- ffmpeg available for audio extraction (video .mkv only; .m4a and .flac skip extraction)
* The Workflow
@@ -43,13 +43,13 @@ Classification is per recording, not per session — a single run can carry this
Find and match recording files with calendar events. *Run sub-steps 1 and 3 (recording list + calendar dump) as a single parallel batch* — they're independent. Sub-step 2 (parse timestamps) and sub-step 4 (matching) work from those two outputs in-memory, so they're sequential after the batch.
-1. **List recordings:** Find all recording files in ~/sync/recordings/ (video .mkv or audio-only .m4a)
+1. **List recordings:** Find all recording files in ~/sync/recordings/ (video .mkv or audio-only .m4a / .flac)
#+begin_src bash
- ls -la ~/sync/recordings/*.mkv ~/sync/recordings/*.m4a 2>/dev/null
+ ls -la ~/sync/recordings/*.mkv ~/sync/recordings/*.m4a ~/sync/recordings/*.flac 2>/dev/null
#+end_src
- Audio-only recordings (.m4a) are used when no screen content is expected. These skip Step 3 (audio extraction) since they're already in a transcribable format.
+ Audio-only recordings (.m4a or lossless .flac) are used when no screen content is expected. These skip Step 3 (audio extraction) since they're already in a transcribable format. FLAC is the recorder's current audio format — its frames are self-contained, so an interrupted recording still decodes.
-2. **Extract timestamps:** Parse date/time from each filename (format: YYYY-MM-DD-HH-MM-SS.mkv or .m4a)
+2. **Extract timestamps:** Parse date/time from each filename (format: YYYY-MM-DD-HH-MM-SS.mkv, .m4a, or .flac)
3. **Match with calendar:** Check ~/.emacs.d/data/*cal.org for meetings at those times
#+begin_src bash
@@ -74,7 +74,7 @@ Per =cross-project.md=: transcribe and label here, then deliver the *labeled tra
** Step 3: Extract Audio (video recordings only)
-*Skip this step for .m4a files* — they are already audio and can go directly to transcription.
+*Skip this step for .m4a and .flac files* — they are already audio and can go directly to transcription.
For .mkv video recordings, extract audio for transcription:
@@ -96,11 +96,11 @@ Output: /tmp/FILENAME.m4a (temporary, deleted after transcription)
#+begin_src bash
# For .mkv files (audio was extracted to /tmp/):
~/.emacs.d/scripts/assemblyai-transcribe /tmp/FILENAME.m4a > ~/sync/recordings/FILENAME.txt
- # For .m4a files (transcribe directly):
+ # For .m4a or .flac files (transcribe directly — AssemblyAI accepts both natively):
~/.emacs.d/scripts/assemblyai-transcribe ~/sync/recordings/FILENAME.m4a > ~/sync/recordings/FILENAME.txt
#+end_src
-2. **Clean up:** Delete intermediate .m4a file after successful transcription (only for .mkv extractions — do NOT delete original .m4a recordings)
+2. **Clean up:** Delete intermediate .m4a file after successful transcription (only for .mkv extractions — do NOT delete original .m4a or .flac recordings)
#+begin_src bash
rm /tmp/FILENAME.m4a
#+end_src
@@ -181,13 +181,13 @@ Present the speaker identification table to Craig for confirmation:
** Step 9: Copy Recording to Meetings Folder
-1. Ensure engagement meetings folder exists and patterns are in .gitignore (~*/meetings/*.mkv~ and ~*/meetings/*.m4a~)
+1. Ensure engagement meetings folder exists and patterns are in .gitignore (~*/meetings/*.mkv~, ~*/meetings/*.m4a~, and ~*/meetings/*.flac~)
2. Copy the recording file with descriptive name:
#+begin_src bash
# Video recordings:
cp ~/sync/recordings/YYYY-MM-DD-HH-MM-SS.mkv {engagement}/meetings/YYYY-MM-DD_HH-MM-meeting-name.mkv
- # Audio-only recordings:
+ # Audio-only recordings (.m4a or .flac — copy with the original extension):
cp ~/sync/recordings/YYYY-MM-DD-HH-MM-SS.m4a {engagement}/meetings/YYYY-MM-DD_HH-MM-meeting-name.m4a
#+end_src
Example: ~deepsat/meetings/2026-02-03_11-02-standup-ipm-grooming.mkv~
diff --git a/claude-templates/.ai/workflows/read-calendar-events.org b/claude-templates/.ai/workflows/read-calendar-events.org
index be66bf4..5eac529 100644
--- a/claude-templates/.ai/workflows/read-calendar-events.org
+++ b/claude-templates/.ai/workflows/read-calendar-events.org
@@ -1,5 +1,5 @@
#+TITLE: Read Calendar Events Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-01
* Overview
diff --git a/claude-templates/.ai/workflows/readability-audit.org b/claude-templates/.ai/workflows/readability-audit.org
new file mode 100644
index 0000000..90ad366
--- /dev/null
+++ b/claude-templates/.ai/workflows/readability-audit.org
@@ -0,0 +1,242 @@
+#+TITLE: Readability Audit Workflow
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-06-28
+
+* Overview
+
+A pass over one file, a set of modules, or the whole tree that makes the code
+*readable to a future maintainer*. It checks four things and fixes the cheap
+ones in place: the file-top commentary, the inline comments, the names, and the
+physical organization of the code. Structural changes that need a real refactor
+(splitting a module, renaming a public symbol) are not done here — they are
+filed as =:refactor:= tasks so they get their own design and test pass.
+
+This is language-agnostic. Where a step names a language-specific tool or
+convention, it's stated as "the project's <X>, if it has one" — read the
+project's =CLAUDE.md= / =notes.org= and the language bundle to resolve the
+concrete tool.
+
+* Where it sits among the code-quality tools
+
+These tools are a pipeline, not duplicates. Knowing which to reach for:
+
+- *readability-audit* (this workflow) — prose and human-reader clarity:
+ comments, file headers, names, and physical organization. Judgment-driven
+ (does this comment lie? does this name reveal intent? can a newcomer place
+ this file in a minute?).
+- =/refactor= — structure on measurable metrics: complexity, duplication,
+ dead-code, the =simplification= lens (behavior-preserving logic/size
+ reduction), and =rename= (executes a codebase-wide symbol rename).
+- =/simplify= — behavior-preserving cleanup of the current diff, applied
+ directly.
+
+The link that keeps them from overlapping: when this audit finds a structural
+problem too big for a comment/name fix — a module to split, a *public* symbol to
+rename across call sites — it *files* a =:refactor:= task rather than doing it
+here. =/refactor= (rename, simplification) or =/start-work= then executes that
+filed task with a proper design and test plan. Readability finds and files;
+=/refactor= transforms.
+
+* Problem We're Solving
+
+Source files drift toward two opposite failure modes, and both hurt the next
+person to open the file:
+
+- *Documentation rot and noise.* Headers carry stale user-manual content
+ (quick-starts, full option matrices, setup walkthroughs) that belongs in user
+ docs; comments restate what the next line already says; comments go out of
+ date and start lying; placeholder =TODO=/=FIXME= stubs and conversational
+ asides accumulate. A blank summary or a missing file-top description leaves a
+ reader with no map.
+- *Structural fog.* Names that don't reveal intent force the reader to decode
+ them; related functions scatter; a public entry point sits far from the
+ private helpers it calls; a file grows to hold several unrelated
+ responsibilities.
+
+Left alone, opening a file costs more every month. The fix is a repeatable audit
+with a clear, checkable standard, run on demand or as files are touched.
+
+* Exit Criteria
+
+For the audited scope:
+
+1. *Every file has an accurate top section* that states what the file does and
+ how it fits the rest of the codebase — terse, no user-manual content, and
+ carrying the project's file-header convention where it has one.
+2. *Every surviving comment earns its place* — it explains a *why* the code
+ can't (a constraint, a workaround and its reason, an ordering dependency, a
+ warning), it is accurate against the current code, and it is terse. Obvious
+ "describe the next line" comments are gone.
+3. *Names reveal intent* — no cryptic abbreviations; the project's
+ public/private visibility convention is applied consistently.
+4. *Related code is co-located* — a public function's private helpers sit right
+ after it; the file reads top-to-bottom by descending abstraction; sections
+ group what belongs together.
+5. *Structural problems too big to fix in a comment pass are filed* as
+ =:refactor:= tasks, not left as a vague note and not half-done inline.
+6. *Nothing broke* — the build is clean and the test suite is green
+ (comment/name edits are behavior-preserving, so this should always hold; it
+ is the proof, not a hope). See "Graceful degradation" for projects without a
+ suite.
+
+* When to Use This Workflow
+
+- "Let's run the readability-audit workflow."
+- "Audit the comments and commentary in <file/area>."
+- "Clean up the structure/organization of <module>."
+- After landing a feature, on the files it touched, before moving on.
+- On a single file you just found hard to read.
+- As a tree-wide sweep: inventory all the source files, audit each, batch the
+ fixes.
+
+Do NOT use this to *perform* the structural refactors themselves (use
+=/refactor= or =/start-work= against a filed task) or to hunt for bugs /
+complexity / duplication (that is =/refactor=, not a readability pass).
+
+* Approach: How We Work Together
+
+** Phase 1 — Scope and inventory
+
+Pick the target: one file, a named module set, or the whole tree. For a sweep,
+list the source files (honor =.aiignore=) and decide coverage. Lean on the
+language's own doc linters as a first filter where they exist — many flag a
+missing or blank file summary and malformed headers; run the project's lint
+target first.
+
+** Phase 2 — Audit each file against the four dimensions
+
+Record findings as =file:line — issue — proposed fix=. The four dimensions:
+
+*** A. File-top commentary (the map)
+
+- Present, and *accurate* against what the file now does.
+- States purpose, the file's role/architecture, and key entry points —
+ *tersely*. A reader should learn what this is and how it connects in a few
+ lines.
+- Carries the project's file-header convention where it has one (a metadata
+ block, a module docstring, a standard header comment). If the project has no
+ header convention, skip this sub-check — don't invent one.
+- Does *not* carry user-manual content — quick-starts, full option matrices,
+ step-by-step setup. That belongs in user docs; move it, don't keep it in the
+ source header.
+- Mechanics are correct for the language: a filled summary line (not blank), the
+ expected section markers, the expected footer.
+
+*** B. Inline comments (why, not what)
+
+- Explains a *why* the code cannot: a workaround *and its reason*, an ordering
+ or load dependency, business-logic rationale, a real warning ("do not reorder
+ these — deadlock").
+- Is *accurate* — matches the current code. A wrong comment is worse than none;
+ fix or delete on sight.
+- Is *terse and useful*. Delete the obvious "describe the next line" comment
+ unless it names a non-obvious constraint. Replace a stale placeholder or a
+ rambling aside with the real one-line reason, or remove it.
+- Convert a comment that's only restating the code into a better *name* instead
+ (see C).
+
+*** C. Names (carry the what/how so comments don't have to)
+
+- Intention-revealing variable and function names; no cryptic single letters or
+ abbreviations outside tight local scopes.
+- The project's public/private convention is applied consistently and correctly:
+ a helper only called within the file is private; a user-facing or
+ intentionally-reusable symbol is public. (Resolve the concrete convention from
+ the language and the project — a naming prefix, an export list, an
+ access modifier.)
+- When a comment exists only to explain a name, rename instead.
+
+*** D. Organization (co-location and ordering)
+
+- Related functions sit together. A public function's private helpers come
+ *right after* it (stepdown / proximity / "reads like a newspaper").
+- The file reads top-to-bottom by descending abstraction.
+- Sections group what belongs together.
+- *Cohesion check:* if the file holds several unrelated responsibilities, or has
+ grown large enough that the top no longer describes one coherent thing, flag a
+ split into layered owners — but see Phase 4: that's a filed refactor, not an
+ inline fix.
+
+** Phase 3 — Apply the cheap, safe fixes inline
+
+Dimensions A, B, and C are *comment- and name-only* and *solo* (no design or
+preference call): apply them directly. After each file (or a batch), verify with
+the project's gates: parse/syntax check, a clean build (no new warnings), and a
+green test suite. Comment/name edits can't change behavior, so green is the proof
+the edit was clean, not a behavior check.
+
+For a tree-wide sweep, drive the uniform rewrites mechanically and verify the
+whole batch at once: a *mechanical applier with a boundary assertion* that
+replaces a well-defined header span is reliable and fast, then one suite run
+covers the batch. Keep the varied cases (header-line fixes, summary fixes that
+must preserve surrounding metadata, inline-comment surgery, generated-file
+headers) as careful per-file edits. (The boundary markers are language-specific;
+the principle — mechanical applier + assert + one suite run for uniform
+rewrites, per-file judgment for varied cases — is not.)
+
+** Phase 4 — File the structural refactors, don't do them here
+
+Dimension D's bigger findings — split a module, rename a *public* symbol across
+call sites, move a function to a different file — are real refactors with their
+own risk and test surface. Do *not* slip them into a readability pass. File each
+as a =:refactor:= task in =todo.org= with the specific finding, so it gets
+=/refactor= or =/start-work= with a proper design and test plan. This is the
+line between the cheap clarity win and the structural change; keeping it sharp is
+what lets the audit stay safe and fast.
+
+** Phase 5 — Verify and commit in logical batches
+
+Full suite green, build clean. Commit the doc/comment changes as =docs:= (or
+=refactor:= where a header/structure normalized) in cohesive batches — one
+commit per coherent slice (a set of condensed commentaries, the
+generated-file-header fixes, the obvious-comment prune), not one mega-commit and
+not one-per-file. Generated files are fixed *in their generator* and then
+regenerated, so the next regen stays compliant.
+
+* Graceful degradation
+
+The audit adapts to what the project provides:
+
+- *No file-header convention* → skip dimension A's metadata sub-check; still
+ check the summary/description for accuracy and terseness.
+- *No test suite* → the green-suite proof in Phases 3 and 5 is unavailable. Fall
+ back to the strongest gate the project has (compile/byte-compile, parse check,
+ linters) and *flag the weaker proof as a known limit* — a behavior-preserving
+ edit is lower-risk, but say plainly that there's no suite to confirm it.
+- *No doc linter* → do the Phase 1 first-filter by reading instead; the audit
+ still runs, just without the cheap pre-pass.
+
+* Principles to Follow
+
+- *Comments explain why; code explains what.* If a comment restates the code,
+ delete it or turn it into a better name.
+- *Accuracy beats completeness.* A wrong or stale comment is worse than no
+ comment. When in doubt, delete.
+- *Terse and useful.* Every comment and every header line earns its place. The
+ source header is not the user manual — move manuals to user docs.
+- *Readable means the next person, fast.* The test of the top-section and the
+ organization is whether a maintainer who has never seen the file can place it
+ and navigate it in under a minute.
+- *Keep the cheap pass cheap.* Comment/name fixes are solo and land inline.
+ Structural splits and public renames are not — they get filed, designed, and
+ tested separately.
+- *Preserve legal and attribution headers verbatim.* Vendored / GPL / copyright
+ notices are never condensed away by a readability pass.
+- *Manual validation is still Craig's.* Solo means no input is needed to *do*
+ the work; visual/behavior confirmation afterward is expected where relevant.
+
+* Living Document
+
+Update this with what real runs teach. Lessons worth keeping as the standard
+sharpens:
+
+- *Interpretation default for "fix blank summary":* when a rewrite shows only a
+ header + summary and omits a metadata block the file already has, keep the
+ existing metadata and replace only the header line and the summary. Its
+ absence from the rewrite means "leave it," not "delete it."
+- *Generated files:* fix the *generator*, then regenerate. Editing the generated
+ file directly is reverted on the next regen.
+- *Vendored files:* preserve the copyright/attribution; do not auto-condense a
+ licensed header.
+- *Mechanical applier + assert + one suite run* is the safe way to do a
+ many-file uniform rewrite; per-file judgment is for the varied cases.
diff --git a/claude-templates/.ai/workflows/rename-artifact.org b/claude-templates/.ai/workflows/rename-artifact.org
index 7b9f15b..a8d1246 100644
--- a/claude-templates/.ai/workflows/rename-artifact.org
+++ b/claude-templates/.ai/workflows/rename-artifact.org
@@ -1,5 +1,5 @@
#+TITLE: Rename an .ai Artifact
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-31
* Summary
diff --git a/claude-templates/.ai/workflows/send-email.org b/claude-templates/.ai/workflows/send-email.org
index 065f925..82d2286 100644
--- a/claude-templates/.ai/workflows/send-email.org
+++ b/claude-templates/.ai/workflows/send-email.org
@@ -1,5 +1,5 @@
#+TITLE: Email Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-01-26
* Overview
diff --git a/claude-templates/.ai/workflows/sentry.org b/claude-templates/.ai/workflows/sentry.org
new file mode 100644
index 0000000..b25fc14
--- /dev/null
+++ b/claude-templates/.ai/workflows/sentry.org
@@ -0,0 +1,227 @@
+#+TITLE: Sentry — Overnight Hygiene Supervisor
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-19
+
+* Overview
+
+Sentry is an interval loop that keeps a project's hygiene current while Craig is away. Each cycle walks a fixed list of passes — roam pull, inbox zero, triage (no mail or messengers), todo cleanup, task audit, working-files hygiene, spec board, link integrity, git health, prep freshness, bug and refactor finding, and (opt-in) solo-task implementation — and commits each pass's writing to a throwaway daily branch. Nothing pushes. In the morning Craig reviews the branch, squash-merges what he wants, and deletes it.
+
+The design goal is a project that greets the morning already tidy, with every judgment call and every destructive action parked in an approval queue rather than executed unattended. Sentry does the mechanical sweeping; Craig does the deciding.
+
+This file is the engine. It owns the entry gates, the branch mechanics, the lock model, the per-cycle pass runner, the digest and approval queue, the skip semantics, and the stop-sentry shutdown. The =agent-lock= helper (=.ai/scripts/agent-lock=) provides the locks. The passes reuse existing workflows (=inbox.org=, =triage-intake.org=, =clean-todo.org=, =task-audit.org=) under sentry's unattended contract.
+
+* When to Use This Workflow
+
+Craig arms sentry at the end of a session, with the machine left running, to have overnight hygiene done by morning.
+
+Triggers:
+
+- "start sentry", "run sentry", "arm sentry", "sentry mode"
+- "let sentry watch this overnight", "keep this tidy overnight"
+- "start sentry hourly", "start sentry every <interval>" (sets the loop interval)
+
+Stop trigger (see Stop Sentry below):
+
+- "stop sentry", "stand down sentry", "sentry off"
+
+Sentry is deliberately *not* auto-armed. Running it in a project is a per-project grant (the =:COMMIT_AUTONOMY:= marker) plus a deliberate launch with Craig at the terminal for the entry gates.
+
+* Prerequisite — the autonomy ticket
+
+Sentry commits unattended. =commits.md= gates commits on Craig's approval, so sentry needs standing, per-project authorization to run at all. Before anything else, read the project's =.ai/notes.org= Workflow State block for:
+
+: :COMMIT_AUTONOMY: yes
+
+If the marker is absent or not =yes=, decline to start and name the marker:
+
+: Sentry needs ":COMMIT_AUTONOMY: yes" in .ai/notes.org Workflow State to run — it commits unattended. Add it to grant, or run the hygiene passes by hand.
+
+No half-running mode: a project without the grant doesn't run sentry's read-only passes either. The grant is one line away, so this is a deliberate opt-in, not a barrier.
+
+A second, *independent* marker gates the solo-task implementation pass (pass 12):
+
+: :SENTRY_MAY_IMPLEMENT: yes
+
+=:COMMIT_AUTONOMY:= lets sentry commit its hygiene sweeps to the branch; =:SENTRY_MAY_IMPLEMENT:= additionally lets it implement solo, decision-free backlog tasks on the branch. The split exists because the two carry different morning costs: hygiene is a two-minute merge, implemented code is a review session. A project can run hygiene-only sentry without the implement pass, and most should until sentry has quiet weeks behind it. Absent =:SENTRY_MAY_IMPLEMENT:=, pass 12 skips; sentry still runs every other pass. Requires =:COMMIT_AUTONOMY:= alongside it — implementing implies committing.
+
+* Entry — interactive, with Craig present
+
+Craig types the sentry trigger, so the first moves run with him at the terminal. Do them in order; each gate that fails stops entry until Craig answers.
+
+1. *Autonomy ticket* — the prerequisite above. Absent → decline and stop.
+
+2. *Dirty-tree gate.* =git diff --quiet HEAD= (tracked modifications only; untracked and gitignored files never block — an inbox drop or scratch file is not in-progress work). If the tracked tree is dirty, describe what's dirty and offer, inline-numbered per =interaction.md=:
+
+ 1. Finish the job — commit the in-progress work first (recommended if it's a coherent unit)
+ 2. Stash it — =git stash= and start sentry on a clean tree
+ 3. Roll back named changes — discard specific files (names them)
+
+ Wait for an answer. Sentry can't start unattended from a dirty state; that's the point.
+
+3. *Green-suite gate.* Run the project's full suite (=make test=, or the project's equivalent — detect it). Read the output. If anything is red, describe the failures and offer to investigate before arming. The loop starts only on a green baseline, because every unattended cycle measures itself against "did I break this?" and a pre-existing red poisons that check.
+
+4. *Prior sentry branch.* =git branch --list 'sentry/*'=. An unmerged =sentry/*= branch from a previous night means the morning review didn't happen. Surface it and offer to squash-merge or delete it now (Craig is present); don't stack a second sentry branch on the first.
+
+5. *Reconcile the project branch.* Fetch and fast-forward-only against upstream — the same reconcile =startup= runs:
+
+ : git fetch --all --prune
+ : git rev-list --left-right --count @{u}...HEAD
+
+ Zero-behind → continue. Behind-only and clean → =git merge --ff-only @{u}=. Diverged → surface to Craig (he's present); don't auto-resolve.
+
+6. *Create the daily branch.* From HEAD:
+
+ : git switch -c "sentry/$(date +%F)-$(uname -n)"
+
+ The host suffix (=uname -n=) stops a same-date collision between the two daily drivers. The working tree now sits on this branch overnight — the launch hands the repo to sentry until the morning merge. Reclaiming it mid-night means stopping sentry first (see Stop Sentry). Note the Emacs buffer-revert caveat to Craig if he has the repo open: files change on disk under him overnight, so buffers want reverting after the morning merge (see =emacs.md=).
+
+7. *Arm the loop.* Start =/loop= at the interval (default hourly; Craig's "every <interval>" phrase overrides) with the per-cycle body being one sentry cycle (the Pass Runner below). Confirm the arming in one line: interval, branch name, project.
+
+* The lock model
+
+Two locks, both served by =.ai/scripts/agent-lock= (names only; the helper owns the paths, which live on tmpfs under =$XDG_RUNTIME_DIR/agent-locks/=, host-local and cleared on reboot).
+
+*Single-runner lock* (=sentry-<project>=, where =<project>= is the repo-root basename: =basename "$(git rev-parse --show-toplevel)"= — the same derivation =wrap-it-up.org='s guard uses, so the two agree on the lock name). Each cycle acquires it at cycle start and releases it at cycle end, and refreshes it between passes (the heartbeat, so a live cycle's lock never ages past one pass). If =/loop= fires again while a previous cycle still holds it, the new cycle's acquire fails and the cycle skips with one digest line — no two cycles run at once. The bounded wait is short (a few seconds); a live cycle means defer, not queue.
+
+*Roam-write lock* (=roam-write=). A pass that edits a file under =~/org/roam= acquires it, runs =capture-guard --wait= (the human-capture layer stays underneath), edits the working tree, triggers =systemctl --user start roam-sync.service=, and releases. The lock spans only edit-plus-trigger. Sentry never runs =git= against =~/org/roam= — roam-sync stays the repo's only committer (the 2026-06-24 one-git-owner rule). Pass 1's =pull --ff-only= is the sole, read-only exception.
+
+Every reclaim of a stale lock surfaces in the digest — the helper prints the reclaim note, and the cycle records it. A reclaim during a genuinely slow pass is possible, so it's never silent.
+
+* The Pass Runner — one contract per pass
+
+Each cycle, after acquiring the single-runner lock and verifying branch state (below), walks the pass list in order. Every pass follows the same four-step contract:
+
+1. *Probe* — a cheap existence check for the pass's target (named per pass below). Absent → the pass is one skip line in the digest and nothing more. This is what makes the pass list portable: passes self-activate where their target exists and stay silent elsewhere, with zero per-project configuration.
+
+2. *Work* — run the pass under the unattended contract. Quick, solo, already-agreed mechanical actions execute. Anything destructive or requiring judgment does *not* execute — it appends to the morning-approval queue (what, why, the exact command or edit that fires on approval). A pass runs fully or not at all; there is no reduced-form pass.
+
+3. *Session-context entry* — a pass that does or queues work appends its digest line to the =session-context.org= Session Log (path resolved via =.ai/scripts/session-context-path=) before its commit, so a crash between them still leaves the trail. Per-pass lines for an all-quiet cycle (every pass probe-skipped or no-op) are not written one by one — the cycle collapses to a single heartbeat at cycle-end (below), so an idle cycle doesn't spray one skip line per pass.
+
+4. *Commit* — if the pass wrote to disk, commit it: =chore(sentry): <pass> — <what changed>=. One commit per writing pass. A probe-skip or a no-op pass writes nothing and commits nothing.
+
+Between passes, refresh the single-runner lock (=agent-lock refresh sentry-<project>=) — the heartbeat.
+
+** Branch-state verification (cycle start, before the passes)
+
+After acquiring the lock, confirm the cycle is safe to run:
+
+- *On the right branch* — HEAD is =sentry/<today>-<host>=. If the loop was armed on a prior day and crossed midnight, the branch keeps the arming date; that's fine, morning teardown handles it. If HEAD is somehow *not* a sentry branch (an interrupted stop, a manual checkout), skip the whole cycle with a digest line rather than committing onto main.
+- *Clean of foreign changes* — =git diff --quiet HEAD= excluding the spine set (=session-context.org= / =session-context.d/=, resolved via =session-context-path=). Sentry's own spine writes must not trip this; a genuinely unexpected dirty tree (something outside the spine changed and wasn't committed by a prior pass) poisons the cycle — skip it with a digest line, the next cycle retries.
+
+* Unattended safety — skip, never degrade
+
+With no one at the terminal, any unsafe state makes the affected scope skip with one digest line, and the next cycle retries. Unsafe states and their scope:
+
+- *Unexpected dirty tree* (non-spine) → skip the whole cycle.
+- *Lost or un-acquirable single-runner lock* → skip the cycle (another cycle holds it, or the helper is missing).
+- *A pass's own precondition unmet* (its probe fails, or a dependency is dirty) → skip that pass only.
+- *Red suite at cycle-end* (see below) → the commits stay on the branch, flagged in the digest for morning review; the cycle doesn't roll back.
+
+Skips are never silent and never partial. Inside a *working* cycle, a pass line means the pass fully ran and a skip line names why it didn't. An *all-quiet* cycle is not a silent skip either: its single =sentry at HH:MM: nothing= heartbeat is the explicit record that every pass found nothing, standing in for a wall of identical skip lines. The anti-silence rule targets a pass that hides work it should have surfaced; a quiet cycle has surfaced that there was none.
+
+** Multi-day stall notification
+
+An unmerged prior =sentry/*= branch at cycle start (the morning review never happened) skips the cycle. After the *second consecutive* cycle skipped for this reason, send one persistent desktop notification naming the project and branch:
+
+: sentry stalled: <branch> unmerged — merge or delete to resume
+
+Then repeat at most daily. Persistent notify matches the paging convention — it stays on screen until dismissed. A multi-day stall never stays silent.
+
+* The pass list (v1)
+
+In order. Each names its detection probe. A pass whose probe fails is one skip line.
+
+1. *Roam pull* — =git -C ~/org/roam pull --ff-only=. Probe: =~/org/roam= is a git clone. Skipped when the roam tree is dirty (roam-sync owns that case) or the clone is absent. Read-only and ff-only — the one narrow exception to "don't touch roam git," so later passes read a fresh tree.
+
+2. *Inbox zero* — run =inbox.org= roam mode under the no-approvals contract: quick+solo+agreed items execute, shared-asset and convention proposals park (prepared diff, =VERIFY= task, sender reply) in the approval queue. Edits to =~/org/roam/inbox.org= take the roam-write lock + =capture-guard=. Probe: the roam clone or a project =inbox/= exists. Tidying the shared roam inbox is allowed from *any* project session, work included — it's housekeeping on a shared resource, not a durable KB-node write, so the work-denylist doesn't gate it (=knowledge-base.md=). Never park it as a cross-project boundary crossing.
+
+3. *Triage intake — mail and messenger sources excluded.* Run =triage-intake.org=, loading only its non-mail, non-messenger source plugins (calendar, PR/ticketing). The mail and messenger plugins — cmail, any Gmail variant, Telegram, Signal, chat DMs — are never loaded by a sentry cycle: Craig ruled 2026-07-21 that sentry doesn't check email or messengers. A manual "triage intake" still scans everything. Probe: the project has at least one *active* triage source that survives that exclusion — a project-specific plugin (=.ai/project-workflows/triage-intake.*.org=), or a non-empty =:TRIAGE_SOURCES:= declaration naming general plugins that exist. Mere presence of the template-synced general plugins does *not* activate the pass; a project that declares no sources, or whose only declared sources are mail or messengers, probe-skips (see =docs/specs/2026-07-20-triage-source-activation-spec.org=). Destructive actions (deleting, archiving, sending) queue; they never cycle unattended.
+
+4. *Todo cleanup* — the =clean-todo.org= mechanics (hygiene pass + =--archive-done= + =--convert-subtasks=). Probe: a root =todo.org=. Note that =--archive-done= is not purely an org-file pass on its first run in a project: it creates =archive/task-archive.org= and appends a =.gitignore= entry, so it produces a real tracked-file commit and correctly trips the cycle-end conditional suite. (archangel, first live run 2026-07-21.)
+
+5. *Task audit* — the *mechanical subset* of =task-audit.org= hourly (staleness counts, structural checks, cookie recomputation); the judgment half (priority regrades, consolidations, merge candidates) runs *once per night* and queues its findings rather than repeating them every cycle. Probe: a root =todo.org=. A full audit every hour is too heavy and re-surfaces the same judgment calls all night. (takuzu, first live run 2026-07-21.) Factual staleness fixes that are unambiguous still execute.
+
+6. *Working-files hygiene* — flag =working/<slug>/= directories whose backing task is closed (a filing candidate per =working-files.md=). Probe: a =working/= directory exists. The filing itself queues (it's a judgment move).
+
+7. *Spec status board* — the =docs-lifecycle= grep for spec keywords, surfacing any =DOING= spec whose bound build parent is closed. Probe: =docs/specs/= exists.
+
+8. *Link integrity* — broken =file:= links in the project's org files, via =lint-org.el=. Probe: =lint-org.el= present. Report-only into the digest; no unattended rewrites.
+
+9. *Git health* — uncommitted drift, unpushed commits on other branches, stale branches, main-behind-origin. Probe: =.git=. Report into the digest.
+
+10. *Prep + symlink freshness* — stale daily-prep docs, broken symlinks. Probe: the prep dir / symlinks exist (work and home only, in practice).
+
+11. *Bug and refactor finding* — hunt for real bugs and worthwhile refactoring opportunities in the project's codebase: static analysis (=shellcheck= for shell, the project's own linters for its languages), config sanity checks, plus one targeted code-reading area per cycle. Rotate the area across cycles and name it in the digest, so coverage accumulates over a night instead of re-reading the same corner. Randomized property sweeps (generate-and-verify against an engine's own invariants) are good quiet-cycle work here, reaching past a frozen test corpus. Expect the pass to go honestly quiet after the first few cycles find the standing defects; a quiet hunt is a result, not a failure. (takuzu, first live run 2026-07-21: three real fixes in the first four cycles, then quiet.) This pass does *not* run the test suite — the entry baseline already ran it, and re-running it hourly is anti-pattern 5; read the entry result instead. Probe: the project carries a codebase — source under version control beyond its org and tooling files. File each verified bug as a graded task in =todo.org= per the severity × frequency matrix (=todo-format.md=), and each refactoring opportunity as a =:refactor:= task, deduped against existing tasks; an unverifiable suspicion is a digest line, not a task. *Find, never fix in this pass* — the finding files a task and stops. A fix happens only in the opt-in implementation pass below, and only after the finding is a filed task that pass then re-verifies from scratch (see the premise rule there). A freshly-found "bug" can be a misread — one was filed and retracted two cycles apart on 2026-07-23 — so the file-then-verify-then-fix pipeline is deliberate: the task is the checkpoint, not a same-breath fix. (Added at Craig's order 2026-07-21, first dogfooded in dotfiles; refactor-finding added 2026-07-24.)
+
+12. *Solo-task implementation (opt-in — =:SENTRY_MAY_IMPLEMENT:=)* — work the backlog's solo, decision-free tasks on the branch. Probe: =.ai/notes.org= Workflow State carries =:SENTRY_MAY_IMPLEMENT: yes= *and* the project holds =:COMMIT_AUTONOMY:= (the implement pass commits). Absent the marker, skip — this pass is off by default, because it turns the morning from a two-minute merge into a code review, and that's the project owner's call. When on: invoke =work-the-backlog.org= under its unattended-loop contract (no pre-flight Q&A — there's no Craig overnight), eligibility =TODO= + =:solo:=, with the defer checklist deciding act-vs-file. The overnight-only tightening: only the *ready* bucket implements (clears every checklist item with zero open decisions); a task needing even one quick decision defers to a =VERIFY= rather than guessing, exactly as the loop caller already does. Commit each logical change to the sentry branch; *never push* — the morning review and merge is the gate, same as every other pass. The full quality bar holds (TDD, suite green before each commit, the isolated adversarial review per =publish= Step 1 with its re-review loop, =/voice=), and the review here runs the *premise check first*: reproduce the bug or confirm the problem is real before judging the diff. The review is the fact-checker that a filed claim never got, and it is what makes fixing-on-a-branch safe (Craig, 2026-07-24). A task that fails its premise check is not implemented — the finding was wrong, and that outcome is a digest line, not a commit. A task whose review never reaches approval — three rounds, a recurring finding, or a =Needs Discussion= verdict — is the same shape: no commit, and a digest line naming the standing findings, so the morning review sees what the reviewer would not pass rather than finding the task silently absent. (Added at Craig's direction 2026-07-24: overnight implement-on-branch, gated and never-pushed.)
+
+(KB lesson promotion — the pass the original proposal listed eleventh — is deferred to vNext. An unattended judgment pass writing to the shared knowledge base waits until sentry has quiet weeks behind it and a designed detection heuristic. See the filed lesson-detection-heuristic task.)
+
+* Cycle-end — conditional suite, then the digest commit
+
+After the passes:
+
+1. *Conditional suite run.* If any pass this cycle modified files *outside* the org/spine set (a code-touching pass, rare but possible via fixtures), run the full suite once. A green run confirms the cycle's commits are safe; a red run flags the digest for morning review — the commits stay on the branch (nothing is pushed, so the morning gate catches it). No per-pass suite runs: the entry run is the green baseline, and hourly per-commit runs would turn a seconds-long cycle into minutes all night. Cycles that only touched org/spine files skip this.
+
+2. *Heartbeat or digest, then commit.* Decide quiet vs working. A *quiet* cycle — every pass probe-skipped or no-op, nothing added to the approval queue — writes a single heartbeat line to the Session Log, =sentry at HH:MM: nothing= (HH:MM local, from =date=), and no per-pass digest block. A *working* cycle — any pass ran, wrote, or queued — writes its full per-pass digest block. Then commit any accumulated spine writes in one sweep: =chore(sentry): digest — <date> <time> cycle= for a working cycle, =chore(sentry): heartbeat — <date> <time>= for a quiet one, so even a quiet cycle leaves a clean tree for the next branch-state check (where the spine is untracked, the mirror-only case, there is nothing to commit and the heartbeat line stays in the working-tree anchor). This is the silent-until-signal policy (see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=): an all-quiet night collapses from a wall of no-op digests to a list of one-line heartbeats, while a cycle that actually did or queued something still writes the full record.
+
+3. *Release the single-runner lock.*
+
+* The digest and the approval queue
+
+*Digest.* A *working* cycle appends its block to the =session-context.org= Session Log (the spine the cycle already writes), so it survives a crash, rides the session archive, and is on screen in the running session. One block per working cycle: the timestamp, then one line per pass (ran + what, or skipped + why), plus any lock reclaim notes. A *quiet* cycle (nothing done or queued) writes no block — just the one heartbeat line =sentry at HH:MM: nothing= (the silent-until-signal policy). The per-pass block is a working-cycle artifact; it still carries one line per pass so a real skip inside a working cycle is never hidden.
+
+*Approval queue.* Destructive and judgment actions accumulate under one heading in the same file — =* Sentry approval queue (<date>)= — newest last. Each item carries three things: *what* (the action), *why* (what triggered it), and the *exact command or edit* that fires on approval. The morning review is Craig reading this heading top to bottom and running or discarding each item.
+
+* Morning teardown — Craig's, documented not automated
+
+Sentry never merges its own branch. In the morning Craig:
+
+1. Reviews the digest and the approval queue in =session-context.org=.
+2. Runs or discards each approval-queue item.
+3. Reviews the branch: =git log main..sentry/<date>-<host>= and the diff.
+4. Squash-merges what he wants (=git switch main && git merge --squash sentry/<date>-<host>=, then one clean commit) or cherry-picks selectively.
+5. Deletes the branch: =git branch -D sentry/<date>-<host>=.
+6. Reverts any Emacs buffers still showing the pre-merge on-disk state (=emacs.md= buffer-revert caveat).
+
+A bad night is discarded by deleting one branch — nothing reached main, nothing was pushed.
+
+In a project that gitignores =.ai/=, the whole spine is untracked, so quiet cycles produce no commits at all and =git log main..sentry/<date>-<host>= understates the night's activity. There the anchor's heartbeat list is the only record of what fired. Read the anchor, not just the log. (archangel, first live run 2026-07-21.)
+
+* Stop Sentry
+
+Trigger: "stop sentry" (and synonyms above). Sentry owns its own shutdown:
+
+1. *Cancel the loop* — stop the =/loop= (=ScheduleWakeup= stop / the loop's stop path). No further cycles.
+2. *Release the single-runner lock* if this context holds it.
+3. *Branch disposition* — offer, inline-numbered:
+ 1. Squash-merge the day's branch into main now (walk the morning teardown steps 3-5 interactively)
+ 2. Leave it named for later review (=sentry/<date>-<host>= stays; review at leisure)
+4. *Approval queue* — offer to walk the queued items now, or carry them (they stay under the heading for whenever Craig reviews).
+
+Stopping sentry is the only way to reclaim the working tree mid-night. The entry gate fronts the handoff; stop-sentry ends it.
+
+* Wrap-up interaction
+
+=wrap-it-up.org= refuses while sentry is live: it detects the single-runner lock (=agent-lock status sentry-<project>= → held) and stops with "sentry is active — say 'stop sentry' first." The shutdown logic lives here, not in wrap-up; wrap-up carries only the one guard.
+
+* Common Mistakes
+
+1. *Running without the =:COMMIT_AUTONOMY:= grant* — sentry commits unattended; the marker is the entry ticket, and its absence is a hard stop, not a degrade.
+2. *Starting from a dirty or red tree* — the entry gates exist because an unattended cycle can't tell Craig's in-progress work from a regression. Answer the gate; don't bypass it.
+3. *Committing onto main* — every writing pass commits to the daily =sentry/*= branch. A cycle that finds HEAD off the sentry branch skips rather than commits.
+4. *Running a =git= write against =~/org/roam=* — roam-sync is the only committer. Sentry edits the tree under the roam-write lock and triggers the sync; it never commits or pushes roam.
+5. *A per-pass suite run* — the suite runs at entry (baseline) and conditionally at cycle-end (only when a pass touched non-org files). Hourly per-commit runs all night is the anti-pattern the suite policy exists to prevent.
+6. *Executing a judgment or destructive action unattended* — those queue for the morning with their exact command. The pass did its detection; Craig makes the call. The one sanctioned exception is pass 12's solo-task implementation, and only because it inherits work-the-backlog's full defer checklist (data-loss and irreversible actions defer, never execute) plus a premise-verifying review, and it commits to the branch rather than acting on anything live.
+7. *A silent skip* — inside a working cycle, every skip writes a digest line naming why; a missing pass with no line reads as "ran clean" when it didn't. The one exception is not a violation: an all-quiet cycle collapses to a single =sentry at HH:MM: nothing= heartbeat instead of one skip line per pass — the heartbeat is the explicit "nothing to do" record, per the silent-until-signal policy.
+8. *Degrading a pass to a reduced form* — a pass runs fully or skips. No half-passes.
+9. *Letting an unmerged branch stall silently* — after two consecutive unmerged-branch skips, the persistent desktop notify cycles. Don't suppress it.
+10. *Merging sentry's branch automatically* — the morning teardown is Craig's. Sentry creates and commits; it never merges or deletes its own branch.
+
+* Living Document
+
+Sentry ships with eleven finding/hygiene passes, one opt-in implementation pass, and a deferred KB pass. The pass list, the interval default, the =:SENTRY_MAY_IMPLEMENT:= default, and the queue-vs-execute line for each pass are the knobs most likely to move with dogfooding. The implement pass especially is new (2026-07-24) and unproven at scale — watch the corrections signal (work-the-backlog's metric for autonomous commits later reverted or hand-fixed) before widening it past the projects that opt in. Fold in what the live trial surfaces — a pass that queues too eagerly, a probe that misfires, a digest line that wants more detail. Refine as the signal arrives.
+
+* History
+
+Built 2026-07-19 from the sentry spec (=docs/specs/2026-07-14-sentry-workflow-spec.org=, ID f6c51f27-d7a2-4b63-9ff9-5ba005a66dfb) — 10 decisions and 12 review findings resolved before the build. Phase 1 shipped the =agent-lock= helper (commit =a8b6cf4=); this file is Phase 2, the engine. Phase 3 reconciles the roam writers (=inbox.org=, =knowledge-base.md=) to acquire the roam-write lock and adds the =wrap-it-up.org= guard.
diff --git a/claude-templates/.ai/workflows/session-harvest.org b/claude-templates/.ai/workflows/session-harvest.org
index c48d689..54a7c09 100644
--- a/claude-templates/.ai/workflows/session-harvest.org
+++ b/claude-templates/.ai/workflows/session-harvest.org
@@ -1,5 +1,5 @@
#+TITLE: Session-Harvest Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-06-11
* Overview
diff --git a/claude-templates/.ai/workflows/spec-create.org b/claude-templates/.ai/workflows/spec-create.org
index f90c511..39758a0 100644
--- a/claude-templates/.ai/workflows/spec-create.org
+++ b/claude-templates/.ai/workflows/spec-create.org
@@ -1,5 +1,5 @@
#+TITLE: Spec-Create Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-06-09
* Overview
@@ -10,7 +10,7 @@ The guiding principle, drawn from how Google, Oxide, Amazon, Basecamp, and the A
It is the front of a trio:
- =spec-create.org= (this one) — author writes the spec.
-- =spec-review.org= — a reviewer gates the spec for implementation-readiness and writes =<spec-basename>-review.org=.
+- =spec-review.org= — a reviewer gates the spec for implementation-readiness and records findings in the spec's =* Review findings= section.
- =spec-response.org= — the author folds the review back in.
The spec this workflow produces has to *pass spec-review's gate* — that gate is the definition of done. So the structure below is built to answer the reviewer's questions up front. Keep it lightweight anyway: a short required spine plus a *readiness-dimensions menu* where each item is either answered or explicitly marked "N/A because…". The best spec is the shortest one that still lets an engineer build it, test it, and ship behavior that matches the user's mental model.
@@ -47,6 +47,8 @@ Capture, in this order:
** Phase 2 — Design, alternatives, decisions
1. *Design* — overview first, then detail. Write the reasoning as *prose, not bullet dumps* — prose exposes weak logic that bullets let you hide. Use bullets only for genuinely enumerable lists. When the thing has an interface, use the *two-altitude* split (Rust RFC): explain it once for a user/caller, once for an implementer.
+
+ *Non-trivial UI.* When the deliverable is a real UI (a panel, a multi-control surface, an interacting visual layout — not a single dialog, a CLI flag, or a one-off prompt), the design isn't settled on the page. Run the research → ~5 distinct working-prototype directions → iterate-one-to-final process in =claude-rules/ui-prototyping.md= before treating the UI design as done, and add a =Prototype iterations= subsection under the spec's status heading linking every iteration (final linked in the design section). A UI design decision moves to =DONE= only once it's been seen working in a prototype.
2. *Alternatives considered* — the load-bearing section authors skip and reviewers need most. For each option, force a why-not with the MADR grammar: "Good, because… / Bad, because… / Neutral, because…". Even one rejected option, with the reason, beats presenting one path as inevitable.
3. *Decisions* — capture each real choice as an org =TODO= task carrying an inline mini-ADR (Nygard's spine):
- The heading is =** TODO <Decision name>=. It flips to =DONE= when the decision-maker agrees with the call; until then it stays =TODO=.
@@ -59,7 +61,7 @@ Capture, in this order:
This is where the spec earns a "Ready" from review: an engineer must be able to build it in steps, know when it's done, and never have to invent product behavior mid-implementation.
-1. *Implementation phases* — decompose the work into phases each small enough to finish in one focused session and each leaving the tree in a working (not half-broken) state. =spec-review= lifts this section straight into =todo.org= tasks, so a spec that can't be phased fails the gate — the absence is itself a finding.
+1. *Implementation phases* — decompose the work into phases each small enough to finish in one focused session and each leaving the tree in a working (not half-broken) state. =spec-review= checks this section decomposes cleanly and =spec-response= lifts it into =todo.org= tasks, so a spec that can't be phased fails the gate — the absence is itself a finding.
2. *Acceptance criteria* — the observable conditions that mean the feature works, written as checkable items. The review's test-surface task mirrors these.
3. *Readiness dimensions* — walk this menu and, for each, either define the behavior or write "N/A because…". The escape hatch keeps a simple spec short; the prompt keeps a hidden decision from slipping into implementation:
- *Data model & ownership* — what's user-authored / generated / cached / remote; who owns each editable region; what persists vs refreshes.
@@ -82,8 +84,9 @@ This is where the spec earns a "Ready" from review: an engineer must be able to
** Phase 5 — Wire it up (conventions)
-- *Filename + location:* =docs/<problem-slug>-spec.org=. Org-mode. The slug names the *problem/feature*, not a date. Must end in =-spec.org=.
-- *Metadata header:* a small table at the top — Status, Owner, Reviewer(s), Date, Related (link to the task/ticket).
+- *Filename + location:* =docs/specs/YYYY-MM-DD-<problem-slug>-spec.org= — formal specs live in =docs/specs/=, never =docs/design/= (that's for notes, brainstorms, inventories; see =claude-rules/docs-lifecycle.md=). Org-mode. The slug names the *problem/feature*; no status suffixes ever — status lives in the file. Must end in =-spec.org=.
+- *Status heading (first element after the file header):* a top-level heading carrying the lifecycle keyword, stamped =DRAFT= at authoring — spec-create owns this flip. It holds an =:ID:= UUID (generate with =uuidgen=) and dated history lines, newest first. The keyword is authoritative; the Metadata =Status= field mirrors it in lowercase. Transitions are three lines in one file (keyword + history line + mirror): spec-review flips =READY=, spec-response flips =DOING= at decomposition, the final build task flips =IMPLEMENTED=. Terminal states always record a reason.
+- *Metadata header:* a small table at the top — Status (the lowercase mirror), Owner, Reviewer(s), Date, Related (link to the task/ticket).
- *Review-and-iteration-history stub:* add a =Review and iteration history= section at the bottom and seed it with the author's first entry. =spec-review= and =spec-response= append provenance entries here, so the heading shape is a contract: =YYYY-MM-DD Day @ HH:MM:SS -ZZZZ — Contributor — Role=, body fields What / Why / Artifacts.
- *Cross-link both ways:* the spec links its task; the task links the spec (replace the task's inline plan with a terse description + a =file:= link to the spec).
@@ -103,7 +106,14 @@ Then it's ready for =spec-review.org=. Snapshot-vs-living rule: keep the spec li
,#+TITLE: <Feature> — Spec
,#+AUTHOR: <author>
,#+DATE: <YYYY-MM-DD>
-,#+TODO: TODO | DONE SUPERSEDED CANCELLED
+,#+TODO: TODO | DONE
+,#+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED
+
+,* DRAFT <spec short name>
+:PROPERTIES:
+:ID: <uuid — generate with uuidgen>
+:END:
+- <YYYY-MM-DD Day @ HH:MM:SS -ZZZZ> — drafted.
,* Metadata
| Status | draft |
diff --git a/claude-templates/.ai/workflows/spec-response.org b/claude-templates/.ai/workflows/spec-response.org
index 2686cf8..7628e49 100644
--- a/claude-templates/.ai/workflows/spec-response.org
+++ b/claude-templates/.ai/workflows/spec-response.org
@@ -5,9 +5,9 @@
* Overview
-The spec-response workflow processes external reviews of a design spec and folds them into the spec until it is implementation-ready. A reviewer (human or another agent) leaves a review file next to the spec — typically produced by its counterpart, the *spec-review* workflow, which writes =<spec-basename>-review.org= and assigns a readiness rubric. Claude works through every recommendation, deciding accept / modify / reject for each, updates the spec for the accepted ones, documents the modified and rejected ones with reasons, then deletes the review file. Repeat for each spec under review.
+The spec-response workflow processes a review's findings and folds them into the spec until it is implementation-ready. A reviewer (human or another agent) records findings in the spec's =* Review findings= section — typically via its counterpart, the *spec-review* workflow, which writes one =TODO= task per finding (=[/]= cookie on the heading) and assigns a readiness rubric. Claude works through every finding, deciding accept / modify / reject, updates the spec body for the accepted ones, and completes each finding task in place: accept and modify finish =DONE= (the modify's change noted in the body), reject finishes =CANCELLED= with the reason. Repeat for each spec under review.
-The output is a spec a reader could implement from, plus a durable record — inside the spec — of why any recommendation was changed or declined, so the reviewer can find the reasoning later without re-litigating.
+The output is a spec a reader could implement from, plus a durable record — inside the spec — of why any finding was changed or declined, so the reviewer can find the reasoning later without re-litigating. The reasoning lives on the completed finding task, not a separate file.
This workflow was first run on 2026-05-23 against the linear-emacs =issue-query-spec.org= and =issue-representation-spec.org= reviews, and written up from that run.
@@ -27,25 +27,25 @@ A review is only useful if every point in it gets a decision. Without a defined
A spec's review is fully processed when:
-1. *Every recommendation has an explicit disposition* — accepted, modified, or rejected. None dropped.
-2. *Accepted recommendations are woven into the spec body* — the spec reads as if they were always there, not appended as a changelog.
-3. *Modified and rejected recommendations are documented in a bottom section* (e.g. "Review dispositions") with a one-paragraph reason each, so the reviewer can find the reasoning.
+1. *Every finding has an explicit disposition* — accepted, modified, or rejected, and its task completed (=DONE= or =CANCELLED=). None dropped.
+2. *Accepted findings are woven into the spec body* — the spec reads as if they were always there, not appended as a changelog.
+3. *Modified and rejected findings carry a one-paragraph reason in the completed task's body*, so the reviewer can find the reasoning.
4. *Review/response provenance is documented in the spec* — iteration count/date, contributor, role, what changed, and why.
5. *Pre-agreed decisions are flipped to =DONE=* — each settled decision's =TODO= becomes =DONE=, and the =[/]= cookie on the spec's =* Decisions= heading reflects the tally.
6. *Cross-spec tensions are reconciled in writing* when related specs were reviewed together.
-7. *The review file is deleted* once 1-6 hold.
+7. *Every finding task is completed* — the =* Review findings= =[/]= cookie reads complete (each finding =DONE= or =CANCELLED=).
8. *Tracking is updated* — the spec's VERIFY/task body notes "review incorporated" and whether it's implementation-ready.
9. *Implementation tasks exist* — once the author confirms the spec is Ready, the project's =todo.org= carries the full implementation-task breakdown (Phase 6), reviewed for completeness, with =:solo:= marked and a Manual-testing task for everything else.
-The whole run is done when no =*-review.org= files remain and each spec is judged implementation-ready (or its remaining blockers are named).
+The whole run is done when every spec's =* Review findings= cookie reads complete and each spec is judged implementation-ready (or its remaining blockers are named).
-*Measurable validation:* a reader scanning the review against the revised spec can find, for every review point, either the change in the body or its disposition at the bottom. Nothing is unaccounted for.
+*Measurable validation:* a reader scanning the spec can find, for every finding, either the change in the body or its disposition on the completed finding task. Nothing is unaccounted for.
* When to Use This Workflow
Trigger when:
-- A reviewer drops a review file alongside a spec — convention: same basename with a =-review.org= suffix (=foo.org= → =foo-review.org=).
+- A reviewer records findings in the spec's =* Review findings= section — typically via the spec-review workflow.
- Craig says "respond to the review" / "let's run the spec-response workflow" / "process the spec reviews."
- Any time a spec needs to absorb structured external feedback and converge to implementation-ready.
@@ -63,7 +63,7 @@ the file should be renamed first. Spec workflows require the -spec.org suffix as
guard against pointing the workflow at tutorial, inventory, or setup docs.
#+end_example
-The review file the response consumes follows the convention =<spec-basename>-review.org=, so a misnamed spec produces a mis-pointed review file too. Fix the spec name first.
+The findings the response consumes live in the spec's own =* Review findings= section, so the spec filename is the handle the workflow keys on — point it at the right =-spec.org= file. Fix the spec name first.
The user resolves the mismatch and re-invokes the workflow. Do not proceed with the response against a misnamed spec.
@@ -71,7 +71,7 @@ The user resolves the mismatch and re-invokes the workflow. Do not proceed with
** Phase 0: Orient
-1. List the review files (=ls docs/*-review.org= or wherever they live). Process them one at a time in whatever order; the user may name an order.
+1. Find specs with open findings — a =* Review findings= section whose =[/]= cookie isn't complete (=TODO= findings remain). Process them one at a time in whatever order; the user may name an order.
2. Re-read the *current* spec, not your memory of it — it may have changed since you wrote it (the user or a linter may have edited it, and pre-agreed decisions may already be encoded in the tracking file).
3. Note any *pre-agreed decisions* the reviewer or user has already settled — in the review's own "Agreed decisions" section, or in the spec's tracking task. These are settled inputs. Don't reopen them; bake them in.
@@ -79,15 +79,15 @@ The user resolves the mismatch and re-invokes the workflow. Do not proceed with
Read the entire review first. Recommendations interact — an early "medium" finding may be subsumed by a "high" one, or two findings may point at the same edit. Decide dispositions with the whole picture in view, not finding-by-finding as you scroll.
-** Phase 2: Decide a disposition for every recommendation
+** Phase 2: Decide a disposition for every finding
-For each recommendation, choose one:
+For each finding, choose one:
-- *Accept* — the recommendation is right as written. Plan the edit.
-- *Modify* — the recommendation is right in spirit but wrong in detail or scope. Adjust it, and record what you changed and why.
-- *Reject* — the recommendation doesn't fit. Record why.
+- *Accept* — the finding is right as written. Plan the edit.
+- *Modify* — the finding is right in spirit but wrong in detail or scope. Adjust it, and record what you changed and why.
+- *Reject* — the finding doesn't fit. Record why.
-*Engage critically.* Rubber-stamping is a failure mode. On a strong review most points are accepts, but actively look for the genuine modify/reject cases — they are where your judgment earns its place. Examples from the first run:
+*Engage critically.* Rubber-stamping is a failure mode. On a strong review most findings are accepts, but actively look for the genuine modify/reject cases — they are where your judgment earns its place. Examples from the first run:
- *Modify:* the review proposed an automatic cache-TTL defcustom. Accepted the goal (fresh data) but deferred TTL to vNext because for a single-user tool an explicit force-refresh + clear-cache command covers it without invalidation complexity.
- *Reject:* the review floated a separate =default-issue-filter= defcustom. Rejected as redundant — a fixed default command plus a default-view preference already covered it.
@@ -101,8 +101,8 @@ When related specs were reviewed together, two reviews can recommend opposite th
** Phase 4: Update the spec
-1. *Weave accepted recommendations into the body.* The spec should read naturally — a new "Selector semantics" section, a revised phase plan, an added test-strategy section — not a list of "review said X so I did Y." The body reflects the decisions; it doesn't narrate them.
-2. *Add a bottom "Review dispositions" section* listing only the *modified* and *rejected* recommendations, each with a short reason. Close it with a one-line "everything else accepted as written" so the reader knows the omissions from this section are accepts, not gaps.
+1. *Weave accepted findings into the body.* The spec should read naturally — a new "Selector semantics" section, a revised phase plan, an added test-strategy section — not a list of "review said X so I did Y." The body reflects the decisions; it doesn't narrate them.
+2. *Complete each finding task in place.* Accept → =DONE=, body noting where it was folded; modify → =DONE=, body noting what you changed and why; reject → =CANCELLED=, body giving the reason. The =[/]= cookie on =* Review findings= tracks progress. The reason on a modified or rejected finding is the durable record — accepted findings are recorded by the body change itself, so they need no separate note. The asymmetry is deliberate.
3. *Update or add a bottom "Review and iteration history" section.* Every response pass gets an entry, even when all findings are accepted. Each entry is an org subheading with a compound id followed by three body fields:
Heading format: =YYYY-MM-DD Day @ HH:MM:SS -ZZZZ — Contributor — Role=
@@ -113,26 +113,28 @@ When related specs were reviewed together, two reviews can recommend opposite th
- *What changed:* compact summary of accepted, modified, and rejected work.
- *Why:* the rationale or decision pressure behind the changes.
- - *Artifacts:* review filename, disposition section, task IDs, source checks, or commits when useful.
+ - *Artifacts:* the relevant findings, task IDs, source checks, or commits when useful.
4. *Flip settled decisions to =DONE=.* Each decision the decision-maker has agreed flips its =TODO= to =DONE=; the =[/]= cookie on the =* Decisions= heading tracks the tally. A contested decision stays =TODO= with the back-and-forth under its =*** Discussion= child header. Decisions still =TODO= should be only what genuinely still blocks, each with an owner and a by-when.
-5. *Raise the spec to implementation-ready:* consolidate decisions up front, add any implementation prerequisites the review surfaced (e.g. a schema-verification checklist), a consolidated test strategy, and a phased plan ordered so dependencies (like an output model everything depends on) come early. *Gate:* the spec Status cannot move past =draft= to implementation-ready while any decision is still =TODO= — the =[/]= cookie must read complete, or the author consciously accepts and records the risk of building with one open.
+5. *Raise the spec to implementation-ready:* consolidate decisions up front, add any implementation prerequisites the review surfaced (e.g. a schema-verification checklist), a consolidated test strategy, and a phased plan ordered so dependencies (like an output model everything depends on) come early. *Gate:* the spec Status cannot move past =draft= to implementation-ready while any decision or any =:blocking:= finding is still =TODO= — both =[/]= cookies must read complete, or the author consciously accepts and records the risk of building with one open. *If this response expanded scope* — folding a finding in added new phases, decisions, or external-dependency assumptions — re-run spec-review's readiness rubric against the *expanded* spec, and file any new gap as a finding or decision before claiming =Ready=. Disposition-completeness gates the *review*; the readiness rubric gates the *spec*. A response can resolve every finding and still be less ready than before, because the answers introduced unproven obligations — the cookies only protect you if the new obligation is actually filed.
6. *Update the status line* to note "review incorporated (<reviewer>, <date>)."
** Phase 5: Close out and iterate
-1. *Delete the review file* — only after every recommendation has a disposition. Its deletion is the signal the review is fully processed.
-2. *Update tracking* — the spec's VERIFY/task body gets a line noting review incorporated, what changed at a high level, which recommendations were modified (pointing at Review dispositions), and whether it's now implementation-ready pending final go.
+1. *Confirm every finding is completed* — the =* Review findings= =[/]= cookie reads complete (every finding =DONE= or =CANCELLED=). The complete cookie is the signal the review is fully processed; there is no file to delete.
+2. *Update tracking* — the spec's VERIFY/task body gets a line noting review incorporated, what changed at a high level, which findings were modified or rejected (pointing at the completed findings), and whether it's now implementation-ready pending final go.
3. *Update the session log* (state changed this turn).
-4. *Move to the next review file.* Repeat Phases 1-5 until none remain.
+4. *Move to the next spec with open findings.* Repeat Phases 1-5 until none remain.
5. *Report* what was accepted-wholesale, what was modified/rejected and why, any cross-spec reconciliations, and the implementation-ready verdict per spec.
** Phase 6: On Ready, build the implementation-task breakdown
This is the *last* step of the workflow, and it runs *only after the author confirms the spec is Ready* — never during review iterations. A Ready spec nobody can act on is unfinished; this phase turns it into tracked work. It applies to every project type (library, application, service, docs set).
-1. *Decide where the tasks live.* If the work is spinning off into its own project/repo, move the parent task into that project's =todo.org= (and relocate the spec with it); otherwise use the current project's =todo.org=. One parent task owns the effort; the phase tasks hang under it.
+*This phase owns the =READY= → =DOING= lifecycle flip* (docs-lifecycle convention): when the decomposition below lands, update the spec's top-level status heading keyword to =DOING=, add a dated history line, and set the Metadata =Status= mirror to =doing= — three lines, one file.
-2. *Create one task per implementation phase* from the spec's =Implementation phases=, in dependency order, so the task set as a whole describes the *full* milestone (e.g. v1) with no gaps. Each task body names the deliverable, its tests, and how it is verified. Carry over deferred/vNext work and any publish/release steps as their own tasks.
+1. *Decide where the tasks live.* If the work is spinning off into its own project/repo, move the parent task into that project's =todo.org= (and relocate the spec with it); otherwise use the current project's =todo.org=. One parent task owns the effort; the phase tasks hang under it. *Stamp the binding:* the parent task's =:PROPERTIES:= drawer gets a =:SPEC_ID:= line holding the spec's status-heading UUID. That property is the durable join task-audit uses to police =DOING= specs (a =DOING= spec whose bound parent is closed, archived, or missing gets flagged).
+
+2. *Create one task per implementation phase* from the spec's =Implementation phases=, in dependency order, so the task set as a whole describes the *full* milestone (e.g. v1) with no gaps. Each task body names the deliverable, its tests, and how it is verified. Carry over deferred/vNext work and any publish/release steps as their own tasks. *Always end the set with the flip task:* a final "flip the spec to IMPLEMENTED (+ dated history line + mirror)" task under the same parent — the tracked obligation that closes the lifecycle loop when the build finishes. Never skip it; "a human remembers" is the failure mode this exists to prevent.
3. *Turn a critical eye on completeness.* Re-read the spec — every phase, every acceptance criterion, every named deliverable, every data-safety/principle rule — and confirm each has a home in a task. The work is not done when the tasks merely exist; it is done when nothing in the spec is left untracked. This completeness pass is mandatory regardless of project type.
@@ -150,7 +152,7 @@ The workflow is complete when these tasks exist, the completeness pass confirms
Accept, modify, or reject — but never silently drop. The reviewer must be able to account for every point.
** Document the no's, not the yes's
-Accepted recommendations live in the spec body (the change *is* the record). Modified and rejected ones need an explicit written reason at the bottom, because the change is invisible and the reasoning would otherwise be lost. The asymmetry is deliberate.
+Accepted findings live in the spec body (the change *is* the record). Modified and rejected ones need an explicit written reason on the completed finding task, because the change is invisible and the reasoning would otherwise be lost. The asymmetry is deliberate.
** Critique, don't rubber-stamp
A review you accept entirely without finding a single thing to push on probably wasn't read critically. Your judgment — including a well-reasoned no — is the value you add.
@@ -164,11 +166,11 @@ When reviews conflict, find the framing where both are right. Silently honoring
** A reject goes back to the reviewer, not just into the file
Recording a reasoned reject is the floor, not the close. Communicate the rejection and its reason to the reviewer — a reject is a two-party event, not a unilateral call. If the reviewer disagrees, that's a discussion: weigh the counter, and if you still can't agree, escalate to whoever owns the decision rather than letting the author's "no" stand by default. "I'm not doing that" with no reason the reviewer can engage is the failure mode. (For a tight solo author-reviewer loop this is lightweight; for a team it's the difference between a review and a rubber-stamp-in-reverse.)
-** The spec reads forward, the dispositions read backward
-The body is written for the implementer (no review archaeology). The dispositions section is written for the reviewer (the reasoning trail). Keep the two audiences separate.
+** The spec reads forward, the findings read backward
+The body is written for the implementer (no review archaeology). A completed finding's reason is written for the reviewer (the reasoning trail). Keep the two audiences separate.
** The history explains provenance, not implementation behavior
-The spec body should still be the implementation contract. The bottom =Review and iteration history= section is for provenance: number of iterations, dates, contributors (including agents), roles, what each pass contributed, and why. Keep it short enough that future readers can understand how decisions evolved without rereading chats, deleted review files, or session logs.
+The spec body should still be the implementation contract. The bottom =Review and iteration history= section is for provenance: number of iterations, dates, contributors (including agents), roles, what each pass contributed, and why. Keep it short enough that future readers can understand how decisions evolved without rereading chats or session logs.
** Re-read before editing
The spec may have changed since you last saw it. Edit the current file, reconcile against the latest tracking state.
@@ -213,3 +215,8 @@ Update this workflow as we learn what works. Capture new disposition patterns, b
- *What:* Reconciled this workflow to spec-create's new Decisions convention (each decision is an org =TODO= task that flips to =DONE= on agreement, with a =[/]= cookie on the =* Decisions= heading and a =*** Discussion= child for disputes). Exit Criterion 5, Phase 2's pre-agreed-decisions step, and Phase 4 steps 4-5 now speak in flip-to-=DONE= terms, and the implementation-ready step gates on the all-=DONE= cookie.
- *Why:* The convention change landed in spec-create.org via an .emacs.d handoff (originated in its keymap-consolidation spec); this workflow still described the retired =State: proposed | accepted | superseded= model.
- *Artifacts:* Handoff =inbox/2026-06-12-1906-from-.emacs.d-spec-create-decisions-todo-note.org=. Paired spec-create.org and spec-review.org edits in the same commit.
+
+** 2026-06-21 Sun @ 23:16:06 -0400 — Claude Code (rulesets) — responder
+- *What:* Folded the review into the spec. Findings are now =* Review findings= =TODO= tasks the responder completes in place (accept/modify → =DONE=, reject → =CANCELLED= with the reason) instead of a "Review dispositions" section; the response is done when the =[/]= cookie reads complete, not when a review file is deleted. Phase 0 finds open work by an incomplete findings cookie; the Phase 4 implementation-ready gate now also requires the findings cookie, and rerun-the-readiness-rubric-on-expanded-scope is folded into that gate (a scope-expanding response must file new obligations as findings or decisions before claiming =Ready=).
+- *Why:* Deleting the review file left the iteration-history =Artifacts= line dangling and lost the verbatim review; keeping the file collided with this workflow's file discovery and its "no review files remain" done-condition. Craig's call: incorporate the review into the document, reusing the decisions machinery so the readiness signal is a cookie. The scope-expansion rerun closes a real gap — a response can resolve every finding and still introduce unreviewed obligations.
+- *Artifacts:* Paired spec-review.org edits in the same commit. Inbox handoffs =2026-06-20-2339-from-home-spec-response-readiness-gate-proposal.org= and =2026-06-21-0156-from-home-companion-to-tonight-s-spec-response.org=.
diff --git a/claude-templates/.ai/workflows/spec-review.org b/claude-templates/.ai/workflows/spec-review.org
index d956f00..0da8e65 100644
--- a/claude-templates/.ai/workflows/spec-review.org
+++ b/claude-templates/.ai/workflows/spec-review.org
@@ -5,9 +5,9 @@
* Overview
-The spec-review workflow evaluates a feature/specification document before implementation and decides one thing: can an engineer implement it confidently, test it thoroughly, and ship behavior that matches the user's mental model? If yes, say so and stop. If no, write a review file next to the spec naming every blocking gap and the concrete change that closes it.
+The spec-review workflow evaluates a feature/specification document before implementation and decides one thing: can an engineer implement it confidently, test it thoroughly, and ship behavior that matches the user's mental model? If yes, say so and stop. If no, record every blocking gap and the concrete change that closes it as findings in the spec's own =* Review findings= section.
-This is the *reviewer* side of a pair. Its counterpart is the spec-response workflow, which the spec's author runs to fold a review back in. The contract between them is the review file: =<spec-basename>-review.org= (e.g. =docs/issue-query-spec.org= → =docs/issue-query-spec-review.org=). spec-review produces it; spec-response consumes it.
+This is the *reviewer* side of a pair. Its counterpart is the spec-response workflow, which the spec's author runs to disposition the findings. The contract between them lives *in the spec*: a =* Review findings= section carrying one =TODO= task per finding, with a =[/]= cookie — the same shape the spec's =* Decisions= section already uses. spec-review writes the findings; spec-response completes them. No separate review file is written, so nothing dangles when a review is processed and the full review/response trail stays in the spec.
The goal is not to prove the spec is clever. It is to leave the implementer with *fewer* hidden decisions, not more prose.
@@ -28,7 +28,7 @@ A review is complete when:
1. *The implementation-readiness gate has been evaluated* and a rubric label assigned (=Ready= / =Ready with caveats= / =Not ready= / =Needs research=).
2. *If ready:* the user is told plainly ("This spec is implementation-ready. I have no further blocking review notes."), and the review stops — no churn for its own sake.
-3. *If not ready:* a =<spec>-review.org= file is written next to the spec, in the standard structure, with every finding specific and actionable (current behavior named, risk explained, change recommended, blocking-or-not stated).
+3. *If not ready:* findings are recorded in the spec's =* Review findings= section as =TODO= tasks (one per finding, =[/]= cookie on the heading), each specific and actionable (current behavior named, risk explained, change recommended, blocking-or-not stated).
4. *The spec's review history is updated* with who reviewed it, when, which iteration it was, what changed or was recommended, and why.
5. *Deferred work is logged* to =todo.org= (v1 = =[#B]=, vNext/someday = =[#D]=), not left only in chat.
6. *Implementation tasks are enumerated* — the spec's =Implementation phases= section is lifted into a drop-in =todo.org= block (one entry per phase plus a test-surface entry), or, if the spec has no phase decomposition, that gap is raised as a finding.
@@ -50,6 +50,11 @@ Run it *early* — design review exists to catch viability problems and costly m
Before Phase 1, verify the file under review ends with =-spec.org=. Every design, decision, or planning document under a project's =docs/= directory carries that suffix as its identifier. The =.org= extension alone is not enough because =docs/= holds non-spec org files too (tutorials, frozen inventories, reference material).
+*Location expectation (docs-lifecycle convention).* Formal specs live in =docs/specs/=. Whether that's enforced depends on whether the project has run its one-time =spec-sort= retrofit:
+
+- =:LAST_SPEC_SORT:= present in =.ai/notes.org= Workflow State → the project has sorted; a =-spec.org= file outside =docs/specs/= fails this precondition. Surface it: "this spec sits outside docs/specs/ — move it (and update inbound links) before review."
+- Marker absent → legacy locations (=docs/= root, =docs/design/=) stay reviewable; add one nudge line to the review output ("this project's docs pile has never been spec-sorted — say 'run spec-sort' to sort it") and proceed. No legacy spec is ever unreviewable during the transition.
+
If the file does not end with =-spec.org=, stop immediately and surface the mismatch:
#+begin_example
@@ -93,19 +98,19 @@ Mark the spec implementation-ready only if *all* of these hold:
- The plan can be phased without shipping broken intermediate states, and phases are small enough to reach a clean stopping point in one focused work session.
- External API assumptions are verified or explicitly listed as prerequisites.
-If all true → tell the user it's ready and stop unless they ask for more. If any false → continue and write the review file. A "ready" at this phase is provisional; confirm it at Phase 3 after the code read.
+If all true → tell the user it's ready and stop unless they ask for more. If any false → continue and record findings (Phase 5). A "ready" at this phase is provisional; confirm it at Phase 3 after the code read.
** Phase 2: Required reading order
Never review a spec in isolation.
1. *Read the existing implementation first.* The code paths the spec would touch: public commands and entry points, internal helpers/boundaries, current data representation, persistence/write-back, async/sync, caching, error handling, existing tests, naming/style. Capture current-state facts with function names and file paths. Don't recommend designs that ignore how the package works today.
-2. *Read related specs and task tracking.* Companion specs, relevant =todo.org= tasks, README/testing docs, prior review files. Record which tasks the spec absorbs, which stay separate, which decisions are already made, which are still open.
+2. *Read related specs and task tracking.* Companion specs, relevant =todo.org= tasks, README/testing docs, prior reviews (in each spec's =* Review findings= and =Review and iteration history=). Record which tasks the spec absorbs, which stay separate, which decisions are already made, which are still open.
3. *Read the target spec end to end — twice.* First for its problem/behavior/phases/assumptions; second looking only for gaps. The second read asks: "What would an implementer still have to invent?"
** Phase 3: Re-run the gate (authoritative)
-After reading code and spec, re-run the Phase 1 gate — this is the pass that counts, because now you can actually judge the items that needed the code: architecture fit, API verification, integration points. If now ready, don't manufacture churn. If not, write the review file.
+After reading code and spec, re-run the Phase 1 gate — this is the pass that counts, because now you can actually judge the items that needed the code: architecture fit, API verification, integration points. If now ready, don't manufacture churn. If not, record findings (Phase 5).
** Phase 4: Evaluate across dimensions
@@ -128,6 +133,8 @@ Work the spec against these. Each is a source of concrete findings, not a box to
- *Performance & scale.* Expected counts (issues/comments/labels/teams/projects/views)? Server-side filtering where possible? Bounded, visible pagination? Cached name→ID lookups? Sync calls in the command path acceptable? Could a save hook or whole-file scan make N network calls? Rendering linear? Full-file rewrites avoided? Long-running operations async/cancellable/observable? Is concurrency/queueing/backpressure defined? Are high-output process filters throttled and cheap? Is progress/ETA exposed only when defensible, and are hung/stalled operations detectable and killable? Identify UI freezes, repeated network calls, unbounded pagination — without premature optimization.
- *Security & privacy.* API keys safe? Debug logs leaking secrets or private issue text? Confirmations before mutating shared workspace objects? Personal vs shared distinguished? Local files holding sensitive descriptions/comments? Anything to redact from messages/logs? Any work-tracker integration may handle private company data.
- *UX & accessibility.* Discoverable commands? Recoverable mistakes? Prompts ordered to the task? Safe, useful defaults? Informative-not-noisy status messages? Does the UI avoid implying unsupported actions are supported? Match the upstream product's permissions/concepts? Are customizations named in user language, with clear defaults and docstrings? For Emacs packages, command names, completion candidates, buffer layout, defcustom names, and message wording *are* the UX.
+- *Operational-panel UI traps.* Applies when the spec covers a user-facing panel, dialog, or control surface; skip otherwise. Lists that mix saved, current, and generated items must name each item's source. Refresh or scan actions must not gate data that could be shown immediately. Add-forms must not ask the user to retype values the system already discovered. Destructive confirmations read in future tense before the action and verified-result tense after it. Diagnostics, performance, logging, and repair affordances are reviewed as one coherent flow before extra pages or buttons are added. A popup launched from a bar, tray, or tool surface should visually belong to that launcher. (Promoted from archsetup's Waybar network-panel review, 2026-06-30.)
+- *Prototype process for non-trivial UI.* Applies when the deliverable is a real UI (a panel, a multi-control surface, an interacting visual layout — not a single dialog or CLI flag); skip otherwise. Verify the =claude-rules/ui-prototyping.md= process ran: category research is cited in Goals/Design, the final prototype is linked in the design section, a =Prototype iterations= subsection under the status heading lists every pass, and each UI design decision is backed by a prototype it was seen working in rather than asserted on the page. A non-trivial-UI spec with decisions but no prototype evidence is a =:blocking:= finding.
- *Test strategy and coverage.* Characterization tests before behavior changes? Pure functions to unit-test? API responses needing fixtures? Command flows needing stubs? Regression tests for prior bugs? Boundary/error cases? What's covered elsewhere and shouldn't be re-tested? Which existing tests must change? How is coverage generated, summarized, and used to find untested/refactor-worthy code? Prefer tests that lock contracts: representation shape, query compilation, sync no-op, conflict refusal, pagination, dirty-buffer protection, log redaction, and long-running/slow-operation behavior via fakes rather than flaky live dependencies.
- *Observability & operations.* How does a user see what the package is doing? Progress messages for long ops? Useful, safe debug logging? Are logs structured enough to isolate issues from a bug report? Are commands provided to inspect/clear caches, test connectivity, diagnose backends/tools, copy redacted debug info, or reproduce command invocations? How are terminal states discovered: completion, failure, partial success, stalled/hung, cancelled, cleanup-unverified, and "needs user action"? Does the product notify only when useful, avoid noisy success spam, and keep non-success states visible until acknowledged? For generated org files, headers should often carry source, filter/view name, refresh time, count, truncation state.
- *Comparable-product sentiment.* When there are obvious adjacent products, research what users love and hate about them from official docs plus current community reports. Do not cargo-cult their feature set; translate findings into the spec's scope. For each loved behavior, say whether the spec provides it, intentionally omits it, or defers it. For each hated behavior, say whether the spec avoids, resolves, inherits, or accepts it.
@@ -136,66 +143,41 @@ Work the spec against these. Each is a source of concrete findings, not a box to
- *Development tooling.* Does the repo give contributors obvious commands for setup, fast tests, specific tests, compile, lint, coverage, cleanup, slow/manual tests, and release checks? Are optional/live tests gated by explicit environment variables? Is the Makefile/script surface consistent with sibling projects?
- *Small enhancement radar.* Are there low-complexity, high-value affordances already provided by the platform that should be surfaced now or explicitly deferred? Examples: archive/compress commands in file managers, built-in history, previews, diagnostics, or doctor commands. Keep the hot path simple; capture the opportunity rather than accidentally losing it.
-** Phase 5: Write the review file
+** Phase 5: Record findings in the spec
-Use this structure for =<spec-basename>-review.org= unless the spec calls for something different:
+Findings live in the spec, not a sibling file. Add (or append to) a =* Review findings= section near the spec's =* Decisions= section, with a =[/]= cookie on the heading. Each finding is a =** TODO= task: the heading is the smallest noun phrase naming the gap; the body names current behavior, the risk, and the recommended change. Tag a blocking (high-priority) finding =:blocking:= — it holds the rubric at =Not ready= until dispositioned; leave non-blocking findings untagged. Findings accumulate across review rounds the way decisions do, and the responder completes each one in place (Phase 4 of spec-response), so the section becomes the full review/response trail.
#+begin_src org
-,#+TITLE: Review: <Spec Title>
-,#+AUTHOR: <reviewer>
-,#+DATE: <date>
-,#+STARTUP: showall
-
-,* Scope reviewed
-What code, tests, docs, and specs you read.
-
-,* Implementation-readiness
-Whether the spec is ready. If not, summarize the blockers.
-
-,* Overall assessment
-The short senior-engineering read: what's right, what's risky, what must be clarified.
-
-,* High-priority findings
-Concrete headings. Each: why it matters and what to change.
-
-,* Medium-priority findings
-Important improvements that shouldn't block all progress.
-
-,* UX observations
-,* Architecture observations
-,* Robustness and performance observations
-,* Test strategy recommendations
-Specific test cases, not generic "add tests".
-,* Documentation and tooling recommendations
-README/user/developer docs, Makefile/package scripts, coverage, debug tools, and customization surface.
-
-,* Suggested spec edits
-Concrete edits to make the spec implementation-ready.
-
-,* Agreed decisions
-Answers reached during review. Omit if none.
-
-,* Open questions
-Only questions that truly block or materially affect implementation.
-
-,* vNext candidates
-Deferred features to capture in task tracking.
+,* Review findings [/]
+,** TODO Comment edit-back is undefined :blocking:
+The spec says fetched comments render as subheadings but doesn't define whether
+editing one syncs back. Linear only lets users edit their own comments. V1 should
+treat fetched comments as remote-owned display content and support only adding new
+comments; editing own comments can be vNext. (blocking)
+,** TODO Empty result and fetch error render identically
+A failed fetch and a successful-but-empty fetch produce the same buffer, so the
+user can't tell "no issues" from "the query broke." Define a distinct empty-state
+message. (non-blocking)
#+end_src
+Where the old review-file sub-sections go now: the scope-reviewed and overall-assessment narrative goes in the =Review and iteration history= entry (Phase 6); suggested spec edits are the recommended-change line in each finding's body; agreed decisions flip the spec's own =* Decisions= tasks; open questions are =:blocking:= findings or open decisions; vNext candidates are logged to =todo.org= as =[#D]= (Phase 6). The Phase 4 review dimensions are where findings come *from* — not headings to reproduce in the spec.
+
** Phase 6: Assign the rubric and update tracking
Assign one label consistently:
-- =Ready= — no blocking open questions; implementation can start. Requires no decision in the spec's =* Decisions= section to still be =TODO= (the =[/]= cookie reads complete; =SUPERSEDED= and =CANCELLED= count as resolved) — a decision still =TODO= holds the rubric at =Not ready=, or =Ready with caveats= if the author consciously accepts and records the risk.
+- =Ready= — no blocking open questions; implementation can start. Requires both cookies complete: no decision in =* Decisions= and no =:blocking:= finding in =* Review findings= still =TODO= (the =[/]= cookies read complete; =SUPERSEDED=/=CANCELLED= and a completed or rejected finding count as resolved) — a still-=TODO= decision or =:blocking:= finding holds the rubric at =Not ready=, or =Ready with caveats= if the author consciously accepts and records the risk. A non-blocking finding left =TODO= is author's discretion and does not hold the rubric.
- =Ready with caveats= — can start if the caveats are accepted and tracked.
- =Not ready= — blocking ambiguity / missing decisions would force implementers to invent product behavior.
- =Needs research= — external API/library/platform assumptions must be verified first.
The most useful reviews move a spec from =Not ready= to =Ready with caveats= or =Ready= once decisions are captured.
+*The =Ready= verdict flips the spec's lifecycle status.* spec-review owns the =DRAFT= → =READY= transition (docs-lifecycle convention): on assigning =Ready= (or =Ready with caveats= the author accepts), update the spec's top-level status heading keyword to =READY=, add a dated history line under it naming the review that passed, and set the Metadata =Status= mirror to =ready= — three lines, one file. Any other rubric label leaves the keyword where it stands (a re-review that finds new blockers on a =READY= spec demotes it back to =DRAFT= the same three-line way, with the reason in the history line).
+
Finding severity maps to blocking power: *high-priority findings block =Ready=* — they hold the rubric at =Not ready= (or =Ready with caveats= if the author accepts and tracks them) until dispositioned; *medium-priority findings are the author's discretion* and don't block. State the blocking status on each finding so the author running spec-response knows which ones gate the rubric.
-Then update the spec's review history. Specs should carry a bottom section named =Review and iteration history= (or the nearest existing equivalent) that tracks each material author/reviewer pass. Add a concise entry for this review even when the spec is ready and no review file is written.
+Then update the spec's review history. Specs should carry a bottom section named =Review and iteration history= (or the nearest existing equivalent) that tracks each material author/reviewer pass. Add a concise entry for this review even when the spec is ready and no findings are recorded.
Each entry is an org subheading with a compound id followed by three body fields.
@@ -207,27 +189,11 @@ Body fields:
- *What changed or was recommended:* high-signal summary, not a duplicate of the whole review.
- *Why:* the decision pressure or rationale that caused the contribution.
-- *Artifacts:* links to the review file, response/disposition section, commits, task IDs, or source checks when useful.
+- *Artifacts:* links to the relevant findings, commits, task IDs, or source checks when useful.
If the spec has no such section, add it at the bottom. Keep the history short and cumulative; it is provenance for future readers, not a session transcript.
-*Emit implementation tasks (drop-in for =todo.org=).* Read the spec's =Implementation phases= section and turn it into a paste-ready block in the review file, under a heading =Implementation tasks (drop-in for todo.org)=. One =** TODO= entry per phase, plus a final entry for the test surface. The point: the handoff to whoever implements is one paste, not a re-read of the spec, and a spec that can't be decomposed into phases fails this step, surfacing a shape problem before =Ready=.
-
-Per-phase entry, following =todo-format.md= (terse heading names the phase; body holds the one-line deliverable plus a pointer back to the spec; tags on the heading):
-
-#+begin_example
-** TODO [#B] <phase name — smallest noun phrase> :feature:
-<what this phase delivers, one line>. Spec: [[file:<spec path>]] (Implementation phases, phase N).
-#+end_example
-
-Final test-surface entry, mirroring the spec's =Acceptance criteria= when present:
-
-#+begin_example
-** TODO [#B] <feature> — test surface :test:
-Unit: <...>. Integration: <...>. E2e / manual-verify: <acceptance criteria as checkable items>. Spec: [[file:<spec path>]] (Acceptance criteria).
-#+end_example
-
-Priority and tags follow the deferred-work rule below. Emit the block in the review file; the author pastes it into =todo.org= during spec-response, or you log it directly when you're also closing the loop. If the spec has no =Implementation phases= section, don't invent one — that absence is the finding, and the step becomes the prompt to ask the author to add a phase decomposition before the spec can be =Ready=.
+*Check the spec decomposes into phases.* A =Ready= spec needs an =Implementation phases= section an implementer can turn into one task per phase plus a test surface. Confirm it's present and decomposable — each phase small enough to reach a clean stopping point in one focused session, with no broken intermediate states. If it's missing or can't be phased, file that as a =:blocking:= finding; don't invent the phases. The phase-to-task breakdown itself is spec-response's job (its Phase 6 reads =Implementation phases= directly once the author confirms =Ready=); the reviewer only verifies the section exists and is sound.
Then log deferred work to =todo.org=: v1 implementation = =[#B]= (unless urgent or speculative); vNext/someday = =[#D]=. Tag =:feature:= / =:bug:= / =:refactor:= / =:test:= / =:quick:= / =:solo:= only when accurate. Don't leave important deferred decisions only in chat.
@@ -256,8 +222,11 @@ Every material comment should be tagged by force: blocking, should-fix, or optio
** Make feedback author-usable
Review comments should be specific, neutral, and actionable: quote or name the spec behavior, explain the risk, recommend the smallest concrete change, and say how the author can verify the fix. Avoid personal language, rhetorical questions, vague "this needs work" comments, and comments that require the author to infer the desired edit.
+** Keep review and response roles explicit
+If the user asks for review plus "enhance the spec" in the same turn, produce the findings first. Make only low-risk provenance and tracking edits unless the user clearly wants the reviewer to respond too. Don't silently resolve product decisions on the author's behalf — a proposed default belongs in a finding until it's accepted, modified, or rejected.
+
** Preserve iteration provenance
-Future reviewers and implementers need to know not just the current decision, but how the spec got there: how many review/response loops happened, who contributed, what they changed or recommended, and why. Keep that record in the spec itself under =Review and iteration history= so the trail survives deleted review files, chat loss, and agent handoffs.
+Future reviewers and implementers need to know not just the current decision, but how the spec got there: how many review/response loops happened, who contributed, what they changed or recommended, and why. Keep that record in the spec itself under =Review and iteration history= so the trail survives chat loss and agent handoffs.
** Be strict about ownership
Especially for org-mode features: a user treats visible text as editable unless the representation says otherwise. Make generated-vs-editable explicit.
@@ -265,6 +234,9 @@ Especially for org-mode features: a user treats visible text as editable unless
** Never depend on an unverified API shape
If the spec assumes fields/mutations/enums, they're verified against current schema/docs/live responses, or listed as a research prerequisite. =Needs research= is a real, useful verdict.
+** Source external-dependency checks in the finding
+When a finding turns on a current external-dependency fact (release version, API capability, platform behavior, package availability, hosted-service terms), cite the checked source in the finding body. Stale dependency assumptions are common, and the next reviewer needs to tell "verified this pass" from "remembered from prior context."
+
** Favor small pure cores and thin IO layers
Push findings toward separable, unit-testable pure functions surrounded by thin command/transport layers.
@@ -354,3 +326,8 @@ Sources:
- *What:* Two refinements to the same-day decisions convention after Craig's review: the gate item and =Ready= rubric now read "no decision is still =TODO=" with =SUPERSEDED= and =CANCELLED= counting as resolved (spec-create's template defines them as done-class keywords via a =#+TODO:= header), and a spec still on the retired =State:= field model explicitly fails the gate item until converted — closing the vacuous-pass hole on old specs.
- *Why:* Review of the freshly-landed convention flagged that TODO/DONE alone lost the old model's superseded state and that the gate as written would silently pass a spec with no decision tasks at all. Craig chose the two done-class keywords and the auto-added =#+TODO:= header (the in-file header is what makes custom keywords portable).
- *Artifacts:* Paired spec-create.org edits (keyword scheme + template header) in the same commit.
+
+** 2026-06-21 Sun @ 23:16:06 -0400 — Claude Code (rulesets) — responder
+- *What:* Moved findings from a sibling =<spec>-review.org= file into the spec itself. Findings are now =** TODO= tasks under a =* Review findings= section with a =[/]= cookie, mirroring =* Decisions=; =:blocking:= marks high-priority. Phase 5 records findings in the spec instead of writing a review file; the Phase 6 =Ready= rubric gates on both the decisions and the findings cookie; the implementation-task drop-in (which lived in the review file) is gone, leaving the reviewer to verify the spec decomposes into phases and spec-response to build the breakdown. Also added two reviewer-practice principles harvested from a home spec-review: keep review and response roles explicit, and source external-dependency checks in the finding.
+- *Why:* The delete-the-review-file convention left the iteration-history =Artifacts= line dangling and dropped the verbatim review; keeping the file instead collided with spec-response's file discovery and its "no review files remain" done-condition. Craig's call: incorporate the review into the document, reusing the decisions machinery so the readiness signal is a cookie, not a file's presence or absence. The role-explicit and source-checking practices came in from the home finance-report spec via inbox handoffs.
+- *Artifacts:* Paired spec-response.org edits in the same commit. Inbox handoffs =2026-06-20-2339-from-home-spec-response-readiness-gate-proposal.org=, =2026-06-21-0156-from-home-companion-to-tonight-s-spec-response.org=, and the home-edited =2026-06-21-0156-from-home-spec-review.org=.
diff --git a/claude-templates/.ai/workflows/startup.org b/claude-templates/.ai/workflows/startup.org
index 59c9c54..2262eea 100644
--- a/claude-templates/.ai/workflows/startup.org
+++ b/claude-templates/.ai/workflows/startup.org
@@ -1,5 +1,5 @@
#+TITLE: Startup Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-04-25
* Summary
@@ -10,8 +10,8 @@ The workflow is structured into four phases. *Phase A.0* is a sequential pre-fli
Quick contract — runs / produces:
- *Phase A.0* (sequential): refresh rulesets, then the project repo.
-- *Phase A* (parallel batch): timestamp, session-context check, guarded =.ai/= sync, recent sessions, inbox-status, cross-agent status, notes.org, staleness, language-bundle freshness.
-- *Phase B* (parallel batch): read the crash-recovery anchor if present, the recent session summaries, new inbox items, pending cross-agent messages.
+- *Phase A* (parallel batch): timestamp, session-context check, guarded =.ai/= sync, recent sessions, inbox-status, notes.org, staleness, language-bundle freshness.
+- *Phase B* (parallel batch): read the crash-recovery anchor if present, the recent session summaries, new inbox items.
- *Phase C* (interactive): surface findings, process the inbox, run project startup-extras, ask priorities.
* Execution
@@ -29,10 +29,16 @@ Inside a rulesets session, the project-repo refresh below covers this — the ru
#+begin_src bash
rs="$HOME/code/rulesets"
if [ -d "$rs/.git" ]; then
- if (cd "$rs" && git diff --quiet --ignore-submodules HEAD -- 2>/dev/null); then
+ gate="$rs/claude-templates/bin/git-worktree-gate"
+ if [ -x "$gate" ] && "$gate" sync-safe "$rs"; then
+ (cd "$rs" && git pull --ff-only origin main 2>&1) | tail -3
+ elif [ ! -x "$gate" ] \
+ && (cd "$rs" && git diff --quiet --ignore-submodules HEAD -- 2>/dev/null); then
+ # Bootstrap fallback for a checkout old enough not to have the shared
+ # gate yet. The pull that follows installs it for subsequent starts.
(cd "$rs" && git pull --ff-only origin main 2>&1) | tail -3
else
- echo "rulesets: dirty working tree — using as-is, skipping pull"
+ echo "rulesets: changes beyond untracked inbox deliveries — using as-is, skipping pull"
fi
else
echo "rulesets: not a git checkout — skipping"
@@ -40,10 +46,12 @@ fi
#+end_src
Behavior:
-- *Clean working tree* → fast-forward pull. =git pull --ff-only= refuses any merge or rebase, so the operation is either a no-op (already current) or a clean advance.
-- *Dirty working tree* → skip the pull. Don't auto-stash and don't auto-merge — those would either lose work or invite conflicts at the worst possible moment (session start).
+- *Clean working tree, or untracked deliveries only beneath =inbox/=* → fast-forward pull. =git pull --ff-only= refuses any merge or rebase, so the operation is either a no-op (already current) or a clean advance. Inbox files are queue input, not source-tree work, and do not block other projects from receiving rulesets updates.
+- *Any staged or tracked change, dirty submodule, Git operation in progress, or untracked file outside =inbox/=* → skip the pull. Don't auto-stash and don't auto-merge — those would either lose work or invite conflicts at the worst possible moment (session start).
- *Non-fast-forward history* → =--ff-only= aborts with an error. Surface that to the user; the rsync still proceeds against the working tree as-is.
+*Template-freshness policy (applies to every dirty-check in the synced workflows).* The shared =git-worktree-gate sync-safe= policy is the source of truth: untracked files beneath =inbox/= and gitignored files do not block a pull, fast-forward, or monitoring gate; every other staged, tracked, untracked, submodule, or in-progress-operation state does. Projects must not fall behind merely because somebody sent them a task, but an arbitrary scratch file is not silently treated as safe. One deliberate exception remains: the rsync WIP-guard below is narrower than the repository gate and counts untracked files within rulesets' own synced source paths, because an untracked half-written template is exactly the WIP it exists to hold back.
+
*** Install rulesets symlinks into ~/.claude (idempotent)
A skill, rule, or bin script added to rulesets and pushed reaches each machine's *files* on the next pull, but not its =~/.claude= *symlink* — =make install= only links what isn't already linked, and =git pull= doesn't run it. So a newly-added skill stays silently uninstalled until someone re-runs =make install= by hand. The flush skill sat in that gap from 2026-06-02 until a manual install on 2026-06-05. Running =make install= here, right after the rulesets pull, closes it: "add a skill, commit, push" becomes enough for it to reach every machine on the next session.
@@ -72,8 +80,11 @@ if [ -d .git ]; then
current=$(git symbolic-ref --short HEAD 2>/dev/null)
dirty=0
- if ! git diff --quiet --ignore-submodules HEAD -- 2>/dev/null \
- || [ -n "$(git status --porcelain --untracked-files=no)" ]; then
+ gate="$HOME/code/rulesets/claude-templates/bin/git-worktree-gate"
+ if [ -x "$gate" ]; then
+ "$gate" sync-safe "$PWD" >/dev/null 2>&1 || dirty=1
+ elif ! git diff --quiet --ignore-submodules HEAD -- 2>/dev/null \
+ || [ -n "$(git status --porcelain --untracked-files=no)" ]; then
dirty=1
fi
@@ -105,8 +116,8 @@ fi
#+end_src
Behavior, per branch:
-- *Behind only, current branch, clean tree* → =git merge --ff-only= advances HEAD.
-- *Behind only, current branch, dirty tree* → fetched but not advanced. Surface so Craig can ff manually after dealing with the dirty state.
+- *Behind only, current branch, sync-safe tree* → =git merge --ff-only= advances HEAD. An untracked =inbox/= delivery is sync-safe.
+- *Behind only, current branch, sync-blocking tree* → fetched but not advanced. Surface so Craig can ff manually after dealing with the reported state.
- *Behind only, non-checkout branch* → =git fetch . upstream:branch= advances the ref without touching the working tree.
- *Diverged* (ahead and behind) → leave alone. Surface for Craig to resolve. Don't auto-rebase or auto-merge.
- *Ahead only* or *up to date* → silent no-op.
@@ -124,7 +135,7 @@ These calls have no dependencies on each other. Issue them all together in one m
sc=$(.ai/scripts/session-context-path 2>/dev/null || echo .ai/session-context.org)
[ -e "$sc" ] && echo "present: $sc" || echo "absent: $sc"
#+end_src
-3. *Sync =.ai/= from templates — but only when the synced source paths in rulesets are clean.* Guard the three rsyncs behind a check that =claude-templates/.ai/{protocols.org,workflows/,scripts/}= have no uncommitted changes. Otherwise Phase A copies in-flight rulesets WIP (tracked edits or new untracked files) into this project's =.ai/workflows/= and =.ai/scripts/=, where it shows up as drift the user didn't author. Skipping once is cheap — the next session with rulesets clean catches up. The check is scoped to the synced paths, so unrelated rulesets dirt (a stray =session-context.org=, scratch files) doesn't needlessly block the sync.
+3. *Sync =.ai/= from templates — but only when the synced source paths in rulesets are clean.* Guard the three rsyncs behind a check that =claude-templates/.ai/{protocols.org,workflows/,scripts/}= have no uncommitted changes. Otherwise Phase A copies in-flight rulesets WIP (tracked edits or new untracked files) into this project's =.ai/workflows/= and =.ai/scripts/=, where it shows up as drift the user didn't author. Skipping once is cheap — the next session with rulesets clean catches up. The check is scoped to the synced paths, so unrelated rulesets dirt (a stray =session-context.org=, scratch files) doesn't needlessly block the sync. A second guard skips the same rsyncs when the *project* branch is behind its upstream (=git rev-list --left-right --count @{u}...HEAD= with =behind > 0=): syncing templates onto a stale committed =.ai/= baseline measures the diff against old content, so it comes out huge and conflicts when the branch later reconciles to upstream, whose history already carries the newer templates. It composes with the rulesets-clean guard — a stable rulesets source and a current project branch are both required before the sync runs.
#+begin_src bash
rs="$HOME/code/rulesets"
@@ -132,26 +143,75 @@ These calls have no dependencies on each other. Issue them all together in one m
claude-templates/.ai/protocols.org \
claude-templates/.ai/workflows/ \
claude-templates/.ai/scripts/ 2>/dev/null)
- if [ -z "$synced_dirty" ]; then
+ # Skip the sync when the project branch hasn't reached its upstream. Syncing
+ # templates onto a behind baseline measures the diff against stale committed
+ # .ai/, producing confusing drift that conflicts when the branch reconciles —
+ # the newer .ai/ is already in upstream. behind==0 (up-to-date or ahead-only)
+ # means HEAD contains all of upstream, so the baseline is current. No upstream
+ # (new/unpushed branch) → rev-list fails → proj_behind stays 0, sync runs.
+ proj_behind=0
+ if [ -d .git ]; then
+ counts=$(git rev-list --left-right --count '@{u}...HEAD' 2>/dev/null) \
+ && [ "$(printf '%s' "$counts" | cut -f1)" -gt 0 ] 2>/dev/null \
+ && proj_behind=1
+ fi
+
+ if [ -n "$synced_dirty" ]; then
+ echo "rulesets has uncommitted changes under the synced template paths — skipping .ai/ sync this session (catches up when rulesets is clean):"
+ echo "$synced_dirty" | sed 's/^/ /'
+ elif [ "$proj_behind" -eq 1 ]; then
+ echo "project branch is behind upstream — skipping .ai/ sync this session (templates never land on a stale baseline; the sync runs once the branch is current)"
+ else
rsync -a "$rs/claude-templates/.ai/protocols.org" .ai/protocols.org
rsync -a --delete "$rs/claude-templates/.ai/workflows/" .ai/workflows/
rsync -a --delete --exclude='__pycache__' --exclude='.pytest_cache' --exclude='*.pyc' \
"$rs/claude-templates/.ai/scripts/" .ai/scripts/
echo ".ai/ synced from templates"
- else
- echo "rulesets has uncommitted changes under the synced template paths — skipping .ai/ sync this session (catches up when rulesets is clean):"
- echo "$synced_dirty" | sed 's/^/ /'
fi
#+end_src
4. =\ls -t .ai/sessions/ 2>/dev/null | head -5= — list 5 most recent session files. The backslash bypasses any =ls= alias in the user's profile. Without it, bare =ls -t= silently returns no output under =exa= (a common =ls= replacement) — which makes a sessions directory full of files look empty, and the agent then skips Phase B step 2.
5. =\ls -la inbox/ 2>/dev/null= — inventory the inbox. Same reason for the backslash escape, applied uniformly across the Phase A =ls= calls.
-6. =cross-agent-status 2>/dev/null || true= — snapshot of pending cross-agent messages across local projects. This is layer A of the cold-start design from =cross-agent-comms.org=: pending messages from other agents (delivered while no session was active here) get surfaced on session start. The =|| true= keeps Phase A from failing if =cross-agent-status= isn't installed yet — older projects without the script still boot cleanly. If HALT is active, =cross-agent-status= prints a banner; surface that prominently in Phase C.
-7. Read =.ai/notes.org= — Project-Specific Context, Active Reminders, Pending Decisions sections (skip About This File).
-8. Read =.ai/project-workflows/startup-extras.org= if it exists.
-9. =[ -f todo.org ] && .ai/scripts/task-review-staleness.sh todo.org 7 || true= — count top-level tasks overdue for review (the daily task-review habit's startup nudge). The =[ -f todo.org ]= guard skips projects without a root todo.org; =|| true= keeps Phase A from failing if the script isn't synced yet. Threshold 7 days is one review cycle of slack — softer than the wrap-up health check's 30-day alarm.
-10. =bash ~/code/rulesets/scripts/sync-language-bundle.sh "$PWD" 2>/dev/null || true= — language-bundle freshness for the current project. Fingerprint-detects which bundle (if any) the project has, auto-fixes drifted rulesets-owned files (=.claude/rules/*.md=, =.claude/hooks/*=, =githooks/*=), and surfaces drift in =settings.json= without writing it (a project may have customized it). =CLAUDE.md= is deliberately left untracked — it's seed-only in =install-lang= and project-owned afterward, mirroring how =diff-lang= skips it. Quiet when there's no bundle or everything's clean. Hardcodes the rulesets path because =languages/= is the canonical source and lives only there — the same absolute-path dependency the rsyncs already carry. =|| true= keeps Phase A from failing on older checkouts where the script isn't present yet. The =.ai/= rsyncs and this call write to disjoint paths (=.ai/= vs =.claude/=/=githooks/=), so the batch stays parallel-safe.
-11. =[ -f "$HOME/org/roam/inbox.org" ] && grep -cE '^\*\* ' "$HOME/org/roam/inbox.org" || true= — count items in the roam global inbox (=~/org/roam/inbox.org=), the inbox-zero startup nudge. Silent if the roam clone isn't on this machine. Phase C reads the file when the count is non-zero, splits total vs items related to this project, and surfaces the offer (see =inbox-zero.org=). Read-only; never files at startup.
+6. Read =.ai/notes.org= — Project-Specific Context, Active Reminders, Pending Decisions sections (skip About This File).
+7. Read =.ai/project-workflows/startup-extras.org= if it exists.
+8. =[ -f todo.org ] && .ai/scripts/task-review-staleness.sh todo.org 7 || true= — count top-level tasks overdue for review (the daily task-review habit's startup nudge). The =[ -f todo.org ]= guard skips projects without a root todo.org; =|| true= keeps Phase A from failing if the script isn't synced yet. Threshold 7 days is one review cycle of slack — softer than the wrap-up health check's 30-day alarm.
+9. =bash ~/code/rulesets/scripts/sync-language-bundle.sh "$PWD" 2>/dev/null || true= — language-bundle freshness for the current project. Fingerprint-detects which bundle (if any) the project has, auto-fixes drifted rulesets-owned files (=.claude/rules/*.md=, =.claude/hooks/*=, =githooks/*=), and surfaces drift in =settings.json= without writing it (a project may have customized it). =CLAUDE.md= is deliberately left untracked — it's seed-only in =install-lang= and project-owned afterward, mirroring how =diff-lang= skips it. Quiet when there's no bundle or everything's clean. Hardcodes the rulesets path because =languages/= is the canonical source and lives only there — the same absolute-path dependency the rsyncs already carry. =|| true= keeps Phase A from failing on older checkouts where the script isn't present yet. The =.ai/= rsyncs and this call write to disjoint paths (=.ai/= vs =.claude/=/=githooks/=), so the batch stays parallel-safe.
+10. =[ -f "$HOME/org/roam/inbox.org" ] && grep -cE '^\*\* ' "$HOME/org/roam/inbox.org" || true= — count items in the roam global inbox (=~/org/roam/inbox.org=), the roam-mode startup nudge. Silent if the roam clone isn't on this machine. Phase C reads the file when the count is non-zero, splits total vs items related to this project, and surfaces the offer (see =inbox.org= roam mode). Read-only; never files at startup.
+11. KB surface prep (the read + contribute startup nudges; see =docs/specs/2026-06-16-encourage-kb-contribution-spec.org=). Gated on the agent KB clone. Counts =:agent:= nodes, lists up to 5 whose content matches the current project basename (titles only; a few most-recent nodes as a fallback when nothing matches), and resolves the best-practices node path. Read-only; silent when the clone is absent. Phase C surfaces the relevant titles (consult) and the best-practices link (contribute).
+
+ The best-practices lookup matches the node's *filename*, not its content. A roam node's slug lives only in its filename, so the earlier content-grep (=rg -l 'agent-kb-best-practices'=) matched nothing and the contribute nudge silently pointed at an empty path in every project, every session, for as long as it shipped. =find= rather than a glob keeps the probe identical under bash and zsh (zsh aborts on an unmatched glob) — the same reason the spec-sort probe below uses =find=.
+
+ #+begin_src bash
+ ra="$HOME/org/roam/agents"
+ if [ -d "$ra" ]; then
+ proj=$(basename "$PWD")
+ echo "kb-total: $(rg -l '#\+filetags:.*:agent:' "$ra" 2>/dev/null | wc -l)"
+ echo "kb-bestpractices: $(find "$ra" -maxdepth 1 -name '*agent-kb-best-practices*.org' -print -quit 2>/dev/null)"
+ matches=$(rg -il "$proj" "$ra" 2>/dev/null | head -5)
+ [ -z "$matches" ] && matches=$(\ls -t "$ra"/*.org 2>/dev/null | head -3)
+ echo "kb-relevant-titles:"
+ for f in $matches; do rg -m1 '^#\+title:' "$f" 2>/dev/null | sed 's/^#+title:/ -/'; done
+ fi
+ #+end_src
+
+12. Spec-sort probe (the docs-lifecycle retrofit nudge; see the docs-lifecycle spec in =docs/specs/=). Read-only; prints one line when the project has an unsorted docs pile — a =docs/design/= directory or stray =docs/*-spec.org= root files — and no =:LAST_SPEC_SORT:= marker in =.ai/notes.org=. Silent for projects with nothing to sort or an already-stamped marker (the marker permanently clears it).
+
+ #+begin_src bash
+ { [ -d docs/design ] || [ -n "$(find docs -maxdepth 1 -name '*-spec.org' -print -quit 2>/dev/null)" ]; } \
+ && ! grep -qs ':LAST_SPEC_SORT:' .ai/notes.org \
+ && echo "spec-sort: unsorted docs present" || true
+ #+end_src
+
+ The stray-root check uses =find= rather than a glob so the probe behaves identically under bash and zsh (=compgen= is bash-only, and zsh aborts on an unmatched glob).
+
+13. Host-identity probe (see the host-identity rule in =claude-rules/=). Read-only; flags fixed machine-identity claims in the project's tracked/synced docs — the "This machine is ratio" trap, false on every machine but the one that wrote it. Silent when nothing matches.
+
+ #+begin_src bash
+ grep -inE '\b(this|the current) (machine|host|box|laptop|workstation) is ' \
+ CLAUDE.md .ai/notes.org 2>/dev/null | head -3 || true
+ #+end_src
+
+ Fleet descriptions ("the fleet is ratio and velox") and runtime derivations ("run =uname -n= to find the hostname") don't match — only current-identity assertions do. Fixture-verified under bash and zsh.
Notes on the rsync commands:
- Trailing slashes on both source and destination matter — they tell rsync to sync /contents/ rather than nest a directory inside.
@@ -159,6 +219,7 @@ Notes on the rsync commands:
- protocols.org is a single file, no =--delete= needed.
- The =scripts/= sync excludes Python build artifacts (=__pycache__/=, =.pytest_cache/=, =*.pyc=). Running rulesets' own pytest leaves these in =claude-templates/.ai/scripts/tests/=, and =rsync -a= copies by disk presence regardless of =.gitignore=, so without the excludes every consuming project's tree gets polluted with machine-specific cache files. The excludes also protect existing dest copies from =--delete= cleanup, so a project that already received the cache must remove it once by hand.
- The sync is guarded to skip when rulesets has uncommitted changes under the synced source paths. =rsync -a --delete= copies the working tree by disk presence, so without the guard a downstream session started while rulesets had in-flight WIP would pull that WIP into its =.ai/workflows/= and =.ai/scripts/=, surfacing as drift the user never authored (and tempting a fake "chore: sync .ai tooling" commit). The guard is scoped to the synced paths, not the whole repo, so unrelated rulesets dirt doesn't block the sync. From the jr-estate handoff 2026-05-29.
+- The sync is also guarded to skip when the *project* branch is behind its upstream (=proj_behind=). Phase A.0 correctly declines to fast-forward a diverged or behind-and-dirty branch, but the rsync would then land templates on the stale committed =.ai/= baseline — a huge diff measured against old content that conflicts once the branch reconciles to upstream's newer templates. Skipping is safe: the sync runs next session once the branch is current. Not an auto-discard — startup never =git checkout=s drift away, because a legitimate local stopgap in a synced file is indistinguishable from accidental drift by content alone (home reverted an intentional =flashcard-to-anki.py= fix this way on 2026-06-22). Prevention is safe; blind cleanup-after is not. Phase C's template-sync-churn safety net still surfaces any pre-existing dirt for a human decision. From the home handoff 2026-07-04.
- The sync touches only =protocols.org=, =workflows/=, and =scripts/=. The project-owned dirs =project-workflows/= and =project-scripts/= are deliberately *outside* the synced set, so a project's own workflows and scripts survive startup. This is why a project script that a workflow imports must live in =.ai/project-scripts/=, never =.ai/scripts/= — the latter is wiped to match the template by =--delete= on every startup. Naming: a script imported as a Python module needs an importable name (underscores, e.g. =zlibrary_api.py=); a CLI-invoked script can stay kebab-case like the template tooling (=cmail-action.py=).
Rationale: Every call in Phase A is read-only or writes to a distinct path. Running them sequentially wastes round-trips; running them in parallel gives Claude the complete starting picture in one round-trip.
@@ -170,7 +231,6 @@ These calls depend on Phase A outputs, but are independent of each other. Issue
1. *Read =.ai/session-context.org= if Phase A reported it exists.* The file is the crash-recovery anchor — if it's there, the previous session was interrupted and the context lives only in this file.
2. *Read each of the 5 most recent session files* from Phase A's =\ls -t .ai/sessions/= output. Read just the =* Summary= section of each — not the full file. The Summary gives Active Goal / Decisions / Data Collected / Findings / Files Modified / Next Steps. That's enough to pick up where things left off. Drill into a specific =* Session Log= later only if you need the /why/ or sequence on something. *If Phase A's listing came back empty, sanity-check with =\ls -la .ai/sessions/= before treating empty as definitive — sessions/ should normally be populated, and an empty result usually means the listing got swallowed somewhere, not that the directory is genuinely empty.*
3. *Read each new inbox file* from Phase A's =\ls -la inbox/= output. For =.eml= files, defer to Phase C — those need the extract script (below) rather than a raw Read.
-4. *Process pending cross-agent messages.* For each project with a pending count >0 in Phase A's =cross-agent-status= output (typically the current project; cross-project pending is surfaced too but only acted on if the user asks), run =cross-agent-recv <message-file>= on the file path =cross-agent-status= named. The script returns a structured decision (=process= / =dedup= / =query= / =reject=) per the protocol. For =process=, read the message body to determine the action. For =query=, prepare a clarifying reply. For =reject=, surface to user with the reason. For =dedup=, no action — silent retry already handled. Surface all decisions in Phase C alongside other findings.
Rationale: Reads are independent and benign. Batching them means the whole session-history view + inbox view lands in one round-trip instead of one per file.
@@ -184,7 +244,11 @@ This phase touches the user and runs sequentially:
- Mention Pending Decisions from notes.org.
- Briefly note significant template updates noticed during sync (new workflows, protocol changes).
- *Task-review nudge.* If the Phase A staleness count (step 11) is greater than zero, surface one line: "=<N>= top-level tasks unreviewed for >7 days — say 'let's do a task review' to run a cycle." If zero, say nothing.
- - *Roam inbox nudge.* If the Phase A roam-inbox count is greater than zero, read =~/org/roam/inbox.org=, split total vs items related to this project (claimed by the =<project>:= prefix, plus any unprefixed item whose topic plainly concerns this project), and surface one line: "Roam inbox: =<N>= total, =<M>= appear related to this project — say 'inbox zero' to file them." Offer it as a priority option; never auto-file. If the count is zero or the file is absent, say nothing. See =inbox-zero.org=.
+ - *Roam inbox nudge.* If the Phase A roam-inbox count is greater than zero, read =~/org/roam/inbox.org=, split total vs items related to this project (claimed by the =<project>:= prefix, plus any unprefixed item whose topic plainly concerns this project), and surface one line: "Roam inbox: =<N>= total, =<M>= appear related to this project — say 'inbox zero' to file them." Offer it as a priority option; never auto-file. If the count is zero or the file is absent, say nothing. See =inbox.org= roam mode.
+ - *KB consult nudge (read side).* If the Phase A KB-surface prep returned any =kb-relevant-titles=, surface one line listing them (capped 5): "KB lessons that may be relevant: =<title>=; =<title>=… — open the node before related work." The titles are declarative, so the list alone tells you whether to open one. Gated on the roam clone; silent when the clone is absent or nothing relevant surfaced. See the best-practices node and =knowledge-base.md=.
+ - *KB contribute nudge (write side).* Once per session, surface one line pointing at the best-practices node (the =kb-bestpractices= path from Phase A): "Learned something durable? See =<path>= for how to write a KB node — contributing cross-project facts is welcome (personal projects only; work/unknown projects never write per =knowledge-base.md=)." Light encouragement, never a gate. Gated on the roam clone; silent when absent.
+ - *Spec-sort nudge.* If the Phase A spec-sort probe printed =spec-sort: unsorted docs present=, surface one line: "this project's docs pile has never been spec-sorted — say 'run spec-sort' to sort it." If the probe was silent, say nothing. A project with nothing to sort never sees the line; a stamped =:LAST_SPEC_SORT:= marker permanently clears it. See the docs-lifecycle rule and the spec in =docs/specs/=.
+ - *Host-identity flag.* If the Phase A host-identity probe printed any match, surface it with the file:line and the fix: "this doc asserts a fixed machine identity — false on every other machine; replace with a runtime derivation (run =uname -n=), per the host-identity rule." The probe flags for judgment, never blocks. Silent when the probe is silent.
- *Language-bundle sync.* If the Phase A step-12 call (=sync-language-bundle.sh=) printed anything, surface it. =fixed= lines are informational — the drift was already repaired (note that =.claude/= is now dirty if the project commits it). A =drift= line on =settings.json= is surface-only and needs the printed =make install-<lang> PROJECT=.= to reconcile; flag it so the user can decide. If the call was silent, say nothing.
- *Newly-installed symlinks.* If the Phase A.0 =make install= step printed any =link= / =relink= / =WARN= line, surface it. A =link= line means a skill, rule, hook, or script added to rulesets is now linked into =~/.claude= for the first time on this machine. For a newly-linked *skill*, check the agent's available-skills list: if the harness already registered it mid-session, note it's available and move on; if it's absent, stop and tell Craig to restart the agent so it loads (whether a mid-session reload works is harness-version-dependent). For a newly-linked *hook*, note that the harness reads hooks at session start — it fires from the next session (or after Craig opens =/hooks= once); its settings.json wiring travels with the tracked file, so the link is usually the only missing piece. A =WARN ... not a symlink= line is a real collision at the target path — surface it; it needs a human. If the step printed only "nothing new to link", say nothing.
- *Template-sync churn (safety net).* Check whether Phase A's rsync left uncommitted churn in the synced =.ai/= paths — accumulated from a prior session that crashed before wrap-up, or freshly added this session when rulesets advanced. Without surfacing, it builds up silently until it blocks Phase A.0's auto-ff (git won't ff a dirty tree). Skip in the rulesets repo itself (there =.ai/= is a committed mirror, kept honest by the pre-commit hook). The check is sequential here, after the rsync has finished — not a Phase A step, to keep that batch race-free.
@@ -197,8 +261,7 @@ This phase touches the user and runs sequentially:
#+end_src
If it reports a count, surface one line: wrap-up's Step 4.0 will commit it as =chore: sync .ai tooling from templates=, or offer to commit it now. If silent, say nothing. This is the crashed-session counterpart to the wrap-up commit step (the primary fix). From the 2026-05-31 jr-estate + work handoffs.
- - *Surface pending cross-agent messages.* If =cross-agent-status= reported any pending messages, list them with their =cross-agent-recv= decision (process / query / reject) per file. For =process= messages in this project's inbox, propose handling now or after the current task. For pending in other projects, mention the count so the user knows to switch projects when ready. If HALT was active, surface that prominently — cross-agent activity is paused until =cross-agent-resume= clears it.
-2. *Process inbox if non-empty.* Mandatory — don't ask, just delegate to [[file:process-inbox.org][process-inbox.org]]. That workflow owns the value gate (advances an existing TODO / improves the project / serves the mission), the per-source rejection flow (Craig / project handoff / script), the priority-scheme check before filing, and the =.eml= extraction path. Single source of truth for the discipline.
+2. *Process inbox if non-empty.* Mandatory — don't ask, just delegate to [[file:inbox.org][inbox.org]] process mode. That mode owns the value gate (advances an existing TODO / improves the project / serves the mission), the per-source rejection flow (Craig / project handoff / script), the priority-scheme check before filing, and the =.eml= extraction path. Single source of truth for the discipline.
3. *Execute project-specific startup extras* (the contents of =.ai/project-workflows/startup-extras.org= read in Phase A). If the file didn't exist, skip.
4. *Ask about priorities.* "What would you like to work on, or is there something urgent you need?"
- If urgent: proceed immediately.
diff --git a/claude-templates/.ai/workflows/status-check.org b/claude-templates/.ai/workflows/status-check.org
index efff16d..4a9972c 100644
--- a/claude-templates/.ai/workflows/status-check.org
+++ b/claude-templates/.ai/workflows/status-check.org
@@ -1,5 +1,5 @@
#+TITLE: Status Check Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-02
* Overview
diff --git a/claude-templates/.ai/workflows/summarize-emails.org b/claude-templates/.ai/workflows/summarize-emails.org
index 6ac5e6f..c9c7001 100644
--- a/claude-templates/.ai/workflows/summarize-emails.org
+++ b/claude-templates/.ai/workflows/summarize-emails.org
@@ -1,5 +1,5 @@
#+TITLE: Summarize Emails Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-14
* Overview
diff --git a/claude-templates/.ai/workflows/suspend.org b/claude-templates/.ai/workflows/suspend.org
new file mode 100644
index 0000000..166f9c9
--- /dev/null
+++ b/claude-templates/.ai/workflows/suspend.org
@@ -0,0 +1,143 @@
+#+TITLE: Session Suspend Workflow
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-06-28
+
+* Overview
+
+This workflow captures the live state of a session when Craig must leave
+abruptly, so a future session resumes with nothing lost. It is the fast,
+capture-only workflow for departure: it writes down where every thread stands,
+notes any uncommitted work, then STOPS — no cleanup, no archive, no teardown.
+
+Triggered by Craig saying "suspend the session," "suspend," "I need to go,"
+"stick a pin in everything," or similar. "I need to go" is broad — if it reads
+as a conversational aside rather than a request to suspend, confirm before
+running.
+
+* Where suspend sits among its neighbors
+
+Three workflows touch the session anchor (=.ai/session-context.org=); keep them
+straight:
+
+- =flush= ([[file:../../flush/SKILL.md]] / =/flush=) — *stay and sharpen.*
+ Refreshes the anchor in place, prompts Craig to type =/clear=, and a hook
+ resumes the *same* logical session in a fresh context. Craig is still here.
+- *suspend* (this workflow) — *leave.* Captures richly into the anchor, leaves
+ the file in place, detaches the tmux client so the session parks in the
+ re-attachable set, and Craig walks away. The next session is a cold startup
+ that detects the present anchor and resumes from it — or Craig re-attaches the
+ still-live session directly.
+- =wrap-it-up= ([[file:wrap-it-up.org][wrap-it-up.org]]) — *end.* Writes the
+ Summary, archives the anchor into =.ai/sessions/=, commits + pushes, and runs
+ the phrase-dependent teardown.
+
+Suspend and flush share one core — capture into the anchor, leave it in place.
+They differ in the exit (leave vs clear-and-continue) and the resume path
+(startup vs the =/clear= hook). Suspend reuses flush's capture discipline (its
+Phase 1 anchor-refresh) rather than restating it, and adds a richer,
+resume-weighted Session Log entry because it's written for a cold resume after a
+gap, not a same-session reset.
+
+* Suspend vs wrap-up — the one structural difference
+
+=wrap-it-up= ARCHIVES =.ai/session-context.org= (renames it into
+=.ai/sessions/=); its absence at the next startup is the signal that the last
+session ended cleanly.
+
+Suspend does the opposite: it LEAVES =.ai/session-context.org= in place. Its
+presence at startup is exactly the signal that the previous session was
+interrupted, so the startup workflow reads it and resumes. Suspend provides only
+the *capture* half — startup's existing interrupted-session path (Phase A checks
+for the anchor, Phase B reads it, Phase C offers to resume) is the *resume* half,
+already built.
+
+So: never archive, never rename the context file in a suspend. Capture into it
+and leave it.
+
+* What gets captured
+
+The point is zero lost information, weighted toward RESUME. Into the
+=* Session Log= of =.ai/session-context.org=, append one dated
+=** YYYY-MM-DD ... — SUSPENDED= entry holding:
+
+1. *Open threads — resume here.* For each active or pending thread: the topic,
+ its status (ACTIVE / PINNED / SET ASIDE / DEFERRED), the immediate next
+ step, and the pointers needed to act on it cold (files + line numbers,
+ commit SHAs, the specific finding or decision). This is the core; spend the
+ most words here. Order newest / most-active first.
+2. *Pending decisions / open questions* awaiting Craig — anything blocked on
+ his input, with enough context that the answer is actionable.
+3. *Shipped this session* — a terse list of what landed, each with its commit
+ SHA, so the resume knows what is already done and need not re-derive it.
+4. *Uncommitted work* — anything modified on disk but not committed, named
+ file by file, so the resume knows what state the tree is in.
+5. *Key findings not yet recorded elsewhere* — anything learned this session
+ that isn't already in a commit, a file, or memory, so it survives.
+6. *Background work* — any running task, agent, or job, and how to check it.
+7. *Resume hint* — the single most likely "start here" next action.
+
+Also update the top of =* Summary= (Active Goal) with a one-line SUSPENDED
+pointer to the entry, so startup reading the top sees the current state even
+when the Summary body is from an earlier thread.
+
+* Steps
+
+1. *Write the SUSPENDED entry* into the Session Log, per "What gets captured"
+ above. Timestamp with =date "+%Y-%m-%d %a @ %H:%M:%S %z"=.
+2. *Update the Active Goal pointer* at the top of =* Summary=.
+3. *Record uncommitted work, don't force-commit it.* A suspend records state, it
+ does not tidy it. Name every uncommitted change in the SUSPENDED entry and
+ leave the tree as it is — on an abrupt departure, a dirty tree (like any
+ crash) is safer than a blind commit of arbitrary mid-work state. (If a
+ project defines a standing always-commit set in its own workflow, commit only
+ that set — but the default shared behavior is to leave the tree alone.)
+4. *Leave =.ai/session-context.org= in place.* Do not archive it.
+5. *Brief handoff* — one or two lines: what was captured, where the resume
+ pointer is, the most-active thread. This is the last thing Craig sees before
+ the view detaches (Step 6), so deliver it complete.
+6. *Detach the tmux client.* As the final action, detach the client viewing the
+ =aiv-<project>= session so it drops out of Craig's active view while staying
+ alive in the background. This is a DETACH, not a teardown: the session and the
+ agent process keep running, nothing is killed, no context is lost.
+
+ #+begin_src bash
+ sess=$(tmux display-message -p '#S' 2>/dev/null)
+ [ -n "$sess" ] && tmux detach-client -s "$sess"
+ #+end_src
+
+ Run it as the very last tool call, after the handoff text has rendered — tmux
+ preserves the pane, so Craig sees the full handoff when he re-attaches. Unlike
+ wrap-up's teardown (which must defer to a =Stop= hook because it kills the
+ session the agent runs in, which would cut off the valediction), detach runs
+ inline: it disconnects the view but leaves the agent's session alive, so
+ nothing is cut off. Degrade gracefully — if not inside tmux (=$TMUX= unset, no
+ session), skip silently and the session simply stays attached.
+
+ Why detach on every suspend: Craig cycles his live agent sessions in Emacs
+ with alt-space, and rotates through everything — including re-attaching
+ detached ai-term sessions — with shift+alt+space. A suspended session left
+ attached clutters the active rotation; detaching parks it in the
+ re-attachable set, which is what makes suspend-and-walk-away work. Re-attach
+ is one keystroke (shift+alt+space) or =tmux attach -t aiv-<project>=.
+
+* What suspend does NOT do
+
+Speed over completeness. A suspend deliberately skips everything wrap-it-up
+does beyond capture:
+
+- No =* Summary= rewrite beyond the one-line Active Goal pointer.
+- No todo.org cleanup / archive-done.
+- No KB / memory promotion sweep.
+- No Linear / board reconciliation.
+- No session-record archive (the file stays live).
+- No teardown. Suspend DETACHES the tmux client (Step 6) but never kills the
+ session: the =aiv-<project>= session and the agent process stay alive in the
+ background, only the view disconnects. It drops no =Stop=-hook teardown
+ sentinel, so the wrap-teardown hook stays dormant. Teardown — killing the
+ session — is wrap-it-up's job, not suspend's; detach is the lighter move that
+ parks a still-live session.
+- No blind commit of working files (step 3).
+- No valediction. A suspend is a pause, not a goodbye.
+
+If Craig later wants the clean end, he runs wrap-it-up, which picks up the
+captured state and finishes the job.
diff --git a/claude-templates/.ai/workflows/sync-email.org b/claude-templates/.ai/workflows/sync-email.org
index 52a7caf..863b400 100644
--- a/claude-templates/.ai/workflows/sync-email.org
+++ b/claude-templates/.ai/workflows/sync-email.org
@@ -1,5 +1,5 @@
#+TITLE: Sync Email Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-02-01
* Overview
diff --git a/claude-templates/.ai/workflows/task-audit.org b/claude-templates/.ai/workflows/task-audit.org
index 67ce496..aa50176 100644
--- a/claude-templates/.ai/workflows/task-audit.org
+++ b/claude-templates/.ai/workflows/task-audit.org
@@ -1,5 +1,5 @@
#+TITLE: Task Audit Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-22
* Overview
@@ -61,6 +61,8 @@ For each open task, read its body and cross-check its claims against the actual
- *Calendar* — did a scheduled event happen; is a SCHEDULED/DEADLINE date now past.
- *Meeting recordings* — if a task hinges on "did this conversation happen / what was said," check the recording queue (e.g. =~/sync/recordings/=) and transcribe via =process-meeting-transcript.org= if the answer lives in an un-transcribed recording. (This is exactly how a "did the interview happen?" task gets resolved instead of guessed.)
+*Spec lifecycle reconcile (docs-lifecycle convention).* If the project has a =docs/specs/=, run the =:SPEC_ID:= query as part of this phase: for each spec whose top-level status heading reads =DOING=, find the =todo.org= task whose =:SPEC_ID:= property matches the spec's =:ID:=. Flag the spec NEEDS-USER when that bound parent is =DONE=/=CANCELLED=, archived, or missing — the build finished (or evaporated) without the =IMPLEMENTED= flip, exactly the drift this check exists to catch. Check the parent's own keyword, not its children (completed children become dated entries and the final flip task is a child, so child-counting misleads).
+
Assign each task a bucket (CURRENT / STALE / NEEDS-USER) and, for STALE, the specific factual update.
*Scale tactic.* For a large open-task set, dispatch read-only investigation sub-agents over batches of tasks (parallel-safe per =subagents.md= — independent read-only domains). Each returns a per-task bucket + suggested update. *Never* let sub-agents write to =todo.org= concurrently — apply all edits serially in the main thread (concurrent writes to one file race and lose work).
@@ -79,11 +81,41 @@ For every STALE task, edit it in the main thread:
- *Ensure priority is set per the project scheme.* The top of the project's =todo.org= should carry the priority legend (=[#A]= through =[#D]=). Every task should carry an explicit priority cookie. If a cookie is missing, or no longer matches the reconciled facts, assign the right level per the legend. If the level is unambiguous from the body, do it autonomously; if it's a judgment call (especially the [#A] / [#B] line for important-but-not-urgent work), flag NEEDS-USER. Also enforce the [#A]-discipline rule from the legend — an [#A] task without a =SCHEDULED:= or =DEADLINE:= line is mis-graded and is either down-graded to [#B] (when reconciled facts say "important but not urgent") or surfaced as NEEDS-USER for the user to date.
- *Ensure a type tag is set.* Every task carries one type tag from the project's tag legend (typically =:feature:= / =:chore:= / =:spec:= / =:bug:=). If missing or wrong, assign or correct it from the body when the type is unambiguous. If two tags fit (a refactor that also fixes a bug; a spec that's also a chore), flag NEEDS-USER rather than picking one silently.
- *Enforce the project's declared tag vocabulary.* If the project's tag legend declares an *exhaustive* set of allowed tags, strip from each task any tag outside that set — the heading and parent section already carry topic/scope context, so ad-hoc tags only fragment the vocabulary and defeat tag-based filtering. Normalize near-duplicate spellings to the canonical tag (a plural to its singular, say). Where the legend does not declare the set closed, leave existing tags alone; this step applies only where the allowed set is exhaustive by design.
-- *Re-assess the =:quick:= and =:solo:= tags* — reconciliation can change a task's effort or autonomy: a resolved dependency may make a stuck task =:solo:=, a scope cut may make it =:quick:=, and new complexity surfaced by the sources can invalidate either. Add or remove the tags per the definitions in the project's tag legend (and [[file:task-review.org][task-review.org]]) when the reconciled facts make the call clear. When they don't — an effort estimate you can't pin down, a =:solo:= gate you can't confirm — it's a NEEDS-USER flag, not a guess.
+- *Re-assess the =:quick:= and =:solo:= tags (mandatory — an audit that skips this is incomplete).* Reconciliation can change a task's effort or autonomy: a resolved dependency may make a stuck task =:solo:=, a scope cut may make it =:quick:=, and new complexity surfaced by the sources can invalidate either. Add or remove the tags per the hard definitions in [[file:../../claude-rules/todo-format.md][todo-format.md]] ("Hard definitions: :solo: and :quick:"; task-review carries the same three-gate walk). Autonomous execution reads =:solo:= as its eligibility gate and trusts the tag, so a stale one is a run-time hazard, not cosmetic drift. When the call isn't clear — an effort estimate you can't pin down, a =:solo:= gate you can't confirm — it's a NEEDS-USER flag, not a guess.
- Bump =:LAST_REVIEWED:= on each edited task.
Follow =todo-format.md= for completion mechanics (depth-based DONE vs dated-rewrite) and the working-files / link-hygiene rules when moving artifacts.
+** Phase C.5 — Consolidate related tasks (interactive)
+
+Phase C's *Consolidate duplicates* bullet folds tasks that track the *same* thing. This step is the broader case: tasks that aren't duplicates but are really *one effort* fragmented across the list. A spread-out effort — several tasks all circling "make the tooling agent-agnostic," say — is harder to see, plan, and finish as a whole than one task, or one parent with the pieces as children.
+
+After the Phase C edits, read the open-task set as a whole and look for *clusters*: tasks that share a goal, a subsystem, or an obvious sequence. Use judgment over the task bodies, not a keyword heuristic — adjacency is a semantic call, and a brittle title-match both misses real clusters and invents false ones.
+
+For each cluster, surface it to Craig (inline numbered options per =interaction.md=, no popup) with a recommendation, offering the two shapes:
+
+- *Merge* — fold the cluster into one task when the members are genuinely the same work split up (near-duplicates, or steps with no independent value). The merged task keeps the strongest priority, unions the type tags, and absorbs each member's body as a dated note or a short list; the absorbed tasks close per =todo-format.md= (a =**= task → =CANCELLED= + =CLOSED:= with a one-line "merged into <task>", or deletion if it carried nothing unique).
+- *Parent with children* — when the members are related but distinct (each ships independently or has its own value), promote a parent task and re-home the members beneath it as sub-tasks, so the list shows the effort as a unit without losing the individual pieces.
+
+Never merge or re-parent autonomously — which tasks belong together, and whether they're one-work or related-distinct, is a judgment only Craig ratifies. Propose, don't apply, until he picks. A cluster he declines stays as separate tasks; don't re-surface it every audit (note the decline in the session log).
+
+When no clear cluster exists, say so in one line and move on — most audits won't find one, and forcing a merge fragments worse than it consolidates.
+
+** Phase C.6 — Retire completed parents and promote stragglers (interactive)
+
+Phase C.5 consolidates related *open* tasks. This step retires parent tasks whose work is *finished*, so completed containers don't linger in Open Work as scaffolding.
+
+Run =todo-cleanup.el --convert-subtasks= first (it's part of the =clean-todo= / wrap-up cleanup, and =open-tasks.org= runs it too) so every completed sub-task is a dated event-log entry rather than a lingering =DONE= keyword. The closure logic below reads "open child" as a child heading still carrying a task keyword (=TODO=/=DOING=/=WAITING=/=VERIFY=/=NEXT=/=PROJECT=/=STALLED=/=DELEGATED=); a dated entry is correctly not open.
+
+Two shapes, both proposed to Craig (inline numbered options per =interaction.md=, no popup) before applying:
+
+- *Zero open children → close the parent.* A parent whose child *tasks* are all resolved (now dated) and that carries no open child task is finished: close it per =todo-format.md= (=**= parent → =DONE=/=CANCELLED= + =CLOSED:=), and it moves to Resolved on the next =--archive-done=. If the work resurfaces later, a fresh task is created then; a completed container shouldn't sit open as a placeholder.
+- *One or two open children → promote, then close.* When a parent has only one or two open children, pull them out and rewrite them as standalone =**= level-2 tasks — give each a priority per the project scheme, and make the heading stand alone without the parent's context — then close the now-childless parent and let it move to Resolved. The former children become first-class Open Work tasks; the retired parent stops being scaffolding for one or two stragglers.
+
+*The leaf-with-notes carve-out (important).* "Zero open children" is not the same as "done." A =**= leaf task whose only descendants are dated *notes* — a captured "Ideas", "Goals", or "Current State" entry, not a real completed sub-task — is unstarted work with a note attached, not a finished container. Do not close it. Tell the two apart by intent: a container reads as a grouping (a =PROJECT= keyword, an explicit "parent grouping ..." line, or several dated entries that were genuinely separate sub-tasks that shipped); a leaf-with-notes is a single feature/bug task whose title names unstarted work and whose lone dated child is a design note. When the call is ambiguous, flag it NEEDS-USER rather than closing.
+
+Never close or promote autonomously past the ambiguous line — surface the candidates with a recommendation and let Craig ratify, the same interactive stance as Phase C.5. Clear container completions (a =PROJECT= whose every child is dated) can be proposed as a batch; leaf-with-notes ambiguities are flagged individually. Verify open-vs-done counts against the actual headings (a real scan of the subtree), not a fragile regex that a shell's =\b= support can silently break — a miscount here closes live work.
+
** Phase D — Flag the judgment calls (interactive)
Present the NEEDS-USER bucket as a short, scannable list — one line per task, naming the decision or the fact required. Adjudicate with the user one item at a time (inline numbered options per =interaction.md=, no popup). Apply the user's calls as they come (which may itself produce more autonomous updates, or new tasks).
@@ -132,3 +164,7 @@ Two Phase C behaviors added, both surfaced by an Emacs-config =todo.org= audit:
- *Tag-vocabulary enforcement.* That project declares a closed tag set (=bug=, =feature=, =refactor=, =test=, =quick=, =solo=); the audit had to strip ~44 ad-hoc tags that had accumulated across the file. The prior workflow only checked that a type tag was *present* — it had no concept of an exhaustive allowed set. The new bullet enforces a declared closed vocabulary and leaves open-vocabulary projects untouched.
- *Code-complete-but-unverified closing.* Many tasks had shipped (tests green, live in the daemon) but stayed open awaiting a manual or visual verification, so they accumulated as half-open. Leaving them open is noise; auto-closing them would violate "never claim a fix verified before the user confirms." The fix routes the pending human check into the project's =Manual testing and validation= parent (dedup-checked) per =verification.md='s manual-verification hand-off, then closes the implementation task. The work is done and the check is tracked; a failed check promotes to a bug.
+
+** 2026-07-01 — Retire completed parents (Phase C.6)
+
+Added Phase C.6: retire a parent task once its child *tasks* are all done. Zero open children → close the parent; one or two open children → promote them to standalone level-2 tasks, then close. Surfaced by an Emacs-config =todo.org= audit where several PROJECT containers had all children complete. Depends on =todo-cleanup.el --convert-subtasks= running first so completed sub-tasks are dated (not lingering =DONE= keywords) and the open-child count is accurate. Carries a leaf-with-notes carve-out: a =**= leaf task whose only descendant is a dated design note ("Ideas"/"Goals") is unstarted work, not a finished container, and must not be closed — the ambiguous case is flagged NEEDS-USER. The step also warns against counting open-vs-done with a fragile regex (a =\b= that a given shell/awk silently drops miscounts and closes live work).
diff --git a/claude-templates/.ai/workflows/task-review.org b/claude-templates/.ai/workflows/task-review.org
index 69e172d..7ea2e8e 100644
--- a/claude-templates/.ai/workflows/task-review.org
+++ b/claude-templates/.ai/workflows/task-review.org
@@ -1,5 +1,5 @@
#+TITLE: Task Review Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-20
* Overview
@@ -57,7 +57,9 @@ Keep is the common case — most tasks are still right and just need re-stamping
*** Tagging =:quick:= — small tasks
-While reviewing each task, estimate its effort. If you judge it *30 minutes or less* and it doesn't already carry =:quick:=, add the tag to the heading line. If the heading and body don't tell you how long it'll take, *ask Craig* — don't guess. A wrong =:quick:= is worse than none: the tag exists so Craig can grab a genuinely small task in a spare moment, and a mislabeled one wastes that moment.
+The =:quick:= and =:solo:= assessments (this section and the next) are *mandatory* for every reviewed task except a Kill — a review that skips them is incomplete. The hard definitions live in [[file:../../claude-rules/todo-format.md][todo-format.md]] ("Hard definitions: :solo: and :quick:"); autonomous execution (work-the-backlog / the no-approvals speedrun) reads =:solo:= as its eligibility gate and trusts the author's tag, so the run-time gate is only as trustworthy as this pass.
+
+While reviewing each task, estimate its effort. If you judge it *30 minutes or less* and it doesn't already carry =:quick:=, add the tag to the heading line. If the heading and body don't tell you how long it'll take, *ask Craig* — don't guess. A wrong =:quick:= is worse than none: the tag exists so Craig can grab a genuinely small task in a spare moment, and a mislabeled one wastes that moment. =:quick:= is an effort hint only, never an eligibility gate — size does not decide what runs autonomously.
This is orthogonal to the action chosen — a task can be kept (or re-graded, or marked DOING) *and* tagged =:quick:= in the same pass. Skip the assessment on a Kill, since it's leaving the pool. Tags go on the heading line per [[file:../../claude-rules/todo-format.md][todo-format.md]], sharing one =:tag1:tag2:= cluster.
@@ -67,7 +69,7 @@ While reviewing each task, judge whether Claude could build *and* verify it with
1. *Buildable* — Claude has the capability and access to do the work.
2. *Verifiable by Claude* — an objective or local check exists that Claude can run itself. Craig's routine spot-checking does not count against this, and neither does handing off a residual human-in-the-loop confirmation as a structured manual-testing reminder (the =verification.md= "Handing Off Manual Verification" pattern). The disqualifier is having no verification path of Claude's own at all — when the success criterion is only judgeable by Craig's eyes or subjective taste.
-3. *No upfront decision* — no design or preference call Craig must make before Claude can begin.
+3. *No deliberation* — no open design question and no "weigh these approaches" with real tradeoffs. At most one or two *quick, upfront-answerable* factual decisions are allowed — the speedrun preset batches those into its pre-flight Q&A, so they don't break the hands-off run. A genuine design or preference call disqualifies.
If any gate is shaky, leave the tag off. Like =:quick:=, a wrong =:solo:= is worse than none — it tells Craig he can hand the task off and walk away, so a mislabeled one wastes that trust. When the heading and body don't make all three gates clear, ask Craig instead of guessing.
@@ -90,12 +92,14 @@ Set =:LAST_REVIEWED:= to today's date (from above) in the task's =:PROPERTIES:=
Body...
#+end_example
-The exact date string matters: =task-review-staleness.sh= and the wrap-up health check both parse =:LAST_REVIEWED: YYYY-MM-DD=.
+Format: =:LAST_REVIEWED:= takes a bare ISO date (=2026-05-20=) or an org-native inactive timestamp (=[2026-05-20 Tue]=, matching the =CREATED:=/=CLOSED:= cookies beside it); =task-review-staleness.sh= and the wrap-up health check normalize both to the date. A value that is neither is a data error — the staleness script warns loudly to stderr (naming the file, line, and value) and leaves the task out of the stale count rather than silently reporting a freshly-reviewed task as never-reviewed. Stamp a clean date and the warning never fires.
*** Killing a task
Follow the completion rules in [[file:../../claude-rules/todo-format.md][todo-format.md]]. A killed top-level =**= task stays task-shaped: change the keyword to =CANCELLED=, add a =CLOSED: [YYYY-MM-DD Day]= line under the heading (generate with =date "+%Y-%m-%d %a"=), and leave the priority and tags intact. It's then a candidate for =--archive-done= at the next cleanup. Don't stamp =:LAST_REVIEWED:= on a kill — it's leaving the review pool anyway.
+A killed *sub-task* (=***= or deeper, under a parent task) instead becomes a dated event-log entry per the depth rule — but you don't have to hand-format it here. =todo-cleanup.el --convert-subtasks= (run in the =clean-todo= and wrap-up cleanup passes) rewrites any level-3+ DONE/CANCELLED/FAILED heading into its dated form mechanically from the =CLOSED= cookie, so a keyword-plus-=CLOSED= close at depth gets normalized on the next cleanup rather than lingering. =lint-org.el= flags any that slip through (checker =subtask-done-not-dated=).
+
* Phase D: Close out
When the batch is done (or Craig calls it early):
diff --git a/claude-templates/.ai/workflows/triage-intake.cmail.org b/claude-templates/.ai/workflows/triage-intake.cmail.org
index d818c72..8d8abfb 100644
--- a/claude-templates/.ai/workflows/triage-intake.cmail.org
+++ b/claude-templates/.ai/workflows/triage-intake.cmail.org
@@ -1,5 +1,5 @@
#+TITLE: Triage Intake — cmail (Proton) Source
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-26
# Source plugin for the triage-intake engine. See triage-intake.org for the
diff --git a/claude-templates/.ai/workflows/triage-intake.github-prs.org b/claude-templates/.ai/workflows/triage-intake.github-prs.org
index c1bc796..644421c 100644
--- a/claude-templates/.ai/workflows/triage-intake.github-prs.org
+++ b/claude-templates/.ai/workflows/triage-intake.github-prs.org
@@ -1,5 +1,5 @@
#+TITLE: Triage Intake — Personal GitHub PRs Source
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-26
# Source plugin for the triage-intake engine. See triage-intake.org for the
diff --git a/claude-templates/.ai/workflows/triage-intake.org b/claude-templates/.ai/workflows/triage-intake.org
index b257f2d..55cc939 100644
--- a/claude-templates/.ai/workflows/triage-intake.org
+++ b/claude-templates/.ai/workflows/triage-intake.org
@@ -1,5 +1,5 @@
#+TITLE: Triage Intake Workflow (Engine)
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-01
* Summary
@@ -11,12 +11,14 @@ Think of it as the ER intake queue: every new message, invite, and PR notificati
*This file is the engine.* It carries no sources of its own. Every source it scans comes from a *source plugin* — a =triage-intake.<source>.org= file the engine loads at Phase 0. The engine is source-agnostic and project-agnostic; the project- and account-specific knowledge lives entirely in the plugins. To add a source, drop a plugin file. To change one, edit its plugin. Never wire a source into this file.
+*Which sources a project pulls is a per-project choice.* A *project-specific* plugin (=.ai/project-workflows/triage-intake.*.org=, never synced) is active by presence — dropping it is the declaration. A *general* plugin (=.ai/workflows/triage-intake.*.org=, template-synced into every project — personal Gmail, cmail, calendar, Telegram, GitHub PRs) is active only when the project names its basename in a =:TRIAGE_SOURCES:= line in =.ai/notes.org= Workflow State (space-separated basenames, e.g. =:TRIAGE_SOURCES: personal-gmail cmail=). A project that declares nothing and owns no project plugin pulls nothing. This is the Phase 0 activation gate — presence is capability, the declaration is activation (see =docs/specs/2026-07-20-triage-source-activation-spec.org=).
+
Distinct from =daily-prep.org=:
- *daily-prep* — heavier, once daily, builds the day's plan + standup brief + meeting prep + time blocks.
- *triage-intake* — fast, repeatable, just answers "what's new since last check?"
-Quick contract — what it does: fans out across source plugins, classifies every item into Action / FYI / Noise-keep / Noise-trash, synthesizes one deduped summary, writes each Action item to =todo.org= as a =:quick:reactive:= task, and executes star/mark-read/trash on confirmation.
+Quick contract — what it does: fans out across source plugins, classifies every item into Action / FYI / Noise-keep / Noise-trash, surfaces one three-section digest (==TASKS== / ==FYI== / ==MISC==) with a two-option close offer, then *closes by default*: files each TASKS item to =todo.org= as a =:quick:reactive:= task, runs the star/mark-read/trash hygiene on every scanned account, advances the sentinel, and tears down anything it started. The close runs unless Craig explicitly holds it.
** When to Use This Workflow
@@ -37,6 +39,8 @@ Typical timing:
Do *not* use when running daily-prep — daily-prep already does this as Phase 3.
+Also runs unattended as sentry's triage pass (=sentry.org=, pass 3): sentry invokes this engine under its no-approvals contract, where destructive actions (deleting, archiving, sending) queue for the morning-approval review instead of firing. The trigger phrases above are unchanged — a manual "triage intake" always routes here directly.
+
* Execution
@@ -56,16 +60,18 @@ ls .ai/workflows/triage-intake.*.org .ai/project-workflows/triage-intake.*.org 2
The glob exclude is automatic: =triage-intake.*.org= matches the plugins but not this engine file (=triage-intake.org= has no second dot-segment), so the engine never loads itself.
After globbing, for each plugin file:
-1. Read it.
-2. Evaluate its =ENABLED= precondition. If false, *announce the skip with its reason* ("skipping linear — mcp__linear not present") and move on.
-3. The surviving set is the source list for Phases A-D.
+1. *Activation gate.* A *general* plugin (from =.ai/workflows/=, template-synced into every project) is active only if its basename appears in the project's =:TRIAGE_SOURCES:= declaration (=.ai/notes.org= Workflow State — a space-separated list of source basenames). If it isn't declared, it is *inactive*: announce it ("inactive: personal-gmail — not in :TRIAGE_SOURCES:") and skip it. A *project-specific* plugin (from =.ai/project-workflows/=, never synced) is always active — dropping it there is itself the per-project declaration. This is what stops the synced general plugins from self-activating in projects that aren't triage targets: presence is capability, the declaration is activation (see =docs/specs/2026-07-20-triage-source-activation-spec.org=). An absent or empty =:TRIAGE_SOURCES:= means no general sources are active; a project with no declaration and no project plugin has no active sources, so triage no-ops there.
+2. Read it.
+3. Evaluate its =ENABLED= precondition. If false, *announce the skip with its reason* ("skipping linear — mcp__linear not present") and move on.
+4. The surviving set — active and enabled — is the source list for Phases A-D.
-*Announce the loaded set before scanning* so the omission can't hide:
+*Announce the loaded set before scanning* so the omission can't hide — inactive (undeclared) plugins are named too, so a general plugin left out of =:TRIAGE_SOURCES:= is a visible choice, not a silent drop:
#+begin_example
-Loaded 5 source plugins:
- general: personal-gmail, personal-calendar, cmail, github-prs
+Loaded 2 source plugins (:TRIAGE_SOURCES: personal-gmail cmail):
+ general: personal-gmail, cmail
project: deepsat-gmail
+ inactive (undeclared): personal-calendar, github-prs, telegram
skipped: linear (mcp__linear not present)
#+end_example
@@ -91,23 +97,27 @@ Every item lands in one bucket. Plugins refine these with source-specific bias a
Per-source bias (a work email account leans keep for audit value; a personal account leans trash on high noise volume) lives in each plugin's =Classify= section. Read it from there; don't re-derive it here.
-*** Phase C: Synthesize a single summary
+*** Phase C: Synthesize — the three-section digest
-One markdown summary surfaced inline to Craig. Order:
+One digest surfaced inline to Craig — notable items only, written as prose bullets a reader can absorb without knowing the source taxonomy. It is *not* a per-source roll-call; the plugins' =Render= shapes feed the classification, they are no longer displayed as blocks. (Format ratified by Craig 2026-07-18 after a sweep where the follow-up "summarize the notable items" digest was the report he actually wanted first.)
-0. *Scan failures — first, loud, always.* Any loaded source whose scan failed, hung, was killed, or was skipped for an operational reason renders at the very top of the summary, before Top signals:
+Order:
+
+0. *Scan failures — first, loud, always.* Any loaded source whose scan failed, hung, was killed, or was skipped for an operational reason renders at the very top of the summary, before everything else:
#+begin_example
⚠ SCAN FAILED: <source> — <reason, one line> — <what's now unknown>
#+end_example
- A failed scan is never folded into "quiet." Quiet means the scan ran and found nothing; a failure means the sweep is blind on that channel, and the reader must know which. The same applies to a precondition skip the user hasn't standing-approved (e.g. a messaging client that needs a temporary server spin-up): run the lifecycle or report the failure — don't silently narrow the sweep.
+ A failed scan is never folded into "quiet." Quiet means the scan ran and found nothing; a failure means the sweep is blind on that channel, and the reader must know which. The same applies to a precondition skip the user hasn't standing-approved (e.g. a messaging client that needs a temporary server spin-up): run the lifecycle or report the failure — don't silently narrow the sweep. Backlog banners (a plugin's pre-anchor-residue probe firing) render here too — loud, above the sections.
+
+1. *==TASKS==* — every work item that needs Craig or his sign-off. Sorted in two groups: *solo-executable first* (see the solo test in Phase D), then the rest; within each group, priority order (blocking someone / deadline inside 48h first). Each item is one short prose bullet naming who, what, and why-now, with the source link or locator.
+2. *==FYI==* — substantive work context worth seeing, no action owed. Priority order.
+3. *==MISC==* — everything outside the project: personal mail, family, household, personal calendar. Priority order. An outside-project item that needs Craig still lands here (with its action-ness stated inline), not in TASKS — TASKS is the work queue. MISC items are *not* filed into this project's =todo.org=; they surface in the digest and are gone after the close unless Craig reroutes them to their owning project (the "and reroute" modifier, below).
-1. *Top signals to act on* — bullet list of 3-7 items, ordered by urgency, *Action only*. Each bullet links to the source (permalink, thread URL, PR number).
-2. *Per-source breakdown* — one short section per loaded source *that has changes*, in =ORDER=, using that plugin's =Render= shape: Action items detailed, FYI items as a short list, Noise as a tally only ("Noise: 12 trash candidates, 4 keep, 0 starred").
-3. *Suggested actions* — explicit list of state changes Craig could take this run (trash these N messages, mark-read these M, star this Action item, respond to this invite, merge PRs #X and #Y, etc.). This line stays whenever there are queued actions, regardless of how quiet the sweep was.
+A section with nothing in it is omitted. After the sections, the offer (see Phase D). The sweep's final line is always the run timestamp (=date "+%A %Y-%m-%d %H:%M %Z"=).
-*Deltas only.* The summary reports what *changed* since the anchor: a new invite, a new/moved/cancelled calendar event, a new message needing attention. A source with no changes gets no block — no "Calendar — quiet", no "PRs — nothing new" roll-call. A sweep where nothing changed anywhere renders as a single line:
+*Deltas only.* The digest reports what *changed* since the anchor: a new invite, a new/moved/cancelled calendar event, a new message needing attention. A source with no changes contributes nothing — no "Calendar — quiet", no "PRs — nothing new" roll-call. A sweep where nothing changed anywhere renders as a single line plus the timestamp:
#+begin_example
17:39 sweep: no changes
@@ -117,11 +127,40 @@ One markdown summary surfaced inline to Craig. Order:
Scan failures are the standing exception: a failed or skipped scan always renders loudly per point 0 above and is never folded into the no-change line — "no changes" is a claim about channels the sweep could actually see.
-Format target: scannable in 30 seconds, full read in 2 minutes. Don't pad.
+Format target: scannable in 30 seconds, full read in 2 minutes. Don't pad. The old long-form report (anchor line, per-source breakdown, itemized suggested-actions list) is available *on request* — it is no longer the default surface.
+
+**** The offer — exactly this, right after the sections
+
+#+begin_example
+1. Close the triage — file todo.org tasks for the TASKS items, run the standard close
+2. Close the triage and execute the solo TASKS now, after filing the rest
+Append "and reroute" to either option to send the outside-project items to their owning projects.
+Or tell me what you'd like handled differently.
+#+end_example
+
+Option 2 renders *only when solo-executable TASKS exist*. The reroute line renders only when MISC is non-empty. No other options, no itemized action menu — the close (Phase D) owns the routine hygiene.
+
+*The reroute modifier.* "1 and reroute" / "2 and reroute" (or a bare "reroute" right after a close) means: in addition to the close, deliver every outside-project item the sweep surfaced to the project that owns it. Routing goes through =inbox-send= per the cross-project rule — a handoff into the owner's =inbox/=, never a direct write to a foreign =todo.org= — and the owner's own inbox processing files it by its conventions. This is the persistence path for MISC: without a reroute, MISC items are surfaced-only.
+
+*** Phase D: Close — the default, not an option
+
+*Closing the triage is the next action after the digest, no exceptions* — unless Craig explicitly says not to ("hold the triage", "don't close yet"). If his reply picks option 1 or 2, close per the option. If his reply is anything else — a question, a redirect, a new task — *close the triage first* (as option 1), then handle what he asked. An unclosed triage strands the noise unprocessed and the sentinel stale; Craig ruled 2026-07-18 that the close, including the mail hygiene on every scanned account, is unconditional.
+
+The close, in order:
+
+1. *File every TASKS item into =todo.org=* as its own =:quick:reactive:= task (format below). Dedupe against existing tasks first — fold into an existing task's body when one already covers the topic.
+2. *Mail and message hygiene on every scanned account*: trash the Noise-trash set, mark-read the Noise-keep set, star what was flagged for keeping. This runs *without itemized confirmation* — the digest's tallies are the notice, and the actions dispatch through each plugin's =Actions= verbs. Trash is recoverable (Gmail 30-day trash; cmail =\Deleted= flag), which is what makes the no-confirmation batch safe.
+3. *Clear unacked items* that the sweep found resolved.
+4. *If option 2: execute the solo TASKS.* The solo test — mechanical, standing-approved, and producing *no prose under Craig's name*: calendar RSVPs, ticket-state moves the publishing overlay already authorizes, mark-read/star/trash. Anything that sends words as Craig (a Slack reply, an email, a PR comment, a Linear comment) is *never* solo — it stays a filed task and goes through the normal draft gate when worked. Destructive or hard-to-reverse actions beyond mail hygiene (branch deletes, PR merges) also stay confirm-gated per their plugins.
+5. *If "and reroute": route the outside-project items.* For each MISC item (and any surfaced item this project doesn't own), send a handoff to the owning project's inbox: =inbox-send <project> --text "..."= carrying what it is, the source locator (message id, thread, event id), and why it routed. Resolve the owner against =inbox-send --list=; when ownership is ambiguous, ask before sending — a wrong-project handoff costs more than one question. Never write another project's =todo.org= directly (cross-project rule). Note in the close's status line which items went where.
+6. *Advance the sentinel* — write the held Phase A capture into the sentinel's *content* (see "Capture the Phase A timestamp"): =echo "$PHASE_A_TS $(date -d "@$PHASE_A_TS" '+%Y-%m-%d %H:%M:%S %z')" > .ai/last-triage-intake=. Do not use plain =touch= (writes mtime to /now/ and strands items posted between Phase A and end of run) and do not use =touch -d "@$PHASE_A_TS"= (correct timestamp but mtime is per-machine — won't survive a fresh clone or cross-machine sync).
+7. *Tear down anything the sweep started* (e.g. the telegram lifecycle's leave-no-trace shutdown).
+
+After the close, report one status line — what shipped, the sentinel time — and stop. The close *is* the exit; there is no separate confirmation loop.
-**** Sub-step: write each Action item into =todo.org= as its own =:quick:= task
+**** Task-filing format (=todo.org=)
-After surfacing the summary inline, append every Action item — regardless of source — to =todo.org= as its own top-level =** TODO= heading carrying the =:quick:= tag plus =:reactive:= and any relevant person/entity tag.
+Append every TASKS item — regardless of source — as its own top-level =** TODO= heading carrying the =:quick:= tag plus =:reactive:= and any relevant person/entity tag.
Each Action item is one task. Don't group items by source under =** Email Response=, =** PR Review=, etc. sub-headings. Each response is its own filterable task so Craig can re-prioritize, =SCHEDULE:= / =DEADLINE:=, or tag individually.
@@ -142,30 +181,113 @@ Rules:
- *Record the source locator in the task body* so a reply can be routed back to where the request came from — the channel + thread id for chat, the repo + PR number, the message id for mail. The general rule: a reply goes back to the *origin* of the request, not a fixed notification channel. (Project plugins may add stricter routing rules in their own files.)
- Placement: append at end of =* Work Open Work= (just before =* Work Incubate=) unless the project's =todo.org= has a designated triage section near the top (=* Triage= or =* Inbox=).
-This sub-step makes triage-intake's findings *persist* in =todo.org= instead of evaporating after the inline summary.
+The filing makes triage-intake's findings *persist* in =todo.org= instead of evaporating after the inline digest. Every close action dispatches to the owning source plugin's =Actions= verb (trash, mark-read, star, respond, merge, comment, attachment-fetch) — the engine doesn't hardcode action commands; it reads them from the loaded plugins.
-*** Phase D: Execute actions on confirmation
+*** Exit Criteria
-Wait for Craig's go-ahead before running any state changes. Default to single-confirmation for the whole batch ("yes" → run everything proposed). Craig may also pick a subset ("trash personal but hold the work account") or hand back a different plan ("trash all but star the expense thread and queue PR merges for after lunch").
+The close is the exit. Once the close completes (tasks filed, hygiene run, sentinel advanced, teardown done), report one status line — what shipped, the sentinel time — and stop. The old stay-open-until-confirmed loop is retired (Craig, 2026-07-18): the digest plus the two-option offer is the whole interaction, and the close runs by default. The only way the workflow stays open is Craig explicitly saying not to close.
-Each action dispatches to the owning source plugin's =Actions= verb (trash, mark-read, star, respond, merge, comment, attachment-fetch). The engine doesn't hardcode action commands — it reads them from the loaded plugins. Read each plugin's =Actions= section for the exact command.
+*** KB capture (only if the sweep surfaced something durable)
-After actions complete, write the Phase A capture into the sentinel's *content* (see "Capture the Phase A timestamp"): =echo "$PHASE_A_TS $(date -d "@$PHASE_A_TS" '+%Y-%m-%d %H:%M:%S %z')" > .ai/last-triage-intake=. Do not use plain =touch= (writes mtime to /now/ and strands items posted between Phase A and end of run) and do not use =touch -d "@$PHASE_A_TS"= (correct timestamp but mtime is per-machine — won't survive a fresh clone or cross-machine sync).
+If this sweep surfaced a durable, cross-project fact — a recurring pattern across sources, a reference pointer worth keeping, an environment gotcha — consider writing it to the agent KB as one =:agent:= node (see the best-practices node and =knowledge-base.md=; personal projects only, work never writes). One line of judgment, not a step: an all-quiet sweep surfaces nothing and writes nothing. Never blocking, never padded onto a no-signal run.
-*Do not close the workflow yet.* See Exit Criteria below.
-*** Exit Criteria
+* Auto mode (unattended monitoring)
+
+Auto mode is a self-running variant of the engine for when Craig is away from the desk but wants tight awareness — a loop that runs the standard sweep on a short interval, *accumulates* findings rather than mutating state, and hands Craig a gated checkpoint to commit the batch. It composes two things: the *delivery* (a =/loop= in the live session) and the *behavior* (accumulate-don't-mutate sweeps with a checkpoint). The one-shot run above is unchanged; auto mode is an additional way to run the same Phase 0 / A-D engine.
+
+** Trigger and delivery
+
+- "auto triage" / "auto triage-intake" / "watch the desk" / "monitor the triage" — start auto mode.
+- Default interval *20 minutes*; Craig sets it.
+
+Auto mode runs as a =/loop= in the *live session*, not a detached cron job:
+
+#+begin_src
+/loop 20m run an auto-mode triage-intake sweep per triage-intake.org
+#+end_src
+
+Running in the live session means MCP auth (Slack, Gmail, Linear) is inherited from the session — the headless-auth wall that blocks a detached cron run does not apply. A durable cross-session schedule is out of scope here; that belongs to the morning-ops orchestrator, which can later invoke auto mode's accumulate behavior as its triage limb. The close/stop commands below require a live session by design.
+
+*** Phone delivery — push each signal sweep via =agent-text=
+
+Auto mode exists for when Craig is away from the desk, so a sweep that surfaces something worth seeing is delivered to his phone, not just printed into a session he isn't watching. After a sweep that renders the full three sections — one with real deltas or an unacked-list change (see "End-of-sweep output" below) — send that same output to his phone over Signal with =agent-text=:
+
+#+begin_src bash
+agent-text "$SWEEP_SUMMARY"
+#+end_src
-The workflow stays open until Craig has *explicitly* either:
+The pushed text is the *fuller* three-section shape, not a terse one-liner: the per-source deltas, the responses-awaiting-acknowledgment list, and the timestamp, led by a ⚠ SCAN FAILED banner if any source failed.
-1. *Confirmed* that the executed actions are sufficient and nothing more is needed this round, or
-2. *Handed back a different plan* (e.g., "actually hold the PR merges, address #131 first").
+*Signal-only — never on a quiet sweep.* An empty sweep (the =triage intake at HH:MM: nothing= heartbeat) does *not* push to the phone. Silent-until-signal (see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=) governs the phone channel too, so the phone stays silent until a sweep has real signal. The in-session heartbeat still prints as proof the loop ran; the phone is reserved for something that actually needs Craig. (Craig's ruling, 2026-07-20: the higher-cost channel doesn't buzz with "nothing.")
-A successful Phase D run is *not* an exit signal. After the action batch returns, surface what shipped and wait. Don't volunteer "done" or "all set" — those are exit-claim phrases that pre-empt Craig's call. Use a status report ("17 actions succeeded, sentinel written at 12:19") and stop.
+If =agent-text= isn't on =PATH=, fall back to inline delivery and say so once.
-If Craig has been silent for a while after Phase D and the surface looks closed-out, *ask*: "Anything else on this triage, or are we good to close out?" Don't auto-terminate.
+*Reply polling is deferred.* The send half ships here; polling the phone for Craig's replies (the =phone-recv= half of the retired ntfy design) waits on the reply-correlation follow-up. With the Signal account linked on more than one device, a reply fans out to every device and neither knows which page it answers — that has to be resolved before auto mode reads replies back. Until then auto mode pushes but does not poll, and Craig acts on a pushed summary from wherever he picks it up.
-This rule prevents the failure mode where the workflow self-declares done and the next exchange has to relitigate what state things are in.
+** Preconditions and Close-out
+
+Auto mode borrows the inbox monitor-mode gates (=inbox.org= monitor mode): do not start on a dirty worktree or a red test suite — a close's batch commit would otherwise sweep up unrelated changes — and leave the tree clean and green when the loop stops. Surface a blocker with inline numbered options per =interaction.md= and wait.
+
+** A sweep: accumulate, don't mutate
+
+Each sweep runs Phase 0 (load *both* plugin dirs — the loud requirement still holds) and Phases A-D's scan / classify / synthesize, but performs *none* of the normal run's mutations:
+
+- Does NOT advance the sentinel. The scan window grows from the last *close* until the next close: every sweep scans from the existing sentinel up to now, so nothing between sweeps is dropped.
+- Does NOT write =todo.org= Action tasks — accumulates them for the close.
+- Does NOT take mail actions (trash / mark-read / star).
+- Does NOT commit.
+- DOES update an active daily-prep in Update mode and re-open it on change (per =daily-prep.org=).
+- DOES report, deltas-only, with loud scan-failure banners (Phase C rules unchanged).
+
+** End-of-sweep output — three sections, or one heartbeat
+
+*Silent-until-signal (see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=).* An *empty sweep* — no deltas since the previous sweep and no change to the awaiting-acknowledgment list — collapses to a single heartbeat line and nothing else: =triage intake at HH:MM: nothing= (HH:MM local, from =date=). Detection still runs in full (Phase 0 plus the A-D scan, against the session's inherited MCP auth); only the output collapses, so a long unattended run stops filling the session with identical "no changes" blocks. A sweep with real deltas or an unacked-list change prints the full three sections below, and — when away — pushes them to Craig's phone via =agent-text= (see "Phone delivery" above). The empty-sweep heartbeat is never pushed.
+
+1. *Deltas* — what changed since the *previous sweep* (the standard Phase C summary scoped to the inter-sweep delta).
+2. *Responses awaiting your acknowledgment* — every Slack reply, email, or message directed at Craig that he hasn't acknowledged or had the agent answer. A *running list carried forward across sweeps* until Craig acks each item or closes the triage. An away user's first need is "who's waiting to hear back from me," which a delta-only sweep loses the moment it scrolls past.
+3. *Timestamp* — the current date, time, and timezone on the sweep's own final line, so an away reader sees how fresh the summary is without computing it. Print it on every sweep that prints these three sections. On an *empty* sweep there is no separate timestamp line — the heartbeat (=triage intake at HH:MM: nothing=) is itself the freshness stamp and the proof the loop ran. Generate it with:
+
+ #+begin_src bash
+ date "+%A %Y-%m-%d %H:%M:%S %Z (%z)"
+ #+end_src
+
+** The unacked list — durable state
+
+The awaiting-acknowledgment list lives in =.ai/triage-intake-unacked.org=, so it survives a session crash, a =/clear=, or a restart — the away-from-desk case auto mode exists for. It's project-local state, tracked the same way as the sentinel (=.ai/last-triage-intake=), created on first need.
+
+Shape — one =** = heading per awaiting item:
+
+#+begin_example
+#+TITLE: Triage Intake — Responses Awaiting Acknowledgment
+# Maintained by triage-intake auto mode. One heading per item; acked items are removed.
+
+** Dana — 2pm reschedule invite
+:PROPERTIES:
+:SOURCE: personal-calendar
+:LOCATOR: <event id or thread url — the dedupe key>
+:SINCE: 2026-06-15 10:42
+:END:
+She's waiting on a yes/no to the move.
+#+end_example
+
+- *Add* — a sweep appends any new directed-at-Craig response not already listed, deduped on =LOCATOR=.
+- *Carry forward* — every sweep re-renders the full list in its second section, whether or not it changed this sweep.
+- *Ack* — "ack <item>" (e.g. "ack the Dana thread") removes that heading; "ack all" clears the list.
+- *Close* — a close empties the list as part of processing (each item is either actioned or filed).
+
+** Close and stop — the checkpoint
+
+The mutations are gated behind two commands:
+
+- *"close the triage"* — run the full close per Phase D: take the accumulated mail hygiene, file the accumulated TASKS items to =todo.org=, reroute if asked ("close the triage and reroute"), empty the unacked list, then *advance the sentinel* — capture the close run's Phase A timestamp, do the mutations, write that timestamp to =.ai/last-triage-intake= exactly as a normal run does (per "Capture the Phase A timestamp") — and commit + push the batch. Then *keep looping* (next sweep on the normal interval). This is the "flush the batch and carry on" checkpoint.
+- *"stop the triage"* — the same close processing, then *stop the loop* and revert to manual (on-demand) triage.
+
+A close is the only point auto mode advances the sentinel or commits. Between closes the engine state is untouched — that is what makes a 20-minute sweep cheap and non-destructive, and it preserves the engine invariant: the sentinel still means "everything before this timestamp has been scanned," it just advances once per close instead of once per run.
+
+** Why a separate mode
+
+The standard engine is one-shot and mutating — right for an at-the-desk "what's new?" glance, wrong for unattended polling: run every 20 minutes it would advance the sentinel past unprocessed items, spray reactive todos, take mail actions, and commit noise without review. Auto mode separates the cheap, frequent *watching* from the deliberate, gated *committing*, and adds the away-user's missing primitive — the running unacked-responses list.
* Reference
@@ -185,7 +307,7 @@ A plugin file declares exactly one source through a fixed shape:
*Body sections:*
- =** Scan= — the command(s) that fetch new/unread items since =<anchor>=, emitting raw items.
- =** Classify= — the source's per-bucket bias and noise patterns. *Deltas* from the engine's shared four-bucket model below, not a re-derivation.
-- =** Render= — the source's block in the Phase C summary. "Omit if empty."
+- =** Render= — the source's classification shape. Since the 2026-07-18 digest format, Render blocks are *inputs to the Phase C synthesis*, not displayed sections — the digest is source-agnostic (TASKS / FYI / MISC). Keep the shape: it defines what the source considers reportable, and the long-form breakdown (on request) still uses it.
- =** Actions= — the executable state-changes, one verb per line: =verb :: command template (parameterized by item id)=.
Template:
@@ -263,33 +385,37 @@ If both fail, fall through to the resolution order above (prep doc → session f
** Output Template
-The summary follows this shape (deltas only: a source with no changes gets no block; when *nothing* changed anywhere, the whole summary collapses to the one-line form below — plus any scan-failure banners and the suggested-actions line if actions are queued):
+The digest follows this shape (deltas only: a source with no changes contributes nothing; when *nothing* changed anywhere, the whole digest collapses to the one-line form below plus the timestamp — scan-failure banners always render regardless):
#+begin_example
17:39 sweep: no changes
#+end_example
-When there are changes, render one block per changed source in =ORDER=, using each plugin's =Render= shape:
+When there are changes:
#+begin_example
-**Anchor:** <previous run timestamp> → now (<elapsed> elapsed)
-**Loaded:** <general plugins> + <project plugins> (skipped: <disabled, with reason>)
+<⚠ SCAN FAILED / backlog banners, if any>
+
+==TASKS==
+- <solo-executable items first, then the rest; priority order within each;
+ one prose bullet each: who, what, why-now, source link>
+
+==FYI==
+- <substantive work context, no action owed; priority order>
-**Top signals to act on:**
-1. <terse Action description with link>
-2. ...
+==MISC==
+- <everything outside the project — personal mail, family, household;
+ priority order; action-ness stated inline>
-<one block per loaded source, in ORDER — see each plugin's Render>
+1. Close the triage — file todo.org tasks for the TASKS items, run the standard close
+2. Close the triage and execute the solo TASKS now, after filing the rest
+Append "and reroute" to either option to send the outside-project items to their owning projects.
+Or tell me what you'd like handled differently.
-**Suggested actions:**
-- Trash N noise items
-- Mark-read M keep items
-- Respond to <invite>
-- Merge PRs #X and #Y
-- ...
+<run timestamp — always the final line>
#+end_example
-Order matters: top-signals first because that's what Craig reads in 30 seconds between meetings. Per-source detail second. Suggested actions last because they require a decision.
+Option 2 appears only when solo-executable TASKS exist; the reroute line only when MISC is non-empty. Sections with nothing in them are omitted. No per-source blocks, no itemized suggested-actions list — the close owns the routine hygiene. The long-form per-source breakdown is available on request only.
** Common Mistakes
@@ -297,11 +423,11 @@ Order matters: top-signals first because that's what Craig reads in 30 seconds b
1. *Globbing only =.ai/workflows/= and missing the project plugins.* The single most damaging failure mode — the sweep runs with half its sources and the omission is invisible (a missing source looks identical to a quiet one). Phase 0 globs *both* =.ai/workflows/triage-intake.*.org= and =.ai/project-workflows/triage-intake.*.org=, every run, and announces the loaded set.
2. *Running Phase A sequentially.* Send every enabled source's scan in one message — the whole point is parallelism.
3. *Wiring a source into the engine.* Sources live in plugin files, never here. If you find yourself editing this file to add an account, repo, or channel, stop — write or edit a =triage-intake.<source>.org= plugin instead.
-4. *Executing actions without explicit confirmation.* Phase D runs only after Craig says "yes" or picks a subset.
+4. *Leaving the triage open, or re-asking about the routine hygiene.* The close is the default next action after the digest (Craig's 2026-07-18 ruling — no exceptions unless he says hold). Mail hygiene (trash/mark-read/star) runs at close without itemized confirmation. What still needs explicit confirmation: anything sending prose under Craig's name, and destructive actions beyond mail hygiene (branch deletes, PR merges).
5. *Forgetting to set the sentinel at the end.* Without it, the next run re-scans the same window.
6. *Using mtime instead of content for the sentinel.* Plain =touch= writes /now/ to mtime, stranding items posted between Phase A and end of run. =touch -d "@$PHASE_A_TS"= fixes the time but mtime is per-machine — git tracks content, not metadata, so the anchor doesn't survive a clone or cross-machine sync. Always write the epoch into the file's *content*.
7. *Running this alongside daily-prep.* Daily-prep already does this as Phase 3 — don't duplicate.
-8. *Mixing Action and FYI in the top-signals list.* Top signals = Action only. FYI lives in the per-source detail.
+8. *Mixing sections.* TASKS = work items needing Craig only; work FYIs never appear there. Outside-project items always land in MISC, even when they carry an action. Solo items lead TASKS; priority order inside every section.
9. *Reporting a failed or skipped scan as a quiet source.* A hung receive, a dead daemon, or a skipped spin-up looks identical to "no new messages" in the output unless it's flagged. The 2026-06-10 sweep shipped with Signal silently missing because the scan hung on an account lock. Failures lead the summary, in their own banner line.
10. *Rendering a per-source quiet roll-call.* "Calendar — quiet" / "PRs — nothing new" lines on every silent source bury the one change that matters and pad a no-change sweep into a report. Deltas only: changed sources get blocks, unchanged sources get nothing, and an all-quiet sweep is one line (Craig's 2026-06-11 ruling in Phase C).
@@ -314,6 +440,15 @@ Update the engine as the orchestration pattern evolves; update a plugin as its s
*** Updates and Learnings
+**** 2026-07-20: Phone delivery for signal sweeps (=agent-text=, send half)
+Auto mode now pushes a full-three-section sweep to Craig's phone over Signal via =agent-text=, the away-from-desk delivery the retired ntfy design carried before ntfy was torn down (2026-07-04). Transport is =agent-text= (the renamed Signal pager), not ntfy. Signal-only by Craig's 2026-07-20 ruling: a quiet sweep's =nothing= heartbeat never reaches the phone — silent-until-signal governs the phone channel too, so the higher-cost channel only fires when a sweep has real signal, while the in-session heartbeat stays as proof the loop ran. Falls back to inline when =agent-text= is absent. Only the send half ships; reply polling (the old =phone-recv=) waits on the reply-correlation follow-up, because a Signal reply fans out to every linked device and neither knows which page it answers.
+
+**** 2026-07-18: Three-section digest + close-by-default (Phase C/D rewrite)
+Craig's ruling after a 42h-gap sweep where the long-form report (top signals + per-source breakdown + 7-option action menu) was followed by "summarize the notable items" — and the digest that answered it was the report he wanted first. Phase C now renders ==TASKS== (work items needing Craig, solo-executable first, priority order) / ==FYI== (work context, no action owed) / ==MISC== (everything outside the project, actions stated inline), then exactly two options (close-and-file / close-and-execute-solo, the latter only when solo items exist), timestamp last. Per-source blocks and the itemized action menu are gone from the default surface (long form on request). Phase D became the close: it runs as the next action no matter what Craig replies (unless he explicitly holds), includes the mail hygiene on every scanned account without itemized confirmation, files the TASKS, clears resolved unacked items, advances the sentinel, and tears down started services. Solo = mechanical + standing-approved + no prose under Craig's name; prose sends and destructive non-mail actions stay gated. The stay-open-until-confirmed exit loop is retired. Same-day addendum: the "and reroute" modifier ("1 and reroute") — MISC items are surfaced-only by default (never filed to this project's todo.org); appending the modifier delivers each outside-project item to its owner's inbox via inbox-send per the cross-project rule.
+
+**** 2026-06-15: Auto mode (unattended monitoring)
+Added a self-running mode for when Craig is away but wants tight awareness — a =/loop= in the live session running accumulate-don't-mutate sweeps with "close the triage" / "stop the triage" as the gated checkpoint. Born the morning Craig cleared his day for a family emergency and wanted the desk watched while in and out. Design decisions (work-project proposal, ratified by Craig 2026-06-15): the unacked-responses list is durable in =.ai/triage-intake-unacked.org= (survives a crash/clear, the away-from-desk case it exists for); the sentinel advances only at close, preserving the scanned-before invariant; delivery is an in-session loop so MCP auth is inherited (a detached cron schedule belongs to the morning-ops orchestrator, not here, because of the headless-auth wall); it stays a mode of this engine, distinct from but reusable by that orchestrator. Same-day addendum (work, 2026-06-15): each sweep ends with a date/time/timezone stamp on its own final line (printed on quiet sweeps too, as proof the loop ran) so an away reader gauges freshness at a glance.
+
**** 2026-05-01: Initial creation
Extracted from daily-prep's Phase 3 pattern as a standalone, lightweight, between-meetings sweep.
@@ -327,7 +462,7 @@ The sentinel is checked into git, but git tracks content, not mtime — so an mt
Craig, via the work project's same-day handoff: "we only need to report if anything's changed when we do triage intake." Sweep summaries report deltas only — a new invite, a new/moved/cancelled event, a new message needing attention. Unchanged sources get no block (the "Calendar — quiet" roll-call is retired), and an all-quiet sweep renders as a single "HH:MM sweep: no changes" line. Failures keep their loud banner (never folded into the no-change line) and the suggested-actions line stays when actions are queued. Same ruling: the telegram plugin's dev-community group traffic is dropped from reports entirely unless Craig asks (see that plugin's 2026-06-11 note).
**** 2026-06-10: Loud failure surfacing (Phase C item 0 + Common Mistake 9)
-Craig: "highlight any failures in daily triage loudly. I get important communication from all these channels." Trigger: the 2026-06-10 sweep shipped with Signal silently missing — a standalone receive hung on the account lock while the signel daemon owned it, and the failure looked identical to a quiet source. Failures now lead the summary in a ⚠ SCAN FAILED banner; the telegram plugin's failure path points at this rule.
+Craig: "highlight any failures in daily triage loudly. I get important communication from all these channels." Trigger: the 2026-06-10 sweep shipped with Signal silently missing — a standalone receive hung on the signal-cli account lock while another client held it, and the failure looked identical to a quiet source. Failures now lead the summary in a ⚠ SCAN FAILED banner; the telegram plugin's failure path points at this rule.
**** 2026-05-26: Refactor into engine + source plugins
Split the monolithic workflow into a source-agnostic engine (this file) and per-source plugins named =triage-intake.<source>.org=. The engine carries the anchor/sentinel logic, the four-bucket model, the Phase A-D orchestration, the todo.org persistence convention, and the exit criteria. Each source's scan/classify/render/action knowledge moved to its own plugin. General plugins (personal-gmail, personal-calendar, cmail, github-prs) live in =.ai/workflows/= and are template-synced; project-specific plugins (a work project's Linear, work Gmail, work Slack, enterprise PRs) live in the project's =.ai/project-workflows/= and are never synced. Phase 0 globs *both* directories — the loud requirement, because missing the project dir silently halves the sweep. Naming convention: first dot is the engine/plugin boundary, deeper dots reserved for sub-adapters. This removed all DeepSat/Linear specifics from the engine; they become work-project plugins.
diff --git a/claude-templates/.ai/workflows/triage-intake.personal-calendar.org b/claude-templates/.ai/workflows/triage-intake.personal-calendar.org
index bf7d543..b5ee67a 100644
--- a/claude-templates/.ai/workflows/triage-intake.personal-calendar.org
+++ b/claude-templates/.ai/workflows/triage-intake.personal-calendar.org
@@ -1,5 +1,5 @@
#+TITLE: Triage Intake — Personal Calendar Source
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-26
# Source plugin for the triage-intake engine. See triage-intake.org for the
diff --git a/claude-templates/.ai/workflows/triage-intake.personal-gmail.org b/claude-templates/.ai/workflows/triage-intake.personal-gmail.org
index aa0554d..7fb1231 100644
--- a/claude-templates/.ai/workflows/triage-intake.personal-gmail.org
+++ b/claude-templates/.ai/workflows/triage-intake.personal-gmail.org
@@ -1,5 +1,5 @@
#+TITLE: Triage Intake — Personal Gmail Source
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-26
# Source plugin for the triage-intake engine. See triage-intake.org for the
@@ -21,10 +21,29 @@ Personal Gmail unread in the inbox since the anchor:
mcp__google-docs-personal__listMessages q="is:unread in:inbox after:<anchor-epoch>" maxResults=100
#+end_src
-⚠ *Express the cutoff as the literal UNIX epoch* — =after:1778856990=, not =after:YYYY/MM/DD=. Gmail's =after:YYYY/MM/DD= operator only supports day resolution; the =YYYY/MM/DD HH:MM:SS= form is NOT valid syntax — Gmail parses the space as a term separator, treats =HH:MM:SS= as a search term that never matches, and returns 0 results, silently masking unread mail. The engine supplies =<anchor-epoch>= because this source declares =ANCHOR: epoch=.
+⚠ *Express every anchor cutoff as the literal UNIX epoch* — =after:1784177122= and =before:1784177122= for the same anchor, never the =YYYY/MM/DD= form. This governs *both* anchored queries: the scan above and the backlog-residue probe below. They must meet at the same instant or mail falls between them permanently. Gmail's day-resolution operators fail two different ways: =after:YYYY/MM/DD HH:MM:SS= is not valid syntax at all — Gmail parses the space as a term separator, treats =HH:MM:SS= as a search term that never matches, and returns 0 results, silently masking unread mail — while =before:YYYY/MM/DD= is valid but excludes the named day entirely, so pairing it with a second-resolution scan leaves the whole anchor day covered by neither query. The engine supplies =<anchor-epoch>= because this source declares =ANCHOR: epoch=.
+
+The rule binds the *anchor* windows only. The date-slice walk below deliberately uses =before:<oldest-full-day-seen>= at day resolution — safe there because consecutive slices overlap and get deduped by message id.
⚠ *Do NOT add =-category:promotions -category:social=.* That filter masked 67 promo+social messages across two runs (2026-05-04, 2026-05-06), both needing a follow-up sweep. Pull the full unfiltered set; the trash-leaning bias in Classify handles promotions and social directly.
+⚠ *The MCP caps at =maxResults=100= and exposes NO =pageToken= parameter.* The response carries a =nextPageToken=, but the tool can't consume it, so a pile over 100 is silently truncated — the tail below the cap never gets classified, and every later anchored sweep skips it (it predates the new anchor). This is exactly how a 300+ backlog accumulated invisibly by 2026-07-08. Two consequences:
+
+- *Never treat a 100-row result as complete.* When a scan returns exactly 100, walk the tail in *date slices*: re-query with =before:<oldest-full-day-seen>= (day resolution), repeat until a page returns fewer than 100, dedupe by message id across slices (the day-resolution boundary overlaps).
+- *Never report =resultSizeEstimate= as a count.* It's unreliable — observed stuck at "201" across three different queries whose real union exceeded 300.
+
+*** Backlog-residue check (every sweep — cheap, mandatory)
+
+The anchored scan is blind to anything unread from *before* the anchor. After it, run one probe for pre-anchor residue:
+
+#+begin_src text
+mcp__google-docs-personal__listMessages q="is:unread in:inbox before:<anchor-epoch>" maxResults=5
+#+end_src
+
+The cutoff is the epoch, matching the scan's =after:<anchor-epoch>= — see the epoch rule above.
+
+If it returns any messages, surface one loud line in the sweep summary: "Backlog: unread predating the anchor exists (N+ shown; date-slice to inventory)" and offer a backlog sweep. Never fold the residue into a quiet sweep — an anchored "no changes" claim is only true for the window the scan saw. (Added 2026-07-08 after ~300 pre-anchor unread accumulated unseen; the probe returns actual messages, so it works where the estimate lies. Shipped with a day-resolution cutoff that hid the entire anchor day; fixed to epoch 2026-07-16 after a home sweep reported the backlog clear while two July-15 messages sat unread.)
+
** Classify
Bias: *trash-leaning* — personal Gmail is high noise volume.
diff --git a/claude-templates/.ai/workflows/triage-intake.telegram.org b/claude-templates/.ai/workflows/triage-intake.telegram.org
index 9caa4e1..1319da5 100644
--- a/claude-templates/.ai/workflows/triage-intake.telegram.org
+++ b/claude-templates/.ai/workflows/triage-intake.telegram.org
@@ -1,5 +1,5 @@
#+TITLE: Triage Intake — Telegram Source
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-06-09
# Source plugin for the triage-intake engine. See triage-intake.org for the
@@ -30,12 +30,27 @@ Telega does not autostart with the Emacs daemon. "Down" is its normal state
unless Craig has Telegram open in Emacs. The scan therefore runs the full
lifecycle every time, never skips because the server is down:
+⚠ *DOWN / not-loaded is the TRIGGER to launch, never a reason to skip or fail.*
+This is the exact mistake two projects (work + home, 2026-07-24) made: they
+probed telega, saw =(telega-server-live-p)= nil or telega not =featurep=, and
+reported =SCAN FAILED: telegram — not loaded= or a silent SKIP — a *blind*
+sweep — instead of running Step 1 to start it. A down or unloaded telega is the
+normal entry state; =(telega t)= both LOADS the package and STARTS the docker
+server (work confirmed: down → =(telega t)= → Ready, 18 chats). So the plugin
+MUST run Step 1's launch whenever telega is down/unloaded, wait for Ready, then
+scan. =SCAN FAILED= is reserved for a launch that was actually ATTEMPTED and did
+not reach Ready (image missing, server crash on start, daemon unreachable) —
+never for the pre-launch down state itself. The =:ENABLED:= guard above tests
+whether telega is INSTALLED (=fboundp=), not whether the server is up; a down
+server never disables the source.
+
1. Record prior state: TELEGA_WAS_RUNNING via (telega-server-live-p).
2. Launch (only if not running):
emacsclient -e "(progn (setq telega-use-docker t) (telega t) 'started)"
- The setq is mandatory defense: tdlib segfaults outside docker mode
- (2026-06-09), and Craig's daemon currently has telega-use-docker nil.
- Wait ~2s for Ready, then (telega--loadChats 'main) until telega--chats
+ The setq is mandatory defense: tdlib crashed in native mode when this was
+ set up (2026-06-09) — a separate matter from the SEGFAULT gotcha, which is
+ about the loadChats argument — and Craig's daemon defaults to nil.
+ Wait ~2s for Ready, then (telega--loadChats '(:@type "chatListMain")) until telega--chats
is populated.
3. Check messages: the maphash unread scan in ** Scan Step 2 (filters the
messageContactRegistered join-notice noise).
@@ -48,10 +63,13 @@ lifecycle every time, never skips because the server is down:
Verify: telega-server-live-p → nil, no zevlg/telega-server container in
docker ps. If Craig had it running, leave it untouched.
-If any lifecycle step fails (docker image missing, server crash, daemon
-unreachable), the sweep reports it as SCAN FAILED at the top of the summary
-per the engine's failure rule — never as a silent skip. Craig gets real
-traffic here.
+If any lifecycle step fails *after the launch was attempted* (docker image
+missing, server crash on start, daemon unreachable, Ready never reached), the
+sweep reports it as SCAN FAILED at the top of the summary per the engine's
+failure rule — never as a silent skip. This does NOT cover the ordinary
+pre-launch down state: a down server means "run Step 1," not "SCAN FAILED."
+Craig gets real traffic here, so a blind sweep that skipped the launch is worse
+than a clean failure — it hides real unread messages behind a false all-clear.
** Scan
@@ -85,22 +103,58 @@ TELEGA_WAS_RUNNING=$(emacsclient -e "(and (fboundp 'telega-server-live-p) (teleg
*** Step 1 — start (docker mode) if not already running, wait for Ready
#+begin_src bash
-# `(telega t)` starts without popping the root buffer. Docker mode (the stable
-# path — see the SEGFAULT gotcha) reconnects the persisted ~/.telega session in
-# ~2s. Then load the main chat list so telega--chats populates.
+# `(telega t)` starts without popping the root buffer. Docker mode reconnects the
+# persisted ~/.telega session in ~2s. Then load the main chat list so
+# telega--chats populates.
+#
+# The `(setq telega-use-docker t)` is mandatory and must come BEFORE `(telega t)`:
+# tdlib crashed in native mode when this was first set up (2026-06-09), and the
+# daemon's default is nil unless something (e.g. an Emacs-config :custom) has
+# already forced it. It was missing here while the Quick Reference required it —
+# a session that started telega without it on a native-mode daemon would take the
+# untested path. Match the Quick Reference exactly.
+#
+# Note this is a SEPARATE concern from the SEGFAULT gotcha below: that gotcha is
+# about the `loadChats` argument, and the deaths it explains happened in docker
+# mode. Docker mode is not a defense against it, and it is not evidence for
+# docker mode. Keep both.
emacsclient -e "(progn
+ (setq telega-use-docker t)
(unless (and (fboundp 'telega-server-live-p) (telega-server-live-p)) (telega t))
'started)"
# Poll until Ready with chats synced, or a crash/timeout. Background this with an
# until-loop so the wait doesn't block; exit on Ready-with-chats OR an abnormal
# server exit. Then force a chat-list load if the hash is thin:
-emacsclient -e "(progn (ignore-errors (telega--loadChats 'main)) (ignore-errors (telega--loadChats 'main)) 'loaded)"
+# NOTE: the chat-list argument must be a TL object, not the symbol 'main.
+# `telega--loadChats' puts it straight into the request as :chat_list, and a
+# bare symbol kills the server outright (see the SEGFAULT gotcha below).
+#
+# The liveness check on the tail is the load's only failure signal. `ignore-errors'
+# catches nothing here, because a bad argument kills the server process rather than
+# signalling in elisp, so without this the call returns 'loaded either way.
+# The `fboundp' guard matches Step 0: if the launch failed outright telega is not
+# loaded, and that should read as 'server-died like any other failure rather than
+# signalling void-function.
+emacsclient -e "(progn (ignore-errors (telega--loadChats '(:@type \"chatListMain\"))) (ignore-errors (telega--loadChats '(:@type \"chatListMain\"))) (if (and (fboundp 'telega-server-live-p) (telega-server-live-p)) 'loaded 'server-died))"
#+end_src
On a persisted session telega reaches status "Ready" within ~2s; the chat list
loads over a few more. If =(hash-table-count telega--chats)= is 0 or thin,
re-issue =telega--loadChats= and poll until it stabilizes.
+⚠ *=server-died= is SCAN FAILED, never a quiet account.* A server that dies
+during the load leaves a thin =telega--chats= hash, and a thin hash reads exactly
+like an account with little unread. That is the same false all-clear the
+down/not-loaded rule exists to prevent, arriving one step later in the lifecycle.
+It also fits the SCAN FAILED definition above: the launch was attempted and did
+not hold. So on =server-died=, report SCAN FAILED rather than scanning, and never
+report a low unread count from that run.
+
+This is the independent evidence the SEGFAULT gotcha asks for when it says to
+treat a short chat list as a real short list. Without the check there is no way
+to tell the two apart, which is how the =loadChats= crash stayed invisible
+through two investigations.
+
*** Step 2 — read unread, classified by last-message type
The single most important filter: =messageContactRegistered=. Telegram counts a
@@ -157,24 +211,61 @@ stays non-nil). =telega-server-kill= is what actually stops the server. Call
left in =docker ps=. Skipping this whole branch when =TELEGA_WAS_RUNNING= is t is
the point of Step 0: never tear down a session Craig is actively using.
-⚠ *SEGFAULT GOTCHA — crashes are spontaneous; treat server death as routine.*
-The dockerized =telega-server= (=zevlg/telega-server:latest=, image built
-2026-06-04, tdlib 1.8.64) SIGSEGVs (exit 139) *on its own*, minutes-to-hours
-into a session — 11 host coredumps between 2026-06-09 and 2026-06-11, several at
-times when no triage verb was running. The 2026-06-11 investigation reproduced
-the crash-free verbs and the spontaneous deaths side by side: coredump
-backtraces show a corrupted stack (memory corruption in the musl build), and
-no newer image exists upstream. Earlier theories — "native mode is the trigger",
-"toggle-read is the trigger" — were timing coincidences; the verbs are sound.
+⚠ *SEGFAULT GOTCHA — this was our bug, not tdlib's. Root-caused 2026-07-28.*
+=telega-server= dies with =Unexpected char 'm' in plist value= followed by
+=Assertion failed: false (telega-dat.c: tdat_plist_value: 500)=. The cause was
+this workflow: Step 1 called =(telega--loadChats 'main)=.
+
+The chain. =telega--loadChats= is a raw TL wrapper — it drops its argument into
+the request as =:chat_list= with no conversion. =telega-server--send= then
+=prin1='s the whole plist, and =telega--tl-pack= passes atoms through untouched,
+so the symbol goes out on the wire bare as =main=. The C parser
+(=server/telega-dat.c=, =tdat_plist_value=) accepts only =(=, =[=, ="=, =-=, a
+digit, =t=, =:=, or =n= to start a value. It hits =m=, prints that line, and
+calls =assert(false)=, which aborts the process. The =m= in the error is
+literally the first character of =main=.
+
+The symbol shorthand is real but belongs to a different layer:
+=telega-filter.el= and =telega-folders.el= convert =(eq cl-fspec 'main)= into
+='(:@type "chatListMain")=. The raw TL layer never does. telega's own callers
+always pass the object (=telega.el:290=, =telega-tdlib-events.el:516=).
+
+Proved by experiment, not inference (2026-07-28): from a live Ready server,
+=(telega--loadChats 'main)= killed it within seconds and added one coredump,
+with that exact assertion; a restart plus =(telega--loadChats '(:@type
+"chatListMain"))= survived three consecutive calls with no new coredump and no
+assertion.
+
+*The previous entry here was wrong and cost real time.* It recorded the deaths
+as spontaneous musl memory corruption and declared "the verbs are sound", which
+sent later investigations at the docker image and tdlib versions instead of at
+this file. The corrupted stack in the backtraces is what an =assert= abort looks
+like, not independent evidence of a memory bug. If crashes are ever seen again
+with *no* triage verb running, that is a genuinely separate cause and needs its
+own investigation — do not reuse the old spontaneous-crash story to explain it.
+
+*This crash kills a scan; it does not silently shorten one.* An earlier draft of
+this section claimed the reported "19 chats of ~50" was truncation caused by the
+bad call. That was wrong, and work disproved it at the wire level on 2026-07-28:
+with the corrected call their count is 19 before the first load and 19 after five
+(four on =chatListMain=, one on =chatListArchive=). Nineteen is the real size of
+that account. The same reading here — 19 stable across three corrected loads —
+was already sitting in the evidence and should have retired the claim before it
+was written down. Treat a short chat list as a real short list unless something
+independently shows the server died mid-sync.
+
+=ignore-errors= around the call never helped — the failure is the server process
+dying, not an elisp signal, so there is nothing for it to catch. That is why the
+death is easy to miss from inside elisp, and why a caller should check
+=(process-live-p (telega-server--proc))= after a load rather than trusting a
+returned value.
Operationally: docker mode stays mandatory (=telega-use-docker= = t; the setq
before =(telega t)= is still the right defense), and *every action batch checks
the server first* — =(process-live-p (telega-server--proc))= — restarting via
-=(telega t)= when dead and re-checking Ready before firing verbs. A mid-sweep
-death is recoverable, not an abort: restart, confirm Ready, resume. Durable-fix
-candidates if the crashing gets worse: pin a pre-2026-06 image digest, build
-=telega-server= natively against tdlib, or report upstream to zevlg with the
-coredumps (=coredumpctl list /usr/bin/telega-server=).
+=(telega t)= when dead and re-checking Ready before firing verbs. Any argument
+handed to a =telega--*= TL wrapper must be a TL object or a plain
+string/number/list, never a bare symbol.
Defense in depth: even if the server does die, the scan still works because it
reads the cached =telega--chats= hash, not a live query. A dead server is
diff --git a/claude-templates/.ai/workflows/work-the-backlog.org b/claude-templates/.ai/workflows/work-the-backlog.org
new file mode 100644
index 0000000..ea3f402
--- /dev/null
+++ b/claude-templates/.ai/workflows/work-the-backlog.org
@@ -0,0 +1,266 @@
+#+TITLE: Work the Backlog
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-02
+
+* Overview
+
+The single home for the autonomous task-execution loop: take a set of marked, solo-doable tasks from the project's =todo.org= and work them unattended, each held to the full quality bar, under a fixed safety contract. Spec: =rulesets/docs/specs/2026-06-16-autonomous-batch-execution-spec.org=.
+
+Two callers feed it, differing only in how they build the task set and which session mode they pass:
+
+- The *inbox auto-loop* (=inbox.org= auto mode) chains here after its routing completes, with a tag/priority query, file-only mode, cap 1.
+- The *no-approvals speedrun* preset feeds an explicit ordered list with autonomous-commit + always-push + paging-on, after a pre-flight Q&A that front-loads every decision.
+
+This workflow owns the execution logic — eligibility gate, defer checklist, quality bar, run cap. Callers own input assembly and mode selection. Capture-routing (inbox surfaces) stays entirely in =inbox.org=; this file never reads an inbox.
+
+* When to Use This Workflow
+
+Invoked by its two callers, or directly by phrase:
+
+- *Speedrun triggers:* "speedrun", "no approvals speedrun", "speedrun these: <task set>" — run the no-approvals speedrun preset (below). The word "speedrun" always routes here, even when the phrase also says "no approvals": plain =no-approvals.org= is the general session mode; the speedrun is this workflow's preset over an explicit task set.
+- *Loop caller:* =inbox.org= auto mode chains here after its routing (below). Not phrase-triggered.
+
+Manual fallback: "work the backlog" / "work the backlog with <task set>" — gather the three inputs below (ask for whichever are missing, defaulting to file-only mode; default cap is the list length for an explicit set, 1 for a query) and run the loop.
+
+* Inputs — the caller contract
+
+A caller hands this workflow three things:
+
+1. *A task set* — an ordered list of candidate task headings from the project's =todo.org=. Either an explicit ordered list (speedrun) or the result of a tag/priority query (the loop). The loop does not care how the set was assembled; it receives an ordered list of candidates.
+2. *A session mode* — two orthogonal flags:
+ - *Commit autonomy:* =file-only= (default) or =autonomous-commit=. See "Commit autonomy" below.
+ - *Paging:* on or off. End-of-set only.
+3. *A run cap* — the hard maximum number of tasks to complete this run.
+
+It returns a per-task outcome and a run summary.
+
+* Outcomes — the per-task vocabulary
+
+Every task in the set ends in exactly one of:
+
+- =implemented-committed= — implemented, committed (and pushed per the project's flow) under =autonomous-commit=.
+- =implemented-diff-surfaced= — implemented, diff surfaced, *not* committed (=file-only=).
+- =deferred-VERIFY= — a defer-checklist hit; a =VERIFY= filed naming what's missing or risky.
+- =dropped-by-craig= — removed from the run at the speedrun pre-flight Q&A ("skip this").
+- =skipped-ineligible= — failed the mechanical eligibility gate.
+- =failed= — implementation was attempted and abandoned: the tree is left working (never commit a broken state), the failure is surfaced in the run summary, and the run continues to the next task.
+
+The run summary lists each task with its outcome, plus the remaining set when the cap stopped the run.
+
+* The loop
+
+For the task set, in order, until the run cap is hit:
+
+1. *Eligibility gate* (below). Ineligible → record =skipped-ineligible=, next task.
+2. *Scope read* of the relevant code. Cheap; just enough to run the defer checklist.
+3. *Defer checklist* (below). Any hit → defer: file the =VERIFY= naming the gap and record =deferred-VERIFY= (or, under the speedrun preset, route a quick-question gap to the pre-flight Q&A), next task.
+4. *Implement* under the project's commit discipline: TDD red→green→refactor, then the isolated adversarial review (=publish= Step 1) with its re-review loop, fix all Critical/Important findings, then close the task per =todo-format.md='s completion rules. Decompose into as many logical commits as the change needs — size is not capped. If implementation fails partway, leave the tree working, record =failed=, surface it, and continue to the next task.
+5. *Commit autonomy branch:*
+ - =file-only= → surface the diff, do *not* commit. Record =implemented-diff-surfaced=.
+ - =autonomous-commit= → =/voice personal= on the message, commit individually, push per the project's flow. Record =implemented-committed=.
+6. *Record metrics* for the task (the JSONL append — see Metrics below).
+7. Decrement the cap. At zero, stop.
+
+After the set: if the paging flag is set, fire the end-of-set page (below). Surface the run summary either way.
+
+* Eligibility gate — mechanical, no judgment
+
+A task is autonomous-safe when *both* hold. This layer is a lookup, not a judgment; all the judgment lives in the defer checklist.
+
+1. *Status is =TODO=* — never =VERIFY=, =DOING=, =DONE=, or =CANCELLED=. =VERIFY= marks "awaiting Craig's input"; auto-implementing one defeats the check it represents. The do-not-implement set is safe-by-omission: anything not plainly =TODO= (plus any project-declared "hold" marker) is out.
+2. *Tagged =:solo:=* — the autonomy tag, resolved against the project's priority/tag scheme header in =todo.org= (never hardcoded). =:solo:= carries the hard definition in =todo-format.md=: completable and verifiable without Craig beyond at most one or two quick decisions answerable up front, no design deliberation. A project whose scheme declares a different autonomous-safe tag set overrides the default.
+
+Terminology: *speedrunnable means tagged =:solo:=*. It does not mean =:quick:= or require =:quick:solo:=. The =TODO= status check above is the execution-state gate over that speedrunnable set.
+
+Priority and =:next:= drive *ordering* within the eligible set, not eligibility ([#A] before [#B] before [#C], then the author's ordering). =:quick:= is an effort hint for batching and duration estimates — never a gate.
+
+Task *size* is deliberately absent from this gate. A large but well-specified, decision-free task is in scope and gets decomposed into per-logical-commit chunks during implementation. Size never sends a task away; only *deliberation* or *risk* does (the checklist below).
+
+*No scheme header → don't run.* The gate reads =:solo:= semantics from the project's scheme header; a =todo.org= without one leaves the tag undefined (=todo-format.md= makes the header mandatory). Surface that the header is missing and stop rather than guessing eligibility.
+
+* The defer checklist — act vs file
+
+After the scope read, run each eligible candidate through the checklist. Each item is a concrete, answerable question, not an adjective. *Any* hit — or any "unsure" — defers the task. Only a task that clears every item is implemented.
+
+1. *Test-writability (the keystone).* Can I write the failing test from the task text — plus any decisions gathered up front — without inventing a requirement? *No / unsure* → underspecified. Under the speedrun preset, if the gap is one or two quick answerable questions, route it to the pre-flight Q&A; otherwise file a =VERIFY= noting what's missing. Under the unattended loop, file the =VERIFY= (no one to ask). *Open-ended goals are a specific, recognizable failure of this item:* a task phrased as an absence ("find bugs until none remain," "refactor until nothing worthwhile is left," "clean it up") has no writable acceptance test and so isn't really =:solo:=, even when tagged. Don't guess a stopping point — defer it and note that it needs measurable acceptance criteria (bound the surface, characterization net, dispositioned findings, objective floor — see =todo-format.md='s "Making an open-ended task measurable"). Once those are in the task body, it becomes runnable.
+2. *Data-loss / irreversible / external operation.* Does implementing it require any of: =rm= of non-scratch data, =git reset --hard= / force-push, =DROP= / =DELETE= / =TRUNCATE=, file truncate/overwrite of persisted content, a schema or data migration, any external or shared-state mutation, any credential touch? *Yes* → do NOT implement; file a =VERIFY= naming the risk. This is the hard safety gate; an upfront answer never overrides it without an explicit checkpoint.
+3. *Already-satisfied.* Does the scope read show the desired end-state already holds? *Yes* → file a =VERIFY= noting it and move on. Don't make a no-op change.
+4. *Design deliberation.* Does the task carry an unresolved design question, a "weigh these approaches" with real tradeoffs, or a TBD that isn't a quick factual answer? *Yes* → under the speedrun preset, if it collapses to one or two quick questions, route to the pre-flight Q&A; otherwise file and surface as a =/start-work= candidate. Under the loop, file. The discriminator is *quick-answerable question* vs *deliberation* — never task size.
+
+When genuinely unsure which side a task falls on, defer — a wrong auto-implement costs a revert *and* the next-session correction.
+
+** Filing the deferral =VERIFY=
+
+Every checklist hit files a =VERIFY= in the project's =todo.org=, per =todo-format.md='s VERIFY rules:
+
+- *Dedup first.* If a =VERIFY= sibling for this deferral already exists (a prior run filed it), don't file another — record the outcome as =deferred-VERIFY= with a "previously filed" note and move on. The deferred task keeps its =TODO= status and tags, so without this check every subsequent run would re-defer and re-file.
+- *Placement:* sibling of the deferred task (the deferred task is the trigger) — a =**= task gets its =VERIFY= at =**=, a =***= sub-task gets it at =***= under the same parent, never deeper.
+- *Heading:* carries the question or risk on its own ("VERIFY <topic> — migration touches persisted rows").
+- *Body:* which checklist item hit, what's missing or risky, and what answer or action would make the task runnable. For an already-satisfied hit, the evidence that the end-state already holds.
+
+** Routing a quick-question gap (speedrun only)
+
+Under the speedrun preset, a checklist-1 or checklist-4 hit that collapses to one or two quick answerable questions routes to the pre-flight Q&A instead of deferring (see the preset section below). The discriminator: a *quick question* is a factual or preference pick answerable in one line without weighing tradeoffs ("cap at 5 or 8?", "which config key name?"); *deliberation* is anything that needs tradeoffs weighed, options explored, or code read by Craig. A task needing three or more questions isn't quick-question-gapped — it's underspecified; file the =VERIFY=. Checklist item 2 (data-loss / irreversible) never routes to the Q&A: an upfront answer doesn't override the hard safety gate.
+
+The unattended loop has no one to ask — every hit defers there.
+
+* Per-task quality bar
+
+Autonomy changes who approves, not what quality means. Per task, non-negotiable:
+
+- *TDD* per =testing.md=: red first, green, refactor. The keystone checklist item already proved the failing test is writable.
+- *Verification* per =verification.md=: fresh evidence, full suite green before any commit.
+- *Isolated adversarial review* before every commit, dispatched per the =publish= skill's Step 1 — never an inline self-review, however small the diff. Critical and Important findings block until fixed, and each fix goes back to the *same* reviewer until it approves. Minor findings never earn another round.
+ - *When the review can't reach approval* — three rounds without it, a finding that recurs after being reported fixed, or a =Needs Discussion= verdict — the unattended run has no one to ask. Record the task =failed= with the standing findings in its result, leave the tree working, and continue to the next task. Never commit past a blocking finding because nobody is awake to adjudicate — an unreviewed commit landing overnight is the outcome this gate exists to prevent.
+- *=/voice personal=* on every commit message on the =autonomous-commit= path (or the patterns walked inline if the skill is unavailable), message printed inline so the log shows what landed.
+- *Task closure* per =todo-format.md=: depth-based completion (keyword + =CLOSED:= at level 2, dated rewrite at level 3+).
+- *One logical change per commit.* A large task becomes several commits, not one omnibus.
+
+* Commit autonomy
+
+=file-only= is the default: surface the diff, never commit. =autonomous-commit= is honored only when the project carries the commit-autonomy waiver, read fresh each run — never from memory of past runs or "this project usually allows it."
+
+The waiver lives in the project's =.ai/notes.org= *Workflow State* section as marker lines, the same shape as the workflow markers already there:
+
+#+begin_example
+:COMMIT_AUTONOMY: yes
+:LOOP_MAY_COMMIT: yes
+#+end_example
+
+- =:COMMIT_AUTONOMY: yes= — the project has the waiver. An =autonomous-commit= request (the speedrun preset, or a manual run asking for it) is honored.
+- =:LOOP_MAY_COMMIT: yes= — the *unattended loop caller* may also commit. It requires =:COMMIT_AUTONOMY:= alongside it; the split exists because "Craig-initiated speedrun may commit" and "the recurring loop may commit unattended" are different levels of trust. Without this flag the loop stays =file-only= even when the project holds the waiver.
+
+An absent marker means no. Anything other than a plain =yes= value also means no. The read is one grep of the Workflow State section — a lookup, not a judgment.
+
+*The degrade contract.* When a caller requests =autonomous-commit= and the required marker is missing, degrade to =file-only= and surface it in both the run intro and the run summary: "autonomous-commit requested, no :COMMIT_AUTONOMY: waiver in notes.org — running file-only." Never honor the request without the marker, and never drop to file-only silently — the first commits into a project that didn't opt in, the second hides why nothing got committed.
+
+* Bounding the run
+
+The cap is a hard per-run task ceiling passed by the caller — the kill switch a runaway can't exceed:
+
+- *Loop caller default: 1.* Implement the highest-priority eligible candidate, record, stop; the next tick continues.
+- *Speedrun: the length of the explicit list*, capped at a ceiling — the human bounded the set by naming it.
+
+Even the speedrun stops at the cap and surfaces (and, with paging on, pages) the remainder. The cap bounds task *count*, not cost; a token budget is logged as vNext.
+
+* Context hygiene — auto-flush between tasks
+
+Task boundaries are clean boundaries by construction: the previous task is closed and committed (or filed), nothing is half-edited. When the context window grows heavy mid-run, run the flush skill's *auto mode* between tasks: checkpoint the session anchor with the remaining task set, session mode, and cap in Next Steps (so the resumed context continues the run blind), arm the self-injection (=.ai/scripts/self-inject.sh= via =tmux run-shell -b=), and end the turn. The fresh context resumes from the anchor and works on. Unattended runs only — the keystroke-collision hazard and the full mechanism live in the flush skill.
+
+* End-of-set page
+
+With paging on, fire one page when the set is done or the cap is hit — end-of-set only, never per-task:
+
+#+begin_src sh
+notify info "Page" "<project>: <N> done, <M> remaining — <one-line summary>" --persist
+#+end_src
+
+=--persist= keeps it on screen until dismissed, and =info= is the notification urgency convention (persistent but never crash-scary). The notification fires when the set completes *or* the cap stops the run, either way exactly once. The message carries the project name, the completed count, and the remaining count (with skipped tasks noted in the run summary) so Craig can confirm ready and name the next project in one reply. =notify= is the desktop channel (the "page me" surface); a run that expects Craig to be away also fires =agent-text= with the same message (the Signal phone channel, "text me"). See protocols.org "Reaching Craig".
+
+* Metrics
+
+Each task outcome appends one JSON line to the project's =.ai/metrics/work-the-backlog.jsonl= — git-tracked, append-only, =jq=-queryable. Create the directory and file on the first append. Logging is a side effect only: a failed append surfaces a warning in the run summary but never blocks, reorders, or aborts execution.
+
+One record per task, written at the moment its outcome is decided:
+
+| Field | Meaning |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =ts= | ISO-8601 timestamp of the task outcome |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =run_id= | UUID shared by every record in one run (=uuidgen= at run start) |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =project= | project basename |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =caller= | =loop= / =speedrun= / =manual= |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =task= | the task heading (slug) |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =outcome= | =implemented-committed= / =implemented-diff= / =deferred-verify= / =skipped-ineligible= / |
+| | =dropped-by-craig= / =failed= |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =defer_reason= | =underspecified= / =data-loss= / =already-satisfied= / =needs-deliberation= — set on |
+| | =deferred-verify= records only |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =upfront_decision= | =true= when a pre-flight answer was recorded and used for this task |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =wall_clock_s= | seconds from task start to outcome |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =commit_sha= | committed tasks: the commit SHA (comma-separated when the task decomposed into several); empty |
+| | otherwise |
+|--------------------+-------------------------------------------------------------------------------------------------|
+| =review_findings= | count of =/review-code= Critical + Important findings on this task |
+|--------------------+-------------------------------------------------------------------------------------------------|
+
+The =outcome= slugs map one-to-one onto the outcome vocabulary above (=implemented-diff= is =implemented-diff-surfaced=; =deferred-verify= is =deferred-VERIFY=). Per-run rollups (attempted / completed / deferred / dropped, wall-clock total, findings per commit) are computed at synthesis, not stored per record. The =commit_sha= field is what the synthesis step's corrections signal keys on — whether a later commit reverted or hand-fixed an autonomous one — so never omit it on a committed task.
+
+* Caller: the inbox auto-loop
+
+=inbox.org= auto mode chains here as an explicit second step *after* its routing completes — never as a phase inside inbox processing. When a cycle files new items and Craig answers "run this batch next?" with yes, auto mode invokes this workflow with:
+
+- *Task set:* the eligibility query over the queued/filed items — status =TODO= + =:solo:= per the scheme header, priority-ordered.
+- *Session mode:* =file-only=, paging off. (A project carrying both =:COMMIT_AUTONOMY:= and =:LOOP_MAY_COMMIT:= markers opts the loop into commits — see Commit autonomy above.)
+- *Cap: 1.* The highest-priority eligible candidate runs, gets recorded, and the loop's next tick (or the next yes) continues from there.
+
+The loop has no human at kickoff of each task, so a needs-quick-decisions task defers with a =VERIFY= — the pre-flight Q&A is a speedrun capability, not a loop one. Startup and wrap-up never invoke this workflow.
+
+* Preset: the no-approvals speedrun
+
+The named preset is a label for one flag combination, not a second code path: *explicit ordered list + =autonomous-commit= + always-push + paging-on*, with every approval front-loaded into a single pre-flight step. "No approvals" means all input first, then hands-off — not no input ever. =autonomous-commit= still requires the =:COMMIT_AUTONOMY:= waiver (Commit autonomy above); without it the preset degrades to =file-only= and says so in the pre-flight intro.
+
+When Craig names a task set and says "speedrun":
+
+1. *Gather* the named task set.
+2. *Scope-read and classify* each task against the eligibility gate + defer checklist: *ready* (clears everything), *needs-quick-decisions* (one or two upfront-answerable questions — checklist item 1 or 4), or *drop* (data-loss/irreversible, or deliberation that isn't a quick question).
+3. *Order* the list — priority, then the author's ordering / =:next:=.
+4. *Intro the work* — present the ordered plan: what will run, what was dropped and why, and the batched questions for the needs-quick-decisions tasks.
+5. *Craig answers each question or says "skip this"* — a skip removes the task (recorded =dropped-by-craig=; the task itself stays =TODO=); an answer is recorded so implementation works from the decision, not a guess.
+6. *Run the finalized list autonomously* — no further approvals until done. Cap = the list length (the human bounded the set by naming it), still one commit per logical change, always-push per the project's flow, auto-flushing between tasks when the context grows heavy (see Context hygiene above).
+7. *End-of-set page* with completed + remaining + skipped.
+
+The batch-ask (step 4-5) is one message: each question names its task, puts the recommended answer at item 1 when there is one (per =interaction.md= — inline numbered, no popup), and offers "skip this" as the last option. Before the run starts, write each answer into its task's body in =todo.org= as a dated line — the implementation works from the recorded decision, and the record survives the session. The Q&A fires only under this preset; the loop caller never asks (its decision-needing tasks defer).
+
+*** Per-item disposition rule
+
+For every item the run picks up (this holds for any executing caller, including an auto-inbox-zero run given a standing yes):
+
+- *Feature-level task* → write a spec first (=spec-create=), don't implement directly. The spec is the run's deliverable for that item.
+- *Needs decisions you can't confidently guess* → file it as a =VERIFY= carrying the question (under this preset, one or two quick questions route to the pre-flight Q&A instead).
+- *Well-defined* → implement it, taking the time it needs.
+
+This extends the defer checklist: the checklist decides *act vs file*; this rule decides the *shape* of the act.
+
+* Synthesis: metrics → org-roam KB
+
+Trigger: "synthesize backlog metrics" (optionally a weekly scheduled run). This is the read side of the metrics log — Craig's ask was "gather data and create org-roam articles we can look at later," and this step is the second half. It is read-only over the logs plus exactly one KB write.
+
+1. *Gather the JSONL union.* Discover =.ai/metrics/work-the-backlog.jsonl= across the project roots (dirs carrying =.ai/protocols.org= under =~/code=, =~/projects=, =~/.emacs.d=). Classify each project per =knowledge-base.md= (work-root denylist, never inference) before reading it into the union.
+2. *Enforce personal-only.* A work-classified or unknown project's metrics never enter the KB write — they stay in that project's own log. Report the exclusion per the KB refusal contract: the classification, a one-line redacted summary, and where the data stayed.
+3. *Compute the rollups and trends.* Per run: attempted / completed / deferred (by reason) / dropped / failed, wall-clock total, commits landed, review findings per commit. Trends across runs: completion rate over time, defer-reason distribution, findings-per-commit trend.
+4. *Compute the corrections signal* — the key metric. For each =commit_sha= in the window, check that project's history for a later commit (within ~14 days) that reverts it or carries a fix touching the same files. A clean run is one whose autonomous commits survive untouched; a flagged run is what Craig reviews by hand. This is a cheap proxy, not proof — it flags candidates, it doesn't convict.
+5. *Write one KB node* at =~/org/roam/agents/YYYYMMDDHHMMSS-backlog-metrics-<window>.org= per =knowledge-base.md=: =:agent:metrics:= filetags, a concise title, the rollup table, the trend narrative, and =[[id:...]]= links to prior synthesis nodes so the series is traceable. Pull before writing, commit and push after — the normal KB session discipline.
+
+The KB node is the artifact Craig reads later: "are the runs completing more and getting corrected less?" should read off the trend table without touching raw logs. Synthesis never mutates the JSONL, todo.org, or any project tree.
+
+* Common Mistakes
+
+1. *Implementing a =VERIFY= or =DOING= task.* The gate is status =TODO= only — a =VERIFY= exists precisely because Craig's input is pending.
+2. *Treating =:quick:= as eligibility.* It's an effort hint. =:solo:= is the gate.
+3. *Deferring on size.* A large, well-specified, decision-free task runs — decomposed into logical commits. Size is not a checklist item.
+4. *Guessing past the keystone.* If the failing test isn't writable from the task text, the task isn't ready. Inventing the requirement is the failure the checklist exists to stop.
+5. *Rationalizing through the data-loss list.* "The migration is small" doesn't clear checklist item 2. Enumerated operations defer, full stop.
+6. *Committing in =file-only= mode.* The diff is the deliverable; the commit is Craig's.
+7. *One omnibus commit for the whole run.* Every logical change is its own reviewed commit.
+8. *Skipping =/review-code= or =/voice= because nobody's watching.* Autonomy removes interaction gates, never engineering-discipline gates (same contract as =no-approvals.org=).
+9. *Running past the cap.* The cap is the kill switch; hitting it means stop and surface, even mid-set.
+10. *Paging per-task.* One page, end of set.
+11. *Honoring =autonomous-commit= from memory.* The waiver is the marker line in =notes.org=, read fresh each run. "This project usually allows it" isn't a read.
+12. *Re-filing the same deferral =VERIFY= every run.* The deferred task stays =TODO=, so a run that skips the existing-sibling check spams =todo.org= with duplicates.
+13. *Routing a data-loss hit to the pre-flight Q&A.* Checklist item 2 is the hard gate — an upfront answer never clears it without an explicit checkpoint.
+
+* Living Document
+
+Refine as the dogfooding signal arrives — the metrics log and the corrections-in-next-session signal are the feedback loop. Fold recurring adjustments in rather than accumulating caller-side workarounds.
+
+* History
+
+Created 2026-07-02 as Phase 1 of the autonomous-batch execution spec, reconciling the inbox-zero "Phase E" proposal and the =.emacs.d= speedrun proposal into one execution loop. The auto-inbox-zero execute step in =inbox.org= reverted to routing-only in the same change so this file is the loop's only home. Phases 2-6 (same day) wired both callers, pinned the commit-autonomy waiver markers, fleshed the defer/Q&A/page mechanics, and added the metrics record + KB synthesis step.
diff --git a/claude-templates/.ai/workflows/wrap-it-up.org b/claude-templates/.ai/workflows/wrap-it-up.org
index 2d79795..a9a5895 100644
--- a/claude-templates/.ai/workflows/wrap-it-up.org
+++ b/claude-templates/.ai/workflows/wrap-it-up.org
@@ -1,10 +1,10 @@
#+TITLE: Session Wrap-Up Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-04-20
* Overview
-This workflow defines the process for ending a Claude Code session cleanly. It finalizes the session record, commits + pushes all work, and provides a warm handoff.
+This workflow defines the process for ending a Claude Code session cleanly. It finalizes the session record, commits + pushes all work, and provides a warm handoff. A bare wrap also tears the session down (kills the ai-term buffer + tmux session, restoring geometry); a qualified wrap keeps the buffer, and a shutdown wrap powers the machine off. The teardown variants are set by the trigger phrase (see Teardown mode below) and act only at the very end, in Step 6.
Triggered by Craig saying "wrap it up," "that's a wrap," "let's call it a wrap," or similar.
@@ -24,15 +24,53 @@ The wrap-up is complete when:
2. *File is archived.* =.ai/session-context.org= has been renamed to =.ai/sessions/YYYY-MM-DD-HH-MM-description.org=. The old path no longer exists.
3. *todo.org is clean.* Cleanup script ran. Any auto-fixes are staged for the wrap-up commit. Orphan planning lines surfaced for manual fix if there are any.
4. *Linear board is honest* (skip if project doesn't use Linear). Any Dev-Review ticket whose PR has merged was moved to Done or PM Acceptance per the classification rule.
-5. *Git state is clean.* All changes committed + pushed to all remotes. Working tree clean.
-6. *Valediction delivered.* Brief, warm closing with key accomplishments and reminders.
+5. *Git state is certified clean.* All changes are committed + pushed to all remotes, =git-worktree-gate certify= succeeded at the current HEAD, and the working tree has no staged, unstaged, untracked, submodule, or in-progress-operation state.
+6. *Valediction delivered.* Brief, warm closing with key accomplishments and reminders, ending with =session wrapped.= on its own line as the signoff marker.
The absence of =.ai/session-context.org= is the signal that the last session wrapped up cleanly. Its presence at session start means the previous session was interrupted.
+* Teardown mode (set from the trigger phrase)
+
+The wrap itself — Steps 1 through 5 — is identical in every mode. The trigger phrase only decides what Step 6 does once commit + push and the valediction are done. Resolve the mode from the phrase before starting:
+
+- *Teardown* (the default) — bare "wrap it up", "that's a wrap", "let's call it a wrap". The full wrap, then Step 6 kills the ai-term buffer + the =aiv-<project>= tmux session (which takes =claude= with it) and restores the saved window geometry. This is Craig's typical end-of-day case.
+- *No-teardown* — "wrap it up with summary" or "wrap it up and summarize". The full wrap, but Step 6 leaves the buffer and session intact so the summary stays readable. The explicit qualifier is what opts out of teardown.
+- *Shutdown* — "wrap it up and shutdown". The full wrap, then Step 6 gates on this being the only live ai-term session and powers the machine off. Shutdown supersedes teardown (killing the buffer is moot if the box is going down).
+
+Why teardown waits for Step 6 and runs through a hook, never inline: teardown kills the very tmux session =claude= runs in, so an inline kill would cut the valediction off before it renders. Step 6 instead drops a sentinel after everything else is verified, and the =Stop= hook (=ai-wrap-teardown.sh=) does the actual teardown when this response ends — by which point the valediction has already been delivered.
+
+This depends on three functions in =.emacs.d/modules/ai-term.el= (=cj/ai-term-quit=, =cj/ai-term-live-count=, =cj/ai-term-shutdown-countdown=) and on the =Stop= hook being wired in =settings.json= (=hooks/settings-snippet.json=). If =emacsclient= or the daemon is unreachable, the sentinel is cleared and the session simply stays up — teardown degrades to a no-op, never a wedge.
+
* The Workflow
+** Step 0: Refuse if sentry is live
+
+Before anything else, check whether sentry is running in this project. Sentry holds the working tree on its =sentry/<date>-<host>= branch and commits unattended; wrapping underneath it would archive the session anchor and tear down the buffer while the loop is still firing into it. If sentry's single-runner lock is held, stop and point at the shutdown path:
+
+#+begin_src bash
+proj="$(basename "$(git rev-parse --show-toplevel 2>/dev/null || pwd)")"
+if [ -x .ai/scripts/agent-lock ] && .ai/scripts/agent-lock status "sentry-$proj" | grep -q '^held'; then
+ echo "sentry is active — say 'stop sentry' first"
+ exit 1
+fi
+#+end_src
+
+The stop-sentry operation (defined in =sentry.org=) owns the shutdown: it cancels the loop, disposes of the branch, and walks the approval queue. Wrap-up carries only this one guard; a =stale= lock (a crashed cycle) doesn't block — only a live =held= lock does.
+
** Step 1: Finalize the Summary
+*** Work the Before-Close Queue (before the Summary)
+
+If the session anchor (=.ai/session-context.org=) carries a =* Before-Close Queue= heading with items, work them now, oldest-first, before writing the Summary, so any resulting edits ride this wrap's commit and get described in it. The queue is the "put X on the list" shorthand (see =protocols.org=, Colloquialisms and Expansions): session-scoped work Craig deferred to wrap time.
+
+Per item: do it if it's clear and bounded, or promote it to a =todo.org= task if it turns out to need its own session. Never drop an item silently. Remove each line as it's handled; if one can't be finished, surface it in the valediction (Step 5) and either leave a follow-up task or state why it's dropped.
+
+If there's no =* Before-Close Queue= heading, or it's empty, this step is a silent no-op.
+
+*** Early KB reflection (capture while fresh, before the Summary)
+
+Before distilling the Summary, while the session is still fresh, ask: what did this session learn worth remembering, for yourself or a future agent? Reflect and stage any candidate durable facts — a decision and its why, an environment gotcha, a reference pointer, a transferable lesson. Self-answer silently; this adds no interactive turn (Craig already authorized the wrap). The candidates flow straight into the KB promotion check below, which does the actual writing and the receipt — this is the capture half, that is the commit half, one pipeline, one receipt. Reflecting here rather than reconstructing learnings after the Summary is the point: the early ask is what keeps the receipt from defaulting to "promoted 0" out of fatigue.
+
Read through the =* Session Log= in =.ai/session-context.org=. Populate (or refine) the =* Summary= section:
- *Active Goal* — one or two sentences describing the session's focus
@@ -84,21 +122,21 @@ idseg="${AI_AGENT_ID:+${AI_AGENT_ID}-}"
mv "$sc" ".ai/sessions/${now}-${idseg}DESCRIPTION.org"
#+end_src
-Replace =DESCRIPTION= with your picked slug. (=AI_AGENT_ID= should be filename-safe; the recommended =host.project.runtime.shortid= shape already is.)
+Replace =DESCRIPTION= with your picked slug. (=AI_AGENT_ID= should be filename-safe and unique per run; the recommended =host.project.runtime.<epoch>= shape is both. The epoch on the tail keeps a re-run of the same logical agent from resolving to a prior run's leftover anchor. See protocols.org "Agent-scoped path".)
** Step 3: todo.org cleanup (hygiene + archive completed work)
If the project has a =todo.org= at its root, run the cleanup script before committing. Two passes, both fast and idempotent: a hygiene pass and an archive pass.
-*** Roam inbox sweep (inbox-zero)
+*** Roam inbox sweep (inbox roam mode)
-Before the cleanup scripts, sweep the roam global inbox (=~/org/roam/inbox.org=) for items that belong to this project, so any imported tasks get linted and ride the wrap commit. Delegate to [[file:inbox-zero.org][inbox-zero.org]] for the claimed set.
+Before the cleanup scripts, sweep the roam global inbox (=~/org/roam/inbox.org=) for items that belong to this project, so any imported tasks get linted and ride the wrap commit. Delegate to [[file:inbox.org][inbox.org]] roam mode for the claimed set.
#+begin_src bash
[ -f "$HOME/org/roam/inbox.org" ] && grep -cE '^\*\* ' "$HOME/org/roam/inbox.org" || true
#+end_src
-Skip-fast when nothing matches: if the roam clone isn't on this machine, or no item is prefixed for this project, this is a silent no-op. When claimed items exist, run inbox-zero's Phase B–C (file each into =todo.org=, then remove them from the shared inbox in a separate roam commit). Report the total count and how many appeared related to this project, per inbox-zero's scan-summary rule.
+Skip-fast when nothing matches: if the roam clone isn't on this machine, or no item is prefixed for this project, this is a silent no-op. When claimed items exist, run roam mode's Phase B–D (file each into =todo.org=, then remove them from the shared inbox and let =roam-sync= commit + push the edit). Report the total count and how many appeared related to this project, per roam mode's scan-summary rule.
*** Hygiene pass
@@ -121,6 +159,22 @@ Run the report-only variant first if you want to see what would change without w
emacs --batch -q -l .ai/scripts/todo-cleanup.el --check todo.org
#+end_src
+*** Convert done sub-tasks to dated entries
+
+#+begin_src bash
+[ -f todo.org ] && emacs --batch -q -l .ai/scripts/todo-cleanup.el --convert-subtasks todo.org
+#+end_src
+
+=--convert-subtasks= rewrites every heading at level 3 or deeper whose TODO state is DONE/CANCELLED/FAILED into a dated event-log entry (=<stars> YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <text>=), dropping the keyword, priority cookie, and tags, and removing the now-redundant =CLOSED:= line. This enforces the =todo-format.md= depth rule that a completed *sub-task* (a heading under a parent task) becomes dated history, not a lingering DONE keyword — a shape an interactive org close (=org-log-done= → DONE + CLOSED) never applies and =--archive-done= (level-2 only) never reaches. The timestamp comes from each entry's own =CLOSED= cookie; a date-only close yields =00:00:00=. Heading text is kept verbatim. Idempotent (an already-dated heading has no keyword to match), and a done sub-task with no parseable =CLOSED= is flagged and left alone rather than stamped with a fabricated date.
+
+Run this *before* =--archive-done= so that when a completed level-2 parent is archived, its sub-tasks already carry their dated form. Any rewrites show up in the wrap-up commit's diff for review before push.
+
+Preview without writing:
+
+#+begin_src bash
+emacs --batch -q -l .ai/scripts/todo-cleanup.el --convert-subtasks --check todo.org
+#+end_src
+
*** Archive completed work
#+begin_src bash
@@ -135,6 +189,16 @@ Preview the moves without writing:
emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done --check todo.org
#+end_src
+*** Clear temp/
+
+#+begin_src bash
+[ -d temp ] && find temp -mindepth 1 -delete && echo "temp/ cleared"
+#+end_src
+
+=temp/= holds throwaway artifacts — discarded prototypes, scratch output, intermediate data (see =working-files.md=). It's gitignored in every project, so nothing here rides a commit and nothing is recoverable from git once deleted. Clearing it at wrap is what keeps ephemeral work from silting up across sessions, and it's the counterpart to =working/=, which is tracked and *never* cleared here.
+
+Two guards. Confirm before deleting if =temp/= holds anything a reasonable reader would call in-progress rather than throwaway — misfiled work belongs in =working/=, so move it there instead of deleting it. And skip the step entirely in a project where =temp/= is not gitignored, since that means the project is using the directory for something else.
+
*** Sync child priorities
#+begin_src bash
@@ -170,9 +234,14 @@ else
followups=".ai/lint-followups.org"
fi
[ -f todo.org ] && emacs --batch -q -l .ai/scripts/lint-org.el \
- --followups-file="$followups" todo.org
+ --fix --followups-file="$followups" todo.org
#+end_src
+The =--fix= flag is required for the writes: lint-org's CLI default is
+report-only (a linter reports, it doesn't write), and this wrap-up pass is
+the deliberate exception that applies fixes — its diff rides the wrap-up
+commit for review.
+
=lint-org= runs =org-lint= over =todo.org=, auto-applies four mechanical
categories (=item-number= counters, bare =#+begin_src= → =#+begin_example=,
multi-line planning-info merged onto one line, =**X.**= → =*X.*=), and
@@ -204,7 +273,7 @@ For an interactive walk of the judgments mid-day, run =/lint-org todo.org=.
*** Inbox sanity check (surface unprocessed handoffs)
-If the project has an =inbox/= directory, verify it holds nothing but =.gitkeep=, =lint-followups.org= (the lint-org pipeline file the next morning's daily-prep consumes), and any explicitly-deferred =PROCESSED-*= files before the wrap completes. An inbox that arrived at session start with handoffs from other projects, or that received handoffs mid-session, needs the =process-inbox.org= workflow to run and apply its value-gate dispositions. Wrapping with a dirty inbox silently defers the work to next session and accumulates handoff debt that the sender can't see.
+If the project has an =inbox/= directory, verify it holds nothing but =.gitkeep=, =lint-followups.org= (the lint-org pipeline file the next morning's daily-prep consumes), and =PROCESSED-*= files before the wrap completes. An inbox that arrived at session start with handoffs from other projects, or that received handoffs mid-session, needs =inbox.org= process mode to run and apply its value-gate dispositions. Wrapping with an unprocessed inbox silently defers the work to next session and accumulates handoff debt that the sender can't see.
#+begin_src bash
unprocessed=$(find inbox -maxdepth 1 -type f \
@@ -213,7 +282,7 @@ unprocessed=$(find inbox -maxdepth 1 -type f \
! -name 'PROCESSED-*' \
2>/dev/null | wc -l)
if [ "$unprocessed" -gt 0 ]; then
- echo "wrap-up: inbox/ has $unprocessed unprocessed item(s). Run process-inbox.org before wrapping, or explicitly defer each item with a one-line reason in the valediction."
+ echo "wrap-up blocked: inbox/ has $unprocessed unprocessed item(s). Run inbox.org process mode before wrapping."
find inbox -maxdepth 1 -type f \
! -name '.gitkeep' \
! -name 'lint-followups.org' \
@@ -222,11 +291,37 @@ if [ "$unprocessed" -gt 0 ]; then
fi
#+end_src
-If the count is zero or the project has no =inbox/= directory, the check is a silent no-op. If non-zero, the wrap is incomplete by default. The user resolves each item (process now, defer with reason in the valediction, or delete with rationale) before the validation checklist passes.
+If the count is zero or the project has no =inbox/= directory, the check is a silent no-op. If non-zero, the wrap is blocked. Process each item through its value-gate disposition, or delete it only when that workflow's rationale authorizes deletion, before continuing.
The check exempts =lint-followups.org= explicitly because lint-org runs earlier in the same wrap-up workflow and writes its judgment items to that file in =inbox/= by design. The file is a pipeline artifact for the next morning's =daily-prep=, not a handoff that needs the value gate.
-This integrates with =process-inbox.org=, which stamps =:LAST_INBOX_PROCESS:= in =notes.org='s *Workflow State* section on completion. Wrap-up doesn't double-stamp. It only ensures the inbox carries nothing but the expected pipeline artifacts at session end.
+This integrates with =inbox.org= process mode, which stamps =:LAST_INBOX_PROCESS:= in =notes.org='s *Workflow State* section on completion. Wrap-up doesn't double-stamp. It only ensures the inbox carries nothing but the expected pipeline artifacts at session end.
+
+*** Cross-project router (optional — route filed keepers to their home projects)
+
+Runs directly after the inbox sanity check. The split between the two: the sanity check *gates* the wrap (a dirty inbox blocks until resolved); the router is *optional* (skipping it never blocks anything — the candidates just stay local until a future wrap). Spec: =docs/specs/wrapup-routing-spec.org= (D7/D8/D9).
+
+The candidate set is exactly the local tasks carrying a =:ROUTE_CANDIDATE:= property — keepers that inbox process mode filed this session whose inferred home is another project. Never scan the standing backlog.
+
+#+begin_src bash
+.ai/scripts/route-batch --list
+#+end_src
+
+*Empty set = zero interaction.* =--list= prints nothing when there are no candidates; continue the wrap silently — no prompt, no "0 items" line.
+
+When candidates exist, surface the batch as one line per task — the task heading, the destination project, the delivery mode (=inbox-send= file handoff), and the engine's confidence — then offer exactly two options: *go* (route the whole batch) or *skip* (leave everything local). Derive each confidence label by running the engine on the task's heading + body (=python3 .ai/scripts/route_recommend.py --item "..." --exclude "$(basename "$PWD")"=); label weak matches visibly ("weak — verify the destination") so a low-confidence route gets a human glance before the keystroke.
+
+On *go*:
+
+#+begin_src bash
+.ai/scripts/route-batch --go
+#+end_src
+
+Per candidate, the helper writes the task's subtree (children ride along; =:ROUTE_CANDIDATE:= stripped, headings promoted to top level) to a one-task handoff, delivers it via =inbox-send <destination> --file= (so the =from-<this-project>= provenance is stamped and the destination's inbox process mode dispositions it as a single item), and only after a successful send removes the subtree from the local =todo.org= — a single-file local edit the wrap is already committing. A failed send leaves that task in place and exits non-zero; report it and continue the wrap. Never write the destination's =todo.org= directly; its own inbox processing files the task per its conventions.
+
+On *skip*, leave every candidate in place, marker included — they resurface next wrap.
+
+Mis-routes are recoverable: the receiving project rejects via inbox process mode's reject-from-another-project flow, which returns the item to this project's inbox with the rationale. That reject path is why removing the local source on send is safe.
*** Review-habit health check (surface a slipped daily task-review)
@@ -406,17 +501,17 @@ Behavior:
git status --short
#+end_src
-*Default policy: end every session with an empty =git status=.* The wrap is incomplete while anything remains dirty. There is no "leave it alone" default — every leftover gets an active resolution. The only way for a file to stay dirty across the wrap is the user explicitly saying "defer this one, leave it dirty." Surface each leftover with a concrete recommendation; the user has to actively opt out for the dirt to persist.
+*Hard policy: end every session with an empty =git status=.* The wrap is incomplete while anything remains dirty. There is no deferral exception and no "wrapped with known changes" state: unresolved dirt means the session remains open and wrap-up does not occur.
This inverts the older "intentional carryover" default, which let pre-existing dirty state accumulate across sessions silently. Carryover that lives for days or weeks is almost always one of: a forgotten commit from a prior wrap, a stale change that should be discarded, or genuine in-flight work that needs an explicit stash/branch home. None of those should default to "leave it dirty."
**** Three kinds of leftover
-| Pattern | What it is | Recommended action (apply unless user defers) |
+| Pattern | What it is | Recommended action |
|---+---+---|
| Generated, runtime, or lock files that no human edits — e.g., =.claude/scheduled_tasks.lock=, =.pytest_cache/=, build outputs, IDE state, editor swap files | *Runtime artifact* — created by tooling or the harness, not by the user, and shouldn't be tracked | Add the matching pattern to =.gitignore= (project-level, not =~/.gitignore_global=). For tracked files, =git rm --cached <path>=. Stage =.gitignore= and any =rm --cached= changes in *one* follow-up commit (=chore: gitignore X=), push. Re-run =git status= to confirm clean. |
| Modified or created during the session but not staged into the wrap-up commit | *Forgotten change* — real session work that should have been in the wrap commit but missed it | Stage and create a follow-up commit. Don't =--amend= the wrap-up commit once pushed (diverging history without a clear win). Push the follow-up to all remotes. |
-| Was dirty at session start and still dirty at session end — work this session deliberately didn't touch | *Pre-existing dirt that needs a decision* — could be a missed commit from a prior wrap, stale abandoned work, or real in-flight work without a home | Investigate (show diff + check the originating session). Recommend one of: (a) commit now if the work is complete, (b) stash with a descriptive message if it's genuine WIP, (c) =git checkout -- <path>= / =git clean -f <path>= if stale and unwanted, (d) move to a feature branch if it's longer-running, (e) user explicitly defers and accepts the dirt. Do not silently leave dirty. |
+| Was dirty at session start and still dirty at session end — work this session deliberately didn't touch | *Pre-existing dirt that needs a decision* — could be a missed commit from a prior wrap, stale abandoned work, or real in-flight work without a home | Investigate (show diff + check the originating session). Recommend one of: (a) commit now if the work is complete, (b) stash with a descriptive message if it's genuine WIP, (c) =git checkout -- <path>= / =git clean -f <path>= if stale and unwanted, or (d) move to a feature branch if it's longer-running. Do not silently leave dirty. |
**** Per-file flow
@@ -424,18 +519,40 @@ For each leftover line in =git status --short=:
1. Identify which of the three kinds above it matches.
2. State what the file is (one line) and the recommended action.
-3. Apply the action unless the user explicitly defers.
-4. Re-run =git status --short= after each follow-up commit until empty (or until every remaining line is an explicit user-deferred entry).
+3. Apply the action when it is safe and authorized.
+4. Re-run =git status --short= after each follow-up commit until empty.
The pre-existing-dirt case (third row) is the one this rule most cares about. Treat each pre-existing-dirty file as a question that must get an answer this session, not as "carryover that's fine to inherit." A file that was dirty for a week before this session probably isn't going to get cleaner by waiting another week. Look at the diff, check the originating session's notes, and recommend a real resolution.
-**** When the user defers
+**** When cleanup cannot be completed
+
+Stop the wrap. Do not deliver the valediction, print =session wrapped.=, drop a teardown/shutdown sentinel, or describe the session as complete. Report:
-If the user does say "leave this one dirty for now" after seeing the recommendation, that is fine — log the deferral in the valediction so the next session knows it was an explicit choice, not a miss. Format: "Deferred (per Craig's decision today): =path/to/file= — <one-line reason>". Without that note, the next session can't distinguish "we agreed to defer" from "we forgot again."
+1. Every remaining path and its exact Git state.
+2. What the file is and why the agent cannot safely resolve it alone.
+3. The concrete action or decision Craig needs to provide to make the tree clean.
+
+An explicit decision to keep a file dirty changes the outcome from "wrapping" to "leaving the session interrupted." It never satisfies this workflow.
+
+*** Final clean-tree certificate — hard gate
+
+After all commits are pushed and every leftover appears resolved, run the shared gate:
+
+#+begin_src bash
+gate="$(command -v git-worktree-gate 2>/dev/null || true)"
+[ -n "$gate" ] || gate="$HOME/code/rulesets/claude-templates/bin/git-worktree-gate"
+if [ ! -x "$gate" ]; then
+ echo "wrap blocked: git-worktree-gate is unavailable; install rulesets tooling and retry"
+ exit 1
+fi
+"$gate" certify "$PWD"
+#+end_src
+
+The certificate lives inside the Git directory, so it does not dirty the worktree. It records the exact verified HEAD. A non-zero result is a hard stop governed by "When cleanup cannot be completed" above. Step 5 is unreachable until certification succeeds.
** Step 5: Valediction
-Brief, warm closing. 3-4 sentences max.
+Only after the final clean-tree certificate succeeds, deliver a brief, warm closing. 3-4 sentences max.
Include:
- What was accomplished (specific, not generic)
@@ -444,6 +561,8 @@ Include:
Tone: warm but professional. No emoji unless Craig has explicitly requested. Acknowledge effort when session was long or difficult.
+End on a clear signoff: the *last* line of the valediction is always =session wrapped.= on its own line (lowercase, with the period, nothing after it). It's the unmistakable end-of-session marker, so don't trail it with another sentence. This is the last user-facing output — Step 6's teardown is silent.
+
Example:
#+begin_example
That's a wrap. Today we restructured the entire claude-templates
@@ -456,8 +575,49 @@ from earlier) and archsetup's layout-navigate tests. Both are
ratio-local uncommitted state.
Good session. Talk tomorrow.
+
+session wrapped.
#+end_example
+** Step 6: Session teardown (mode-dependent)
+
+The last action of the wrap, and only after Step 4's commit + push is verified and the Step 5 valediction is composed. The teardown itself happens when this response ends (via the =Stop= hook), so the valediction always renders first. Act by the mode resolved up front:
+
+*** No-teardown mode
+
+Do nothing. The buffer, the =aiv-<project>= tmux session, and =claude= all stay up so the summary stays readable. The wrap is complete.
+
+*** Teardown mode (default)
+
+Confirm commit + push and the final clean-tree certificate succeeded (Exit Criteria 5 — never tear down over unpushed or dirty work), then drop the sentinel:
+
+#+begin_src bash
+touch "/tmp/ai-wrap-teardown-$(basename "$PWD")"
+#+end_src
+
+That is the whole step. Don't run any =tmux kill-session=, =emacsclient=, or buffer kill inline — the =Stop= hook reads the sentinel when this response ends and runs =cj/ai-term-quit=, which kills the =aiv-<project>= session (taking =claude= with it), kills the vterm buffer, and restores geometry. The basename of =$PWD= is the key the hook matches, so the sentinel names the session it tears down.
+
+*The sentinel is session-scoped.* If certification fails, the =Stop= hook blocks and leaves the sentinel armed on purpose, so a wrap blocked by a dirty tree retries on a later stop without re-running this workflow. It does *not* survive the session: =session-start-disarm.sh= clears it at =SessionStart=, because a wrap that never certified is not a pending teardown once its session is gone. Before that hook existed, an uncertified sentinel sat armed indefinitely and fired in whatever session next reached a clean tree — work's 2026-07-27 11:37 wrap killed the 13:20 session mid-work, and archsetup's sat armed on a live terminal for two days. If teardown is still wanted in a new session, run this workflow again.
+
+*** Shutdown mode
+
+Confirm commit + push succeeded, then evaluate the safety gate *before* committing to the shutdown — never power the box off out from under another live session:
+
+#+begin_src bash
+emacsclient -e '(cj/ai-term-live-count)'
+#+end_src
+
+- *Count > 1* — another ai-term session is alive. ABORT the shutdown. List the other live =aiv-*= sessions, drop *no* sentinel, and tell Craig in the valediction that it fell back to a normal wrap (no poweroff, no teardown). This gate is the load-bearing safety of the whole feature.
+- *Count = 1* — this session is the only one. Drop the shutdown sentinel:
+
+ #+begin_src bash
+ touch "/tmp/ai-wrap-shutdown-$(basename "$PWD")"
+ #+end_src
+
+ The =Stop= hook fires =cj/ai-term-shutdown-countdown= when this response ends: it re-checks the gate, runs an abort-able 10→1 countdown in the Emacs echo area (=C-g= cancels), then =sudo shutdown now=. Shutdown supersedes teardown — do *not* also drop the teardown sentinel.
+
+If =emacsclient= isn't resolvable or the daemon is down, the gate can't run — abort the shutdown, fall back to a normal wrap, and say so. Don't power off on an unverifiable gate.
+
* Common Mistakes to Avoid
1. *Skipping Step 1 (Summary)* — the file becomes the record; an empty Summary makes it hard to scan at catch-up
@@ -469,7 +629,8 @@ Good session. Talk tomorrow.
7. *Leaving =.ai/session-context.org= in place* — its presence means "interrupted session", confuses next startup
8. *Long preachy valediction* — brief beats thorough
9. *Leaving runtime/generated files dirty without gitignoring them* — pollutes every future =git status= and erodes trust in "working tree clean" as a signal. Fix =.gitignore= during the wrap, not later.
-10. *Treating "was dirty at session start, still dirty now" as fine by default* — that's how a forgotten commit from two sessions ago turns into "carryover" for two weeks. Every pre-existing dirty file needs an active resolution recommendation this session. Deferral is allowed only with an explicit user choice, logged in the valediction.
+10. *Treating "was dirty at session start, still dirty now" as fine* — that's how a forgotten commit from two sessions ago turns into "carryover" for two weeks. Every pre-existing dirty file must be resolved or the wrap remains blocked.
+11. *Calling a blocked cleanup a wrap* — if the strict gate fails, report the paths and needed decisions; do not valedict, certify completion, or tear down.
* Validation Checklist
@@ -479,19 +640,23 @@ Before considering wrap-up complete:
- [ ] The Summary ends with the =KB: promoted N / consulted yes-no= line (promotion check ran)
- [ ] File renamed to =.ai/sessions/YYYY-MM-DD-HH-MM-description.org=
- [ ] =.ai/session-context.org= no longer exists
-- [ ] =todo-cleanup.el= ran — hygiene pass + =--archive-done= + =--sync-child-priority= (if =todo.org= exists at project root)
+- [ ] =todo-cleanup.el= ran — hygiene pass + =--convert-subtasks= + =--archive-done= + =--sync-child-priority= (if =todo.org= exists at project root)
- [ ] =lint-org.el= ran on =todo.org= — mechanical fixes applied, judgments appended to follow-ups file (if =todo.org= exists)
- [ ] Any orphan-planning-line warnings reviewed (fix or accept)
-- [ ] Inbox carries nothing but expected pipeline artifacts (=.gitkeep=, =lint-followups.org=, =PROCESSED-*= prefixes), OR each remaining handoff has an explicit deferral logged in the valediction
+- [ ] Inbox carries nothing but expected committed or ignored pipeline artifacts (=.gitkeep=, =lint-followups.org=, =PROCESSED-*= prefixes); any untracked inbox delivery was processed before wrap
- [ ] Linear Dev-Review sweep ran; any merged-PR tickets moved to Done or PM Acceptance (skip if project doesn't use Linear)
- [ ] Template-sync churn committed as its own =chore: sync .ai tooling from templates= (consuming projects only; skipped in rulesets), or surfaced if a synced path didn't match canonical
-- [ ] After wrap-up commit + push, =git status --short= is empty OR every remaining line has an explicit user-deferred decision logged in the valediction
+- [ ] After wrap-up commit + push, =git-worktree-gate certify "$PWD"= succeeded at the current HEAD
- [ ] Each leftover was investigated and the user saw a concrete resolution recommendation
- [ ] Runtime artifacts added to =.gitignore=, follow-up commit pushed, =git status= re-verified
- [ ] Forgotten changes committed in a follow-up and pushed
-- [ ] Pre-existing dirty files resolved (committed / stashed / discarded / moved to a feature branch) or explicitly deferred with a one-line reason in the valediction
+- [ ] Pre-existing dirty files resolved (committed / stashed / discarded / moved to a feature branch); otherwise wrap stopped with an actionable blocker report
- [ ] Current branch pushed to ALL remotes (verified with =git remote -v=)
- [ ] All other local branches with a tracking upstream pushed to their remote
- [ ] Any untracked-upstream branches surfaced for manual =git push -u=
+- [ ] Step 6 teardown matches the trigger phrase: no-teardown leaves the buffer; teardown drops only =/tmp/ai-wrap-teardown-<project>=; shutdown gates on =cj/ai-term-live-count= = 1 and drops only =/tmp/ai-wrap-shutdown-<project>=
+- [ ] No teardown/shutdown sentinel was dropped before commit + push was verified
+- [ ] The teardown hook can re-verify the clean-tree certificate before consuming a sentinel
+- [ ] Shutdown aborted (fell back to normal wrap, logged in the valediction) when another =aiv-*= session was live or the gate couldn't run
- [ ] Commit message follows format (no =session:=, no Claude attribution)
- [ ] Valediction delivered (brief, specific, warm)
diff --git a/claude-templates/AGENTS.md b/claude-templates/AGENTS.md
new file mode 100644
index 0000000..97fd001
--- /dev/null
+++ b/claude-templates/AGENTS.md
@@ -0,0 +1,19 @@
+# Agent Entry Point
+
+You are this project's agent, whichever model or harness is running you.
+
+1. If the project has `.ai/protocols.org`, read it now and follow all
+ instructions. It is the single entry point for session behavior:
+ startup, session logging, inbox processing, wrap-up.
+2. Behavioral rules live in `.claude/rules/*.md` in the project (when
+ present) and `~/.claude/rules/*.md` globally. Read them; they bind
+ every session regardless of harness.
+3. A `/name` reference in any rule or workflow (`/voice`,
+ `/review-code`, `/brainstorm`, ...) resolves to a file:
+ `~/.claude/skills/<name>/SKILL.md` or `~/.claude/commands/<name>.md`.
+ Read that file and follow it. If it is absent, say so and apply the
+ rule's documented fallback rather than skipping the gate.
+4. Harness mechanics these files assume (hooks, `/clear`, popup tools)
+ may not exist in your harness. Degrade per each rule's own fallback
+ language. Never silently skip a verification or approval gate because
+ the tooling that enforces it is missing.
diff --git a/claude-templates/bin/agent-page b/claude-templates/bin/agent-page
new file mode 100755
index 0000000..728ee78
--- /dev/null
+++ b/claude-templates/bin/agent-page
@@ -0,0 +1,12 @@
+#!/bin/bash
+# agent-page — deprecated alias for agent-text.
+#
+# The Signal phone tool was renamed agent-text on 2026-07-20, when the
+# notification vocabulary split into "text me" (Signal) and "page me" (desktop).
+# This shim keeps old callers and other machines working until they re-install
+# and pick up agent-text directly. Remove it in a later cleanup once nothing
+# references agent-page.
+#
+# Source: ~/code/rulesets/claude-templates/bin/agent-page
+
+exec "$(dirname "$(readlink -f "$0")")/agent-text" "$@"
diff --git a/claude-templates/bin/agent-text b/claude-templates/bin/agent-text
new file mode 100755
index 0000000..86aa933
--- /dev/null
+++ b/claude-templates/bin/agent-text
@@ -0,0 +1,57 @@
+#!/bin/bash
+# agent-text — text Craig's phone over Signal, from any machine or agent runtime.
+# The Signal half of the notification vocabulary: "text me" reaches the phone,
+# "page me" is the desktop channel (notify). See protocols.org "Reaching Craig".
+#
+# Usage: agent-text <message...>
+#
+# The Signal identity (+15045173983) is registered in velox's signal-cli, and
+# any daily driver linked as a device of that account (ratio, 2026-07-20) can
+# send directly too. So the dispatch is: if the account is registered in the
+# local signal-cli, send directly; otherwise ssh-relay the send to velox over
+# the tailnet. A direct send from a linked device still lands when velox is
+# down (the reason ratio was linked). The recipient is Craig's Signal account
+# UUID; his phone number reads as unregistered in Signal's directory, so never
+# target the number. Verified end to end 2026-07-13 (velox) and 2026-07-20
+# (ratio, direct).
+#
+# This is the AWAY channel. At his desk, use the desktop channel instead:
+# notify info "Title" "Message" --persist
+# See protocols.org "Reaching Craig" for choosing between them.
+#
+# Known caveats (full runbook in rulesets docs/design/): a relay from a
+# non-linked machine needs velox up on the tailnet, and each device holding the
+# account wants a periodic `receive` (staleness warnings appear otherwise); the
+# signal-receive timer handles that.
+#
+# Source: ~/code/rulesets/claude-templates/bin/agent-text
+# Install: make -C ~/code/rulesets install
+
+SIGNAL_ACCOUNT="+15045173983"
+CRAIG_UUID="b1b5601e-6126-47f8-afaa-0a59f5188fde"
+VELOX_HOST="velox.tailf3bb8c.ts.net"
+
+if [ $# -eq 0 ]; then
+ echo "usage: agent-text <message...>" >&2
+ exit 2
+fi
+
+msg="$*"
+
+# The account is local if this machine's signal-cli holds it: the registered
+# primary (velox) or any linked device. Those send directly.
+if signal-cli listAccounts 2>/dev/null | grep -q "$SIGNAL_ACCOUNT"; then
+ signal-cli -a "$SIGNAL_ACCOUNT" send -m "$msg" "$CRAIG_UUID"
+ rc=$?
+else
+ # printf %q hardens the message for the remote shell.
+ ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new \
+ "$VELOX_HOST" \
+ "signal-cli -a $SIGNAL_ACCOUNT send -m $(printf '%q' "$msg") $CRAIG_UUID"
+ rc=$?
+fi
+
+if [ "$rc" -ne 0 ]; then
+ echo "agent-text: phone message failed (velox down or unreachable?); fall back to the desktop channel: notify info 'Message' '<message>' --persist" >&2
+fi
+exit "$rc"
diff --git a/claude-templates/bin/ai b/claude-templates/bin/ai
index 63dc2e9..65d0ab7 100755
--- a/claude-templates/bin/ai
+++ b/claude-templates/bin/ai
@@ -1,16 +1,23 @@
#!/bin/bash
-# ai — Claude Code session launcher (unified aix + hey)
+# ai — agent session launcher (unified aix + hey)
#
# Usage:
-# ai Select one or more projects via fzf and open each in
-# an 'ai' tmux session window (creates session if needed).
+# ai Pick the agent first (claude, codex/ChatGPT, or any
+# local ollama model), then select one or more projects
+# via fzf; each opens in an 'ai' tmux session window.
# Git-aware: fetches, annotates with ↑/↓/dirty, auto-pulls
-# clean-and-behind repos before opening.
+# clean-and-behind repos before opening. --runtime or
+# AI_RUNTIME skips the agent pick.
#
# ai <dir>... Single-project mode. Opens each given directory directly
# in the 'ai' session (new window or switch to existing).
# Use '.' for current directory. Git prep per dir.
#
+# ai --runtime <rt> Launch with a different agent CLI: claude (default),
+# codex, or local (codex --oss against this machine's
+# ollama; model per AI_LOCAL_MODEL, default gpt-oss:120b).
+# Also settable via AI_RUNTIME.
+#
# ai --attach Attach to the existing 'ai' session without changes.
#
# ai -h | --help Show this help.
@@ -25,7 +32,72 @@
# would kill the script.
SESSION="ai"
-CLAUDE_CMD="claude"
+RUNTIME="${AI_RUNTIME:-claude}"
+LOCAL_MODEL="${AI_LOCAL_MODEL:-gpt-oss:120b}"
+
+# Map the runtime name to the agent CLI a pane launches. All three take the
+# opening instructions as a positional prompt. AGENT_BIN is the binary the
+# dependency check probes; AGENT_CMD is the full launch command (the local
+# runtime rides codex's open-source provider against the machine's ollama —
+# model per AI_LOCAL_MODEL, default gpt-oss:120b, verified on ratio's
+# Strix Halo 2026-07-13).
+resolve_agent_cmd() {
+ case "$RUNTIME" in
+ claude)
+ AGENT_BIN="claude"
+ AGENT_CMD="claude"
+ ;;
+ codex)
+ AGENT_BIN="codex"
+ AGENT_CMD="codex"
+ ;;
+ local)
+ AGENT_BIN="codex"
+ AGENT_CMD="codex --oss --local-provider=ollama -m $LOCAL_MODEL"
+ ;;
+ *)
+ echo "ai: unknown runtime '$RUNTIME' — valid runtimes: claude, codex, local" >&2
+ exit 2
+ ;;
+ esac
+}
+
+# One line per launchable agent, claude first (Enter-Enter keeps the old
+# muscle memory). Local models appear only when both codex (the CLI that
+# drives them) and a live ollama answer; a dead server just drops the lines.
+build_runtime_choices() {
+ command -v claude >/dev/null 2>&1 && echo "claude — Claude Code"
+ command -v codex >/dev/null 2>&1 && echo "codex — ChatGPT (Codex CLI)"
+ if command -v codex >/dev/null 2>&1 && command -v ollama >/dev/null 2>&1; then
+ timeout 3 ollama list 2>/dev/null | tail -n +2 | awk 'NF {print "local:" $1 " — ollama"}'
+ fi
+}
+
+# Interactive runtime pick for the bare-`ai` flow. Sets RUNTIME (and
+# LOCAL_MODEL for a local pick) and re-resolves the agent command.
+# Returns 1 when the pick is cancelled.
+pick_runtime() {
+ local choice
+ choice=$(build_runtime_choices | fzf --height=30% --reverse --prompt='agent> ') || return 1
+ [ -z "$choice" ] && return 1
+ case "$choice" in
+ claude*) RUNTIME="claude" ;;
+ codex*) RUNTIME="codex" ;;
+ local:*)
+ RUNTIME="local"
+ LOCAL_MODEL="${choice#local:}"
+ LOCAL_MODEL="${LOCAL_MODEL%% *}"
+ ;;
+ esac
+ resolve_agent_cmd
+}
+
+# Run in the pane's shell just before Claude launches. `stty susp undef` clears
+# the tty's SIGTSTP (C-z) character for this pane only, so an accidental C-z is
+# passed through to Claude as input rather than suspending the session to the
+# shell. Scoped here so C-z keeps working as job control in every other
+# terminal, shell, and program.
+LAUNCH_PREFIX="stty susp undef; "
# Format the per-project opening line passed to claude. Takes the project
# directory's basename; returns a string of the form
@@ -39,16 +111,63 @@ build_instructions() {
}
usage() {
- sed -n '2,20p' "$0" | sed 's|^# \?||'
+ sed -n '2,23p' "$0" | sed 's|^# \?||'
exit 0
}
-for cmd in fzf tmux claude; do
- if ! command -v "$cmd" &>/dev/null; then
- echo "ai: $cmd is not installed" >&2
- exit 1
+check_deps() {
+ for cmd in fzf tmux "$AGENT_BIN"; do
+ if ! command -v "$cmd" &>/dev/null; then
+ echo "ai: $cmd is not installed" >&2
+ exit 1
+ fi
+ done
+}
+
+# ---------- pure decision cores (no tmux/git I/O; unit-tested directly) ----------
+
+# Decide what a git-prep pass should do from a repo's already-computed state.
+# Inputs: has_upstream (1/0), dirty (1/0), ahead, behind. Echoes one of:
+# none — no upstream, or in sync: nothing to do
+# pull — clean and purely behind: safe to fast-forward
+# report — ahead, dirty, or behind-while-dirty: show a summary, don't pull
+_git_prep_action() {
+ local has_upstream="$1" dirty="$2" ahead="$3" behind="$4"
+ [ "$has_upstream" -eq 1 ] || {
+ echo none
+ return
+ }
+ if [ "$dirty" -eq 0 ] && [ "$ahead" -eq 0 ] && [ "$behind" -gt 0 ]; then
+ echo pull
+ elif [ "$ahead" -gt 0 ] || [ "$behind" -gt 0 ] || [ "$dirty" -eq 1 ]; then
+ echo report
+ else
+ echo none
fi
-done
+}
+
+# Re-order "name<TAB>wid" lines (stdin) into the launcher's window order:
+# non-project windows alphabetically, then project windows alphabetically.
+# $1 is a newline-separated list of project window names.
+_order_windows() {
+ local project_names="$1" wname wid others="" projects=""
+ while IFS=$'\t' read -r wname wid; do
+ [ -z "$wname" ] && continue
+ if printf '%s\n' "$project_names" | grep -qxF "$wname"; then
+ projects+="${wname}"$'\t'"${wid}"$'\n'
+ else
+ others+="${wname}"$'\t'"${wid}"$'\n'
+ fi
+ done
+ others=$(printf '%s' "$others" | sort -t$'\t' -k1,1f)
+ projects=$(printf '%s' "$projects" | sort -t$'\t' -k1,1f)
+ printf '%s\n%s\n' "$others" "$projects" | sed '/^$/d'
+}
+
+# Emit the window id whose name (field 1 of "name<TAB>wid" stdin) equals $1.
+_match_window_id() {
+ awk -F'\t' -v n="$1" '$1 == n { print $2; exit }'
+}
# ---------- shared helpers ----------
@@ -66,13 +185,16 @@ create_window() {
wid=$(tmux new-window -a -t "$SESSION:{end}" -n "$name" -c "$dir" -P -F '#{window_id}')
sleep 0.1
instructions=$(build_instructions "$name")
- tmux send-keys -t "$wid" "$CLAUDE_CMD \"$instructions\"" Enter
+ tmux send-keys -t "$wid" "${LAUNCH_PREFIX}$AGENT_CMD \"$instructions\"" Enter
echo "$wid"
}
# Add a directory to candidates only if it's a Claude-template project.
maybe_add_candidate() {
local dir="$1"
+ # The "~/" is a deliberate literal display prefix, re-expanded downstream via
+ # ${c/#\~/$HOME}; it must not expand here, so SC2088 doesn't apply.
+ # shellcheck disable=SC2088
[ -f "$dir/.ai/protocols.org" ] && candidates+=("~/${dir#"$HOME"/}")
}
@@ -80,6 +202,7 @@ maybe_add_candidate() {
build_candidates() {
candidates=()
maybe_add_candidate "$HOME/.emacs.d"
+ maybe_add_candidate "$HOME/.dotfiles"
if [ -d "$HOME/code" ]; then
while IFS= read -r d; do
maybe_add_candidate "$d"
@@ -108,12 +231,36 @@ fetch_candidates() {
wait
}
+# Resolve the shared state gate installed beside this launcher. Keeping the
+# policy in one executable prevents startup, the picker, and wrap-up from
+# developing different meanings of "safe to sync."
+_git_gate_path() {
+ local gate="${GIT_WORKTREE_GATE:-}"
+ [ -n "$gate" ] || gate="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/git-worktree-gate"
+ [ -x "$gate" ] && printf '%s\n' "$gate"
+}
+
+# True (exit 0) when strict wrap would reject the worktree.
+_git_is_dirty() {
+ local dir="$1" gate
+ gate="$(_git_gate_path)" || return 0
+ ! "$gate" strict "$dir" >/dev/null 2>&1
+}
+
+# True (exit 0) when startup sync must stop. Untracked inbox deliveries are
+# safe queue input; every tracked, staged, or other untracked change blocks.
+_git_blocks_sync() {
+ local dir="$1" gate
+ gate="$(_git_gate_path)" || return 0
+ ! "$gate" sync-safe "$dir" >/dev/null 2>&1
+}
+
# Return " (↑N ↓N dirty)" or " (✓)" if clean.
git_status_indicator() {
local dir="$1" upstream ahead=0 behind=0 parts=()
[ -d "$dir/.git" ] || return 0
- upstream=$(git -C "$dir" rev-parse --abbrev-ref --symbolic-full-name @{u} 2>/dev/null || true)
+ upstream=$(git -C "$dir" rev-parse --abbrev-ref --symbolic-full-name "@{u}" 2>/dev/null || true)
if [ -n "$upstream" ]; then
ahead=$(git -C "$dir" rev-list --count "$upstream..HEAD" 2>/dev/null || echo 0)
behind=$(git -C "$dir" rev-list --count "HEAD..$upstream" 2>/dev/null || echo 0)
@@ -123,10 +270,12 @@ git_status_indicator() {
parts+=("no upstream")
fi
- if ! git -C "$dir" diff --quiet 2>/dev/null \
- || ! git -C "$dir" diff --cached --quiet 2>/dev/null \
- || [ -n "$(git -C "$dir" ls-files --others --exclude-standard 2>/dev/null)" ]; then
- parts+=("dirty")
+ if _git_is_dirty "$dir"; then
+ if _git_blocks_sync "$dir"; then
+ parts+=("dirty")
+ else
+ parts+=("inbox")
+ fi
fi
if [ ${#parts[@]} -gt 0 ]; then
@@ -148,27 +297,21 @@ annotate_candidates() {
candidates=("${annotated[@]}")
}
-# Pull if clean, behind, not ahead. No-op otherwise.
+# Pull if sync-safe, behind, not ahead. Inbox-only queue input is sync-safe.
auto_pull_if_clean() {
local dir="$1" upstream ahead behind
[ -d "$dir/.git" ] || return 0
+ _git_blocks_sync "$dir" && return 0
- if ! git -C "$dir" diff --quiet 2>/dev/null \
- || ! git -C "$dir" diff --cached --quiet 2>/dev/null \
- || [ -n "$(git -C "$dir" ls-files --others --exclude-standard 2>/dev/null)" ]; then
- return 0
- fi
-
- upstream=$(git -C "$dir" rev-parse --abbrev-ref --symbolic-full-name @{u} 2>/dev/null || true)
+ upstream=$(git -C "$dir" rev-parse --abbrev-ref --symbolic-full-name "@{u}" 2>/dev/null || true)
[ -z "$upstream" ] && return 0
ahead=$(git -C "$dir" rev-list --count "$upstream..HEAD" 2>/dev/null || echo 0)
- [ "${ahead:-0}" -gt 0 ] 2>/dev/null && return 0
-
behind=$(git -C "$dir" rev-list --count "HEAD..$upstream" 2>/dev/null || echo 0)
- [ "${behind:-0}" -eq 0 ] 2>/dev/null && return 0
- git -C "$dir" pull --ff-only --quiet 2>/dev/null || true
+ # dirty=0 and has_upstream=1 are guaranteed by the early returns above.
+ [ "$(_git_prep_action 1 0 "${ahead:-0}" "${behind:-0}")" = pull ] &&
+ git -C "$dir" pull --ff-only --quiet 2>/dev/null || true
}
# Strip " (annotation)" suffix from fzf output so downstream gets raw paths.
@@ -181,7 +324,7 @@ read_selections() {
# Re-order windows: non-project windows at base-index, projects alphabetically after.
sort_windows() {
- local windows others="" projects="" base_idx project_names=""
+ local windows base_idx project_names="" ordered
base_idx=$(tmux show-option -gv base-index 2>/dev/null || echo 0)
windows=$(tmux list-windows -t "$SESSION" -F '#{window_name}'$'\t''#{window_id}')
@@ -190,83 +333,65 @@ sort_windows() {
project_names+="$(basename "${c/#\~/$HOME}")"$'\n'
done
- while IFS=$'\t' read -r wname wid; do
- [ -z "$wname" ] && continue
- if echo "$project_names" | grep -qxF "$wname"; then
- projects+="${wname}"$'\t'"${wid}"$'\n'
- else
- others+="${wname}"$'\t'"${wid}"$'\n'
- fi
- done <<<"$windows"
- others=$(echo -n "$others" | sort -t$'\t' -k1,1f)
- projects=$(echo -n "$projects" | sort -t$'\t' -k1,1f)
-
- local all
- all=$(printf '%s\n' "$others" "$projects" | sed '/^$/d')
+ ordered=$(printf '%s\n' "$windows" | _order_windows "$project_names")
+ [ -z "$ordered" ] && return 0
+ # First pass parks every window above the live range so the second pass can
+ # reassign the target indices without colliding with a window already there.
local i=900
while IFS=$'\t' read -r _n wid; do
+ [ -z "$wid" ] && continue
tmux move-window -s "$wid" -t "$SESSION:$i"
i=$((i + 1))
- done <<<"$all"
+ done <<<"$ordered"
i=$base_idx
- if [ -n "$others" ]; then
- while IFS=$'\t' read -r _n wid; do
- tmux move-window -s "$wid" -t "$SESSION:$i"
- i=$((i + 1))
- done <<<"$others"
- fi
- if [ -n "$projects" ]; then
- while IFS=$'\t' read -r _n wid; do
- tmux move-window -s "$wid" -t "$SESSION:$i"
- i=$((i + 1))
- done <<<"$projects"
- fi
+ while IFS=$'\t' read -r _n wid; do
+ [ -z "$wid" ] && continue
+ tmux move-window -s "$wid" -t "$SESSION:$i"
+ i=$((i + 1))
+ done <<<"$ordered"
}
# Find existing window id in ai session by window name; empty if none.
find_window_id() {
- local name="$1"
- tmux list-windows -t "$SESSION" -F '#{window_name}'$'\t''#{window_id}' 2>/dev/null \
- | awk -F'\t' -v n="$name" '$1 == n {print $2; exit}'
+ tmux list-windows -t "$SESSION" -F '#{window_name}'$'\t''#{window_id}' 2>/dev/null |
+ _match_window_id "$1"
}
# Git prep for a single directory. Uses FETCH_HEAD cache to skip back-to-back
# fetches. Pulls automatically if clean-and-behind; prints one-line summary
# if diverged/dirty/ahead.
prep_git_single() {
- local dir="$1" gitdir upstream ahead=0 behind=0 dirty="" age fetch_stale=1 parts=()
+ local dir="$1" gitdir upstream ahead=0 behind=0 dirty=0 age fetch_stale=1 parts=()
git -C "$dir" rev-parse --is-inside-work-tree >/dev/null 2>&1 || return 0
gitdir=$(git -C "$dir" rev-parse --git-dir 2>/dev/null)
if [ -f "$gitdir/FETCH_HEAD" ]; then
- age=$(( $(date +%s) - $(stat -c %Y "$gitdir/FETCH_HEAD" 2>/dev/null || echo 0) ))
+ age=$(($(date +%s) - $(stat -c %Y "$gitdir/FETCH_HEAD" 2>/dev/null || echo 0)))
[ "$age" -lt 600 ] && fetch_stale=0
fi
[ "$fetch_stale" -eq 1 ] && git -C "$dir" fetch --quiet 2>/dev/null || true
- upstream=$(git -C "$dir" rev-parse --abbrev-ref --symbolic-full-name @{u} 2>/dev/null || true)
+ upstream=$(git -C "$dir" rev-parse --abbrev-ref --symbolic-full-name "@{u}" 2>/dev/null || true)
[ -z "$upstream" ] && return 0
ahead=$(git -C "$dir" rev-list --count "$upstream..HEAD" 2>/dev/null || echo 0)
behind=$(git -C "$dir" rev-list --count "HEAD..$upstream" 2>/dev/null || echo 0)
-
- if ! git -C "$dir" diff --quiet 2>/dev/null \
- || ! git -C "$dir" diff --cached --quiet 2>/dev/null \
- || [ -n "$(git -C "$dir" ls-files --others --exclude-standard 2>/dev/null)" ]; then
- dirty="dirty"
- fi
-
- if [ -z "$dirty" ] && [ "${ahead:-0}" -eq 0 ] && [ "${behind:-0}" -gt 0 ]; then
- echo "ai: pulling $behind commit(s) from $upstream..." >&2
- git -C "$dir" pull --ff-only --quiet
- elif [ "${ahead:-0}" -gt 0 ] || [ "${behind:-0}" -gt 0 ] || [ -n "$dirty" ]; then
- [ "${ahead:-0}" -gt 0 ] && parts+=("↑$ahead")
- [ "${behind:-0}" -gt 0 ] && parts+=("↓$behind")
- [ -n "$dirty" ] && parts+=("$dirty")
- echo "ai: $(basename "$dir") — ${parts[*]}" >&2
- fi
+ _git_blocks_sync "$dir" && dirty=1
+
+ case "$(_git_prep_action 1 "$dirty" "${ahead:-0}" "${behind:-0}")" in
+ pull)
+ echo "ai: pulling $behind commit(s) from $upstream..." >&2
+ git -C "$dir" pull --ff-only --quiet
+ ;;
+ report)
+ [ "${ahead:-0}" -gt 0 ] && parts+=("↑$ahead")
+ [ "${behind:-0}" -gt 0 ] && parts+=("↓$behind")
+ [ "$dirty" -eq 1 ] && parts+=("dirty")
+ echo "ai: $(basename "$dir") — ${parts[*]}" >&2
+ ;;
+ esac
}
# ---------- modes ----------
@@ -283,7 +408,10 @@ attach_mode() {
# Open a single project (or focus existing window).
single_mode() {
local arg="$1" dir name wid existing
- dir="$(cd "$arg" 2>/dev/null && pwd)" || { echo "ai: cannot access '$arg'" >&2; return 1; }
+ dir="$(cd "$arg" 2>/dev/null && pwd)" || {
+ echo "ai: cannot access '$arg'" >&2
+ return 1
+ }
if [ ! -f "$dir/.ai/protocols.org" ]; then
echo "ai: $dir has no .ai/protocols.org — not a Claude-template project" >&2
@@ -311,7 +439,7 @@ single_mode() {
local instructions
wid=$(tmux new-session -d -s "$SESSION" -n "$name" -c "$dir" -P -F '#{window_id}')
instructions=$(build_instructions "$name")
- tmux send-keys -t "$wid" "$CLAUDE_CMD \"$instructions\"" Enter
+ tmux send-keys -t "$wid" "${LAUNCH_PREFIX}$AGENT_CMD \"$instructions\"" Enter
fi
sort_windows
@@ -371,12 +499,12 @@ multi_mode() {
local instructions
first_wid=$(tmux new-session -d -s "$SESSION" -n "$name" -c "$dir" -P -F '#{window_id}')
instructions=$(build_instructions "$name")
- tmux send-keys -t "$first_wid" "$CLAUDE_CMD \"$instructions\"" Enter
+ tmux send-keys -t "$first_wid" "${LAUNCH_PREFIX}$AGENT_CMD \"$instructions\"" Enter
for entry in "${selected[@]:1}"; do
dir="${entry/#\~/$HOME}"
name="$(basename "$dir")"
auto_pull_if_clean "$dir"
- create_window "$dir" "$name" > /dev/null
+ create_window "$dir" "$name" >/dev/null
done
else
# Add windows to existing session
@@ -395,21 +523,99 @@ multi_mode() {
attach_session
}
+# Print the launch command a real run would send to the pane, then exit.
+# Exists for the launcher's bats tests: exercises runtime resolution and the
+# opening line with no tmux or fzf involved.
+print_launch_mode() {
+ local arg="$1" dir name
+ dir="$(cd "$arg" 2>/dev/null && pwd)" || {
+ echo "ai: cannot access '$arg'" >&2
+ exit 1
+ }
+ if [ ! -f "$dir/.ai/protocols.org" ]; then
+ echo "ai: $dir has no .ai/protocols.org — not an agent-template project" >&2
+ exit 1
+ fi
+ name="$(basename "$dir")"
+ printf '%s "%s"\n' "$AGENT_CMD" "$(build_instructions "$name")"
+ exit 0
+}
+
# ---------- dispatch ----------
-case "${1:-}" in
- -h|--help)
- usage
- ;;
- --attach)
- attach_mode
- ;;
- "")
- multi_mode
- ;;
- *)
- for arg in "$@"; do
- single_mode "$arg"
- done
- ;;
-esac
+# Argument parsing + mode dispatch. Wrapped so the file can be sourced (by the
+# launcher's bats tests) to exercise individual functions without running a
+# real launch. When executed as a program, BASH_SOURCE[0] equals $0 and the
+# dispatch runs exactly as before; when sourced, it's skipped.
+main() {
+ print_launch=""
+ runtime_explicit="${AI_RUNTIME:+1}"
+ while [ $# -gt 0 ]; do
+ case "$1" in
+ -h | --help)
+ usage
+ ;;
+ --runtime)
+ [ -z "${2:-}" ] && {
+ echo "ai: --runtime needs a value — valid runtimes: claude, codex, local" >&2
+ exit 2
+ }
+ RUNTIME="$2"
+ runtime_explicit=1
+ shift 2
+ ;;
+ --runtime=*)
+ RUNTIME="${1#--runtime=}"
+ runtime_explicit=1
+ shift
+ ;;
+ --print-launch)
+ print_launch=1
+ shift
+ ;;
+ --print-runtimes)
+ build_runtime_choices
+ exit 0
+ ;;
+ *)
+ break
+ ;;
+ esac
+ done
+
+ resolve_agent_cmd
+
+ if [ -n "$print_launch" ]; then
+ [ $# -eq 0 ] && {
+ echo "ai: --print-launch needs a project directory" >&2
+ exit 2
+ }
+ print_launch_mode "$1"
+ fi
+
+ case "${1:-}" in
+ --attach)
+ check_deps
+ attach_mode
+ ;;
+ "")
+ # Bare `ai`: pick the agent first (skipped when --runtime or AI_RUNTIME
+ # already chose), then the familiar project multi-select.
+ if [ -z "$runtime_explicit" ]; then
+ pick_runtime || exit 0
+ fi
+ check_deps
+ multi_mode
+ ;;
+ *)
+ check_deps
+ for arg in "$@"; do
+ single_mode "$arg"
+ done
+ ;;
+ esac
+}
+
+if [ "${BASH_SOURCE[0]}" = "${0}" ]; then
+ main "$@"
+fi
diff --git a/claude-templates/bin/git-worktree-gate b/claude-templates/bin/git-worktree-gate
new file mode 100755
index 0000000..e453fd1
--- /dev/null
+++ b/claude-templates/bin/git-worktree-gate
@@ -0,0 +1,185 @@
+#!/usr/bin/env bash
+# git-worktree-gate — one definition of safe Git state for startup and wrap.
+#
+# Modes:
+# strict [DIR] Require an entirely empty worktree.
+# sync-safe [DIR] Permit untracked inbox/ deliveries, but nothing else.
+# certify [DIR] Strict-check, then record the verified HEAD in the git dir.
+# verify [DIR] Strict-check and require the recorded HEAD to still match.
+#
+# Ignored files are deliberately outside Git's clean-worktree contract.
+
+set -u
+
+mode="${1:-}"
+repo="${2:-.}"
+
+usage() {
+ echo "usage: git-worktree-gate {strict|sync-safe|certify|verify} [DIR]" >&2
+ exit 2
+}
+
+case "$mode" in
+ strict|sync-safe|certify|verify) ;;
+ *) usage ;;
+esac
+
+root="$(git -C "$repo" rev-parse --show-toplevel 2>/dev/null)" || {
+ echo "git-worktree-gate: $repo is not inside a Git worktree" >&2
+ exit 2
+}
+gitdir="$(git -C "$root" rev-parse --absolute-git-dir 2>/dev/null)" || {
+ echo "git-worktree-gate: cannot resolve the Git directory for $root" >&2
+ exit 2
+}
+certificate="$gitdir/ai-wrap-clean"
+
+quote_path() {
+ printf '%q' "$1"
+}
+
+operation_in_progress() {
+ local marker
+ for marker in MERGE_HEAD CHERRY_PICK_HEAD REVERT_HEAD BISECT_LOG; do
+ [ -e "$gitdir/$marker" ] && {
+ printf '%s' "$marker"
+ return 0
+ }
+ done
+ for marker in rebase-merge rebase-apply sequencer; do
+ [ -d "$gitdir/$marker" ] && {
+ printf '%s' "$marker"
+ return 0
+ }
+ done
+ return 1
+}
+
+describe_entry() {
+ local xy="$1" path="$2" original="${3:-}"
+ local index="${xy:0:1}" worktree="${xy:1:1}" label=""
+
+ if [ "$xy" = "??" ]; then
+ label="untracked; add and commit it, move it outside the repository, or remove it if unwanted"
+ elif [ "$xy" = "!!" ]; then
+ label="ignored"
+ elif [[ "$xy" = *U* || "$xy" = "AA" || "$xy" = "DD" ]]; then
+ label="unmerged; resolve the conflict and commit the result"
+ elif [ "$index" != " " ] && [ "$worktree" != " " ]; then
+ label="staged and unstaged changes; review both layers, then commit or restore them"
+ elif [ "$index" != " " ]; then
+ label="staged change; commit it or unstage and restore it"
+ else
+ label="unstaged tracked change; commit it or restore it"
+ fi
+
+ printf ' %s ' "$xy"
+ quote_path "$path"
+ if [ -n "$original" ]; then
+ printf ' (from '
+ quote_path "$original"
+ printf ')'
+ fi
+ printf ' — %s\n' "$label"
+}
+
+check_state() {
+ local policy="$1" xy path original="" blocked=0 op=""
+ local status_tmp="" status_err="" status_detail=""
+ local -a report=()
+
+ if op="$(operation_in_progress)"; then
+ report+=(" Git operation in progress: $op — finish or abort it")
+ blocked=1
+ fi
+
+ status_tmp="$(mktemp "$gitdir/ai-worktree-status.tmp.XXXXXX")" || {
+ echo "wrap blocked: cannot allocate a Git-state check file" >&2
+ return 1
+ }
+ status_err="$(mktemp "$gitdir/ai-worktree-status.err.XXXXXX")" || {
+ rm -f "$status_tmp"
+ echo "wrap blocked: cannot allocate a Git-state error file" >&2
+ return 1
+ }
+
+ if ! git -C "$root" status --porcelain=v1 -z \
+ --untracked-files=all --ignore-submodules=none \
+ >"$status_tmp" 2>"$status_err"; then
+ status_detail="$(head -1 "$status_err")"
+ [ -n "$status_detail" ] || status_detail="unknown Git error"
+ report+=(" git status failed — $status_detail")
+ blocked=1
+ else
+ while IFS= read -r -d '' entry; do
+ xy="${entry:0:2}"
+ path="${entry:3}"
+ original=""
+ if [[ "${xy:0:1}" = "R" || "${xy:0:1}" = "C" ]]; then
+ IFS= read -r -d '' original || true
+ fi
+
+ if [ "$policy" = "sync-safe" ] \
+ && [ "$xy" = "??" ] \
+ && [[ "$path" = inbox/* ]]; then
+ continue
+ fi
+
+ report+=("$(describe_entry "$xy" "$path" "$original")")
+ blocked=1
+ done <"$status_tmp"
+ fi
+ rm -f "$status_tmp" "$status_err"
+
+ if [ "$blocked" -ne 0 ]; then
+ if [ "$policy" = "sync-safe" ]; then
+ echo "sync blocked: rulesets has changes other than untracked inbox deliveries" >&2
+ else
+ echo "wrap blocked: Git worktree is not completely clean" >&2
+ fi
+ printf '%s\n' "${report[@]}" >&2
+ return 1
+ fi
+ return 0
+}
+
+case "$mode" in
+ strict)
+ check_state strict
+ ;;
+ sync-safe)
+ check_state sync-safe
+ ;;
+ certify)
+ check_state strict || exit 1
+ head="$(git -C "$root" rev-parse HEAD 2>/dev/null)" || {
+ echo "wrap blocked: cannot resolve HEAD" >&2
+ exit 1
+ }
+ tmp="$(mktemp "$gitdir/ai-wrap-clean.tmp.XXXXXX")" || exit 1
+ chmod 600 "$tmp"
+ {
+ printf 'head=%s\n' "$head"
+ printf 'root=%s\n' "$root"
+ } >"$tmp"
+ mv "$tmp" "$certificate"
+ ;;
+ verify)
+ check_state strict || exit 1
+ [ -f "$certificate" ] || {
+ echo "wrap blocked: no clean-tree certificate exists; rerun the final wrap verification" >&2
+ exit 1
+ }
+ certified_head="$(sed -n 's/^head=//p' "$certificate" | head -1)"
+ certified_root="$(sed -n 's/^root=//p' "$certificate" | head -1)"
+ current_head="$(git -C "$root" rev-parse HEAD 2>/dev/null)" || exit 1
+ [ "$certified_root" = "$root" ] || {
+ echo "wrap blocked: clean-tree certificate belongs to a different worktree" >&2
+ exit 1
+ }
+ [ -n "$certified_head" ] && [ "$certified_head" = "$current_head" ] || {
+ echo "wrap blocked: HEAD changed after clean-tree certification; rerun the final wrap verification" >&2
+ exit 1
+ }
+ ;;
+esac
diff --git a/claude-templates/bin/install-ai b/claude-templates/bin/install-ai
new file mode 100755
index 0000000..3283c4a
--- /dev/null
+++ b/claude-templates/bin/install-ai
@@ -0,0 +1,23 @@
+#!/usr/bin/env bash
+# install-ai — PATH-facing launcher for the fresh-project bootstrapper.
+#
+# make install symlinks this into ~/.local/bin/install-ai (same bin loop that
+# links `ai` and `agent-text`), so `install-ai [--track|--gitignore] [PROJECT]`
+# runs from anywhere. The real logic lives in scripts/install-ai.sh; this
+# resolves its own location through the ~/.local/bin symlink and execs that
+# script by its true repo path, so the script's own repo-root computation
+# (dirname "$0"/..) stays correct. dotfiles needs no copy — the symlink always
+# points at the canonical.
+set -euo pipefail
+
+# Resolve this file through any symlink chain to its real location in the repo.
+source="${BASH_SOURCE[0]}"
+while [ -L "$source" ]; do
+ dir="$(cd -P "$(dirname "$source")" && pwd)"
+ source="$(readlink "$source")"
+ [[ "$source" != /* ]] && source="$dir/$source"
+done
+bindir="$(cd -P "$(dirname "$source")" && pwd)" # <repo>/claude-templates/bin
+repo="$(cd -P "$bindir/../.." && pwd)" # <repo>
+
+exec "$repo/scripts/install-ai.sh" "$@"
diff --git a/docs/design/2026-05-28-generic-agent-runtime-spec.org b/docs/design/2026-05-28-generic-agent-runtime-spec.org
index 01be6d4..7d7a549 100644
--- a/docs/design/2026-05-28-generic-agent-runtime-spec.org
+++ b/docs/design/2026-05-28-generic-agent-runtime-spec.org
@@ -3,6 +3,10 @@
#+DATE: 2026-05-28
#+STARTUP: showall
+* Status note (2026-06-16)
+
+The cross-agent-comms subsystem this spec references as an existing substrate (=cross-agent-send= / =-recv= / =-watch= / =-status= / =-discover= / =-halt= / =-resume=, the =inbox/from-agents/= file-IPC protocol) was *removed* on 2026-06-16 as unused — every real cross-project handoff goes through =inbox-send= instead. Sections below that propose extending the cross-agent protocol (e.g. "Cross-agent updates", the =machine.project.agent-id= targeting) are historical: if this arc is revived, that layer would be rebuilt on =inbox-send=, not the deleted scripts.
+
* Introductory note
Craig asked for a design pass on making =rulesets= generic rather than
diff --git a/docs/design/2026-05-28-pattern-catalog-prompt-labels-and-defaults.org b/docs/design/2026-05-28-pattern-catalog-prompt-labels-and-defaults.org
index cc2cb77..9ee29d6 100644
--- a/docs/design/2026-05-28-pattern-catalog-prompt-labels-and-defaults.org
+++ b/docs/design/2026-05-28-pattern-catalog-prompt-labels-and-defaults.org
@@ -28,7 +28,7 @@ A generic =pearl--with-sentinel SENTINEL CANDIDATES= helper lets each call site
** Why for the catalog
-This is the "no hidden affordances" pattern from the earlier note ([[file:2026-05-28-0003-from-pearl-rulesets-followup-no-empty-input.org]]) sharpened with a second rule: *if the affordance is visible, its label has to match what picking it does*. Visibility without accuracy is its own problem. A label that says "none" when the behavior is "any" is no better than an invisible empty-input idiom — both leave the user holding the wrong model.
+This is the "no hidden affordances" pattern from the earlier note (a processed pearl-rulesets inbox item, since removed) sharpened with a second rule: *if the affordance is visible, its label has to match what picking it does*. Visibility without accuracy is its own problem. A label that says "none" when the behavior is "any" is no better than an invisible empty-input idiom — both leave the user holding the wrong model.
Catalog shape suggestion: this is the same principle as Pattern 3, in a follow-up form. Either a single entry that captures both halves (visible + accurate) or two cross-linked entries.
diff --git a/docs/design/2026-06-02-flush-promotion.org b/docs/design/2026-06-02-flush-promotion.org
index 9d9d8a3..ff4e579 100644
--- a/docs/design/2026-06-02-flush-promotion.org
+++ b/docs/design/2026-06-02-flush-promotion.org
@@ -1,5 +1,5 @@
#+TITLE: Flush Promotion — Handoff Bundle (from work)
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-06-02
* Provenance
diff --git a/docs/design/2026-06-02-pattern-catalog-spec.org b/docs/design/2026-06-02-pattern-catalog-spec.org
index e74b8ae..6bb105a 100644
--- a/docs/design/2026-06-02-pattern-catalog-spec.org
+++ b/docs/design/2026-06-02-pattern-catalog-spec.org
@@ -1,5 +1,5 @@
#+TITLE: Cross-Project Pattern Catalog — Spec
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-06-02
One-page spec for capturing reusable design patterns so they travel from one project to the next instead of being re-derived. Drafted for Craig's spec-review; the five open questions from the [[file:../../todo.org][todo.org]] task carry a recommended call each, marked DECISION.
diff --git a/docs/design/2026-06-15-auto-triage-intake-spec.org b/docs/design/2026-06-15-auto-triage-intake-spec.org
new file mode 100644
index 0000000..b41c99c
--- /dev/null
+++ b/docs/design/2026-06-15-auto-triage-intake-spec.org
@@ -0,0 +1,47 @@
+#+TITLE: Proposed engine addition — Auto mode (auto triage-intake)
+#+DATE: 2026-06-15
+
+* What this adds
+
+A new *mode* of the triage-intake engine: *auto mode* (auto triage-intake). It's a self-running monitor for when Craig is away from the desk but wants tight awareness — a loop that runs the standard triage on a short interval, accumulates rather than mutating, and hands Craig a controlled checkpoint to commit the batch.
+
+Origin: 2026-06-15, the morning Craig had to clear his day for a family emergency and wanted the desk watched while he was in and out. He asked for 20-minute sweeps that summarize what's come in for him, with an explicit command to process and commit the batch.
+
+* The mode
+
+** Cadence
+A loop (CronCreate / =/loop=) fires the triage on an interval — default *20 minutes*. Craig sets the interval.
+
+** Accumulate, don't mutate (the core difference from a normal run)
+A normal triage run writes the sentinel, creates =:quick:reactive:= todos, takes mail actions on confirmation, and is a one-shot. An auto-mode *sweep* does none of the mutations:
+
+- Does NOT advance the sentinel — the scan window grows from the last *close* until the next close, so nothing is dropped between sweeps.
+- Does NOT create todo.org tasks — accumulates them (in the session log) for the close.
+- Does NOT take mail actions (trash / mark-read / star).
+- Does NOT commit.
+- DOES run the full plugin scan, DOES update an active daily-prep (Update mode) and re-open it on change, DOES report.
+
+** End-of-sweep output
+Each sweep ends with two short sections:
+1. *Deltas* — what changed since the last sweep (one line if nothing).
+2. *Responses awaiting your acknowledgment* — every Slack reply, email, or message directed at the user that he hasn't explicitly acknowledged or had the agent answer. This is a *running list carried forward* across sweeps until the user acks each item or closes the triage. It exists because an away user's main need is "who's waiting to hear back from me," which a delta-only sweep loses the moment it scrolls past.
+
+** Close / stop commands (the checkpoint)
+The mutations are gated behind two user commands:
+
+- *"close the triage"* — finish the mail + sentinel processing, add all accumulated todos (asking the user questions along the way), commit and push, then *keep looping* (next sweep on the normal interval). This is the "flush the batch and carry on" checkpoint.
+- *"stop the triage"* — the same close processing, then *stop the loop* and revert to manual triage (run only when asked).
+
+* Why it's worth adding to the engine
+
+The standard engine is one-shot and mutating, which is right for an at-the-desk "what's new?" glance. It's wrong for unattended monitoring: running it every 20 minutes would spray reactive todos, advance the sentinel past unprocessed items, and commit noise without review. Auto mode separates the cheap, frequent *watching* from the deliberate, gated *committing* — and adds the away-user's missing primitive, the running unacknowledged-responses list.
+
+* Companion notes
+
+- The interval loop is the delivery mechanism (CronCreate session-only, or a durable schedule); the mode is the behavior. They compose.
+- "Responses awaiting acknowledgment" tracking needs a small piece of state. In a session it can live in the session-context log; if the engine wants it durable across sessions, a tiny =.ai/triage-unacked.org= (or similar) is the natural home — flagged as a design choice for the rulesets side to decide.
+- Deltas-only reporting (the 2026-06-11 ruling) and loud scan-failure surfacing both still apply inside each sweep.
+
+* Addendum (work, 2026-06-15 17:16)
+
+Add a third end-of-sweep output line: the current date/time/timezone, on its own final line, via =date "+%A %Y-%m-%d %H:%M:%S %Z (%z)"=. Reason: on an away day with frequent unattended sweeps, the per-sweep stamp shows how fresh each summary is at a glance. Sweep output sections become: (a) Deltas, (b) Responses awaiting acknowledgment, (c) the timestamp. Implemented 2026-06-15; the stamp prints on quiet sweeps too, as proof the loop ran.
diff --git a/docs/design/2026-06-15-fix-speedrun-workflow-proposal.org b/docs/design/2026-06-15-fix-speedrun-workflow-proposal.org
new file mode 100644
index 0000000..c1c7077
--- /dev/null
+++ b/docs/design/2026-06-15-fix-speedrun-workflow-proposal.org
@@ -0,0 +1,21 @@
+#+TITLE: Proposed reusable workflow: "fix speedrun" mode, available t
+#+SOURCE: from .emacs.d
+#+DATE: 2026-06-15 19:22:56 -0500
+
+Proposed reusable workflow: "fix speedrun" mode, available to all coding projects.
+
+Origin: a 2026-06-15 .emacs.d theme-studio session. Craig batched a list of quick wins / small fixes and asked me to run them autonomously.
+
+The shape that worked:
+- Entry: Craig names an ordered set of tasks (or points at a tagged subset) and says "fix speedrun" / "no approvals until done".
+- Execution: work the set in order, no per-step approval gates. Each task still runs the full quality bar (TDD red->green, /review-code, /voice on the commit), and each is committed + pushed as its own logical commit when green ("always push this session" pairs naturally).
+- Ambiguity handling: if a task turns out underspecified or already-satisfied, do not guess-implement — file a VERIFY noting why and move on. (This session: "raise max spans to 5" — every cap was already 8.)
+- Exit / handoff: when the set is done, PAGE the user (proactive push notification) with the project name, the completed task(s), and a numbered list of remaining tasks. The user confirms ready + names the next project in one reply.
+
+Open questions for the canonical:
+- Where the page fires (every task vs end-of-set) and via what (push notification).
+- How it composes with the existing no-approvals + always-push session modes (is "fix speedrun" just a named preset of those plus the end page?).
+- Whether it should auto-pull the task set from a tag/priority query rather than an explicit list.
+- Guardrails: it should refuse to speedrun tasks that need design decisions or carry data-loss risk without a checkpoint (e.g. the unused-tile flag here was biased-safe deliberately).
+
+This is a rulesets-owned concept (cross-project), so sending it here rather than building it locally. No local stopgap made.
diff --git a/docs/design/2026-06-15-spec-storage-lifecycle-proposal.org b/docs/design/2026-06-15-spec-storage-lifecycle-proposal.org
new file mode 100644
index 0000000..7dbba72
--- /dev/null
+++ b/docs/design/2026-06-15-spec-storage-lifecycle-proposal.org
@@ -0,0 +1,44 @@
+#+TITLE: Proposal — spec storage location + lifecycle-status convention
+
+* What I'm proposing
+
+Two coupled documentation conventions, surfaced while triaging ~28 design docs in .emacs.d. Both belong in rulesets — the spec-create workflow (=.ai/workflows/spec-create.org=) and likely a new docs-lifecycle rule under =claude-rules/=.
+
+** A. Separate specs from working notes by location
+Today .emacs.d/docs/design/ holds everything in one bucket: formal specs (goals/decisions/phases/acceptance) jumbled with brainstorms, inventories, reviews, and idea-lists. You can't tell specs from notes without opening each file.
+
+Proposal: formal specs live in =docs/specs/=; =docs/design/= keeps working notes, brainstorms, inventories, and reviews. A "spec" is a doc proposing a buildable change with a Decisions section and phases; everything else is a note.
+
+** B. Make a spec's lifecycle status glanceable
+Specs carry no lifecycle status today, so which shipped, which are open, and which are dead is invisible. I had to run a four-agent sweep reading every spec against the code to reconstruct it (6 implemented, 8 in-progress, 12 not-started, 1 superseded). A status convention makes that a one-line scan.
+
+Two parts:
+- Filename suffix (Craig's idea): =<topic>-spec.org= (draft / not-started) → =-spec-doing.org= → =-spec-implemented.org= → =-spec-superseded.org= / =-spec-cancelled.org=. Visible in =ls=, greppable by glob, impossible to forget.
+- Authoritative Status field in the spec's Metadata table, on every spec (retrofit old ones). The filename is the index; the field is the record — it also carries a dated history line a filename can't hold.
+
+* Why
+
+Triage. A collection of specs with no status field and no location split degrades into "open it to find out." The cost compounds with every spec added. The two conventions together make the answer visible from a directory listing.
+
+* Recommendations / refinements
+
+- Pair filename suffix + Status field; the field is authoritative, the filename is the at-a-glance index. Don't let the filename be the only record (it's volatile and loses the why/when).
+- Link safety is the real cost. Both the move to =docs/specs/= and every status rename break =[[file:...]]= links (e.g. todo.org → a spec's Related link). Mitigate one of two ways, and pick one as the standard:
+ - Switch cross-doc links to =org-id= (=[[id:...]]=), which survive moves and renames; or
+ - A helper that moves/renames + relinks inbound references + stamps the Status field + appends a dated history line, run on each transition — so a status change is one command, not manual link surgery.
+- Vocabulary: draft (no suffix) / doing / implemented / superseded / cancelled. An in-progress-but-partial spec is "doing."
+
+* The bigger design choice rulesets should weigh
+
+Filename-encoded status is one option; the other is the org TODO keyword on the spec's top heading. These specs already carry =#+TODO: TODO | DONE SUPERSEDED CANCELLED=, so the top-heading keyword could be the status — scannable via a docs/specs org-agenda view, greppable, and it never breaks a link because the filename stays stable. The tradeoff is ls-visibility (filename wins) versus link-stability and zero-rename transitions (org-keyword wins). Worth deciding deliberately; my lean is filename suffix for the listing-level visibility Craig wants, paired with the Status field, with org-id links to neutralize the rename cost.
+
+* Generalization — a reusable pattern
+
+This is not spec-specific. The shape is: lifecycle state visible in the artifact's name (or location), an authoritative status record inside the artifact, rename-safe linking, and formal artifacts separated from working notes by location. Reach for it whenever triaging a growing collection of processed documents or media — brainstorms, inbox items, working-files, a transcription/recording queue, anything with a draft → in-progress → done/dropped lifecycle. Worth capturing as a general docs-lifecycle convention in claude-rules/, with the spec-create workflow as the first concrete instance.
+
+* Follow-up
+
+- Decide the canonical mechanism (filename suffix vs org-TODO-keyword) and the link-safety standard (org-id vs relink-helper).
+- Update spec-create to emit into =docs/specs/= with a Status field and the chosen status mechanism documented.
+- If you want the relink-helper, it's a natural =.ai/scripts/= addition that downstream projects get via the template sync.
+- Send a note back if you want .emacs.d to pilot the chosen approach before it's generalized.
diff --git a/docs/design/2026-06-16-inbox-zero-phase-e-proposal.diff b/docs/design/2026-06-16-inbox-zero-phase-e-proposal.diff
new file mode 100644
index 0000000..e3d8ee8
--- /dev/null
+++ b/docs/design/2026-06-16-inbox-zero-phase-e-proposal.diff
@@ -0,0 +1,65 @@
+--- claude-templates/.ai/workflows/inbox-zero.org 2026-06-13 13:18:35.988799778 -0500
++++ /proc/self/fd/12 2026-06-16 00:18:07.592944696 -0500
+@@ -1,6 +1,3 @@
+-#+TITLE: Inbox Zero Workflow
+-#+AUTHOR: Craig Jennings & Claude
+-#+DATE: 2026-06-13
+
+ * Overview
+
+@@ -19,7 +16,7 @@
+ Reused from three callers so the steps live in one place:
+ - *Startup* (read-only nudge) — count the items, identify which appear related to this project, surface both numbers, offer processing as one of the startup options. Never auto-files.
+ - *Wrap-up* (Step 3 sub-step) — sweep items that belong here before the cleanup scripts, so imported tasks lint and ride the wrap commit.
+-- *On demand* — "inbox zero", "empty the inbox", "process the roam inbox", "triage my roam inbox".
++- *On demand* — "inbox zero", "empty the inbox", "process the roam inbox", "triage my roam inbox". The on-demand caller (and the recurring loop that invokes it) also runs *Phase E* below: after routing the inbox, it works the project's actionable backlog, implementing eligible =:next:= / =:quick:+:solo:= tasks. Startup and wrap-up skip Phase E.
+
+ Each project touches the roam inbox at least twice a session this way: once at startup, once at wrap-up.
+
+@@ -69,6 +66,46 @@
+
+ Report: moved (with their new priorities and tags), folded, dropped-as-done. Then the residue: foreign items (left for their owners, count only) and unowned items (count plus the headings that appear related to this project, for manual claim or prefix). Same "summarize what we kept" shape.
+
++* Phase E — Execute actionable tagged tasks (autonomous; on-demand / loop caller only)
++
++After routing the inbox, the on-demand and loop callers work the project's actionable backlog. *Startup (read-only) and wrap-up (winding down) skip this phase entirely* — it runs only when a human or the recurring loop invokes the workflow to make progress, never as a side effect of session bookkeeping.
++
++** Eligibility gate
++
++Scan =todo.org= (both items freshly filed in Phase B and the existing backlog). A task is a candidate when ALL hold:
++
++1. Status is =TODO= — not =VERIFY=, =DOING=, =DONE=, or =CANCELLED=. =VERIFY= means "awaiting Craig's manual confirmation"; never auto-implement a VERIFY. Those are the manual-testing verifications this phase deliberately leaves alone.
++2. Tagged =:next:=, OR tagged BOTH =:quick:= AND =:solo:=.
++3. Implementable solo — no input or undecided judgment call from Craig.
++4. Estimated at roughly 30 minutes or less of focused work.
++
++** Act-vs-file decision
++
++For each candidate, after a quick scope read of the relevant code:
++
++- *Clear, bounded, solo, ≤ ~30 min* → implement it now (below).
++- *Needs Craig's input, a decision, or design discussion* → do NOT implement. Leave it filed, add a one-line note on the task naming the input it needs, and surface it.
++- *An hour or more* → do NOT implement. Leave it filed and surface it as a larger task for a dedicated =/start-work= session.
++
++When unsure which side a task falls on, file rather than implement. A wrong auto-implement costs more than a deferred task.
++
++** Implementing a candidate
++
++Per task, follow the project's commit discipline — the per-project waiver: no approval gate, but TDD + =/review-code= + =/voice personal= on every commit, no AI attribution:
++
++1. Trace to root cause; write the failing test first (Red → Green → Refactor).
++2. Live-reload into the running daemon and verify per the emacs reload-and-verify loop.
++3. Close the task per =todo-format.md= (top-level → =DONE= + =CLOSED:=; sub-task → dated log rewrite). When the only residue is Craig's manual check, file a =VERIFY= child under "Manual testing and validation" and close the originating task (the codified manual-verification-handoff pattern).
++4. =/review-code --staged= → fix all Critical/Important → =/voice personal= on the message → commit individually. Push per the project's flow.
++
++** Bounding the run
++
++Default to one task per run: implement the highest-priority eligible candidate (=[#A]= before =[#B]= before =[#C]=), commit, then stop and let the next tick or the next on-demand invocation continue. A caller may work more than one in a run when the eligible tasks are small and clearly independent — but each gets its own test and its own commit, and the run stays reviewable. Never batch unrelated changes into one commit.
++
++** Surface
++
++Report what was implemented (task + commit), what was deferred and why (needs-input / too-large), and what stays filed.
++
+ * Skip conditions
+
+ - No =~/org/roam/inbox.org= → silent no-op.
diff --git a/.ai/workflows/inbox-zero.org b/docs/design/2026-06-16-inbox-zero-phase-e-proposal.org
index aa7c273..5fa7e12 100644
--- a/.ai/workflows/inbox-zero.org
+++ b/docs/design/2026-06-16-inbox-zero-phase-e-proposal.org
@@ -1,5 +1,5 @@
#+TITLE: Inbox Zero Workflow
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-06-13
* Overview
@@ -19,7 +19,7 @@ This version routes each item to its one owning project, identified by an explic
Reused from three callers so the steps live in one place:
- *Startup* (read-only nudge) — count the items, identify which appear related to this project, surface both numbers, offer processing as one of the startup options. Never auto-files.
- *Wrap-up* (Step 3 sub-step) — sweep items that belong here before the cleanup scripts, so imported tasks lint and ride the wrap commit.
-- *On demand* — "inbox zero", "empty the inbox", "process the roam inbox", "triage my roam inbox".
+- *On demand* — "inbox zero", "empty the inbox", "process the roam inbox", "triage my roam inbox". The on-demand caller (and the recurring loop that invokes it) also runs *Phase E* below: after routing the inbox, it works the project's actionable backlog, implementing eligible =:next:= / =:quick:+:solo:= tasks. Startup and wrap-up skip Phase E.
Each project touches the roam inbox at least twice a session this way: once at startup, once at wrap-up.
@@ -69,6 +69,46 @@ The roam inbox lives in a git repo (=~/org/roam=, auto-synced by the =roam-sync=
Report: moved (with their new priorities and tags), folded, dropped-as-done. Then the residue: foreign items (left for their owners, count only) and unowned items (count plus the headings that appear related to this project, for manual claim or prefix). Same "summarize what we kept" shape.
+* Phase E — Execute actionable tagged tasks (autonomous; on-demand / loop caller only)
+
+After routing the inbox, the on-demand and loop callers work the project's actionable backlog. *Startup (read-only) and wrap-up (winding down) skip this phase entirely* — it runs only when a human or the recurring loop invokes the workflow to make progress, never as a side effect of session bookkeeping.
+
+** Eligibility gate
+
+Scan =todo.org= (both items freshly filed in Phase B and the existing backlog). A task is a candidate when ALL hold:
+
+1. Status is =TODO= — not =VERIFY=, =DOING=, =DONE=, or =CANCELLED=. =VERIFY= means "awaiting Craig's manual confirmation"; never auto-implement a VERIFY. Those are the manual-testing verifications this phase deliberately leaves alone.
+2. Tagged =:next:=, OR tagged BOTH =:quick:= AND =:solo:=.
+3. Implementable solo — no input or undecided judgment call from Craig.
+4. Estimated at roughly 30 minutes or less of focused work.
+
+** Act-vs-file decision
+
+For each candidate, after a quick scope read of the relevant code:
+
+- *Clear, bounded, solo, ≤ ~30 min* → implement it now (below).
+- *Needs Craig's input, a decision, or design discussion* → do NOT implement. Leave it filed, add a one-line note on the task naming the input it needs, and surface it.
+- *An hour or more* → do NOT implement. Leave it filed and surface it as a larger task for a dedicated =/start-work= session.
+
+When unsure which side a task falls on, file rather than implement. A wrong auto-implement costs more than a deferred task.
+
+** Implementing a candidate
+
+Per task, follow the project's commit discipline — the per-project waiver: no approval gate, but TDD + =/review-code= + =/voice personal= on every commit, no AI attribution:
+
+1. Trace to root cause; write the failing test first (Red → Green → Refactor).
+2. Live-reload into the running daemon and verify per the emacs reload-and-verify loop.
+3. Close the task per =todo-format.md= (top-level → =DONE= + =CLOSED:=; sub-task → dated log rewrite). When the only residue is Craig's manual check, file a =VERIFY= child under "Manual testing and validation" and close the originating task (the codified manual-verification-handoff pattern).
+4. =/review-code --staged= → fix all Critical/Important → =/voice personal= on the message → commit individually. Push per the project's flow.
+
+** Bounding the run
+
+Default to one task per run: implement the highest-priority eligible candidate (=[#A]= before =[#B]= before =[#C]=), commit, then stop and let the next tick or the next on-demand invocation continue. A caller may work more than one in a run when the eligible tasks are small and clearly independent — but each gets its own test and its own commit, and the run stays reviewable. Never batch unrelated changes into one commit.
+
+** Surface
+
+Report what was implemented (task + commit), what was deferred and why (needs-input / too-large), and what stays filed.
+
* Skip conditions
- No =~/org/roam/inbox.org= → silent no-op.
diff --git a/docs/design/2026-06-16-inbox-zero-phase-e-sender-note.org b/docs/design/2026-06-16-inbox-zero-phase-e-sender-note.org
new file mode 100644
index 0000000..08e7650
--- /dev/null
+++ b/docs/design/2026-06-16-inbox-zero-phase-e-sender-note.org
@@ -0,0 +1,27 @@
+#+TITLE: inbox-zero Phase E — autonomous execution of :next: / :quick:+:solo: tasks
+
+* What changed
+
+Added a new "Phase E — Execute actionable tagged tasks" to inbox-zero.org (edited locally in .emacs.d as a stopgap; the edited file is attached alongside this note). After routing the roam inbox, the on-demand and loop callers now scan todo.org and autonomously implement eligible tasks.
+
+Eligibility gate (all must hold): status TODO (not VERIFY/DOING/DONE/CANCELLED); tagged :next: OR both :quick: and :solo:; solo-doable without Craig's input; ~30 min or less. VERIFY is explicitly excluded — in this project VERIFY means "awaiting Craig's manual confirmation," i.e. the manual-testing verifications that must NOT be auto-implemented. Anything needing input or an hour-plus is filed and surfaced, never implemented. Each implementation follows the project commit discipline (TDD + /review-code + /voice personal + individual commit, no AI attribution). Startup (read-only) and wrap-up (winding down) skip Phase E. Default is one task per run, highest priority first.
+
+* Why
+
+Craig wants the 30-minute inbox-zero loop to also burn down the small actionable backlog autonomously, not just route capture. The existing tag convention (:next:, :quick:+:solo:) already marks exactly the solo-doable work, so Phase E acts on it under a conservative act-vs-file gate. Companion change: .emacs.d is scheduling inbox-zero on a 30-minute loop with Phase E enabled.
+
+* Requesting considerations for a general (cross-project) improvement
+
+The local edit is a stopgap; the durable, canonical version is yours to shape. Open questions for the canonical:
+
+1. Commit autonomy. Phase E as written assumes .emacs.d's per-project waiver (no per-commit approval gate). Most projects do NOT have that waiver. The canonical Phase E should probably gate on whether autonomous commits are allowed, defaulting to file-only (no auto-commit, surface a diff) unless a project opts in.
+
+2. Tag resolution. The eligibility tags and the per-project priority/tag scheme vary. Phase E hardcodes :next: / :quick:+:solo:. Better: read the project's todo.org priority/tag scheme header to resolve which tags mean "autonomous-safe."
+
+3. Do-not-implement markers. "Manual testing verification" is encoded here as status VERIFY (faithful to todo-format.md). Other projects may use VERIFY differently or not at all. Worth a canonical definition of the "do-not-auto-implement" marker set.
+
+4. Bounding and kill switch. A loop firing every 30 min that implements + commits unattended is powerful. Consider a per-run cap (or a token budget) and an explicit kill switch beyond "default one task per run."
+
+5. Seam. Whether autonomous execution belongs inside inbox-zero at all, or in its own workflow (e.g. work-the-backlog.org) that the loop chains after inbox-zero. Coupling capture-routing with autonomous-implementation may be the wrong seam — separating them keeps inbox-zero's three existing callers (startup, wrap-up, on-demand) clean.
+
+Sent from .emacs.d (helper id n/a — primary session).
diff --git a/docs/design/2026-06-17-flashcard-multitag-note.md b/docs/design/2026-06-17-flashcard-multitag-note.md
new file mode 100644
index 0000000..cfe960d
--- /dev/null
+++ b/docs/design/2026-06-17-flashcard-multitag-note.md
@@ -0,0 +1,28 @@
+# Flashcard tooling: multi-tag headings + --tag-filter (for curated subset decks)
+
+Source: work project, 2026-06-17. Two synced scripts edited locally as a stopgap; please reconcile into the rulesets canonical so the next sync doesn't revert them.
+
+## What changed and why
+
+Craig wanted a curated "DeepSat Fundamentals" Anki deck: the 100 most fundamental cards out of the 465-card deepsat.org org-drill deck, marked with a second org tag (`:fundamental:`) so they stay findable/grep-able in the source.
+
+The blocker: both `flashcard-to-anki.py` and `flashcard-stats.py` keyed cards on a heading ending in exactly ` :drill:` (`CARD_RE = ^\*\*\s+(.+?)\s+:drill:\s*$`). Adding any second org tag turns the heading into `... :fundamental:drill:`, which that regex does not match — so the 100 tagged cards would silently drop from the full-deck apkg and be undercounted by stats. The "passing gate skips your file" failure mode.
+
+## flashcard-to-anki.py
+
+- `CARD_RE` now matches a trailing org tag block (`^\*\*\s+(.*?)\s+(:[A-Za-z0-9_@#%:]+:)\s*$`); a heading is a card when `drill` is among its tags. Other org tags ride along as Anki tags next to the section tag.
+- Card body is now bounded by any L1/L2 heading (`HEADING_RE = ^\*{1,2}\s`) instead of only `* ` or a drill heading.
+- New `--tag-filter <tag>`: emit only cards carrying that org tag (e.g. `--tag-filter fundamental` → the 100-card subset).
+- New `--guid-salt <s>`: salt note GUIDs so a derived subset deck gets its own GUID space. Without it, the subset's notes share fronts with the full deck, Anki dedupes on GUID, and the subset deck imports empty. Default (no salt) is unchanged — `guid_for(front)` — so the existing deepsat deck's GUIDs and SRS state are untouched.
+
+Generation used: `flashcard-to-anki.py deepsat.org --tag-filter fundamental --deck "DeepSat Fundamentals" --guid-salt fundamentals`.
+
+## flashcard-stats.py
+
+- Same `CARD_RE` broadening + `HEADING_RE` body bound, and the card guard now checks `drill` membership in the tag block. Verified: full deck still counts 465 after 100 cards were multi-tagged.
+
+## Companion files to reconcile
+
+- Both rulesets copies: `~/code/rulesets/.ai/scripts/` and `~/code/rulesets/claude-templates/.ai/scripts/` (the synced source).
+- `claude-templates/.ai/scripts/tests/flashcard-sync.bats` — worth adding a multi-tag case (a `:foo:drill:` heading still parses; `--tag-filter foo` returns only those) so this doesn't regress.
+- Regression checked locally: full deck parses to 465 with and without the change; `--tag-filter fundamental` returns exactly 100.
diff --git a/docs/design/2026-06-17-flashcard-multitag-stats.py b/docs/design/2026-06-17-flashcard-multitag-stats.py
new file mode 100755
index 0000000..3c984e7
--- /dev/null
+++ b/docs/design/2026-06-17-flashcard-multitag-stats.py
@@ -0,0 +1,332 @@
+#!/usr/bin/env python3
+"""Inventory + authoring-quality checks for an org-drill deck source file.
+
+Reports counts and flags two tiers of issue.
+
+Blocking WARNs (exit 1):
+- PROPERTIES drawer count not matching card count
+- Cards missing :ID: (risks SRS-state loss across rewrites)
+- `*** Answer` sub-headers (should be 0 per flashcard-review.org)
+- Non-prompt headings (topic-as-heading not yet rewritten)
+- #+TITLE missing, or carrying source-tool jargon ("org-drill")
+- Answer leakage: a card whose question echoes most of its own answer
+ (Source: citation lines and created-date lines are excluded from the
+ overlap, and range/category cards that recall numbers are exempted)
+- Duplicate / near-duplicate fronts (interference between confusable cards)
+
+Non-blocking NOTEs (exit unaffected):
+- Overloaded backs (long answer — candidate to split into atomic cards)
+- List-shaped backs (enumeration — candidate to split or use overlapping cloze)
+- Binary yes/no prompts (low retrieval effort — candidate to reformulate)
+
+Exits 0 when no blocking warnings are present, 1 otherwise, 2 on bad usage.
+Use as a gate before regenerating the Anki deck or running flashcard-sync.
+
+The fuzzy checks (leakage, duplicate, overloaded) are tuned by the LEAKAGE_*
+and BACK_WORD_LIMIT constants below; loosen them if a real deck trips false
+positives.
+
+Usage:
+ flashcard-stats.py <file.org>
+"""
+from __future__ import annotations
+
+import re
+import sys
+from pathlib import Path
+
+# A level-2 card heading carries a trailing org tag block that includes
+# `drill`. The block may hold more than one tag (e.g. ":fundamental:drill:"),
+# so match the whole block and check membership rather than pinning :drill:
+# as the literal last tag.
+CARD_RE = re.compile(r"^\*\*\s+(.+?)\s+(:[A-Za-z0-9_@#%:]+:)\s*$")
+HEADING_RE = re.compile(r"^\*{1,2}\s")
+ANSWER_RE = re.compile(r"^\*\*\*\s+Answer\b")
+PROP_START_RE = re.compile(r"^\s*:PROPERTIES:\s*$")
+PROP_END_RE = re.compile(r"^\s*:END:\s*$")
+ID_RE = re.compile(r"^\s*:ID:\s+(\S+)\s*$")
+TITLE_RE = re.compile(r"^#\+TITLE:\s*(.+?)\s*$", re.IGNORECASE)
+SOURCE_TOOL_RE = re.compile(r"\borg[-\s]?drill\b", re.IGNORECASE)
+PLANNING_RE = re.compile(r"^\s*(SCHEDULED|DEADLINE|CLOSED):\s")
+SOURCE_LINE_RE = re.compile(r"^\s*source:\s", re.IGNORECASE)
+CREATED_LINE_RE = re.compile(r"^\s*:?created:?\s", re.IGNORECASE)
+RANGE_RE = re.compile(r"\d[^\n]*[-–—]\s*\d")
+THRESHOLD_RE = re.compile(r"[<>≤≥]\s*\d")
+BULLET_RE = re.compile(r"^\s*([-+*]|\d+[.)])\s+")
+BINARY_LEAD_RE = re.compile(
+ r"^\s*(is|are|was|were|does|do|did|can|could|should|would|will|has|have|had)\b",
+ re.IGNORECASE,
+)
+
+# A heading qualifies as "prompt form" if it contains `?` or starts with one of
+# these imperative verbs (directive prompts like "Spell these out" and
+# "Introduce yourself" are valid even without `?`).
+IMPERATIVE_VERBS = frozenset({
+ "spell", "describe", "explain", "name", "list", "give",
+ "show", "tell", "define", "compare", "identify", "outline",
+ "introduce", "walk", "state", "recite", "recall", "summarize",
+})
+
+# Function words ignored when comparing a question against its answer.
+STOPWORDS = frozenset({
+ "the", "a", "an", "is", "are", "was", "were", "of", "to", "in", "on",
+ "for", "and", "or", "with", "what", "who", "whom", "when", "where", "why",
+ "how", "which", "does", "do", "did", "tell", "me", "about", "their", "this",
+ "that", "it", "as", "at", "by", "be", "your", "you", "they", "them",
+})
+
+# Tuning knobs for the fuzzy checks.
+LEAKAGE_RATIO = 0.8 # share of a question's content words echoed in its answer
+LEAKAGE_MIN_WORDS = 3 # ignore very short questions, where overlap is noise
+BACK_WORD_LIMIT = 60 # words on a card back before it's flagged as overloaded
+
+
+def is_prompt_form(heading: str) -> bool:
+ """True if the heading reads as a question or imperative prompt."""
+ if "?" in heading:
+ return True
+ first_word = heading.split(None, 1)[0].lower().rstrip(":,;")
+ return first_word in IMPERATIVE_VERBS
+
+
+def content_words(text: str) -> set[str]:
+ """Lowercased alphanumeric tokens of length >= 3, minus stopwords."""
+ return {w for w in re.findall(r"[a-z0-9]+", text.lower())
+ if len(w) >= 3 and w not in STOPWORDS}
+
+
+def leakage_ratio(heading: str, body: str) -> float:
+ """Fraction of the question's content words that reappear in the answer.
+
+ A high ratio means the answer is largely restated in the question, so the
+ card can be answered by recognition rather than recall. Returns 0.0 for a
+ question with fewer than LEAKAGE_MIN_WORDS content words, where overlap is
+ just noise.
+ """
+ hw = content_words(heading)
+ if len(hw) < LEAKAGE_MIN_WORDS:
+ return 0.0
+ return len(hw & content_words(body)) / len(hw)
+
+
+def prose_body(body: str) -> str:
+ """Body with Source: citation and created-date lines removed.
+
+ Those lines are metadata, not the answer. A Source line's URL slug often
+ repeats the question's words, and a created date is bookkeeping — neither
+ should count toward answer-leakage overlap.
+ """
+ return "\n".join(
+ ln for ln in body.splitlines()
+ if not SOURCE_LINE_RE.match(ln) and not CREATED_LINE_RE.match(ln)
+ )
+
+
+def has_distinct_numeric_recall(heading: str, body: str) -> bool:
+ """True if the answer carries numeric ranges/thresholds the question lacks.
+
+ A range/category card ("What are the HbA1c ranges across normal,
+ prediabetes, and diabetes?") echoes its categories in the answer, but the
+ recalled content is the numbers, which the question doesn't give away — so
+ high word overlap isn't leakage.
+ """
+ body_nums = bool(RANGE_RE.search(body) or THRESHOLD_RE.search(body))
+ head_nums = bool(RANGE_RE.search(heading) or THRESHOLD_RE.search(heading))
+ return body_nums and not head_nums
+
+
+def is_leaky(heading: str, body: str) -> bool:
+ """True if a card leaks its answer, after excluding citation lines and
+ numeric-recall (range/category) cards."""
+ prose = prose_body(body)
+ if leakage_ratio(heading, prose) < LEAKAGE_RATIO:
+ return False
+ return not has_distinct_numeric_recall(heading, prose)
+
+
+def normalize_heading(heading: str) -> str:
+ """Collapse a heading to a comparison key (lowercase, alnum + single spaces)."""
+ return re.sub(r"\s+", " ", re.sub(r"[^a-z0-9 ]", " ", heading.lower())).strip()
+
+
+def is_binary_prompt(heading: str) -> bool:
+ """True for yes/no or 'A or B' prompts, which need little retrieval effort."""
+ if BINARY_LEAD_RE.match(heading):
+ return True
+ return bool(re.search(r"\bor\b", heading, re.IGNORECASE)) and heading.rstrip().endswith("?")
+
+
+def back_word_count(body: str) -> int:
+ return len(body.split())
+
+
+def is_list_back(body: str) -> bool:
+ """True if the answer body is mostly an org list (an enumeration card)."""
+ lines = [ln for ln in body.splitlines() if ln.strip()]
+ if len(lines) < 2:
+ return False
+ bullets = sum(1 for ln in lines if BULLET_RE.match(ln))
+ return bullets >= 2 and bullets * 2 >= len(lines)
+
+
+def parse_cards(lines: list[str]) -> tuple[list[dict], int]:
+ """Parse :drill: cards from org lines.
+
+ Returns (cards, prop_count). Each card is a dict with heading, has_id,
+ has_answer, and body (the answer text with PROPERTIES drawers, planning
+ lines, and `*** Answer` headers removed, approximating the rendered back).
+ """
+ cards: list[dict] = []
+ prop_count = 0
+ i = 0
+ n = len(lines)
+ while i < n:
+ m = CARD_RE.match(lines[i])
+ if not m or "drill" not in [t for t in m.group(2).split(":") if t]:
+ i += 1
+ continue
+ heading = m.group(1).strip()
+ i += 1
+ has_id = False
+ has_answer = False
+ in_drawer = False
+ body_lines: list[str] = []
+ while i < n:
+ line = lines[i]
+ if HEADING_RE.match(line):
+ break
+ if PROP_START_RE.match(line):
+ prop_count += 1
+ in_drawer = True
+ elif in_drawer and PROP_END_RE.match(line):
+ in_drawer = False
+ elif in_drawer:
+ if ID_RE.match(line):
+ has_id = True
+ elif ANSWER_RE.match(line):
+ has_answer = True
+ elif PLANNING_RE.match(line):
+ pass
+ else:
+ body_lines.append(line)
+ i += 1
+ cards.append({
+ "heading": heading,
+ "has_id": has_id,
+ "has_answer": has_answer,
+ "body": "\n".join(body_lines).strip(),
+ })
+ return cards, prop_count
+
+
+def find_duplicate_fronts(cards: list[dict]) -> list[tuple[str, str]]:
+ """Return (first, dup) heading pairs that normalize to the same key."""
+ seen: dict[str, str] = {}
+ dups: list[tuple[str, str]] = []
+ for c in cards:
+ key = normalize_heading(c["heading"])
+ if not key:
+ continue
+ if key in seen:
+ dups.append((seen[key], c["heading"]))
+ else:
+ seen[key] = c["heading"]
+ return dups
+
+
+def main() -> int:
+ if len(sys.argv) != 2:
+ print(f"usage: {sys.argv[0]} <file.org>", file=sys.stderr)
+ return 2
+
+ path = Path(sys.argv[1]).expanduser().resolve()
+ if not path.is_file():
+ print(f"error: {path} not found", file=sys.stderr)
+ return 2
+
+ lines = path.read_text(encoding="utf-8").splitlines()
+
+ title: str | None = None
+ for line in lines[:20]:
+ m = TITLE_RE.match(line)
+ if m:
+ title = m.group(1).strip()
+ break
+
+ cards, prop_count = parse_cards(lines)
+
+ no_id = [c["heading"] for c in cards if not c["has_id"]]
+ not_prompt = [c["heading"] for c in cards if not is_prompt_form(c["heading"])]
+ answer_count = sum(1 for c in cards if c["has_answer"])
+ leaky = [c["heading"] for c in cards if is_leaky(c["heading"], c["body"])]
+ dups = find_duplicate_fronts(cards)
+ overloaded = [c["heading"] for c in cards if back_word_count(c["body"]) > BACK_WORD_LIMIT]
+ listy = [c["heading"] for c in cards if is_list_back(c["body"])]
+ binary = [c["heading"] for c in cards if is_binary_prompt(c["heading"])]
+
+ print(f"{path.name} — drill deck stats")
+ print()
+ print(f"Deck title: {title if title else '(no #+TITLE)'}")
+ print(f"Cards: {len(cards)}")
+ drawer_status = "match" if prop_count == len(cards) else f"mismatch (expected {len(cards)})"
+ print(f"PROPERTIES drawers: {prop_count} ({drawer_status})")
+ print(f"*** Answer sub-headers: {answer_count} ({'clean' if answer_count == 0 else 'workflow violation'})")
+ print(f"Cards missing :ID:: {len(no_id)}")
+ print(f"Cards with non-prompt heading: {len(not_prompt)}")
+ print(f"Cards with possible answer leakage: {len(leaky)}")
+ print(f"Duplicate / near-duplicate fronts: {len(dups)}")
+ print()
+
+ warnings = 0
+
+ def emit_list(items: list[str]) -> None:
+ for h in items[:5]:
+ print(f" - {h}")
+ if len(items) > 5:
+ print(f" - ... and {len(items) - 5} more")
+
+ def warn(msg: str, items: list[str] | None = None) -> None:
+ nonlocal warnings
+ warnings += 1
+ print(f"WARN: {msg}")
+ if items:
+ emit_list(items)
+
+ def note(msg: str, items: list[str] | None = None) -> None:
+ print(f"NOTE: {msg}")
+ if items:
+ emit_list(items)
+
+ if title is None:
+ warn("no #+TITLE: line found; deck name will fall back to the file basename")
+ elif SOURCE_TOOL_RE.search(title):
+ warn(f"#+TITLE contains source-tool jargon ('{title}'); the deck name shows in Anki — drop 'Org-Drill' for a name that reads well on the consumption side")
+ if answer_count:
+ warn(f"{answer_count} cards have *** Answer sub-headers (drop per flashcard-review.org)")
+ if prop_count != len(cards):
+ warn(f"PROPERTIES count {prop_count} does not match card count {len(cards)}")
+ if no_id:
+ warn(f"{len(no_id)} cards missing :ID:; losing identity risks SRS-state loss across rewrites", no_id)
+ if not_prompt:
+ warn(f"{len(not_prompt)} cards have non-prompt headings (no '?' and no imperative-verb start); likely topic-as-heading not yet rewritten", not_prompt)
+ if leaky:
+ warn(f"{len(leaky)} cards may leak their answer (question echoes >= {int(LEAKAGE_RATIO * 100)}% of its own answer's key words); reformulate so the answer is recalled, not recognized", leaky)
+ if dups:
+ warn(f"{len(dups)} duplicate / near-duplicate fronts (interference between confusable cards); disambiguate or merge",
+ [f"{a} == {b}" for a, b in dups])
+
+ if overloaded:
+ note(f"{len(overloaded)} cards have a long answer (> {BACK_WORD_LIMIT} words); candidates to split into atomic cards", overloaded)
+ if listy:
+ note(f"{len(listy)} cards have a list-shaped answer; enumeration cards recall poorly — candidates to split or use overlapping cloze", listy)
+ if binary:
+ note(f"{len(binary)} cards are binary (yes/no or 'A or B'); low retrieval effort — candidates to reformulate open-ended", binary)
+
+ if warnings == 0:
+ print("clean (with non-blocking notes above)" if (overloaded or listy or binary) else "clean")
+ return 0
+ return 1
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/docs/design/2026-06-17-flashcard-multitag-to-anki.py b/docs/design/2026-06-17-flashcard-multitag-to-anki.py
new file mode 100755
index 0000000..3764acf
--- /dev/null
+++ b/docs/design/2026-06-17-flashcard-multitag-to-anki.py
@@ -0,0 +1,294 @@
+#!/usr/bin/env -S uv run --script
+# /// script
+# requires-python = ">=3.11"
+# dependencies = [
+# "genanki>=0.13",
+# ]
+# ///
+"""Convert an org-drill file into an Anki .apkg deck.
+
+Parses org-drill structure:
+ - Top-level "* Section" headings become tags on every card under them.
+ - Each "** Card name :drill:" entry becomes a card. Front = heading
+ text (sans the org tag block). Back = entry body with newlines
+ converted to <br>.
+
+A card heading may carry more than one org tag (e.g.
+"** Question :fundamental:drill:"). Any heading whose trailing tag block
+includes `drill` is a card; the other org tags ride along as Anki tags
+next to the section tag. Pass --tag-filter <tag> to emit only the cards
+carrying that org tag (e.g. a curated "fundamentals" subset).
+
+Deck name defaults to the input basename, case preserved. Deck and model
+IDs are derived from the deck name via stable hash so re-importing the
+same deck updates existing cards instead of duplicating them.
+
+Note GUIDs default to a hash of the card front, so re-running against the
+same source preserves SRS state. A derived subset deck (one built with
+--tag-filter) should pass --guid-salt so its notes get a distinct GUID
+space and Anki treats it as a separate deck rather than merging its cards
+into a full deck that shares the same fronts.
+
+Output defaults to ~/sync/phone/anki/<input-basename>.apkg. The .apkg is
+a mobile-Anki artifact the phone picks up from its sync dir, so it lands
+there rather than next to the org source.
+
+Usage:
+ flashcard-to-anki.py <input.org>
+ flashcard-to-anki.py <input.org> --deck "My Deck Name"
+ flashcard-to-anki.py <input.org> --output /path/to/deck.apkg
+ flashcard-to-anki.py <input.org> --tag-filter fundamental \
+ --deck "DeepSat Fundamentals" --guid-salt fundamentals
+
+Requires genanki, which uv resolves automatically via the PEP 723
+script metadata above. No venv or system install needed.
+"""
+from __future__ import annotations
+
+import argparse
+import hashlib
+import re
+import sys
+from pathlib import Path
+
+import genanki
+
+# 32-bit integer space genanki accepts. Start above the conventional
+# "user model" floor so collisions with hand-written decks stay
+# unlikely.
+ID_BASE = 1_500_000_000
+ID_RANGE = 500_000_000
+
+
+def stable_id(name: str, salt: str) -> int:
+ """Derive a deterministic 32-bit id from `name` and a `salt`.
+
+ Same (name, salt) pair always returns the same id, so re-running
+ against the same source produces a stable deck/model id pair and
+ Anki imports update existing cards in place rather than duplicating.
+ """
+ h = hashlib.sha256(f"{salt}:{name}".encode()).hexdigest()
+ return ID_BASE + (int(h[:8], 16) % ID_RANGE)
+
+
+def make_model(deck_name: str) -> genanki.Model:
+ return genanki.Model(
+ stable_id(deck_name, "model"),
+ f"{deck_name} (Craig)",
+ fields=[{"name": "Front"}, {"name": "Back"}],
+ templates=[
+ {
+ "name": "Card 1",
+ "qfmt": "{{Front}}",
+ "afmt": '{{FrontSide}}<hr id="answer">{{Back}}',
+ }
+ ],
+ css=(
+ ".card { font-family: sans-serif; font-size: 18px; "
+ "color: #222; background: #fafafa; line-height: 1.45; }\n"
+ "hr#answer { margin: 14px 0; }\n"
+ ),
+ )
+
+
+def section_to_tag(title: str) -> str:
+ return re.sub(r"[^a-z0-9]+", "-", title.lower()).strip("-")
+
+
+def escape_html(s: str) -> str:
+ return (
+ s.replace("&", "&amp;")
+ .replace("<", "&lt;")
+ .replace(">", "&gt;")
+ )
+
+
+def strip_org_metadata(body_lines: list[str]) -> list[str]:
+ """Drop :PROPERTIES: drawers, planning lines, and created-date lines.
+
+ Org-drill needs these in the source file (SRS state lives in the
+ PROPERTIES drawer; SCHEDULED carries the next-review date), but they
+ are noise on the back of an Anki card. A created/added date never
+ belongs on a card, so a stray "Created:" or ":CREATED:" body line is
+ dropped too.
+ """
+ cleaned: list[str] = []
+ in_drawer = False
+ planning_re = re.compile(r"^\s*(SCHEDULED|DEADLINE|CLOSED):\s")
+ created_re = re.compile(r"^\s*:?created:?\s", re.IGNORECASE)
+ drawer_start_re = re.compile(r"^\s*:PROPERTIES:\s*$")
+ drawer_end_re = re.compile(r"^\s*:END:\s*$")
+ for line in body_lines:
+ if in_drawer:
+ if drawer_end_re.match(line):
+ in_drawer = False
+ continue
+ if drawer_start_re.match(line):
+ in_drawer = True
+ continue
+ if planning_re.match(line) or created_re.match(line):
+ continue
+ cleaned.append(line)
+ return cleaned
+
+
+# A level-2 heading carrying a trailing org tag block. Group 1 is the
+# front text, group 2 the colon-delimited tag block (e.g. ":fundamental:drill:").
+CARD_RE = re.compile(r"^\*\*\s+(.+?)\s+(:[A-Za-z0-9_@#%:]+:)\s*$")
+# Any level-1 or level-2 heading — used to bound a card's body.
+HEADING_RE = re.compile(r"^\*{1,2}\s")
+SECTION_RE = re.compile(r"^\*\s+(.+?)\s*$")
+
+
+def parse(
+ org_text: str, tag_filter: str | None = None
+) -> list[tuple[str, str, list[str]]]:
+ """Return [(front, back_html, anki_tags), ...] for every :drill: card.
+
+ A card is any level-2 heading whose trailing org tag block includes
+ `drill`. Additional org tags become Anki tags alongside the section
+ tag. When `tag_filter` is set, only cards carrying that org tag are
+ returned.
+ """
+ cards: list[tuple[str, str, list[str]]] = []
+ current_section: str | None = None
+
+ lines = org_text.splitlines()
+ i = 0
+ while i < len(lines):
+ line = lines[i]
+
+ sec = SECTION_RE.match(line)
+ if sec:
+ current_section = sec.group(1).strip()
+ i += 1
+ continue
+
+ m = CARD_RE.match(line)
+ tags = [t for t in m.group(2).split(":") if t] if m else []
+ if m and "drill" in tags:
+ front = m.group(1).strip()
+ body_lines: list[str] = []
+ i += 1
+ while i < len(lines):
+ nxt = lines[i]
+ if HEADING_RE.match(nxt):
+ break
+ body_lines.append(nxt)
+ i += 1
+ body_lines = strip_org_metadata(body_lines)
+ while body_lines and not body_lines[0].strip():
+ body_lines.pop(0)
+ while body_lines and not body_lines[-1].strip():
+ body_lines.pop()
+ back_html = "<br>".join(escape_html(ln) for ln in body_lines)
+
+ org_tags = [t for t in tags if t != "drill"]
+ if tag_filter and tag_filter not in org_tags:
+ continue
+ anki_tags: list[str] = []
+ if current_section:
+ anki_tags.append(section_to_tag(current_section))
+ anki_tags.extend(org_tags)
+ if not anki_tags:
+ anki_tags = ["drill"]
+ cards.append((front, back_html, anki_tags))
+ continue
+
+ i += 1
+
+ return cards
+
+
+def build(
+ cards: list[tuple[str, str, list[str]]],
+ deck_name: str,
+ guid_salt: str | None = None,
+) -> genanki.Deck:
+ deck = genanki.Deck(stable_id(deck_name, "deck"), deck_name)
+ model = make_model(deck_name)
+ for front, back, tags in cards:
+ guid = (
+ genanki.guid_for(guid_salt, front)
+ if guid_salt
+ else genanki.guid_for(front)
+ )
+ note = genanki.Note(
+ model=model,
+ fields=[front, back],
+ tags=tags,
+ guid=guid,
+ )
+ deck.add_note(note)
+ return deck
+
+
+def default_deck_name(input_path: Path) -> str:
+ return input_path.stem
+
+
+def default_output_path(input_path: Path) -> Path:
+ anki_dir = Path.home() / "sync" / "phone" / "anki"
+ return anki_dir / f"{input_path.stem}.apkg"
+
+
+def main() -> int:
+ parser = argparse.ArgumentParser(
+ description="Convert an org-drill file into an Anki .apkg deck.",
+ )
+ parser.add_argument(
+ "input",
+ type=Path,
+ help="Path to the org-drill source file.",
+ )
+ parser.add_argument(
+ "--deck",
+ help="Deck name. Defaults to the input basename.",
+ )
+ parser.add_argument(
+ "--output",
+ type=Path,
+ help="Output .apkg path. Defaults to "
+ "~/sync/phone/anki/<input-basename>.apkg.",
+ )
+ parser.add_argument(
+ "--tag-filter",
+ help="Emit only cards carrying this org tag (e.g. 'fundamental').",
+ )
+ parser.add_argument(
+ "--guid-salt",
+ help="Salt note GUIDs with this string so a derived subset deck "
+ "gets its own GUID space and Anki keeps it separate from a "
+ "full deck sharing the same card fronts.",
+ )
+ args = parser.parse_args()
+
+ input_path: Path = args.input.expanduser().resolve()
+ if not input_path.is_file():
+ print(f"error: {input_path} not found", file=sys.stderr)
+ return 1
+
+ org_text = input_path.read_text(encoding="utf-8")
+ deck_name = args.deck or default_deck_name(input_path)
+ output_path: Path = (args.output or default_output_path(input_path)).expanduser().resolve()
+ output_path.parent.mkdir(parents=True, exist_ok=True)
+
+ cards = parse(org_text, tag_filter=args.tag_filter)
+ if not cards:
+ if args.tag_filter:
+ print(
+ f"error: no :drill: cards tagged :{args.tag_filter}: in {input_path}",
+ file=sys.stderr,
+ )
+ else:
+ print(f"error: no :drill: cards found in {input_path}", file=sys.stderr)
+ return 1
+
+ deck = build(cards, deck_name, guid_salt=args.guid_salt)
+ genanki.Package(deck).write_to_file(str(output_path))
+ print(f"wrote {output_path} ({len(cards)} cards, deck '{deck_name}')")
+ return 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/docs/design/2026-06-17-ntfy-agent-comms-proposal.org b/docs/design/2026-06-17-ntfy-agent-comms-proposal.org
new file mode 100644
index 0000000..0961b47
--- /dev/null
+++ b/docs/design/2026-06-17-ntfy-agent-comms-proposal.org
@@ -0,0 +1,89 @@
+#+TITLE: Proposal — Promote the ntfy phone channel into a general agent-comms tool
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-06-17
+
+* Why this is in rulesets' inbox
+
+The home project built a working, private phone-notification channel for Craig on 2026-06-17 (self-hosted ntfy over Tailscale). Craig wants rulesets to consider promoting it from a one-way notification system into a *general two-way communication tool* between him and his agents — and, critically, to move it off pure polling toward event-driven delivery (an inbound message can trigger an action or notify an agent, not just sit in a queue waiting to be polled).
+
+This is a proposal, not a change to anything rulesets owns. It documents exactly what exists, what ntfy makes possible, and the open design decisions rulesets would own. It also relates directly to the cross-agent-comms scripts that were retired from the templates in this same session — ntfy may be the transport layer that effort was missing.
+
+* Part 1 — What exists now (as-built, verified)
+
+- *Server:* ntfy in Docker on =ratio= at =~/docker/ntfy/= (=compose.yml= + =server.yml=, =data/= volume, =restart=unless-stopped=, healthcheck on =/v1/health=). Listens container :80, published to =127.0.0.1:2586=.
+- *Tailnet exposure:* =tailscale serve --bg --http=80 http://127.0.0.1:2586= → reachable at =http://ratio.tailf3bb8c.ts.net= (tailnet only, no public exposure). Disable with =tailscale serve --http=80 off=.
+- *Transport security:* plain HTTP, but every byte rides inside the WireGuard mesh (Tailscale), so it is encrypted end to end. The Tailscale account does not support TLS certs, and TLS would be redundant on the tailnet anyway. If ever exposed publicly, TLS + stronger auth become mandatory.
+- *Auth:* =auth-default-access: deny-all=. User =cj= has read-write on topics =claude= and =infra=. Anonymous is denied — verified 403 on both publish AND subscribe without a token. Token =tk_…= never expires. App login is username =cj= + a short password.
+- *Phone:* Pixel 6, ntfy F-Droid build (no Firebase / Google Play Services), WebSocket instant delivery. Already on the tailnet.
+- *Publisher wrapper:* =~/.local/bin/phone-notify= (on ratio only). Reads =~/.config/phone-notify/config= (chmod 600: URL, token, topic). Supports =-t/--title=, =-p/--priority=, =-T/--tags=, =--topic=, =--click=, =--url=.
+- *Verified two-way:* agent → phone push lands instantly; phone → publish to the topic lands on the server and is readable by the agent (Craig sent "It did, in fact, land." from the app and the agent polled and saw it).
+
+* Part 2 — The ntfy building blocks rulesets can use
+
+** Publish (agent → phone), already wired
+- =curl -H "Authorization: Bearer <token>" -d "msg" <url>/<topic>= or =phone-notify=.
+- Rich features available, unused so far: =Priority= (1-5), =Tags= (emoji/keywords), =Click= (URL opened on tap), =Actions= (tappable buttons — =view= a URL, =http= fire a request, =broadcast= an Android intent), =Attach= (files/images), =Markdown=, scheduled/delayed delivery (=At:= / =Delay:= header), and email/call forwarding.
+
+** Read (agent ← phone)
+- One-shot poll, all cached: =GET /<topic>/json?poll=1= (needs the token).
+- Only-new since a point: =?since=<id|timestamp|duration>= (e.g. =?since=5m= or =?since=<last-seen-id>=). This is the basis of a =phone-recv= helper that prints only messages newer than the last one seen.
+- Cache window is 12h (=cache-duration= in server.yml), so on-demand polling never misses a recent message.
+
+** Subscribe with side effects (the event-driven primitive)
+- =ntfy subscribe <topic> '<command>'= holds a persistent connection (WebSocket / JSON stream) and runs =<command>= for *every* inbound message, with fields exposed as environment variables (=$message=, =$title=, =$topic=, =$tags=, =$priority=, etc.).
+- This is the answer to "not all polling": a long-running subscriber reacts the instant a message arrives.
+
+* Part 3 — Making it event-driven (Craig's core ask)
+
+Three tiers, increasing capability and difficulty:
+
+** Tier A — Subscriber daemon routes inbound (clean, doable now)
+A systemd *user* service on ratio (always-on):
+#+begin_src
+ntfy subscribe --since=<last> claude /usr/local/bin/ntfy-inbound-handler
+#+end_src
+=ntfy-inbound-handler= classifies the message and routes it:
+- Append to a watched queue (a project =inbox/= or a dedicated comms file) → the next agent session picks it up at a task boundary (already in protocols: inbox check at task boundaries).
+- Fire desktop =notify= so a human at a screen sees it immediately.
+- Tag-based dispatch: =#task= → file as a TODO; =#infra= → infra queue; etc.
+
+This gets us instant reaction with zero polling, and it degrades gracefully — if nothing is listening, the message still sits in the queue.
+
+** Tier B — Inbound spawns a *new* agent session
+The handler invokes the =ai= launcher (or a scheduled/cron Claude run) to process the message autonomously. An inbound phone text becomes an agent action — "remind me to X" from anywhere, "what's the status of Y", "approve the pending commit". This is where it stops being a notifier and becomes a remote control for the agent fleet. Ties into the harness cron/schedule features and the retired cross-agent-comms intent.
+
+** Tier C — Notify / interrupt a *live* agent session (hardest, harness-dependent)
+A turn-based session has no native external interrupt. Honest options to explore:
+- The session runs a background subscriber/poll loop; the harness re-invokes the agent when backgrounded work emits or completes (the background-Bash + Monitor + ScheduleWakeup / =/loop= dynamic-pacing mechanisms).
+- A =/loop= that polls the topic every N seconds (still polling, but bounded and cheap).
+- Whatever the harness exposes for inbound push into a live session (e.g. a RemoteTrigger / inbound-PushNotification path) — needs experimentation.
+
+Recommendation: ship Tier A first (high value, low risk), prototype Tier B, treat Tier C as research.
+
+* Part 4 — The general-comms vision (beyond notifications)
+
+- *Channels as topics:* =claude= (agent ↔ Craig), =infra= (server/health/backup alerts — the DEGRADED-pool class), per-project topics, a cross-agent topic.
+- *Bidirectional chat:* Craig texts his agent from anywhere over Tailscale; the agent replies. Effectively private, self-hosted "SMS with your agent."
+- *Approval buttons:* the publish =Actions= feature can render Approve / Reject buttons on the phone. For the commits.md approval gates (commit message, PR body, PR review) when Craig is away from the desk, a tapped button fires a webhook the handler turns into "proceed." This is a concrete, high-value use.
+- *Attachments:* agent sends a generated screenshot/report to the phone; Craig sends a photo to the agent.
+
+* Part 5 — What rulesets would own / decide
+
+1. *Canonical tooling:* promote =phone-notify= (send) and add =phone-recv= (check-since) as rulesets bin scripts, synced to all machines via dotfiles/templates. Today =phone-notify= lives only on ratio.
+2. *Config + secret convention:* where the server URL + token live per machine (=~/.config/phone-notify/config= chmod 600 today), and whether the token should be a rulesets-managed GPG-encrypted secret distributed via dotfiles.
+3. *The subscriber daemon:* a reference =ntfy-inbound-handler= + a systemd user-unit template, plus the routing convention (tags → destinations).
+4. *Protocol conventions:* topic naming, a message format/tag vocabulary for routing, and how inbound maps to the existing =inbox/= and (retired) cross-agent-comms protocols.
+5. *Harness integration:* how — if at all — to wake or notify a live/new agent session on inbound. The Tier C research.
+6. *Relationship to cross-agent-comms:* decide whether ntfy is the transport that replaces the just-retired scripts, and whether agent↔agent messaging rides the same server (a dedicated topic) or stays separate.
+
+* Part 6 — Open questions
+
+- Multi-machine token distribution (per-machine config vs encrypted-in-dotfiles).
+- Daemon placement: one always-on subscriber on ratio vs per-machine subscribers.
+- Inbound integration with the existing inbox + the retired cross-agent protocols.
+- Live-session interrupt feasibility (entirely harness-dependent — needs a spike).
+- Whether agent↔agent comms and agent↔Craig comms share a server or are isolated.
+
+* Companion artifact
+
+The full as-built runbook (concrete values, server.yml, the verification checklist, the security model) lives in the home project at =working/phone-notifications/spec.org=. This proposal is the forward-looking half; that file is the operational record of what was deployed.
diff --git a/docs/design/2026-06-18-triage-intake-phone-push-note.org b/docs/design/2026-06-18-triage-intake-phone-push-note.org
new file mode 100644
index 0000000..2f6502b
--- /dev/null
+++ b/docs/design/2026-06-18-triage-intake-phone-push-note.org
@@ -0,0 +1,11 @@
+#+TITLE: WORKFLOW UPDATE — triage-intake.org auto mode gains a phone
+#+SOURCE: from work
+#+DATE: 2026-06-18 15:15:47 -0500
+
+WORKFLOW UPDATE — triage-intake.org auto mode gains a phone (ntfy) delivery step. Supersedes/consolidates the earlier 2026-06-18-1512 send.
+
+WHAT we did: added a new subsection 'Push each sweep to Craig's phone (ntfy) — the primary delivery' under 'Trigger and delivery' in the auto-mode section. It makes phone-notify (the self-hosted ntfy channel over Tailscale) the primary delivery for every auto-mode sweep, pushing the fuller End-of-sweep output (per-source deltas + open-PR/Linear state + the awaiting-ack list + a one-line verdict + a timestamp; SCAN FAILED banner if any source failed), and polling phone-recv each sweep for Craig's replies. Falls back to inline if phone-notify is absent.
+
+WHY: auto mode exists precisely for when Craig is away from the desk (out of office for a while, on vacation — which is the case right now, traveling 6/17-24). An inline-only report is useless if he is not at the screen; the whole point is reaching his phone. We ran it live this way all day and it worked, so we codified the delivery rather than re-deriving it each time. Craig confirmed this is his standard pattern: he starts auto-triage before leaving the office / on vacation.
+
+Companion: the reference_phone_notify_ntfy_channel harness memory documents the channel + the high-bar caveat. A separate task is producing standalone ntfy setup instructions (install + start/stop the service) so the channel can be brought up on other machines.
diff --git a/docs/design/2026-06-18-triage-intake-phone-push-workflow.org b/docs/design/2026-06-18-triage-intake-phone-push-workflow.org
new file mode 100644
index 0000000..a0bb416
--- /dev/null
+++ b/docs/design/2026-06-18-triage-intake-phone-push-workflow.org
@@ -0,0 +1,427 @@
+#+TITLE: Triage Intake Workflow (Engine)
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-05-01
+
+* Summary
+
+
+Lightweight, between-meetings sweep across whatever sources are plugged in — email, calendar, chat, open PRs, ticketing. Classifies what came in since the last check (Action / FYI / Noise-keep / Noise-trash), produces a single synthesized summary, and offers to execute the routine actions (trash, mark-read, star, respond, merge, attachment fetch).
+
+Think of it as the ER intake queue: every new message, invite, and PR notification is a "patient" walking through the door. This workflow is the triage nurse looking at the queue and telling Craig what needs attention now, what's just FYI, and what can be cleared.
+
+*This file is the engine.* It carries no sources of its own. Every source it scans comes from a *source plugin* — a =triage-intake.<source>.org= file the engine loads at Phase 0. The engine is source-agnostic and project-agnostic; the project- and account-specific knowledge lives entirely in the plugins. To add a source, drop a plugin file. To change one, edit its plugin. Never wire a source into this file.
+
+Distinct from =daily-prep.org=:
+- *daily-prep* — heavier, once daily, builds the day's plan + standup brief + meeting prep + time blocks.
+- *triage-intake* — fast, repeatable, just answers "what's new since last check?"
+
+
+Quick contract — what it does: fans out across source plugins, classifies every item into Action / FYI / Noise-keep / Noise-trash, synthesizes one deduped summary, writes each Action item to =todo.org= as a =:quick:reactive:= task, and executes star/mark-read/trash on confirmation.
+
+** When to Use This Workflow
+
+Trigger phrases:
+
+- "Run a triage-intake"
+- "Triage intake"
+- "What's new" / "What's new since I last checked"
+- "Do a sweep" / "Do a triage sweep"
+- "Check email, calendar, and PRs"
+
+Typical timing:
+
+- Between meetings (1-2 minute glance)
+- After a long focused-work block
+- Before context-switching to a new task
+- When ambient anxiety about "did I miss something?" creeps in
+
+Do *not* use when running daily-prep — daily-prep already does this as Phase 3.
+
+
+* Execution
+
+** Phase 0 — Load source plugins (MANDATORY — do not skip)
+
+The engine has no sources baked in. It discovers them by globbing *two* directories, and you MUST glob *both*:
+
+#+begin_src bash
+ls .ai/workflows/triage-intake.*.org .ai/project-workflows/triage-intake.*.org 2>/dev/null
+#+end_src
+
+- =.ai/workflows/triage-intake.*.org= — *general* source plugins, template-synced (personal Gmail, personal calendar, cmail/Proton, Telegram, personal GitHub PRs).
+- =.ai/project-workflows/triage-intake.*.org= — *PROJECT-SPECIFIC* source plugins, never synced, owned by this project (e.g. a work project's Linear, work Gmail, work Slack, enterprise-GitHub PRs).
+
+⚠ *THE #1 FAILURE MODE — read this twice.* Globbing only =.ai/workflows/= and silently missing every project plugin. If you skip =.ai/project-workflows/=, the sweep runs with *half its sources* and Craig never learns what it dropped — the omission is invisible, because a missing source looks identical to a quiet source in the output. There is no error, no empty block, no warning. The sweep just lies by omission. *Glob both directories. Always.*
+
+The glob exclude is automatic: =triage-intake.*.org= matches the plugins but not this engine file (=triage-intake.org= has no second dot-segment), so the engine never loads itself.
+
+After globbing, for each plugin file:
+1. Read it.
+2. Evaluate its =ENABLED= precondition. If false, *announce the skip with its reason* ("skipping linear — mcp__linear not present") and move on.
+3. The surviving set is the source list for Phases A-D.
+
+*Announce the loaded set before scanning* so the omission can't hide:
+
+#+begin_example
+Loaded 5 source plugins:
+ general: personal-gmail, personal-calendar, cmail, github-prs
+ project: deepsat-gmail
+ skipped: linear (mcp__linear not present)
+#+end_example
+
+If the project directory glob returns nothing, say so explicitly ("no project plugins in .ai/project-workflows/") rather than staying silent — silence is indistinguishable from forgetting to look.
+
+
+** Approach: Phases A → D
+
+*** Phase A: Fan-out (one parallel batch)
+
+Issue every enabled source's =Scan= command in a single message, with the anchor substituted in each source's declared format. They have no dependencies and benefit from running concurrently.
+
+Per-source subagent escalation: if a source's scan is expected to return more than its =SUBAGENT_OVER= count (e.g. personal Gmail after a multi-day gap), dispatch a subagent for that source. The subagent applies Phase B classification and returns the synthesized buckets, not the raw item list.
+
+*** Phase B: Classify per source (shared four-bucket model)
+
+Every item lands in one bucket. Plugins refine these with source-specific bias and noise patterns in their =Classify= section; they do not redefine the buckets.
+
+- *Action* — needs Craig to do something: an explicit ask, a decision needed, blocked-on-Craig, a mergeable PR, an invite needing a response, a deadline inside 48h.
+- *FYI* — substantive context worth seeing, but no action owed.
+- *Noise-keep* — low value but worth retaining (audit trail, receipts).
+- *Noise-trash* — safe to discard: newsletters, marketing, social digests, bot pings, redundant aggregator digests, wrong-recipient mail, past-event artifacts.
+
+Per-source bias (a work email account leans keep for audit value; a personal account leans trash on high noise volume) lives in each plugin's =Classify= section. Read it from there; don't re-derive it here.
+
+*** Phase C: Synthesize a single summary
+
+One markdown summary surfaced inline to Craig. Order:
+
+0. *Scan failures — first, loud, always.* Any loaded source whose scan failed, hung, was killed, or was skipped for an operational reason renders at the very top of the summary, before Top signals:
+
+ #+begin_example
+ ⚠ SCAN FAILED: <source> — <reason, one line> — <what's now unknown>
+ #+end_example
+
+ A failed scan is never folded into "quiet." Quiet means the scan ran and found nothing; a failure means the sweep is blind on that channel, and the reader must know which. The same applies to a precondition skip the user hasn't standing-approved (e.g. a messaging client that needs a temporary server spin-up): run the lifecycle or report the failure — don't silently narrow the sweep.
+
+1. *Top signals to act on* — bullet list of 3-7 items, ordered by urgency, *Action only*. Each bullet links to the source (permalink, thread URL, PR number).
+2. *Per-source breakdown* — one short section per loaded source *that has changes*, in =ORDER=, using that plugin's =Render= shape: Action items detailed, FYI items as a short list, Noise as a tally only ("Noise: 12 trash candidates, 4 keep, 0 starred").
+3. *Suggested actions* — explicit list of state changes Craig could take this run (trash these N messages, mark-read these M, star this Action item, respond to this invite, merge PRs #X and #Y, etc.). This line stays whenever there are queued actions, regardless of how quiet the sweep was.
+
+*Deltas only.* The summary reports what *changed* since the anchor: a new invite, a new/moved/cancelled calendar event, a new message needing attention. A source with no changes gets no block — no "Calendar — quiet", no "PRs — nothing new" roll-call. A sweep where nothing changed anywhere renders as a single line:
+
+#+begin_example
+17:39 sweep: no changes
+#+end_example
+
+(Craig, 2026-06-11: "we only need to report if anything's changed when we do triage intake. did someone send me a new invite? did christine throw something on my calendar that wasn't there earlier? did someone cancel a meeting?")
+
+Scan failures are the standing exception: a failed or skipped scan always renders loudly per point 0 above and is never folded into the no-change line — "no changes" is a claim about channels the sweep could actually see.
+
+Format target: scannable in 30 seconds, full read in 2 minutes. Don't pad.
+
+**** Sub-step: write each Action item into =todo.org= as its own =:quick:= task
+
+After surfacing the summary inline, append every Action item — regardless of source — to =todo.org= as its own top-level =** TODO= heading carrying the =:quick:= tag plus =:reactive:= and any relevant person/entity tag.
+
+Each Action item is one task. Don't group items by source under =** Email Response=, =** PR Review=, etc. sub-headings. Each response is its own filterable task so Craig can re-prioritize, =SCHEDULE:= / =DEADLINE:=, or tag individually.
+
+Format:
+
+#+begin_example
+*** TODO [#B] Merge PR #42 on archsetup (approved, CI green) — [[https://github.com/<user>/archsetup/pull/42][PR #42]] :quick:reactive:
+*** TODO [#B] Respond to the 2pm reschedule invite from Dana :quick:reactive:
+*** TODO [#B] Reply to the contract-terms email thread :quick:reactive:
+#+end_example
+
+Rules:
+
+- Heading is plain prose. Lead with the verb (Read / Re-review / Reply / Respond / Address / Merge / Schedule).
+- Priority: default =[#B]= for fresh reactive items. Bump to =[#A]= only if blocking someone or a deadline lands inside 7 days.
+- Tags: always =:quick:= + =:reactive:=. Add person/entity tags when the dependency is sharp.
+- Link the source in the heading when it has a URL (GitHub PR, mail thread, chat permalink). Use org's =[[url][label]]= form so the heading stays clickable in Emacs.
+- *Record the source locator in the task body* so a reply can be routed back to where the request came from — the channel + thread id for chat, the repo + PR number, the message id for mail. The general rule: a reply goes back to the *origin* of the request, not a fixed notification channel. (Project plugins may add stricter routing rules in their own files.)
+- Placement: append at end of =* Work Open Work= (just before =* Work Incubate=) unless the project's =todo.org= has a designated triage section near the top (=* Triage= or =* Inbox=).
+
+This sub-step makes triage-intake's findings *persist* in =todo.org= instead of evaporating after the inline summary.
+
+*** Phase D: Execute actions on confirmation
+
+Wait for Craig's go-ahead before running any state changes. Default to single-confirmation for the whole batch ("yes" → run everything proposed). Craig may also pick a subset ("trash personal but hold the work account") or hand back a different plan ("trash all but star the expense thread and queue PR merges for after lunch").
+
+Each action dispatches to the owning source plugin's =Actions= verb (trash, mark-read, star, respond, merge, comment, attachment-fetch). The engine doesn't hardcode action commands — it reads them from the loaded plugins. Read each plugin's =Actions= section for the exact command.
+
+After actions complete, write the Phase A capture into the sentinel's *content* (see "Capture the Phase A timestamp"): =echo "$PHASE_A_TS $(date -d "@$PHASE_A_TS" '+%Y-%m-%d %H:%M:%S %z')" > .ai/last-triage-intake=. Do not use plain =touch= (writes mtime to /now/ and strands items posted between Phase A and end of run) and do not use =touch -d "@$PHASE_A_TS"= (correct timestamp but mtime is per-machine — won't survive a fresh clone or cross-machine sync).
+
+*Do not close the workflow yet.* See Exit Criteria below.
+
+*** Exit Criteria
+
+The workflow stays open until Craig has *explicitly* either:
+
+1. *Confirmed* that the executed actions are sufficient and nothing more is needed this round, or
+2. *Handed back a different plan* (e.g., "actually hold the PR merges, address #131 first").
+
+A successful Phase D run is *not* an exit signal. After the action batch returns, surface what shipped and wait. Don't volunteer "done" or "all set" — those are exit-claim phrases that pre-empt Craig's call. Use a status report ("17 actions succeeded, sentinel written at 12:19") and stop.
+
+If Craig has been silent for a while after Phase D and the surface looks closed-out, *ask*: "Anything else on this triage, or are we good to close out?" Don't auto-terminate.
+
+This rule prevents the failure mode where the workflow self-declares done and the next exchange has to relitigate what state things are in.
+
+
+* Auto mode (unattended monitoring)
+
+Auto mode is a self-running variant of the engine for when Craig is away from the desk but wants tight awareness — a loop that runs the standard sweep on a short interval, *accumulates* findings rather than mutating state, and hands Craig a gated checkpoint to commit the batch. It composes two things: the *delivery* (a =/loop= in the live session) and the *behavior* (accumulate-don't-mutate sweeps with a checkpoint). The one-shot run above is unchanged; auto mode is an additional way to run the same Phase 0 / A-D engine.
+
+** Trigger and delivery
+
+- "auto triage" / "auto triage-intake" / "watch the desk" / "monitor the triage" — start auto mode.
+- Default interval *20 minutes*; Craig sets it.
+
+Auto mode runs as a =/loop= in the *live session*, not a detached cron job:
+
+#+begin_src
+/loop 20m run an auto-mode triage-intake sweep per triage-intake.org
+#+end_src
+
+Running in the live session means MCP auth (Slack, Gmail, Linear) is inherited from the session — the headless-auth wall that blocks a detached cron run does not apply. A durable cross-session schedule is out of scope here; that belongs to the morning-ops orchestrator, which can later invoke auto mode's accumulate behavior as its triage limb. The close/stop commands below require a live session by design.
+
+*** Push each sweep to Craig's phone (ntfy) — the primary delivery
+
+Auto mode exists for when Craig is away from the desk (out of office for a while, on vacation), so the report's primary delivery is a push to his phone, not an inline message he won't be looking at. After every sweep, send the End-of-sweep output to his phone via =phone-notify= (the self-hosted ntfy channel over Tailscale; see the phone-notify reference memory for usage and the high-bar caveat). Push on *every* sweep, including a quiet "no changes" one — the timestamp line is the proof the loop ran.
+
+The pushed summary is the *fuller* shape, not a terse one-liner: per-source deltas (Slack / work-email / Linear / PRs / calendar / Telegram, noise tallies included), the current open-PR + Linear state, the awaiting-acknowledgment list, a one-line verdict on whether anything needs Craig, and the timestamp. Lead with a ⚠ SCAN FAILED banner if any source failed.
+
+Poll =phone-recv= at the top of each sweep for anything Craig sends back (delivery is not pushed to the agent); act on his requests and reply via =phone-notify=. Note that =phone-recv= echoes the agent's own outgoing pushes back, so only treat a message as inbound from Craig when it is not one of the sweep summaries.
+
+If =phone-notify= isn't installed on the host (it lives on ratio for now), fall back to inline delivery and say so once.
+
+** Preconditions and Close-out
+
+Auto mode borrows the inbox-monitor gates (=monitor-inbox.org=): do not start on a dirty worktree or a red test suite — a close's batch commit would otherwise sweep up unrelated changes — and leave the tree clean and green when the loop stops. Surface a blocker with inline numbered options per =interaction.md= and wait.
+
+** A sweep: accumulate, don't mutate
+
+Each sweep runs Phase 0 (load *both* plugin dirs — the loud requirement still holds) and Phases A-D's scan / classify / synthesize, but performs *none* of the normal run's mutations:
+
+- Does NOT advance the sentinel. The scan window grows from the last *close* until the next close: every sweep scans from the existing sentinel up to now, so nothing between sweeps is dropped.
+- Does NOT write =todo.org= Action tasks — accumulates them for the close.
+- Does NOT take mail actions (trash / mark-read / star).
+- Does NOT commit.
+- DOES update an active daily-prep in Update mode and re-open it on change (per =daily-prep.org=).
+- DOES report, deltas-only, with loud scan-failure banners (Phase C rules unchanged).
+
+** End-of-sweep output — three sections
+
+1. *Deltas* — what changed since the *previous sweep* (the standard Phase C summary scoped to the inter-sweep delta; one line if nothing: "HH:MM sweep: no changes").
+2. *Responses awaiting your acknowledgment* — every Slack reply, email, or message directed at Craig that he hasn't acknowledged or had the agent answer. A *running list carried forward across sweeps* until Craig acks each item or closes the triage. An away user's first need is "who's waiting to hear back from me," which a delta-only sweep loses the moment it scrolls past.
+3. *Timestamp* — the current date, time, and timezone on the sweep's own final line, so an away reader sees how fresh the summary is without computing it. Print it on *every* sweep, including a quiet "no changes" one — on a quiet sweep the stamp is the proof the loop ran. Generate it with:
+
+ #+begin_src bash
+ date "+%A %Y-%m-%d %H:%M:%S %Z (%z)"
+ #+end_src
+
+** The unacked list — durable state
+
+The awaiting-acknowledgment list lives in =.ai/triage-intake-unacked.org=, so it survives a session crash, a =/clear=, or a restart — the away-from-desk case auto mode exists for. It's project-local state, tracked the same way as the sentinel (=.ai/last-triage-intake=), created on first need.
+
+Shape — one =** = heading per awaiting item:
+
+#+begin_example
+#+TITLE: Triage Intake — Responses Awaiting Acknowledgment
+# Maintained by triage-intake auto mode. One heading per item; acked items are removed.
+
+** Dana — 2pm reschedule invite
+:PROPERTIES:
+:SOURCE: personal-calendar
+:LOCATOR: <event id or thread url — the dedupe key>
+:SINCE: 2026-06-15 10:42
+:END:
+She's waiting on a yes/no to the move.
+#+end_example
+
+- *Add* — a sweep appends any new directed-at-Craig response not already listed, deduped on =LOCATOR=.
+- *Carry forward* — every sweep re-renders the full list in its second section, whether or not it changed this sweep.
+- *Ack* — "ack <item>" (e.g. "ack the Dana thread") removes that heading; "ack all" clears the list.
+- *Close* — a close empties the list as part of processing (each item is either actioned or filed).
+
+** Close and stop — the checkpoint
+
+The mutations are gated behind two commands:
+
+- *"close the triage"* — run the full close: take the accumulated mail actions, add the accumulated Action items to =todo.org= as =:quick:reactive:= tasks (asking Craig the questions a normal Phase C/D would), empty the unacked list, then *advance the sentinel* — capture the close run's Phase A timestamp, do the mutations, write that timestamp to =.ai/last-triage-intake= exactly as a normal run does (per "Capture the Phase A timestamp") — and commit + push the batch. Then *keep looping* (next sweep on the normal interval). This is the "flush the batch and carry on" checkpoint.
+- *"stop the triage"* — the same close processing, then *stop the loop* and revert to manual (on-demand) triage.
+
+A close is the only point auto mode advances the sentinel or commits. Between closes the engine state is untouched — that is what makes a 20-minute sweep cheap and non-destructive, and it preserves the engine invariant: the sentinel still means "everything before this timestamp has been scanned," it just advances once per close instead of once per run.
+
+** Why a separate mode
+
+The standard engine is one-shot and mutating — right for an at-the-desk "what's new?" glance, wrong for unattended polling: run every 20 minutes it would advance the sentinel past unprocessed items, spray reactive todos, take mail actions, and commit noise without review. Auto mode separates the cheap, frequent *watching* from the deliberate, gated *committing*, and adds the away-user's missing primitive — the running unacked-responses list.
+
+
+* Reference
+
+** Source Plugin Contract
+
+A source plugin is a file named =triage-intake.<source>.org=. The first dot after =triage-intake= is the engine/plugin boundary; the segment after it is the source id. Hyphens stay *inside* a segment (=triage-intake.personal-gmail.org= is engine =triage-intake=, source =personal-gmail=). Deeper dots (=triage-intake.<source>.<sub>.org=) are reserved for sub-adapters — YAGNI for now, but the namespace accommodates them at no cost.
+
+A plugin file declares exactly one source through a fixed shape:
+
+*Property drawer* on the top-level =* Source:= heading:
+- =ORDER= — integer. Output ordering in the per-source breakdown (lower = earlier).
+- =ENABLED= — the precondition the engine evaluates before loading the source. The source is skipped — *with an announced reason* — when it's false. Forms: =always=, a shell test (=command -v gh && gh auth status=), or =mcp <server> present=.
+- =ANCHOR= — the cutoff format this source consumes: =epoch=, =iso8601=, =day=, or =none= (state-based source with no since-window — e.g. live IMAP unread, or open-PR state). The engine computes the anchor once and substitutes it in the requested format.
+- =SUBAGENT_OVER= — integer. If the scan is expected to return more than this many items, dispatch a subagent for the source so its raw output stays out of main context. The subagent applies Phase B and returns buckets only, not the raw list.
+
+*Body sections:*
+- =** Scan= — the command(s) that fetch new/unread items since =<anchor>=, emitting raw items.
+- =** Classify= — the source's per-bucket bias and noise patterns. *Deltas* from the engine's shared four-bucket model below, not a re-derivation.
+- =** Render= — the source's block in the Phase C summary. "Omit if empty."
+- =** Actions= — the executable state-changes, one verb per line: =verb :: command template (parameterized by item id)=.
+
+Template:
+
+#+begin_example
+
+** Source: <id>
+:PROPERTIES:
+:ORDER: <n>
+:ENABLED: <precondition>
+:ANCHOR: epoch | iso8601 | day | none
+:SUBAGENT_OVER: <n>
+:END:
+
+*** Scan
+<command(s) that fetch new/unread items since <anchor>>
+
+*** Classify
+<bias + noise patterns; deltas from the shared four-bucket model>
+
+*** Render
+"<Source label> — N <unit>" block; omit if empty.
+
+*** Actions
+- <verb> :: <command, parameterized by <id>>
+#+end_example
+
+
+** Anchor: Since When?
+
+The workflow needs a "scan since" timestamp. Resolution order:
+
+1. *Sentinel file content:* first whitespace-delimited token in =.ai/last-triage-intake= is the Phase A scan-kickoff epoch from the most recent successful run (see "Capture the Phase A timestamp" below). Most accurate.
+2. *Sentinel file mtime* (back-compat): if the file exists but is empty, read its mtime — that's the older mtime-based convention that pre-dates the content-based change. Still accurate on the machine that wrote it.
+3. *Most recent prep doc:* if no sentinel content or readable mtime, anchor on the latest =daily-prep/YYYY-MM-DD-daily-prep.org= mtime.
+4. *Most recent session file:* if none of the above, anchor on the most recent =.ai/sessions/= file's mtime.
+5. *Session start:* fall back to the current session's start time. Last resort.
+
+The engine computes the anchor *once* and exposes it in every format a plugin might request (=epoch=, =iso8601=, =day=). Each plugin's =ANCHOR= field says which it consumes; the engine substitutes that form into the plugin's =<anchor>= placeholder. Sources with =ANCHOR: none= are state-based (live unread, open-PR state) and get no cutoff substituted — they report current state, and Phase B uses the anchor only to flag what's *new since* last check.
+
+*** Capture the Phase A timestamp
+
+Just before issuing the Phase A batch, capture the current epoch seconds:
+
+#+begin_src bash
+PHASE_A_TS=$(date +%s)
+#+end_src
+
+Hold this value through Phases B, C, and D. At end of run, *write* the captured timestamp into the sentinel's content (not its mtime):
+
+#+begin_src bash
+echo "$PHASE_A_TS $(date -d "@$PHASE_A_TS" '+%Y-%m-%d %H:%M:%S %z')" > .ai/last-triage-intake
+#+end_src
+
+The file ends up with a single line like =1778683109 2026-05-13 09:38:29 -0500= — epoch first (machine-readable, parsed by reading the first token), human-readable timestamp second.
+
+*Why content, not mtime:* the sentinel is checked into git. Git tracks content, not mtime, so an mtime-based sentinel is per-machine: one machine's anchor stays on that machine; a fresh clone gets the file but the mtime is whenever the clone happened, not the actual triage time. Writing the epoch as content means the anchor travels with the repo and stays accurate after a fetch + pull on any machine.
+
+*Why Phase A and not end-of-run:* Phase A runs at one moment, but Phases B-D may take 5-30 minutes. Items posted to any source /during/ Phases B-D land between the Phase A scan time and the eventual end-of-run time. If the sentinel were set to the end-of-run time, those items would silently fall through the cracks: the next triage's Phase A would skip the gap window and never see them. Anchoring the sentinel to Phase A's scan time guarantees the next run's window starts where this run's window ended, with zero gap.
+
+*** Reading the sentinel
+
+When the workflow needs the anchor at the start of a new run:
+
+#+begin_src bash
+# Content-first, mtime-fallback.
+ANCHOR_EPOCH=$(awk 'NR==1 {print $1; exit}' .ai/last-triage-intake 2>/dev/null)
+if [ -z "$ANCHOR_EPOCH" ] && [ -f .ai/last-triage-intake ]; then
+ ANCHOR_EPOCH=$(stat -c %Y .ai/last-triage-intake)
+fi
+#+end_src
+
+If both fail, fall through to the resolution order above (prep doc → session file → session start).
+
+
+** Output Template
+
+The summary follows this shape (deltas only: a source with no changes gets no block; when *nothing* changed anywhere, the whole summary collapses to the one-line form below — plus any scan-failure banners and the suggested-actions line if actions are queued):
+
+#+begin_example
+17:39 sweep: no changes
+#+end_example
+
+When there are changes, render one block per changed source in =ORDER=, using each plugin's =Render= shape:
+
+#+begin_example
+**Anchor:** <previous run timestamp> → now (<elapsed> elapsed)
+**Loaded:** <general plugins> + <project plugins> (skipped: <disabled, with reason>)
+
+**Top signals to act on:**
+1. <terse Action description with link>
+2. ...
+
+<one block per loaded source, in ORDER — see each plugin's Render>
+
+**Suggested actions:**
+- Trash N noise items
+- Mark-read M keep items
+- Respond to <invite>
+- Merge PRs #X and #Y
+- ...
+#+end_example
+
+Order matters: top-signals first because that's what Craig reads in 30 seconds between meetings. Per-source detail second. Suggested actions last because they require a decision.
+
+
+** Common Mistakes
+
+1. *Globbing only =.ai/workflows/= and missing the project plugins.* The single most damaging failure mode — the sweep runs with half its sources and the omission is invisible (a missing source looks identical to a quiet one). Phase 0 globs *both* =.ai/workflows/triage-intake.*.org= and =.ai/project-workflows/triage-intake.*.org=, every run, and announces the loaded set.
+2. *Running Phase A sequentially.* Send every enabled source's scan in one message — the whole point is parallelism.
+3. *Wiring a source into the engine.* Sources live in plugin files, never here. If you find yourself editing this file to add an account, repo, or channel, stop — write or edit a =triage-intake.<source>.org= plugin instead.
+4. *Executing actions without explicit confirmation.* Phase D runs only after Craig says "yes" or picks a subset.
+5. *Forgetting to set the sentinel at the end.* Without it, the next run re-scans the same window.
+6. *Using mtime instead of content for the sentinel.* Plain =touch= writes /now/ to mtime, stranding items posted between Phase A and end of run. =touch -d "@$PHASE_A_TS"= fixes the time but mtime is per-machine — git tracks content, not metadata, so the anchor doesn't survive a clone or cross-machine sync. Always write the epoch into the file's *content*.
+7. *Running this alongside daily-prep.* Daily-prep already does this as Phase 3 — don't duplicate.
+8. *Mixing Action and FYI in the top-signals list.* Top signals = Action only. FYI lives in the per-source detail.
+9. *Reporting a failed or skipped scan as a quiet source.* A hung receive, a dead daemon, or a skipped spin-up looks identical to "no new messages" in the output unless it's flagged. The 2026-06-10 sweep shipped with Signal silently missing because the scan hung on an account lock. Failures lead the summary, in their own banner line.
+10. *Rendering a per-source quiet roll-call.* "Calendar — quiet" / "PRs — nothing new" lines on every silent source bury the one change that matters and pad a no-change sweep into a report. Deltas only: changed sources get blocks, unchanged sources get nothing, and an all-quiet sweep is one line (Craig's 2026-06-11 ruling in Phase C).
+
+
+* History / Design Notes
+
+** Living Document
+
+Update the engine as the orchestration pattern evolves; update a plugin as its source evolves. Source-specific learnings belong in the plugin's own file, not here.
+
+*** Updates and Learnings
+
+**** 2026-06-15: Auto mode (unattended monitoring)
+Added a self-running mode for when Craig is away but wants tight awareness — a =/loop= in the live session running accumulate-don't-mutate sweeps with "close the triage" / "stop the triage" as the gated checkpoint. Born the morning Craig cleared his day for a family emergency and wanted the desk watched while in and out. Design decisions (work-project proposal, ratified by Craig 2026-06-15): the unacked-responses list is durable in =.ai/triage-intake-unacked.org= (survives a crash/clear, the away-from-desk case it exists for); the sentinel advances only at close, preserving the scanned-before invariant; delivery is an in-session loop so MCP auth is inherited (a detached cron schedule belongs to the morning-ops orchestrator, not here, because of the headless-auth wall); it stays a mode of this engine, distinct from but reusable by that orchestrator. Same-day addendum (work, 2026-06-15): each sweep ends with a date/time/timezone stamp on its own final line (printed on quiet sweeps too, as proof the loop ran) so an away reader gauges freshness at a glance.
+
+**** 2026-05-01: Initial creation
+Extracted from daily-prep's Phase 3 pattern as a standalone, lightweight, between-meetings sweep.
+
+**** 2026-05-07: Anchor the sentinel to Phase A scan time, not run-end time
+Gap-window bug: a run had Phase A fire at 13:35 and the sentinel set at 15:04, so an item posted at 14:20 would be skipped by the next run (the sentinel claimed everything before 15:04 was scanned when Phase A only reached 13:35). Fix: capture =PHASE_A_TS= just before Phase A, hold it through B-D, write it to the sentinel at end of run. The sentinel means "everything before this timestamp has been scanned," the only invariant that prevents items falling through the cracks.
+
+**** 2026-05-13: Move the sentinel from mtime to content (cross-machine survivability)
+The sentinel is checked into git, but git tracks content, not mtime — so an mtime anchor is per-machine. Fix: write the captured epoch into the sentinel's content (=EPOCH ISO-8601=), read with =awk 'NR==1 {print $1}'=, mtime as back-compat fallback.
+
+**** 2026-06-11: Deltas-only reporting (Phase C + Output Template + Common Mistake 10)
+Craig, via the work project's same-day handoff: "we only need to report if anything's changed when we do triage intake." Sweep summaries report deltas only — a new invite, a new/moved/cancelled event, a new message needing attention. Unchanged sources get no block (the "Calendar — quiet" roll-call is retired), and an all-quiet sweep renders as a single "HH:MM sweep: no changes" line. Failures keep their loud banner (never folded into the no-change line) and the suggested-actions line stays when actions are queued. Same ruling: the telegram plugin's dev-community group traffic is dropped from reports entirely unless Craig asks (see that plugin's 2026-06-11 note).
+
+**** 2026-06-10: Loud failure surfacing (Phase C item 0 + Common Mistake 9)
+Craig: "highlight any failures in daily triage loudly. I get important communication from all these channels." Trigger: the 2026-06-10 sweep shipped with Signal silently missing — a standalone receive hung on the account lock while the signel daemon owned it, and the failure looked identical to a quiet source. Failures now lead the summary in a ⚠ SCAN FAILED banner; the telegram plugin's failure path points at this rule.
+
+**** 2026-05-26: Refactor into engine + source plugins
+Split the monolithic workflow into a source-agnostic engine (this file) and per-source plugins named =triage-intake.<source>.org=. The engine carries the anchor/sentinel logic, the four-bucket model, the Phase A-D orchestration, the todo.org persistence convention, and the exit criteria. Each source's scan/classify/render/action knowledge moved to its own plugin. General plugins (personal-gmail, personal-calendar, cmail, github-prs) live in =.ai/workflows/= and are template-synced; project-specific plugins (a work project's Linear, work Gmail, work Slack, enterprise PRs) live in the project's =.ai/project-workflows/= and are never synced. Phase 0 globs *both* directories — the loud requirement, because missing the project dir silently halves the sweep. Naming convention: first dot is the engine/plugin boundary, deeper dots reserved for sub-adapters. This removed all DeepSat/Linear specifics from the engine; they become work-project plugins.
+
diff --git a/docs/design/2026-06-21-anki-titlefix-proposal.org b/docs/design/2026-06-21-anki-titlefix-proposal.org
new file mode 100644
index 0000000..08b8c13
--- /dev/null
+++ b/docs/design/2026-06-21-anki-titlefix-proposal.org
@@ -0,0 +1,57 @@
+#+TITLE: Proposal — flashcard-to-anki.py deck name should come from #+TITLE
+
+From: home session, 2026-06-21. Two attached files are the edited
+canonical scripts (flashcard-to-anki.py + its test). Applied locally in
+home as a stopgap; this is the durable proposal for the rulesets
+canonical. Please reconcile and re-sync.
+
+* The bug (longstanding)
+
+flashcard-to-anki.py's default_deck_name returned input_path.stem (the
+filename), so every deck generated through flashcard-sync (which passes no
+--deck) was named after the file, e.g. "personal-drill" / "health-drill"
+/ "kit", not the curated #+TITLE.
+
+flashcard-review.org already documents the intended behavior: "The
+#+TITLE line drives ... the Anki deck name on the phone" and "derives the
+Anki deck ID from the deck name." The script never matched the doc.
+deepsat only looked correct because its first run used an explicit
+--deck "DeepSat Flashcards".
+
+* The fix
+
+default_deck_name(input_path, org_text) now scans for a #+TITLE: line
+(case-insensitive, surrounding whitespace trimmed) and returns it; falls
+back to input_path.stem when there's no non-empty #+TITLE. main() passes
+the already-read org_text. Help text + module docstring updated.
+
+TDD: the two old deck-name tests asserted the buggy basename behavior —
+rewrote them. New tests cover title-driven naming, trimming,
+case-insensitive #+title, basename fallback (no title), and basename
+fallback (blank title). Full file: 29 pass.
+
+No companion script changes needed: flashcard-sync passes no --deck so it
+picks up the new default automatically, and flashcard-stats.py already
+reads #+TITLE. flashcard-review.org needs no change (the script now
+matches what it already says).
+
+* Migration caveat (worth a line in the doc if you want)
+
+Deck ID derives from the deck name, so this fix changes the ID for any
+deck previously generated without --deck. On next import those land as
+new decks; the old basename-named decks keep their review history and
+must be deleted by hand. The workflow's existing "Stable-ID caveat"
+already covers the mechanics. In home this affected personal-drill,
+health-drill, kit (regenerated this session as Personal / Health / KIT,
+with titles also stripped of "Flashcards"/"Drill" per Craig). deepsat is
+unaffected (already title-named).
+
+* Related idea (separate, not in these files) — apkg → org-drill converter
+
+deepsat-fundamentals.apkg (100-card DeepSat subset, made once with
+--deck "DeepSat Fundamentals") has no saved .org source anywhere. Craig
+wants an apkg → org-drill converter — the inverse of flashcard-to-anki.py
+— to recover orphaned decks and pull phone-authored cards back into the
+org source-of-truth. Flagging as a candidate rulesets tool alongside the
+flashcard-* family; deepsat-fundamentals is the concrete first use case.
+Not built yet; raising for the backlog.
diff --git a/docs/design/2026-06-21-apkg-to-orgdrill-buildreq.org b/docs/design/2026-06-21-apkg-to-orgdrill-buildreq.org
new file mode 100644
index 0000000..37a866f
--- /dev/null
+++ b/docs/design/2026-06-21-apkg-to-orgdrill-buildreq.org
@@ -0,0 +1,68 @@
+#+TITLE: Build request — apkg → org-drill converter (inverse of flashcard-to-anki.py)
+
+From: home session, 2026-06-21. Craig wants this built (backlogged, not
+urgent). Standalone build request — the earlier anki-title-fix-proposal
+only mentioned it in passing; this is the real ask.
+
+* Why
+
+The flashcard pipeline is one-directional (org-drill → apkg). Decks
+authored or curated on the phone, and orphaned apkgs whose .org source
+was never saved, can't get back into the org source-of-truth. Concrete
+case: deepsat-fundamentals.apkg — a 100-card DeepSat subset generated
+once with --deck "DeepSat Fundamentals" — has no .org source anywhere on
+ratio, velox, or in work git history. The converter recovers it and makes
+phone → org round-tripping possible.
+
+* What — contract (inverse of flashcard-to-anki.py)
+
+Input: an Anki =.apkg= (a zip containing collection.anki2 / .anki21
+sqlite, plus a media blob).
+Output: an org-drill =.org= file in the house canonical shape that
+flashcard-stats.py / flashcard-to-anki.py already agree on.
+
+Mapping (mirror flashcard-to-anki.py's parse/build):
+- Deck name (from the apkg) → =#+TITLE:=.
+- Each note → =** <Front> :drill:= with the Back as the body.
+- Card tag → top-level =* Section= grouping (inverse of section_to_tag;
+ cards sharing a tag collect under one section; the slug won't round-trip
+ to the exact original section title, so this is best-effort — emit the
+ tag as the section heading and let a human retitle).
+- Back HTML → org: convert =<br>= back to newlines; unescape
+ =&amp;/&lt;/&gt;=; strip the =<hr id="answer">= the card template adds
+ (the Back field itself shouldn't contain it, but guard anyway).
+- Generate a fresh =:ID:= UUID per card in a =:PROPERTIES:= drawer so the
+ output is immediately org-drill-valid and round-trips back through
+ flashcard-to-anki.py. (Note: GUIDs in flashcard-to-anki.py are derived
+ from the front text, not the :ID:, so a regenerated apkg still matches
+ existing phone cards by front — call that out in the docstring.)
+
+Edge cases to cover in tests (Normal/Boundary/Error):
+- Multiple decks in one apkg (emit one file per deck, or error asking for
+ a deck filter — pick one and document it).
+- Notes with multiple fields / non-basic note types (the pipeline only
+ models Front/Back — skip or warn on others, don't silently drop).
+- HTML entities, embedded =<br>=, and any =Source:= footer surviving
+ round-trip.
+- Empty back; media references (flag, since org side has no media path).
+- collection.anki2 vs .anki21 schema differences.
+
+* Where it lives
+
+Rulesets-owned, beside the flashcard-* family
+(=claude-templates/.ai/scripts/=): suggest =anki-to-flashcard.py= (or
+=apkg-to-orgdrill.py= — your naming call). Add tests under
+=scripts/tests/=. A new file can't be built downstream — home/.ai/scripts/
+is wiped to match the template by the startup =--delete= rsync — so this
+has to be built in the rulesets canonical. PEP 723 uv-run script like its
+sibling; genanki isn't needed for reading (stdlib =zipfile= + =sqlite3=
+suffice), so it has no runtime deps.
+
+* Acceptance
+
+Round-trip test: take a known org-drill source, run it through
+flashcard-to-anki.py, run the result back through this converter, and
+assert the cards (front/back/section) match the original (modulo
+regenerated :ID:s and best-effort section titles). Plus: run it on the
+real deepsat-fundamentals.apkg and hand the recovered .org back so its
+source can be filed (work project).
diff --git a/docs/design/2026-06-21-flashcard-stats-refutation-proposal.org b/docs/design/2026-06-21-flashcard-stats-refutation-proposal.org
new file mode 100644
index 0000000..bbbe175
--- /dev/null
+++ b/docs/design/2026-06-21-flashcard-stats-refutation-proposal.org
@@ -0,0 +1,57 @@
+#+TITLE: Proposal — flashcard-stats.py refutation / claim-prompt mode
+
+From: home session, 2026-06-21. Backlog, not urgent. Relates to the
+refutation-drill deck being built in the home project.
+
+* Problem
+
+A new card family doesn't fit the linter: the *refutation / claim-prompt*
+card. Its heading is a bare false claim ("The earth is flat.") and its
+body is the rebuttal. This is a legit org-drill simple card (org-drill is
+happy), but flashcard-stats.py — built for Q&A decks — trips two BLOCKING
+checks on every such card, both false positives:
+
+- *non-prompt heading*: a declarative claim has no '?' and no
+ imperative verb, so it reads as "topic-as-heading not yet rewritten".
+ But for this family the declarative claim IS the intended prompt.
+- *answer leakage*: the claim's words necessarily reappear in the
+ refutation, so front/back overlap is high. But the answer (the rebuttal)
+ is not given away by the claim — there's no actual leakage.
+
+Concrete: the home refutation-drill.org (6 cards) reports 6 non-prompt
+headings + 1 leakage WARN, so flashcard-sync's gate blocks it entirely.
+The deck currently has to be generated with the flashcard-to-anki.py
+override, losing the safety net.
+
+* Proposed fix
+
+A per-deck opt-in marker that switches the two checks off for that file
+only. Two options (your call):
+
+1. A file-level keyword: =#+DECK_KIND: refutation= near the top. When
+ present, flashcard-stats skips the non-prompt-heading check and the
+ answer-leakage check for the whole file (keeps the others:
+ missing-:ID:, *** Answer sub-headers, duplicate fronts, the
+ non-blocking NOTEs).
+2. A per-card tag: cards tagged =:claim:= (alongside =:drill:=) are
+ exempted from those two checks individually.
+
+Option 1 is simpler and matches how this deck works (the whole file is
+one family). Option 2 is finer-grained if a deck ever mixes families.
+
+Either way: document the new card family in flashcard-review.org (a
+"Refutation / claim-prompt cards" subsection under Canonical Card Shape —
+heading is the bare claim, body is snap-response + backups + named-fallacy
++ restate, Source footer), and note that flashcard-sync then works
+normally on these decks.
+
+* Affected files
+- =flashcard-stats.py= — the check skip + (option 1) keyword parse / (option 2) tag check.
+- =flashcard-review.org= — document the family + the marker.
+- =flashcard-to-anki.py= / =flashcard-sync= — no change needed (they don't gate on heading form).
+- Tests: add cases for a refutation-marked file passing despite declarative headings + claim/answer overlap.
+
+* Companion context
+The home deck's card format and the org-drill-fine / Anki-linter-fights
+finding are written up in home:refutation-drill-sources.org (Tooling
+note). The override command is documented there too.
diff --git a/docs/design/2026-06-21-host-identity-guard-proposal.org b/docs/design/2026-06-21-host-identity-guard-proposal.org
new file mode 100644
index 0000000..f389825
--- /dev/null
+++ b/docs/design/2026-06-21-host-identity-guard-proposal.org
@@ -0,0 +1,54 @@
+#+TITLE: From archsetup — hardcoded machine identity in CLAUDE.md (consider fleet-wide)
+#+DATE: 2026-06-21
+
+* What we did
+
+Built a Super+F Dirvish popup in the archsetup/dotfiles + .emacs.d projects,
+modeled on the existing Super+Shift+N org-capture popup (launcher script names an
+emacsclient frame, Hyprland window rules float it, an Emacs command runs in the
+frame and q closes it). Cross-project: dotfiles half committed from archsetup,
+Emacs half handed off to .emacs.d's inbox.
+
+* The bug it surfaced
+
+While stowing on this machine, =make stow hyprland= pulled the *velox* host tier,
+and =uname -n= returned =velox=. But archsetup's CLAUDE.md asserted, as a fixed
+fact, "This machine is **ratio**." It was simply wrong on velox — a stale
+identity baked into a per-project doc that travels to every machine via git.
+
+I'd been reasoning from that line all session (e.g. "the touchpad-auto reminder
+is velox-only, and we're on ratio, so skip it") — exactly backwards. A hardcoded
+"this machine is X" in a synced/tracked project file is a latent trap on any
+multi-machine setup: the file is identical on every host, so the claim is false
+on every host but one.
+
+* The fix (this project)
+
+Replaced the fixed identity with a runtime instruction. The attached CLAUDE.md
+now reads, in the Notes section:
+
+ Never assume which machine this is — always run =uname -n= to find the hostname
+ (the =hostname= binary is absent, so =uname -n= is the source of truth;
+ =uname -r= is the kernel release, not the host). The fleet is ratio
+ (workstation) and velox (laptop), both Hyprland (Wayland)...
+
+(Craig initially said =uname -r=; that's the kernel release. =uname -n= is the
+nodename/hostname, which is what the stow host-tier logic already keys on.)
+
+* Why this is a rulesets concern
+
+This isn't an archsetup-only quirk. Any project whose CLAUDE.md / notes get
+synced or cloned across machines can hardcode environment identity — current
+host, current OS, "the laptop", an IP, a display name — and be wrong everywhere
+the doc lands but the origin. rulesets governs how every project's CLAUDE.md and
+rules are shaped, so it's the right layer to consider a general guard:
+
+- A rule (claude-rules) along the lines of: don't assert mutable
+ environment/host identity as a fixed fact in a tracked/synced project file;
+ derive it at runtime (=uname -n= for host, etc.) and name the command.
+- Possibly a startup or codify-time lint that flags "this machine is <name>" /
+ "the current host is" style claims in CLAUDE.md.
+
+Sending the edited CLAUDE.md (attached separately) plus this note so the rulesets
+session can decide whether to codify the broader pattern. Proposal, not a
+directive — your value gate applies.
diff --git a/docs/design/2026-06-22-inbox-zero-capture-hardening.org b/docs/design/2026-06-22-inbox-zero-capture-hardening.org
new file mode 100644
index 0000000..69acf94
--- /dev/null
+++ b/docs/design/2026-06-22-inbox-zero-capture-hardening.org
@@ -0,0 +1,39 @@
+#+TITLE: inbox-zero Phase D wedges live org-capture sessions on the roam inbox
+
+* The bug
+
+Phase D of =inbox-zero.org= removes claimed items from =~/org/roam/inbox.org= by editing the file on disk (Edit / sed / Write). That collides with any live org-capture session Craig has open against the same file.
+
+org-capture works through an *indirect buffer* cloned from the target file. When the inbox-zero disk write lands and Emacs reverts the main =inbox.org= buffer underneath, the indirect capture buffers are left pointing at stale state and wedge — they can no longer finalize cleanly with =C-c C-c=. The visible symptom is org-capture failing, and one or more orphaned =CAPTURE-*inbox.org= / =CAPTURE-N-inbox.org= buffers piling up as Craig retries.
+
+Hit live on 2026-06-22 during a home-session inbox-zero: I filed three home items, wrote =inbox.org= on disk, and Craig's open capture wedged, leaving two orphaned =CAPTURE-inbox.org= buffers. No data was lost that time (the orphaned buffers held only existing file content, not a freshly-typed item), but that was luck — had he typed an item into the capture before it wedged, finalizing the stale buffer afterward would have written it back against the post-edit file and could have clobbered the routing or a foreign item.
+
+* Why it's worth fixing in the canonical
+
+=inbox-zero.org= is a rulesets-owned synced workflow that runs in every project (startup nudge, wrap-up sub-step, on demand), and the roam inbox is the single shared file all of them edit. Craig edits that same file live in Emacs and captures into it constantly. So the collision window recurs in every project, every session — not a home-only quirk. A local fix in home gets reverted by the next template sync, so the durable fix has to land in rulesets.
+
+* Proposed fix (recommended: guard before the disk write)
+
+Add a pre-edit guard to Phase D, before removing any claimed items:
+
+1. If Emacs is reachable (=emacsclient -e t= succeeds), check for live capture buffers targeting the roam inbox:
+
+ #+begin_src bash
+ emacsclient -e '(mapcar #(quote buffer-name)
+ (seq-filter (lambda (b) (string-match-p "CAPTURE.*inbox" (buffer-name b)))
+ (buffer-list)))' 2>/dev/null
+ #+end_src
+
+2. If any =CAPTURE-*inbox*= buffer exists, *stop before editing* and surface it: "You have a live org-capture session open against the roam inbox — finalize (=C-c C-c=) or abort (=C-c C-k=) it before I route items, otherwise the edit will wedge the capture." Resume Phase D only once it's clear. This mirrors the existing pull-before-edit / surface-and-stop discipline already in Phase D.
+
+3. Independently, when Emacs has =inbox.org= open and *unmodified* (the common case, no live capture), the disk edit is benign — Emacs reverts a clean buffer without complaint. Optionally trigger an explicit =revert-buffer= via emacsclient afterward so the buffer is immediately consistent rather than lazily on next focus.
+
+* Alternative (heavier): do the removal through Emacs when it's running
+
+Instead of editing on disk, when Emacs is reachable, perform the claimed-item removal inside the running daemon (find the buffer, delete the items, save), and fall back to the disk edit only when Emacs isn't running. This keeps Emacs's buffer authoritative and sidesteps the disk/buffer divergence entirely. It's more code and more failure surface for arbitrary item removal, so I'd lean on the guard above unless you want the stronger guarantee.
+
+* Note for whoever builds it
+
+The =emacs.md= rule already covers "don't make Craig restart Emacs; push changes into the running daemon." This is the same principle one layer out: don't edit a file *on disk* that the running daemon is actively editing/capturing into. Worth a line in =emacs.md= too, or at least a cross-reference from inbox-zero Phase D.
+
+Origin: home, 2026-06-22.
diff --git a/docs/design/2026-06-23-install-lang-claude-md-gap.org b/docs/design/2026-06-23-install-lang-claude-md-gap.org
new file mode 100644
index 0000000..cf16256
--- /dev/null
+++ b/docs/design/2026-06-23-install-lang-claude-md-gap.org
@@ -0,0 +1,31 @@
+#+TITLE: install-lang CLAUDE.md gap — non-elisp projects get a wrong or missing CLAUDE.md
+#+DATE: 2026-06-23
+
+Surfaced while running the archangel .ai/ conversion you sent (the 2026-06-20 handoff). archangel is a bash project — 437 =.sh= files; the only =.el=/=.py= in the tree are under =work/x86_64/airootfs/= (archiso staging, not source). Per your handoff I installed both elisp and python bundles. The result exposed two coupled issues that block CLAUDE.md consistency across projects.
+
+* The two findings
+
+1. *Only the elisp bundle ships a CLAUDE.md template.* =languages/elisp/CLAUDE.md= exists; =languages/python/=, =go/=, =typescript/= ship none. =install-lang.sh= guards on =[ -f "$SRC/CLAUDE.md" ]=, so a bundle without a template silently contributes nothing — no line printed, no file seeded.
+
+2. *No shell/bash bundle exists* (only elisp, go, python, typescript). archangel and archsetup are bash projects with no bundle that fits.
+
+* The consequence
+
+- Install python (or go/ts) alone → project gets *no* CLAUDE.md.
+- Install elisp + anything → project gets the elisp stub, whose first line is "Elisp project." Because install-lang seeds CLAUDE.md only on first install and never overwrites without FORCE=1, install order doesn't matter — the elisp template is the only one available, so it always wins.
+- Net: archangel, a bash project, ended up with a CLAUDE.md headed "Elisp project." An inaccurate CLAUDE.md is worse than none — it mislabels the project for every future session.
+
+* Proposals (rulesets' call)
+
+1. *Add a shell/bash language bundle.* This is the real gap for archangel/archsetup and any other shell-heavy project.
+2. *Give every bundle its own CLAUDE.md template*, or ship a language-neutral default so install-lang always seeds an accurate (or at least non-misleading) header. A stub that says "<LANG> project — customize this" is only safe when the bundle actually matches the language.
+3. *Consider the multi-bundle case* — when a project installs more than one bundle, the CLAUDE.md "Project" line shouldn't hardcode a single language picked by which template happened to exist.
+
+* Companion files to reconcile
+
+- =scripts/install-lang.sh= — the seed-on-first-install / no-overwrite logic (sections 3 and 3b) is correct; the gap is the missing templates and missing bash bundle, not this logic.
+- =languages/elisp/CLAUDE.md= — the only template today; pattern to replicate per language.
+
+* What archangel did locally (stopgap)
+
+Installed both bundles as you asked; the generic =.claude/rules/= and gitignore hygiene are the real gain there. I flagged the elisp-stub mismatch to Craig and offered to hand-write archangel's CLAUDE.md as a bash ISO-build project. That local fix doesn't address the cross-project pattern — hence this note.
diff --git a/docs/design/2026-06-23-wrap-teardown-shutdown-proposal.org b/docs/design/2026-06-23-wrap-teardown-shutdown-proposal.org
new file mode 100644
index 0000000..a47aa2d
--- /dev/null
+++ b/docs/design/2026-06-23-wrap-teardown-shutdown-proposal.org
@@ -0,0 +1,124 @@
+#+TITLE: Proposal — wrap-it-up teardown + "wrap it up and shutdown" variant
+
+* Source
+
+Raised by Craig in a home-project session, 2026-06-23, after talking the
+design through. Two related additions to =wrap-it-up.org=. Both touch the
+Claude-session lifecycle (workflow + hook + the =ai-term= buffer/tmux pair),
+so they're rulesets — with one companion piece that has to live in
+=.emacs.d/modules/ai-term.el= (flagged below). Originally floated as an
+archsetup task; archsetup owns the Hyprland/waybar layer, not the
+Claude-session lifecycle, so it was re-routed here.
+
+* Architecture this depends on (so the design is grounded)
+
+- =ai-term.el= (=.emacs.d=) is the in-Emacs launcher: a vertical-split vterm
+ buffer running a tmux session named =aiv-<project-basename>= (prefix
+ =aiv-=). Layering: =claude= process → tmux session =aiv-<proj>= → Emacs
+ vterm buffer.
+- Killing the tmux session takes the =claude= process with it, so "quit
+ Claude Code" is a *consequence* of killing =aiv-<proj>=, not a separate
+ step.
+- Hooks already exist under =~/.claude/hooks/= (e.g. =session-clear-resume.sh=,
+ =precompact-priorities.sh=) — the teardown trigger fits that pattern.
+- =sudo= is =NOPASSWD: ALL= on Craig's machines, so =sudo shutdown now= runs
+ unattended.
+
+* Item 1 — wrap-up also removes the buffer, quits Claude, removes the tmux session
+
+Recommend: yes, with one structural rule — the wrap-up runs *inside* the
+things it tears down, so teardown is self-terminating and must be the last,
+decoupled action, or the valediction may not flush before the session dies.
+
+Design:
+1. *Teardown lives in =ai-term.el=* (companion, see below): one function
+ =cj/ai-term-quit= that kills the =aiv-<proj>= tmux session (takes =claude=
+ with it), kills the vterm buffer, and restores the saved window geometry —
+ =ai-term.el= already owns the buffer↔session pair and the geometry logic.
+2. *Trigger from a Stop / SessionEnd hook, not inline.* Wrap-up does all its
+ git/archive work, delivers the valediction, then drops a sentinel (flag
+ file, e.g. =/tmp/ai-wrap-teardown-<session>=). The hook fires when Claude
+ finishes, sees the sentinel, and runs =cj/ai-term-quit= via =emacsclient=.
+ Decoupling guarantees the valediction lands before the session dies.
+3. *Gate on commit+push verified* — never tear down before the session record
+ is pushed (wrap-up's existing Step 4 / validation checklist already
+ enforces push; teardown is strictly after it).
+4. *Phrase split — teardown IS the default* (Craig's decision 2026-06-23).
+ Bare "wrap it up" does the full wrap AND removes the buffer/session/quits —
+ that's his typical case. The non-destructive variant gets the explicit
+ qualifier: "wrap it up with summary" summarizes + commits + pushes +
+ archives but keeps the buffer (no teardown), so the summary stays readable.
+ So: "wrap it up" → teardown; "wrap it up with summary" → no teardown;
+ "wrap it up and shutdown" → wrap + poweroff (supersedes teardown, Item 2).
+
+* Item 2 — "wrap it up and shutdown": 10-count then =sudo shutdown now=
+
+Recommend: yes, but the safety gate is load-bearing and the countdown has a
+rendering gotcha.
+
+Design:
+1. *"Only ai-term left" = hard blocking precondition*, evaluated BEFORE the
+ countdown. Count live sessions (=tmux ls | grep '^aiv-'= or
+ =pgrep -fc claude=). If more than this one is alive, ABORT the shutdown,
+ list what's running, and fall back to a normal wrap. Never power the box
+ off out from under another active Claude session. This is the most
+ important part of the item.
+2. *The live countdown can't run through Claude's tool output.* The Bash tool
+ buffers stdout until the command returns, so a =for i in $(seq 10 -1 1);
+ sleep 1= prints all ten at once at the end, not one per second. It has to
+ run detached or in Emacs:
+ - tty writer: =for i in $(seq 10 -1 1); do printf '\rShutting down in %2d…'
+ "$i" > /dev/tty; sleep 1; done; sudo shutdown now= (backgrounded), or
+ - an Emacs =run-at-time= timer printing 10→1 in the echo area, then
+ =(shell-command "sudo shutdown now")=.
+3. *Make it abort-able* (Ctrl-C / keypress cancels). A 10-second countdown's
+ whole purpose is a last-chance window; a non-cancellable one is just a
+ delay.
+4. *Sequencing.* "...and shutdown" supersedes Item 1's teardown — if the box
+ is powering off, killing the buffer/session first is moot. Wrap (commit +
+ push + archive) → session-count gate → countdown → =shutdown=.
+
+Packaging: a small rulesets bin script (e.g. =ai-wrap-shutdown=) doing the
+gate → abort-able countdown → shutdown, invoked by the workflow after the wrap
+commit/push. Countdown either in that script (tty) or handed to Emacs.
+
+* Companion — required change in =.emacs.d/modules/ai-term.el=
+
+Item 1's teardown function =cj/ai-term-quit= must live in =ai-term.el= (it
+owns =aiv-<proj>= session naming, the vterm buffer, and geometry restore).
+rulesets owns the workflow + hook + bin script that *call* it; =.emacs.d= owns
+the function itself. Spec for the =.emacs.d= side:
+
+- =cj/ai-term-quit (&optional project)= — resolve the =aiv-<basename>= session
+ for the current/!named project, =tmux kill-session= it, =kill-buffer= the
+ associated vterm buffer, restore saved geometry. Idempotent / no-op if the
+ session or buffer is already gone. Callable from =emacsclient -e= so the
+ Stop hook can invoke it headlessly.
+- (Optional) a count helper =cj/ai-term-live-count= so the Item-2 gate can ask
+ Emacs how many ai-term sessions are live, as an alternative to =tmux ls= /
+ =pgrep=.
+
+When rulesets builds the workflow/hook side, route this companion to
+=.emacs.d= (inbox-send) so the two land together.
+
+* Open decisions for Craig
+
+- Phrase set: DECIDED (2026-06-23) — "wrap it up" tears down (default);
+ "wrap it up with summary" wraps without teardown; "wrap it up and shutdown"
+ is the poweroff variant. Remaining nuance: confirm the exact non-destructive
+ qualifier wording is "with summary" (vs e.g. "and summarize").
+- Countdown home: tty-writer bin script vs Emacs timer. (Emacs timer reads
+ cleaner inside the vterm and is trivially abort-able.)
+- Session-count mechanism for the gate: =tmux ls=, =pgrep claude=, or
+ =cj/ai-term-live-count=.
+
+* Verify
+
+- Item 1: bare "wrap it up" → valediction renders fully, THEN buffer +
+ =aiv-<proj>= session + claude all gone, geometry restored; "wrap it up with
+ summary" → wrap completes but the buffer stays intact (no teardown).
+- Item 2 gate: with a second =aiv-*= session alive, "wrap it up and shutdown"
+ refuses, lists the other session, and does a normal wrap (no poweroff).
+- Item 2 happy path: sole session → 10→1 renders one-per-second, is
+ cancellable, then =shutdown= fires.
+- Teardown never runs before commit+push is verified.
diff --git a/docs/design/2026-06-27-bug-priority-matrix-cover-note.org b/docs/design/2026-06-27-bug-priority-matrix-cover-note.org
new file mode 100644
index 0000000..d6398e0
--- /dev/null
+++ b/docs/design/2026-06-27-bug-priority-matrix-cover-note.org
@@ -0,0 +1,5 @@
+#+TITLE: Cover note for the bug-priority-matrix-proposal.md just sent
+#+SOURCE: from emacs-wttrin
+#+DATE: 2026-06-27 23:57:30 -0400
+
+Cover note for the bug-priority-matrix-proposal.md just sent. WHAT: a Severity × Frequency bug-priority matrix (P1-P4) distilled from wttrin's priority scheme today. It derives a bug's priority from two facts (severity, frequency) instead of opinion, maps each cell to a release vehicle, and — for projects running the todo-format [#A]-[#D] scheme — has bugs derive their letter from the matrix while features keep their roadmap judgment. Includes a special-category rule (privacy/security/safety graded on severity alone) and a tie-in that makes a 'no open [#A]' release gate fact-based. WHY SEND: this is a generic defect-management pattern, not wttrin-specific, so it belongs upstream to distribute with any code-centric project (a claude-rule and/or a todo.org priority-scheme template snippet — your call on placement and whether it folds into todo-format.md or stands alone). TWO ASKS: (1) consider a companion NON-CODING matrix for work-product/ops/decision defects — the same frequency×severity shape likely generalizes beyond bugs; (2) treat this as a LIVING DOCUMENT — adjust bands and wording as projects use it until it's as mature as the coding matrix. SOURCING: the matrix is a standard industry pattern; I adapted a generic version and the wttrin worked-example. Nothing work-confidential is copied — please keep the upstream artifact generic and uncited to any private reference. wttrin keeps its local copy in todo.org as the stopgap until you distribute the canonical.
diff --git a/docs/design/2026-06-27-bug-priority-matrix-proposal.md b/docs/design/2026-06-27-bug-priority-matrix-proposal.md
new file mode 100644
index 0000000..4b49769
--- /dev/null
+++ b/docs/design/2026-06-27-bug-priority-matrix-proposal.md
@@ -0,0 +1,70 @@
+# Bug Priority — Severity × Frequency Matrix (proposed rule for code projects)
+
+Applies to: code-centric projects (a `:bug:`-tracking `todo.org`, an issue tracker, or any defect backlog).
+
+Status: proposed. Distilled from wttrin's priority scheme on 2026-06-27. Generic defect-management pattern — adapt the bands per project. Treat as a living document; refine the wording and bands as projects use it, the way wttrin's matrix will mature in use.
+
+## The problem
+
+Without a systematic scheme, bugs get prioritized by who's loudest, who has the most authority, or who's most frustrated. Most bugs drift to "medium," and no one can say what medium means. Whether a bug is "high priority" stays a matter of opinion, so the release decision becomes a judgment call every time.
+
+## The approach — derive priority from two facts
+
+Priority should fall out of two relatively objective factors, not an argument:
+
+1. Severity — how bad is it when the bug occurs? (Service down / data loss / security or privacy leak at one end; cosmetic nit at the other.)
+2. Frequency — how often will a user hit it? (Every user every time at one end; rare edge case at the other.)
+
+Put severity on one axis and frequency on the other. Each cell maps to a priority level; each level maps to a release vehicle.
+
+```
+| Frequency / Severity | Critical | Major | Minor | Cosmetic |
+|------------------------+----------+-------+-------+----------|
+| Every user, every time | P1 | P1 | P2 | P3 |
+| Most users, frequently | P1 | P2 | P3 | P4 |
+| Some users, sometimes | P2 | P3 | P3 | P4 |
+| Rare edge case | P2 | P3 | P4 | P4 |
+```
+
+Response per level:
+
+- P1 / Critical — fix in the current release or patch. Blocks a release.
+- P2 / High — next patch release.
+- P3 / Medium — next major release.
+- P4 / Low — backlog, fix when convenient.
+
+(A P0 / showstopper tier — all hands, emergency release — sits above P1 for projects that need it.)
+
+## Adapt the bands per project
+
+The axis labels are generic; each project defines what Critical/Major/Minor/Cosmetic and the frequency rows mean in its own terms. wttrin's worked example:
+
+- Critical: no weather at all on the common path (crash, hard error), wrong weather shown as if correct (silent data error), or a privacy/security leak.
+- Major: a feature broken but with a workaround.
+- Minor: degraded but usable.
+- Cosmetic: a pure visual nit.
+- Frequency rows map to default-config-every-time → common-action → feature/config-specific → unusual-input/environment.
+
+## Special category — severity alone
+
+Some defects don't correlate with frequency: privacy/security leaks, compliance violations, safety issues. Grade these on severity alone — one occurrence with the right consequences is a showstopper no matter how rarely it'd be noticed. (wttrin's worked case: a home address leaked into public git history — Critical on severity alone, scrubbed 2026-06-26.)
+
+## Integrating with a project's `[#A]`–`[#D]` priority scheme
+
+Projects that already run a letter-priority scheme (the rulesets `todo-format.md` convention) keep it for features — deciding feature X is more urgent than feature Y is a legitimate roadmap judgment. Bugs are different: their letter should be derived, not argued.
+
+Rule: a `:bug:`'s letter comes from the matrix, not a choice.
+
+- P1, P2 → `[#A]` (release-gating)
+- P3 → `[#C]` (scheduled fix, has a workaround)
+- P4 → `[#D]` (backlog)
+
+(The exact letter mapping is a per-project knob; this is wttrin's.) Features keep their `[#A]`–`[#D]` judgment as a roadmap call; only bugs derive their letter from the matrix.
+
+## Tie-in to release criteria
+
+When a project gates a release on a soak or a "no open [#A]" rule, the matrix makes that gate fact-based: "no open [#A] bugs" becomes "no open P1/P2," which the matrix produces. A severity-graded soak follows naturally — a P1/P2/P3 found in the window resets the clock and must be fixed; a P4 logs to backlog without resetting.
+
+## The payoff
+
+Once each priority maps to a release vehicle, you've pre-decided most issues and only discuss the genuine edge cases. The release decision stops being a soul-search and becomes arithmetic. When there's disagreement on a bug's priority, the matrix wins.
diff --git a/docs/design/2026-06-29-green-baseline-proposal.org b/docs/design/2026-06-29-green-baseline-proposal.org
new file mode 100644
index 0000000..47de18d
--- /dev/null
+++ b/docs/design/2026-06-29-green-baseline-proposal.org
@@ -0,0 +1,72 @@
+#+TITLE: Proposal: ensure a green test run before starting work
+#+AUTHOR: Craig Jennings (via .emacs.d session)
+#+DATE: 2026-06-29
+
+* Why
+
+While running a multi-task refactor speedrun in =.emacs.d=, the very first full
+=make test= surfaced a failing test (=test-system-cmd-restart-emacs-no-service-aborts=)
+that turned out to be *pre-existing* -- it failed on clean HEAD, unrelated to
+the work. It had nothing to do with the task; it just happened to fail on this
+machine (a native-comp mock that bypasses =symbol-function= redefinition, real
+check passing because the box has =emacs.service=).
+
+Two costs landed because the red was already there when work began:
+
+1. Every later "did I break this?" suite run carried a known failure, so the
+ green bar became "only that one fails" instead of a clean pass I could read
+ at a glance. Easy to let a *new* regression hide behind the familiar red.
+2. The work assumed the tree was in a known-good state. It wasn't, and nothing
+ in the workflow forced that assumption to be checked first.
+
+The fix is cheap and general: run the suite *before* starting work, and clear
+(or explicitly triage) any failure before the work begins. A green start
+confirms we're in the known-good place we think we are, and any issue is fixed
+before it can be confused with our own changes.
+
+* Proposed change 1 -- claude-rules/verification.md
+
+Add a section (suggested placement: right after =## The Rule=, before
+=## What Fresh Means=). Proposed text, ready to paste:
+
+#+begin_example
+## Green Baseline Before Starting Work
+
+Run the test suite before you start work, not only before you finish. A clean run at the start confirms the tree is in the known-good state you assume it is, so the baseline you build on and measure your changes against is actually green.
+
+If the suite is red before you touch anything, fix or explicitly triage the failure first. A pre-existing failure left in place poisons every later "did I break this?" check: you can't separate your own regressions from the noise, and the end-of-work run stops being readable as pass/fail at a glance. Work that assumes a known-good base may also be built on a broken assumption you never saw.
+
+When a pre-existing failure genuinely can't be fixed before the work begins (out of scope, or it needs a decision), record it as a tracked task with the diagnosis and carry its name forward. The green bar for the rest of the work is then explicitly "only this known failure remains," not a silent tolerance for red.
+
+This is the start-of-work counterpart to the Before Committing gate below: one confirms the ground is solid before you build, the other confirms you didn't crack it.
+#+end_example
+
+* Proposed change 2 -- start-work skill, Pre-work phase
+
+The start-work skill already has a Pre-work phase (eligibility, fetch-and-reconcile
+against base, source-code check that the problem still exists). Add a green-baseline
+step to that phase:
+
+- Run the project's test suite before claiming the work.
+- If it's fully green, proceed.
+- If it's red, fix the failure first, or (when out of scope / needs a decision)
+ file a tracked task with the diagnosis and carry its name forward as the only
+ tolerated failure for this work.
+- Surface the baseline result so "we started from green" is on the record.
+
+This makes the verification.md principle operational at the exact moment it
+matters -- the start of a task -- the same way the Verify phase and the
+Review-and-Publish flow operationalize the end-of-work gates.
+
+* Note on why this came as a proposal, not a direct edit
+
+=.emacs.d='s =.claude/rules/*.md= are symlinks into =~/code/rulesets/claude-rules/=,
+so editing =verification.md= from the downstream session would modify the
+rulesets canonical directly. Per the cross-project rule, downstream sessions
+send rulesets the proposed change rather than editing its canonical in place.
+Hence this note instead of a local stopgap edit. The start-work skill isn't
+installed on this machine to edit anyway.
+
+The test that surfaced all this is already fixed in =.emacs.d= (commit on main:
+mock =executable-find= at the boundary instead of the helper). The durable
+process change is the part that belongs here.
diff --git a/docs/design/2026-06-29-lint-org-structural-checkers-proposal.org b/docs/design/2026-06-29-lint-org-structural-checkers-proposal.org
new file mode 100644
index 0000000..c464aca
--- /dev/null
+++ b/docs/design/2026-06-29-lint-org-structural-checkers-proposal.org
@@ -0,0 +1,55 @@
+#+TITLE: lint-org.el — four structural heading checkers org-lint doesn't cover
+
+* What changed (from .emacs.d, 2026-06-29)
+
+Added four custom judgment checkers to =lint-org.el=, following the existing
+=lo--check-tables= / =lo--check-level2-dated-headers= pattern (custom scans run
+after the org-lint pass, emitting judgment items, never auto-fixed):
+
+- =indented-heading= — a line of whitespace + stars + space OUTSIDE any block.
+ org parses a heading only at column 0, so leading whitespace silently demotes
+ a would-be heading to body text: the task vanishes from the agenda and never
+ archives. The worst defect class (an invisible task) and entirely silent
+ today. Skips indented stars inside =#+begin_/#+end_= blocks (legit content).
+- =empty-heading= — a line of bare stars with no title.
+- =malformed-priority-cookie= — a =[#x]=-shaped token org rejected (lowercase,
+ multi-char, non-letter) left stranded where a cookie would be. Checks only the
+ first cookie token per heading; skips verbatim-wrapped =[#D]= in dated-log
+ titles.
+- =level2-done-without-closed= — a level-2 DONE/CANCELLED with no CLOSED line.
+ Directly supports the todo-cleanup aging step (sent separately today): an
+ undated completed task gets force-archived immediately, so flagging it lets
+ the human add CLOSED first.
+
+Two attached files (edited canonical candidates): =lint-org.el=,
+=tests/test-lint-org.el=.
+
+* Why
+
+org-lint validates links, drawers, blocks, and babel — but NOT heading
+well-formedness. On Craig's .emacs.d todo.org a missing org-bullet in the live
+buffer prompted the question "is the file structurally okay?", and org-lint
+(even unfiltered, all checkers) reported nothing actionable. These four close
+the gap. They are general (any org file), not project-specific.
+
+* Design notes for the canonical
+
+- All four are regex-based, NOT org-element/keyword-based, so they don't depend
+ on which TODO keywords the batch Emacs happens to recognize (lint-org.el does
+ not set =org-todo-keywords=). The =level2-done-without-closed= done set is a
+ defconst =lo-done-keywords= (DONE/CANCELLED) for easy extension.
+- *Gotcha worth carrying in the canonical:* =case-fold-search= defaults to t, so
+ a naive =[A-Z]= cookie check accepts =[#a]= as valid and =\(DONE\|CANCELLED\)=
+ matches the title words "done"/"cancelled". Both letter-sensitive checkers
+ bind =case-fold-search nil=. (Caught by a failing test before it shipped.)
+- Wired into =lo-process-file= after =lo--check-level2-dated-headers=. Judgment
+ output already flows through the existing report + followups-file machinery.
+- 8 new ERT tests (good-input-silent + bad-input-flagged for each, plus
+ block-skip and verbatim-skip boundary cases). 44/44 green. Zero false
+ positives on a real 5600-line todo.org.
+
+* Note
+
+=make task-sorted= in .emacs.d now runs =lint-org.el todo.org= after the
+archive, so these checkers also gate the task-hygiene target. Makefiles aren't
+template-synced; that wiring is project-local (noted for context).
diff --git a/docs/design/2026-06-29-todo-cleanup-aging-proposal.org b/docs/design/2026-06-29-todo-cleanup-aging-proposal.org
new file mode 100644
index 0000000..5a18990
--- /dev/null
+++ b/docs/design/2026-06-29-todo-cleanup-aging-proposal.org
@@ -0,0 +1,64 @@
+#+TITLE: todo-cleanup.el — add Resolved-section file-aging to --archive-done
+
+* What changed (from .emacs.d, 2026-06-29)
+
+Extended =todo-cleanup.el='s =--archive-done= mode (the =make task-sorted=
+target) with a SECOND step, run after the existing Open Work -> Resolved move:
+
+- *Age the Resolved section.* Level-2 DONE/CANCELLED subtrees whose CLOSED date
+ is older than =tc-archive-retain-days= (default 7) — AND any with no parseable
+ CLOSED date — move out of the in-file Resolved section to =tc-archive-file=
+ (default =archive/task-archive.org= beside the todo file). Only tasks closed
+ within the last week stay in todo.org itself.
+
+Two files are attached (the edited canonical candidates):
+- =todo-cleanup.el=
+- =tests/test-todo-cleanup.el=
+
+* Why
+
+Craig's .emacs.d todo.org had grown to 768KB / 9616 lines, ~44% of it a
+243-task in-file "Resolved" section. The existing =--archive-done= only moved
+closures Open Work -> Resolved (same file), so the file grew without bound. The
+new step keeps only the last week of closed tasks in the file and sheds the rest
+to a git-tracked archive sibling. After this run: 207 aged out, todo.org
+9616 -> 5625 lines.
+
+* Design notes for the canonical
+
+- New defvars: =tc-archive-retain-days= (7; nil disables the step, preserving
+ legacy in-file-only behavior), =tc-archive-reference-date= ((YEAR MONTH DAY),
+ nil=real today — mockable for deterministic tests), =tc-archive-file= (nil =>
+ =archive/task-archive.org= beside the todo file).
+- Policy: KEEP iff CLOSED date present AND within the window (cutoff inclusive).
+ Older OR undated => archive. The undated->archive call is deliberate ("keep
+ the last week and that's it"); an earlier undated->keep version left 14 legacy
+ undated tasks behind and read as two weeks.
+- The aging step honors =--check= (previews + reports, writes nothing).
+- Report: an additive "N aged subtree(s) moved to task-archive.org" line, only
+ when N>0, so the existing real-mode-no-op silence tests are unaffected.
+- Archive file scaffold: =#+TITLE: Task Archive= / =#+FILETAGS: :archive:= /
+ =* Resolved (archived)=; aged subtrees append as level-2 children; created on
+ first use, appended to thereafter (one scaffold, never duplicated).
+- Tests: =tc-test--reset= now sets the aging knobs OFF (retain nil) so the
+ existing in-file-move + sync tests are untouched by the wall clock; a new
+ =tc-test--age= harness re-enables them with a fixed reference date and a temp
+ archive file. 6 new tests (old+undated move, cutoff-inclusive stay, disabled,
+ idempotent, check-no-write, straggler pipeline, append-preserves). 34/34 green.
+
+* Cross-project consideration for your value gate
+
+Default is ON (retain 7) for ALL consuming projects once this syncs. A project's
+first =task-sorted= after the sync will shed everything in its Resolved section
+older than a week to a new =archive/task-archive.org=. That's the intended
+feature, but flag it — projects with a large historical Resolved section will see
+a big first-run move (git-tracked, recoverable). Adjust the default or gate it if
+you'd rather it be opt-in per project.
+
+* Companion (project-local, NOT synced)
+
+.emacs.d's Makefile =task-sorted= target now also runs =lint-org.el todo.org=
+after the archive, as a structural-safety pass (org-lint catches links/drawers/
+blocks; we separately verified heading-level structure by hand). Makefiles aren't
+template-synced, so this is per-project — noting it in case the pattern is worth
+documenting alongside the tool.
diff --git a/docs/design/2026-06-30-daily-drivers-tailscale-correction.org b/docs/design/2026-06-30-daily-drivers-tailscale-correction.org
new file mode 100644
index 0000000..9e2bb52
--- /dev/null
+++ b/docs/design/2026-06-30-daily-drivers-tailscale-correction.org
@@ -0,0 +1,9 @@
+#+TITLE: Correction to claude-rules/daily-drivers.md: it states 'the
+#+SOURCE: from .emacs.d
+#+DATE: 2026-06-30 13:20:30 -0400
+
+Correction to claude-rules/daily-drivers.md: it states 'the agent can't reach the other machine; the point is to surface...'. That assumption is now demonstrably false. On 2026-06-30, from velox, I drove ratio directly over tailscale ssh — git fetch + reset --hard to repair ratio's .emacs.d after a history rewrite, plus scp'd a file across. Both daily drivers are on the same tailnet and reachable from each other.
+
+Gotcha worth capturing in the rule: the BARE hostname does not resolve (ssh ratio -> 'Could not resolve hostname'), which makes it look unreachable. The tailscale IP (e.g. 100.71.182.1) and the MagicDNS name (ratio.tailf3bb8c.ts.net) DO resolve and connect. First connection from a given address fails host-key verification under BatchMode; -o StrictHostKeyChecking=accept-new clears it. 'tailscale status' lists every node's IP + online state.
+
+Suggested rule change: reframe daily-drivers.md from 'can't reach, so surface it' to 'CAN reach over tailscale ssh — so the agent can directly sync/verify/repair the other daily driver, not just flag it'. Keep the flag-it guidance as the fallback for when tailscale is actually down. Add the bare-hostname-doesn't-resolve / use-tailscale-IP-or-MagicDNS gotcha. uname -n still tells you which machine you're on.
diff --git a/docs/design/2026-07-02-auto-flush-mechanism-note.org b/docs/design/2026-07-02-auto-flush-mechanism-note.org
new file mode 100644
index 0000000..fbe06ae
--- /dev/null
+++ b/docs/design/2026-07-02-auto-flush-mechanism-note.org
@@ -0,0 +1,20 @@
+#+TITLE: AUTO-FLUSH capability — proven live in the archsetup session
+#+SOURCE: from archsetup
+#+DATE: 2026-07-02 01:26:20 -0400
+
+AUTO-FLUSH capability — proven live in the archsetup session 2026-07-02, Craig asks that it be promoted to all projects and recommended as part of the no-approvals speedrun to keep sessions sharp.
+
+Problem: /clear is a user-only keystroke, so long autonomous sessions either bloat or hit arbitrary auto-compaction. Craig can't always be around to type it.
+
+Mechanism (companion script: self-inject.sh, sent separately to this inbox):
+1. At a clean task boundary, the agent refreshes .ai/session-context.org exactly as the flush skill does (checkpoint with Active Goal / Decisions / Next Steps).
+2. It derives its own tmux pane: match pane_pid from 'tmux list-panes -a' against its process ancestry (the ai launcher runs every agent session inside tmux, so this holds everywhere).
+3. It arms the injection VIA THE TMUX SERVER — tmux run-shell -b "sleep 25; tmux send-keys -t %N -l '/clear'; tmux send-keys -t %N Enter; sleep 15; tmux send-keys -t %N -l 'go — auto-flush resume: read .ai/session-context.org and continue per Next Steps'; tmux send-keys -t %N Enter" — and immediately ends its turn so the prompt is idle when the keys land.
+4. /clear fires the SessionStart hook (which already points a fresh context at notes.org + session-context.org), and the injected resume line starts the next turn. Zero human keystrokes.
+
+Gotchas learned the hard way:
+- A detached child (setsid/nohup/&) of a tool call DIES when the tool call ends; only tmux run-shell -b (server-owned) survives the turn boundary.
+- Under run-shell the process is a child of the tmux server, so ancestry-based pane detection can't run there — derive the pane first from the agent's shell, pass it explicitly.
+- Collision: if the user is typing when the keys fire, the injection merges into their input (a real /clear became '/clearto' mid-word). Fine for unattended sessions; warn the user to keep hands off the armed window if present.
+
+Suggested integration: an 'auto' mode on the flush skill (checkpoint, then self-inject instead of prompting the user), plus a line in the no-approvals speedrun workflow to auto-flush at clean boundaries when context grows heavy. The script could live in claude-templates' .ai/scripts/ so every project gets it on sync.
diff --git a/docs/design/2026-07-13-runtime-portability-inventories.org b/docs/design/2026-07-13-runtime-portability-inventories.org
new file mode 100644
index 0000000..b9e8244
--- /dev/null
+++ b/docs/design/2026-07-13-runtime-portability-inventories.org
@@ -0,0 +1,84 @@
+#+TITLE: Runtime Portability — Hook, MCP, and Memory Inventories
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-13
+
+Decision-prep for the generic-agent-runtime arc (spec: [[file:2026-05-28-generic-agent-runtime-spec.org]]). Three of the eight gap child tasks are inventories; this doc is their deliverable. Each section ends with a verdict and any decision left for Craig.
+
+* Hook parity inventory
+
+What's wired today (global =.claude/settings.json=, symlinked to =~/.claude/settings.json=):
+
+- PreToolUse AskUserQuestion hard-deny — blocks the popup-choice tool, forcing inline numbered options per =interaction.md=. Exists because prose alone failed.
+- PreCompact =precompact-priorities.sh= — saves priorities before context compaction.
+- SessionStart =session-title.sh= (cosmetic title) and =session-clear-resume.sh= (resume-after-/clear plumbing).
+- Stop =ai-wrap-teardown.sh= — sentinel-gated session teardown.
+
+Shipped but unwired (reference copies in =hooks/= + =settings-snippet.json=): =git-commit-confirm.py=, =gh-pr-create-confirm.py=, =destructive-bash-confirm.py=. Not active under Claude either, so they impose no parity requirement today.
+
+Language bundles wire PostToolUse validators per edit (=validate-bash.sh=, =validate-el.sh=, =validate-go.sh=, plus python/typescript equivalents) — and every bundle also ships the same validation as a =githooks/pre-commit=.
+
+Per-hook portability verdicts:
+
+- AskUserQuestion deny: moot off Claude. The popup tool is Claude Code's; Codex-style and local harnesses have no equivalent tool to deny. No port needed.
+- PostToolUse validators: covered off Claude. The bundles' git pre-commit hooks run the same validators, so enforcement survives at commit time instead of edit time. Acceptable downgrade, no work needed.
+- SessionStart clear-resume: not portable as-is; /clear semantics are Claude's. This is the same surface as the "session plumbing per runtime" child task — resolve it there, not as a hook port.
+- session-title: cosmetic, skip on other runtimes.
+- PreCompact priority-save: genuine gap. Codex-style harnesses compact without a hook point. Downgrade is prose (the session-context write triggers in protocols.org already say "save before compaction"), which is weaker but is also the pre-2026 Claude behavior. Accept prose, or shorten sessions on those runtimes.
+- Stop wrap-teardown: partial gap. Codex has a turn-completion notify command that could run the same sentinel-gated script; local harnesses vary. Port the script trigger where a hook point exists; where none does, teardown becomes a manual step in the wrap phrase.
+
+Verdict: only PreCompact and Stop carry real porting work, and both have acceptable degraded modes. Nothing here blocks a pilot.
+
+* MCP portability check
+
+Locally-configured servers (user scope, =~/.claude.json=): linear, notion, figma, slack-deepsat, google-calendar, google-docs-personal, google-docs-work, drawio, google-keep. No project-scope servers anywhere. All are ordinary MCP server definitions, portable to any MCP-speaking harness (Codex CLI reads them from its config; most local harnesses that matter speak MCP now). Port = translate the server blocks + carry the auth material. Mechanical.
+
+claude.ai-managed connectors (session-injected, not in local config): Gmail, and the claude.ai variants of Calendar/Drive. These do not travel. Overlap analysis: calendar and docs are already covered by the locally-configured servers; Gmail's workflow-preferred path is already =cmail-action= (local CLI), so the managed Gmail connector is convenience, not dependency.
+
+Named gap: signal-mcp. protocols.org "Paging Craig" names it as the only supported away-from-desk page path, and it is not in any local config — it lives claude.ai-side. Off Claude there is currently no page channel at all. This lands squarely on the open Signal-pager task ([#C], todo.org): the signal-cli runbook that task produces would be the runtime-neutral paging path. Recommend noting runtime-portability as an added motivation on that task.
+
+Verdict: portable except paging; paging's fix is already a filed task.
+
+* Memory story for non-Claude agents
+
+Claude auto-memory (=~/.claude/projects/<enc>/memory/=, MEMORY.md index) is harness-owned: written by Claude's memory tooling, loaded by its session start. A non-Claude agent should neither read nor write it.
+
+The designed cross-agent store already exists: the org-roam KB (=knowledge-base.md= — reading is plain rg over files, writing is one node per fact, both runtime-neutral). Project state lives in file artifacts every agent reads anyway: todo.org, notes.org, session anchors, docs/.
+
+One wording gap: knowledge-base.md's "Capture, then promote" section names harness memory as the capture layer. For a non-Claude agent the capture layer is the session log itself. Recommended one-sentence addition there: an agent without harness memory captures into its session log and promotes from it at wrap-up. Shared-asset edit, needs approval.
+
+Verdict: no build. One approved sentence in knowledge-base.md closes it.
+
+* Skill and command parity
+
+What rulesets ships: 11 skills (SKILL.md, model-invokable — add-tests, debug, five-whys, flush, frontend-design, pairwise-tests, playwright-js, playwright-py, review-code, root-cause-trace, voice) and 18 commands (plain .md prompts under =.claude/commands/=, user-invoked). Every body is markdown instructions; nothing executable lives in the registration layer.
+
+The load-bearing observation: a =/name= reference is just a pointer to a file on disk. Any harness that reads files can execute all 29 artifacts through one resolution rule in its bootstrap entry file: "a skill or command reference (=/voice=, =/review-code=, =/brainstorm=) resolves to =~/.claude/skills/<name>/SKILL.md= or =~/.claude/commands/<name>.md= — read the file and follow it." That single sentence, emitted by the instruction-bootstrap install target, makes the whole library portable with zero per-skill work. commits.md's existing "/voice unavailable — walk the patterns inline" fallback generalizes the same way.
+
+What that rule does not carry:
+
+- Auto-invocation. Claude Code triggers skills from their descriptions; other harnesses won't. The high-value auto-triggers (voice and review-code inside the publish flow) don't actually need it — commits.md invokes them by name at fixed flow points, and the resolution rule covers a by-name invocation. The convenience triggers (debug, frontend-design firing on topic match) degrade to on-request. Acceptable.
+- Native slash registration. Codex-style harnesses have their own custom-prompt mechanism; the 18 commands could also be registered there for ergonomics (an optional install nicety, not a requirement — the resolution rule already works).
+- flush. Its mechanics are Claude Code's (/clear, the self-inject resume hook). Explicitly not ported; the session-plumbing child owns that surface.
+- Harness built-ins (code-review ultra, plan mode, artifacts). Not rulesets' to port; flows that name them need per-runtime alternatives or graceful absence, same as today when they're unavailable.
+
+Verdict: no per-skill porting matrix needed. One resolution sentence in the bootstrap entry file (instruction-bootstrap child), optional native registration as an install nicety, flush explicitly excluded.
+
+* Session plumbing per runtime
+
+Four pieces, examined:
+
+- =session-context-path= is already runtime-aware by design — the =AI_AGENT_ID= shape carries a runtime segment (=host.project.runtime.epoch=) and the resolver is pure bash. Nothing to do.
+- The anchor cycle (suspend entry, startup's interrupted-session recovery, wrap-up's archive rename) is plain files driven by workflow prose. Any agent that reads protocols.org executes it identically. Nothing to do.
+- =self-inject.sh= is harness-agnostic: it types arbitrary strings into a tmux pane. The Claude-bound part is only the *payload* — "/clear" plus a resume line that leans on the SessionStart hook. A Codex auto-flush is a payload variant: the clear command differs (confirm the exact one when wiring) and the second injected line must itself carry the resume instruction ("read .ai/session-context.org and resume") since no hook fires. Arguably simpler than the Claude path.
+- =session-clear-resume.sh= (the SessionStart hook) is Claude-only, and the payload variant above makes it unnecessary off Claude.
+
+Verdict: no build. The flush *skill* stays Claude-official; a codex auto-flush is a documented payload variant to wire the day a codex session actually wants it, not before.
+
+* Decisions for Craig
+
+All four approved by Craig, 2026-07-13:
+
+1. PreCompact downgrade off Claude: prose-only "save before compaction" accepted.
+2. Stop-teardown off Claude: port via Codex notify where available, manual teardown elsewhere.
+3. Runtime-portability motivation note added to the Signal-pager task.
+4. knowledge-base.md capture-layer sentence added (non-Claude runtimes capture into the session log and promote from there).
diff --git a/docs/design/2026-07-14-sentry-workflow-proposal.org b/docs/design/2026-07-14-sentry-workflow-proposal.org
new file mode 100644
index 0000000..d29ceb8
--- /dev/null
+++ b/docs/design/2026-07-14-sentry-workflow-proposal.org
@@ -0,0 +1,150 @@
+#+TITLE: Sentry Workflow — Proposal for rulesets
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-13
+
+* Status
+Proposal from the work project to the rulesets session, for implementation as a
+general (shared-layer) workflow: =.ai/workflows/sentry.org= plus a small
+supporting script or two under =.ai/scripts/=.
+
+* Problem
+
+An agent session left running overnight can keep a project clean and healthy on
+a cadence, but the current pieces don't coordinate. Two gaps:
+
+1. *No agent-vs-agent collision guard on shared public files.* =capture-guard=
+ only detects Emacs org-capture buffers (the human case) on one target file,
+ and it's advisory, not a lock. There is no flock anywhere in =.ai/scripts/=.
+ Multiple sessions (several projects, sometimes both daily drivers) can
+ read-modify-write =~/org/roam/inbox.org= and the =agents/= KB nodes at once.
+ The only coordination is the git pull/commit/push discipline in
+ knowledge-base.md, which is optimistic — a pull/push race forks a
+ =*sync-conflict*= file, and the KB search already globs those out, so we know
+ the forks happen and get dropped.
+
+2. *No single serialized runner.* Firing inbox-zero and triage-intake as
+ separate hourly crons means they can overlap each other and other agents.
+
+Sentry is one supervisor loop that runs a fixed sequence of hygiene passes
+serially, holds locks so it never collides with itself or other agents, and
+checkpoints aggressively for crash recovery.
+
+* What sentry is
+
+A single workflow, driven by one interactive =/loop= (one cron), that on each
+fire walks an ordered list of passes inside a single turn. It replaces the
+per-task crons. Design invariants:
+
+- *Serial.* Passes run one at a time, in order. They never overlap each other.
+- *Single-runner lock.* Sentry takes a project-local flock
+ (=.ai/.sentry.lock=) at entry. If a prior fire is still running, this fire
+ skips instead of doubling up.
+- *Shared-file lock.* Any pass that writes a public roam file first acquires a
+ roam-write lock (flock on =~/org/roam/.roam-write.lock=) held across the whole
+ read-modify-write-commit-push cycle, so it serializes against other agents on
+ the same host. =capture-guard= still runs underneath for the human case. (The
+ roam-write lock is new; it's the general guard the dropped KB-guard note would
+ have proposed, folded in here.)
+- *Non-destructive by default.* Read-only and idempotent-filing passes run
+ fully. Anything destructive or judgment-heavy (triage trash/mark-read, task
+ regrades, spec flips) queues its proposal into the digest for morning
+ approval, never fires unattended.
+- *Idempotent + isolated.* Every pass is safe to repeat. A pass that errors
+ doesn't abort the rest; its failure is a loud banner in the digest.
+
+* Crash-recovery spine (applies to every pass)
+
+Crash recovery has been a live consideration. Two enforced steps wrap each pass:
+
+1. *Session-context between passes.* Before the next pass starts, the current
+ pass appends a dated entry to =.ai/session-context.org= recording what it
+ did. A crash mid-sequence leaves a readable "got this far" trail, so the next
+ session resumes without re-deriving state.
+2. *Local commit, no push, on any disk change.* Every pass that writes to disk
+ gets its own =chore(sentry): <pass> — <what changed>= commit on the current
+ branch, unpushed. A crash loses at most one pass. Craig reviews the stack in
+ the morning and pushes deliberately. Conventional messages, no AI
+ attribution. (Open decision: commit on the current branch vs a dedicated
+ =sentry/<date>= branch — see Open decisions.)
+
+The unit per pass is therefore: do the work → log to session-context → commit.
+That triplet is the recovery boundary.
+
+* Passes (v1 — all ten, plus the first-run KB promotion)
+
+Ordered so read-only/pull passes precede writes, and the heaviest external
+sweep sits mid-run:
+
+1. *Roam pull* — =git -C ~/org/roam pull --ff-only= so later reads are fresh.
+ Read-only.
+2. *Inbox zero* — inbox.org roam mode. Route roam-inbox items this project
+ owns. Acquires the roam-write lock for the inbox edit.
+3. *Triage intake* — triage-intake.org. External-accounts sweep, classify,
+ file Action tasks. Destructive actions (trash/mark-read) queue for morning;
+ =mbsync -a= and the sentinel advance run.
+4. *Todo cleanup* — clean-todo.org + todo-cleanup.el + lint-org.el. Convert
+ level-3 done subtasks to dated logs, archive done subtrees, flag lint.
+5. *Task audit* — task-audit.org. Re-check =:solo:=/=:quick:= tags,
+ priority-scheme conformance, bug severity×frequency grades, stale =DOING=
+ specs with closed parents, blocked/blocker reciprocity. Regrades queue for
+ morning.
+6. *Working-files hygiene* — flag =working/<slug>/= dirs whose task is DONE but
+ not filed to =assets/=. Report only; filing is a morning decision.
+7. *Spec status board* — the docs-lifecycle grep. Flag =DOING= specs with
+ closed parents, or specs stuck in =DRAFT=.
+8. *Link integrity* — lint-org broken =file:= links across todo/notes/prep.
+9. *Git health* — uncommitted drift, unpushed commits, stale local branches,
+ whether main is behind origin.
+10. *Prep + symlink freshness* — verify tomorrow's prep and both symlinks
+ resolve.
+11. *KB lesson promotion (first run of the session, and thereafter as new
+ lessons accrue)* — promote recent durable lessons out of the fast capture
+ layer. CLASSIFICATION-GATED:
+ - Personal project → write to the roam KB as =agents/= nodes (pull, lock,
+ write, commit, push), per knowledge-base.md.
+ - Work / denylisted project → NEVER touch roam. Promote to the project's own
+ durable store (e.g. =deepsat/knowledge.org=) instead, and emit the
+ one-line refusal note from the knowledge-base.md refusal contract so
+ nothing is lost silently.
+
+* Digest output
+
+One consolidated, timestamped summary per fire: run time first, then per-pass
+one-liners (what it found / did), the morning-approval queue (proposed
+trash/mark-read, regrades, files-to-move), and any failures as loud banners at
+the top. Delta-only where a pass supports it (triage): a pass with nothing to
+report is one line.
+
+* Cadence
+
+One hourly sentry to start (passes short-circuit fast when there's nothing to
+do). A slower evening interval (every 2–3 hours) is a config knob if hourly
+proves noisy.
+
+* Open decisions for rulesets
+
+1. *Commit target* — per-pass commits on the current branch (simplest, matches
+ "commit but don't push") vs a dedicated =sentry/<date>= branch (keeps the
+ working branch clean, costs a branch dance). Proposal leans current branch.
+2. *Roam-write lock scope* — host-local flock only (solves same-host
+ multi-agent, the common case) vs also hardening roam-sync to surface
+ cross-host conflicts loudly. Proposal: ship the flock now, note the
+ cross-host limit, treat roam-sync conflict-surfacing as a separate follow-up.
+3. *Interval default* — hourly vs evening-only. Proposal: hourly, config knob.
+4. *Which passes gate on a green tree* — e.g. skip todo-cleanup commits if the
+ project suite is red, matching inbox monitor-mode's clean-tree gate.
+
+* Cover note from sender (work, 2026-07-14 00:02)
+
+Intro for the sentry-proposal.org file just sent. This is a NEW general workflow proposal (.ai/workflows/sentry.org + supporting scripts under .ai/scripts/), not an edit to an existing synced file.
+
+What it is: one serialized supervisor loop that runs a project's evening hygiene passes on a cadence, with locks so it never collides with itself or with other agents. It replaces firing inbox-zero + triage-intake as separate crons.
+
+Why now: an agent left running overnight can keep a project clean, but the current pieces don't coordinate. capture-guard only handles the Emacs-capture case on one file and holds no lock; there is no flock anywhere in .ai/scripts/; and separate crons let inbox-zero and triage overlap each other and other agents on shared roam files, which already fork *sync-conflict* files that the KB search globs out.
+
+Three design points worth your judgment:
+1. Crash-recovery spine on every pass (crash recovery has been a live concern): enforce a session-context.org update BETWEEN passes, and a local commit (no push) after any disk change, so a crash loses at most one pass.
+2. A new roam-write flock (~/org/roam/.roam-write.lock) as the shared-file guard. This absorbs a separate 'guard all public KB files' note we considered and dropped. The lock is the better home for that concern than a bespoke per-file guard.
+3. The KB lesson-promotion pass is classification-gated: personal projects promote to roam agents/ nodes; a denylisted work project promotes to its own local store (deepsat/knowledge.org) and never touches roam, per knowledge-base.md's refusal contract.
+
+Companion files to reconcile: knowledge-base.md (the gated promotion + refusal contract), inbox.org (roam mode + the new lock), triage-intake.org (runs as a sentry pass; keep its own triggers), capture-guard (becomes one layer under the roam-write lock, not the whole guard), and clean-todo.org / task-audit.org / lint-org.el (invoked as passes). Open decisions for you are listed at the bottom of the proposal file: commit target, lock scope, interval default, and which passes gate on a green tree.
diff --git a/docs/design/2026-07-15-subproject-pattern-proposal.org b/docs/design/2026-07-15-subproject-pattern-proposal.org
new file mode 100644
index 0000000..61ade18
--- /dev/null
+++ b/docs/design/2026-07-15-subproject-pattern-proposal.org
@@ -0,0 +1,22 @@
+#+TITLE: Proposal: promote the "subproject" pattern into the claude-r
+#+SOURCE: from home
+#+DATE: 2026-07-15 23:24:07 -0500
+
+Proposal: promote the "subproject" pattern into the claude-rules layer.
+
+WHAT WE DID (in home, 2026-07-15)
+We named and built a pattern for what happens when a standalone project is consolidated into another project. home absorbed nine former projects on 2026-06-11 (danneel, finances, jr-estate, kit, health, documents, clipper, elibrary, philosophy). We now call each folded-in unit a "subproject": a former standalone project that lives as a subdirectory, shares the parent's .ai/ scope (one session, one toolchain), but keeps its own files, assets, and history self-contained under its subdir.
+
+The pattern has four parts:
+1. Vocabulary — parent project, subproject, consolidation (the process), subproject archiving. Terminology so we can talk precisely about where a subproject's documents/assets live and archive them cleanly.
+2. Subproject brief — a read-first orientation doc at <subproject>/<subproject>-brief.org (situation, current state, reading list, and for deep ones cast/timeline/key-facts-with-verify-flags). The parent's notes.org holds a thin index + a read-first rule + pointers to urgent items only, never a copy of subproject substance ("one fact, one home").
+3. Lifecycle criteria — when to CREATE a subproject (consolidate: domain fit, low independent cadence, shared-scope safety re: privacy/remote posture, no independent collaborators), and when/how to ARCHIVE one cleanly (terminal state -> move the self-contained subdir to <parent>/archive/, reconcile pointers/links/tags, manifest note).
+4. Instrumentation — a use/miss log, metrics weighted toward Craig-caught errors over self-report, review folded into task-audit, and a falsifiable kill criterion.
+
+WHY IT'S GENERAL (not just a home thing)
+Consolidating a project into another drops what the former project's own startup used to auto-load (its full depth), so sessions start cold and re-read files. That failure recurs whenever ANY project folds into another. The pattern also gives consolidation and archiving a named, repeatable shape, and the self-contained-subdir rule is what makes archiving a move-plus-pointer-reconcile instead of an excavation.
+
+THE PROPOSAL
+Promote this to a rulesets claude-rule (e.g. claude-rules/subprojects.md): the vocabulary, the brief convention, the parent-vs-subproject content criterion, and the create/archive criteria. Any parent/hub project would then inherit it. home's .ai/subprojects.org (attached next) becomes the local instance/index under the general rule. It's a companion to working-files.md (in-progress artifact layout) and docs-lifecycle.md (formal-doc lifecycle) — same family, different scope: this one governs whole-subproject orientation and lifecycle.
+
+Attached: the home instance .ai/subprojects.org as the reference draft to adapt. Apply your value gate; this is a proposal, not a synced-file edit.
diff --git a/docs/design/2026-07-15-subprojects-convention-home-instance.org b/docs/design/2026-07-15-subprojects-convention-home-instance.org
new file mode 100644
index 0000000..b031a41
--- /dev/null
+++ b/docs/design/2026-07-15-subprojects-convention-home-instance.org
@@ -0,0 +1,282 @@
+#+TITLE: Subprojects — Convention
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-15
+
+# The reference spec for the subproject pattern: what a subproject is, when to
+# create one (consolidation), how to keep it legible (the brief + content
+# split), and how to archive one cleanly. Not read every session — the
+# always-on parts (the subproject index + the read-first rule) live in
+# notes.org and point here. Read this when consolidating a project, creating
+# or maintaining a subproject brief, or archiving a subproject.
+
+* Vocabulary (say it the same way every time)
+
+The terminology is the point — it's what lets us talk precisely about where a
+subproject's documents and assets live, and archive them cleanly.
+
+- *Parent project* (a.k.a. hub) — a top-level project that owns an =.ai/=
+ scope: its own protocols, session cadence, memory, inbox, todo. home, work,
+ rulesets, .emacs.d are parent projects. Reached by launching an agent
+ session in its directory.
+- *Subproject* — a formerly-standalone project that has been *consolidated*
+ into a parent and now lives as a subdirectory of it (=danneel/=,
+ =finances/=, =kit/= …). A subproject shares the parent's =.ai/= tooling and
+ session — it is not a separate agent scope — but keeps its own working
+ files, assets, notes, and history self-contained under its one subdirectory.
+- *Consolidation* — the process of turning a standalone project into a
+ subproject of a parent. (home's nine subprojects were consolidated on
+ 2026-06-11.) The inverse — promoting a subproject back to a standalone
+ project — is rare but allowed; it's just consolidation run backward.
+- *Subproject archiving* — retiring a completed or abandoned subproject
+ cleanly. Because a subproject is self-contained under one subdirectory,
+ archiving is a move plus a pointer reconcile, not an excavation.
+- *Subproject brief* — the read-first orientation doc for one subproject, at
+ =<subproject>/<subproject>-brief.org=.
+
+The cross-project boundary rule (=claude-rules/cross-project.md=) governs
+*parent* projects, which are separate =.ai/= scopes. It does *not* apply
+between a parent and its subprojects — they share one scope.
+
+* Why the pattern exists
+
+As a standalone project, each of these units auto-loaded its own notes.org —
+its full depth — at every startup, so recall felt instant. Consolidated into a
+parent, a subproject is one compressed section competing with the others, and
+the parent's startup surfaces hub-wide state, not any one subproject's depth.
+So a session starts cold on a given subproject and re-reads several files on
+demand. This convention restores fast, accurate orientation without
+re-bloating what loads every session — and gives consolidation and archiving a
+named, repeatable shape.
+
+* Document & asset location (what makes clean archiving possible)
+
+Everything for a subproject lives under =<subproject>/= — its =notes.org=,
+=assets/=, working files, event log, briefs. Links *within* a subproject are
+relative to the subproject dir, so the whole tree can move without breaking
+them. The parent's own top-level =assets/= is for parent-level material only,
+never a subproject's. This self-containment is exactly what lets archiving be
+a single =mv= plus a pointer reconcile. Keep it strict: a subproject's asset
+in the parent's shared =assets/= is a future broken link.
+
+* The two layers (parent vs subproject)
+
+- *Always-on layer* — the parent's notes.org. Read at every startup, so every
+ line is a tax paid whether or not it's relevant today. Holds only what must
+ be seen session-independently.
+- *On-demand layer* — the subproject brief + the subproject's notes.org.
+ Loaded only when working that subproject. Holds the depth.
+
+The whole design is: sort content by *when it's needed*, keep the always-on
+layer thin, put the depth one open away.
+
+* Creating a subproject (consolidation criteria)
+
+Consolidate a standalone project into a parent as a subproject when *all* hold:
+
+1. *Domain fit* — the project is a facet of a domain the parent already owns
+ (danneel is home life-admin; finances is home).
+2. *Low independent cadence* — it no longer needs its own session rhythm.
+ You'd rather reach it from the parent than launch it standalone, and its
+ standalone startup/sync/session overhead now outweighs the isolation.
+3. *Shared-scope safety* — it does not need a git/remote/visibility posture
+ distinct from the parent. A public code project, a team repo, or anything
+ with its own CI or gitignore-privacy posture *stays standalone* — folding
+ it would break the parent's privacy model (see
+ =claude-rules/git-hosting-privacy-model= and the gitignore-vs-track
+ decision). Personal/documentation units fold; code-with-a-public-remote
+ does not.
+4. *No independent collaborators* depend on it as a standalone deliverable.
+
+Keep it standalone when any of those fail: active independent development, its
+own team or remote, a distinct visibility/security posture, or activity high
+enough that competing in the parent's hub would bury it.
+
+*The consolidation act itself* (mechanics, mirroring home's 2026-06-11 fold):
+move the project tree under the parent as =<parent>/<subproject>/=; import its
+tasks into the parent's todo with an area =:tag:= and a =:MIGRATED_FROM:=
+property; fold its live reminders/decisions into the parent's notes as
+pointers (see the content split); build the subproject brief; record the fold
+in a consolidation manifest (=docs/consolidation-manifest-<subproject>.org=).
+Retire the source to =~/projects/.retired/<name>= once parity is verified.
+
+* The subproject brief
+
+** Path (fixed — discovery is mechanical)
+
+=<subproject>/<subproject>-brief.org= (=danneel/danneel-brief.org=,
+=finances/finances-brief.org=). The subproject name is the directory name, so
+the brief name follows for free. One predictable name means the read-first
+rule never has to hunt; put descriptive richness *inside*, not in the filename.
+
+** Spine (scales from light to deep)
+
+Required in every brief, even a fifteen-line one:
+- *One-paragraph situation* — what this subproject is, in plain language.
+- *Current state / open threads* — the live stuff. The only high-churn
+ section; everything else is stable.
+- *Reading list* — the subproject's deeper files and what each holds. The
+ brief *points into* notes.org and the rest; it does not copy them.
+
+Add only where the subproject warrants it (danneel, jr-estate, finances earn
+all of these; elibrary or documents may earn none):
+- Cast / contacts.
+- Timeline (for matters with history).
+- Key facts / numbers, with explicit *verify* flags anywhere the underlying
+ record disagrees with itself. A brief that asserts a wrong number is worse
+ than no brief — flag, don't guess.
+
+** Read-first rule (the behavior)
+
+When a session's work targets a subproject — a file under =<subproject>/=, its
+=:tag:=, or Craig naming the topic — open =<subproject>/<subproject>-brief.org=
+*first*, before touching anything else in it. If the subproject has no brief
+yet, creating it is the first step of the work (lazy backfill), and the same
+pass reconciles that subproject's parent-vs-subproject content.
+
+* Parent-vs-subproject content criterion
+
+The test for any piece of content: *if next session I'm working a different
+subproject, do I still need to see this?* Three bins:
+
+1. *Yes, and it's a fact or policy* → hub-level → the parent's notes.org.
+ Machine names, mail policy, calendar access, the hub map, the subproject
+ index.
+2. *Yes, but only because it's urgent* — a deadline or blocking status that
+ would be missed if it only lived where I might not look this week → the
+ parent holds a *one-line pointer* (item + date + "see the subproject
+ brief"); the substance stays in the subproject.
+3. *No* → the subproject. Working depth: contacts, contract facts, case
+ theory, evidence, document index, and any reminder that isn't time-critical.
+
+One hard rule over both axes: *one fact, one home.* The parent never copies
+subproject substance — it points. The parent holds the trigger; the subproject
+holds the truth. Every parent↔subproject duplicate is a second surface that
+goes stale.
+
+* Archiving a subproject (archiving criteria + clean process)
+
+Archive a subproject when it reaches a *terminal* state:
+- *Complete* — the work is done and closed (a settlement signed, a probate
+ closed, a trip finished): no live threads, no open tasks, nothing
+ time-sensitive in the parent still pointing at it; or
+- *Abandoned / superseded* — dropped, or replaced by other work.
+
+Clean archive process (self-containment is what makes each step a one-liner):
+1. Close or settle its tasks in the parent's todo (DONE / CANCELLED), per the
+ todo completion rules.
+2. Move the whole subproject tree to the parent's archive:
+ =<parent>/archive/<subproject>/= (keep it intact — archiving is reversible,
+ not deletion).
+3. Update inbound =file:= links that pointed into it (grep first, per the
+ keep-links-current rule).
+4. Move or retire its brief with it, and drop its row from the active
+ subproject index (or mark it =archived= there).
+5. Strip its pointers/reminders from the parent's always-on layer, and retire
+ its =:tag:= from active use.
+6. Record the archive in a manifest note
+ (=docs/archive-manifest-<subproject>.org=): why, when, where it moved,
+ what links were repointed. Mirrors the consolidation manifest.
+
+The =:LAST_UPDATED:= on a brief plus "no open tasks under the tag" is the
+signal a subproject is a candidate for archiving — surfaced at the review
+cadence, never auto-archived.
+
+* Freshness
+
+- *Wrap-up hook*: at wrap-up, for each subproject touched this session, refresh
+ its brief's Current-state section and bump its =:LAST_UPDATED:= date. Stable
+ sections rarely change, so this is cheap.
+- *Staleness nudge*: a brief whose =:LAST_UPDATED:= is old while its subproject
+ was recently active is worth surfacing at startup.
+
+* Self-improvement loop
+
+Every real use emits a signal, and the signal is *captured*, not silently
+acted on:
+- Brief answers the question → *hit*.
+- Brief is missing something, stale, or wrong → *miss*, logged in
+ subprojects-log.org (a home-side artifact; no rulesets file).
+- Each miss drives *two* updates: *local* (fix that brief) and, if the same
+ kind of miss recurs across subprojects, *structural* (fix this spine/spec,
+ not just the one file). The structural half is the actual self-improvement —
+ grow-and-refine from real use rather than rewriting from scratch (the ACE
+ idea behind the =codify= skill).
+
+* Metrics
+
+Split by who can trust them.
+
+*Process signals (I observe, cheaply):*
+- *Orientation cost* — files/tool-calls before my first correct action in a
+ subproject. Target: ~1 (the brief), occasionally a second deep dive.
+- *Brief hit rate* — fraction of subproject questions answered from the brief
+ alone. Should climb.
+- *Freshness gap* — days between =:LAST_UPDATED:= and last subproject activity.
+ Target ~0.
+
+*Outcome signals (Craig is ground truth — weight these over the above):*
+- *Craig-caught errors* — the real quality bar. Target: trend to zero. If
+ briefs work, Craig stops catching me.
+- *Felt speed* — did dropping into the subproject feel fast and right.
+
+*Honesty caveat.* Most process signals are self-observed, and self-report is
+biased toward "hit." So the external signals (Craig-caught errors, felt speed)
+outrank mine, and the log is auditable so Craig can spot-check my self-grades.
+The instrumentation must stay lighter than the briefs, or it rots like an
+unmaintained brief. Small N, no clean A/B — this is directional evidence, not
+proof.
+
+* Review cadence
+
+Fold a "subprojects health" check into the existing task-audit cadence: read
+the log, look at the hit-rate / correction / freshness trends, flag any
+subproject that's gone terminal (archive candidate), and check the
+parent-vs-subproject criterion — anywhere the parent holds a copy where it
+should hold a pointer. Decide continue / adjust / kill for the pattern itself.
+
+* Kill criteria (falsifiable)
+
+After ~6-8 subproject-sessions across ≥3 subprojects, drop or rework the
+pattern if:
+- the correction rate isn't falling, or Craig is still catching errors; or
+- the freshness gap keeps reopening despite the wrap-up hook; or
+- maintenance costs more time than the orientation it saves.
+
+Success is the mirror: orientation down to ~1 read, Craig catching ~nothing,
+briefs current, and I'm right fast in any subproject cold.
+
+* Rollout
+
+- Define the convention (this file) + the index scaffold in notes.org. Done
+ 2026-07-15.
+- *Lazy backfill*: the next session that touches a subproject with no brief
+ creates the brief as its first step — cost paid exactly when it's used.
+- *Reconcile in the same pass*: building a subproject's brief includes
+ reconciling its parent-vs-subproject content — push substance down, leave
+ pointers up, drop dead reminders.
+- *Promotion*: this pattern is home-instanced but general. Consolidation into
+ subprojects will recur whenever any project folds into another, so this is a
+ candidate to promote into the rulesets rules layer (proposed to rulesets
+ 2026-07-15).
+
+* Adoption status
+
+| Subproject | Brief exists | Reconciled | Notes |
+|------------+--------------+------------+---------------------------------------|
+| danneel | yes | yes | First worked instance, 2026-07-15. |
+|------------+--------------+------------+---------------------------------------|
+| clipper | no | no | Lazy backfill on next touch. |
+|------------+--------------+------------+---------------------------------------|
+| documents | no | no | Lazy backfill on next touch. |
+|------------+--------------+------------+---------------------------------------|
+| elibrary | no | no | No notes.org yet; lightest subproject.|
+|------------+--------------+------------+---------------------------------------|
+| finances | no | no | Deep; earns the full spine. |
+|------------+--------------+------------+---------------------------------------|
+| health | no | no | Has its own notes.org. |
+|------------+--------------+------------+---------------------------------------|
+| jr-estate | no | no | Deep; earns the full spine. |
+|------------+--------------+------------+---------------------------------------|
+| kit | no | no | Lazy backfill on next touch. |
+|------------+--------------+------------+---------------------------------------|
+| philosophy | no | no | Lazy backfill on next touch. |
diff --git a/docs/design/2026-07-16-polyglot-bundle-collision.txt b/docs/design/2026-07-16-polyglot-bundle-collision.txt
new file mode 100644
index 0000000..b7e1257
--- /dev/null
+++ b/docs/design/2026-07-16-polyglot-bundle-collision.txt
@@ -0,0 +1,30 @@
+Proposal: language bundles collide on coverage-makefile.txt in a polyglot project, and the second install loses silently.
+
+Context: scaffolded a new project (clock-panel) on 2026-07-16 with both the python and typescript bundles installed into the same project.
+
+What happened. The second install printed:
+
+ [skip] coverage-makefile.txt already exists (use FORCE=1 to overwrite)
+
+Both bundles ship a file at that same path. Python installed first and won, so the project ended up with Python's coverage fragment only. The TypeScript one (c8, Istanbul json-summary) never landed. The line reads like routine idempotence, not a dropped deliverable, so it's easy to skim past. I only caught it by checking which targets were actually in the file.
+
+The workaround is bad. FORCE=1 gets the TypeScript fragment but overwrites the Python one, so you can't have both without renaming by hand between installs. FORCE=1 also re-seeds CLAUDE.md, which is fine on a fresh project and destructive on a customized one. What I did in clock-panel: install python, mv coverage-makefile.txt coverage-makefile-python.txt, install typescript FORCE=1, mv coverage-makefile.txt coverage-makefile-typescript.txt. Both fragments now coexist under language-suffixed names.
+
+The deeper problem, which the filename collision only hints at: both fragments define targets with the SAME names.
+
+ coverage:
+ coverage-summary:
+
+So even with both files present, a polyglot project can't copy both into one Makefile. It gets duplicate targets. The fragments assume they're the only language in the project. Renaming the files doesn't fix that, it just makes both sets of instructions visible while leaving the conflict for whoever pastes them.
+
+Options, in the order I'd weigh them:
+
+Suffix the shipped filename per bundle (coverage-makefile-python.txt, coverage-makefile-typescript.txt) so installs never collide. Cheap, and it makes the multi-bundle case work by default. It leaves the target-name conflict.
+
+Namespace the targets too (coverage-python / coverage-typescript, with a coverage target that runs both). Fixes the real problem for polyglot projects. Bigger change, and it makes single-language projects type a longer target name unless there's an alias.
+
+At minimum, make the skip line loud when the skipped file came from a DIFFERENT bundle than the one that wrote it. A skip that means "already yours" and a skip that means "another language's file is here and yours is being dropped" currently look identical.
+
+Worth deciding whether polyglot projects are a supported case at all. If they are, the bundles need a collision story. If they aren't, install-lang could say so when it detects a second bundle going into a project that already has one, rather than half-installing.
+
+Also flagged separately in a note from home today: the same install-lang run is where I'd look for other per-bundle files that could collide on a shared filename. I only checked coverage-makefile.txt.
diff --git a/docs/design/2026-07-17-dated-log-planning-line-strip-proposal.md b/docs/design/2026-07-17-dated-log-planning-line-strip-proposal.md
new file mode 100644
index 0000000..2448600
--- /dev/null
+++ b/docs/design/2026-07-17-dated-log-planning-line-strip-proposal.md
@@ -0,0 +1,15 @@
+Proposal: two linked gaps that let a closed sub-task keep polluting the org agenda.
+
+WHAT HAPPENED (home, 2026-07-17)
+Six completed sub-tasks under a DONE parent (a finished trip) had each been closed correctly to the dated event-log form per todo-format.org's depth-based completion rule — keyword, priority, and tags dropped, heading rewritten to "YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <past-tense>". But every one kept its old active SCHEDULED: <date> planning line. A dated-log heading has no TODO keyword, and org-agenda renders ANY headline carrying an active SCHEDULED, so all six showed on the daily agenda as weeks-overdue ("Sched.31x") long after the work and the parent were done. They're invisible to a keyword scan (no TODO) and survive archive-done (parent isn't level-2 CANCELLED), so nothing catches them.
+
+ROOT CAUSE
+The completion rewrite strips keyword/priority/tags but nothing strips the planning line, and an interactive org close (org-log-done) only stamps CLOSED — it never removes a pre-existing SCHEDULED/DEADLINE. So the stale timestamp lingers and pins the entry to the agenda permanently.
+
+TWO FIXES, both rulesets-owned:
+
+1. todo-format.org — the sub-task completion rule (the "*** and deeper — rewrite to a dated event-log entry" section). Add an explicit step: remove any SCHEDULED:/DEADLINE: planning line when rewriting to the dated form. The completion time already lives in the heading; an active planning date on a historical log entry is always wrong. The VERIFY dated-completion path and todo-cleanup.el --convert-subtasks want the same treatment (the batch converter pulls the timestamp from CLOSED and keeps heading text — it should also drop the planning line).
+
+2. lint-org.el — add a checker (proposed name: dated-log-heading-active-timestamp). Flag any dated-log heading (a heading matching the YYYY-MM-DD ... @ ... form, no TODO keyword) that still carries an active <...> SCHEDULED or DEADLINE. This is the mechanical backstop for #1, the same way subtask-done-not-dated backstops the depth rule. It would have caught all six here.
+
+Local fix already applied in home (stripped the six lines by hand); this is about the durable convention + lint so the next instance is prevented and caught. No home-specific detail needs to travel — the gap is general to any project using the dated-log completion form.
diff --git a/docs/design/2026-07-17-todo-cleanup-dated-seal-proposal.md b/docs/design/2026-07-17-todo-cleanup-dated-seal-proposal.md
new file mode 100644
index 0000000..61d457c
--- /dev/null
+++ b/docs/design/2026-07-17-todo-cleanup-dated-seal-proposal.md
@@ -0,0 +1,38 @@
+Proposal: change todo-cleanup.el's --archive-done aging to a dated-seal model
+
+Origin: work project, 2026-07-17. Reconciling a stale rotation convention with the tool's actual behavior. Craig ratified the design in-session.
+
+## The problem
+
+`--archive-done` today does two things: move level-2 DONE/CANCELLED out of Open Work into Resolved, then age Resolved by moving subtrees older than `tc-archive-retain-days` (default 7) into one rolling file `tc-archive-file` (default archive/task-archive.org).
+
+Two mismatches with the intended convention:
+
+1. Retention is 7 days; the convention wants the last month kept inline in Resolved.
+2. One rolling file forever; the convention wants periodic dated files so todo.org's archive is browsable by seal date.
+
+An earlier attempt to name files by calendar quarter (resolved-YYYY-QN.org) hit a seam: with a one-month retention window, a task closed in the last month of a quarter isn't archived until after the quarter boundary, so a quarter-named file can't hold it without mislabeling.
+
+## The design (ratified)
+
+Redefine the sealed file by a predicate that's always true of its contents, and date it by the seal run rather than by calendar quarter:
+
+- A level-2 DONE/CANCELLED subtree is archived when its CLOSED date is older than one month, OR its CLOSED date can't be parsed (archive regardless — a keyword-complete task with no readable close date is cruft, not live work).
+- The working (open) file keeps the existing name task-archive.org. A seal renames it to resolved-YYYY-MM-DD.org (the seal date) and starts a fresh working file.
+- The dated file means "everything sealed as of that date," not a quarter — so a late-quarter close archived after a boundary is never mislabeled. Cadence (quarterly) becomes independent of correctness; slip is harmless.
+
+## Concrete changes to todo-cleanup.el
+
+1. `tc-archive-retain-days` default 7 → 31 (or expose it; the value should reflect "one month"). Reconcile the test at test-todo-cleanup.el:362 that asserts 7-day aging.
+2. Aging predicate: archive when `closed < now - retain-days` OR `closed` is unparseable. Today an unparseable-CLOSED subtree is already moved out in aging per the docstring ("those with no parseable CLOSED date are moved out") — keep that, and make it explicit in the contract.
+3. Add a seal step (new flag, e.g. `--seal`, or fold into a boundary check): when invoked, rename the current `tc-archive-file` (task-archive.org) → `resolved-<today>.org` beside it and let the next aging recreate a fresh working file. Whether the seal fires automatically at a quarter boundary or stays a manual flag is your call — the manual quarterly rotation task covers it until then.
+4. Tests: update the file-aging tests (test-todo-cleanup.el:362, :378, :491, :513, :518) for the new retain default and the seal/rename behavior. The gitignore-inheritance logic (tc--ensure-archive-gitignored) applies unchanged to both the working and sealed names.
+
+## Companion state already applied in the work project (for reference, not to sync)
+
+- todo.org grew a `* Work Resolved` section; 30 level-2 DONE/CANCELLED moved there.
+- archive/task-archive.org renamed to archive/resolved-2026-07-17.org (grandfathered H1 seal), inbound link updated.
+- archive/README.org rewritten to the dated-seal contract above.
+- The rotation task rewritten to the new procedure.
+
+The rulesets change is only the todo-cleanup.el behavior + its tests. The README lives per-project (this one is already updated); if rulesets ships a canonical archive README template, align it to the dated-seal contract too.
diff --git a/docs/design/2026-07-18-colloquialisms-and-the-list-proposal.md b/docs/design/2026-07-18-colloquialisms-and-the-list-proposal.md
new file mode 100644
index 0000000..7254447
--- /dev/null
+++ b/docs/design/2026-07-18-colloquialisms-and-the-list-proposal.md
@@ -0,0 +1,22 @@
+#+TITLE: PROPOSAL: a "Colloquialisms and Expansions" convention + the
+#+SOURCE: from home
+#+DATE: 2026-07-18 18:04:42 -0500
+
+PROPOSAL: a "Colloquialisms and Expansions" convention + the "the list" before-close-queue norm. Craig recommends other projects adopt both. Home has implemented it locally as the reference; sending it up so rulesets can decide whether to make it a shared norm.
+
+THE NORM: "the list" = a before-close FIFO queue
+When Craig says "put X on the list" / "add X to the list", X is appended to a Before-Close Queue: a FIFO queue of tasks and actions to finish before the session closes (wrap-up). Work oldest-first. Process the queue at wrap-up before teardown, surfacing anything unfinished rather than dropping it. Session-scoped: resets when the session anchor is archived at wrap. Anything that must outlive the session is a todo.org task instead.
+
+Home implementation (reference):
+- The norm is documented in home .ai/notes.org under a new "* Colloquialisms and Expansions" section (notes.org is read every startup, so the agent honors it without a workflow change).
+- The queue itself lives in the session anchor (session-context.org) under a "* Before-Close Queue" heading, seeded empty.
+
+THE BROADER IDEA: a "Colloquialisms and Expansions" shorthand dictionary
+A per-project (or shared) reference mapping Craig shorthand phrases to their expansion, so the agent applies them without asking. Seed entries Craig named:
+1. "the list" -> append to the Before-Close Queue (above).
+2. "tell <project> <message>" -> drop the message in that project inbox via inbox-send (python3 .ai/scripts/inbox-send.py <project> --text "..."), the sanctioned cross-project handoff, never a direct write.
+
+WHY RULESETS SHOULD CARE
+Both are cross-project by nature. "tell <project>" already rides inbox-send, which every project has. The before-close queue is a session-lifecycle concept that wrap-it-up owns, and wrap-it-up is a rulesets-owned synced workflow. If this becomes a shared norm, the durable wiring is: (a) a colloquialisms reference shipped in the template (or a protocols.org section), and (b) a wrap-it-up step that processes the Before-Close Queue before teardown. Home cannot wire (b) durably from downstream because wrap-it-up.org is synced, so it is stubbed via the notes.org norm for now and needs the canonical change to stick.
+
+RECOMMENDATION FROM CRAIG: roll this shorthand out to the other projects.
diff --git a/docs/design/2026-07-20-signal-pager-runbook.org b/docs/design/2026-07-20-signal-pager-runbook.org
new file mode 100644
index 0000000..f31ed18
--- /dev/null
+++ b/docs/design/2026-07-20-signal-pager-runbook.org
@@ -0,0 +1,198 @@
+#+TITLE: Signal Pager Runbook
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-20
+
+The operational reference for the agent pager — how a page reaches Craig's
+phone, how his replies come back, how the account stays healthy, and the
+signal-cli setup behind it. This is the Signal successor to the retired ntfy
+runbook. Canonical home is rulesets because the pager is cross-machine tooling.
+
+* What the pager is
+
+One Signal identity, =+15045173983=, registered in *velox's* signal-cli
+(account file 465310, velox is the primary device). It is a dedicated pager
+number, not Craig's personal Signal. Pages go *from* that identity *to* Craig's
+own Signal account, which fires a normal mobile push on his phone.
+
+As of 2026-07-20 the identity spans two devices: velox (primary) and ratio
+(linked device "ratio-pager"). Any machine holding the account sends directly;
+a machine that doesn't relays to velox over the tailnet.
+
+Two constants the tooling depends on:
+
+- Pager account: =+15045173983= (primary on velox, linked on ratio).
+- Recipient: Craig's Signal account UUID =b1b5601e-6126-47f8-afaa-0a59f5188fde=.
+ His phone *number* reads as unregistered in Signal's directory — always
+ target the UUID, never the number.
+
+velox is the laptop that travels with Craig, so the pager account rides with
+him; ratio holds it too, so a page still lands when velox is down.
+
+* Choosing a channel
+
+Two trigger words, two channels, and both work from any agent runtime (nothing
+here is Claude-specific). protocols.org "Reaching Craig" is the short version
+pointed at every project; this runbook is the full one for the Signal side.
+
+- *"page me"* — desktop notification, stays up until dismissed:
+
+ #+begin_src bash
+ notify info "Title" "Message" --persist
+ #+end_src
+
+- *"text me"* — the phone, over Signal:
+
+ #+begin_src bash
+ agent-text "Message for Craig's phone"
+ #+end_src
+
+- *"text and page me"* — both. The default when a run can't tell whether he's
+ away: the desktop one is free and the phone one reaches him if he is.
+
+* Sending a text
+
+=agent-text= (shipped at =claude-templates/bin/agent-text=, installed to
+=~/.local/bin= by =make -C ~/code/rulesets install=) is the interface. It hides
+the machine topology by checking whether the account is registered in the
+local signal-cli:
+
+- If the account is local (velox's primary or a linked device like ratio), it
+ sends directly — no velox dependency.
+- Otherwise it ssh-relays the send to velox over the tailnet.
+- On failure (velox down or unreachable from a non-linked machine) it prints the
+ desktop fallback line and exits non-zero, so a caller can tell the page did not
+ land.
+
+The raw command it runs, for reference or a manual send from velox:
+
+#+begin_src bash
+signal-cli -a +15045173983 send -m "your message" b1b5601e-6126-47f8-afaa-0a59f5188fde
+#+end_src
+
+From another machine, the same send relayed over the tailnet:
+
+#+begin_src bash
+ssh velox.tailf3bb8c.ts.net \
+ "signal-cli -a +15045173983 send -m 'your message' b1b5601e-6126-47f8-afaa-0a59f5188fde"
+#+end_src
+
+Prefer =agent-text= over the raw command — it hardens the message for the remote
+shell and handles the fallback. Reach for the raw form only when debugging.
+
+* Reading replies
+
+Craig replies to a page straight from Signal on his phone. The reply is a normal
+data message *to* the pager account, so it is waiting in the pager's inbound
+queue until something receives it.
+
+Drain the queue and read what is there:
+
+#+begin_src bash
+# On velox:
+signal-cli -a +15045173983 receive --timeout 10
+# From another machine:
+ssh velox.tailf3bb8c.ts.net "signal-cli -a +15045173983 receive --timeout 10"
+#+end_src
+
+=receive= prints every queued envelope and exits 0 once the queue drains or the
+timeout elapses. A text reply from Craig arrives as an envelope from his UUID
+carrying a =Body:= line — that line is the reply text. Most envelopes are
+delivery/read receipts and typing indicators (no =Body:=); the reply you want is
+the data message with body text. For a script that waits on a reply, add
+=--send-read-receipts= so his phone shows the page was read, and parse stdout for
+the =Body:= line on an envelope from =b1b5601e-…=.
+
+Note: =receive= is destructive — it consumes the queue. Whatever drains the
+queue (an on-demand read, or the warm-keeping timer below) is what sees the
+reply, and it is seen once. An agent that pages and then waits for an answer
+should do its own =receive= rather than race the timer.
+
+* Keeping the account warm (receive timer)
+
+Signal expects a registered account to receive regularly. Left alone, the pager
+account drifts stale — signal-cli warns "Messages have been last received N days
+ago" (observed at 47 days on 2026-07-20 before a manual drain reset it). A stale
+account is a reliability risk on the one channel that reaches Craig when he is
+away.
+
+The fix mirrors roam-sync: a systemd user timer that drains the queue on a
+cadence, keeping the account warm and, as a bonus, picking up async replies. With
+the account linked on both machines, each device wants its own regular receive,
+so the timer runs on *both* velox and ratio (the shared =common= dotfiles
+package, same home as roam-sync).
+
+- Script: =scripts/signal-receive.sh= (rulesets, so both machines get it on
+ =git pull=). It no-ops cleanly on a machine that lacks the account.
+- Units: =scripts/signal-receive.service= + =.timer= (reference copies under
+ =scripts/systemd/=; the stowed copies live in =common/.config/systemd/user/=
+ of the dotfiles repo, so both machines get them).
+- Cadence: every 15 minutes (=OnUnitActiveSec=15min=), matching roam-sync.
+
+Enable on each machine (one-time, per daily-drivers.md's one-time-setup class):
+
+#+begin_src bash
+# After the dotfiles + rulesets pull, on each daily driver:
+systemctl --user daemon-reload
+systemctl --user enable --now signal-receive.timer
+systemctl --user status signal-receive.service # confirm a clean receive
+#+end_src
+
+* signal-cli setup notes
+
+- *Version:* signal-cli 0.14.5 on velox (2026-07-20).
+- *Accounts:* velox's signal-cli holds the pager identity =+15045173983= as the
+ registered primary (account file 465310). ratio's signal-cli holds two
+ accounts: Craig's personal number =+15103169357= (its own primary,
+ note-to-self only — no phone push) and the pager identity as a *linked device*
+ (Device 2, "ratio-pager", linked 2026-07-20). Both accounts coexist; target
+ the pager with =-a +15045173983=. A future daily driver joins the same way.
+- *signal-mcp:* on velox, Claude sessions may also expose a =signal-mcp= tool
+ (=send_message_to_user=, same pager identity) configured in velox's global
+ =~/.claude.json=. It works there but is invisible from any other machine and
+ from non-Claude runtimes, so =agent-text= is the portable habit. The old
+ =page-signal= shell script was removed 2026-06-12 — do not resurrect it.
+- *Linking a device:* to add a second signal-cli as a linked device of the pager
+ account (see the open decision below), provision it from the new machine and
+ approve the link from the account holder:
+
+ #+begin_src bash
+ # On the new machine — prints a tsdevice:/ URI (render as QR to approve):
+ signal-cli link -n "ratio-pager"
+ # Approve from velox (the primary device):
+ signal-cli -a +15045173983 addDevice --uri "tsdevice:/?uuid=…"
+ #+end_src
+
+* Decision — linked device (2026-07-20)
+
+The topology question — ssh-relay only vs. registering daily drivers as linked
+devices — was decided in favor of linked devices. ssh-relay only was simpler
+(one identity, one receive point) but had a single point of failure: a page
+failed when velox was down or off the tailnet.
+
+Registering ratio as a linked device removes that: ratio sends directly, so a
+page lands even when velox is down. The costs, both paid: linked-device
+provisioning per machine (the =link= / =addDevice= handshake above), and each
+device wanting its own regular =receive= — so the warm-keeping timer moved from a
+velox-only home to the shared =common= package, running on both.
+
+Adding another daily driver later is the same handshake plus a dotfiles stow;
+the timer and =agent-text= already generalize to "any machine holding the
+account."
+
+* History
+
+- 2026-07-04 — home retired ntfy (self-hosted on ratio) and tore it down,
+ switching agent paging to Signal. Handoff to rulesets to document and own.
+- 2026-07-13 — reconciled to one pager identity on velox; =agent-page= shipped
+ (direct on velox, ssh-relay elsewhere, desktop fallback); protocols.org "Paging
+ Craig" rewritten around the two channels.
+- 2026-07-20 — this runbook; receive-timer script + units added; a manual drain
+ cleared the 47-day staleness live; ratio linked as a device of the pager
+ account and its direct send verified; the tool (still named =agent-page= that
+ morning) generalized to send directly from any machine holding the account;
+ receive timer moved to the shared =common= package and enabled on both machines.
+- 2026-07-20 (later) — notification vocabulary split: "page me" is the desktop
+ channel, "text me" is Signal, "text and page me" is both. The tool was renamed
+ =agent-page= → =agent-text= to match, with a deprecated =agent-page= shim
+ delegating to it. protocols.org section renamed "Paging Craig" → "Reaching
+ Craig".
diff --git a/docs/design/task-review.org b/docs/design/task-review.org
index 6c6dac7..c9ae023 100644
--- a/docs/design/task-review.org
+++ b/docs/design/task-review.org
@@ -1,5 +1,5 @@
#+TITLE: Design: Daily Task-Review Habit
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-05-16
#+OPTIONS: toc:nil num:nil
diff --git a/docs/design/wrapup-routing-spec.org b/docs/design/wrapup-routing-spec.org
deleted file mode 100644
index 0091806..0000000
--- a/docs/design/wrapup-routing-spec.org
+++ /dev/null
@@ -1,181 +0,0 @@
-#+TITLE: Wrap-Up Inbox/Transcript Routing — Spec
-#+AUTHOR: Craig Jennings
-#+DATE: 2026-06-13
-#+TODO: TODO | DONE SUPERSEDED CANCELLED
-
-* Metadata
-| Status | ready for review |
-|----------+-----------------------------------------------------|
-| Owner | Craig Jennings |
-|----------+-----------------------------------------------------|
-| Reviewer | Codex (spec-review) |
-|----------+-----------------------------------------------------|
-| Related | [[file:../../todo.org][todo.org: wrap-up routing task]] · [[file:2026-06-13-wrapup-inbox-transcript-routing-proposal.org][archsetup proposal]] |
-|----------+-----------------------------------------------------|
-
-* Summary
-
-At wrap-up, an inbox handoff that belongs to another project has nowhere to go but the current project's =todo.org= or a deferral. This adds an optional routing step to =wrap-it-up.org=: surface the items that belong elsewhere, recommend a destination project for each, and move the whole batch there on one confirmation. A parallel step files meeting-transcript recordings into the right project's =assets/=.
-
-* Problem / Context
-
-=process-inbox.org= dispositions each handoff as act / fold / file / reject, and "file as TODO" lands the task in the *current* project's =todo.org=. When the real home is a different project, the choices today are: file it locally and let it rot in the wrong tracker, hand-edit two projects' =todo.org= files, or defer it and carry the debt to next session.
-
-The wrap-up's existing Step 3 "Inbox sanity check" only counts unprocessed items and blocks the wrap until they clear. It answers "is the inbox clean?" — it doesn't route anything.
-
-Meeting transcripts have the same homelessness: a recording dropped during a session belongs in some project's =assets/=, but nothing moves it there at wrap.
-
-The friction is small per-item but recurring, and the manual cross-project edit is error-prone (two files, two repos, easy to leave one half-done).
-
-* Goals and Non-Goals
-
-** Goals
-- At wrap-up, surface inbox items (and transcripts) whose home is a different project, with a recommended destination each.
-- Move the whole batch on one confirmation ("go with recommendations") or leave it entirely ("skip"). No per-item triage.
-- Move a task/event item as a proper org task into the destination's "Open Work" section per =todo-format.md=; move a transcript as a flat-filed artifact per =working-files.md=.
-- Keep the move atomic and visible (it shows in the destination's next git diff, with a provenance note).
-- Discover any project with a =todo.org= as a candidate destination, not only =.ai/= projects.
-
-** Non-Goals
-- Not a wrap gate. A skip is a clean, complete wrap.
-- Not per-item triage. The interaction is batch-level: go or skip.
-- Not a replacement for =process-inbox.org='s value gate. Routing assumes the item is already an accepted keeper.
-- Not a confidence-free auto-mover. A low-confidence destination recommendation says so, and the batch "go" stays trustworthy because the surfaced list is reviewable before the keystroke.
-
-** Scope tiers
-- v1: task/event routing to a destination project's =todo.org=. The interaction, the recommendation engine, the atomic move helper, the widened project discovery.
-- Out of scope: per-item destination editing, an interactive correction loop, moving items that aren't accepted keepers.
-- vNext: meeting-transcript filing (gated on the unresolved source-location decision and the file-vs-file+extract question — see Decisions).
-
-* Design
-
-** User-facing (the wrap interaction)
-
-The router is a new sub-step of =wrap-it-up.org='s Step 3, running after the existing inbox sanity check. Its input is filed keepers, not raw inbox files (decision: Reading B): tasks =process-inbox= accepted and filed into the local =todo.org= this session whose inferred home is a different project. When the router finds such a keeper, it surfaces it in a list, one line each: the task, the recommended destination project, and a confidence marker when the inference is weak. Then two options, batch-level:
-
-1. Go with the recommendations — apply every recommended move.
-2. Skip — leave the whole batch in place. A skip is a clean wrap.
-
-That is the entire interaction. No per-item walk. The surfaced list is the review surface; the single keystroke is trustworthy because the list was reviewable and low-confidence recommendations flagged themselves.
-
-A move of a task/event relocates it into the destination project's "Open Work" section as a proper org task (terse heading, body for detail, tags on the heading line, per =todo-format.md=), and removes it from the source. A skipped or unroutable item stays where it is; the existing sanity check still governs whether the wrap is clean.
-
-** Implementer (the mechanics)
-
-*Candidate set (what the router considers).* Reading B means the router does not scan the whole local backlog — it would otherwise suggest moving legitimate local tasks every wrap. The candidate set is keepers =process-inbox= filed this session whose inferred home differs from the current project. How those are marked is an implementation detail for Phase 3/4: either =process-inbox= tags a cross-project-candidate keeper at file time, or the router infers from a =CREATED= stamp dated this session plus content. The reviewer should pin which; the design constraint is "session-filed inbox keepers only, never the standing backlog."
-
-*Destination discovery.* Widen the project-discovery filter from "directory with a =.ai/protocols.org= marker" (what =inbox-send.py= and the =ai= launcher use) to "directory with a =todo.org= containing a level-1 'Open Work' heading." A plain code repo Craig keeps a =todo.org= in is a valid destination; an =.ai/= directory is not required.
-
-*Destination anchor.* Reuse =todo-cleanup.el='s existing matcher: =tc--find-section= locates the unique level-1 heading containing "Open Work" (case-insensitive) and returns =nil= / ='multiple= when absent or ambiguous. A destination whose =todo.org= lacks a clean Open Work heading is surfaced and skipped, never guessed at.
-
-*The move helper.* A small tool inserts a task subtree under a named project's "Open Work" heading and removes the source atomically — extend =todo-cleanup.el= (it already owns the section matcher and the subtree-move logic for =--archive-done=) or add a sibling =.ai/scripts= tool. Hand-editing across two repos is the error-prone path this replaces.
-
-*Recommendation engine.* Infer the destination from the item's content — project names, file paths, topic words — matched against the discovered project list. Conservative by design: a weak match is labeled low-confidence so "go" stays a safe single keystroke. The engine is the interesting, uncertain part; it earns the spec.
-
-*Cross-project write discipline.* Moving an item into project X's =todo.org= writes into X's scope (=cross-project.md=). The batch "go" authorizes it, but the move stays visible (X's next git diff) and leaves a one-line provenance note on the moved task naming the source project.
-
-* Alternatives Considered
-
-** Per-item triage instead of batch go/skip
-- Good, because it gives precise control over each destination.
-- Bad, because it taxes the common case (a batch that's all-correct, or all-stay) with a walk. Craig explicitly asked for two options, not a triage loop.
-- Neutral, because per-item correction could return as a vNext refinement if batch-only proves too blunt.
-
-** Fold the router into the existing Inbox sanity check step
-- Good, because one inbox step is simpler than two.
-- Bad, because the sanity check *gates* the wrap (blocks until clean) and the router is *optional* (skip is clean). Merging a blocking check with an optional action muddies both.
-- Neutral, because the two share discovery code while staying separate steps. (Resolved: D1 keeps them separate, with the router acting on filed keepers rather than inbox files.)
-
-** Reuse process-inbox's "file as TODO" with a destination argument
-- Good, because it avoids a second mechanism.
-- Bad, because =process-inbox= runs per-item mid-session against the local project; the router runs at wrap, batch-level, cross-project. Different cadence, different scope.
-- Neutral, because both ultimately call the same atomic move helper — the helper is the shared primitive, the two callers stay distinct.
-
-* Decisions [6/6]
-
-** DONE Reuse the Open Work matcher for destination anchoring
-- Context: the move needs a reliable insertion point in the destination =todo.org=; guessing risks corrupting another project's file.
-- Decision: We will reuse =todo-cleanup.el='s =tc--find-section "open work"= matcher, which already handles the unique / missing / ambiguous cases, and skip+surface any destination without a clean Open Work heading.
-- Consequences: easier — no new parser, consistent with =--archive-done=. Harder — destinations must carry the "Open Work" heading convention, so a project with a differently-named section is silently unroutable until it conforms.
-
-** DONE Move atomically through a helper, never hand-edit two repos
-- Context: a move touches two files in two repos; a half-done move loses or duplicates a task.
-- Decision: We will route every move through one helper (extend =todo-cleanup.el= or a sibling =.ai/scripts= tool) that inserts under the destination's Open Work heading and removes the source as one operation.
-- Consequences: easier — no partial-move corruption, one place to test. Harder — a new helper to build and cover with tests before the router can ship.
-
-** DONE Cross-project writes stay visible and carry provenance
-- Context: writing into another project's =todo.org= crosses the =cross-project.md= scope boundary.
-- Decision: We will treat the batch "go" as the authorization, leave the move visible in the destination's git diff, and stamp a one-line provenance note (source project + date) on each moved task.
-- Consequences: easier — the boundary rule is honored without a per-move prompt. Harder — the destination's next session sees an externally-authored task it didn't file, so the provenance note is load-bearing, not decorative.
-
-** DONE Separate router step, operating on filed keepers (Reading B)
-- Context: the sanity check gates the wrap on inbox/ contents; the router is optional. The deeper question was the router's input — raw inbox files (Reading A, which overlaps the sanity check) or already-filed keepers that belong elsewhere (Reading B, a todo-routing concern).
-- Decision: We will keep the router a separate optional sub-step after the sanity check, and its input is Reading B: accepted keepers process-inbox filed into the local =todo.org= whose inferred home is another project. The sanity check stays a pure inbox gate; the router is a todo-routing action that shares only the destination-discovery code.
-- Consequences: easier — each step has one job, the gate can't be muddied by an optional action, and the router never competes with the inbox gate over the same files. Harder — the candidate set (which local tasks the router considers) needs a marking mechanism (see the Implementer "candidate set" note); Reading A's "dispose raw inbox files at wrap" convenience is given up.
-
-** DONE Transcript routing deferred to vNext
-- Context: transcripts file as artifacts, not tasks, and a meeting usually produces both a recording to keep and action items to track. Two unknowns block it: where recordings accumulate (a recordings inbox, a downloads dir, wherever the meeting tooling drops them), and whether filing should also extract action items into the destination's =todo.org=.
-- Decision: We will defer transcript routing to vNext. Both the source-location dependency and the file-only-vs-extract-action-items question are deferred with it, to be settled when the vNext work is specced. v1 ships task routing only.
-- Consequences: easier — v1 isn't blocked on the unresolved source location. Harder — until vNext, a meeting recording still has no automatic home; only its action items (if filed as tasks) route through v1.
-
-** DONE Keep defer-and-stage and the router as distinct policies
-- Context: the 2026-06-12 Skeptical Review added a defer-and-stage path in =process-inbox.org= that files a =[#B]= VERIFY for shared-asset proposals parked for review. That also turns an inbox item into a =todo.org= task — overlapping surface with this router.
-- Decision: We will keep them distinct. Defer-and-stage parks a proposal-under-review locally as a VERIFY; the router moves an accepted keeper to its home project as a TODO. They differ on review status (proposal vs accepted) and destination (local vs cross-project), and share only the atomic move helper, not the policy. Reading B makes the split clean: the router acts on accepted keepers, never on proposals under review.
-- Consequences: easier — two clear, non-competing policies on one shared primitive. Harder — the workflow prose must name the boundary so a future reader doesn't collapse them and reintroduce the ambiguity.
-
-* Implementation phases
-
-** Phase 1 — Widened project discovery
-A discovery function returning every project with a =todo.org= that has a clean Open Work heading, reusing =tc--find-section=. Unit-tested against fixtures: =.ai/= project, plain-code-repo-with-todo, todo-without-Open-Work (excluded), ambiguous-Open-Work (excluded). Leaves the tree working — nothing calls it yet.
-
-** Phase 2 — Atomic cross-project move helper
-Extend =todo-cleanup.el= (or sibling tool) with a "move this subtree into project X's Open Work" operation that inserts at the destination and removes the source as one step, stamping the provenance line. ERT coverage: successful move, missing-destination-heading refusal, source-removal-on-success, no-partial-move-on-failure.
-
-** Phase 3 — Recommendation engine + candidate-set marking
-Infer destination from item content against the discovered list, with a confidence label. Pure function over (item, project-list) → (destination, confidence). Unit-tested: strong match (project named in item), weak match (topic-only → low-confidence), no match (stays put). Also settle the candidate-set marking (tag at file time vs CREATED-this-session inference) so the router considers only session-filed inbox keepers, never the standing backlog.
-
-** Phase 4 — Wrap-up step wiring
-Add the router sub-step to =wrap-it-up.org= Step 3: surface the batch, the two options, apply-on-go via the Phase 2 helper. Per the D1/D5 decisions once settled. Sync the =.ai/= mirror.
-
-** Phase 5 — Transcript routing (vNext, gated on the transcript decision)
-Only after the transcript-scope decision resolves. File a recording into the destination =assets/= per =working-files.md=, batch go/skip mirroring the task router.
-
-* Acceptance criteria
-- [ ] At wrap, an inbox item naming another project is surfaced with that project as the recommended destination.
-- [ ] "Go" moves every recommended item into its destination's Open Work section as a valid org task with a provenance line, and removes it from the source.
-- [ ] "Skip" leaves every item in place and the wrap completes cleanly.
-- [ ] A destination =todo.org= without a clean Open Work heading is surfaced and skipped, never corrupted.
-- [ ] A low-confidence recommendation is visibly labeled in the surfaced list.
-- [ ] A plain code repo with a =todo.org= (no =.ai/=) is a valid destination.
-- [ ] A failed move leaves both source and destination unchanged (no partial move).
-
-* Readiness dimensions
-- Data model & ownership: items are org subtrees; the destination owns the moved task after the move (provenance note records origin). N/A for remote/cached state — all local files.
-- Errors, empty states & failure: missing/ambiguous Open Work heading → skip+surface; failed move → atomic no-op; empty routable set → router stays silent (no prompt).
-- Security & privacy: N/A — local org files, no credentials or external services.
-- Observability: the move shows in the destination's git diff plus the provenance line; the surfaced batch list is the pre-move view.
-- Performance & scale: bounded by inbox size (single digits) and project count (tens); no hot path.
-- Reuse & lost opportunities: reuses =tc--find-section= and todo-cleanup's subtree-move; widens existing discovery rather than adding a parallel one.
-- Architecture fit & weak points: the recommendation engine is the weak point (a wrong-confident destination is the worst failure) — mitigated by the confidence label and reviewable batch list.
-- Config surface: possibly a discovery-root list (defaults to =~/projects/=, =~/code/=, matching =inbox-send.py=). Name it if it needs to be user-visible.
-- Documentation plan: =wrap-it-up.org= step prose; a note in =cross-project.md= that the router is a sanctioned cross-project write path.
-- Dev tooling: ERT for the elisp helper + discovery; the existing =make test= picks up new test files by glob.
-- Rollout, compatibility & rollback: additive workflow step; rollback is removing the sub-step. No persisted-data migration.
-- External APIs & deps: none.
-
-* Risks, Rabbit Holes, and Drawbacks
-- *Recommendation accuracy is the rabbit hole.* A confidently-wrong destination silently files a task in the wrong project. Dodge: keep the engine conservative, label low confidence, and keep the batch list reviewable before the keystroke. Don't chase a clever inference model in v1.
-- *Two inbox-touching steps* (sanity check + router) risk reading as redundant. Dodge: the D1 decision states the gate-vs-optional split in the workflow prose.
-- *Scope creep into transcripts* before the source-location question is answered would stall v1. Dodge: transcripts are explicitly vNext behind decision D4.
-
-* Review and iteration history
-
-** 2026-06-13 Sat @ 01:23:13 -0500 — Claude Code (rulesets) — author
-- What: initial draft. Problem, goals/scope tiers, two-altitude design, alternatives, six decisions (three DONE from grounding, three TODO for Craig), five implementation phases, acceptance criteria, readiness dimensions, risks.
-- Why: the archsetup 2026-06-13 handoff cleared the spec bar in inbox triage and was filed spec-bound rather than applied. This draft turns the proposal into a reviewable design with the open questions isolated as decision tasks.
-- Artifacts: proposal source at =docs/design/2026-06-13-wrapup-inbox-transcript-routing-proposal.org=; grounded against =wrap-it-up.org= Step 3, =todo-cleanup.el= =tc--find-section=, and =inbox-send.py= discovery.
-
-** 2026-06-13 Sat @ 01:36:28 -0500 — Craig Jennings + Claude Code (rulesets) — author
-- What: resolved all three open decisions. The router's input is Reading B (filed keepers that belong elsewhere, not raw inbox files), so D1 keeps it a separate sub-step from the inbox gate and D5 keeps it distinct from the defer-and-stage router; D4 defers transcript routing to vNext. Reworked the design (input definition, a candidate-set note bounding the router to session-filed keepers) and Phase 3 to match. Cookie now [6/6]; Status moved to ready-for-review.
-- Why: Craig chose Reading B after the A-vs-B input ambiguity surfaced as the root under D1 and D5. Reading B keeps the inbox gate, the router, and defer-and-stage each simple instead of entangling three mechanisms.
-- Artifacts: this spec; the candidate-set marking mechanism is the one detail flagged for spec-review to pin.
diff --git a/docs/specs/2026-06-16-autonomous-batch-execution-spec.org b/docs/specs/2026-06-16-autonomous-batch-execution-spec.org
new file mode 100644
index 0000000..a42adc3
--- /dev/null
+++ b/docs/specs/2026-06-16-autonomous-batch-execution-spec.org
@@ -0,0 +1,393 @@
+#+TITLE: Autonomous-Batch Task Execution — Spec
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-06-16
+#+TODO: TODO | DONE
+#+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED
+
+* IMPLEMENTED Autonomous-Batch Task Execution — Spec
+:PROPERTIES:
+:ID: 90f623cd-fdbe-4f5c-b63d-b2f84d9151cf
+:END:
+- 2026-07-02 Thu @ 05:26:07 -0400 — DOING → IMPLEMENTED: all six phases built (work-the-backlog.org, both callers, the waiver gate, checklist/Q&A/page mechanics, metrics record, KB synthesis) and the live trial validated — run c726f526, 3/3 tasks as reviewed commits with the pre-flight Q&A, page, and metrics all exercised. Craig confirmed and granted :LOOP_MAY_COMMIT:.
+- 2026-07-02 Thu @ 00:44:59 -0400 — READY → DOING: spec-response decomposition ran — the speedrun build parent in todo.org carries the :SPEC_ID: binding, one task per phase (1-6) plus the live-trial validation and the flip-to-IMPLEMENTED task. Phase 0 had already landed 2026-07-01.
+- 2026-07-02 Thu @ 00:17:01 -0400 — retrofitted by spec-sort; status set to READY (evidence-based, human-confirmed)
+
+* Metadata
+| Status | implemented |
+|----------+--------------------------------------------------------------------|
+| Owner | Craig Jennings |
+|----------+--------------------------------------------------------------------|
+| Reviewer | Craig Jennings |
+|----------+--------------------------------------------------------------------|
+| Date | 2026-06-16 |
+|----------+--------------------------------------------------------------------|
+| Related | [[file:../design/2026-06-16-inbox-zero-phase-e-proposal.org][Phase E proposal]]; [[file:../design/2026-06-15-fix-speedrun-workflow-proposal.org][speedrun proposal]] |
+|----------+--------------------------------------------------------------------|
+
+* Summary
+
+Two proposals arrived within a day of each other describing the same capability: have Claude work a batch of small, well-marked tasks autonomously, with a full quality bar per task and no per-step approval gate. The inbox-zero "Phase E" proposal drives it from a tag/priority query on a recurring loop; the "speedrun" proposal drives it from an explicit ordered list a human dictates in-session. This spec reconciles both into one feature: a single dedicated workflow, =work-the-backlog.org=, that holds the task-execution logic, with two thin callers feeding it. It also designs the instrumentation that measures whether the autonomy is actually paying off.
+
+* Problem / Context
+
+Craig has a standing backlog of small, solo-doable fixes across several projects, already marked with a tag convention (=:next:=, =:quick:+:solo:=). Doing them by hand one at a time is the bottleneck — the context-switch and the per-commit approval ceremony dominate the actual work. He wants Claude to burn these down unattended: on a recurring loop for the routed inbox case, and on demand when he batches a named list and says "speedrun, no approvals until done." The speedrun is the away-from-desk / working-on-something-else mode, so it must be able to take on larger tasks too — not only sub-30-minute ones — or it forces him to stay at the desk for anything non-trivial.
+
+Two separate proposals tried to answer this:
+
+- *Phase E* (in =inbox-zero.org=, edited in =.emacs.d= as a stopgap) bolted autonomous execution onto the inbox-zero workflow's on-demand and loop callers. The sender flagged the seam as the open question: coupling capture-routing with autonomous-implementation pollutes inbox-zero's three existing callers (startup, wrap-up, on-demand), two of which must never execute anything.
+- *speedrun* (a =.emacs.d= theme-studio session that worked well) is the same execution loop driven by an explicit ordered task set, with end-of-set paging and always-push.
+
+They overlap almost entirely. The execution loop — eligibility gate, act-vs-file decision, per-task quality bar, bounded run — is identical. Only the *input* differs (tag query vs explicit list) and the *session mode* differs (loop default vs no-approvals + always-push + page). Building them as two features would duplicate the execution logic and let the two copies drift. The forces: keep inbox-zero's callers clean, share one execution loop, and make the autonomy safe enough to run unattended on a 30-minute timer without Craig watching.
+
+A second, explicit ask from Craig: instrument this so its effectiveness is measurable. "Gather data on this and create some org-roam articles we can look at later." Autonomous execution that silently makes bad commits is worse than no autonomy; the only way to know which it is, is to measure tasks completed vs deferred vs reverted, and human corrections in the following session, over time.
+
+* Goals and Non-Goals
+
+** Goals
+- One workflow, =work-the-backlog.org=, owns the task-execution loop. Both input shapes (tag query, explicit list) and both session modes feed it.
+- inbox-zero's three existing callers stay clean: the loop caller chains into =work-the-backlog= *after* routing; startup and wrap-up never touch it.
+- The *no-approvals speedrun* is a thin named preset, not a second implementation: autonomous-commit + always-push + end-of-set page, fed an explicit ordered list, with all approvals front-loaded into a single pre-flight step (below) so the run itself is uninterrupted.
+- Eligibility is decided by *crisp, checkable criteria*, not adjectives: a mechanical tag/status gate (=:solo:= + status =TODO=), then a per-task defer checklist whose keystone is "can I write the failing test from the task text without inventing a requirement?" Task *size* is explicitly not a gate — a large task is decomposed into per-logical-commit chunks, not deferred.
+- The autonomy tags (=:solo:=, =:quick:=) carry hard definitions in =todo-format.md= and are applied + enforced as a mandatory step in the task-review and task-audit workflows, so the run-time gate trusts the author's tag instead of re-deriving it.
+- Commit autonomy defaults to file-only (surface a diff, no auto-commit). A project opts into autonomous commit+push explicitly via its per-project waiver.
+- Hard guardrails: refuse any task carrying data-loss / irreversible / external-state risk without a checkpoint; gather any one-or-two quick decisions a task needs *up front* (speedrun) rather than guessing; file a =VERIFY= for anything underspecified or needing design deliberation; a per-run cap / kill switch beyond "one task per run."
+- A lightweight per-run metrics log plus a periodic synthesis step that writes org-roam KB articles summarizing the trend.
+
+** Non-Goals
+- *Not* a replacement for =/start-work=. Tasks needing deliberation or design stay with =/start-work= and its approval gates. This feature only touches the marked, solo set — regardless of size.
+- *Not* a new tag convention. It reads the project's own priority/tag scheme header; it never invents or hardcodes tags across projects.
+- *Not* an inbox-routing change. =inbox-zero.org= keeps its A-D phases. The Phase E text added in =.emacs.d= as a stopgap is *removed* and its logic moves here.
+- *Not* a multi-project orchestrator. One run works one project's backlog. Cross-project handoff stays with =inbox-send= and the paging reply.
+- *Not* a credential-handling or external-API feature. Tasks that touch secrets or external mutations are out of the eligible set by the guardrail.
+
+** Scope tiers
+- *v1:* =work-the-backlog.org=; crisp =:solo:= / =:quick:= definitions in =todo-format.md= plus their mandatory application in task-review and task-audit; the eligibility gate (=:solo:= + status =TODO=, read against the project's scheme header); the act-vs-file *defer checklist* (test-writability keystone, enumerated data-loss list, already-satisfied, design-deliberation); the no-approvals speedrun's pre-flight decision-gathering step; file-only commit default with per-project opt-in; the loop caller wiring and inbox-zero Phase E removal; the speedrun preset with end-of-set =notify --persist= page; the per-run metrics log (structured JSONL).
+- *Out of scope:* a token-budget kill switch (cap is a task count in v1); cross-project batch runs; a dashboard or live UI over the metrics.
+- *vNext (log to todo.org):* the periodic org-roam synthesis step if it doesn't make v1; a token/cost budget alongside the task-count cap (more pressing now that task size is uncapped — a single large task can run long in the unattended loop); auto-detection of "human corrected my autonomous commit" from the next session's diff.
+
+* Design
+
+** Overview
+
+The architecture is one execution workflow with two callers and one preset, plus an instrumentation sidecar.
+
+#+begin_example
+ inbox-zero loop caller ──(after Phase D routing)──┐
+ ├──▶ work-the-backlog.org ──▶ metrics log (JSONL)
+ no-approvals speedrun ──(explicit ordered list)──┘ │
+ = pre-flight Q&A + autonomous-commit + push + page ▼
+ periodic synthesis ──▶ org-roam KB articles
+#+end_example
+
+=work-the-backlog.org= is the only place the execution loop lives. It takes a *task set* (however assembled) and a *session mode* (which gates commit autonomy and paging), and works the set under a fixed safety contract. The two callers differ only in how they build the task set and which session mode they pass.
+
+This is the seam the Phase E sender asked for: separating capture-routing (inbox-zero) from autonomous-implementation (work-the-backlog) keeps inbox-zero's startup and wrap-up callers — which must never execute anything — untouched. The loop caller is the only one of inbox-zero's callers that chains forward into execution, and it does so as an explicit second step after routing completes, not as a phase buried inside inbox-zero.
+
+** The execution loop (two-altitude: caller's view)
+
+A caller hands =work-the-backlog= three things:
+
+1. *A task set* — either an explicit ordered list of task headings (speedrun), or the result of a tag/priority query against =todo.org= (the loop). The workflow does not care which; it receives an ordered list of candidate tasks.
+2. *A session mode* — =file-only= (default) or =autonomous-commit= (requires the project's per-project waiver), and a paging flag.
+3. *A run cap* — the maximum number of tasks to complete this run.
+
+It returns: per-task outcome (implemented+committed / implemented+diff-surfaced / deferred-VERIFY / dropped-by-craig / skipped-ineligible), and a metrics record per task.
+
+** The execution loop (implementer's view)
+
+For the task set, in order, until the run cap is hit:
+
+1. *Eligibility gate* (below). Ineligible → record =skipped-ineligible=, next task.
+2. *Scope read* of the relevant code. Cheap; just enough to run the defer checklist.
+3. *Defer checklist* (below). Any hit → record the deferral reason (or, under the speedrun preset, route the quick-question gap to the pre-flight Q&A), next task.
+4. *Implement* under the project's commit discipline: TDD red→green→refactor, then =/review-code --staged=, fix all Critical/Important, then close the task per =todo-format.md=. Decompose into as many logical commits as the change needs — size is not capped.
+5. *Commit autonomy branch:*
+ - =file-only= → surface the diff, do *not* commit. Record =implemented-diff-surfaced=.
+ - =autonomous-commit= → =/voice personal= on the message, commit individually, push per the project's flow. Record =implemented-committed=.
+6. *Record metrics* for the task (the JSONL append, below).
+7. Decrement the cap. At zero, stop.
+
+After the set: if the paging flag is set, fire the end-of-set page (below). Surface the run summary.
+
+** Eligibility gate (mechanical — no judgment)
+
+A task is autonomous-safe when *both* hold. This layer is a lookup, not a judgment; all the judgment lives in the defer checklist below.
+
+1. *Status is =TODO=* — never =VERIFY=, =DOING=, =DONE=, or =CANCELLED=. =VERIFY= is the "awaiting Craig's manual confirmation" marker; auto-implementing one defeats the manual check it represents. The do-not-implement set is safe-by-omission: anything not plainly =TODO= (plus any project-declared "hold" marker) is out.
+2. *Tagged =:solo:=* — the autonomy tag, resolved against the project's priority/tag scheme header (not hardcoded). =:solo:= carries a hard definition (see Tag definitions, below): the task is completable without Craig's involvement beyond at most one or two quick decisions answerable up front, with no design deliberation. A project whose scheme declares a different autonomous-safe tag set overrides the default. Priority / =:next:= drive *ordering* within the eligible set, not eligibility.
+
+Task *size* is deliberately absent from this gate. The old "≤ ~30 minutes / one logical commit" criterion is removed: a large but well-specified, decision-free task is in scope and is decomposed into per-logical-commit chunks during implementation. Size never sends a task to =/start-work=; only *deliberation* or *risk* does (the checklist below). This is what makes the speedrun usable as an away-from-desk mode rather than a sub-30-minute-only mode.
+
+*** Tag definitions (land in =todo-format.md=, enforced in task-review + task-audit)
+
+- *=:solo:= — autonomy.* The task can be completed without Craig's involvement, except for at most one or two quick decisions that can be stated and answered before the run starts. No open design question, no "weigh these approaches," no waiting on Craig mid-task. This is the eligibility tag.
+- *=:quick:= — effort hint only.* A small, fast task. Informational for batching and estimating a run's duration; *not* an eligibility gate (size no longer gates).
+
+Both tags are applied at task creation and *re-checked as a mandatory step* in the task-review and task-audit workflows, so the run-time gate can trust the author's tag rather than re-derive autonomy and effort from the task body. A task-review or task-audit that skips the =:solo:= / =:quick:= assessment is incomplete.
+
+** Act-vs-file decision (the defer checklist)
+
+After the scope read, run each eligible candidate through the checklist below. Each item is a concrete, answerable question, not an adjective. *Any* hit — or any "unsure" — sends the task to defer (or, for a quick-decision gap under the speedrun preset, to the pre-flight Q&A). Only a task that clears every item is implemented.
+
+1. *Test-writability (the keystone).* Can I write the failing test from the task text — plus any decisions gathered up front — without inventing a requirement? *No / unsure* → underspecified. Under the speedrun preset, if the gap is one or two quick answerable questions, route it to the pre-flight Q&A; otherwise file a =VERIFY= noting what's missing. Under the unattended loop, file the =VERIFY= (no one to ask). This replaces the old "clear / bounded / underspecified" adjectives with an action that fails loudly: if the red test isn't writable, the task isn't ready.
+2. *Data-loss / irreversible / external operation.* Does implementing it require any of: =rm= of non-scratch data, =git reset --hard= / force-push, =DROP= / =DELETE= / =TRUNCATE=, file truncate/overwrite of persisted content, a schema or data migration, any external or shared-state mutation, any credential touch? *Yes* → do NOT implement; file a =VERIFY= naming the risk. This is the hard safety gate; an upfront answer never overrides it without an explicit checkpoint. Replaces the vague "data-loss risk" with an enumerated, greppable set.
+3. *Already-satisfied.* Does the scope read show the desired end-state already holds? *Yes* → file a =VERIFY= noting it (the "raise max spans to 5 — every cap was already 8" case) and move on. Don't make a no-op change.
+4. *Design deliberation.* Does the task carry an unresolved design question, a "weigh these approaches" with real tradeoffs, or a TBD that isn't a quick factual answer? *Yes* → under the speedrun preset, if it collapses to one or two quick questions, route to pre-flight Q&A; otherwise file and surface as a =/start-work= candidate. Under the loop, file. The discriminator is now *quick-answerable question* vs *deliberation* — not task size.
+
+A task that clears 1–4 is implemented under the project's commit discipline, decomposed into as many logical commits as the change needs. When genuinely unsure which side a task falls on, defer — a wrong auto-implement costs a revert *and* the next-session correction the metrics are designed to catch.
+
+** Pre-flight decision gathering (the no-approvals speedrun's only interaction)
+
+The speedrun preset front-loads every approval into one step before the run, so the run itself is uninterrupted — that is what "no approvals" means. It is *not* "no input ever"; it is "all input first, then hands-off."
+
+When Craig kicks off a speedrun over an explicit list:
+
+1. *Gather* the named task set.
+2. *Scope-read and classify* each task against the eligibility gate + defer checklist: ready (clears the checklist), needs-quick-decisions (one or two upfront-answerable questions — checklist item 1 or 4), or drop (data-loss / irreversible, or design deliberation that isn't a quick question).
+3. *Order* the list (priority, then the author's ordering / =:next:=).
+4. *Intro the work* — present the ordered plan: what will run, what was dropped and why, and the batched questions for the needs-quick-decisions tasks.
+5. *Craig answers each question, or says "skip this"* → a skipped task is removed from the run (recorded =dropped-by-craig=); an answered task has the answer recorded so implementation works from the decision, not a guess.
+6. *Run the finalized list autonomously* — no further approvals until done.
+7. *End-of-set page* with completed + remaining + skipped.
+
+The unattended *loop* caller has no human at kickoff, so it cannot gather decisions: there, a needs-quick-decisions task simply defers (files its note) like any other checklist hit. The pre-flight Q&A is a speedrun-preset capability, not a loop one.
+
+** Session modes and the no-approvals speedrun preset
+
+Two orthogonal session-mode dimensions feed the loop:
+
+- *Commit autonomy:* =file-only= (default) or =autonomous-commit=. =autonomous-commit= is honored only when the project carries the per-project waiver (=.emacs.d= and =rulesets= have it; most projects do not). Absent the waiver, a request for =autonomous-commit= degrades to =file-only= and says so.
+- *Paging:* on or off. End-of-set only.
+
+The *no-approvals speedrun* is the named preset = =autonomous-commit= + always-push + paging-on, fed an *explicit ordered list*, run after the pre-flight decision-gathering step above. It is not a separate code path; it is a label for that combination of mode flags plus the explicit-list input, with the pre-flight Q&A as its only interactive moment. The loop caller, by contrast, runs =file-only= (unless the project has the waiver and opts the loop into commits) with paging off, fed the *tag query*, with no pre-flight step.
+
+** Bounding the run and the kill switch
+
+Default cap: one task per run for the loop caller — implement the highest-priority eligible candidate (=[#A]= before =[#B]= before =[#C]=), record, then stop and let the next tick continue. The speedrun preset works the whole explicit list in order (the human bounded it by naming it), still one commit per logical change.
+
+The kill switch is a hard per-run task cap passed by the caller, independent of "one per run": even the speedrun stops at the cap and pages with the remainder listed. A loop that fires every 30 minutes and commits unattended needs a ceiling that a runaway can't exceed. With task size now uncapped, the count cap no longer bounds *cost* — a single large task can run long — so a token/cost budget is the most pressing vNext addition.
+
+** End-of-set paging
+
+When the set is done (or the cap is hit), if paging is on, fire one page — end-of-set only, never per-task:
+
+#+begin_src sh
+notify alarm "Page" "<project>: <N> done, <M> remaining — <one-line summary>" --persist
+#+end_src
+
+=--persist= keeps it on screen until dismissed (the page-me convention). The message carries the project name, the completed count, and the remaining count, so Craig can reply confirming ready + naming the next project in one turn. The page-signal wrapper removed 2026-06-12 is reconciled to =notify= here — there is no separate page-signal call.
+
+* Alternatives Considered
+
+** Fold execution into inbox-zero (the Phase E stopgap shape)
+- Good, because it's the smallest diff — the loop caller already runs inbox-zero, so execution is "one more phase."
+- Bad, because it couples capture-routing with implementation. inbox-zero has three callers; startup and wrap-up must never execute. A Phase E inside inbox-zero forces both to carry a "skip Phase E" caveat and risks a future caller running it by accident.
+- Neutral, because the eligibility-gate and defer-checklist text is identical either way — only its *home* differs.
+
+** Two separate features (keep Phase E and speedrun distinct)
+- Good, because each proposal ships as written with no reconciliation work.
+- Bad, because the execution loop is duplicated in two places and will drift; a guardrail tightened in one won't reach the other. Two ways to do autonomous execution is two things to audit.
+- Neutral, because the input and session-mode differences are real — but they're thin caller-level differences, not a reason to fork the engine.
+
+** Keep the task-size gate (defer anything over ~30 minutes)
+- Good, because it bounds per-task cost and blast radius with a single number.
+- Bad, because it defeats the away-from-desk use case — anything non-trivial bounces back to Craig, so he can't actually leave. Size correlates poorly with risk; a large mechanical refactor is safer than a tiny change to persisted state.
+- Neutral, because the things size was a proxy for (risk, cost) are covered directly — risk by the data-loss checklist, cost by the run cap (and the vNext token budget). The defer checklist's deliberation item, not size, is what routes genuine =/start-work= tasks out.
+
+** Autonomous-commit as the default
+- Good, because it's faster end-to-end with no diff to review.
+- Bad, because most projects lack the per-project waiver, and an unattended loop committing to a project that never opted in is exactly the failure the file-only default prevents. The blast radius of a bad autonomous commit is a revert plus lost trust in the loop.
+- Neutral, because the projects that *do* want it (=.emacs.d=, =rulesets=) opt in explicitly, so the capability is available where it's wanted without being the default everywhere.
+
+* Decisions [8/8]
+
+** DONE Eligibility tag set and where it's read
+- Owner / by-when: Craig / spec-review
+- Context: Projects' priority/tag schemes vary, and the =todo-format.md= scheme header is the declared per-project source of truth. Task size is no longer a gate, so eligibility rests on the autonomy tag, not an effort cap.
+- Decision: Eligibility = status =TODO= AND the =:solo:= autonomy tag, resolved against the project's scheme header (a project may declare a different autonomous-safe set). Priority / =:next:= drive ordering, not eligibility. =:quick:= is an effort hint, never a gate.
+- Consequences: easier — one workflow works across projects with different vocab, and the gate is a pure lookup; harder — a project with no/malformed scheme header needs a fallback, and the default (=:solo:=) must be defined precisely enough that two projects agree.
+
+** DONE Crisp =:solo:= / =:quick:= definitions, enforced in task-review + task-audit
+- Owner / by-when: Craig / spec-review
+- Context: The run-time gate is only as crisp as the tags. Today =:quick:= / =:solo:= are listed in the scheme header with no hard definition, and nothing enforces that tasks get assessed for them.
+- Decision: Define =:solo:= (completable without Craig beyond at most one-or-two upfront-answerable quick decisions; no design deliberation) and =:quick:= (small/fast effort hint only) in =todo-format.md=, and make assessing both a *mandatory step* in the task-review and task-audit workflows. A review/audit that skips the assessment is incomplete.
+- Consequences: easier — authoring-time judgment by the human who knows the answer, and the run-time gate trusts the tag; harder — task-review and task-audit grow a required step, and existing untagged tasks need a back-fill pass.
+
+** DONE The do-not-auto-implement marker set
+- Owner / by-when: Craig / spec-review
+- Context: =VERIFY= means "awaiting Craig's manual confirmation"; other projects may use markers differently.
+- Decision: Do-not-implement = any status that is not =TODO=, plus any project-declared "hold" marker. Safe-by-omission: exclude anything not plainly =TODO=.
+- Consequences: easier — portable, and manual-check tasks can't auto-run; harder — richer per-project overrides need marker semantics in the scheme header, which most lack, so the default must stay conservative.
+
+** DONE Pre-flight decision gathering for the speedrun preset
+- Owner / by-when: Craig / spec-review
+- Context: Forcing every decision-needing task to defer wastes the away-from-desk use case — many tasks need only one or two quick answers Craig could give at kickoff. The speedrun is interactive at its start but must be hands-off after.
+- Decision: The speedrun preset gathers + orders the set, intros the work, and batches all needed quick decisions into one pre-flight Q&A; Craig answers or says "skip this" (drops the task); the run then proceeds with zero further approvals. The unattended loop has no kickoff human, so it defers decision-needing tasks instead.
+- Consequences: easier — "no approvals" becomes "all approvals first," which fits working-while-away, and larger / lightly-underspecified tasks become runnable; harder — the classifier must reliably split quick-question vs real-deliberation, and the recorded answers must reach the implementer so it works from the decision, not a guess.
+
+** DONE Commit-autonomy opt-in mechanism
+- Owner / by-when: Craig / spec-review
+- Context: =file-only= is the default; =.emacs.d= and =rulesets= have a per-project waiver allowing autonomous commits. Where does the workflow *read* that a project has opted in?
+- Decision: Read the opt-in from the project's existing per-project waiver location (=notes.org= Workflow State or =CLAUDE.md=), not a new config file. Two flags: "has commit waiver" and "loop may commit" can differ.
+- Consequences: easier — no new config surface, reuses the existing waiver concept; harder — the waiver location/format must be pinned for deterministic detection, and "waiver yes, loop-commit no" needs the two-flag split.
+
+** DONE Run-cap default and the kill switch shape
+- Owner / by-when: Craig / spec-review
+- Context: The loop default is one task per run; the speedrun works an explicit list. Both need a hard ceiling. Task size is now uncapped, so a single task can be large.
+- Decision: The caller passes a hard per-run task cap (loop default 1; speedrun = length of the explicit list, capped at a ceiling); stop + page with the remainder when the cap is hit. v1 caps by task count, not token budget.
+- Consequences: easier — a simple caller-controlled integer with a bounded task count; harder — a count cap doesn't bound *cost*, and with size uncapped a single large task can run long, so a token budget is vNext and more pressing than before.
+
+** DONE Metrics log location and format
+- Owner / by-when: Craig / spec-review
+- Context: Per-run metrics must land somewhere structured and queryable, per-project, and survive across sessions for the synthesis step to read.
+- Decision: Append one JSONL record per task to a per-project log at =.ai/metrics/work-the-backlog.jsonl=, git-tracked, with the synthesis step reading the union across projects.
+- Consequences: easier — append-only JSONL is trivial to write and =jq=-queryable, and per-project keeps it local to the work; harder — a git-tracked log adds commit churn, and "union across projects" needs the synthesis step to know where every log lives.
+
+** DONE Synthesis cadence and trigger
+- Owner / by-when: Craig / spec-review
+- Context: Craig wants periodic org-roam articles summarizing the data. What triggers synthesis, and how often?
+- Decision: Run synthesis on an explicit trigger ("synthesize backlog metrics") and optionally a weekly scheduled run, writing one KB node per synthesis under =~/org/roam/agents/= per the knowledge-base rule.
+- Consequences: easier — an explicit trigger means no surprise writes, and the KB rule already governs node shape; harder — a weekly run needs a scheduler entry, and the personal-only write-classification must gate it so work-project metrics never land in the KB.
+
+* Implementation phases
+
+** Phase 0 — Tag definitions + task-review/audit enforcement
+Add the hard =:solo:= / =:quick:= definitions to =todo-format.md=, and add the mandatory tag-assessment step to the task-review and task-audit workflows. Independent of the workflow build; lands first so the eligibility gate has crisp tags to read and existing tasks start getting assessed. Tree stays working: these are rule + workflow prose additions.
+
+** Phase 1 — Extract the execution loop into work-the-backlog.org
+Write =work-the-backlog.org= holding the eligibility gate, defer checklist, per-task quality bar, and run-cap logic — taking a task set + session mode + cap as input. Remove the stopgap "Phase E" text from =inbox-zero.org= (restore it to its A-D shape) in the same change so there's one home, not two. Tree stays working: inbox-zero reverts to routing-only, and the new workflow is callable but not yet wired to the loop.
+
+** Phase 2 — Wire the two callers
+Add the loop caller's chain step (after inbox-zero Phase D, invoke work-the-backlog with the tag query + file-only + cap 1) and the no-approvals speedrun preset (pre-flight decision-gathering → explicit list + autonomous-commit + always-push + paging-on). Both go through the same workflow; only the speedrun runs the pre-flight Q&A. Tree stays working: each caller is independently testable.
+
+** Phase 3 — File-only vs autonomous-commit gate
+Implement the commit-autonomy branch: read the per-project waiver, degrade =autonomous-commit= to =file-only= when absent, surface the degrade. Tree stays working: default file-only behavior is the safe path even before the waiver-read lands.
+
+** Phase 4 — The defer checklist, pre-flight Q&A, and the page
+Implement the act-vs-file defer checklist (test-writability keystone, enumerated data-loss list, already-satisfied, design-deliberation), the speedrun pre-flight decision-gathering (gather → classify → order → intro → batch-ask → skip/answer), the =VERIFY=-on-ambiguity filing, and the end-of-set =notify alarm ... --persist= page. Tree stays working: the checklist only ever *reduces* what runs, and the pre-flight step only runs under the speedrun preset.
+
+** Phase 5 — Metrics log
+Append the per-task JSONL record at each task outcome. Tree stays working: logging is a side effect that doesn't alter execution.
+
+** Phase 6 — Synthesis to org-roam
+Write the synthesis step: read the JSONL union, compute the per-run and trend metrics (below), write a KB node under =~/org/roam/agents/= per the knowledge-base rule, personal-projects-only classification enforced. Tree stays working: synthesis is read-only over the logs plus a KB write.
+
+* Acceptance criteria
+- [ ] =work-the-backlog.org= exists and is the only home for the execution loop; =inbox-zero.org= is back to its A-D routing-only shape with no Phase E.
+- [ ] The loop caller chains into work-the-backlog after routing; startup and wrap-up never invoke it.
+- [ ] The no-approvals speedrun runs as the preset (pre-flight Q&A → autonomous-commit + always-push + end-page) over an explicit ordered list, one commit per logical change.
+- [ ] =:solo:= and =:quick:= carry hard definitions in =todo-format.md=, and task-review + task-audit both refuse to complete without assessing them.
+- [ ] Eligibility = status =TODO= AND =:solo:=, read from the project's scheme header, not hardcoded; a =VERIFY= / =DOING= / =DONE= / =CANCELLED= task is skipped by the gate.
+- [ ] Task size never sends a task to =/start-work=; a large but =:solo:=, well-specified task runs and is decomposed into per-logical-commit chunks.
+- [ ] The defer checklist fires correctly: a task whose red test isn't writable (and isn't a quick-question gap), one carrying an enumerated data-loss operation, an already-satisfied one, and one needing design deliberation are each deferred (or routed to pre-flight Q&A under the speedrun), not implemented.
+- [ ] Under the speedrun preset, a task needing one or two quick decisions is surfaced in the pre-flight Q&A; "skip this" drops it, an answer is recorded and used; the run then proceeds with no further approvals.
+- [ ] Under the unattended loop, a decision-needing task defers (no pre-flight Q&A).
+- [ ] In a project without the commit waiver, an =autonomous-commit= request degrades to file-only and says so; no commit is made.
+- [ ] The run stops at the per-run cap and pages with the remaining tasks listed.
+- [ ] Each task outcome appends one JSONL record to =.ai/metrics/work-the-backlog.jsonl=.
+- [ ] The synthesis step reads the logs and writes a KB node under =~/org/roam/agents/=; it refuses to write for work-classified projects.
+
+* Effectiveness measurement
+
+This section answers Craig's explicit ask: measure whether autonomous-batch execution is actually effective, and build the "gather data → org-roam articles" loop.
+
+** What "effective" means here
+
+The autonomy is effective if it completes real work that *stays* completed — i.e. tasks land green and the next session doesn't have to undo or fix them. The two failure modes to catch are (1) the loop defers everything (over-cautious, no value delivered) and (2) the loop implements badly (commits that get reverted or hand-corrected next session). Both are measurable.
+
+** Per-run metrics (the JSONL record)
+
+One record per task, appended to =.ai/metrics/work-the-backlog.jsonl= at each task outcome:
+
+| Field | Meaning |
+|-------------------+--------------------------------------------------------------------|
+| =ts= | ISO timestamp of the task outcome |
+|-------------------+--------------------------------------------------------------------|
+| =run_id= | UUID shared by all tasks in one run |
+|-------------------+--------------------------------------------------------------------|
+| =project= | project basename |
+|-------------------+--------------------------------------------------------------------|
+| =caller= | =loop= or =speedrun= |
+|-------------------+--------------------------------------------------------------------|
+| =task= | task heading (slug) |
+|-------------------+--------------------------------------------------------------------|
+| =outcome= | implemented-committed / implemented-diff / deferred-verify / |
+| | skipped-ineligible / dropped-by-craig (skipped at pre-flight) |
+|-------------------+--------------------------------------------------------------------|
+| =defer_reason= | underspecified / data-loss / already-satisfied / needs-deliberation |
+|-------------------+--------------------------------------------------------------------|
+| =upfront_decision=| true if a pre-flight answer was recorded and used for this task |
+|-------------------+--------------------------------------------------------------------|
+| =wall_clock_s= | seconds from task start to outcome |
+|-------------------+--------------------------------------------------------------------|
+| =commit_sha= | for committed tasks; empty otherwise |
+|-------------------+--------------------------------------------------------------------|
+| =review_findings= | count of /review-code Critical+Important findings on this task |
+|-------------------+--------------------------------------------------------------------|
+
+Per-run rollups computed at synthesis (not stored per record): tasks attempted, completed, VERIFY-deferred, dropped-by-craig, reverted; wall-clock total; commits landed; review findings per commit.
+
+** The corrections signal (the key metric)
+
+The hardest and most valuable metric is *human corrections in the following session* — did Craig revert or hand-fix an autonomous commit? v1 captures the cheap proxy: at synthesis, for each =commit_sha=, check whether a later commit touching the same files reverted it or carries a "fix"/"revert" of that change within N days. A clean run is one where the autonomous commits survive untouched. (Auto-detecting "this later commit corrected that autonomous one" precisely is a vNext refinement; the proxy — reverted-or-touched-soon-after — is good enough to flag a problem run for human review.)
+
+** Where the data lands
+
+Per-project git-tracked JSONL at =.ai/metrics/work-the-backlog.jsonl=. Append-only, =jq=-queryable, survives across sessions and machines via the normal project sync. Git-tracked so the history is auditable and the synthesis step can read it from any clone.
+
+** The synthesis loop (gather → article)
+
+On the "synthesize backlog metrics" trigger (and optionally a weekly scheduled run):
+
+1. Read the JSONL union across the personal projects the synthesizer can see.
+2. Compute the rollups and the trend: completion rate over time, defer-reason distribution, review-findings-per-commit trend, and the corrections-signal flag count.
+3. Write one org-roam KB node under =~/org/roam/agents/YYYYMMDDHHMMSS-backlog-metrics-<window>.org= per the knowledge-base rule — filetags =:agent:metrics:=, a concise title, the rollup table, the trend narrative, and =[[id:...]]= links to prior synthesis nodes so the series is traceable.
+4. Enforce the KB write-classification: *personal projects only*. A work-classified project's metrics never write to the KB — they stay in that project's own =.ai/metrics/= log and the synthesizer reports the refusal per the KB refusal contract.
+
+The KB node is the artifact Craig reviews later — "are the autonomous runs completing more and getting corrected less over the last month?" reads off the trend table without re-querying raw logs.
+
+* Readiness dimensions
+
+- *Data model & ownership:* The task set is read from =todo.org= (project-owned, user-authored). The metrics JSONL is generated, append-only, git-tracked, project-owned. KB nodes are agent-generated under =~/org/roam/agents/= (never overwriting Craig's hand-authored nodes — link only). No editable region is co-owned.
+- *Errors, empty states & failure:* Empty task set → report "nothing eligible" and stop. Malformed scheme header → fall back to the default tag reading and surface the fallback. A task that fails mid-implementation → leave the tree working (don't commit a broken state), record the failure outcome, surface it, continue to the next task. No silent data loss: the data-loss guardrail refuses irreversible tasks outright.
+- *Security & privacy:* Tasks touching credentials or external mutations are excluded by the data-loss / external-state checklist item. The KB write is personal-projects-only; work metrics never leave the project. No secrets in the JSONL (task slugs and SHAs only).
+- *Observability:* The end-of-set page surfaces the run outcome. The per-task surface (implemented / deferred + reason / dropped / skipped) is the live progress view. The metrics log + KB synthesis is the long-run observability. A bad run is isolable from the JSONL (which task, which outcome, which review findings).
+- *Performance & scale:* Expected counts are small — a handful of tasks per run, one run per 30-min tick. No bottleneck at this scale. The cap bounds the worst case on task count; with size uncapped, a single large task is the cost outlier the vNext token budget addresses. Synthesis over months of JSONL is still a small file (one record per task).
+- *Reuse & lost opportunities:* Reuses =todo-format.md= for task close + the tag definitions, =/review-code= and =/voice personal= for the quality bar, =notify= for paging, the knowledge-base rule for KB writes, the per-project waiver for commit-autonomy, and task-review / task-audit for tag enforcement. No new config file (the opt-in rides the existing waiver). The execution loop is the one new shared asset.
+- *Architecture fit & weak points:* Integration points — inbox-zero loop caller (chain after Phase D), the per-project waiver location, =todo.org= scheme header, task-review / task-audit, =~/org/roam/agents/=. Weak point: the commit-autonomy gate depends on deterministically reading the waiver; mitigated by defaulting to file-only when the read is ambiguous (fail safe, not open). Second weak point: a 30-min loop committing unattended with uncapped task size; mitigated by the hard count cap and file-only default, with the token budget as the vNext backstop.
+- *Config surface:* Per-project — commit-autonomy opt-in (via existing waiver), optional loop-commit flag, optional autonomous-safe tag override in the scheme header. Per-call — task set, session mode, run cap. Defaults: file-only, paging-off (loop) / paging-on (speedrun), cap 1 (loop).
+- *Documentation plan:* The workflow file itself is the user/operator doc (matches inbox-zero.org's self-documenting style). The =.emacs.d= stopgap note and the speedrun proposal are superseded by this spec; no separate migration doc needed beyond removing the Phase E text.
+- *Dev tooling:* N/A for new build targets — the workflows are prose, exercised by invocation. The metrics JSONL is =jq=-inspectable by hand; a tiny rollup helper may be added under =.ai/scripts/= if the synthesis prose proves to need it (decided at Phase 6, not a v1 prerequisite).
+- *Rollout, compatibility & rollback:* Rollout is removing Phase E from inbox-zero and adding work-the-backlog — both prose changes, instantly reversible. Compatibility: inbox-zero's three callers are unchanged except the loop caller gaining a forward chain. Rollback: delete work-the-backlog and the loop chain step; inbox-zero is already back to A-D. The file-only default means the worst pre-rollback state is surfaced diffs, not committed changes.
+- *External APIs & deps:* =notify alarm "Page" "<msg>" --persist= verified against =/home/cjennings/.local/bin/notify= and the page-me workflow. =~/org/roam/= KB write path and node shape verified against the knowledge-base rule. No external API calls.
+
+* Risks, Rabbit Holes, and Drawbacks
+
+- *The corrections signal is a proxy, not ground truth.* "A later commit touched the same files" over-counts (legitimate follow-up work) and under-counts (a correction in a different file). It's a flag for human review, not a verdict. Don't rabbit-hole on making it precise in v1 — the proxy plus a human glance is the design.
+- *Waiver detection drift.* If the per-project waiver location moves or its format changes, the commit-autonomy gate could mis-read. Mitigation: fail safe to file-only. Pin the waiver format in the Phase 3 decision before building.
+- *Unattended-commit blast radius.* The headline risk. Mitigated four ways: file-only default, the hard cap, the data-loss checklist item, and the metrics loop (which makes a bad run visible after the fact even if the first three let something through). With task size uncapped, the cost dimension of this risk grows — the vNext token budget is the planned fifth layer.
+- *Scope creep into /start-work territory.* Size is intentionally no longer the brake. The brake is the defer checklist's design-deliberation item plus the "when unsure, defer" rule — keep item 4 strict so genuine deliberation-class tasks still route out even when they're tagged =:solo:= by mistake.
+- *Pre-flight classifier error.* The speedrun's gather step has to split quick-answerable-question from real-deliberation. Misclassifying a deliberation task as a quick question puts a half-baked decision into an autonomous run. Mitigation: when the question isn't answerable in one or two lines, treat it as deliberation and drop it from the run, not as a pre-flight question.
+
+* Testing / Verification / Rollout
+
+Verification is by invocation against a project's real =todo.org=: run the loop caller in file-only mode and confirm it surfaces diffs without committing; run the speedrun against a small explicit list in a waiver-carrying project and confirm the pre-flight Q&A fires, "skip this" drops a task, an answer is recorded and used, then one commit per logical change + the end page; plant a =VERIFY=-status task, a data-loss task, an already-satisfied task, and a large-but-=:solo:= task and confirm the first three are skipped/refused while the large one runs and decomposes; confirm the JSONL grows one record per task; run synthesis and confirm a KB node lands (personal project) or is refused (work project). Rollout is the Phase 0-6 sequence, each leaving the tree working; the file-only default makes early phases safe to ship before the commit and paging phases land.
+
+* References / Appendix
+
+- [[file:../design/2026-06-16-inbox-zero-phase-e-proposal.org][Phase E proposal (inbox-zero stopgap)]] and [[file:../design/2026-06-16-inbox-zero-phase-e-sender-note.org][its sender note with the 5 open questions]].
+- [[file:../design/2026-06-15-fix-speedrun-workflow-proposal.org][speedrun proposal]] (file retains its original on-disk name pending a rename pass).
+- [[file:../../.ai/workflows/inbox.org][inbox.org (canonical A-D; was inbox-zero.org)]] — the routing workflow this feature decouples from.
+- =~/code/rulesets/claude-rules/knowledge-base.md= — the org-roam write contract the synthesis step follows.
+
+* Review and iteration history
+** 2026-06-16 Tue — author
+- What: initial draft reconciling the Phase E and fix-speedrun proposals into one work-the-backlog.org feature, plus the effectiveness-measurement instrumentation.
+- Why: two overlapping proposals arrived within a day; building them separately would duplicate the execution loop and let it drift. Craig also asked explicitly for measurement + org-roam synthesis.
+- Artifacts: this spec; the two source proposals under docs/design/ (the Phase E proposal, diff, and sender note now filed there).
+** 2026-06-28 Sun — revision (Craig)
+- What: removed the task-size gate (size no longer defers; large tasks decompose into per-commit chunks); recast the act-vs-file rule as a crisp four-item defer checklist keyed on test-writability; added crisp =:solo:= / =:quick:= definitions destined for =todo-format.md= and made their assessment mandatory in task-review + task-audit; added the speedrun's pre-flight decision-gathering step (batch the quick questions up front, "skip this" drops a task, then run hands-off); renamed "fix speedrun" → "no-approvals speedrun" in prose. Status stays draft pending ratification of the revised decisions.
+- Why: the original criteria were adjectives, not checkable; the size gate forced Craig to stay at his desk for anything non-trivial, defeating the away-from-desk use case; and decision-needing tasks were over-deferred when many need only a quick upfront answer.
+** 2026-06-29 Mon — ratified
+- What: Craig ratified all eight revised decisions; Status → ready. Implementation-ready across Phase 0 (tag definitions + task-review/audit enforcement) through Phase 6 (synthesis).
+- Why: the crisp defer checklist and the pre-flight-Q&A design resolved the "criteria too soft" and "size shouldn't gate" concerns that held the spec in draft.
diff --git a/docs/specs/2026-06-16-encourage-kb-contribution-spec.org b/docs/specs/2026-06-16-encourage-kb-contribution-spec.org
new file mode 100644
index 0000000..3d08ed5
--- /dev/null
+++ b/docs/specs/2026-06-16-encourage-kb-contribution-spec.org
@@ -0,0 +1,206 @@
+#+TITLE: Encourage Org-Roam KB Contribution Across Workflows — Spec
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-06-16
+#+TODO: TODO | DONE
+#+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED
+
+* READY Encourage Org-Roam KB Contribution Across Workflows — Spec
+:PROPERTIES:
+:ID: f67f5f45-5aa1-4a5a-8704-d636e4e16f75
+:END:
+- 2026-07-02 Thu @ 00:17:01 -0400 — retrofitted by spec-sort; status set to READY (evidence-based, human-confirmed)
+
+* Metadata
+| Status | ready |
+|----------+------------------------------------------------|
+| Owner | Craig Jennings |
+|----------+------------------------------------------------|
+| Reviewer | Craig Jennings |
+|----------+------------------------------------------------|
+| Date | 2026-06-16 |
+|----------+------------------------------------------------|
+| Related | [[file:../../todo.org][rulesets todo.org]] |
+|----------+------------------------------------------------|
+
+* Summary
+
+The org-roam KB already exists (=knowledge-base.md=: =~/org/roam/agents/=, =:agent:= filetag, capture-then-promote, personal-vs-work write boundary), but nothing in the daily workflow loop encourages agents to use it. The wrap-up's =KB: promoted N / consulted yes-no= receipt is the only touchpoint, and it fires at the very end when the session's learnings have already faded. This feature wires four light prompts into the synced template workflows — startup, triage-intake, inbox-zero, wrap-it-up — plus one curated best-practices node in the KB, so contributing durable knowledge becomes a habit the workflows nudge rather than a rule agents forget.
+
+* Problem / Context
+
+The KB rule is sound but passive. An agent reads =knowledge-base.md= once at rule-load and then never gets reminded to consult or contribute, so the KB stays nearly empty and never reaches the critical mass where consulting it pays off. The compounding asset Craig wants — a cross-project store that gets more valuable as it grows — needs a contribution habit, and habits in this system come from workflow prompts, not from a rule sitting in the background.
+
+Three gaps:
+
+1. *No quality guidance.* =knowledge-base.md= says what goes in (durable facts) and where (=agents/= nodes), but not /how/ to write a good node — atomic, descriptively titled, linked. An agent following the rule literally can still produce a junk drawer of vague, unlinked notes that no future agent can find or trust.
+2. *No mid-session capture prompts.* Triage-intake and inbox-zero both surface durable signal (a recurring pattern across messages, a reference pointer worth keeping) and then drop it. Nothing tells the agent "that was worth a node."
+3. *The only contribution prompt is too late.* Wrap-up's KB promotion check runs in Step 1, after the session, when the agent is reconstructing learnings from the log rather than capturing them while fresh.
+
+* Goals and Non-Goals
+
+** Goals
+- Curate a best-practices node in the KB that teaches agents how to write good nodes, drawing on established note-taking guidance.
+- Link that node from startup with a light, one-line encouragement to contribute through the session.
+- Add a short end-of-flow KB reminder to triage-intake and inbox-zero.
+- Add an early wrap-up prompt that asks what the agent learned worth remembering, feeding the existing =KB: promoted N= receipt.
+- Keep every prompt light and non-blocking — encouragement, never a gate.
+
+** Non-Goals
+- *Not* changing =knowledge-base.md='s write boundary, schema, or the work/personal classification. The feature builds on that rule unchanged.
+- *Not* adding a blocking gate anywhere. No workflow stalls or fails because a node wasn't written.
+- *Not* automating node creation. The agent decides what's durable; the prompts only ask the question.
+- *Not* a second receipt or metric. Wrap-up's =KB: promoted N / consulted yes-no= line stays the single instrumentation point.
+- *Not* touching the wrap-up's existing Step 1 KB-promotion sub-section's schema — the new early prompt /feeds/ it, it doesn't replace it.
+
+** Scope tiers
+- v1: the four workflow edits + the one curated best-practices node. All synced templates, so the edits propagate to every project on next startup.
+- Out of scope: a contribution-rate dashboard, per-project KB stats, auto-suggesting nodes from session content.
+- vNext: a "consult the KB before this task" prompt in start-work / spec-create (deferred — log to todo.org).
+
+* Design
+
+The feature is four small prompt insertions plus one authored artifact. The design work is mostly about /placement/ and /wording/: these are synced templates, so a prompt that reads as nagging gets paid forward to every project on every run. The governing constraint is "light enough that an agent welcomes it, specific enough that it actually fires."
+
+** The best-practices node (the artifact)
+
+The node lives at =~/org/roam/agents/<timestamp>-agent-kb-best-practices.org=, authored by hand (not agent-generated), with the standard =:agent:reference:= filetags so it's a first-class KB node agents can find by the same =rg= the rule already documents. It is the one node startup links to, and the substance the workflow prompts point at instead of re-explaining note-taking inline.
+
+Its content is curated from the established note-taking literature — Sönke Ahrens' systematization of Luhmann's Zettelkasten, Andy Matuschak's evergreen-notes practice, and the org-roam community's own guidance — distilled to the handful of principles that matter for an /agent/ writing /durable facts/, not a human building a thinking environment. Proposed outline:
+
+1. *Why the KB exists* — one paragraph: a cross-project, cross-machine asset that compounds. Consulting it saves re-deriving; contributing to it pays the next agent forward.
+2. *One idea per node (atomicity).* Each node holds a single durable fact. Atomicity is what makes a note linkable and findable — a node about three things links cleanly to none of them. (Ahrens; zettelkasten.de atomicity guide.)
+3. *Descriptive, declarative titles.* The title states the claim, not the topic: "SSH auth routes through gpg-agent with a separate cache TTL" beats "SSH notes." A title you can read as a standalone statement is one a future agent can scan and trust without opening the node. (Matuschak evergreen notes; org-roam community practice.)
+4. *Link liberally.* Use =[[id:...]]= to connect a new node to related ones; the value is in the network, not the isolated note. Link to Craig's hand-authored nodes, never edit them. (Matuschak "densely linked"; the linking principle.)
+5. *Capture, then promote.* Harness memory is the fast capture layer; the KB is for facts that cleared the durability bar. Don't promote everything — promote what transfers. (Mirrors =knowledge-base.md='s capture-then-promote.)
+6. *What goes in / what stays out.* Restate the rule's inclusion bar tersely (durable, cross-project, the why behind a decision, environment gotchas, reference pointers) and the exclusion bar (session state, task state, high-churn facts, secrets, anything the repo already records).
+7. *The write boundary.* One line pointing at =knowledge-base.md=: personal projects only, work and unknown projects never write — with the refusal contract. The node /defers/ to the rule here rather than restating the denylist, so there's one source of truth for the boundary.
+8. *Sources.* The citations below, as a reference footer.
+
+Two-altitude note: for a /reading/ agent the node is "how do I tell a good node from a bad one before I trust it?"; for a /writing/ agent it's "what shape should this fact take before I commit it?" The outline serves both — principles 2-4 are the writing checklist, 6-7 are the reading/eligibility filter.
+
+** The four workflow prompts (placement + wording)
+
+Each is the minimum that fires reliably without nagging. Exact insertion points and proposed copy are in Implementation phases below; the design rationale per prompt:
+
+- *Startup (link + light encouragement).* Startup already reads =notes.org= and surfaces nudges in Phase C. The KB encouragement rides there as one line, not a new phase — it points at the best-practices node and frames the session's contribution as welcome, not required. It fires once per session at the top, setting the frame; the other three prompts collect on it.
+- *Triage-intake (end-of-flow reminder).* Placed at the very end of Phase D / Exit Criteria, after actions ship — the moment the agent has just seen a sweep's worth of signal and might recognize a durable pattern. One line, conditional in spirit ("if anything here was durable…"), never a blocking step before close-out.
+- *Inbox-zero (end-of-flow reminder).* Same shape, placed in Phase D (Surface) after the moved/folded/dropped report — the agent has just triaged a batch and may have spotted a reference pointer worth keeping.
+- *Wrap-up (early prompt feeding the existing receipt).* Placed at the /start/ of Step 1, before the Summary is finalized, while the session is fresh — "what did you learn worth remembering, for yourself or a future agent?" The answer flows into the existing Step 1 KB-promotion sub-section and its =KB: promoted N / consulted yes-no= receipt. The early prompt and the existing check are one pipeline: the prompt captures while fresh, the existing sub-section does the promotion and writes the receipt. No second receipt.
+
+** How the early wrap-up prompt feeds the existing receipt
+
+The existing wrap-up Step 1 already has a "KB promotion check" sub-section that asks the promotion question and writes =KB: promoted N / consulted yes-no=. The new early prompt is not a second check — it's a /relocation of the asking/ to the top of Step 1 so the question lands while the session is fresh rather than after the Summary is reconstructed. The existing sub-section keeps ownership of the actual promotion (writing the =agents/= nodes per schema) and the receipt line. Concretely: the early prompt asks and collects candidate facts into the session's working notes; the existing sub-section consumes those candidates, writes the nodes, and emits the one receipt. This avoids duplication by making the early prompt a /capture/ step and the existing check the /commit + receipt/ step of the same pipeline.
+
+* Alternatives Considered
+
+** A blocking gate ("you must write ≥1 node to wrap up")
+- Good, because it would guarantee contributions and grow the KB fast.
+- Bad, because it manufactures junk — agents would write a throwaway node to clear the gate, polluting exactly the asset the feature is meant to grow. It also fights the "light, non-nagging" constraint head-on.
+- Neutral, because the receipt already gives visibility into contribution rate without forcing it.
+
+** Inlining the best-practices guidance into each workflow prompt
+- Good, because the guidance is right there at the point of use; no indirection.
+- Bad, because it's four copies of the same note-taking advice in four synced templates — duplication that drifts, and four times the prompt length, which reads as nagging. One linked node keeps each prompt to one line.
+- Neutral, because a one-node-plus-links shape is exactly what the best-practices node /teaches/, so the design eats its own dogfood.
+
+** Putting the encouragement only in =knowledge-base.md= (no workflow edits)
+- Good, because it's the least change — one rule edit, no template churn.
+- Bad, because that's the status quo that produced the problem: a rule read once at load and then forgotten. Habits in this system come from workflow prompts, not background rules.
+- Neutral, because the rule still carries the authoritative boundary; the workflow prompts are the habit layer on top.
+
+* Decisions [6/6]
+
+** DONE Where exactly does the startup link land — Phase A read, Phase C nudge, or notes.org?
+- Owner / by-when: Craig / before implementation
+- Context: Startup has three candidate homes for the KB encouragement: a Phase A parallel read of the best-practices node (costs context every session), a Phase C surfaced nudge (one line, conditional, consistent with the existing roam-inbox and task-review nudges), or a static line in each project's =notes.org= Active Reminders (per-project, not synced, drifts). The Phase C nudge matches the established nudge pattern and costs nothing when there's nothing to say.
+- Decision: We will add the encouragement as a one-line Phase C nudge in startup.org, pointing at the best-practices node by its KB path, surfaced once near the other Phase C nudges.
+- Consequences: easier — consistent with existing nudge mechanics, synced to every project, no per-session read cost; harder — one more line competing for attention in the Phase C surface, so the wording has to earn its place and stay terse.
+
+** DONE Is the startup nudge unconditional, or gated on the KB clone being present?
+- Owner / by-when: Craig / before implementation
+- Context: =~/org/roam/= isn't on every machine. The existing roam-inbox nudge already guards on the clone's presence ([ -f ~/org/roam/inbox.org ]). An unconditional KB nudge would fire on machines where the agent can't act on it.
+- Decision: We will gate the startup nudge on the roam clone being present, reusing the existing presence check, so the encouragement only appears where the agent can act on it.
+- Consequences: easier — no dead nudge on KB-less machines, mirrors the roam-inbox guard; harder — one more conditional in Phase C, and a machine without the clone gets no encouragement at all (acceptable — it can't contribute there anyway).
+
+** DONE Does the early wrap-up prompt stop and ask Craig, or self-answer silently?
+- Owner / by-when: Craig / before implementation
+- Context: Wrap-up is meant to be quick — Craig already authorized the wrap, and the existing KB-promotion check self-answers (the agent decides what's durable; work projects skip the write). An early prompt that /stops and asks Craig/ "what did you learn?" would add an interactive turn to a flow designed not to have them. But a purely silent self-answer risks the agent skipping the reflection.
+- Decision: We will have the agent self-answer the early prompt — reflect on session learnings and stage candidate facts — without stopping to ask Craig, matching the wrap-up's no-extra-turns design; the candidates flow into the existing promotion check which writes the nodes and receipt.
+- Consequences: easier — preserves wrap-up cadence, no new interactive gate, one pipeline from reflect to receipt; harder — relies on the agent actually reflecting rather than rubber-stamping "nothing learned," which the receipt makes visible over time but doesn't enforce.
+
+** DONE Do triage-intake and inbox-zero reminders fire every run, or only when the run surfaced something durable?
+- Owner / by-when: Craig / before implementation
+- Context: Both workflows run frequently (triage-intake between meetings, inbox-zero twice a session). A reminder on /every/ run is the textbook nag-fatigue failure — a line the agent learns to skip. A reminder gated on "this run surfaced a pattern / reference pointer worth keeping" fires rarely and stays meaningful, but requires the agent to make that judgment, which is softer than a mechanical condition.
+- Decision: We will make both reminders conditional in spirit — a single line phrased as "if anything here was durable, write it to the KB" that the agent acts on only when the run actually surfaced something, rather than an unconditional step; an all-quiet triage sweep or an empty inbox-zero run emits no KB line.
+- Consequences: easier — the reminder stays rare and credible, never pads a no-change sweep, fits triage-intake's deltas-only discipline; harder — "durable-looking" is an agent judgment with no mechanical check, so the reminder's effectiveness rides on the best-practices node teaching that judgment well.
+
+** DONE Best-practices node: agent-authored once, or hand-authored by Craig?
+- Owner / by-when: Craig / before implementation
+- Context: =knowledge-base.md= says agents never edit Craig's hand-authored nodes. The best-practices node is /about/ how agents write nodes — if an agent authors it, future agents may treat it as fair game to edit; if Craig hand-authors it, it's protected and stable but he writes it. Given it's a foundational reference the whole feature points at, stability matters.
+- Decision: We will have Craig hand-author the best-practices node from the outline in this spec, so it's a protected, stable reference; the spec supplies the full drafted content for him to review and commit.
+- Consequences: easier — the node is stable and protected from agent edits, one authoritative reference; harder — Craig writes (or reviews-and-commits) it rather than delegating, and updates to it are his call, not an agent's.
+
+** DONE Read side: how does startup surface lessons to consult, not just encourage contribution?
+- Owner / by-when: Craig / ratified 2026-06-20
+- Context: The original spec only strengthened the /write/ side — startup encourages contributing (D1) but never surfaces existing KB lessons to /read/. The wrap-up receipt data shows "consulted no" across recent sessions: agents don't reach for the KB because nothing brings it to their attention at the moment work starts. =knowledge-base.md='s "search the KB first" is reactive and read-once-at-rule-load. A proactive surfacing at startup is the missing counterpart to D1. The cost constraint is the same one D1 dodged: a full Phase A read of matching nodes would spend context every session.
+- Decision: We will add a second startup Phase C nudge (alongside D1's contribute-link, gated on the same roam-clone presence check) that surfaces KB lessons relevant to the current project — a count plus the nodes' declarative /titles only/ (no full-node read), capped at ~5. Relevance is matched cheaply on the project basename and obvious topic words against node titles/filetags/paths, with a most-recent fallback when nothing matches. The agent opens a node on demand. Titles are declarative by the best-practices node's own rule, so a title alone tells the agent whether to open it.
+- Consequences: easier — closes the "consulted no" half with near-zero context cost (titles only), reuses the Phase C nudge pattern and the roam guard, and the consult and contribute nudges sit together as one KB surface; harder — relevance matching is a heuristic that can miss or mis-surface, and it adds a second KB line to Phase C, so both must stay terse to avoid nudge fatigue. If the receipt shows consults rising but the surfaced titles are noise, tighten the match.
+
+* Implementation phases
+
+** Phase 1 — Author the best-practices node
+Write =~/org/roam/agents/<timestamp>-agent-kb-best-practices.org= from the outline in Design, with a generated =:ID:=, =#+title:=, =:filetags: :agent:reference:=, the eight content sections, =[[id:...]]= links to any existing related =:agent:= nodes, and the sources footer. Commit + push the roam repo per =knowledge-base.md='s session discipline. Leaves the KB with one new reference node and nothing else touched.
+
+** Phase 2 — Wire the startup encouragement (contribute + consult)
+Add two one-line Phase C nudges to =claude-templates/.ai/workflows/startup.org= (canonical side), both gated on the roam-clone presence check: (1) D1's contribute-link pointing at the best-practices node by path, and (2) D6's consult-surface listing project-relevant KB node titles (count + titles only, capped ~5, project-basename match with recent fallback). A Phase A read counts =:agent:= nodes cheaply so Phase C only does the title surfacing when there's something to show. Run =scripts/sync-check.sh --fix=, commit both canonical + mirror. Propagates to every project on next startup.
+
+** Phase 3 — Wire the three remaining prompts
+Add the end-of-flow KB reminder to =triage-intake.org= (end of Phase D / Exit Criteria) and =inbox-zero.org= (Phase D Surface), and the early KB prompt to =wrap-it-up.org= (top of Step 1, feeding the existing promotion check). All on the canonical side, then sync-check + commit. Each edit is one short block; the tree stays working after each.
+
+** Phase 4 — Verify propagation + receipt linkage
+Confirm the four edits survive a startup sync into a test project, the wrap-up early prompt's output reaches the existing =KB: promoted N / consulted yes-no= receipt (no duplicate receipt), and the best-practices node is reachable by the =rg= the rule documents.
+
+* Acceptance criteria
+- [ ] Best-practices node exists at =~/org/roam/agents/= with =:agent:reference:= tags, is found by =rg '#\+filetags:.*:agent:' ~/org/roam/=, and cites its sources.
+- [ ] Startup surfaces a single KB-contribution line in Phase C, gated on the roam clone, pointing at the node — and stays silent when the clone is absent.
+- [ ] Startup also surfaces a KB-consult line in Phase C (D6): project-relevant node titles (count + titles only, capped ~5), gated on the clone, silent when nothing matches and the clone is absent.
+- [ ] Triage-intake and inbox-zero each emit one KB reminder line only when the run surfaced something durable; an all-quiet run emits none.
+- [ ] Wrap-up asks the "what did you learn?" reflection early in Step 1, and its candidates feed the existing promotion check — producing exactly one =KB: promoted N / consulted yes-no= receipt, not two.
+- [ ] No workflow blocks, stalls, or fails because a node wasn't written.
+- [ ] All four workflow edits are on the canonical =claude-templates/.ai/= side, mirror synced, sync-check clean.
+
+* Readiness dimensions
+- Data model & ownership: KB nodes are agent-written under =agents/=; the best-practices node is Craig-authored and protected. No new persisted state beyond the one node and the four template edits. Wrap-up receipt ownership unchanged.
+- Errors, empty states & failure: roam clone absent → all KB prompts silently no-op (reuse existing presence guards). Work/unknown project → write boundary in =knowledge-base.md= still refuses with its contract; prompts fire but the agent declines to write per the rule. No silent data loss — nothing is deleted.
+- Security & privacy: no secrets in nodes (rule's exclusion bar). Work-confidential facts never written (the boundary). The best-practices node is reference-only, no sensitive content.
+- Observability: the existing =KB: promoted N / consulted yes-no= receipt is the single metric; grepping session archives for =KB:= answers "are agents using this?" No new instrumentation added.
+- Performance & scale: four one-line prompts; negligible. The startup nudge is a Phase C surface line, not a Phase A read, so no per-session context cost from loading the node.
+- Reuse & lost opportunities: reuses the existing Phase C nudge pattern, the roam-clone presence guard, the wrap-up promotion check + receipt, and =knowledge-base.md='s boundary. Nothing reinvented.
+- Architecture fit & weak points: the four workflows are synced templates; canonical-vs-mirror edit discipline applies (CLAUDE.md). Weak point — nag fatigue if the reminders fire unconditionally; mitigated by the conditional-in-spirit decision. Weak point — the reminders rely on agent judgment ("durable-looking"); mitigated by the best-practices node teaching that judgment.
+- Config surface: none. No new knobs; the prompts are unconditional copy gated only on the existing roam-clone check.
+- Documentation plan: the best-practices node /is/ the user-facing doc. =knowledge-base.md= stays the authoritative rule; this feature adds no new rule file. No migration doc needed.
+- Dev tooling: =scripts/sync-check.sh --fix= keeps canonical + mirror aligned (enforced by =githooks/pre-commit=). =make test= covers the repo's existing gates; no new test target needed for prose-only workflow edits.
+- Rollout, compatibility & rollback: edits propagate via the startup rsync to every project on next session — no migration. Rollback is reverting the four template edits + deleting the node; nothing persisted depends on them. Fully reversible.
+- External APIs & deps: none — no API calls, no new dependencies. The only external surface is the =~/org/roam/= git repo, already in use by the rule.
+
+* Risks, Rabbit Holes, and Drawbacks
+- *Nag fatigue* — the central risk. Four prompts across four frequently-run workflows can train agents to skip them. Dodge: one line each, conditional in spirit, the startup line gated, the triage/inbox reminders firing only on real signal. If the receipt shows agents tuning them out, cut the lowest-value prompt rather than adding more.
+- *Junk-node accumulation* — encouraging contribution without a quality bar grows a junk drawer. Dodge: the best-practices node /is/ the quality bar, and the exclusion list keeps high-churn / session-state facts out. Craig prunes at will (the rule already grants this).
+- *Receipt double-counting* — if the early wrap-up prompt writes its own receipt, the metric breaks. Dodge: the early prompt is explicitly a capture step feeding the existing check; only the existing sub-section emits the receipt. Acceptance criterion guards this.
+
+* References / Appendix
+Sources for the best-practices node's curated content:
+- Sönke Ahrens, /How to Take Smart Notes/ — atomicity, own-words, linking: [[https://www.soenkeahrens.de/en/takesmartnotes][soenkeahrens.de]]; principle of atomicity: [[https://zettelkasten.de/atomicity/guide/][zettelkasten.de atomicity guide]].
+- Andy Matuschak, /Evergreen notes/ — concept-oriented, densely linked, write for yourself: [[https://notes.andymatuschak.org/Evergreen_notes_should_be_concept-oriented][notes.andymatuschak.org]].
+- Org-roam community practice — declarative titles, atomic nodes, capture-then-refine: [[https://www.orgroam.com/manual.html][Org-roam manual]]; [[https://lucidmanager.org/productivity/taking-notes-with-emacs-org-mode-and-org-roam/][lucidmanager.org org-roam guide]].
+- Existing rule this builds on: =~/code/rulesets/claude-rules/knowledge-base.md=.
+
+* Review and iteration history
+** 2026-06-16 Tue — author
+- What: initial draft.
+- Why: Craig wants the org-roam KB to compound into a cross-project asset; needs the workflow wiring + curated best-practices node speced before building.
+- Artifacts: this spec; four target workflows (startup, triage-intake, inbox-zero, wrap-it-up); =knowledge-base.md=.
+** 2026-06-20 Sat — ratified + read-side added
+- What: ratified all five original decisions; added decision D6 (read-side startup consult-nudge) and threaded it through Design, Phase 2, and acceptance. Status draft → approved.
+- Why: receipt data showed the write-only design left "consulted no" across recent sessions. Craig asked for the reverse of contribution — surfacing relevant lessons to read at startup. D6 is that counterpart.
+- Artifacts: this spec; startup.org (now two Phase C nudges); the lint level-2-dated-header checker tracked separately.
diff --git a/docs/specs/2026-07-01-docs-lifecycle-spec.org b/docs/specs/2026-07-01-docs-lifecycle-spec.org
new file mode 100644
index 0000000..91c1603
--- /dev/null
+++ b/docs/specs/2026-07-01-docs-lifecycle-spec.org
@@ -0,0 +1,361 @@
+#+TITLE: Docs Lifecycle — Spec
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-01
+#+TODO: TODO | DONE
+#+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED
+
+* IMPLEMENTED Docs lifecycle
+:PROPERTIES:
+:ID: 80b0787b-4a60-4c82-8a16-b383d3e3c8f2
+:END:
+- 2026-07-04 Sat @ 11:46:31 -0500 — DOING → IMPLEMENTED: all four build phases shipped (docs-lifecycle rule + spec-workflow updates, spec-sort helper + 30-test bats suite, rulesets pilot sorting the pile and standing up the status board, startup nudge gated on :LAST_SPEC_SORT:) plus the follow-up file:→id: link conversion. Remaining manual validation (startup nudge fires/clears, moved-spec links click through in Emacs) tracked as its own task; a failed check promotes to a bug.
+- 2026-07-01 Wed @ 23:34:15 -0400 — READY → DOING: spec-response decomposition ran — build parent in todo.org carries the :SPEC_ID: binding, one task per phase plus the flip-to-IMPLEMENTED task and the manual-testing child. First live exercise of the transition-ownership table.
+- 2026-07-01 Wed @ 23:22:50 -0400 — DRAFT → READY: Codex re-review found all fourteen review findings closed and no remaining blocking implementation-readiness gaps.
+- 2026-07-01 Wed @ 22:54:41 -0400 — verify pass on the second responder round: all five fixes held, findings 1-9 unregressed, verdict ready; three minor nits folded in (scoped id-link criterion, untracked-copy cleanup in the recovery recipe, two stale prose spots). Stays DRAFT pending the reviewers' flip.
+- 2026-07-01 Wed @ 22:46:52 -0400 — second responder pass: all five re-review findings fixed (fourteen of fourteen closed); stays DRAFT — the READY flip belongs to the reviewers this round.
+- 2026-07-01 Wed @ 22:41:33 -0400 — READY → DRAFT: Codex re-review found five new blocking implementation-readiness gaps after the response pass.
+- 2026-07-01 Wed @ 22:41:21 -0400 — DRAFT → READY: dual independent review (Codex + fresh-context Claude agent, both initially Not ready), all nine findings fixed, verify pass by the original reviewer returned ready; flip authorized by Craig.
+- 2026-07-01 Wed @ 22:13:00 -0400 — drafted from the five decisions settled 2026-06-28 (todo.org "Spec storage location + lifecycle-status convention").
+
+* Metadata
+| Status | implemented |
+|----------+------------------------------------------------------------------|
+| Owner | Craig Jennings |
+|----------+------------------------------------------------------------------|
+| Reviewer | Craig Jennings |
+|----------+------------------------------------------------------------------|
+| Date | 2026-07-01 |
+|----------+------------------------------------------------------------------|
+| Related | [[file:../design/2026-06-15-spec-storage-lifecycle-proposal.org][source proposal]]; todo.org "Spec storage location + |
+| | lifecycle-status convention" |
+|----------+------------------------------------------------------------------|
+
+* Summary
+
+Formal specs and working notes currently share one directory per project, and a spec's lifecycle state (drafted, in progress, shipped, dead) is invisible without opening the file. This spec adopts two coupled conventions — a location split (=docs/specs/= for formal specs, =docs/design/= for notes) and an authoritative in-file status carried by an org TODO keyword on a top-level status heading — plus =org-id= links for rename-safety, a general =docs-lifecycle= rule capturing the shape, and a one-time confirmed retrofit that sorts every project's existing pile.
+
+* Problem / Context
+
+.emacs.d triaged ~28 design docs and had to run a four-agent sweep reading every spec against the code to reconstruct which had shipped (6 implemented, 8 in progress, 12 not started, 1 superseded). Nothing in the filename, location, or file records the state, so the answer to "what's open?" degrades into "open every file and infer." rulesets has the same shape: 41 files in =docs/design/= of which only 3 carry a formal spec spine, plus two =-spec.org= files misfiled at the =docs/= root. The cost compounds with every doc added, and every project inherits the problem through the shared spec-create workflow.
+
+Two forces beyond triage cost:
+
+- *Links are load-bearing.* =todo.org= tasks, session archives, and sibling docs link specs by =file:= path. Any convention that renames or moves files on every status change (the filename-suffix approach) breaks those links repeatedly across a cross-linked, template-synced doc set.
+- *The convention is worthless if legacy docs stay misfiled* (Craig, 2026-06-28). Template sync distributes rules and workflows but cannot perform a one-time per-project migration, so the design must include a reach mechanism that gets each project's existing pile sorted once.
+
+* Goals and Non-Goals
+
+** Goals
+- A directory listing answers "which docs are specs, and what state is each in" without opening files.
+- Status transitions cost one small in-file edit (keyword + history line + Metadata mirror) — no rename, no link surgery.
+- Cross-doc spec links survive moves and renames.
+- The shape is captured once as a general rule (=docs-lifecycle=) so future artifact collections (brainstorm piles, recording queues) can reuse it.
+- Every existing project's =docs/design/= pile gets sorted exactly once, with human confirmation on each classification.
+
+** Non-Goals
+- No automation of status flips — the keyword is edited by whoever changes the state (spec-create, spec-review, spec-response, or a human), not by a watcher.
+- No retroactive rewriting of session archives or git history that reference old paths; only live inbound links (=todo.org=, =notes.org=, docs) are updated by the retrofit.
+- No new tracking database or index file — the files are the index.
+
+** Scope tiers
+- v1: the location split, the status-heading convention, the org-id link standard, the =docs-lifecycle= rule, spec-create/spec-review/spec-response updates, the retrofit helper + startup nudge, and the rulesets pilot.
+- Out of scope: applying the lifecycle shape to non-doc collections (the rule documents the pattern; adopting it elsewhere is per-collection work).
+- vNext: an org-agenda custom view over =docs/specs/*.org= keyed on the status keywords (nice-to-have once the keywords exist; log to todo.org).
+
+* Design
+
+** The location split
+
+- =docs/specs/= — formal specs only. A *spec* is a doc proposing a buildable change that carries a =Decisions= section and =Implementation phases= (the spec-create spine). Filenames keep the existing =YYYY-MM-DD-<topic>-spec.org= shape — the =-spec.org= suffix stays because spec-review's Phase 0 precondition keys on it; only the *status* suffixes from the original proposal are dropped.
+- =docs/design/= — everything else: brainstorms, inventories, proposals, research notes, frozen source material. Review findings live inside the spec they review (current spec-review behavior), so standalone review files are legacy notes and stay in =docs/design/=.
+
+** The status heading (the authoritative record)
+
+Each spec's first element after the file header is a single top-level *status heading* carrying the org TODO keyword:
+
+#+begin_example
+,#+TODO: TODO | DONE
+,#+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED
+
+,* DOING <spec short name>
+:PROPERTIES:
+:ID: <uuid>
+:END:
+- <dated one-line history entries, newest first>
+#+end_example
+
+- *The keyword is authoritative.* The Metadata table's =Status= field mirrors it in lowercase for readers already in the table, and a status transition updates keyword + history line + mirror in the same edit; on disagreement the heading wins.
+- *Two keyword sequences, no collisions.* The lifecycle sequence *joins* — never replaces — the =TODO | DONE= sequence that the =* Decisions= and =* Review findings= task machinery depends on. The two sequences share no keyword (the old header's =SUPERSEDED CANCELLED= done-states migrate to the lifecycle sequence; a legacy =CANCELLED= decision heading still parses as a done-state there, so =[/]= cookies stay mechanically correct). The retrofit rewrites each legacy header to carry both lines.
+- *Vocabulary:* =DRAFT= (being written) → =READY= (review passed, buildable) → =DOING= (implementation in progress) → =IMPLEMENTED= / =SUPERSEDED= / =CANCELLED= (terminal).
+- *Transition ownership — every flip has a named owner:*
+ - =DRAFT= — spec-create stamps it at authoring time.
+ - =DRAFT= → =READY= — spec-review, on a passing gate (keyword + history line + mirror in the review pass).
+ - =READY= → =DOING= — spec-response, when it decomposes the phases into build tasks. *The decomposition writes the spec-to-task binding:* the =todo.org= parent task it creates (or updates) carries a =:SPEC_ID:= property holding the spec's status-heading UUID. That property is the durable join between the spec and its build work.
+ - =DOING= → =IMPLEMENTED= — the session that completes the final implementation phase. To make that a tracked obligation rather than a memory, spec-response's phase-to-task breakdown *always emits a final task*: "flip the spec to IMPLEMENTED + history line," as a child of the bound parent. Safety net: task-audit's reconcile pass runs one query — for each =docs/specs/*.org= whose keyword is =DOING=, find the =todo.org= task with the matching =:SPEC_ID:=; flag the spec when that parent is =DONE=/=CANCELLED=, archived, or missing. Checking the *parent's* keyword (not "are all child tasks closed") sidesteps both the flip-task chicken-and-egg (the parent only closes after the flip task ran) and =--convert-subtasks= rewriting completed children into dated entries (dated children never affect the parent's keyword). This is the mechanism whose absence produced the .emacs.d six-shipped-specs-with-no-record failure; "a human remembers" is explicitly not the design.
+ - =SUPERSEDED= / =CANCELLED= — whoever makes the call, with the reason in the history line.
+- *Glanceability without opening files:* one grep gives the full board —
+
+ #+begin_src sh
+ rg -H '^\* (DRAFT|READY|DOING|IMPLEMENTED|SUPERSEDED|CANCELLED) ' docs/specs/
+ #+end_src
+
+ and because the keyword sits on a real org heading, an org-agenda view over =docs/specs/= works for free (the vNext item).
+- *The heading body is the dated status history* — one line per transition (=YYYY-MM-DD Day @ HH:MM:SS -ZZZZ — <what changed, by whom>=), the record a filename could never carry.
+- Why a dedicated status heading rather than restructuring each spec under one top-level heading: demoting every section in every existing spec is a large, link-hostile rewrite; a prepended heading is additive, retrofittable by script, and leaves the familiar flat section layout untouched.
+
+** Rename-safe links
+
+The status heading carries an =:ID:= UUID, assigned at authoring time (and by the retrofit for legacy specs). The target state is that cross-doc references to a spec use =[[id:<uuid>]]= rather than =file:= paths, so any future move can't orphan them. =file:= links remain fine for intra-doc anchors and for notes that never move. The KB's existing id-resolution recipe applies: =rg ':ID:[[:space:]]+<uuid>' docs/=.
+
+*Staged conversion — ids assigned now, links converted only when clickable.* =org-id-locations= only indexes agenda files and files org has visited, so a fresh =:ID:= in =docs/specs/= won't resolve on click in a live Emacs until the id index learns about project docs. =org-id-extra-files= is not a glob mechanism — it's a literal file list, only consulted under =org-id-track-globally= — so "point it at the globs" is not executable as written. The sequencing is therefore:
+
+1. *Pilot and retrofit rewrite =file:= links only* (path recomputation per the relink contract). Every link stays clickable throughout; no conversion window exists.
+2. *:ID: properties are still assigned* during the sort — harmless, and they make the later conversion mechanical.
+3. *Link conversion to =id:= is a separate follow-up pass*, gated on .emacs.d landing an executable id-index mechanism: enumerate each project's =docs/specs/*.org= into =org-id-extra-files= as real file names (a small function globbing at startup, with =org-id-track-globally= t), or a periodic =org-id-update-id-locations= over that enumeration — verified by clicking a known id link. The Phase 4 note to .emacs.d carries this ask; the =rg= recipe is the fallback for non-Emacs consumers either way.
+
+** The =docs-lifecycle= rule (the generalization)
+
+A new =claude-rules/docs-lifecycle.md= captures the reusable shape, with spec-create as the first instance:
+
+1. Separate formal artifacts from working notes by location.
+2. Lifecycle state lives *in* the artifact, on a scannable, greppable carrier (an org keyword heading), with a dated history.
+3. Links use rename-safe identifiers.
+4. A growing collection earns this treatment when "which of these are live?" starts requiring a file-by-file read.
+
+** The retrofit (reach mechanism for existing piles)
+
+A synced helper, =spec-sort=, run once per project. *Canonical placement:* like every synced asset, the helper and all workflow edits land in rulesets' canonical tree first — =claude-templates/.ai/scripts/spec-sort= with its bats tests in =claude-templates/.ai/scripts/tests/= (the glob-discovered suite), workflow changes in =claude-templates/.ai/workflows/= — then =scripts/sync-check.sh --fix= propagates the committed =.ai/= mirror and both sides commit together. A mirror-only edit is reverted by the next sync; nothing in this feature is exempt from that contract. Downstream projects receive everything through the normal startup rsync. The run itself, per project:
+
+1. *Classify* each =docs/**/*.org= outside =docs/specs/= by one predicate: a doc carrying *both* a =Decisions= heading *and* an =Implementation phases= heading is a spec candidate; everything else is a note. (A =Metadata= table alone does not qualify — real counter-case: =docs/design/task-review.org= has a Metadata table and no spine, and is a note.) The heuristic *proposes*; a human confirms every move (classification is a judgment call — Craig, 2026-06-28).
+2. *Move* confirmed specs to =docs/specs/=, *renaming to carry the =-spec.org= suffix* when the file lacks it (spec-review's Phase 0 precondition requires it — a retrofitted spec must be reviewable in its new home). Prepend the status heading, assign an =:ID:=, and rewrite the keyword header to the two-sequence form above. *The proposed keyword is evidence-based, not laundered:* the doc's own Status field is one signal among several, because stale Status fields are exactly what caused the original .emacs.d sweep. For each candidate the helper shows an evidence panel — the current Status value, the decision/finding cookie states, the state and heading of any =todo.org= task that links or binds to the doc, the most recent history/review entries, and (where cheap) whether artifacts the phases name actually exist — and proposes the keyword the evidence supports. When the evidence is inconclusive, the default is the most conservative *non-terminal* state it supports (never a terminal one). =IMPLEMENTED= / =SUPERSEDED= / =CANCELLED= are never applied without an explicit human-stated reason, recorded in the status-history line.
+3. *Relink* under an explicit contract:
+ - *Rewritten roots (project-owned):* =todo.org=, =.ai/notes.org=, =docs/**=, =.ai/project-workflows/=, =.ai/project-scripts/=. The rewrite recomputes each link's relative path from the linking file's directory to the new location. *All rewrites stay =file:= links* — conversion to =[[id:...]]= is the separate follow-up pass gated on the Emacs id-index mechanism (see Rename-safe links), never part of a sort run.
+ - *Reported, never rewritten:* =.ai/sessions/= archives (frozen history), git history, and synced template paths (=.ai/workflows/=, =.ai/scripts/=, =.ai/protocols.org=) — a downstream edit there is reverted by the next template sync, so the report names the canonical rulesets file that needs the edit instead.
+ - *Supported link shapes:* org =[[file:...]]= links, relative or project-root-anchored, with or without a description. Bare-path mentions in prose or scripts are *reported for manual handling*, never rewritten.
+ - *Safety:* dry-run report is the default; =--apply= writes, under a fail-safe contract sized to the fact that one run mutates filenames, links, headers, and =.ai/notes.org= together:
+ - *Clean-worktree preflight.* =--apply= refuses on a dirty git tree (=git status --porcelain= non-empty) unless =--allow-dirty= is passed, which prints exactly what recovery loses. A clean tree is what makes recovery trivially safe.
+ - *Validate, then write.* The full move + relink plan — every source, destination, and link edit — is computed and validated first (every link parses, every target is unambiguous, every destination path is free), written to a plan file for inspection, and only then executed from that recorded plan. Ambiguous cases (two candidates sharing a basename, an unparseable link) block validation: listed, untouched, non-zero exit until each is resolved or explicitly waived.
+ - *Failure mid-apply is not a shrug.* Any write failure or a failed post-apply residue grep stops the run, names what was and wasn't applied (from the plan), and prints the recovery recipe — =git restore= over the plan's touched paths *plus* deletion of the plan's newly-created destination paths (=git restore= reverts tracked edits but doesn't remove untracked copies the move created). Safe by construction because preflight required a clean tree; the project is never silently left half-migrated.
+ - After a successful apply, the residue grep for each old path across the rewritten roots must return zero or =spec-sort= exits non-zero naming the residue.
+4. *Stamp* =:LAST_SPEC_SORT: YYYY-MM-DD= in =.ai/notes.org='s =* Workflow State= section — the same surface as =:LAST_AUDIT:= and =:LAST_INBOX_PROCESS:=, created idempotently (append the section if the file lacks it) exactly as task-audit already does.
+
+*The startup nudge — concrete contract.* Phase A's parallel batch gains one read-only probe:
+
+#+begin_src bash
+{ [ -d docs/design ] || [ -n "$(find docs -maxdepth 1 -name '*-spec.org' -print -quit 2>/dev/null)" ]; } \
+ && ! grep -qs ':LAST_SPEC_SORT:' .ai/notes.org \
+ && echo "spec-sort: unsorted docs present" || true
+#+end_src
+
+(Phase 4 refined the stray-root check from =compgen= to =find=: =compgen= is bash-only and zsh aborts on an unmatched glob, so the original snippet false-negatived on stray root specs under zsh.)
+
+(The probe also fires on stray =docs/*-spec.org= root files, so a project whose only misfiled specs sit at the =docs/= root still gets nudged.)
+
+Phase C surfaces one line when the probe printed ("this project's docs pile has never been spec-sorted — say 'run spec-sort' to sort it") and stays silent otherwise. Projects with nothing to sort — no =docs/design/= and no stray root specs — never see it; a stamped marker permanently clears it.
+
+* Alternatives Considered
+
+** Filename status suffix (=-spec-doing.org=, =-spec-implemented.org=)
+- Good, because the state is visible in a bare =ls= with no tooling.
+- Bad, because every transition renames a file in a cross-linked, template-synced doc set — each rename is link surgery or a broken link, and the churn lands in git history and inbound =todo.org= links.
+- Neutral, because the ls-visibility it buys is matched by the one-line =rg= over status headings.
+- Rejected 2026-06-28 (Craig chose org-keyword over his earlier filename-suffix lean).
+
+** Status field in the Metadata table only (no keyword)
+- Good, because the field already exists and needs no new structure.
+- Bad, because a table cell is neither org-agenda-scannable nor reliably greppable across format drift, and it carries no dated history.
+- Neutral, because the field stays anyway — as the in-table mirror.
+
+** Relink-helper instead of org-id (keep =file:= links, fix them on every move)
+- Good, because readers see plain paths.
+- Bad, because it makes every future move a tooling event, and one missed run silently breaks links — the failure mode is invisible until someone clicks.
+- Neutral, because the retrofit needs relink logic once regardless; org-id just makes it a one-time need.
+
+* Decisions [5/5]
+
+All five were settled with Craig on 2026-06-28 (recorded in todo.org; migrated here per that note).
+
+** DONE Location split — adopt
+- Context: specs and notes share one directory; telling them apart requires opening files.
+- Decision: =docs/specs/= for formal specs (Decisions + phases spine); =docs/design/= for notes. Documented in spec-create and the docs-lifecycle rule.
+- Consequences: easier — a listing answers "what's formal"; harder — one-time migration and link updates (the retrofit).
+
+** DONE Status mechanism — org keyword authoritative, no filename suffix
+- Context: filename suffix vs org keyword; suffix wins =ls= visibility, keyword wins link stability and zero-rename transitions.
+- Decision: the org TODO keyword on the spec's top status heading is authoritative, mirrored by the Metadata =Status= field. No status suffixes in filenames.
+- Consequences: easier — a transition is one keyword edit and links never break; harder — glanceability needs the one-line =rg= (or the vNext agenda view) instead of bare =ls=.
+- (Refined in review, 2026-07-01: "one keyword edit" became "three lines in one file" — keyword + history line + Metadata mirror. The ratified decision stands; see Review findings.)
+
+** DONE Link safety — org-id for cross-doc spec links
+- Context: both the migration move and any future rename break =file:= links.
+- Decision: specs carry =:ID:= UUIDs on the status heading; cross-doc references use =[[id:...]]=.
+- Consequences: easier — moves are free; harder — following a link outside org needs the =rg ':ID:'= lookup.
+- (Refined in review, 2026-07-01: the decision stands; the *sequencing* is staged — IDs are assigned at sort time, but link conversion to =id:= waits for the executable Emacs id-index mechanism, so no window exists where converted links don't click. See Review findings.)
+
+** DONE Generalize as a =docs-lifecycle= rule
+- Context: the shape (in-artifact lifecycle state, formal-vs-notes split, rename-safe links) recurs for any processed-document collection.
+- Decision: capture it in =claude-rules/docs-lifecycle.md= with spec-create as the first instance.
+- Consequences: easier — the next collection reuses a decided pattern; harder — the rule must stay honest as the spec instance evolves.
+
+** DONE Retrofit existing files across ALL projects
+- Context: template sync distributes conventions but cannot perform a per-project one-time migration; legacy piles would stay misfiled forever.
+- Decision: ship a confirmed classify-move-relink helper (=spec-sort=) plus a startup nudge gated on =:LAST_SPEC_SORT:=; the helper proposes, a human confirms. Pilot on rulesets first.
+- Consequences: easier — every project converges without manual archaeology; harder — the helper needs real relink logic and tests, and classification stays a judgment call.
+
+* Review findings [14/14]
+:PROPERTIES:
+:ID: cc77a7f6-e4c3-488a-ac3b-e739420a5c2b
+:END:
+
+Two independent reviews (Codex, 2026-07-01 22:22; a fresh-context Claude agent, 2026-07-01 22:25) converged on =Not ready= with the same worst finding. All nine findings were dispositioned accept and fixed in the responder pass below; each carries its response.
+
+** DONE Org TODO vocabulary drops decision and finding task states :blocking:
+(Codex; the Claude reviewer found the same, adding that keywords must be unique across sequences so a naive two-line fix collides on =SUPERSEDED=/=CANCELLED=.) The spec's example header replaced the file-level keyword vocabulary, so =TODO=/=DONE= stopped being task states and the =[/]= cookies that gate readiness went vacuous — this file itself was the first casualty.
+Response: the scheme is now two collision-free sequences — =TODO | DONE= for decisions/findings, =DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED= for lifecycle (the old header's =SUPERSEDED CANCELLED= done-states migrate to the lifecycle sequence, and a legacy =CANCELLED= decision still parses as a done-state, so cookies stay correct). This file's own header now carries both lines; the Design section documents the two-sequence rule and the retrofit rewrites legacy headers to it. New acceptance criterion: cookies must compute by org, not hand counting.
+
+** DONE Relink behavior is too vague for a safe migration :blocking:
+(Codex; the Claude reviewer independently flagged the synced-.ai/ slice — a downstream rewrite there is reverted by the next template sync, e.g. =startup.org:154='s reference to a spec candidate.) The retrofit named no scan scope, link-shape list, rewrite rule, residue policy, or dry-run format — the implementer would have had to invent the migration's data-safety contract.
+Response: the retrofit section now carries the explicit contract: rewritten roots (=todo.org=, =.ai/notes.org=, =docs/**=, project-owned =.ai/= dirs), reported-never-rewritten surfaces (=.ai/sessions/=, git history, synced template paths — with the canonical rulesets file named in the report), supported link shapes (org =file:= links; bare paths report-only), relative-path recomputation, dry-run default with =--apply=, post-apply residue grep gating exit status, and refuse-loudly on ambiguity.
+
+** DONE Sort marker and startup nudge do not name the actual state surface :blocking:
+(Codex; the Claude reviewer rated the same gap minor — Codex's version was sharper: startup reads =.ai/notes.org=, not a root =notes.org=, and Workflow State may not exist.)
+Response: the marker is pinned to =.ai/notes.org='s =* Workflow State= (the =:LAST_AUDIT:= / =:LAST_INBOX_PROCESS:= surface), created idempotently as task-audit already does; the Design section now spells the Phase A probe command, its exact fire condition, and the Phase C one-liner.
+
+** DONE Phase order can strand legacy specs behind the new review precondition :blocking:
+(Codex; the Claude reviewer found the same at medium severity.) Hardening spec-review's path precondition in Phase 1 while piles stay unsorted until Phases 3-4 would make every legacy spec unreviewable in the gap.
+Response: Phase 1 now carries the compatibility rule — legacy =-spec.org= locations stay reviewable (with a "run spec-sort" nudge) until the project stamps =:LAST_SPEC_SORT:=; the precondition hardens only after. Acceptance criterion 5 updated to match.
+
+** DONE No owner for the DOING → IMPLEMENTED flip :blocking:
+(Claude reviewer.) spec-create owns =DRAFT= and spec-review owns =DRAFT= → =READY=, but implementation finishes outside the spec trio, and "a human edits it" is the exact mechanism whose failure produced this spec (.emacs.d's six shipped-but-unmarked specs).
+Response: the Design section now has a transition-ownership table naming an owner for every flip. =READY= → =DOING= belongs to spec-response; =DOING= → =IMPLEMENTED= is a tracked obligation — spec-response's phase-to-task breakdown always emits a final "flip the spec" task — with task-audit's reconcile pass as the safety net (flag any =DOING= spec whose implementation tasks are all closed). Phase 1 includes both workflow edits.
+
+** DONE Classification heuristic is precedence-ambiguous
+(Claude reviewer.) "Decisions plus phases or Metadata table" reads two ways, and =docs/design/task-review.org= (Metadata table, no spine) classifies differently under each.
+Response: one predicate now — spec candidate iff the doc carries *both* a =Decisions= heading *and* an =Implementation phases= heading; a Metadata table alone does not qualify. The task-review.org counter-case is cited in the retrofit step.
+
+** DONE spec-sort never renames moved files to the -spec.org suffix
+(Claude reviewer.) spec-review's Phase 0 hard-requires the suffix, so a retrofitted legacy spec without it would be unreviewable in its new home.
+Response: retrofit step 2 now renames moved files to carry =-spec.org= when they lack it; the relink pass covers the rename like any move. Acceptance criterion 3 checks the suffix on the re-homed root specs.
+
+** DONE Clicked id: links won't resolve in Craig's Emacs
+(Claude reviewer.) =org-id-locations= indexes only agenda and visited files, so fresh =:ID:=s in =docs/specs/= are invisible-until-clicked broken — the convention would trade visible link breakage for invisible breakage.
+Response: named as an explicit .emacs.d-side prerequisite in the Rename-safe-links section (=org-id-extra-files= over =docs/specs/= globs, or periodic =org-id-update-id-locations=), carried in the Phase 4 note to .emacs.d, with the =rg= recipe as the interim fallback.
+
+** DONE Acceptance criterion 2 contradicts the Metadata Status mirror
+(Claude reviewer.) "Exactly one keyword edit" was irreconcilable with the mandated mirror update.
+Response: a transition is now defined everywhere as three lines in one file — keyword, history line, mirror — still no rename and no link edits. Goals, Design, and criterion 2 all say the same thing.
+
+** DONE Synced helper placement ignores the canonical/mirror split :blocking:
+The spec says to build =.ai/scripts/spec-sort= and update =.ai/workflows/= behavior, but rulesets' current contract is that =claude-templates/.ai/= is canonical and the repo-root =.ai/= tree is only the committed mirror kept honest by =scripts/sync-check.sh=. =CLAUDE.md= explicitly warns that mirror-only edits get silently reverted by the next sync, and =make test= runs the mirror-side tests only after the canonical copy has been synced. V1 should say every shared workflow/script edit lands in =claude-templates/.ai/{workflows,scripts}/= first, then =scripts/sync-check.sh --fix= updates the mirror; =spec-sort= tests should be placed in the synced script-test tree and the acceptance criteria should include =sync-check= / workflow-integrity where relevant. (blocking)
+Response: the retrofit section now opens with the canonical-placement contract (helper + tests in =claude-templates/.ai/scripts{,/tests}/=, workflow edits canonical-side, =sync-check --fix= propagates, both sides commit together); Phases 1 and 2 name it per artifact; new acceptance criterion requires =sync-check= to exit clean after the build commits.
+
+** DONE Task-audit safety net has no spec-to-task binding :blocking:
+The spec says task-audit flags a =DOING= spec whose implementation tasks are all closed, but current =task-audit.org= audits open =todo.org= tasks and has no model for scanning =docs/specs/=, finding a spec's implementation tasks, or deciding "all closed" after =todo-cleanup.el --convert-subtasks= rewrites completed child tasks into dated entries. The added final "flip to IMPLEMENTED" task also means there may always be one open task, so a naive "all tasks closed" check never fires. V1 should define the binding spec-response writes into =todo.org= (for example a parent task property or stable link to the spec ID), the exact audit query, how converted dated entries count, and whether the final flip task is excluded from or satisfies the reconciliation rule. (blocking)
+Response: spec-response's decomposition now stamps a =:SPEC_ID:= property (the spec's status-heading UUID) on the build parent task — the durable binding. The audit query is defined: for each =DOING= spec, find the task with matching =:SPEC_ID:=; flag when that parent is closed, archived, or missing. Checking the parent's keyword (not "all children closed") dissolves both the flip-task chicken-and-egg and the dated-entry conversion concern. New acceptance criterion exercises the flag.
+
+** DONE spec-sort apply path can leave a half-migrated tree :blocking:
+The retrofit contract has dry-run by default and a post-apply residue grep, but it does not say what happens when =--apply= has moved files and then a relink, parse, or residue check fails. Because the operation mutates filenames, links, headers, IDs, and =.ai/notes.org= together, a partial failure can strand the project in the exact mixed state the tool is meant to prevent. V1 should require a clean-worktree preflight (or an explicit dirty-tree refusal/override), validate the full move/relink plan before the first write, write from a single recorded plan, and define recovery behavior for every failed apply: no files moved, automatic rollback, or a printed =git restore= / =git revert= recovery recipe that is safe for uncommitted local edits. (blocking)
+Response: the relink contract's safety block now specifies the fail-safe apply: clean-worktree preflight (refuse on dirty, explicit =--allow-dirty= override that prints what recovery loses), full plan computed + validated + written to a plan file before the first write, execution from the recorded plan, and mid-apply failure stopping with a named applied/not-applied breakdown plus the =git restore= recovery recipe — safe by construction because preflight required a clean tree. Bats covers the preflight and the forced-failure recovery output (Phase 2, plus a new acceptance criterion).
+
+** DONE org-id Emacs prerequisite is not executable as written :blocking:
+The spec says the .emacs.d-side fix can be =org-id-extra-files= over =docs/specs/= globs, but Emacs' own docstring says =org-id-extra-files= is a list of additional files and is only relevant when =org-id-track-globally= is set; it does not establish that project glob strings will be expanded or that every project root will be discovered. The rollout also converts links during the rulesets pilot before the Phase 4 note asks .emacs.d to make clicked =id:= links resolvable. V1 should either keep =file:= links until the Emacs support has landed, or specify the executable Emacs-side implementation precisely: how project =docs/specs/*.org= files are enumerated into =org-id-extra-files= or fed to =org-id-update-id-locations=, when it runs, how it is tested, and how rollout avoids a window where converted links do not click through. (blocking)
+Response: took the fork that removes the window entirely — the pilot and every sort run rewrite =file:= links only; =:ID:= properties are still assigned (harmless, enables later mechanics); conversion to =id:= is a separate follow-up pass gated on .emacs.d landing an executable id-index mechanism, now specified concretely (enumerate =docs/specs/*.org= into =org-id-extra-files= as real file names under =org-id-track-globally=, or feed the enumeration to =org-id-update-id-locations=; verified by clicking a known link). Decision 3 carries a sequencing-refinement note; a new acceptance criterion asserts zero =id:= links exist after the pilot.
+
+** DONE Status confirmation can still encode stale reality :blocking:
+The retrofit proposes lifecycle status from a doc's current =Status= field or review history, then asks a human to confirm. Those are the same stale/incomplete signals that caused the original .emacs.d sweep: shipped specs and dead specs were only knowable by reading code/tasks against the spec. If =spec-sort= only confirms a guessed keyword, the pilot can produce a clean-looking board whose state is still wrong. V1 should define status-confirmation evidence: for each spec candidate, what sources the helper shows (current Status, decision/finding cookies, linked =todo.org= parent state, recent history, matching implementation files/tests), what default is allowed when evidence is inconclusive, and that =IMPLEMENTED= / =SUPERSEDED= / =CANCELLED= require an explicit reason in the status history line. (blocking)
+Response: retrofit step 2 now defines the evidence panel the helper shows per candidate (Status value, cookie states, bound/linking =todo.org= task state, recent history entries, cheap existence checks on phase-named artifacts) with the keyword proposed from the evidence, not the Status field alone. Inconclusive evidence defaults to the most conservative non-terminal state; =IMPLEMENTED= / =SUPERSEDED= / =CANCELLED= always require an explicit human-stated reason recorded in the history line.
+
+* Implementation phases
+
+** Phase 1 — Rule + template updates
+Write =claude-rules/docs-lifecycle.md=. Update spec-create (emit into =docs/specs/=, the two-sequence keyword header, status heading with =:ID:= in the template, transition mechanics), spec-review (path expectation with the compatibility rule below; flipping =DRAFT= → =READY= on a passing review updates keyword + history + mirror), spec-response (owns =READY= → =DOING=; its decomposition stamps the =:SPEC_ID:= binding on the build parent and always emits the final "flip to IMPLEMENTED" task), and task-audit (one reconcile bullet running the =:SPEC_ID:= query: a =DOING= spec whose bound parent is closed, archived, or missing gets flagged). All four are synced assets: edits land in =claude-templates/.ai/= (and =claude-rules/=), the mirror follows via =sync-check --fix=, both commit together. *Compatibility rule:* spec-review keeps accepting legacy =-spec.org= locations (=docs/= root, =docs/design/=) until the project's =:LAST_SPEC_SORT:= is stamped, nudging "run spec-sort" when it meets one; only after the stamp does the =docs/specs/= precondition harden. No legacy spec is ever unreviewable during the transition. Tree stays working: new specs land in the new shape; old specs remain reviewable until their project sorts.
+
+** Phase 2 — The =spec-sort= helper
+Build =claude-templates/.ai/scripts/spec-sort= (classify → evidence-based confirm → plan + validate → move + rename + prepend status heading + assign =:ID:= → relink =file:= references → stamp =:LAST_SPEC_SORT:=), with bats coverage in =claude-templates/.ai/scripts/tests/= (glob-discovered by =make test=) for classification, the evidence/confirm gate, plan validation, moving + renaming, relinking, the clean-worktree preflight, mid-apply failure recovery output, idempotence, and the marker stamp. Mirror synced via =sync-check --fix= in the same commit. Tree stays working: the script is callable but nothing invokes it yet.
+
+** Phase 3 — Pilot on rulesets
+Run =spec-sort= against rulesets' own =docs/= (41 design files, 3 spec-spine candidates, 2 stray root specs). Fix what the pilot surfaces before any other project runs it. Tree stays working: moves are confirmed one by one, links updated in the same pass.
+
+** Phase 4 — Startup nudge + broadcast
+Add the Phase A probe + Phase C nudge line (the concrete contract in the Design retrofit section). Send .emacs.d a note that the convention is live, its ~28-doc pile is ready to sort, and the id-index mechanism is its side of the staged link conversion: enumerate each project's =docs/specs/*.org= into =org-id-extra-files= as real file names (with =org-id-track-globally= t) or feed that enumeration to a periodic =org-id-update-id-locations=, verified by clicking a known id link. The =id:= link-conversion pass across projects runs only after that lands — it is follow-up work, not part of v1's sort runs. Tree stays working: the nudge is one read-only line per session until acted on; every link is a working =file:= link until conversion day.
+
+* Acceptance criteria
+- [ ] =rg '^\* (DRAFT|READY|DOING|IMPLEMENTED|SUPERSEDED|CANCELLED) ' docs/specs/= lists every rulesets spec with its state, and the answer matches reality.
+- [ ] A status transition on a spec changes exactly three lines in one file — the keyword, a history line, and the Metadata mirror — with no rename and no link edits.
+- [ ] Every doc remaining in rulesets =docs/design/= is a note (lacks the Decisions + Implementation-phases spine); both stray =docs/= root specs are re-homed and carry the =-spec.org= suffix.
+- [ ] All inbound links in the rewritten roots resolve after the pilot, and the post-apply residue grep returns zero.
+- [ ] The spec's own decision/finding =[/]= cookies compute correctly under the two-sequence keyword header (org, not hand counting).
+- [ ] spec-create emits new specs into =docs/specs/= in the new shape; spec-review accepts legacy locations until =:LAST_SPEC_SORT:= is stamped and refuses them after.
+- [ ] Every helper/workflow artifact of this feature lives canonical-side (=claude-templates/.ai/=, =claude-rules/=) with the mirror in sync — =scripts/sync-check.sh= exits clean after the build commits.
+- [ ] A =DOING= spec whose =:SPEC_ID:=-bound parent task is closed or missing is flagged by task-audit's reconcile pass (exercised in the pilot or a fixture).
+- [ ] =spec-sort --apply= on a dirty worktree refuses (absent the override); a forced mid-apply failure in the bats suite yields the named-recovery output, not a half-migrated tree.
+- [ ] After the pilot, no link the sort *rewrote* uses =[[id:...]]= form and no rewritten root gained a new =id:= link targeting a spec (conversion is the gated follow-up); every rewritten link is a resolving =file:= link. The check scopes to actual rewritten and spec-target links — literal prose mentions of the id syntax (which already exist in =todo.org= and older specs) don't count, so a naive whole-file grep is the wrong implementation.
+- [ ] A project with an unsorted =docs/design/= gets the startup nudge; one confirmed =spec-sort= run clears it via =:LAST_SPEC_SORT:=.
+
+* Readiness dimensions
+- Data model & ownership: the spec file owns its state; the Metadata mirror is display-only. No external index to drift.
+- Errors, empty states & failure: =spec-sort= on a project with no =docs/= is a silent no-op; an ambiguous classification is surfaced, never auto-moved; a relink pass that finds zero inbound links is normal.
+- Security & privacy: N/A because the docs are already in-repo; no new exposure surface.
+- Observability: the status grep is the dashboard; =spec-sort= prints every proposed move and every rewritten link.
+- Performance & scale: N/A because collections are tens of files; everything is one-shot or grep-speed.
+- Reuse & lost opportunities: reuses org TODO keywords, org-id, the existing scheme-header pattern of declared vocabularies, and spec-review's in-file findings convention.
+- Architecture fit & weak points: weak point is the classification heuristic — mitigated by the confirm gate. The status heading is additive, so old readers of spec files see one extra heading and nothing breaks.
+- Config surface: none new — one marker line (=:LAST_SPEC_SORT:=) in the existing Workflow State section.
+- Documentation plan: the docs-lifecycle rule is the documentation; spec-create's template is the worked example.
+- Dev tooling: =spec-sort= ships with bats tests under the existing glob-discovered suite.
+- Rollout, compatibility & rollback: additive per project, one project at a time, rulesets first. Rollback of a sort is =git revert= of the pilot commit (moves + relinks are one commit).
+- External APIs & deps: N/A — plain files, =rg=, =uuidgen=.
+
+* Risks, Rabbit Holes, and Drawbacks
+- *Relink misses an inbound link shape* (org radio links, bare paths in scripts). Dodge: the pilot greps for the old path after moving and fails loudly on any residue.
+- *Heuristic over-classifies notes as specs.* Dodge: the confirm gate is mandatory; the helper never moves unconfirmed.
+- *Keyword vocabulary drift* between this spec, the rule, and spec-create's template. Dodge: the rule names the vocabulary once and the others link it.
+
+* Testing / Verification / Rollout
+bats for =spec-sort= (classification, the evidence/confirm gate, plan validation, move + rename, relink, the clean-worktree preflight, forced mid-apply failure recovery output, idempotence, marker stamp). The pilot run on rulesets is the live verification; the post-move residue grep is the acceptance check. Rollout is per-project via the startup nudge, each run human-confirmed.
+
+* References / Appendix
+- Source proposal: [[file:../design/2026-06-15-spec-storage-lifecycle-proposal.org]] (.emacs.d handoff, 2026-06-15).
+- Decisions record: todo.org "Spec storage location + lifecycle-status convention" (settled 2026-06-28).
+- This file is the convention's first resident: it lives in =docs/specs/=, carries the status heading + =:ID:=, and drops the status filename suffix.
+
+* Review and iteration history
+** 2026-07-01 Wed @ 22:13:00 -0400 — Claude — author
+- What: initial draft, written from the five pre-ratified decisions.
+- Why: the queued-specs half of the 2026-06-30 session goal; decisions were settled 2026-06-28 and needed migration into a buildable spec.
+- Artifacts: todo.org task "Spec storage location + lifecycle-status convention"; source proposal above.
+
+** 2026-07-01 Wed @ 22:22:34 -0400 — Codex — reviewer
+- What changed or was recommended: rubric =Not ready=. Four blocking findings were added: preserve Org task keywords while adding lifecycle status, make =spec-sort= relinking executable and failure-safe, define the actual =.ai/notes.org= marker/startup-nudge contract, and avoid stranding legacy specs behind a stricter path precondition before retrofit.
+- Why: current rulesets workflows still depend on =TODO= / =DONE= decision and finding tasks, startup state lives in =.ai/notes.org=, and the repo still contains formal specs outside =docs/specs/= until the migration runs.
+- Artifacts: Review findings section; current-state checks against =.ai/workflows/spec-create.org=, =.ai/workflows/spec-review.org=, =.ai/workflows/startup.org=, =scripts/sync-check.sh=, and =todo.org=.
+
+** 2026-07-01 Wed @ 22:25:00 -0400 — Claude (fresh-context agent) — reviewer
+- What: rubric =Not ready=. Independently found Codex's keyword-vocabulary blocker (adding the cross-sequence uniqueness wrinkle) and the stranded-legacy-specs and marker-surface gaps, plus five findings of its own: no owner for the =DOING= → =IMPLEMENTED= flip (blocking), the precedence-ambiguous classification heuristic, the missing =-spec.org= rename in spec-sort, org-id click-resolution in a live Emacs, and the criterion-2/mirror contradiction.
+- Why: fresh-eyes adversarial pass requested by Craig after his own read found nothing; the two reviews converging on the same worst bug from independent context is the confidence signal.
+- Artifacts: Review findings section (findings 5-9); spot-checks against real repo files (=docs/design/task-review.org=, the two stray root specs, =startup.org:154=).
+
+** 2026-07-01 Wed @ 22:46:52 -0400 — Claude — second responder pass
+- What: fixed all five of Codex's re-review findings in place (fourteen of fourteen closed): canonical-placement contract for every synced artifact (+ sync-check acceptance criterion), the =:SPEC_ID:= spec-to-task binding with the parent-keyword audit query (dissolving the flip-task chicken-and-egg), the fail-safe =--apply= contract (clean-tree preflight, validate-then-write from a recorded plan, named recovery), staged id-link conversion (pilot rewrites =file:= links only; =id:= conversion gated on the concrete .emacs.d id-index mechanism — the fork Craig approved), and evidence-based status confirmation (evidence panel, conservative non-terminal default, reasons required for terminal states). Status stays DRAFT; the READY flip belongs to the reviewers this round.
+- Why: Craig approved fixing all five ("1", 2026-07-01), including the keep-file:-links-through-pilot fork.
+- Artifacts: per-finding responses inline; the fixed Design/phase/criteria sections.
+
+** 2026-07-01 Wed @ 23:22:50 -0400 — Codex — reviewer
+- What changed or was recommended: rubric =Ready=. No new blocking findings. The second responder pass closed all five Codex re-review blockers without regressing the first nine findings, and the spec now gives implementers concrete contracts for canonical synced assets, =:SPEC_ID:= task binding, fail-safe =spec-sort --apply= behavior, staged id-link conversion, evidence-based status confirmation, phase sequencing, and test coverage.
+- Why: the current spec can be implemented and tested without hidden product decisions; remaining vNext work is separately tracked.
+- Artifacts: status heading flipped to =READY=; =* Decisions= [5/5]; =* Review findings= [14/14]; Emacs batch cookie check.
+
+** 2026-07-01 Wed @ 22:41:21 -0400 — Claude (fresh-context agent) — verify pass; Claude — READY flip
+- What: the original reviewer re-read the fixed spec against its own nine findings: all held, none regressed, verdict ready. It re-ran the classification predicate live (exactly 5 candidates; task-review.org excluded) and confirmed org computes the cookies. Two non-blocking minors folded in before the flip: a refinement note under Decision 2 (whose frozen body still said "one keyword edit") and a wider nudge probe that also fires on stray =docs/*-spec.org= root files. Status flipped DRAFT → READY.
+- Why: Craig authorized the flip contingent on the verify pass clearing; it did.
+- Artifacts: the status heading's history line; verify-pass report in the session record.
+
+** 2026-07-01 Wed @ 22:41:33 -0400 — Codex — reviewer
+- What changed or was recommended: rubric =Not ready=. Five new blocking findings were added after the response pass: make shared workflow/script edits obey the =claude-templates/.ai/= canonical plus =.ai/= mirror contract; define how task-audit binds a =DOING= spec to its implementation tasks; make =spec-sort --apply= failure-safe; turn the org-id Emacs prerequisite into an executable rollout step; and require status confirmation to be evidence-based rather than a rubber-stamp of stale fields.
+- Why: the response fixed the original keyword/relink/precondition issues but introduced new integration points in synced template assets, task-audit, Emacs id resolution, and migration safety that are not yet buildable from the spec.
+- Artifacts: Review findings section; checks against =CLAUDE.md=, =scripts/sync-check.sh=, =.ai/workflows/task-audit.org=, =.ai/workflows/startup.org=, =.ai/notes.org=, current =docs/= inventory, and Emacs batch/docstring checks for Org TODO cookies and =org-id-extra-files=.
+
+** 2026-07-01 Wed @ 22:30:06 -0400 — Claude — responder
+- What: merged both reviews into one findings ledger (nine findings, all dispositioned accept) and fixed all nine in place: two-sequence keyword header (applied to this file itself), transition-ownership table with the tracked flip-to-IMPLEMENTED task, single classification predicate, the -spec.org rename step, the full relink data-safety contract, the =.ai/notes.org= marker + Phase A/C startup contract, the legacy-location compatibility rule, the org-id Emacs prerequisite, and the three-line transition definition. Acceptance criteria updated to match.
+- Why: Craig approved fixing all nine ("1", 2026-07-01); none touched the five ratified decisions.
+- Artifacts: Review findings section (responses inline per finding); the fixed sections themselves.
diff --git a/docs/specs/2026-07-14-sentry-workflow-spec.org b/docs/specs/2026-07-14-sentry-workflow-spec.org
new file mode 100644
index 0000000..307d0be
--- /dev/null
+++ b/docs/specs/2026-07-14-sentry-workflow-spec.org
@@ -0,0 +1,278 @@
+#+TITLE: Sentry Workflow — Spec
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-14
+#+TODO: TODO | DONE
+#+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED
+
+* IMPLEMENTED sentry workflow
+:PROPERTIES:
+:ID: f6c51f27-d7a2-4b63-9ff9-5ba005a66dfb
+:END:
+- 2026-07-19 Sun @ 05:04:00 -0500 — IMPLEMENTED: all four phases built, committed, and pushed (agent-lock a8b6cf4, engine ccc9c26, companions c6383e9). Full suite green throughout. The overnight live trial is handed to Craig as a structured manual-testing task; its findings file as follow-up tasks, not a build gate.
+- 2026-07-19 Sun @ 04:35:57 -0500 — DOING: build started. Decomposed into the four implementation phases as todo.org build tasks under the sentry parent (SPEC_ID-bound); running in no-approvals + auto-flush mode.
+- 2026-07-19 Sun @ 04:35:57 -0500 — READY: Craig completed his deep read and approved the spec for build. Gate passed.
+- 2026-07-14 Tue @ 02:03:28 -0500 — all 12 review findings dispositioned live with Craig and folded into the design; decisions now 10/10; still DRAFT pending Craig's deep read.
+- 2026-07-14 Tue @ 00:52:03 -0500 — drafted, all nine design decisions resolved live with Craig during the authoring session.
+
+* Metadata
+| Status | implemented |
+|----------+------------------------------------------|
+| Owner | Craig Jennings |
+|----------+------------------------------------------|
+| Reviewer | (spec-review, next session) |
+|----------+------------------------------------------|
+| Related | [[file:../../todo.org][todo.org — sentry build task]] |
+|----------+------------------------------------------|
+
+* Summary
+
+Sentry is one supervisor workflow that runs a project's hygiene passes serially on a cadence, holds locks so it never collides with itself or other agents, commits its work to a host-suffixed side branch for morning review, and checkpoints between passes so a crash loses at most one pass. It replaces the idea of firing inbox-zero and triage-intake as separate uncoordinated crons.
+
+* Problem / Context
+
+An agent session left running overnight can keep a project clean, but the pieces don't coordinate. There is no lock anywhere in .ai/scripts/ (verified 2026-07-14): capture-guard covers only the Emacs org-capture case on one file, and it's advisory. Multiple sessions (several projects, sometimes both daily drivers) can read-modify-write ~/org/roam/inbox.org and the agents/ KB nodes at once. The only coordination is git optimism, and we know it loses: the KB search already globs out *sync-conflict* files because pull/push races fork them.
+
+Separate crons compound it: inbox-zero and triage-intake fired independently can overlap each other and any interactive session on the same shared files.
+
+Proposal origin: the work project, 2026-07-13 (preserved at [[file:../design/2026-07-14-sentry-workflow-proposal.org][docs/design/2026-07-14-sentry-workflow-proposal.org]]). This spec supersedes the proposal where they differ: the skeptical review found four conflicts with existing discipline, each resolved as a Decision below.
+
+* Goals and Non-Goals
+
+** Goals
+- One serialized runner per project: passes never overlap each other, and a fire that finds a prior fire still running skips.
+- A shared-file lock that actually serializes same-host agents writing public roam files.
+- Crash-lossless to one-pass granularity: session-context entry plus a commit after every disk-writing pass.
+- Main stays clean: overnight commits land on a host-suffixed sentry branch so cross-machine pulls and the template sync keep working.
+- Nothing destructive fires unattended: trash/mark-read, regrades, and file moves queue for morning approval.
+- Portable across projects with zero configuration: passes detect their targets and skip where not applicable.
+
+** Non-Goals
+- No report-only or degraded pass mode. A pass runs fully or it doesn't run (Craig's explicit direction). The morning-approval queue is a permanent division of labor, not a degraded mode.
+- No cross-host locking. The lock is host-local; cross-host races stay with roam-sync's abort-loudly rebase.
+- No unattended /schedule contract. Sentry runs inside a live session Craig starts, driven by /loop, the same boundary inbox.org draws for auto inbox zero.
+- Sentry never runs git against the roam repo. roam-sync remains that repo's only committer.
+
+** Scope tiers
+- v1: the sentry.org engine (ten mechanical passes, entry gates, branch mechanics, digest, the stop-sentry operation), the agent-lock helper script with bats tests, the roam-write lock adopted by every roam write path, companion-file reconciliations (knowledge-base.md write recipe, inbox.org core §5, roam-sync.sh header, triage-intake.org note, wrap-it-up.org active-sentry guard), INDEX.org entry.
+- Out of scope: changes to roam-sync.sh behavior; changes to the wrap-up teardown feature.
+- vNext (logged to todo.org): cross-host roam conflict surfacing; the KB lesson-promotion pass (blocked on the lesson-detection heuristic task); a fully-unattended /schedule variant if the interactive shape proves too narrow.
+
+* Design
+
+** For Craig (the user altitude)
+
+You type the trigger in a project session ("run sentry", or the /loop line it expands to). Sentry first checks its entry ticket: the project's notes.org must carry :COMMIT_AUTONOMY: yes, or sentry declines to start and names the marker. Then the entry gates run while you're still at the terminal: a dirty tracked tree stops and describes what's dirty, file by file, and offers numbered options (finish the job, stash it, or name changes to roll back); a red test suite stops the same way with an offer to investigate. The loop starts only once both are answered.
+
+Launching is a handoff — the launch contract: sentry owns the repo until the morning merge. Working in that repo mid-night means saying "stop sentry" first (it cancels the loop and walks the branch and queue disposition), and open Emacs buffers on project files want a revert after launch and again after the merge. "Wrap it up" while sentry is active refuses and points at "stop sentry".
+
+Each fire (hourly by default) walks the pass list serially and appends one timestamped digest to the session log: failures as banners on top, then one line per pass, then the morning-approval queue. A pass with nothing to do is one line. Anything destructive or judgment-shaped waits in that queue for you.
+
+In the morning you review the sentry/<date>-<host> branch (a stack of small unpushed chore(sentry) commits), squash-merge it into your working branch, delete it, and answer the approval queue. If a night's branch is still unmerged when sentry next starts, that fire skips and says so.
+
+** For the implementer
+
+*The engine.* A synced template workflow, .ai/workflows/sentry.org, driven by /loop <interval> (default hourly). Each fire acquires the single-runner lock, verifies branch state, walks the pass list, writes the digest, commits the spine residue, and releases the lock. The runner re-touches the lock between passes (the heartbeat), so a live run's lock is never older than one pass. Every pass follows one contract: probe, work, session-context entry, then a commit when the pass wrote to disk. The probe asks whether the pass's target exists in this project; when it doesn't, the pass is one skip line and nothing more. The session-context entry precedes the commit so a crash between them still leaves the trail.
+
+*Branch mechanics.* At loop start (after the entry gates and a fetch-and-ff-only reconcile of the project branch — a diverged branch joins the interactive gate), sentry creates sentry/YYYY-MM-DD-<host> from HEAD and checks it out; the host suffix prevents a same-date collision between the daily drivers. Commits are conventional, one per writing pass: chore(sentry): <pass> — <what changed>. Nothing pushes. At entry, an existing unmerged sentry/* branch means skip-and-note. No second branch stacks on the first. Morning teardown (review, squash-merge, delete) is Craig's, documented in the workflow, never automated.
+
+*Locks.* One helper script, .ai/scripts/agent-lock, serves both locks. flock is unusable here: every Bash call an agent makes is its own short-lived shell, so a flock dies with the call that acquired it. Instead: atomic mkdir as acquire (create-and-win or fail-and-lose), a metadata file inside recording PID, hostname, and ISO timestamp, and age-based staleness reclaim so a crashed run's lock expires instead of wedging every later fire. Subcommands: acquire <name> [--ttl], release <name>, status <name>. Lock homes: /run/user/<uid>/agent-locks/<name>/ for both locks — the helper owns the path scheme, callers pass names — with ~/.cache/agent-locks/ as the fallback where no runtime dir exists. tmpfs makes a lock host-local by construction, keeps it out of every repo (a lock inside ~/org/roam would ride roam-sync's git add -A to the other machine as a phantom hold), and clears it on reboot. Contention is a bounded wait (~30s), then defer-and-note. With the heartbeat keeping a live lock young, TTLs size to the longest single pass; every reclaim surfaces in the digest.
+
+*Roam writes.* A pass touching a public roam file acquires the roam-write lock, runs capture-guard --wait (the human-capture layer stays underneath), edits the working tree, triggers roam-sync with systemctl --user start roam-sync.service, and releases the lock. Sentry never runs a git write against ~/org/roam, and pass 1's ff-only pull is its only git read of that repo. The 06-24 one-git-owner rule holds. The lock spans only edit-plus-trigger; roam-sync itself is serialized by systemd (a oneshot unit never runs concurrently with itself).
+
+*Unattended safety.* With no one at the terminal, any unsafe state (unexpected dirty tree, red suite, lost lock, unmerged prior branch) makes the affected scope (the pass, or the whole fire, whichever the state poisons) skip with one digest line. The next fire retries; if the state persists to morning, the interactive entry gate handles it with Craig present. Skips are never silent and never partial: no pass runs in a reduced form. Two refinements keep the machinery honest: the dirty check excludes the spine set (session-context files, path resolved via session-context-path) so sentry's own bookkeeping can't trip it, and after the second consecutive fire skipped on an unmerged prior branch, sentry sends one persistent desktop notification naming the project and branch, then repeats at most daily — a multi-day stall never stays silent.
+
+*Pass list (v1), in order, each with its detection probe:*
+
+1. Roam pull — git -C ~/org/roam pull --ff-only, skipped when the tree is dirty (roam-sync owns that case) or the clone is absent. Read-only; the one narrow exception to "don't touch roam git," pull-only and ff-only, so later reads are fresh.
+2. Inbox zero — inbox.org roam mode, run under the engine's no-approvals contract (quick+solo+agreed items execute; shared-asset proposals park; everything lands in the morning queue). Probe: roam clone or project inbox/ exists.
+3. Triage intake — triage-intake.org. Probe: triage plugins present. Destructive actions queue.
+4. Todo cleanup — clean-todo.org mechanics. Probe: root todo.org.
+5. Task audit — task-audit.org. Probe: root todo.org. Regrades queue.
+6. Working-files hygiene — flag working/<slug>/ dirs whose task is closed. Probe: working/ exists. Filing queues.
+7. Spec status board — the docs-lifecycle grep. Probe: docs/specs/ exists.
+8. Link integrity — broken file: links in the project's org files. Probe: lint-org.el present.
+9. Git health — drift, unpushed commits, stale branches, main-behind-origin. Probe: .git.
+10. Prep + symlink freshness. Probe: the prep dir / symlinks exist (work and home only, in practice).
+(KB lesson promotion, the proposal's eleventh pass, is deferred to vNext — see the KB finding and the filed lesson-detection-heuristic task. v1 ships the ten mechanical passes above.)
+
+*Digest.* Appended per fire to the session-context.org Session Log (which the spine already writes), so it survives a crash, rides the session archive, and is on screen in the running session. The morning-approval queue accumulates under one heading in the same file — each item carries what, why, and the exact command or edit that fires on approval.
+
+* Alternatives Considered
+
+** Separate crons per task (status quo direction)
+- Good, because each piece stays independently simple.
+- Bad, because nothing serializes them: the observed sync-conflict forks are this cost, already paid.
+- Bad, because N crons means N session lifecycles to manage instead of one.
+
+** flock for both locks (as proposed)
+- Good, because it's the standard tool and kernel-enforced.
+- Bad, because it binds to a living process; agent tool calls are short-lived shells and /loop turns share none, so the lock evaporates on return. Disqualifying.
+
+** Per-pass commits on the current branch (as proposed)
+- Good, because commits sit where the edits apply; no morning merge step.
+- Bad, because unpushed commits on main on two daily drivers diverge main by morning, the exact state startup's fast-forward refuses and the template-sync guard blocks on. Disqualifying with two machines.
+
+** Degraded report-only mode for unsafe states
+- Good, because some findings still land when committing is unsafe.
+- Bad, because it blurs the contract: a pass that half-ran reads as having run. Craig rejected it outright; skip-and-note is the replacement.
+
+** Per-project pass manifest instead of detection
+- Good, because explicit control over what runs where.
+- Bad, because it's one more thing to maintain and forget; detection activates a pass the day its target appears. Rejected for v1.
+
+* Decisions [10/10]
+
+** DONE Commit target: host-suffixed sentry branch
+- Context: overnight commits must survive two daily drivers that both pull main; unpushed commits on main diverge it across machines, breaking startup fast-forward and the template-sync guard.
+- Decision: We will commit each writing pass to sentry/YYYY-MM-DD-<host>, created from HEAD at loop start, never pushed. Morning flow: review, squash-merge, delete. An unmerged prior sentry branch at entry skips the fire.
+- Consequences: easier — main's ref stays clean for cross-machine sync; a bad night is discarded by deleting one branch. Harder — a daily merge step; the working tree sits on a non-main branch overnight.
+
+** DONE Roam-write lock scope: host-local only
+- Context: observed conflicts are same-host multi-agent; cross-host races go through roam-sync's rebase, which aborts loudly.
+- Decision: We will ship the host-local lock and file cross-host conflict-surfacing as a separate task.
+- Consequences: easier — small, testable v1. Harder — a true cross-host race still forks; we accept the rarity.
+
+** DONE Interval default: hourly, with a knob
+- Context: passes short-circuit in seconds when idle; the real cost is digest noise, not compute.
+- Decision: We will default to hourly and expose the interval as the /loop argument.
+- Consequences: easier — projects stay clean through the night. Harder — a capture-heavy evening produces several small commits; the knob is the escape hatch.
+
+** DONE Entry gates are interactive, not silent
+- Context: Craig types the sentry command in, so the first fire runs with him at the terminal.
+- Decision: We will stop at entry on a dirty tracked tree (describe what's dirty; offer finish-the-job / stash / rollback-named-changes) and on a red suite (describe failures; offer to investigate). The loop starts only after both are answered. Untracked files never block.
+- Consequences: easier — no guessing about his in-progress work. Harder — sentry can't start unattended from a dirty state; that's the point.
+
+** DONE No report-only mode; unsafe unattended states skip
+- Context: a degraded pass mode blurs whether a pass ran; Craig rejected it explicitly.
+- Decision: We will make every pass run fully or not at all. Unattended unsafe states (dirty tree, red suite, lost lock, unmerged prior branch) skip the poisoned scope with one digest line; the next fire retries. The morning-approval queue for destructive/judgment actions stays — it's a permanent contract, not degradation.
+- Consequences: easier — a pass line in the digest means it fully ran. Harder — a persistent unsafe state means zero hygiene until morning; accepted.
+
+** DONE Roam git discipline: roam-sync stays the only committer
+- Context: the proposal had sentry commit-and-push roam under the lock; the 06-24 fix made roam-sync the repo's single git owner because the tree is chronically dirty from live captures.
+- Decision: We will wrap only edit-plus-trigger in the roam-write lock; sentry never runs git against ~/org/roam (pass 1's ff-only pull is the sole, read-only exception).
+- Consequences: easier — one git owner, no mid-rebase states from agents. Harder — an edit lands remotely only when roam-sync fires; the manual trigger closes most of that gap.
+
+** DONE Lock mechanics: mkdir-atomic helper with staleness reclaim
+- Context: flock can't span tool calls (short-lived shells, no shared process across /loop turns).
+- Decision: We will build .ai/scripts/agent-lock — atomic mkdir acquire, PID/host/timestamp metadata, age-based reclaim, acquire/release/status subcommands, bats-tested — and use it for both locks.
+- Consequences: easier — locks survive between calls and self-clear after crashes. Harder — TTL tuning; a reclaim during a genuinely slow pass is possible, so the helper surfaces every reclaim rather than reclaiming silently.
+
+** DONE Autonomy gate: :COMMIT_AUTONOMY: yes is the entry ticket
+- Context: commits.md gates commits on approval; sentry commits unattended, so it needs standing, per-project authorization.
+- Decision: We will have sentry decline to start in any project whose notes.org lacks :COMMIT_AUTONOMY: yes, naming the marker.
+- Consequences: easier — running sentry somewhere is a deliberate grant; no half-running mode to reason about. Harder — read-only passes don't run in ungranted projects either; that's acceptable, since launch is one marker away.
+
+** DONE Pass portability: detection over configuration
+- Context: several proposed passes are work/home-specific; run verbatim elsewhere they're vacuous or error.
+- Decision: We will open every pass with a cheap existence probe and skip with one digest line when the target is absent.
+- Consequences: easier — zero config, passes self-activate when targets appear. Harder — an intentionally-unwanted pass needs a vNext exclusion marker if that ever becomes real.
+
+** DONE Suite policy: entry run plus conditional fire-end run, no per-pass runs
+- Context: the verification discipline requires a full suite run before every commit, but sentry commits per pass, hourly, mostly touching org files the suite doesn't exercise; a per-commit run would turn a seconds-long fire into minutes, all night.
+- Decision: We will run the suite once at entry (the green baseline the gates require) and again at fire-end only when a pass modified files outside the org/spine set. No per-pass runs. The deviation is justified by the unpushed branch and the morning review gating everything before push.
+- Consequences: easier — idle fires stay cheap and the rare code-touching pass is still caught before its commits age. Harder — a suite break introduced by an org-only edit (possible via fixtures) surfaces at morning review rather than at the offending commit.
+
+* Review findings [12/12]
+
+** DONE Overnight working-tree ownership is undefined
+Resolved 2026-07-14 (Craig): launch contract with in-place checkout. Launching sentry hands the repo to sentry until the morning merge; reclaiming it mid-night means stopping the loop first. The entry gate (clean tree, Craig present) fronts the handoff, the unattended dirty-skip backstops anything that slips, and the workflow documents the Emacs buffer-revert caveat at launch and after the morning merge. Worktree isolation was rejected because untracked inbox drops exist only in the main tree, which blinds the inbox pass; plumbing commits were rejected as fragile.
+
+** DONE Roam-write lock guards nothing unless every roam writer acquires it
+Resolved 2026-07-14 (Craig): Phase 3 strengthened from notes to mandatory write-path changes. inbox.org core §5 gains acquire/release around the Phase D edit; knowledge-base.md's write recipe gains the same around its write block. Both degrade gracefully when the helper isn't installed (proceed unlocked — today's behavior); an interactive caller finding the lock busy does a bounded wait (~30s) then surfaces to the user rather than proceeding unlocked.
+
+** DONE Roam lock location gets committed by roam-sync
+Resolved 2026-07-14 (Craig): both locks live under /run/user/<uid>/agent-locks/<name>/, with ~/.cache/agent-locks/ as the fallback where no runtime dir exists. The agent-lock helper owns the path scheme; callers pass names, never paths. tmpfs gives host-locality by construction and reboot self-clearing for free.
+
+** DONE Sentry's own spine writes trip its dirty-tree skip
+Resolved 2026-07-14 (Craig), all three parts: the unsafe-state dirty check excludes the spine set (session-context.org / session-context.d/), each fire ends with one digest commit that sweeps accumulated spine writes so read-only fires still leave a clean tree, and the spine path resolves through .ai/scripts/session-context-path so a concurrent agent's anchor is never clobbered.
+
+** DONE Pass 8 depends on a known data-corrupting bug
+Resolved 2026-07-14, mid-review: the lint-org fix landed in 951b6fc — block-type-aware scanning in both helpers, CLI report-only by default with writes behind --fix, and the true corruption path (wrap-org-table's load-time dispatch firing on lint-org's require) guarded to entry-script-only. Pass 8 runs against the fixed linter; the prerequisite is satisfied and no Implementation-phases dependency is needed.
+
+** DONE Unattended inbox-pass semantics are undefined
+Resolved 2026-07-14 (Craig): sentry runs inbox processing under inbox.org's no-approvals contract verbatim — quick+solo+agreed items execute, shared-asset and convention proposals park (prepared diff, VERIFY task, sender reply). Everything the pass did lands in the morning queue, and anything the engine would ask interactively defers to the queue instead of blocking the fire.
+
+** DONE Per-pass commits skip the pre-commit suite run — undecided convention conflict
+Resolved 2026-07-14 (Craig): recorded as a Decision (see Decisions — suite policy). Entry run establishes the green baseline; no per-pass runs; a fire-end suite run fires only when a pass modified files outside the org/spine set, so the rare code-touching pass is caught before its commits age overnight. Justification recorded with both halves of the consequences.
+
+** DONE KB promotion pass has no defined lesson source
+Resolved 2026-07-14 (Craig): pass 11 is cut from v1 and deferred to vNext. KB promotion stays a wrap-up concern — the editorial moment where Craig answers the promotion prompt. An unattended judgment pass writing to the shared KB waits until sentry has quiet weeks behind it AND a designed detection heuristic; the heuristic design is filed in todo.org ([#D] "KB lesson-detection heuristic", which blocks re-adding the pass).
+
+** DONE Lock contention and reclaim mechanics are half-specified
+Resolved 2026-07-14 (Craig): both mechanics adopted. Contention: bounded wait (~30s, capture-guard's --wait shape), then defer — the pass skips with a digest line and the next fire retries. Reclaim: heartbeat refresh — the runner re-touches the lock timestamp between passes, so a live run's lock is never older than one pass and the TTL sizes to the longest single pass (minutes). Every reclaim surfaces in the digest.
+
+** DONE Wrap-up while the loop is live is undefined
+Resolved 2026-07-14 (Craig): wrap-up refuses while sentry is active. It detects the live single-runner lock and stops with "sentry is active — say 'stop sentry' first." The stop-sentry operation is defined in sentry.org and owns the shutdown: cancel the loop, walk the branch disposition (squash-merge now or leave named), walk-or-carry the approval queue. Wrap-up itself gains only the one guard; the shutdown logic lives with sentry.
+
+** DONE Entry should reconcile the project branch before branching
+Resolved 2026-07-14 (Craig): adopted. After the entry gates pass, sentry runs the same fetch-and-ff-only reconcile startup uses, then creates the branch. A diverged branch joins the interactive entry gate (Craig is present at launch) rather than being auto-resolved.
+
+** DONE Multi-day unmerged branch stalls hygiene silently
+Resolved 2026-07-14 (Craig): adopted. After the second consecutive fire skipped for the unmerged-branch reason, sentry sends one persistent desktop notification naming the project and branch ("sentry stalled: <branch> unmerged — merge or delete to resume"), then repeats at most daily. Persistent notify matches the paging convention: it stays on screen until dismissed.
+
+* Implementation phases
+
+** Phase 1 — agent-lock helper
+.ai/scripts/agent-lock (canonical: claude-templates/.ai/scripts/) with bats tests: acquire/release/status, contention, staleness reclaim, metadata. Tree stays working; nothing calls it yet.
+
+** Phase 2 — sentry.org engine
+The workflow file: entry ticket, interactive gates, branch mechanics, pass runner with the probe→work→log→commit contract, digest + approval queue, skip semantics. INDEX.org entry. Mirror synced.
+
+** Phase 3 — companion reconciliations
+knowledge-base.md write recipe (acquire/release the roam-write lock around the write block; edit-plus-trigger replaces inline pull/commit/push); inbox.org core §5 (acquire/release around the Phase D edit, capture-guard staying as the human layer underneath; graceful degradation when the helper is absent); roam-sync.sh header comment; triage-intake.org note (runs as a sentry pass; own triggers kept); wrap-it-up.org active-sentry guard (refuse and point at "stop sentry").
+
+** Phase 4 — verification
+make test green; a live trial night on rulesets (ratio): entry gates exercised, one fire observed end to end, morning branch review performed; follow-up tasks filed from what the trial surfaces.
+
+* Acceptance criteria
+- [ ] agent-lock: two concurrent acquires produce exactly one winner; release frees; a stale lock (aged past TTL) is reclaimed with a surfaced note; a busy lock is a bounded wait then defer; heartbeat refresh keeps a live lock young; locks live under the runtime dir (cache fallback), never inside a repo; bats suite green.
+- [ ] Sentry declines to start without :COMMIT_AUTONOMY: yes, naming the marker.
+- [ ] Entry on a dirty tracked tree stops with the file-by-file description and the three options; untracked files don't trigger it.
+- [ ] A fire on a clean project produces one digest with every pass either run or skipped-with-reason; no other output shape exists.
+- [ ] Writing passes commit individually to sentry/<date>-<host>; main's ref is untouched; nothing is pushed.
+- [ ] A roam-writing pass acquires the roam-write lock, passes capture-guard, edits, triggers roam-sync, releases, and never runs git write commands against ~/org/roam.
+- [ ] An unmerged prior sentry branch at entry causes a skip-and-note, not a second branch.
+- [ ] Destructive/judgment actions appear only in the approval queue, never executed unattended.
+- [ ] On a project missing a pass target (no todo.org, no triage plugins), the pass skips with one line and the rest run.
+- [ ] Entry reconciles the project branch ff-only before creating the sentry branch; a diverged branch stops at the interactive gate.
+- [ ] Sentry's own spine writes never trip the dirty-tree skip (a full fire of read-only passes ends with a clean tree via the digest commit).
+- [ ] "Wrap it up" during an active loop refuses and names "stop sentry"; "stop sentry" cancels the loop and walks the branch and queue disposition.
+- [ ] The second consecutive unmerged-branch skip produces one persistent desktop notification; repeats are at most daily.
+- [ ] The inbox pass parks a shared-asset proposal (VERIFY + prepared diff + sender reply) rather than applying it, and the morning queue lists everything the pass did.
+
+* Readiness dimensions
+- Data model & ownership: locks own their directories under /run/user/<uid>/agent-locks/ (tmpfs, outside every repo); digest and queue live in session-context.org (session-owned, path via session-context-path); commits on the sentry branch are Craig-authored per commits.md.
+- Errors, empty states & failure: every skip names its reason in the digest; a failed pass banners and the rest continue; the digest surfaces lock-reclaim events. No silent outcomes.
+- Security & privacy: no credentials touched; KB promotion honors the knowledge-base.md work-project refusal contract; commit messages follow the no-tooling-enumeration rule where applicable.
+- Observability: the digest is the observability surface (per-fire, timestamped, failures on top); session-context.org persists it across crashes.
+- Performance & scale: idle fire is seconds (probes + short-circuits); heaviest pass is triage (network); hourly cadence assumed fine; the knob covers it.
+- Reuse & lost opportunities: reuses inbox.org, triage-intake.org, clean-todo, task-audit, lint-org.el, capture-guard, roam-sync wholesale; the only new code is agent-lock and the engine file.
+- Architecture fit & weak points: engine+passes mirrors the engine+plugin convention; weak points are lock TTL tuning (reclaim vs slow pass) and /loop session lifetime (a killed session ends sentry; the lock's TTL clears the residue).
+- Config surface: /loop interval; :COMMIT_AUTONOMY: marker; lock TTLs as script flags with defaults. Nothing else.
+- Documentation plan: the workflow file is the doc (When to Use + the morning-review section); INDEX.org entry; no README change needed.
+- Dev tooling: bats for agent-lock via the existing glob-discovered suites; make test covers it with no Makefile edit.
+- Rollout, compatibility & rollback: additive, new files plus comment-level reconciliations; no consuming project changes behavior until Craig launches sentry there. Rollback is deleting the branch and not launching.
+- External APIs & deps: none beyond tools already in use (git, systemctl, bats). N/A for schema verification.
+
+* Risks, Rabbit Holes, and Drawbacks
+- The heartbeat bounds the TTL question to a single pass, but a pass that legitimately outruns its TTL (a slow triage sweep on a bad network) can still be reclaimed under itself. Start generous, surface every reclaim, tune from the digest.
+- /loop lifetime: sentry dies with its session (harness restart, machine sleep). Acceptable for v1: the morning state is a readable branch plus session-context, and nothing corrupts.
+- Digest fatigue: hourly fires on an active evening could stack noise. The interval knob and the one-line-when-idle rule are the mitigations; revisit after a week of use.
+- The sentry branch holds todo.org/notes.org edits overnight; Craig editing those files on main before the morning merge makes the squash-merge conflict. That conflict is small, self-inflicted, and resolvable by hand; the morning-review section documents it.
+
+* Review and iteration history
+
+** 2026-07-14 Tue @ 02:03:28 -0500 — Claude Code (rulesets) — responder
+- What: all 12 findings walked with Craig one by one and dispositioned; each completed in place and folded into the body. Load-bearing calls: launch contract with in-place checkout (worktrees rejected — untracked inbox drops are invisible there); every roam writer adopts the lock; locks relocate to the XDG runtime dir; spine-set exclusion + fire-end digest commit; inbox pass runs the no-approvals contract; suite policy recorded as Decision 10; KB promotion pass cut to vNext (heuristic task filed); heartbeat + bounded-wait lock mechanics; wrap-up refuses during a live loop ("stop sentry" owns shutdown); entry ff-only reconcile; stall notify after two skips. Mid-walk, Craig redirected the pass-8 finding into fixing the lint-org bug immediately — landed as 951b6fc, which also found and closed the true 2026-07-09 corruption path (wrap-org-table's load-time dispatch firing on lint-org's require).
+- Why: Craig chose to resolve every open question together rather than batch them into a written response round.
+- Artifacts: findings section above (12/12 done); todo.org tasks (KB heuristic [#D], bug task closed); commit 951b6fc.
+
+** 2026-07-14 Tue @ 01:15:00 -0500 — Claude Code (rulesets) — reviewer
+- What: adversarial weakness review at Craig's direction — batches of candidate weaknesses checked against the spec, looped until a round produced fewer than five substantive new ones. Round 1: 20 candidates, 11 answered or mitigated by the spec, 9 became findings. Round 2: 7 candidates, 3 answered, 4 folded into findings. Round 3 produced fewer than five substantive candidates; loop terminated. Result: 10 blocking + 2 non-blocking findings recorded above. Rubric: Not ready until the blocking findings are dispositioned. Status stays DRAFT for Craig's deeper review.
+- Why: Craig asked for the weakness loop before his own deep review, in place of a standard first-pass spec-review.
+- Artifacts: findings grounded in code reads — roam-sync.sh:28 (git add -A commits anything inside the roam repo), .ai/scripts/session-context-path (AI_AGENT_ID scoping), todo.org [#B] "Org-table helpers corrupt example blocks" (lint-org mutate-on-lint), inbox.org core §2 no-approvals park path, verification.md pre-commit suite rule.
+
+** 2026-07-14 Tue @ 00:52:03 -0500 — Craig Jennings — author
+- What: initial draft.
+- Why: work project's sentry proposal accepted after skeptical review; nine design decisions resolved live in-session, recorded under Decisions.
+- Artifacts: [[file:../design/2026-07-14-sentry-workflow-proposal.org][origin proposal]]; todo.org sentry build task.
diff --git a/docs/specs/2026-07-20-silent-until-signal-monitors-spec.org b/docs/specs/2026-07-20-silent-until-signal-monitors-spec.org
new file mode 100644
index 0000000..c15d10f
--- /dev/null
+++ b/docs/specs/2026-07-20-silent-until-signal-monitors-spec.org
@@ -0,0 +1,126 @@
+#+TITLE: Silent-Until-Signal Monitor Loops — Spec
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-20
+#+TODO: TODO | DONE
+#+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED
+
+* IMPLEMENTED silent-until-signal monitor loops
+:PROPERTIES:
+:ID: af592bd6-d3e6-47e2-8804-2a287b4d9303
+:END:
+- 2026-07-20 Mon @ 13:34:53 -0500 — READY (Craig approved after his read) → building Phases 2-5 immediately → IMPLEMENTED. Phase 2 (auto triage-intake heartbeat), Phase 3 (auto inbox-zero heartbeat), Phase 4 (shared-policy home: the spec is the single definition, each workflow states its own heartbeat and references it), Phase 5 (manual-testing checklist). All phases shipped.
+- 2026-07-20 Mon @ 12:30:00 -0500 — Phase 1 (sentry quiet-fire heartbeat) implemented ahead of the full READY gate, at Craig's direction — a quiet sentry fire now collapses to =sentry at HH:MM: nothing=, and the "no silent skip" discipline is reconciled (the heartbeat is the explicit "nothing" record). Phases 2-5 (triage, inbox, shared-policy home, verification) still pending Craig's deep read. Spec stays DRAFT.
+- 2026-07-20 Mon @ 12:14:20 -0500 — drafted during the morning session. Design decisions resolved live with Craig off the .emacs.d proposal and the sentry live-trial evidence. DRAFT pending his deep read.
+
+* Metadata
+| Status | draft |
+|----------+--------------------------------------------------------------|
+| Owner | Craig Jennings |
+|----------+--------------------------------------------------------------|
+| Reviewer | (spec-review, next session) |
+|----------+--------------------------------------------------------------|
+| Related | [[file:../../todo.org::*Silent-until-signal for in-session monitor loops][todo.org — silent-until-signal task]] |
+|----------+--------------------------------------------------------------|
+
+* Summary
+
+An in-session monitor loop (sentry, auto triage-intake, auto inbox-zero) fires the model once per interval. Today each fire narrates a full turn even when nothing changed, so a long session fills with walls of "quiet fire" / "no new items" output. The fix is a policy, not a mechanism: every monitor fire does cheap detection first, and on an empty check collapses to a single labelled heartbeat line ("sentry at 03:20: nothing") and stops. Only a fire that finds a genuinely new item spends a full surface-and-judge turn. Detection stays in-session, so the policy applies uniformly to file-based and MCP-auth loops alike.
+
+* Problem / Context
+
+In-session cron/loop monitors surface a visible model turn on every fire. When nothing changed, that turn is pure noise — a per-pass sentry digest full of SKIP lines, or a triage/inbox "nothing new" report. Over a night or a long session the signal (the one fire that found something) drowns in the empty ticks. The rulesets sentry live trial (2026-07-20) demonstrated it directly: fires 1-2 did real work, fires 3-8 were near-identical walls of no-op lines.
+
+Origin: a Craig-approved proposal from .emacs.d (2026-07-20), captured off an auto inbox-zero session filling with empty-check noise.
+
+** Why not an external watcher (the rejected shape)
+
+The proposal's first instinct was a shell-level watcher (systemd timer or backgrounded loop) doing detection outside the model, producing zero output on an empty check and dropping a handoff into inbox/ only on a real item — so the model runs zero turns when nothing changed. That is truly-zero-idle, but it has a disqualifying cost: it moves detection out of the session, and auto triage-intake's sources (Gmail, Slack, Linear) are reachable only through the session's inherited MCP auth. triage-intake.org is explicit that it runs in the live session precisely because "the headless-auth wall that blocks a detached cron run does not apply." A detached watcher cannot scan those sources at all. The external-watcher shape would therefore split the three targets into two incompatible cases (file-detectable vs MCP-auth) and still leave triage unsolved.
+
+The reframing (Craig, 2026-07-20): the noise is a *policy* problem — when to spend a full model turn — not a missing piece of infrastructure. Keeping detection in-session and making the empty fire cheap solves the actual complaint without the watcher, without per-machine daemon setup, and without breaking MCP auth. It applies to all three loops uniformly.
+
+* Goals and Non-Goals
+
+** Goals
+- An empty monitor fire produces exactly one labelled heartbeat line, not a full narrated turn: =<workflow> at HH:MM: nothing=.
+- A fire that detects a genuinely new item does the full surface-and-judge turn unchanged.
+- One uniform policy across sentry, auto triage-intake, and auto inbox-zero — no per-loop special-casing.
+- Detection stays in-session so MCP-auth loops (triage) get the same treatment as file-based loops.
+- Each loop reuses the seen-state it already keeps; no new watcher-owned seen-list.
+
+** Non-Goals
+- No external watcher, systemd timer, or Monitor-tool daemon. Detection is the loop body's own cheap check.
+- No truly-zero-idle (no model turn at all on empty). That needs an external watcher and is the shape rejected above; if it is ever wanted for the file-based loops only, it is a separate vNext, logged not built.
+- No change to what a loop does when it *does* find something — the surface/judge/act behaviour is untouched.
+- No change to the one-shot (non-loop) invocations of these workflows.
+
+* Design
+
+** The policy
+
+Every in-session monitor fire runs in two steps:
+
+1. *Detect (cheap, silent).* Run the loop's existing detection against its seen-state — sentry's pass probes and branch/digest state, triage's sentinel scan, the inbox monitor's disposition check. This is Bash/tool work, not narration.
+2. *Branch on the result.*
+ - *Nothing new* → emit one line, =<workflow> at HH:MM: nothing=, and end the fire. No digest, no per-pass lines, no report.
+ - *Something new* → the full existing turn: surface, judge, act/queue, and its normal richer output.
+
+The heartbeat is the whole output of an empty fire. Its format is fixed: the workflow's short name, =at=, =HH:MM= (local, from =date=), then =: = and the result word (=nothing= for an empty check). Examples: =sentry at 03:20: nothing=, =triage intake at 03:30: nothing=, =inbox zero at 03:30: nothing=.
+
+** Per-workflow application
+
+- *Sentry.* A fire whose passes all probe-skip or no-op collapses to =sentry at HH:MM: nothing=. No per-pass digest block is written for a quiet fire. A fire that runs or queues anything writes its full digest as today. (This supersedes the trial's behaviour, where fires 3-8 each wrote a full no-op digest.)
+- *Auto triage-intake.* An auto-mode sweep that finds nothing across its enabled sources collapses to =triage intake at HH:MM: nothing=. Detection stays in-session, so the MCP sources are scanned normally; only the output on empty changes.
+- *Auto inbox-zero.* A roam-mode cycle that finds no new inbox items collapses to =inbox zero at HH:MM: nothing=.
+
+** Seen-state (no new artifact)
+
+Because detection stays in-session, each loop keeps using the state it already maintains to know what is "new": triage's =.ai/last-triage-intake= sentinel, sentry's branch + digest, the inbox monitor's per-cycle disposition. The proposal's "watcher-owned seen-list replacing the in-anchor Dispositioned list" is dropped — it was an artifact of the external-watcher shape, which is not being built.
+
+** The accepted consequence
+
+An in-session loop still fires the model once per interval; the harness invokes it each time. So "silent" means the empty fire is *cheap* (one detection pass plus one heartbeat line), not *absent*. This is the deliberate trade for uniformity and for keeping the MCP auth that makes triage possible. The heartbeat also doubles as a liveness pulse: a visible "still running, nothing to do" beats silence that is indistinguishable from a stalled loop.
+
+* Decisions
+
+** DONE Policy, not mechanism — detection stays in-session
+CLOSED: [2026-07-20 Mon]
+Craig reframed the proposal: silent-until-signal is a policy about when to spend a full model turn, not a new watcher. Keeping detection in-session dissolves the MCP-auth split (triage's sources need session auth) and needs no per-machine daemon.
+
+** DONE Empty fire → one labelled heartbeat line (not fully silent)
+CLOSED: [2026-07-20 Mon]
+Format =<workflow> at HH:MM: nothing=. A visible pulse is worth one line so a running loop is distinguishable from a stalled one; full silence was the rejected alternative.
+
+** DONE Applies to sentry, auto triage-intake, and auto inbox-zero uniformly
+CLOSED: [2026-07-20 Mon]
+The two MCP-auth loops are first-class targets, not just sentry, precisely because detection stays in-session.
+
+** DONE No external watcher; no truly-zero-idle in v1
+CLOSED: [2026-07-20 Mon]
+The systemd-timer / Monitor-tool / backgrounded-shell shapes are out. Truly-zero-idle (no turn on empty) is the only thing they'd buy, it only works for file-based loops, and it breaks triage. Logged as a possible file-only vNext, not built.
+
+** DONE Reuse each loop's existing seen-state
+CLOSED: [2026-07-20 Mon]
+No new watcher-owned seen-list; the sentinel / branch-digest / disposition each loop already keeps is the detection state.
+
+* Implementation Phases
+
+** Phase 1 — sentry quiet-fire heartbeat — DONE 2026-07-20
+Edit =.ai/workflows/sentry.org= (canonical + mirror): a fire whose passes all probe-skip or no-op writes =sentry at HH:MM: nothing= instead of a full per-pass digest; a fire that runs or queues anything writes the full digest unchanged. Update the digest section and the Common Mistakes "silent skip" note to draw the quiet-fire-vs-working-fire line. Run sync-check. (Separable and the highest-value piece — do first.)
+
+Shipped 2026-07-20: five edits to sentry.org — Pass Runner step 3 (per-pass lines collapse on a quiet fire), Fire-end step 2 (the heartbeat-vs-digest decision + heartbeat commit variant), the digest section (working-block vs quiet-heartbeat), Unattended-safety (quiet fire is not a silent skip), Common Mistakes #7 (the carve-out). Canonical + mirror synced, lint clean.
+
+** Phase 2 — auto triage-intake heartbeat
+Edit =.ai/workflows/triage-intake.org= Auto mode: an empty sweep collapses to =triage intake at HH:MM: nothing=. Preserve in-session detection and the accumulate-don't-mutate contract. Run sync-check.
+
+** Phase 3 — auto inbox-zero heartbeat
+Edit =.ai/workflows/inbox.org= Auto inbox zero mode: an empty cycle collapses to =inbox zero at HH:MM: nothing=. Run sync-check.
+
+** Phase 4 — state the shared policy once
+Factor the policy statement into one place both loops and sentry point at (a short section in =inbox.org= monitor-mode core, or a =claude-rules/= note if it reads as cross-cutting), so the three workflows reference one definition rather than restating it. Decide the home during build.
+
+** Phase 5 — verification
+Workflow prose, no bats surface. Verify by exercising: a sentry quiet fire prints one heartbeat line and writes no digest; a working fire still writes its full digest and commits. For the two MCP loops, confirm an empty sweep prints the heartbeat and a sweep with a planted item still does the full turn. Add a manual-testing checklist entry per =verification.md= where a live check is the only proof.
+
+* Prototype / UI
+
+Not applicable — no UI surface; the deliverable is the one-line heartbeat format.
diff --git a/docs/specs/2026-07-20-triage-source-activation-spec.org b/docs/specs/2026-07-20-triage-source-activation-spec.org
new file mode 100644
index 0000000..dce1ec1
--- /dev/null
+++ b/docs/specs/2026-07-20-triage-source-activation-spec.org
@@ -0,0 +1,127 @@
+#+TITLE: Triage Source Activation — Spec
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-20
+#+TODO: TODO | DONE
+#+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED
+
+* IMPLEMENTED triage source activation
+:PROPERTIES:
+:ID: af73ef0b-cd1d-46f1-9e1d-62695733a4de
+:END:
+- 2026-07-20 Mon @ 13:42:32 -0500 — READY (Craig approved after his read, both open decisions resolved via cj comments: declaration format "good. approved", interactive gate "all of them") → built and IMPLEMENTED same session. Phase 0 activation gate in triage-intake.org, sentry pass-3 probe, engine-intro documentation (the template notes.org has no Workflow State block, so the declaration is documented in triage-intake.org rather than there — deviation from the drafted Phase 3), migration handoffs to home + work, a manual-testing entry. Canonical + mirror synced.
+- 2026-07-20 Mon @ 08:43:23 -0500 — drafted during the morning sentry review. The activation model converged live with Craig off the sentry live-trial finding. DRAFT pending his deep read.
+
+* Metadata
+| Status | draft |
+|----------+--------------------------------------------------------------|
+| Owner | Craig Jennings |
+|----------+--------------------------------------------------------------|
+| Reviewer | (spec-review, next session) |
+|----------+--------------------------------------------------------------|
+| Related | [[file:../../todo.org::*Triage source activation][todo.org — Triage source activation task]] |
+|----------+--------------------------------------------------------------|
+
+* Summary
+
+triage-intake should pull only the sources a project has chosen to pull. Today it runs every plugin it can discover, and the general (personal-account) plugins are template-synced into every project, so every project looks like it wants to triage Craig's personal Gmail, cmail, calendar, and Telegram. The fix is one activation layer: a general plugin runs only when the project names it in a =:TRIAGE_SOURCES:= declaration; a project-specific plugin stays active by its presence, which is already a deliberate per-project act. Sentry's pass-3 probe then reads the same signal.
+
+* Problem / Context
+
+The 8-fire sentry live trial (rulesets, 2026-07-20) surfaced this. Sentry's pass 3 probe is "triage source plugins present for this project," and triage-intake discovers sources by globbing =.ai/workflows/triage-intake.*.org= (general, template-synced) plus =.ai/project-workflows/triage-intake.*.org= (project-specific, never synced). Because the general plugins sync into every project, "plugins present" is true everywhere, so the triage pass self-activates in every project — including rulesets, which is not a triage target.
+
+The harm is two-layered, and sentry's existing safety rule only catches one layer. Sentry's pass-3 line says "destructive actions queue; they never fire unattended," which would hold back the trash/mark-read/star hygiene. But triage-intake also reads the accounts and files each Action item to the local =todo.org= as a =:quick:reactive:= task. Reading and filing are not destructive, so nothing queues them. Under sentry, triage would still authenticate to Craig's personal inboxes overnight and file his personal action items into whatever project the fire runs in. Wrong scope, and it touched real accounts to get there.
+
+The same over-pull exists interactively: running triage-intake by hand in rulesets today would pull personal Gmail too. The trial only made the unattended case visible.
+
+Root cause: the model conflates *presence* with *activation*. A plugin has two existing gates — it must be globbed (presence) and pass its =ENABLED= precondition (capability: "is the gmail MCP reachable?"). Neither answers the question the trial exposed: *should this project pull this source?* For the general plugins, presence comes from sync and capability is true on Craig's machine everywhere, so both gates pass in every project.
+
+Asymmetry that shapes the fix: project-specific plugins do not leak. They live in =.ai/project-workflows/=, are never synced, and exist only where someone deliberately dropped one. A project that polls an RSS feed puts an =triage-intake.rss-*.org= plugin there, and it runs in that project and nowhere else. Only the general synced plugins leak. So the missing activation layer only needs to gate the general plugins.
+
+* Goals and Non-Goals
+
+** Goals
+- Per-project source selection: a project pulls exactly the sources it declares, nothing more.
+- Off-limits by default: a general source that a project has not named is never pulled there.
+- Project-specific sources keep working by presence — dropping the plugin is the declaration (Craig's RSS-feed case works unchanged).
+- One fix covers both paths: the activation layer lives in triage-intake, so interactive and unattended (sentry) runs both respect it.
+- Sentry's pass-3 probe reads the same activation signal rather than mere plugin presence.
+
+** Non-Goals
+- No change to the per-plugin =ENABLED= capability check — it stays the "can I reach this source" gate.
+- No change to the four-bucket classification, the digest shape, or the close behavior.
+- No auto-migration across projects. Projects that pull general sources today declare them via handoff; the change never edits another project's config unattended.
+- No new plugin discovery mechanism — the two-directory glob stays.
+
+* Design
+
+** Two plugin classes, one new activation layer
+
+- *Project-specific plugins* (=.ai/project-workflows/triage-intake.*.org=): active by presence. The plugin exists only because the project author put it there, which is itself the per-project declaration. No further gate. This is where a project's own sources live — an RSS feed, a work Linear, a work Slack.
+- *General plugins* (=.ai/workflows/triage-intake.*.org=, template-synced — personal Gmail, cmail, calendar, Telegram, GitHub PRs): active only when the project names the source in its =:TRIAGE_SOURCES:= declaration. Present-but-undeclared means available-not-active: the plugin is on disk, but this project does not pull it.
+
+** The declaration
+
+A line in the project's =.ai/notes.org= Workflow State block, alongside =:COMMIT_AUTONOMY:= and =:LAST_AUDIT:=:
+
+: :TRIAGE_SOURCES: personal-gmail cmail
+
+Space-separated source names matching general-plugin basenames (=personal-gmail=, =cmail=, =personal-calendar=, =telegram=, =github-prs=). Absent or empty means no general sources are active for this project. Project-specific plugins are unaffected by this line — they run regardless, because presence is their declaration.
+
+** triage-intake Phase 0 change
+
+Phase 0 keeps globbing both directories. The loaded-set computation changes: for each *general* plugin, additionally require its basename to appear in =:TRIAGE_SOURCES:=; skip it with an announced reason otherwise ("skipping personal-gmail — not in :TRIAGE_SOURCES:"). Project-specific plugins skip this check. The =ENABLED= capability check still runs on the survivors. The announce-loaded-set block already exists and gains an "inactive (undeclared)" line so the omission stays visible rather than silent — the same anti-silence discipline Phase 0 already enforces.
+
+** Sentry pass-3 probe change
+
+The probe changes from "triage source plugins present" to "the project has at least one active triage source" — any project-specific plugin present, or a non-empty =:TRIAGE_SOURCES:= intersecting the general plugins on disk. rulesets, declaring nothing and owning no project-specific plugin, probe-skips cleanly.
+
+** Migration
+
+Projects that pull general sources today (home, and possibly work) each add a =:TRIAGE_SOURCES:= line, or their triage goes quiet. This is a per-project handoff, not an automated sweep — the change can't safely guess each project's intended source set. Projects that only ever ran project-specific plugins need no migration.
+
+* Decisions
+
+** DONE Activation gates general plugins only; project-specific stay active-by-presence
+CLOSED: [2026-07-20 Mon]
+Converged live with Craig. Project-specific plugins are already per-project (never synced), so they need no gate; only the general synced plugins leak, so only they need a declaration.
+
+** DONE The activation layer lives in triage-intake Phase 0, not only the sentry probe
+CLOSED: [2026-07-20 Mon]
+Placing it in the engine fixes the interactive over-pull too. Sentry inherits the signal rather than reimplementing it.
+
+** DONE Off-limits by default
+CLOSED: [2026-07-20 Mon]
+An undeclared general source is never pulled. Opt-in is explicit; there is no "pull everything discovered" default.
+
+** DONE Migration is per-project handoffs, never an unattended edit of another project's config
+CLOSED: [2026-07-20 Mon]
+The change can't guess a project's intended source set, and cross-project auto-edits violate the boundary rule.
+
+** DONE Declaration format — =:TRIAGE_SOURCES:= space-separated basenames in notes.org Workflow State
+CLOSED: [2026-07-20 Mon]
+Approved by Craig (2026-07-20). =:TRIAGE_SOURCES: personal-gmail cmail= in notes.org Workflow State, mirroring =:COMMIT_AUTONOMY:=. An empty-but-present marker and an absent one are treated identically — both mean "no general sources active" — so there's no need to distinguish them. A project-specific-only project carries no marker; its absence is correct, since those plugins activate by presence.
+
+** DONE Interactive triage adopts the same gate as unattended
+CLOSED: [2026-07-20 Mon]
+Craig: "all of them" (2026-07-20). The activation gate lives in triage-intake Phase 0 and applies to every path — interactive and unattended alike — so running triage by hand in a project also respects its =:TRIAGE_SOURCES:= declaration. One activation layer, no per-path special-casing.
+
+* Implementation Phases
+
+** Phase 1 — triage-intake Phase 0 activation rule
+Edit =.ai/workflows/triage-intake.org= (canonical =claude-templates/.ai/workflows/= + mirror): add the general-vs-project-specific activation rule to Phase 0, the =:TRIAGE_SOURCES:= read, and the "inactive (undeclared)" announce line. Run sync-check.
+
+** Phase 2 — sentry pass-3 probe
+Edit =.ai/workflows/sentry.org= (canonical + mirror): change the pass-3 probe to "any active triage source present" and note the activation source. Run sync-check.
+
+** Phase 3 — document the declaration
+Document =:TRIAGE_SOURCES:= in the notes.org Workflow State reference (the template =claude-templates/.ai/notes.org= Workflow State block) so new projects see it. Cross-reference from triage-intake.
+
+** Phase 4 — migration handoffs
+=inbox-send= home (and work, if it pulls general sources) the =:TRIAGE_SOURCES:= line each should add, with the reason. No auto-edit.
+
+** Phase 5 — verification
+triage-intake is a workflow (prose), not a script, so there's no bats surface for the activation rule directly. Verify by a scripted check that a project with no declaration loads zero general plugins (a small fixture around the Phase-0 glob-and-filter logic if it's extracted, or a manual-testing checklist entry otherwise). Confirm rulesets probe-skips triage under sentry, and home still pulls its declared sources.
+
+* Prototype / UI
+
+Not applicable — no UI surface.
diff --git a/docs/agent-knowledge-base-spec.org b/docs/specs/agent-knowledge-base-spec.org
index 78ff9bd..d5c0ce7 100644
--- a/docs/agent-knowledge-base-spec.org
+++ b/docs/specs/agent-knowledge-base-spec.org
@@ -1,12 +1,20 @@
#+TITLE: Agent Knowledge Base on Org-roam — Spec
-#+AUTHOR: Craig Jennings & Claude
+#+AUTHOR: Craig Jennings
#+DATE: 2026-06-10
+#+TODO: TODO | DONE
+#+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED
+
+* IMPLEMENTED Agent Knowledge Base on Org-roam — Spec
+:PROPERTIES:
+:ID: 08a5ec99-9e1e-40e4-8241-e8a41e9de49f
+:END:
+- 2026-07-02 Thu @ 00:17:01 -0400 — retrofitted by spec-sort; status set to IMPLEMENTED (reason: v1 (Phases 0-4) shipped 2026-06-10 on Craig's go; KB live at ~/org/roam with the knowledge-base rule installed machine-wide)
* Metadata
-| Status | implemented — v1 (Phases 0-4) shipped 2026-06-10 on Craig's go; manual validation + other-machine clones outstanding (todo.org) |
+| Status | implemented |
| Owner | Craig Jennings |
| Reviewer | Craig Jennings; Codex (2026-06-10) |
-| Related | [[file:../todo.org][todo.org — "Check that memories are sync'd across machines via git"]] |
+| Related | [[file:../../todo.org][todo.org — "Check that memories are sync'd across machines via git"]] |
This spec supersedes the 2026-06-05 draft (formerly docs/design/2026-06-05-org-roam-knowledge-base-spec.org, removed; content in git history), folding in Craig's 2026-06-10 ratification answers and restructuring to the spec-create format.
@@ -308,4 +316,4 @@ Modified recommendations from the 2026-06-10 Codex review, with reasons. Everyth
** 2026-06-10 Wed @ 17:31:10 -0500 — Codex — reviewer
- What changed or was recommended: re-ran the spec-review workflow after the caveat resolution. Rubric: ready. No new blocking or medium-priority findings; no review file written. Confirmed the implementation phases and test-surface tasks are already represented under the existing parent task in todo.org.
- Why: the prior blockers are dispositioned, the work-root denylist is confirmed, the pointer-rule install path matches the current Makefile RULES glob, and v1's manual/agent-runnable verification surface is explicit.
-- Artifacts: this file; [[file:../todo.org][todo.org]] parent task "Check that memories are sync'd across machines via git".
+- Artifacts: this file; [[file:../../todo.org][todo.org]] parent task "Check that memories are sync'd across machines via git".
diff --git a/docs/specs/inbox-workflow-consolidation-spec.org b/docs/specs/inbox-workflow-consolidation-spec.org
new file mode 100644
index 0000000..24ce30a
--- /dev/null
+++ b/docs/specs/inbox-workflow-consolidation-spec.org
@@ -0,0 +1,199 @@
+#+TITLE: Inbox Workflow Consolidation — Spec
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-06-23
+#+TODO: TODO | DONE
+#+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED
+
+* READY Inbox Workflow Consolidation — Spec
+:PROPERTIES:
+:ID: a7fe2a10-dfa8-4ba3-a11a-e7b1288b7573
+:END:
+- 2026-07-02 Thu @ 00:17:01 -0400 — retrofitted by spec-sort; status set to READY (evidence-based, human-confirmed)
+
+* Metadata
+| Status | ready |
+|----------+-------------------------------------------------------------|
+| Owner | Craig |
+|----------+-------------------------------------------------------------|
+| Reviewer | Craig |
+|----------+-------------------------------------------------------------|
+| Related | [[file:../../todo.org][Consolidate inbox/triage workflows + scheduled inbox check]] |
+|----------+-------------------------------------------------------------|
+
+* Summary
+
+Four inbox-named workflows (=inbox-zero=, =process-inbox=, =monitor-inbox=, plus the startup/wrap-up nudges) circle the same disposition logic across three different surfaces. This spec consolidates them into one =inbox= engine with explicit modes, keeps =triage-intake= (external accounts) and =no-approvals= (session mode) separate, and adds an interactive recurring roam check (=auto inbox zero=). The fully-unattended cron pass is named but deferred to vNext.
+
+* Problem / Context
+
+"Too many inbox related workflows" (Craig, roam capture 2026-06-23). The word "inbox" is overloaded onto three genuinely different surfaces, and the workflows that serve them have grown to circle the same logic:
+
+- *Project-local =inbox/= dir* (handoffs from other projects/scripts/Craig) → =process-inbox.org= owns the value gate and disposition; =monitor-inbox.org= is a thin cadence layer on top of it ("loop process-inbox every 15 min + act-vs-file + reply discipline").
+- *Global roam inbox* (=~/org/roam/inbox.org=, GTD capture) → =inbox-zero.org=, whose Phase A already *calls* =process-inbox= for the local dir before doing the roam-routing part.
+- *External accounts* (email / calendar / PRs) → =triage-intake.org= + six source plugins.
+
+So a reader (or a non-Claude agent) facing "deal with my inbox" has to know which of four files to invoke, and the shared concepts — the three-question value gate, the skeptical review, the implement/fold/file/defer/reject disposition, the reply-to-sender discipline, the capture-guard before a roam write, the priority-scheme check before filing — are spread across and cross-referenced between them. The duplication is real (=monitor-inbox= and =inbox-zero= both lean on =process-inbox='s machinery) and the count is the symptom Craig named.
+
+A second gap surfaced in the same capture: there's no documented way to run a *recurring* inbox check, and Craig wants a keyword trigger for it. v1 answers this with an interactive in-session loop (=auto inbox zero=); a fully-unattended cron pass that fires while Craig is away is a larger contract (mutation safety, surfacing-when-away, cross-run state) and is deferred.
+
+* Goals and Non-Goals
+
+** Goals
+- One engine is the single entry point for the inbox surfaces, with the shared value-gate / disposition / reply / capture-guard / priority-scheme logic living in exactly one place.
+- Mode selection is unambiguous from the trigger phrase and the caller (startup, wrap-up, on-demand).
+- Every existing trigger phrase still works, routing to the right mode — no relearning.
+- A documented interactive recurring check (=auto inbox zero=, =/loop=-based). The fully-unattended cron pass (=/schedule=) is vNext, not v1.
+- INDEX.org, protocols.org, and the startup/wrap-up callers reconciled to the new shape with no dangling references.
+- No behavior regression: the value gate, disposition rules, capture-guard, and reply discipline behave exactly as today.
+
+** Non-Goals
+- *Not* merging =triage-intake.org=. External-account triage ("what's new across my email/cal/PRs") is a different domain from "my inbox dirs"; keeping it distinct is correct, not redundancy.
+- *Not* merging =no-approvals.org=. It's a session mode, not an inbox workflow (it's referenced by the monitor cadence, not part of it).
+- *Not* changing value-gate semantics or disposition rules. This is a structural merge, behavior-preserving.
+- *Not* the domain-aware whole-roam-inbox routing (still deferred, unchanged).
+- *Not* the agent-neutral language sweep over these files — that is the parked half of the agent-source task and runs *after* this merge, over fewer files.
+- *Not* renaming =CLAUDE.md=, =.claude/=, or other structural paths.
+
+** Scope tiers
+- v1: merge =process-inbox= + =monitor-inbox= + =inbox-zero= into one =inbox.org= engine with =process= / =monitor= / =roam= modes; preserve all trigger phrases; reconcile INDEX + protocols + startup + wrap-up; add the interactive =auto inbox zero= recurring check (=/loop=).
+- Out of scope: =triage-intake= merge, =no-approvals= merge, domain-aware roam routing, the agent-neutrality sweep.
+- vNext (log to todo.org): the fully-unattended =/schedule= cron pass — needs its own contract (read-only vs may-mutate =todo.org= / =~/org/roam/inbox.org=, how a find surfaces when Craig is away, how dedup state survives across runs, auth/session constraints); a later umbrella unifying =triage-intake='s "what's new" with the inbox engine; the agent-neutrality pass over the consolidated =inbox.org=.
+
+* Design
+
+The consolidation produces one engine file, =inbox.org=, structured as a shared core plus three thin modes. The core holds every concept that today is duplicated or cross-referenced: the three-question value gate, the skeptical review (with the cross-project battery for shared-asset proposals), the disposition ladder (implement-now / fold / file / defer / reject-by-source / park), the reply-to-sender discipline, the capture-guard before any roam-inbox disk write, and the priority-scheme check before filing. A mode is a short front section that says which surface it reads, how it enters and exits, and which core steps it runs.
+
+*Two altitudes.*
+
+For the *user*: the trigger phrase picks the mode, and the phrases are unchanged. "process inbox" / "handle the inbox" → process mode (the local =inbox/= dir). "monitor the inbox" / "watch the inbox" → monitor mode (process mode on a loop, with the act-vs-file and reply discipline and the clean-tree/green-suite gates). "inbox zero" / "process the roam inbox" → roam mode (route the global roam inbox by =<project>:= prefix, sweep empties, capture-guard the write). Startup calls process mode for the local dir and the read-only roam nudge; wrap-up calls process mode then the roam sweep.
+
+For the *implementer*: =inbox.org= is one file. The core sections are written once. Each mode is a section that references core steps by name rather than restating them ("run the value gate (core §X) on each item", "guard and reconcile the roam write (core §Y)"). The old three files are deleted; their content is absorbed, not copied. The =triage-intake= engine and its plugins are untouched and keep their own namespace.
+
+*Routing and callers.* protocols.org's terminology section and the startup workflow's INDEX-driven routing both key off trigger phrases, so the phrase→mode map is the contract. Each caller that today names =process-inbox.org= / =monitor-inbox.org= / =inbox-zero.org= (startup Phase C, wrap-up Step 3, protocols, INDEX) is repointed at =inbox.org= and the relevant mode. INDEX gets one entry for =inbox.org= listing every trigger phrase, grouped by mode.
+
+*Auto inbox zero (the scheduled mode).* The trigger phrase =auto inbox zero= starts a recurring roam-mode pass. On invocation the engine *asks Craig for the interval* (e.g. 30 min, 2 hours), then drives the loop with =/loop <interval>= running roam mode. It's in-session and interactive by design — each cycle reports, and a find waits for Craig's go before any work happens. Per cycle:
+
+- *Nothing found* → no inbox summary. A single acknowledgement line: ran at =HH:MM=, nothing found. Nothing else.
+- *Items found* → summarize the found items, file them as tasks, and *append them to a displayed queue* (the harness task list, =TaskCreate=) so the queue accumulates across cycles. Then ask: "run this batch next?" If Craig says yes, the engine launches into implementing the found items (each through the normal disposition + verify flow); if no, they stay queued for a later go. Subsequent cycles add only newly-found items to the same displayed queue, never re-surfacing what's already there.
+
+The acknowledge-only-on-empty rule keeps a quiet inbox quiet — no noise when there's nothing to do — while a find is always surfaced and gated on Craig's yes. =auto inbox zero= is the interactive =/loop= shape because its execute step waits for a yes, so it is inherently in-session.
+
+A fully-unattended =/schedule= cron pass (firing while Craig is away) is a different contract and is *vNext, not v1*: it can't wait for a yes, so it has to decide up front whether it may mutate =todo.org= and the roam inbox or stays read-only, how a find reaches Craig asynchronously, how dedup state persists between runs that don't share a session, and what session/auth context a cron run carries. v1 ships only the interactive loop; the unattended contract is logged to =todo.org= for its own design pass.
+
+* Alternatives Considered
+
+** Option A — One engine with modes (chosen)
+- Good, because it cuts four inbox-named files to one and puts the shared logic in a single authoritative place, which is exactly the "too many" complaint.
+- Good, because every trigger phrase can re-home to a mode with no user relearning.
+- Bad, because =inbox.org= becomes a larger file with internal mode branching.
+- Neutral, because =triage-intake= and =no-approvals= stay separate either way.
+
+** Option B — Keep three files, extract a shared include
+- Good, because the diffs are smaller and the per-surface entry points stay familiar.
+- Bad, because it does not reduce the file count — Craig's actual complaint is the number of files, and this keeps three plus adds an include.
+- Neutral, because the dedup of logic happens, just without the count reduction.
+
+** Option C — Merge only process-inbox + monitor-inbox, leave inbox-zero
+- Good, because it fixes the tightest, least-ambiguous redundancy (monitor is literally a loop over process) at the lowest risk.
+- Bad, because roam vs local stays two files; the consolidation is partial (4→3, not 4→2).
+- Neutral, because it could be a first phase of Option A rather than a competing end state.
+
+** Option D — Do nothing, just document which file is which
+- Good, because zero risk to load-bearing synced workflows.
+- Bad, because it doesn't reduce the count at all; the complaint stands.
+
+* Decisions [4/4]
+
+** DONE Engine shape — one file with modes vs partial merge
+- Context: Option A (one =inbox.org=, 4→1) maximally addresses "too many" but is the biggest single change to load-bearing synced files. Option C (merge the process/monitor pair only, 4→3) is lower-risk and could be A's first phase.
+- Decision: We will build Option A — one =inbox.org= engine with =process= / =monitor= / =roam= modes. (Craig, 2026-06-23.)
+- Consequences: easier discovery and one home for the logic; harder single-file size and a bigger, higher-blast-radius diff, mitigated by the shared-core + thin-mode structure and the plugin-namespace escape hatch for a mode that wants depth.
+
+** DONE Trigger-phrase routing — preserve all existing phrases
+- Context: protocols + startup route by phrase; users have these in muscle memory.
+- Decision: We will keep every existing trigger phrase, re-homing each to its mode on the one engine, adding only the new =auto inbox zero= phrase. (Craig, 2026-06-23 — accepted as recommended.)
+- Consequences: easier — no relearning, no broken muscle memory; harder — the engine must document a longer phrase→mode table and guard against collisions.
+
+** DONE triage-intake stays separate
+- Context: external-account triage is a different surface; folding it in would re-bloat the engine.
+- Decision: We will leave =triage-intake.org= and its plugins untouched, out of this consolidation. (Craig, 2026-06-23 — accepted as recommended.)
+- Consequences: easier — smaller, coherent inbox engine; harder — two "what's arriving" entry points remain (inbox engine vs triage-intake), documented so the boundary is clear.
+
+** DONE Scheduled-check mechanism + behavior + keyword
+- Context: Craig wants a recurring inbox check with a keyword, an interactive find-then-execute flow, and a running queue.
+- Decision: The trigger phrase is =auto inbox zero=. On invocation it asks Craig for the interval, then runs roam-mode on =/loop <interval>=. Empty cycle → one acknowledgement line (ran at HH:MM, nothing found), no inbox summary. Find → summarize, file as tasks, append to the displayed task queue (=TaskCreate=), and ask "run this batch next?"; on yes, implement the found items; subsequent cycles append only new finds to the same queue. =/schedule= stays available for a fully-unattended pass. (Craig, 2026-06-23.)
+- Consequences: easier — a quiet inbox stays quiet, a find is always gated on a yes, and the queue is one accumulating view; harder — the loop must dedup against already-queued items so it doesn't re-surface them, and the in-session =/loop= shape means the unattended case still needs =/schedule=.
+
+* Review findings [2/2]
+
+** DONE Fully unattended scheduled behavior is not specified :blocking:
+Disposition: accepted via the narrow option. v1 ships only the interactive =auto inbox zero= (=/loop=); the fully-unattended =/schedule= pass is deferred to vNext with its open contract questions named (read-only vs may-mutate, surface-when-away, cross-run dedup state, auth/session). Folded into Summary, Goals, Problem/Context, Scope tiers (vNext), and the Design "Auto inbox zero" subsection; the vNext contract is logged to todo.org. This sequences the unattended pass rather than dropping it, preserving Decision 4's intent.
+The Summary and Goals promise a scheduled unattended inbox check with trigger keywords, but the concrete =auto inbox zero= design is intentionally interactive: it asks for an interval, runs =/loop <interval>= in the live session, and waits for Craig before executing found work. The only unattended behavior is the sentence that =/schedule= remains available for a fully-unattended cloud-cron pass. That leaves an implementer to invent the actual scheduled contract: trigger phrase(s), whether the pass is read-only or may mutate =todo.org= / =~/org/roam/inbox.org=, how findings are surfaced when Craig is away, how dedup state survives across runs, and what auth/session constraints apply. Add a distinct =/schedule= subsection and acceptance criteria for the fully unattended mode, or narrow the Summary/Goals to say v1 ships only the interactive =/loop= mode and log the unattended cron shape as vNext. (blocking)
+
+** DONE Stale-reference verification relies on a checker that does not check workflow links
+Phase 2 and the Risks section rely on the workflow-integrity / INDEX-drift check as the backstop for missed references to =process-inbox.org=, =monitor-inbox.org=, and =inbox-zero.org=. Current =scripts/workflow-integrity.py= checks INDEX coverage, script references, plugin parentage, orientation sections, and duplicate trigger phrases; it does not validate arbitrary =[[file:...org]]= workflow links or prose references in workflows/protocols/rules. That means a deleted-workflow link in =startup.org=, =wrap-it-up.org=, or =protocols.org= can survive the named checker. Keep the grep requirement, but make it an explicit acceptance item with the exact scope: at minimum =rg 'process-inbox|monitor-inbox|inbox-zero' claude-templates/.ai .ai claude-rules= after caller rewrites, allowing only intentional historical/spec/todo mentions. Optionally extend =workflow-integrity.py= to validate local workflow links, but do not imply it already catches this class. (non-blocking)
+
+Disposition: accepted. Added the exact grep as an acceptance item and a Phase 2 step, reworded the Risks "missed caller reference" dodge and the Dev-tooling readiness line so the integrity checker is no longer implied to validate workflow links, and noted the optional =workflow-integrity.py= extension as not-required-for-v1.
+
+* Implementation phases
+
+** Phase 1 — Author the inbox engine
+Write =inbox.org= (canonical =claude-templates/.ai/workflows/=): the shared core (value gate, skeptical review, disposition ladder, reply discipline, capture-guard, priority-scheme check) plus the three mode sections, absorbing the content of the three source files. No caller changes and no deletions yet — the tree still works with the old files in place, the new engine sits alongside for review.
+
+** Phase 2 — Reconcile callers and retire the old files
+Repoint INDEX.org (one =inbox.org= entry, phrases grouped by mode), protocols.org terminology, startup.org Phase C, and wrap-it-up.org Step 3 at =inbox.org= + mode. Delete =process-inbox.org=, =monitor-inbox.org=, =inbox-zero.org=. Grep for stale references — =rg 'process-inbox|monitor-inbox|inbox-zero' claude-templates/.ai .ai claude-rules= — and clear every live caller (the integrity checker covers INDEX coverage and trigger duplication, not workflow links). Run the workflow-integrity / INDEX-drift check. Sync the mirror.
+
+** Phase 3 — Auto inbox zero + scheduled check
+Add the =auto inbox zero= mode to =inbox.org=: ask-for-interval, =/loop <interval>= over roam mode, the empty-cycle acknowledgement, and the find → summarize → file → queue → ask-to-execute flow with cross-cycle dedup against the displayed queue. Document the =/schedule= recipe for the fully-unattended pass alongside it.
+
+** Phase 4 — Verify
+Trigger-phrase coverage (every old phrase resolves to a mode), startup + wrap-up dry-run against the new engine, capture-guard still gates the roam write, INDEX drift clean, mirror in sync.
+
+* Acceptance criteria
+- [ ] Every trigger phrase that today routes to =process-inbox= / =monitor-inbox= / =inbox-zero= resolves to a mode of =inbox.org=.
+- [ ] The three old workflow files are deleted and absent from INDEX; the integrity check reports no orphan or stale entry.
+- [ ] After caller rewrites, =rg 'process-inbox|monitor-inbox|inbox-zero' claude-templates/.ai .ai claude-rules= returns only intentional historical / spec / todo mentions — no live caller reference to a deleted file. (The integrity checker validates INDEX coverage, not arbitrary workflow links, so this grep is the real backstop.)
+- [ ] Startup still processes the local inbox and produces the read-only roam nudge; wrap-up still sweeps the project's roam items.
+- [ ] The capture-guard runs before any roam-inbox disk write in the consolidated engine.
+- [ ] The value gate, disposition ladder, and reply-to-sender discipline are present once and unchanged in behavior.
+- [ ] =auto inbox zero= asks for an interval, then runs roam mode on =/loop <interval>=.
+- [ ] An empty auto cycle emits only a timestamped acknowledgement (ran at HH:MM, nothing found) — no inbox summary.
+- [ ] A find summarizes the items, files them as tasks, appends them to the displayed queue, and asks before executing; "yes" runs the batch; later cycles append only newly-found items, never re-surfacing queued ones.
+- [ ] Canonical and mirror copies are in sync (=sync-check.sh=).
+
+* Readiness dimensions
+- Data model & ownership: the engine reads two files it doesn't own (project =inbox/= dir contents, =~/org/roam/inbox.org=) and writes =todo.org= + the roam file. Ownership unchanged from today; the merge moves no data.
+- Errors, empty states & failure: empty inbox → report and stop (preserved per surface); roam pull blocked or dirty → surface and stop, never auto-stash (preserved); live org-capture on the roam file → capture-guard blocks the write (preserved).
+- Security & privacy: N/A because no credentials or sensitive data; the engine moves task text between local files.
+- Observability: the user sees which mode ran and its disposition summary; INDEX drift check surfaces a mis-wired routing.
+- Performance & scale: N/A because the inputs are small text files triaged by hand-scale counts.
+- Reuse & lost opportunities: the whole point — the shared core is written once instead of three times; =triage-intake='s plugin pattern is intentionally not reused here (different surface).
+- Architecture fit & weak points: integration points are INDEX.org, protocols.org terminology, startup Phase C, wrap-up Step 3. Weak point: a missed caller reference to an old filename breaks routing — mitigated by the explicit stale-reference grep (acceptance item), since the integrity check covers INDEX coverage, not workflow links.
+- Config surface: trigger phrases (the phrase→mode table) and the scheduled-check keyword set + cron expression.
+- Documentation plan: the engine file is the doc; INDEX entry updated; protocols terminology updated. No separate user doc needed.
+- Dev tooling: the workflow-integrity check + startup INDEX-drift check cover INDEX coverage and trigger-phrase duplication; they do *not* validate workflow file links, so the stale-reference grep (acceptance item) is a manual step. Optionally extend =workflow-integrity.py= to validate local =[[file:...org]]= workflow links — not required for v1.
+- Rollout, compatibility & rollback: the merge lands via the template sync; =rsync --delete= removes the three retired files from every consuming project on its next startup, and the new =inbox.org= arrives the same pass. Rollback = git revert of the rulesets commit, then the next sync restores the old files. Trigger-phrase preservation is the compatibility guarantee.
+- External APIs & deps: N/A because no external API; =/schedule= and =/loop= are harness features, not deps.
+
+* Risks, Rabbit Holes, and Drawbacks
+- *Missed caller reference.* A lingering mention of =process-inbox.org= / =monitor-inbox.org= / =inbox-zero.org= in a workflow, protocol, skill, or the INDEX would break routing after the files are deleted. Dodge: =rg 'process-inbox|monitor-inbox|inbox-zero' claude-templates/.ai .ai claude-rules= and clear every live caller before deleting. The workflow-integrity checker validates INDEX coverage and trigger-phrase duplication, *not* arbitrary =[[file:...]]= links, so the grep — not the checker — is the real backstop here.
+- *Single-file sprawl.* One engine with three modes risks becoming the wall-of-text the workflows were split to avoid. Dodge: the shared-core + thin-mode structure and the terseness pass; if a mode wants real depth, it can become a =inbox.<mode>.org= plugin under the engine namespace (the same pattern =triage-intake= uses) rather than bloating the core.
+- *Sequencing with the agent-neutrality sweep.* If the neutrality sweep runs first, it edits three files about to be deleted. Dodge: this consolidation lands first by construction (it's why the sweep was parked).
+
+* Review and iteration history
+** 2026-06-23 Tue @ 21:51:51 -0400 — Claude — author
+- What: initial draft.
+- Why: Craig chose to spec the inbox-workflow consolidation before building a load-bearing 3-to-1 merge of synced workflows.
+- Artifacts: docs/inbox-workflow-consolidation-spec.org; todo.org "Consolidate inbox/triage workflows + scheduled inbox check".
+** 2026-06-23 Tue @ 22:05:00 -0400 — Craig — decision-maker
+- What: resolved all four decisions; added the =auto inbox zero= scheduled mode (ask-for-interval, empty-cycle acknowledgement only, find → summarize → file → displayed queue → ask-to-execute, cross-cycle dedup). Status → ready for review.
+- Why: chose Option A (4→1 engine) and specified the recurring-check behavior in full.
+- Artifacts: Decisions [4/4]; Design "Auto inbox zero" subsection.
+** 2026-06-23 Tue @ 22:15:58 -0400 — Codex — reviewer
+- What: spec-review pass rated the spec =Not ready= and added two findings: the fully unattended =/schedule= behavior is not specified, and stale-reference verification leans on a checker that does not validate workflow links.
+- Why: the current design is strong enough for the consolidation and interactive =/loop= mode, but the stated scheduled/unattended goal would force implementers to invent behavior before shipping.
+- Artifacts: Review findings [0/2].
+** 2026-06-23 Tue @ 22:28:00 -0400 — Claude — responder
+- What: both findings accepted and folded. Finding 1 (blocking) resolved by narrowing v1 to the interactive =auto inbox zero= (=/loop=) and deferring the fully-unattended =/schedule= contract to vNext with its open questions named — Summary, Goals, Problem/Context, Scope tiers, and Design updated. Finding 2 resolved by adding the exact stale-reference grep as an acceptance item + Phase 2 step and dropping the over-claim that the integrity checker validates workflow links. Findings [2/2], Decisions [4/4]; scope narrowed (not expanded), so no readiness-rubric rerun needed. Status → Ready.
+- Why: the scheduled/unattended promise outran what Craig actually specced (he detailed the interactive loop); sequencing the cron pass to vNext keeps v1 honest. The checker genuinely doesn't catch stale workflow links.
+- Artifacts: Decisions [4/4]; Review findings [2/2]; vNext task to be logged for the unattended cron contract.
diff --git a/docs/specs/wrapup-routing-spec.org b/docs/specs/wrapup-routing-spec.org
new file mode 100644
index 0000000..07d9ec6
--- /dev/null
+++ b/docs/specs/wrapup-routing-spec.org
@@ -0,0 +1,226 @@
+#+TITLE: Wrap-Up Inbox/Transcript Routing — Spec
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-06-13
+#+TODO: TODO | DONE
+#+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED
+
+* IMPLEMENTED Wrap-Up Inbox/Transcript Routing — Spec
+:PROPERTIES:
+:ID: 00b47414-2213-4a99-be35-48ceb266fc08
+:END:
+- 2026-07-04 Sat @ 11:49:59 -0500 — DOING → IMPLEMENTED: the task-routing build shipped and is green — route_recommend.py (destination discovery + recommendation, 13 unit tests), the :ROUTE_CANDIDATE: marker wired into inbox process mode, the wrap-it-up router sub-step, and the route-batch delivery helper (9 bats tests). Manual cross-project end-to-end validation tracked as its own task; the transcript-filing half stays a deferred vNext.
+- 2026-07-02 Thu @ 00:17:01 -0400 — retrofitted by spec-sort; status set to DOING (evidence-based, human-confirmed)
+
+* Metadata
+| Status | implemented |
+|----------+-----------------------------------------------------|
+| Owner | Craig Jennings |
+|----------+-----------------------------------------------------|
+| Reviewer | Codex (spec-review) |
+|----------+-----------------------------------------------------|
+| Related | [[file:../../todo.org][todo.org: wrap-up routing task]] · [[file:../design/2026-06-13-wrapup-inbox-transcript-routing-proposal.org][archsetup proposal]] |
+|----------+-----------------------------------------------------|
+
+* Summary
+
+At wrap-up, an inbox handoff that belongs to another project, once accepted and filed locally, has no clean home in the current project's =todo.org=. This adds an optional routing step to =wrap-it-up.org=: surface the filed keepers whose home is elsewhere, recommend a destination for each, and on one confirmation deliver each to that project's =inbox/= via =inbox-send= (one handoff per task), removing it from the local =todo.org=. The destination's own next session files it through =process-inbox=, applying that project's value gate, priority scheme, and =todo-format.md=. A parallel step (vNext) files meeting-transcript recordings into the right project's =assets/=.
+
+* Problem / Context
+
+=process-inbox.org= dispositions each handoff as act / fold / file / reject, and "file as TODO" lands the task in the *current* project's =todo.org=. When the real home is a different project, the choices today are: file it locally and let it rot in the wrong tracker, hand-edit two projects' =todo.org= files, or defer it and carry the debt to next session.
+
+The wrap-up's existing Step 3 "Inbox sanity check" only counts unprocessed items and blocks the wrap until they clear. It answers "is the inbox clean?" — it doesn't route anything.
+
+Meeting transcripts have the same homelessness: a recording dropped during a session belongs in some project's =assets/=, but nothing moves it there at wrap.
+
+The friction is small per-item but recurring, and the manual cross-project edit is error-prone (two files, two repos, easy to leave one half-done).
+
+* Goals and Non-Goals
+
+** Goals
+- At wrap-up, surface filed keepers whose home is a different project, with a recommended destination each.
+- Route the whole batch on one confirmation ("go with recommendations") or leave it entirely ("skip"). No per-item triage.
+- Deliver each routable keeper to the destination's =inbox/= via =inbox-send=, one handoff per task, and remove the keeper from the local =todo.org= on send. The destination files it through its own =process-inbox=.
+- Provenance is automatic: =inbox-send= stamps the source project and date on every handoff (the =from-<source>= filename and =#+SOURCE:= line). The delivery shows in the destination inbox; the removal shows in the source's git diff.
+- The destination set is any project with an =inbox/= — reuse =inbox-send='s existing discovery.
+
+** Non-Goals
+- Not a wrap gate. A skip is a clean, complete wrap.
+- Not per-item triage. The interaction is batch-level: go or skip.
+- Not a replacement for =process-inbox.org='s value gate. Routing assumes the item is already an accepted keeper.
+- Not a confidence-free auto-mover. A low-confidence destination recommendation says so, and the batch "go" stays trustworthy because the surfaced list is reviewable before the keystroke.
+
+** Scope tiers
+- v1: task/event routing by =inbox-send= delivery to the destination's =inbox/=. The interaction, the recommendation engine, the candidate-set marker stamped at file time, reusing =inbox-send='s discovery and delivery.
+- Out of scope: per-item destination editing, an interactive correction loop, moving items that aren't accepted keepers, a new cross-repo =todo.org= move primitive (the superseded direct-move design).
+- vNext: meeting-transcript filing (gated on the unresolved source-location decision and the file-vs-file+extract question — see Decisions).
+
+* Design
+
+** User-facing (the wrap interaction)
+
+The router is a new sub-step of =wrap-it-up.org='s Step 3, running after the existing inbox sanity check. Its input is filed keepers, not raw inbox files (decision: Reading B): tasks =process-inbox= accepted and filed into the local =todo.org= this session whose inferred home is a different project. When the router finds such a keeper, it surfaces it in a list, one line each: the task, the recommended destination project, and a confidence marker when the inference is weak. Then two options, batch-level:
+
+1. Go with the recommendations — route every recommended item (inbox-send to the destination + local removal).
+2. Skip — leave the whole batch in place. A skip is a clean wrap.
+
+That is the entire interaction. No per-item walk. The surfaced list is the review surface; the single keystroke is trustworthy because the list was reviewable and low-confidence recommendations flagged themselves.
+
+On "go", each routable keeper is delivered to its recommended destination's =inbox/= via =inbox-send= (one handoff per task) and removed from the local =todo.org=; the destination's own next session files it through =process-inbox=. A skipped or no-match item stays where it is; the existing sanity check still governs whether the wrap is clean.
+
+** Implementer (the mechanics)
+
+*Candidate set (what the router considers).* Reading B means the router does not scan the whole local backlog — it would otherwise suggest moving legitimate local tasks every wrap. The candidate set is keepers =process-inbox= filed this session whose inferred home differs from the current project, identified by a marker stamped at file time (decision D8): =process-inbox='s "file as TODO" step stamps =:ROUTE_CANDIDATE: <inferred-project>= on any keeper whose inferred home is not the current project. At wrap, the router's candidate set is exactly the local tasks carrying that property — never the standing backlog.
+
+*Destination discovery.* Reuse =inbox-send.py='s existing =discover_projects= (a project is a directory with =.ai/= AND =inbox/=). The destination must have an =inbox/= to receive a handoff, so that is the natural destination set — no new discovery code. A project with a =todo.org= but no =inbox/= cannot receive an inbox handoff and must be bootstrapped first; in practice every active project has an =inbox/=.
+
+*Delivery.* For each candidate, on "go": (1) =inbox-send <destination> --file= a one-task handoff into the destination's =inbox/= (one file per task, so the destination's =process-inbox= dispositions it as a single item), then (2) remove the keeper from the local =todo.org=. Step 1 is a cross-project write, but it uses the =cross-project.md=-sanctioned path (dropping a file in another project's inbox needs no confirmation); step 2 is a single-file edit in the current project's own =todo.org=, which the wrap is already committing. No new cross-repo move primitive, no foreign =todo.org= edit.
+
+*Provenance and filing.* =inbox-send= stamps the source and date automatically (=from-<source>= filename + =#+SOURCE:= line), so the destination's session knows where the item came from. That session files it through its own =process-inbox= — value gate, priority scheme, =todo-format.md= — so the task lands per the destination's conventions rather than as an externally-authored insertion.
+
+*Recovery (mis-route).* If the recommendation engine picks a wrong destination, the receiving session rejects it via =process-inbox='s reject-from-another-project flow (write a response, =inbox-send= it back to the source named in the provenance, delete the local copy). The task returns to the source project's inbox; nothing is lost or corrupted. This is why removing the source on send is safe — the reject path is the undo.
+
+*Recommendation engine.* Infer the destination from the item's content — project names, file paths, topic words — matched against the discovered project list, with a confidence tier: *strong* = a destination project's name or path appears literally in the item; *weak* = topic-word overlap only; *none* = no match, the item stays put and is never surfaced as a route. "Go" routes strong and weak items (weak visibly labeled); a no-match item is left in place. Pure function =(item, project-list) → (destination, confidence)=, unit-tested directly. The engine is the interesting, uncertain part; it earns the spec.
+
+* Alternatives Considered
+
+** Per-item triage instead of batch go/skip
+- Good, because it gives precise control over each destination.
+- Bad, because it taxes the common case (a batch that's all-correct, or all-stay) with a walk. Craig explicitly asked for two options, not a triage loop.
+- Neutral, because per-item correction could return as a vNext refinement if batch-only proves too blunt.
+
+** Fold the router into the existing Inbox sanity check step
+- Good, because one inbox step is simpler than two.
+- Bad, because the sanity check *gates* the wrap (blocks until clean) and the router is *optional* (skip is clean). Merging a blocking check with an optional action muddies both.
+- Neutral, because the two share discovery code while staying separate steps. (Resolved: D1 keeps them separate, with the router acting on filed keepers rather than inbox files.)
+
+** Reuse process-inbox's "file as TODO" with a destination argument
+- Good, because it avoids a second mechanism.
+- Bad, because =process-inbox= runs per-item mid-session against the local project; the router runs at wrap, batch-level, cross-project. Different cadence, different scope.
+- Neutral, because both ultimately call the same atomic move helper — the helper is the shared primitive, the two callers stay distinct.
+
+* Decisions [9/9]
+
+** DONE Reuse the Open Work matcher for destination anchoring
+- Context: the move needs a reliable insertion point in the destination =todo.org=; guessing risks corrupting another project's file.
+- Decision: We will reuse =todo-cleanup.el='s =tc--find-section "open work"= matcher, which already handles the unique / missing / ambiguous cases, and skip+surface any destination without a clean Open Work heading.
+- Consequences: easier — no new parser, consistent with =--archive-done=. Harder — destinations must carry the "Open Work" heading convention, so a project with a differently-named section is silently unroutable until it conforms.
+
+** SUPERSEDED Move atomically through a helper, never hand-edit two repos
+Superseded 2026-06-21 by "Deliver via inbox-send" below. The original plan built a new atomic helper to insert a subtree into a foreign =todo.org= and remove the source. The inbox-route delivers the keeper to the destination's inbox instead, so no cross-repo move primitive is built.
+- Context: a move touches two files in two repos; a half-done move loses or duplicates a task.
+- Decision (superseded): route every move through one helper that inserts under the destination's Open Work heading and removes the source as one operation.
+
+** SUPERSEDED Cross-project writes stay visible and carry provenance
+Superseded 2026-06-21 by "Deliver via inbox-send" below. =inbox-send= already stamps provenance (=from-<source>= filename + =#+SOURCE:= line), so the hand-stamped note is unnecessary; the destination files the item through its own gate rather than receiving an externally-authored insertion.
+- Context: writing into another project's =todo.org= crosses the =cross-project.md= scope boundary.
+- Decision (superseded): treat the batch "go" as authorization, leave the move visible in the destination's git diff, and stamp a one-line provenance note on each moved task.
+
+** DONE Separate router step, operating on filed keepers (Reading B)
+- Context: the sanity check gates the wrap on inbox/ contents; the router is optional. The deeper question was the router's input — raw inbox files (Reading A, which overlaps the sanity check) or already-filed keepers that belong elsewhere (Reading B, a todo-routing concern).
+- Decision: We will keep the router a separate optional sub-step after the sanity check, and its input is Reading B: accepted keepers process-inbox filed into the local =todo.org= whose inferred home is another project. The sanity check stays a pure inbox gate; the router is a todo-routing action that shares only the destination-discovery code.
+- Consequences: easier — each step has one job, the gate can't be muddied by an optional action, and the router never competes with the inbox gate over the same files. Harder — the candidate set (which local tasks the router considers) needs a marking mechanism (see the Implementer "candidate set" note); Reading A's "dispose raw inbox files at wrap" convenience is given up.
+
+** DONE Transcript routing deferred to vNext
+- Context: transcripts file as artifacts, not tasks, and a meeting usually produces both a recording to keep and action items to track. Two unknowns block it: where recordings accumulate (a recordings inbox, a downloads dir, wherever the meeting tooling drops them), and whether filing should also extract action items into the destination's =todo.org=.
+- Decision: We will defer transcript routing to vNext. Both the source-location dependency and the file-only-vs-extract-action-items question are deferred with it, to be settled when the vNext work is specced. v1 ships task routing only.
+- Consequences: easier — v1 isn't blocked on the unresolved source location. Harder — until vNext, a meeting recording still has no automatic home; only its action items (if filed as tasks) route through v1.
+
+** DONE Keep defer-and-stage and the router as distinct policies
+- Context: the 2026-06-12 Skeptical Review added a defer-and-stage path in =process-inbox.org= that files a =[#B]= VERIFY for shared-asset proposals parked for review. That also turns an inbox item into a =todo.org= task — overlapping surface with this router.
+- Decision: We will keep them distinct. Defer-and-stage parks a proposal-under-review locally as a VERIFY; the router moves an accepted keeper to its home project as a TODO. They differ on review status (proposal vs accepted) and destination (local vs cross-project), and share only the atomic move helper, not the policy. Reading B makes the split clean: the router acts on accepted keepers, never on proposals under review.
+- Consequences: easier — two clear, non-competing policies on one shared primitive. Harder — the workflow prose must name the boundary so a future reader doesn't collapse them and reintroduce the ambiguity.
+
+** DONE Deliver via inbox-send to the destination's inbox, not a direct todo.org move (supersedes D2/D3)
+- Owner / by-when: Craig / ratified 2026-06-21 (spec-response)
+- Context: D2/D3 built a new atomic helper that edits a foreign =todo.org= and removes the source, with a hand-stamped provenance note. =inbox-send= + =process-inbox= already do cross-project delivery: inbox-send writes the handoff with =from-<source>= provenance, and the destination's process-inbox files it through that project's own gate. =cross-project.md= names the inbox as the sanctioned cross-scope write path. A verified precondition reversed the old assumption — some projects have =inbox/= but no =todo.org=, so direct-move's discovery silently drops keepers headed there while inbox-route delivers.
+- Decision: We will route each keeper by =inbox-send= into the destination's =inbox/= (one handoff per task) and let the destination's own =process-inbox= file it; we will not edit the destination's =todo.org= directly. D2 (atomic move helper) and D3 (hand-stamped provenance) are superseded — the helper isn't built, and provenance is inbox-send's by construction.
+- Consequences: easier — no new cross-repo write primitive, no foreign-tracker corruption risk, provenance and per-project filing for free, graceful when the destination lacks a =todo.org=. Harder — filing is deferred to the destination's next session (self-resolving, since startup auto-runs =process-inbox= on a non-empty inbox), and a project never opened accumulates a visible inbox backlog rather than a silent foreign insertion.
+
+** DONE Candidate-set marking: tag :ROUTE_CANDIDATE: at process-inbox file time (Option A)
+- Owner / by-when: Craig / ratified 2026-06-21 (spec-response)
+- Context: the router must consider only this-session-filed inbox keepers whose home is elsewhere, never the standing backlog. Two options: tag at file time (process-inbox stamps a marker) or infer from a =CREATED=-this-session stamp + content. =process-inbox= does not stamp =:CREATED:= today, so the inference option would need that paired edit anyway, removing its only advantage.
+- Decision: We will tag at file time. =process-inbox='s "file as TODO" step stamps =:ROUTE_CANDIDATE: <inferred-project>= on any keeper whose inferred home differs from the current project; the router's candidate set is the local tasks carrying it.
+- Consequences: easier — precise (zero standing-backlog false positives), the inference happens once where context is richest, and the marker doubles as the router's "go" trigger. Harder — a paired edit to =process-inbox.org= Phase D ships coupled with the router.
+
+** DONE Source removal is a local todo.org edit on send; recovery via the reject flow
+- Owner / by-when: Craig / ratified 2026-06-21 (spec-response)
+- Context: the review left source-handling vague ("leave the source until the destination confirms by filing"), but there is no confirmation callback, so leaving it duplicates the task once the destination files. The keeper was filed into the *current* project this session and doesn't belong there.
+- Decision: On "go" we will remove the routed keeper from the *current* project's =todo.org= (a local single-file edit, not a cross-repo write) right after the =inbox-send=. If the destination rejects the handoff, =process-inbox='s reject-from-another-project flow returns it to the source's inbox, so the removal is reversible.
+- Consequences: easier — no duplication, the only deletion is from a file we own and are already committing, the reject path is the undo. Harder — a brief window exists where the task lives only as an in-flight inbox handoff (between send and the destination's filing); acceptable because the handoff file is durable and the reject path recovers a mis-route.
+
+* Implementation phases
+
+** Phase 1 — Destination discovery (reuse inbox-send)
+Reuse =inbox-send.py='s =discover_projects= (a directory with =.ai/= AND =inbox/=) as the destination set — no new discovery code. Confirm the destination universe: if a real destination has a =todo.org= but no =.ai/+inbox/=, name it and bootstrap its inbox; otherwise the existing filter already covers it. Leaves the tree working.
+
+** Phase 2 — Candidate-set marking in process-inbox
+Extend =process-inbox.org='s "file as TODO" step (Phase D) to stamp =:ROUTE_CANDIDATE: <inferred-project>= on any keeper whose inferred home differs from the current project (decision D8). Sync the =.ai/= mirror. This is the paired workflow edit that lets the wrap-up router find candidates without scanning the standing backlog. (Replaces the superseded atomic-move helper.)
+
+** Phase 3 — Recommendation engine
+Infer destination from item content against the discovered list, with a confidence tier. Pure function =(item, project-list) → (destination, confidence)=. Unit-tested: strong match (destination project named or path present literally → high) , weak match (topic-word overlap only → low, still routed but labeled), no match (stays put, never surfaced), two-project tie (lowest-confidence / tie-break), empty project list (all stay put). The engine is shared by process-inbox's file-time marker (Phase 2) and the wrap-up router (Phase 4), so it lives where both can call it.
+
+** Phase 4 — Wrap-up step wiring
+Add the optional router sub-step to =wrap-it-up.org= Step 3, after the inbox sanity check: surface the candidate batch (one line each: task, destination, delivery mode, confidence), the two options (go / skip). On "go", for each candidate, =inbox-send= a one-task handoff to the destination's =inbox/= and remove the keeper from the local =todo.org=. Empty candidate set = zero interaction (silent). Name the gate-vs-optional split in the prose (the sanity check gates; the router is optional). Sync the =.ai/= mirror.
+
+** Phase 5 — Transcript routing (vNext, gated on the transcript decision)
+Only after the transcript-scope decision resolves. File a recording into the destination =assets/= per =working-files.md=, batch go/skip mirroring the task router.
+
+* Acceptance criteria
+- [ ] At wrap, a filed keeper naming another project is surfaced with that project as the recommended destination.
+- [ ] "Go" delivers every recommended item as a one-task =from-<source>= handoff into its destination's =inbox/= and removes it from the local =todo.org=.
+- [ ] "Skip" leaves every item in place and the wrap completes cleanly.
+- [ ] An empty candidate set produces zero interaction (no prompt, no "0 items" line).
+- [ ] A weak (low-confidence) recommendation is visibly labeled in the surfaced list; a no-match item is never surfaced as a route.
+- [ ] A candidate whose destination has an =inbox/= but no =todo.org= still delivers (degrades gracefully).
+- [ ] A mis-routed handoff is recoverable via =process-inbox='s reject-from-another-project flow, returning it to the source's inbox.
+- [ ] The router considers only =:ROUTE_CANDIDATE:=-tagged keepers, never the standing backlog.
+
+* Readiness dimensions
+- Data model & ownership: items are org subtrees; the destination owns the moved task after the move (provenance note records origin). N/A for remote/cached state — all local files.
+- Errors, empty states & failure: missing/ambiguous Open Work heading → skip+surface; failed move → atomic no-op; empty routable set → router stays silent (no prompt).
+- Security & privacy: N/A — local org files, no credentials or external services.
+- Observability: the move shows in the destination's git diff plus the provenance line; the surfaced batch list is the pre-move view.
+- Performance & scale: bounded by inbox size (single digits) and project count (tens); no hot path.
+- Reuse & lost opportunities: reuses =tc--find-section= and todo-cleanup's subtree-move; widens existing discovery rather than adding a parallel one.
+- Architecture fit & weak points: the recommendation engine is the weak point (a wrong-confident destination is the worst failure) — mitigated by the confidence label and reviewable batch list.
+- Config surface: possibly a discovery-root list (defaults to =~/projects/=, =~/code/=, matching =inbox-send.py=). Name it if it needs to be user-visible.
+- Documentation plan: =wrap-it-up.org= step prose; a note in =cross-project.md= that the router is a sanctioned cross-project write path.
+- Dev tooling: ERT for the elisp helper + discovery; the existing =make test= picks up new test files by glob.
+- Rollout, compatibility & rollback: additive workflow step; rollback is removing the sub-step. No persisted-data migration.
+- External APIs & deps: none.
+
+* Risks, Rabbit Holes, and Drawbacks
+- *Recommendation accuracy is the rabbit hole.* A confidently-wrong destination silently files a task in the wrong project. Dodge: keep the engine conservative, label low confidence, and keep the batch list reviewable before the keystroke. Don't chase a clever inference model in v1.
+- *Two inbox-touching steps* (sanity check + router) risk reading as redundant. Dodge: the D1 decision states the gate-vs-optional split in the workflow prose.
+- *Scope creep into transcripts* before the source-location question is answered would stall v1. Dodge: transcripts are explicitly vNext behind decision D4.
+
+* Review dispositions
+
+Everything in the 2026-06-21 review was accepted, with one modify:
+
+- *Modified — H1 source-handling.* The review proposed leaving the source keeper in place "until the destination confirms by filing." There is no confirmation callback, so leaving it would duplicate the task once the destination files. Resolved instead (decision D9) to remove the keeper from the *local* =todo.org= on send — a single-file edit in the project we already own and are committing, with =process-inbox='s reject flow as the undo for a mis-route. Keeps the no-foreign-write safety win without the duplication.
+
+Everything else accepted as written: H1 (inbox-route supersedes direct-move; D2/D3 superseded), H1a (one handoff per task), H1b (reuse =inbox-send= discovery; Phase 1), H2 (tag at file time; D8), M1 (confidence tiers defined in Phase 3 + acceptance), M2 (empty-set silence; acceptance), M3 (paired =process-inbox= edit; Phase 2), M4 (=cross-project.md= note adjusted to "the router uses the sanctioned inbox path").
+
+* Review and iteration history
+
+** 2026-06-13 Sat @ 01:23:13 -0500 — Claude Code (rulesets) — author
+- What: initial draft. Problem, goals/scope tiers, two-altitude design, alternatives, six decisions (three DONE from grounding, three TODO for Craig), five implementation phases, acceptance criteria, readiness dimensions, risks.
+- Why: the archsetup 2026-06-13 handoff cleared the spec bar in inbox triage and was filed spec-bound rather than applied. This draft turns the proposal into a reviewable design with the open questions isolated as decision tasks.
+- Artifacts: proposal source at =docs/design/2026-06-13-wrapup-inbox-transcript-routing-proposal.org=; grounded against =wrap-it-up.org= Step 3, =todo-cleanup.el= =tc--find-section=, and =inbox-send.py= discovery.
+
+** 2026-06-13 Sat @ 01:36:28 -0500 — Craig Jennings + Claude Code (rulesets) — author
+- What: resolved all three open decisions. The router's input is Reading B (filed keepers that belong elsewhere, not raw inbox files), so D1 keeps it a separate sub-step from the inbox gate and D5 keeps it distinct from the defer-and-stage router; D4 defers transcript routing to vNext. Reworked the design (input definition, a candidate-set note bounding the router to session-filed keepers) and Phase 3 to match. Cookie now [6/6]; Status moved to ready-for-review.
+- Why: Craig chose Reading B after the A-vs-B input ambiguity surfaced as the root under D1 and D5. Reading B keeps the inbox gate, the router, and defer-and-stage each simple instead of entangling three mechanisms.
+- Artifacts: this spec; the candidate-set marking mechanism is the one detail flagged for spec-review to pin.
+
+** 2026-06-21 Sun @ 01:58:41 -0400 — Claude Code (rulesets) — reviewer
+- What: spec-review pass. Rubric *Not ready*, two blocking findings. H1: the inbox-route alternative (inbox-send each routable keeper to the destination's inbox/, let its own process-inbox file it) supersedes the direct-move design — reshape D2, drop Phase 2 and D3's provenance burden. H2: pin the candidate-set marking to Option A (tag =:ROUTE_CANDIDATE:= at process-inbox file time). Four medium findings (M1 confidence tiers, M2 empty-set silence, M3 paired process-inbox edit phase, M4 cross-project.md note). Full review + drop-in implementation tasks in the review file.
+- Why: Craig challenged D2 directly (why edit a foreign todo.org rather than use the sanctioned inbox-send path). The review confirmed it: inbox-send already emits the exact provenance D3 reinvents, process-inbox already files per-item with the destination's own gate, cross-project.md sanctions the inbox path, and a verified precondition reverses the spec's assumption — chime and yt-sync have inbox/ but no todo.org, so direct-move silently drops keepers headed there while inbox-route degrades gracefully.
+- Artifacts: the review file (since folded into this spec). Next: spec-response to disposition H1/H2 (recommend accept both), which moves the rubric to Ready.
+
+** 2026-06-21 Sun @ 02:06:37 -0400 — Craig Jennings + Claude Code (rulesets) — responder
+- What: folded the spec-review in. Accepted H1 (inbox-route) and H2 (tag at file time); superseded D2 and D3; added D7 (deliver via =inbox-send=), D8 (=:ROUTE_CANDIDATE:= marker at file time), D9 (local source removal + reject-flow recovery). Rewrote Summary, Goals, Design mechanics, Implementation phases (dropped the atomic-move helper — Phase 2 is now the =process-inbox= marker edit), and Acceptance criteria for the inbox-route. One modify (D9) refines H1's vague source-handling. Cookie [9/9]; Status → Ready.
+- Why: Craig's inbox-route challenge held up under review — it reuses the sanctioned cross-project path, gets provenance and per-project filing for free, and degrades gracefully where direct-move drops the task. D9 closes the duplication gap the review left open.
+- Artifacts: review file deleted on this pass. Next: Phase 6 implementation-task breakdown into =todo.org= on the author's go.
diff --git a/flush/SKILL.md b/flush/SKILL.md
index 4c2709a..ca139c1 100644
--- a/flush/SKILL.md
+++ b/flush/SKILL.md
@@ -1,6 +1,6 @@
---
name: flush
-description: Mid-session context flush — the checkpoint half of the wrap/restart rhythm. Refresh the session-context anchor in place, prompt the user to /clear, then resume the same logical session from the anchor without re-running startup. Cheaper tokens and a sharper context window without fragmenting the session into archive files. Agent-callable and agent-initiated: the agent may run the pre-clear checkpoint on its own judgment at a clean task boundary, but /clear is user-only — the agent does all the work, then prompts for the single /clear keystroke. Use when the current task has a clean boundary and the context window is large enough that a reset would sharpen the work. Do NOT use for end-of-day or done-for-now (use wrap-it-up, which archives to .ai/sessions/ and commits), or for a genuine fresh start after being away or on another machine (use startup, which pulls + syncs + surfaces inbox).
+description: Mid-session context flush — the checkpoint half of the wrap/restart rhythm. Refresh the session-context anchor in place, prompt the user to /clear (or in auto mode self-inject it via tmux), then resume the same logical session from the anchor without re-running startup. Cheaper tokens and a sharper context window without fragmenting the session into archive files. Agent-callable and agent-initiated: the agent may run the pre-clear checkpoint on its own judgment at a clean task boundary; interactively /clear stays the user's keystroke, while auto mode ("/flush auto", for unattended runs like the no-approvals speedrun or a recurring loop) arms .ai/scripts/self-inject.sh so tmux types /clear and a resume line at the agent's own idle prompt — zero human keystrokes. Use when the current task has a clean boundary and the context window is large enough that a reset would sharpen the work. Do NOT use for end-of-day or done-for-now (use wrap-it-up, which archives to .ai/sessions/ and commits), or for a genuine fresh start after being away or on another machine (use startup, which pulls + syncs + surfaces inbox).
---
# /flush — Mid-Session Context Checkpoint
@@ -20,9 +20,11 @@ This is the checkpoint half of the wrap/restart rhythm. It is distinct from two
The skill is agent-callable. The agent may also **initiate** a flush on its own judgment when the rhythm calls for it (see below) — it runs the pre-clear checkpoint, then prompts the user to type `/clear`.
-## The hard constraint
+## The constraint, and the auto-mode exception
-`/clear` is a user-only command. The agent **cannot** execute it. "Agent-initiated" means the agent runs the pre-clear checkpoint (refresh the anchor + verify the write landed) on its own, then **prompts** the user: "checkpoint saved, type /clear to reset." The agent proposes and does all the work; the user supplies the single `/clear` keystroke. Never design or imply a flow where the agent self-triggers `/clear`.
+`/clear` is a prompt command — the agent cannot execute it as a tool call. Interactively, "agent-initiated" means the agent runs the pre-clear checkpoint (refresh the anchor + verify the write landed) on its own, then **prompts** the user: "checkpoint saved, type /clear to reset." The user supplies the single `/clear` keystroke.
+
+**Auto mode** (`/flush auto`) is the sanctioned exception for unattended sessions: after the checkpoint, the agent arms the tmux server to type `/clear` and a resume line at its own idle prompt (see Auto mode below). The constraint that never bends is the **gate order**: the anchor write is verified on disk *before* anything arms or prompts a clear. There is no recovering the conversation afterward.
## When the agent should initiate
@@ -68,6 +70,30 @@ That is the wrap/restart rhythm. When both conditions hold, run Phase 1 and end
6. **Hand off the clear.** Tell the user the checkpoint is saved, name the anchor path, and prompt: type `/clear` now, then send any message to resume.
+## Auto mode — self-injected clear for unattended sessions
+
+`/flush auto` runs Phase 1 in full (steps 1-5, including the write-verified gate), then replaces step 6's user prompt with a self-injection. Proven live in the archsetup session 2026-07-02; the mechanism and its gotchas live in `.ai/scripts/self-inject.sh` (synced into every project).
+
+1. **Derive the pane first, synchronously, from this shell:**
+
+ ```bash
+ pane=$(.ai/scripts/self-inject.sh)
+ ```
+
+ This must happen before arming: the armed step runs under the tmux *server*, where ancestry-based pane detection cannot work.
+
+2. **Arm the injection via the tmux server, then end the turn immediately:**
+
+ ```bash
+ tmux run-shell -b ".ai/scripts/self-inject.sh -t $pane 25 '/clear' 15 'go (auto-flush resume: continue per Next Steps)'"
+ ```
+
+ `run-shell -b` is load-bearing — a detached child of a tool call (`setsid`/`nohup`/`&`) dies when the tool call ends; only a server-owned process survives the turn boundary. The delays let the turn fully end before `/clear` lands (25s) and let the `SessionStart` hook finish before the resume line lands (15s).
+
+3. **End the turn.** The prompt must be idle when the keys arrive. The injected `/clear` fires the same `SessionStart(clear)` hook as a hand-typed one; the injected resume line starts the next turn. Zero human keystrokes.
+
+**When to use it:** unattended runs only — the no-approvals speedrun, a recurring loop, any session where nobody is at the keyboard. The collision hazard is real: keys injected while a human is mid-keystroke merge into their input (`/clear` has become `/clearto`). If a user may be present, say the window is armed and to keep hands off, or use the interactive prompt instead.
+
## Phase 2 — Post-clear resume (hook-driven)
This half is driven by the `SessionStart(clear)` hook, not by this skill — but it is documented here so the loop is legible.
diff --git a/hooks/README.md b/hooks/README.md
index 71b3613..9c4268d 100644
--- a/hooks/README.md
+++ b/hooks/README.md
@@ -10,6 +10,9 @@ Machine-wide Claude Code hooks that install into `~/.claude/hooks/` and apply to
| `git-commit-confirm.py` | `PreToolUse(Bash)` | Silent-unless-suspicious gate on `git commit`. Only prompts when the message contains AI-attribution patterns, the message can't be parsed (editor would open), no files are staged, or the git author is unusable. Clean commits pass through without a modal. Parses both HEREDOC and `-m`/`--message` forms. |
| `gh-pr-create-confirm.py` | `PreToolUse(Bash)` | Gates `gh pr create` behind a confirmation modal showing title, base←head, reviewers, labels, assignees, milestone, draft flag, and body (HEREDOC or quoted). |
| `destructive-bash-confirm.py` | `PreToolUse(Bash)` | Gates destructive commands (`git push --force`, `git reset --hard`, `git clean -f`, `git branch -D`, `rm -rf`) with a modal showing the command, local context (branch, uncommitted file counts, targeted paths), and a warning banner. Elevates severity when force-pushing protected branches or targeting root/home/wildcard paths. |
+| `inbox-boundary-check.sh` | `Stop` | Soft-nudges the agent to process pending `inbox/` handoffs before yielding. Blocks the stop once (injects a reason with the pending count) when `inbox-status -q` reports pending items; steps aside on the harness re-entry (`stop_hook_active`) so a mid-task pause or an unprocessable item never wedges. No-ops in any project without an `inbox/` or without `inbox-status`. |
+| `ai-wrap-teardown.sh` | `Stop` | Re-verifies the strict clean-tree certificate and its HEAD before consuming a wrap sentinel. Dirty or uncertified state blocks Stop with an actionable report; verified state tears down the matching ai-term session or starts the guarded shutdown countdown. Emits the appropriate Claude or Codex response shape. |
+| `rulesets-write-boundary.py` | `PreToolUse(Edit\|Write)` | Resolves write targets through symlinks and denies cross-project edits that land inside rulesets. Directs the sender through `inbox-send rulesets`; rulesets sessions themselves pass. Codex maps `apply_patch` to the same matcher. |
Shared library (not a hook): `_common.py` — `read_payload()`, `respond_ask()`, `scan_attribution()`. Installed as a sibling symlink so the two Python hooks can `from _common import …` at runtime.
diff --git a/hooks/_common.py b/hooks/_common.py
index e82f7ed..c1e0578 100644
--- a/hooks/_common.py
+++ b/hooks/_common.py
@@ -65,6 +65,24 @@ def respond_ask(reason: str, system_message: Optional[str] = None) -> None:
print(json.dumps(output))
+def respond_deny(reason: str, system_message: Optional[str] = None) -> None:
+ """Emit a PreToolUse response that blocks the tool call outright.
+
+ Unlike `respond_ask`, the user gets no approve option — the call is denied
+ and `reason` tells the agent why, so it can restructure and retry.
+ """
+ output: dict = {
+ "hookSpecificOutput": {
+ "hookEventName": "PreToolUse",
+ "permissionDecision": "deny",
+ "permissionDecisionReason": reason,
+ }
+ }
+ if system_message:
+ output["systemMessage"] = system_message
+ print(json.dumps(output))
+
+
def read_referenced_file(path: str, max_bytes: int = 1_000_000) -> Optional[str]:
"""Read a local file referenced by -F/--file/--body-file so its text can be
attribution-scanned. Return the text, or None if it can't be safely read
diff --git a/hooks/ai-wrap-teardown.sh b/hooks/ai-wrap-teardown.sh
new file mode 100755
index 0000000..ca73ae6
--- /dev/null
+++ b/hooks/ai-wrap-teardown.sh
@@ -0,0 +1,110 @@
+#!/usr/bin/env bash
+# Stop hook: tear down the ai-term session (or power off) AFTER a wrap-up.
+#
+# wrap-it-up.org drops a sentinel only at the very end of a teardown- or
+# shutdown-mode wrap, once commit+push is verified and the valediction is
+# delivered. This hook fires when Claude stops responding; on every NORMAL
+# stop there is no sentinel, so it is a silent no-op. Only after a wrap that
+# requested teardown does the matching sentinel exist, and only then does this
+# hook act — which is what keeps a routine end-of-turn from killing the
+# session.
+#
+# Decoupling via the Stop hook (rather than an inline workflow step) is what
+# lets the valediction flush before the session dies: teardown kills the very
+# tmux session Claude runs in, so it cannot happen inline.
+#
+# Two sentinels, keyed by the project basename (one ai-term session per
+# project, named aiv-<basename>):
+# /tmp/ai-wrap-teardown-<basename> -> cj/ai-term-quit "<basename>"
+# kill the aiv-<basename> tmux session (takes claude with it), kill the
+# vterm buffer, restore geometry. Defined in .emacs.d/modules/ai-term.el.
+# /tmp/ai-wrap-shutdown-<basename> -> cj/ai-term-shutdown-countdown
+# abort-able 10->1 echo-area countdown, then sudo shutdown now. Shutdown
+# supersedes teardown (killing the buffer is moot if powering off).
+#
+# The sentinel is removed BEFORE the emacsclient call: cj/ai-term-quit kills
+# the tmux session this hook runs inside, so the hook process may not survive
+# the call. Clearing first guarantees the sentinel never lingers to re-fire on
+# a later session in the same project. The emacs daemon is a separate process,
+# so it completes the teardown even when the hook is cut off mid-call.
+#
+# emacsclient absent or daemon unreachable: clear the sentinel and exit 0. The
+# session simply stays up — graceful degradation, never a wedge.
+#
+# Wire in ~/.claude/settings.json (see hooks/settings-snippet.json):
+#
+# "Stop": [
+# { "hooks": [
+# { "type": "command",
+# "command": "~/.claude/hooks/ai-wrap-teardown.sh" } ] } ]
+set -u
+
+payload="$(cat)"
+
+# Stop-hook stdin JSON carries cwd; basename it to the project / aiv- session.
+cwd="$(printf '%s' "$payload" | jq -r '.cwd // empty' 2>/dev/null)"
+[ -z "$cwd" ] && cwd="$PWD"
+proj="$(basename "$cwd")"
+
+teardown_sentinel="/tmp/ai-wrap-teardown-${proj}"
+shutdown_sentinel="/tmp/ai-wrap-shutdown-${proj}"
+
+fire() {
+ # $1 = elisp form. Best-effort: only when emacsclient resolves.
+ command -v emacsclient >/dev/null 2>&1 || return 0
+ emacsclient -e "$1" >/dev/null 2>&1 || true
+}
+
+# A sentinel means wrap-up claimed completion. Re-prove that claim immediately
+# before consuming it: the certificate binds a prior strict check to HEAD, and
+# verify also performs a fresh strict check to catch late writes.
+verify_wrap() {
+ local gate="${GIT_WORKTREE_GATE:-}" detail
+ if [ -z "$gate" ]; then
+ gate="$(command -v git-worktree-gate 2>/dev/null || true)"
+ fi
+ if [ -z "$gate" ] || [ ! -x "$gate" ]; then
+ gate="$HOME/code/rulesets/claude-templates/bin/git-worktree-gate"
+ fi
+ if [ ! -x "$gate" ]; then
+ detail="wrap blocked: git-worktree-gate is unavailable; install rulesets tooling and retry"
+ else
+ detail="$("$gate" verify "$cwd" 2>&1)" && return 0
+ fi
+
+ # Codex and Claude consume different Stop-hook response fields. Codex
+ # command-hook payloads always include model; Claude's do not.
+ if printf '%s' "$payload" | jq -e '.model? != null' >/dev/null 2>&1; then
+ jq -n --arg reason "$detail" \
+ '{continue:false, stopReason:$reason, systemMessage:$reason}'
+ else
+ jq -n --arg reason "$detail" \
+ '{decision:"block", reason:$reason}'
+ fi
+ return 1
+}
+
+consume_certificate() {
+ local dir
+ dir="$(git -C "$cwd" rev-parse --absolute-git-dir 2>/dev/null)" || return 0
+ rm -f "$dir/ai-wrap-clean"
+}
+
+# Shutdown supersedes teardown when both are somehow present.
+if [ -f "$shutdown_sentinel" ]; then
+ verify_wrap || exit 0
+ rm -f "$shutdown_sentinel" "$teardown_sentinel"
+ consume_certificate
+ fire '(cj/ai-term-shutdown-countdown)'
+ exit 0
+fi
+
+if [ -f "$teardown_sentinel" ]; then
+ verify_wrap || exit 0
+ rm -f "$teardown_sentinel"
+ consume_certificate
+ fire "(cj/ai-term-quit \"${proj}\")"
+ exit 0
+fi
+
+exit 0
diff --git a/hooks/git-commit-confirm.py b/hooks/git-commit-confirm.py
index 618ac20..cf01948 100755
--- a/hooks/git-commit-confirm.py
+++ b/hooks/git-commit-confirm.py
@@ -46,6 +46,7 @@ from _common import (
read_payload,
read_referenced_file,
respond_ask,
+ respond_deny,
scan_attribution,
)
@@ -53,6 +54,28 @@ from _common import (
MAX_FILES_SHOWN = 25
MAX_MESSAGE_LINES = 30
+# Recognized full-suite test runners across languages.
+TEST_RUNNER_RE = re.compile(
+ r"\b("
+ r"make\s+(?:test|check)"
+ r"|pytest"
+ r"|go\s+test"
+ r"|cargo\s+test"
+ r"|npm\s+(?:run\s+)?test"
+ r"|yarn\s+test"
+ r"|pnpm\s+(?:run\s+)?test"
+ r"|jest|vitest|bats|tox|rspec|phpunit|ctest"
+ r")\b"
+)
+
+GIT_COMMIT_RE = re.compile(r"(?:^|[\s;&|()\n])git\s+(?:-[^\s]+\s+)*commit\b")
+
+BUNDLED_TEST_REASON = (
+ "Blocked: a test run is bundled with `git commit` in one command, where a "
+ "red suite won't stop the commit. Run the full test suite as its own step, "
+ "read the result, and commit only on zero failures (see verification.md)."
+)
+
UNPARSEABLE_MESSAGE = (
"(commit message not parseable from command line; "
"will be edited interactively)"
@@ -68,6 +91,12 @@ def main() -> int:
if not is_git_commit(cmd):
return 0
+ # Hard gate: a test run bundled into the commit command is denied outright,
+ # because an ungated chain lets a red suite commit anyway.
+ if detect_bundled_test_run(cmd):
+ respond_deny(BUNDLED_TEST_REASON)
+ return 0
+
message = extract_commit_message(cmd)
staged = get_staged_files()
stats = get_diff_stats()
@@ -117,6 +146,34 @@ def collect_issues(message: str, staged: list[str], author: str) -> list[str]:
return issues
+def detect_bundled_test_run(cmd: str):
+ """Return a truthy reason if a `git commit` is chained after a test run in
+ one command via an ungated connector (a red suite wouldn't stop the commit).
+
+ `&&` between the test run and the commit is the one safe connector — the
+ commit runs only on a green suite — and is allowed. Any other chaining (`;`,
+ `&`, `|`, `||`, newline, or a pipe that masks the suite's exit) is flagged.
+ The test runner is matched only in the command *prefix* before `git commit`,
+ so a runner name inside the commit message never trips the detector.
+ """
+ if not cmd:
+ return None
+ commit = GIT_COMMIT_RE.search(cmd)
+ if not commit:
+ return None
+ prefix = cmd[: commit.start()]
+ runs = list(TEST_RUNNER_RE.finditer(prefix))
+ if not runs:
+ return None
+ # Everything between the last test run and the commit. Strip the safe `&&`
+ # connectors; anything left that chains commands means the commit isn't
+ # gated on the suite's success.
+ segment = prefix[runs[-1].end():].replace("&&", "")
+ if re.search(r"[;|\n&]", segment):
+ return BUNDLED_TEST_REASON
+ return None
+
+
def is_git_commit(cmd: str) -> bool:
"""True if the command invokes `git commit` (possibly with env/cd prefix)."""
# Strip leading assignments and subshells; find a `git commit` word boundary
diff --git a/hooks/inbox-boundary-check.sh b/hooks/inbox-boundary-check.sh
new file mode 100755
index 0000000..916c000
--- /dev/null
+++ b/hooks/inbox-boundary-check.sh
@@ -0,0 +1,52 @@
+#!/usr/bin/env bash
+# Stop hook: soft-nudge the agent to process pending inbox/ handoffs before it
+# yields control back to Craig.
+#
+# The "check inbox/ at every task boundary" rule (protocols.org, Inbox
+# Monitoring Cadence) is otherwise prose-only, so it holds only as well as the
+# agent remembers it. A Stop is the agent finishing a turn and about to yield,
+# which is the harness's closest event to the rule's own trigger ("after
+# finishing a unit of work, before reporting back"). When handoffs are pending
+# this hook blocks the yield once and injects a reason, so the agent processes
+# them instead of returning with items unseen.
+#
+# Soft-nudge, not hard-block: on the harness re-entry (stop_hook_active: true)
+# the hook steps aside and lets the turn end. An item the agent genuinely can't
+# process, or a mid-task pause to ask Craig a question (also a Stop), never
+# wedges the session.
+#
+# Self-skips everywhere it doesn't apply: no inbox/ dir, no inbox-status, or a
+# clean inbox each exit 0 silently. One hook, every project, no config.
+#
+# Wire in ~/.claude/settings.json (see hooks/settings-snippet.json), Stop array.
+set -u
+
+payload="$(cat)"
+
+# Re-entry after our own nudge: don't nudge twice, let the turn end.
+active="$(printf '%s' "$payload" | jq -r '.stop_hook_active // false' 2>/dev/null)"
+[ "$active" = "true" ] && exit 0
+
+cwd="$(printf '%s' "$payload" | jq -r '.cwd // empty' 2>/dev/null)"
+[ -z "$cwd" ] && cwd="$PWD"
+cd "$cwd" 2>/dev/null || exit 0
+
+# Nothing to enforce without an inbox to watch.
+[ -d inbox ] || exit 0
+
+# Prefer the project-local inbox-status; fall back to one on PATH. Absent both,
+# degrade to a no-op rather than block on a check we can't run.
+status_bin=".ai/scripts/inbox-status"
+[ -x "$status_bin" ] || status_bin="$(command -v inbox-status 2>/dev/null)" || exit 0
+[ -n "$status_bin" ] || exit 0
+
+# inbox-status -q: exit 1 = pending, 0 = clean, 2 = no inbox. Only 1 nudges.
+out="$("$status_bin" -q 2>/dev/null)"
+[ $? -eq 1 ] || exit 0
+
+count="$(printf '%s' "$out" | grep -oE '[0-9]+' | head -1)"
+[ -n "$count" ] || count="Some"
+
+reason="$count pending inbox handoff(s) in $(basename "$cwd"). Process per inbox.org before yielding."
+printf '{"decision":"block","reason":%s}\n' "$(printf '%s' "$reason" | jq -Rs .)"
+exit 0
diff --git a/hooks/rulesets-write-boundary.py b/hooks/rulesets-write-boundary.py
new file mode 100755
index 0000000..35088ea
--- /dev/null
+++ b/hooks/rulesets-write-boundary.py
@@ -0,0 +1,94 @@
+#!/usr/bin/env python3
+"""Block cross-project Edit/Write calls that resolve into rulesets.
+
+Global Claude rules, hooks, skills, commands, and bin tools are symlinks into
+~/code/rulesets. Editing one from another project's session silently dirties
+rulesets and can block every later startup. The sanctioned path is inbox-send;
+the rulesets session owns the canonical edit and its wrap.
+"""
+
+from __future__ import annotations
+
+import os
+import re
+from pathlib import Path
+from typing import Any, Iterable
+
+from _common import read_payload, respond_deny
+
+
+PATCH_PATH = re.compile(
+ r"^\*\*\* (?:Add|Update|Delete) File: (.+)$|^\*\*\* Move to: (.+)$",
+ re.MULTILINE,
+)
+
+
+def inside(path: Path, parent: Path) -> bool:
+ try:
+ path.relative_to(parent)
+ return True
+ except ValueError:
+ return False
+
+
+def candidate_paths(tool_input: Any) -> Iterable[str]:
+ if isinstance(tool_input, dict):
+ for key, value in tool_input.items():
+ if key in {"file_path", "path"} and isinstance(value, str):
+ yield value
+ elif key in {"patch", "input"} and isinstance(value, str):
+ for match in PATCH_PATH.finditer(value):
+ yield match.group(1) or match.group(2)
+ elif isinstance(tool_input, str):
+ for match in PATCH_PATH.finditer(tool_input):
+ yield match.group(1) or match.group(2)
+
+
+def resolve(raw: str, cwd: Path) -> Path:
+ expanded = Path(os.path.expandvars(os.path.expanduser(raw)))
+ if not expanded.is_absolute():
+ expanded = cwd / expanded
+ return expanded.resolve(strict=False)
+
+
+def main() -> int:
+ payload = read_payload()
+ tool_name = payload.get("tool_name", "")
+ if tool_name not in {"Edit", "Write", "apply_patch"}:
+ return 0
+
+ rulesets = Path(
+ os.environ.get("RULESETS_ROOT", "~/code/rulesets")
+ ).expanduser().resolve(strict=False)
+ cwd = Path(payload.get("cwd") or os.getcwd()).resolve(strict=False)
+
+ # A rulesets session owns its own canonical files.
+ if inside(cwd, rulesets):
+ return 0
+
+ blocked: list[str] = []
+ for raw in candidate_paths(payload.get("tool_input", {})):
+ try:
+ target = resolve(raw, cwd)
+ except (OSError, RuntimeError):
+ blocked.append(f"{raw} (real path could not be verified)")
+ continue
+ if inside(target, rulesets):
+ blocked.append(str(target))
+
+ if not blocked:
+ return 0
+
+ shown = ", ".join(blocked)
+ reason = (
+ "Blocked cross-project write into rulesets: "
+ f"{shown}. This path may have been reached through an installed "
+ "symlink. Send the proposed change with inbox-send rulesets instead, "
+ "then let a rulesets session apply and wrap it."
+ )
+ respond_deny(reason, system_message=reason)
+ return 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/hooks/session-start-disarm.sh b/hooks/session-start-disarm.sh
new file mode 100755
index 0000000..520d8c0
--- /dev/null
+++ b/hooks/session-start-disarm.sh
@@ -0,0 +1,42 @@
+#!/usr/bin/env bash
+#
+# SessionStart: disarm any wrap sentinel left over from a previous session.
+#
+# wrap-it-up drops /tmp/ai-wrap-teardown-<project> (or -shutdown-) to ask the
+# Stop hook to kill the tmux session once the wrap certifies clean. The Stop
+# hook deliberately PRESERVES that sentinel when certification fails, so a wrap
+# blocked by a dirty tree can retry on a later stop in the same session without
+# the user re-running the workflow.
+#
+# Nothing bounded that retry to the session. A sentinel armed by a wrap that
+# never certified survived indefinitely and fired in whatever session next
+# happened to reach a clean tree:
+#
+# work, 2026-07-27. The 11:37 wrap requested teardown, failed certification
+# on a dirty tree, and left the sentinel armed. A fresh session started at
+# 13:20, committed twice during startup, went clean — and the next stop
+# consumed the two-hour-old sentinel and killed the terminal mid-work.
+# archsetup's had been armed for two days on a live attached session.
+#
+# A new session means the wrap that armed the sentinel is gone, so its pending
+# teardown is meaningless: clear it. Within-session retry is untouched, because
+# this only runs at session start. If the user still wants teardown, wrap-it-up
+# re-arms it.
+#
+# Scoped to the current project's sentinels only — a concurrent session in
+# another project keeps its own.
+#
+# Silent and exit 0 always. A SessionStart hook must never block a session from
+# starting, and there is nothing here a user needs told.
+
+set -u
+
+payload="$(cat 2>/dev/null || true)"
+
+cwd="$(printf '%s' "$payload" | jq -r '.cwd // empty' 2>/dev/null)"
+[ -z "$cwd" ] && cwd="$PWD"
+proj="$(basename "$cwd")"
+
+rm -f "/tmp/ai-wrap-teardown-${proj}" "/tmp/ai-wrap-shutdown-${proj}"
+
+exit 0
diff --git a/hooks/settings-snippet.json b/hooks/settings-snippet.json
index a5f9d9c..50e3d31 100644
--- a/hooks/settings-snippet.json
+++ b/hooks/settings-snippet.json
@@ -23,6 +23,12 @@
],
"PreToolUse": [
{
+ "matcher": "Edit|Write",
+ "hooks": [
+ { "type": "command", "command": "~/.claude/hooks/rulesets-write-boundary.py" }
+ ]
+ },
+ {
"matcher": "Bash",
"hooks": [
{ "type": "command", "command": "~/.claude/hooks/git-commit-confirm.py" },
@@ -30,6 +36,14 @@
{ "type": "command", "command": "~/.claude/hooks/destructive-bash-confirm.py" }
]
}
+ ],
+ "Stop": [
+ {
+ "hooks": [
+ { "type": "command", "command": "~/.claude/hooks/inbox-boundary-check.sh" },
+ { "type": "command", "command": "~/.claude/hooks/ai-wrap-teardown.sh" }
+ ]
+ }
]
}
}
diff --git a/hooks/tests/test_git_commit_confirm.py b/hooks/tests/test_git_commit_confirm.py
index 83519ad..4bf95cf 100644
--- a/hooks/tests/test_git_commit_confirm.py
+++ b/hooks/tests/test_git_commit_confirm.py
@@ -95,3 +95,61 @@ def test_oversized_file_falls_through_and_hook_asks(tmp_path, monkeypatch):
# And the hook would ask, because UNPARSEABLE_MESSAGE is a flagged issue.
issues = hook.collect_issues(msg, staged=["a.py"], author="Dev <d@e.com>")
assert any("not parseable" in i for i in issues)
+
+
+# --- bundled test-run + commit: the hard gate ------------------------------
+
+def test_bundle_semicolon_make_test_is_flagged():
+ assert hook.detect_bundled_test_run('make test; git commit -m "x"')
+
+
+def test_bundle_ampersand_gated_is_allowed():
+ # `&&` runs the commit only on a green suite — safe, not flagged.
+ assert hook.detect_bundled_test_run('make test && git commit -m "x"') is None
+
+
+def test_bundle_pytest_semicolon_is_flagged():
+ assert hook.detect_bundled_test_run('pytest ; git commit -m "x"')
+
+
+def test_bundle_npm_test_is_flagged():
+ assert hook.detect_bundled_test_run('npm test; git commit -m "x"')
+
+
+def test_bundle_go_test_is_flagged():
+ assert hook.detect_bundled_test_run('go test ./...; git commit -m "x"')
+
+
+def test_bundle_cargo_test_is_flagged():
+ assert hook.detect_bundled_test_run('cargo test ; git commit -m "x"')
+
+
+def test_bundle_bats_is_flagged():
+ assert hook.detect_bundled_test_run('bats tests/ ; git commit -m "x"')
+
+
+def test_bundle_pipe_masks_exit_is_flagged():
+ # `make test | tee log` exits with tee's status, so && gates on tee, not
+ # the suite — a red suite would still commit. Flag it.
+ assert hook.detect_bundled_test_run('make test | tee log && git commit -m "x"')
+
+
+def test_bundle_or_connector_is_flagged():
+ assert hook.detect_bundled_test_run('make test || git commit -m "x"')
+
+
+def test_runner_only_in_message_is_not_flagged():
+ # "make test" inside the commit message must not trip the detector.
+ assert hook.detect_bundled_test_run('git commit -m "remember to make test"') is None
+
+
+def test_plain_commit_is_not_flagged():
+ assert hook.detect_bundled_test_run('git commit -m "fix: thing"') is None
+
+
+def test_gated_chain_before_commit_is_allowed():
+ assert hook.detect_bundled_test_run('cd proj && pytest && git commit -m "x"') is None
+
+
+def test_empty_command_is_not_flagged():
+ assert hook.detect_bundled_test_run("") is None
diff --git a/hooks/tests/test_rulesets_write_boundary.py b/hooks/tests/test_rulesets_write_boundary.py
new file mode 100644
index 0000000..826a941
--- /dev/null
+++ b/hooks/tests/test_rulesets_write_boundary.py
@@ -0,0 +1,110 @@
+import json
+import os
+import subprocess
+import sys
+from pathlib import Path
+
+
+SCRIPT = Path(__file__).parents[1] / "rulesets-write-boundary.py"
+
+
+def run_hook(payload: dict, rulesets: Path) -> dict | None:
+ proc = subprocess.run(
+ [sys.executable, str(SCRIPT)],
+ input=json.dumps(payload),
+ text=True,
+ capture_output=True,
+ env={**os.environ, "RULESETS_ROOT": str(rulesets)},
+ check=True,
+ )
+ return json.loads(proc.stdout) if proc.stdout else None
+
+
+def test_allows_write_from_rulesets_session(tmp_path):
+ rulesets = tmp_path / "rulesets"
+ rulesets.mkdir()
+ target = rulesets / "file"
+ result = run_hook(
+ {
+ "cwd": str(rulesets),
+ "tool_name": "Write",
+ "tool_input": {"file_path": str(target)},
+ },
+ rulesets,
+ )
+ assert result is None
+
+
+def test_blocks_absolute_cross_project_write(tmp_path):
+ rulesets = tmp_path / "rulesets"
+ other = tmp_path / "other"
+ rulesets.mkdir()
+ other.mkdir()
+ result = run_hook(
+ {
+ "cwd": str(other),
+ "tool_name": "Edit",
+ "tool_input": {"file_path": str(rulesets / "rule.md")},
+ },
+ rulesets,
+ )
+ assert result["hookSpecificOutput"]["permissionDecision"] == "deny"
+ assert "inbox-send rulesets" in result["hookSpecificOutput"][
+ "permissionDecisionReason"
+ ]
+
+
+def test_blocks_write_reached_through_symlink(tmp_path):
+ rulesets = tmp_path / "rulesets"
+ other = tmp_path / "other"
+ installed = tmp_path / "installed"
+ rulesets.mkdir()
+ other.mkdir()
+ (rulesets / "rules").mkdir()
+ installed.symlink_to(rulesets / "rules", target_is_directory=True)
+ result = run_hook(
+ {
+ "cwd": str(other),
+ "tool_name": "Write",
+ "tool_input": {"file_path": str(installed / "todo-format.md")},
+ },
+ rulesets,
+ )
+ assert result["hookSpecificOutput"]["permissionDecision"] == "deny"
+ assert str(rulesets) in result["systemMessage"]
+
+
+def test_blocks_apply_patch_target(tmp_path):
+ rulesets = tmp_path / "rulesets"
+ other = tmp_path / "other"
+ rulesets.mkdir()
+ other.mkdir()
+ patch = (
+ f"*** Begin Patch\n*** Update File: {rulesets / 'file'}\n"
+ "@@\n-old\n+new\n*** End Patch\n"
+ )
+ result = run_hook(
+ {
+ "cwd": str(other),
+ "tool_name": "apply_patch",
+ "tool_input": {"input": patch},
+ },
+ rulesets,
+ )
+ assert result["hookSpecificOutput"]["permissionDecision"] == "deny"
+
+
+def test_allows_unrelated_write(tmp_path):
+ rulesets = tmp_path / "rulesets"
+ other = tmp_path / "other"
+ rulesets.mkdir()
+ other.mkdir()
+ result = run_hook(
+ {
+ "cwd": str(other),
+ "tool_name": "Edit",
+ "tool_input": {"file_path": str(other / "file")},
+ },
+ rulesets,
+ )
+ assert result is None
diff --git a/inbox/PROCESSED-2026-06-11-1703-from-home-consolidation-handoff-rulesets.org b/inbox/PROCESSED-2026-06-11-1703-from-home-consolidation-handoff-rulesets.org
deleted file mode 100644
index f0a86b7..0000000
--- a/inbox/PROCESSED-2026-06-11-1703-from-home-consolidation-handoff-rulesets.org
+++ /dev/null
@@ -1,46 +0,0 @@
-#+TITLE: Home is consolidating all personal ~/projects AI projects into itself — heads-up + okay requested
-#+DATE: 2026-06-11
-
-* What's happening
-
-Craig approved and started a migration that folds every AI-managed project under ~/projects (except work) into the home project as area subdirectories, with full git history preserved via git filter-repo + merge. End state: ~/projects holds home and work only; ~/code stays the place for standalone code projects. One todo.org, one priority scheme, one .ai session for all personal project management.
-
-The spec rode along in this same inbox drop (from-home file with "project-consolidation-spec" in the name). It went through a full spec-review cycle (Codex, two passes: Not ready → Ready) and carries the per-fold manifest contract, a two-mode restore runbook, and a layered rollback story (snapper snapshot, untouched server bares, pre-fold tags, retired dirs, 30-day cooling-off).
-
-* How far we've gone (as of 2026-06-11 ~17:00 CDT)
-
-- Phase 0 done: git-filter-repo verified, snapper snapshot 5621, memory-dir tar in ~/backups/, pre-consolidation tag pushed.
-- Phase 1 done: philosophy folded (pilot). Mode-B rollback drill passed in a disposable clone — the fold is provably removable using only its manifest.
-- Phase 2 done: clipper folded (dress rehearsal). First todo.org import with :MIGRATED_FROM: markers and a fold-time triage; first live-link fix (a dirvish bookmark in ~/.emacs.d).
-- Both sources retired to ~/projects/.retired/ (not deleted). Server bares untouched.
-- Manifests at home:docs/consolidation-manifest-{philosophy,clipper}.org; live inventory gate at docs/consolidation-manifest-inventory.org.
-
-* Learnings and adjustments so far
-
-- A manifest can't embed its own redistribution commit's sha (amend changes it). The manifest row now reads "the commit introducing this manifest"; only the merge sha is recorded literally.
-- git clone --no-local is the right clone shape for filter-repo's fresh-clone safety check; --force is banned from the runbook.
-- The fold-merge branch is read from the source (fb-photo-scraper sits on master, not main — assume nothing).
-- Staged freeze instead of full shutdown: the live inventory gate is regenerated per project immediately before its fold, so Craig can keep using not-yet-folded projects until their turn.
-- Imported [#D] tasks fall outside the task-review staleness pool (it tracks A-C only) — by design, but worth knowing when verifying an import.
-- Source todo.org "Reference" sections (non-task content) merge into the area's notes.org, not into home's todo.org.
-
-* What we'd like rulesets to think about
-
-- Edge cases we may not have seen: anything in the templates, workflows, or scripts that assumes one .ai per project under ~/projects (cross-agent-comms discovery, inbox-send target resolution, the ai launcher's project scan, broadcast).
-- Future-project plans: whether new personal "projects" should now start as areas inside home rather than standalone ~/projects entries, and whether the templates should say so.
-- Any rulesets docs or workflows that name the folding projects as handoff/broadcast targets (finances, jr-estate, etc.) — they'll need updating in our Phase 7 ecosystem pass; a list from your side would help us not miss any.
-- The knowledge-base work-root denylist (~/projects/work) is unaffected.
-
-* Ask: confirmed okay to continue
-
-Reply to home's inbox with a confirmed okay (or concerns) for folding the remaining projects. The remaining list, in planned order:
-
-1. jr-estate (Phase 3 — 704M history, 18-task triage, first memory merge)
-2. danneel (Phase 4 — 613M history, 10-task triage)
-3. finances (Phase 5 — 251M history, 43-task triage)
-4. documents, elibrary, health, kit (Phase 6 folds, together)
-5. website + little-elisper — relocate to ~/code as standalone AI projects (Phase 6)
-6. fb-photo-scraper — delete (unmodified upstream clone; origin recorded in spec D2)
-7. Phase 7 ecosystem pass: link sweep, velox migration, rulesets-reference updates, KB node, cooling clock.
-
-We hold the remaining folds until your reply lands.
diff --git a/inbox/PROCESSED-2026-06-11-1703-from-home-project-consolidation-spec.org b/inbox/PROCESSED-2026-06-11-1703-from-home-project-consolidation-spec.org
deleted file mode 100644
index b557012..0000000
--- a/inbox/PROCESSED-2026-06-11-1703-from-home-project-consolidation-spec.org
+++ /dev/null
@@ -1,350 +0,0 @@
-#+TITLE: Project Consolidation — Fold ~/projects AI Projects into Home — Spec
-#+AUTHOR: Craig Jennings
-#+DATE: 2026-06-11
-
-* Metadata
-| Status | Ready — Codex spec-review confirmed 2026-06-11 |
-| Owner | Craig Jennings |
-| Reviewer | Codex (spec-review, 2026-06-11) |
-| Related | [[file:../todo.org::*Project consolidation into home][todo.org task]] |
-
-* Summary
-
-Fold every AI-managed project under =~/projects/= (except =work=) into the =home= project as area subdirectories, with full git history preserved, so all personal tasks live in one =todo.org= and can be prioritized against each other. The end state: =~/projects/= holds =home= and =work=; =~/code/= holds standalone code projects. Every step is reversible — originals are snapshotted, server bare repos stay untouched until a cooling-off period ends, and a written restore procedure covers both single-project and full rollback.
-
-* Problem / Context
-
-Craig manages 12 AI projects under =~/projects/= beside =home= and =work=. Each has its own =.ai/= session machinery, =todo.org=, inbox, memory dir, and git repo on cjennings.net. That isolation was the design — but the life-management projects (finances, jr-estate, danneel, health, kit, clipper, documents…) are all facets of one life, and their tasks compete for the same hours. Today there is no single surface where a [#A] in jr-estate can be weighed against a [#A] in finances; each project's priorities are graded against siblings only. Sessions fragment the same way: a morning touching finances, danneel, and home means three separate session launches, three inboxes, three memory stores, and cross-project handoff files between them.
-
-The forces: (1) one prioritization surface requires one task file (or at least one repo); (2) these projects are critical — legal disputes, estate settlement, finances — so the migration must be provably reversible; (3) per-project git histories carry evidentiary and reference value (especially danneel and jr-estate) and must survive queryably; (4) the =.ai/= ecosystem (sessions, memories, inboxes, workflows) has per-project state that must merge without loss; (5) other machines (velox at minimum) hold clones whose remotes must not break silently.
-
-Survey of the candidates (2026-06-11):
-
-| Project | What it is | .git | Sessions | Open tasks | Memories | Inbox |
-|------------------+---------------------------------------------+-------+----------+------------+----------+-------|
-| clipper | 19 Clipper St SF rental property | 23M | 6 | 1 | 0 | 0 |
-| danneel | 4319 Danneel construction dispute (legal) | 613M | 45 | 10 | 0 | 0 |
-| documents | Disaster-prep document vault | 61M | 7 | 8 | 0 | 2 |
-| elibrary | Ebook + music library management | 2.4M | 6 | 0 | 2 | 3 |
-| fb-photo-scraper | Third-party FB gallery scraper (clone) | 3.4M | 0 | 0 | 0 | 0 |
-| finances | Personal finance (Craig + Christine) | 251M | 24 | 43 | 1 | 4 |
-| health | Personal health management | 5.9M | 27 | 8 | 1 | 3 |
-| jr-estate | JR estate settlement (legal, trust, taxes) | 704M | 44 | 18 | 7 | 2 |
-| kit | Keep In Touch — relationship management | 17M | 20 | 20 | 3 | 4 |
-| little-elisper | The Little LISPer worked in elisp (study) | 4.7M | 2 | 0 | 0 | 0 |
-| philosophy | Philosophy study + discussion notes | 30M | 5 | 0 | 0 | 0 |
-| website | Personal Hugo site (deployed from homelab) | 8.6M | 10 | 11 | 0 | 3 |
-
-All have origin on cjennings.net except fb-photo-scraper (GitHub clone). All =.ai/= dirs are tracked (personal-project model). No project has a live =session-context.org=. Several have small dirty trees (elibrary 2, finances 1, health 1, kit 4, little-elisper 1 files) that must be resolved pre-fold.
-
-* Goals and Non-Goals
-
-** Goals
-- One repo, one =todo.org=, one priority scheme, one =.ai/= session for all personal (non-work) project management.
-- Full git history of every folded project preserved and queryable in place (=git log <area>/= works).
-- All =.ai/= state merged without loss: session archives, memories, inboxes, someday-maybe, project workflows/scripts.
-- A written, tested restore path for any single project and for the whole migration.
-- =~/code/= becomes the only home for standalone code projects.
-
-** Non-Goals
-- No restructuring of home's existing content (homelab docs, assets, music reconciliation dirs stay where they are; re-nesting them under an =infra/= area is vNext).
-- No change to the =work= project in any way.
-- No task content rewriting beyond re-grading priorities to the unified scheme — bodies, links, and histories move as-is.
-- No server-side bare-repo deletion during the migration (archival is a separate, later, post-cooling step).
-- No renaming of the =home= project.
-
-** Scope tiers
-- v1: fold the nine life-management projects (clipper, danneel, documents, elibrary, finances, health, jr-estate, kit, philosophy); relocate the code-shaped three (website, little-elisper, fb-photo-scraper) per Decision 2; ecosystem updates (emacs agenda, rulesets references, velox).
-- Out of scope: work; any =~/code/= project; home's internal restructure.
-- vNext: server bare-repo archival after cooling-off; optional =infra/= re-nesting of homelab content; per-area README normalization.
-
-* Design
-
-** End-state layout
-
-Each folded project becomes a top-level area directory in home, preserving its internal structure:
-
-#+begin_example
-~/projects/home/
- clipper/ danneel/ documents/ elibrary/
- finances/ health/ jr-estate/ kit/ philosophy/
- assets/ docs/ homelab-inventory/ inbox/ scripts/ (existing home content, unchanged)
- todo.org (unified)
- .ai/ (single session machinery)
-#+end_example
-
-This matches the established area pattern (work's =deepsat/assets/=, the working-files convention's =<area>/assets/=). Each area keeps its own =assets/=, =docs/=, internal org files, and a per-area =.gitignore= carrying its old ignore patterns (git honors nested ignores; prefixing patterns into the root ignore is error-prone).
-
-** Git history — filter-repo then merge
-
-For each source project, on a throwaway clone (never the original). =--no-local= forces a real transport-style clone instead of hardlinked/shared objects — the conservative shape git-filter-repo's manual recommends, and what makes the clone a genuine fresh-clone safety check. =branch= is the source's actual default branch (fb-photo-scraper is on =master=; assume nothing):
-
-#+begin_example
-branch=$(git -C ~/projects/<name> symbolic-ref --short HEAD)
-git clone --no-local ~/projects/<name> /tmp/fold-<name>
-cd /tmp/fold-<name>
-git filter-repo --to-subdirectory-filter <name>
-cd ~/projects/home
-git remote add fold-<name> /tmp/fold-<name>
-git fetch fold-<name>
-git merge --allow-unrelated-histories -m "feat(<name>): fold <name> project into home" "fold-<name>/$branch"
-git remote remove fold-<name>
-#+end_example
-
-=git filter-repo --force= is not part of this runbook. If filter-repo refuses to run, the fresh-clone safety check failed — stop, write down why, and fix the clone rather than overriding.
-
-=filter-repo= rewrites every historical path under =<name>/=, so after the merge =git log <name>/somefile= shows the file's full history with original commit messages, authors, and dates. The original repo and its server bare are never touched — the rewrite happens on the temp clone only.
-
-The merged home =.git= grows to roughly 1.8G (dominated by danneel 613M + jr-estate 704M + finances 251M). Acceptable for a private single-user server; noted as a clone-time cost.
-
-** The .ai/ and task merge (per project, after the git merge)
-
-The git merge lands the project's files under =<name>/=, including its old =.ai/= and =todo.org=. A post-merge commit then redistributes that state:
-
-1. /Sessions:/ =git mv <name>/.ai/sessions/*= into home's =.ai/sessions/=, inserting the area into the name: =YYYY-MM-DD-HH-MM-<desc>.org= → =YYYY-MM-DD-HH-MM-<name>-<desc>.org=. Dated names make collisions near-impossible; the prefix preserves provenance.
-2. /todo.org:/ append the project's open work as a new top-level section =* <Area> Open Work= (matching =* Home Open Work=), and its resolved section likewise. Each imported section heading carries a properties drawer marking provenance — =:MIGRATED_FROM: <name>= and =:MIGRATED_ON: YYYY-MM-DD= — so the import boundary stays visible to future edits and to the restore runbook. Walk the incoming tasks with Craig to re-grade priorities onto the unified scheme (home's A-D impact/urgency ladder, generalized beyond infra) and add an area tag (=:finances:=, =:jrestate:=, =:danneel:=, …). Then delete =<name>/todo.org=.
-3. /notes.org:/ the project's Project-Specific Context moves to =<name>/notes.org= (area-local reference, linked from home's =.ai/notes.org=). Active Reminders and Pending Decisions merge into home's =.ai/notes.org= with area attribution.
-4. /Inbox:/ process each project's inbox to zero before the fold (preferred), or move unprocessed items into home's =inbox/= renamed =YYYY-MM-DD-from-<name>-<orig>.ext=.
-5. /Workflows + scripts:/ copy =<name>/.ai/project-workflows/*= and =project-scripts/*= into home's, after a filename-collision check (13 workflow files exist across sources; any collision is resolved by area-prefixing the incoming file). Then delete the area's old =.ai/= machinery (=protocols.org=, =workflows/=, =scripts/= — all template-synced duplicates).
-6. /someday-maybe.org:/ append under an =* <Area>= header in home's.
-7. /Memory:/ copy =~/.claude/projects/-home-cjennings-projects-<name>/memory/*.md= into home's memory dir (rename on slug collision), append index lines to =MEMORY.md=, dedupe against existing entries, then archive the source memory dir into the backup tar (it lives outside git).
-8. /Links:/ =grep -rn "projects/<name>" ~/projects/home ~/.emacs.d ~/sync/org ~/org/roam= and fix every absolute reference to the new path. Record the before-count, fix, re-run, and classify any remaining hits as historical (session archives, this spec) or live — live hits block the fold's close. Relative links inside the area survive the move untouched because internal structure is preserved.
-
-** Per-fold manifest
-
-Every fold produces a tracked manifest at =docs/consolidation-manifest-<name>.org=, written as the fold proceeds and committed with the redistribution commit. The manifest is the restore contract and the verification record — without it, "every step is reversible" is a slogan. It records:
-
-- /Source state:/ source path, HEAD sha, branch, =git status --porcelain=v1= output at gate time (must be empty), origin URL.
-- /Tracked universe:/ the =git ls-files= listing from the source (or a sha256 of it, with the listing in an appendix block). After the merge, every path must exist under =home/<name>/= or appear in the redistribution map below.
-- /Untracked, ignored, and inbox inventories:/ each file listed with its explicit disposition — processed, moved (to where), archived, or intentionally dropped. Nothing leaves the source tree without a line here.
-- /Redistribution map:/ sessions moved (old → new names), todo.org section markers added (=:MIGRATED_FROM:= headings), notes/someday-maybe merges, workflows/scripts copied (with collision resolutions), memory files copied + =MEMORY.md= lines added, link rewrites made (=path:line=, before → after).
-- /Commits:/ the fold merge commit sha and the redistribution commit sha.
-- /Retired path:/ where the source dir went.
-
-Verification per fold checks path lists against this manifest, not file counts — a count can pass while losing files and fail on intentional redistribution.
-
-** Safety net and restore
-
-Layered, oldest-to-newest:
-
-- /Layer 0 — filesystem snapshot./ Before anything: =snapper create --description pre-consolidation= (ratio's root is btrfs with snapper) plus a belt-and-suspenders tar of the memory dirs: =tar czf ~/backups/claude-memory-pre-consolidation-$(date +%F).tgz -C ~/.claude projects=. Content-only restore is sufficient for the memory tar — memory files are plain-text markdown the harness reads by path; no permissions, ACL, or xattr metadata is load-bearing, so plain =tar czf= is the contract.
-- /Layer 1 — server bare repos untouched./ Origin repos on cjennings.net remain exactly as they are through v1. They hold every byte of every project's history independent of anything done locally.
-- /Layer 2 — pre-fold tags./ Home gets =git tag pre-consolidation= before the first fold and =git tag pre-fold-<name>= before each subsequent one, pushed to origin.
-- /Layer 3 — retired dirs./ After a fold is verified, the source dir moves to =~/projects/.retired/<name>= (not deleted). Deleted only after the cooling-off period.
-- /Cooling-off:/ 30 days minimum after the final fold, and not before velox is migrated. Only then does vNext server archival (move bares to =~/git/archive/=) become eligible.
-
-*** Restore one project — two modes
-
-/Mode A — resurrect standalone (leaves home alone)./ =git clone cjennings@cjennings.net:git/<name>.git ~/projects/<name>=, restore its memory dir from the tar. The folded copy in home stays as a harmless duplicate (or is removed later via Mode B). This is the fast path when the need is "I want the project back," not "the fold was wrong."
-
-/Mode B — remove the folded state from home./ A fold is two commits plus out-of-git side effects; removal must unwind all of it, using the manifest:
-
-1. =git revert <redistribution-commit>= then =git revert -m 1 <merge-commit>=, in that order (newest first). This is the supported path while the fold is recent — before later edits touch the shared files.
-2. If either revert conflicts (todo.org and =.ai/notes.org= are hot files — expected once home has moved on), abort it and instead excise manually from the manifest's redistribution map: delete the =:MIGRATED_FROM: <name>= todo.org sections, the area-prefixed session files, the copied workflows/scripts, and =git rm -r <name>/= — one removal commit citing the manifest.
-3. Out-of-git effects either way: delete the copied memory files and their =MEMORY.md= index lines (named in the manifest); re-fix any link rewrites if the old path is coming back.
-
-/Restore everything:/ snapper rollback (or restore the snapshot's =~/projects/=), restore the memory tar, =git reset --hard pre-consolidation= on home plus a coordinated forced push — acceptable on a single-user remote, with the tag as the anchor.
-
-** Multi-machine
-
-velox (and any other machine with clones) keeps working against the untouched server bares until its own migration step: pull home (which brings all folded content), then retire its local =~/projects/<name>= clones the same way. Nothing breaks in the interim — the old remotes still exist; they're just frozen. The =ai= launcher discovers projects by =.ai/protocols.org= presence, so retired dirs (moved under =.retired/=, outside its scan roots) drop out automatically. =inbox-send= targets shrink the same way; any rulesets doc or workflow that names a folded project as a handoff target gets updated in the final phase.
-
-* Alternatives Considered
-
-** Keep separate repos, unify only the agenda (org-agenda-files spanning all todo.orgs)
-- Good, because zero migration risk and Craig's emacs agenda can already span files.
-- Bad, because it solves only prioritization-viewing, not management: 10 sessions, 10 inboxes, 10 memory stores remain; Claude still can't see or rebalance the whole picture in one session; cross-project handoffs persist.
-- Bad, because priority schemes stay divergent per file.
-- Neutral, because it could serve as an interim state, but it builds nothing toward the end goal.
-
-** git subtree add per project
-- Good, because one command per fold, no external tooling.
-- Bad, because history isn't path-rewritten: =git log <name>/file= doesn't follow into pre-merge history without =--follow= gymnastics, weakening the evidentiary value of danneel/jr-estate histories.
-- Neutral, because content-wise the result is identical; only history ergonomics differ.
-
-** Import working trees only, archive old repos (no history merge)
-- Good, because the home repo stays small and the procedure is trivially simple.
-- Bad, because in-place history is lost — every "when did this clause change" question requires resurrecting an archived repo.
-- Neutral, because Layer-1 bares preserve history regardless; this is about whether history is /at hand/.
-
-** One new "life" super-repo instead of growing home
-- Good, because a clean slate avoids home's existing 109M history and infra identity.
-- Bad, because home is already the hub (biggest session history, the template patterns, Craig's habits) and would itself need folding in — strictly more work for a cosmetic gain.
-
-* Decisions
-
-** D1 — Merge strategy: filter-repo + merge per project
-- State: accepted
-- Context: critical legal/financial histories must stay queryable in place; restore must be possible regardless.
-- Decision: We will fold each project with =git filter-repo --to-subdirectory-filter= on a temp clone, merged with =--allow-unrelated-histories=.
-- Consequences: easier — full per-area history in one repo, originals untouched; harder — home =.git= grows to ~1.8G, and =git-filter-repo= becomes a migration dependency (AUR: =git-filter-repo=).
-
-** D2 — Disposition of the code-shaped three
-- State: accepted (Craig, 2026-06-11)
-- Context: website (Hugo codebase), little-elisper (code study), fb-photo-scraper (third-party clone, no .ai) are code-shaped, and the target model says code lives standalone in =~/code/=.
-- Decision: We will move website and little-elisper to =~/code/= as standalone AI projects (plain =mv= + memory-dir rename, same as the homelab→home rename runbook), and delete fb-photo-scraper (it's an unmodified upstream clone — re-cloneable from =https://github.com/budavariam/traverse_facebook_galleries.git=, branch =master=; recorded here so the proof survives the deletion).
-- Consequences: easier — =~/projects/= reaches the clean end state (home + work); harder — website's 11 open tasks stay in their own todo.org, outside the unified prioritization (acceptable: they're code tasks, not life tasks).
-
-** D3 — Unified todo.org: one file, per-area top-level sections
-- State: accepted
-- Context: cross-area prioritization wants one surface; the staleness script, agenda, and review workflows all operate on one file today.
-- Decision: We will keep a single =todo.org= with =* <Area> Open Work= top-level sections mirroring =* Home Open Work=, unified under home's A-D priority scheme, with area tags on every imported task.
-- Consequences: easier — one review rotation, one grep, one agenda file covers everything; harder — the file grows to roughly 5-6k lines (~110 incoming open tasks), so reads lean on Grep/offset and the section discipline matters more.
-
-** D4 — Per-area task triage at fold time
-- State: accepted
-- Context: each project graded priorities against siblings only; merging without re-grading would make cross-area priorities meaningless.
-- Decision: We will walk each incoming area's open tasks with Craig at fold time, re-grading to the unified scheme (the task-review walk shape, applied per area).
-- Consequences: easier — the unified list is trustworthy from day one; harder — the big folds (finances at 43 tasks) cost a real review session each.
-
-** D5 — Cooling-off before any destruction
-- State: accepted
-- Context: "if it goes south, I need a way to restore."
-- Decision: We will destroy nothing for 30 days after the final fold: source dirs go to =~/projects/.retired/=, server bares stay, snapshots and tags persist. fb-photo-scraper deletion (D2) is the one exception — it's an unmodified upstream clone.
-- Consequences: easier — every layer of the restore path stays live through the risky window; harder — ~2G of retired duplicates sit on disk for a month.
-
-* Implementation phases
-
-Each phase ends with a working tree, a pushed commit, and a verification gate. One project per session is the expected pace; phases 3+ are repetitions of the runbook proven in phase 2.
-
-** Phase 0 — Pre-flight (global prep; gates per project at its turn)
-Install =git-filter-repo=. Snapper snapshot + memory-dir tar. Tag and push =pre-consolidation= on home.
-
-The *live inventory gate* lives at =docs/consolidation-manifest-inventory.org= and is regenerated *per project, immediately before that project's fold* — not all at once. Craig keeps using not-yet-folded projects (staged freeze), so a single up-front table would go stale by Phase 3; the per-fold regeneration is the gate. Fields per project: default branch, origin URL, dirty count (=git status --porcelain=), ahead/behind vs upstream, inbox file count, =session-context.org= presence, memory file count, project-workflow/-script names (collision candidates), and disposition (fold / relocate / delete / out-of-scope). The gate passes only when: clean tree, ahead/behind 0/0, inbox empty, no live session-context. A failing project gets fixed (wrap, commit, process, push) and its row regenerated before its fold begins.
-
-D2 is resolved (2026-06-11); no open decisions remain in this phase.
-
-** Phase 1 — Pilot fold: philosophy
-Smallest life project, no todo.org, no memories, empty inbox. Run the full fold runbook (git merge + redistribution + manifest + verification). Then the *rollback drill*: in a disposable =--no-local= clone of home (or a throwaway branch), run the Mode-B removal runbook against the pilot's manifest and verify the folded state is fully gone — sessions, workflows, =<name>/= tree. Discard the clone. The legal and financial folds must never be the first test of the removal story. This phase hardens the runbook appendix; expect to amend the spec from what's learned (history entry, not rewrite).
-
-** Phase 2 — Dress rehearsal: clipper
-Nearly as small (23M, no memories, empty inbox) but adds the one runbook path the pilot can't exercise: the todo.org import — =* Clipper Open Work= section, =:MIGRATED_FROM:= markers, and a one-task triage with Craig. After this phase every runbook step has run at least once except the memory merge (premieres in Phase 3; lowest-risk step — plain file copies outside git, tar-backed).
-
-** Phase 3 — jr-estate
-The priority fold. 704M history, 18-task triage, 44 sessions, and the first memory merge (7 files). One session.
-
-** Phase 4 — danneel
-613M history, 10-task triage, 45 sessions, 1 project-workflow. One session.
-
-** Phase 5 — finances
-251M history and the heaviest triage (43 tasks). One session, possibly two if the triage runs long.
-
-** Phase 6 — Remaining folds + code relocations (together)
-Fold documents, elibrary, health, kit (~36 incoming tasks, elibrary/health/kit memories, kit's 4 project-workflows collision-checked). Move website and little-elisper to =~/code/= (mv + memory-dir rename per the homelab→home runbook); delete fb-photo-scraper (origin recorded in D2). After this phase =~/projects/= contains home, work, and =.retired/=.
-
-** Phase 7 — Ecosystem pass
-Fix every absolute-path reference (emacs config, org-roam, agenda-files, bookmarks, rulesets docs naming folded projects as inbox-send/cross-agent targets). Migrate velox (pull home, retire its clones). Write the KB node recording the new layout. Start the 30-day cooling clock; file a dated vNext task for server bare archival and =.retired/= deletion.
-
-* Acceptance criteria
-
-- [ ] =~/projects/= contains exactly =home=, =work=, and =.retired/=.
-- [ ] For each folded project, =git log --oneline <name>/ | tail= in home shows its earliest original commits.
-- [ ] Manifest parity per fold: every path in the source's =git ls-files= listing exists under =home/<name>/= or appears in the manifest's redistribution map; every untracked/ignored/inbox file has a recorded disposition. Path-list comparison, not counts.
-- [ ] =todo.org= passes org-lint; every imported task carries an area tag and an A-D priority Craig re-graded; imported sections carry =:MIGRATED_FROM:= markers.
-- [ ] =task-review-staleness.sh --list todo.org 20= run after each fold surfaces the imported area's tasks as depth-2 review units alongside existing areas (proves the rotation spans the whole list).
-- [ ] Per fold: source memory basenames diffed against home's memory dir and =MEMORY.md= entries — all accounted for; source memory dirs are in the backup tar.
-- [ ] Link-rewrite check per fold: the before-grep count is recorded in the manifest, and the after-grep over =~/.emacs.d ~/sync/org ~/org/roam ~/projects/home= returns only hits classified historical (session archives, this spec) — zero live links.
-- [ ] Layer-1 restore drill passes: clone one folded project from its untouched server bare into =/tmp=, confirm it's whole.
-- [ ] Mode-B rollback drill (Phase 1 pilot) passes: the pilot fold is fully removable from a disposable clone using only its manifest.
-- [ ] A fresh Claude session in home can answer "what are my top 5 tasks across all areas?" from the unified todo.org.
-- [ ] velox runs a clean session in home post-migration with no stale-remote errors.
-
-* Readiness dimensions
-
-- Data model & ownership: every file keeps its owner (Craig); =.ai/= state redistributes per the merge map above; memory dirs are the one store outside git — covered by the tar in Phase 0 and the copy step per fold.
-- Errors, empty states & failure: every fold step is git-tracked, so a failed fold is =git reset --hard <pre-fold tag>= plus re-running from the temp clone (which is rebuilt from scratch each attempt). filter-repo failures abort before anything touches home.
-- Security & privacy: all content stays on the private cjennings.net remote; no new exposure surface. The merged repo concentrates sensitive material (legal + financial + health) in one clone — same machines, same threat model as today.
-- Observability: each fold is one merge commit + one redistribution commit, tagged; progress is the phase checklist in todo.org; verification gates are the acceptance criteria run per-fold.
-- Performance & scale: ~1.8G final =.git=; clone cost noted. todo.org at 5-6k lines stays well within Grep/offset workflows. No runtime performance surface.
-- Reuse & lost opportunities: reuses the homelab→home rename runbook (memory-dir rename, link sweep), the task-review walk for triage, snapper for snapshots, and the established area-dir pattern. git-filter-repo over hand-rolled rewrites.
-- Architecture fit & weak points: area dirs match the working-files convention. Weak point: the =.ai/= redistribution is manual and per-project — mitigated by the pilot phase hardening a written runbook before the critical folds.
-- Config surface: none — no knobs. The one dependency is the =git-filter-repo= package.
-- Documentation plan: this spec is the migration doc; the fold runbook and manifest template live in the appendix below (refined by the Phase 2 pilot); the KB node in Phase 6 records the end state for all future agents.
-- Dev tooling: N/A because the migration is one-shot; the runbook commands in Design are the tooling.
-- Rollout, compatibility & rollback: staged per-project rollout, multi-machine sequencing (velox last), layered rollback (snapshot / untouched bares / tags / retired dirs), 30-day cooling before any destruction. Dry-run equivalent: the pilot fold.
-- External APIs & deps: =git-filter-repo= (AUR, stable, widely used — verify installed in Phase 0). No network APIs.
-
-* Risks, Rabbit Holes, and Drawbacks
-
-- /Link rot is the long tail./ Absolute =file:= links to old project paths can lurk in org-roam, emacs bookmarks, calendar event descriptions, and Keep notes. The Phase 6 grep covers the file-based stores; Keep and calendar references can't be grepped — accept that stragglers get fixed on encounter.
-- /todo.org scale./ 5-6k lines is fine for tools, but the agenda view gets dense. If it becomes noise, the vNext escape hatch is per-area =#+CATEGORY= or splitting resolved sections to an archive file — not re-splitting projects.
-- /Triage fatigue./ Re-grading ~110 tasks is the human bottleneck. Mitigation: it's split across the fold phases, and each area's walk uses the existing 7-at-a-time review muscle.
-- /Workflow collisions./ 13 project-workflow files across sources; names look distinct but the check is mandatory per fold.
-- /The merged repo is a bigger blast radius./ A bad force-push or corrupting operation now touches everything. Mitigation: the same layered backups, plus home already carries this responsibility for its own content.
-- /Drawback accepted:/ per-project session isolation disappears — one project's noisy session history now shares a dir with everything. The area prefix on archived session names keeps provenance.
-
-* Appendix — Fold runbook (per project)
-
-Refined by the Phase 2 pilot; until then this is the v0 contract. Every step either succeeds with the expected output or the fold stops — no improvising past a failed step.
-
-1. /Gate./ Regenerate the project's row in the live inventory (Phase 0 fields). Require: clean =git status --porcelain=, ahead/behind 0/0, inbox empty, no =session-context.org=. Fix and regenerate, or stop.
-2. /Manifest open./ Create =docs/consolidation-manifest-<name>.org= from the template below; fill source state and the =git ls-files= listing; inventory untracked/ignored files with dispositions. The Redistribution row reads "the commit introducing this manifest" — a commit sha can't be embedded in its own commit (pilot learning, 2026-06-11); the merge sha is known beforehand and is recorded literally.
-3. /Tag./ =git tag pre-fold-<name> && git push origin pre-fold-<name>=.
-4. /History fold./ The filter-repo + merge block from Design (with =--no-local=, =$branch=, no =--force=). Record the merge sha in the manifest.
-5. /Redistribute./ Steps 1-8 of the =.ai/= and task merge map, recording each move in the manifest's redistribution map as it happens. One commit; record its sha.
-6. /Verify./ Manifest parity (path lists), org-lint on todo.org, staleness-script check, memory diff, link before/after grep. All green or the fold stops here for repair.
-7. /Retire./ =mv ~/projects/<name> ~/projects/.retired/<name>=; record the path. Push home.
-
-** Manifest template
-
-#+begin_example
-,#+TITLE: Consolidation manifest — <name>
-| Source path | ~/projects/<name> |
-| Origin | <url> |
-| Branch / HEAD | <branch> / <sha> |
-| Gate state | clean / 0-0 / inbox 0 / no session-context |
-| Merge commit | <sha> |
-| Redistribution | <sha> |
-| Retired to | ~/projects/.retired/<name> |
-
-,* Tracked universe
-<git ls-files output, or sha256 + appendix>
-
-,* Untracked / ignored / inbox dispositions
-| file | disposition (processed / moved-to / archived / dropped) |
-
-,* Redistribution map
-- Sessions: <old> → <new> …
-- todo.org: section ":MIGRATED_FROM: <name>" added at <heading>
-- Workflows/scripts copied: <names + collision resolutions>
-- Memory: <files copied> + MEMORY.md lines added
-- Link rewrites: <file:line before → after>
-
-,* Link grep
-- Before: <count> | After: <count, all classified historical>
-#+end_example
-
-* Review dispositions
-
-Findings from the 2026-06-11 Codex review (review file deleted on processing per spec-response). Modified items below; *everything else was accepted as written* — H1 (manifest + redistribution-aware restore), H2 (path-list verification, porcelain gate, find-prune fix by removal), H3 (=--no-local=, no =--force=), H4 (live inventory gate), M2 (pilot rollback drill), M3 (fb-photo-scraper re-clone pointer), the UX provenance markers, the documentation appendix, and the memory/link verification commands.
-
-- /M1 (memory tar metadata) — modified:/ the review offered preserving metadata or declaring content-only sufficient. Chose content-only: memory files are plain-text markdown the harness reads by path; no permissions/ACL/xattr metadata is load-bearing. Stated in Layer 0 rather than adding =--xattrs --acls=.
-- /Test strategy item 2 (staleness fixture test) — modified:/ a live =task-review-staleness.sh --list todo.org 20= check after each fold replaces a new fixture-based unit test. The script already has its own bats suite; the migration-specific question ("are imported area tasks depth-2 review units?") is answered better by the live check on the real file, per fold, than by a one-shot fixture.
-- /Open question 2 (revert vs manifest-removal as the supported runbook) — modified:/ the "choose one" framing doesn't survive the time axis. Supported path: revert both commits (redistribution first) while the fold is recent; once shared files have moved on and reverts conflict, the manifest-driven removal commit is the path. Both are now written in Restore Mode B; the manifest makes the fallback safe, which is why it exists.
-
-* Review and iteration history
-
-** 2026-06-11 Thu @ 15:11:32 -0500 — Claude Code (home, with Craig) — author
-- What changed: re-sequenced the implementation phases to Craig's chosen order — philosophy pilot, clipper dress rehearsal (added as its own phase so the todo.org-import path is proven before real data), then jr-estate → danneel → finances by urgency, with the remaining folds + code relocations merged into one closing phase before the ecosystem pass. Phase 0's live inventory gate is now explicitly per-project-at-its-turn, matching the staged-freeze approach (Craig keeps using not-yet-folded projects until their turn).
-- Why: Craig wants the critical legal/financial projects consolidated early after a proven runbook, and chose staged freeze over a full shutdown. The clipper rehearsal closes the gap where the pilot (no todo.org) never exercises task import.
-- Artifacts: todo.org phase tasks re-sequenced to match.
-
-** 2026-06-11 Thu @ 14:13:49 -0500 — Codex — reviewer
-- What changed or was recommended: assigned =Ready= after re-running spec-review against the incorporated spec; no further blocking review notes and no new review file.
-- Why: the prior blockers are now covered by the per-fold manifest contract, live inventory gate, =--no-local= filter-repo runbook, redistribution-aware restore, Phase 2 rollback drill, and manifest-based acceptance criteria.
-- Artifacts: this spec; [[file:../todo.org::*Project consolidation into home][todo.org tracking task]] updated with Ready status and implementation phase tasks.
-
-** 2026-06-11 Thu @ 13:20:54 -0500 — Claude Code (home) — responder
-- What changed: all four blocking findings accepted and woven in — =--no-local= + no-=--force= runbook with a =$branch= variable (H3, H4's master/main catch), a Per-fold manifest section as the restore/verification contract (H1, H2), two-mode single-project restore covering the redistribution commit (H1), Phase 0 live inventory gate (H4), Phase 2 Mode-B rollback drill (M2), manifest-parity acceptance criteria replacing file counts (H2), fb-photo-scraper re-clone pointer in D2 (M3), =:MIGRATED_FROM:= provenance markers (UX), runbook appendix + manifest template (docs). Three points modified with reasons in Review dispositions; nothing rejected.
-- Why: the review's core finding was right — restore and verification only covered the history merge, not the redistribution commit and the untracked-file universe. The manifest is the single artifact that fixes both.
-- Artifacts: review file (deleted on processing); dispositions section above; todo.org tracking task updated.
-
-** 2026-06-11 Thu @ 13:09:29 -0500 — Codex — reviewer
-- What changed or was recommended: assigned =Not ready= and wrote a blocking review focused on exact rollback semantics, per-fold manifests, live inventory gating, =git clone --no-local= for filter-repo safety, and stronger verification than raw file counts.
-- Why: the design direction is sound, but implementation would still require inventing how to unwind redistributed =.ai/=, =todo.org=, inbox, and memory state safely after each fold.
-- Artifacts: review file deleted during the response pass; retained via the dispositions section above.
-
-** 2026-06-11 Thu @ 12:08:03 -0500 — Claude (with Craig) — author
-- What: initial draft.
-- Why: Craig asked for a consolidation design with restore guarantees — one prioritization surface for all personal projects.
-- Artifacts: survey data gathered live from ~/projects on ratio; todo.org task cross-linked.
diff --git a/inbox/PROCESSED-2026-06-11-1705-from-home-addendum-to-today-s-consolidation.org b/inbox/PROCESSED-2026-06-11-1705-from-home-addendum-to-today-s-consolidation.org
deleted file mode 100644
index 392b844..0000000
--- a/inbox/PROCESSED-2026-06-11-1705-from-home-addendum-to-today-s-consolidation.org
+++ /dev/null
@@ -1,5 +0,0 @@
-#+TITLE: Addendum to today's consolidation handoff: first concrete to
-#+SOURCE: from home
-#+DATE: 2026-06-11 17:05:20 -0500
-
-Addendum to today's consolidation handoff: first concrete tooling edge case found. todo-cleanup.el --archive-done assumes exactly one level-1 'Open Work' and one 'Resolved' heading per todo.org; home's consolidated file now has per-area pairs (Home Open Work / Home Resolved, Clipper Open Work / Clipper Resolved, more coming) and the pass skips with 'more than one level-1 heading contains Open Work'. Suggested fix: match each '* <Area> Open Work' with its '* <Area> Resolved' sibling and archive within the pair, falling back to current behavior for single-pair files. Until then home archives manually at wrap-up.
diff --git a/inbox/PROCESSED-2026-06-11-1755-from-work-from-the-work-project-2026-06-11-craig.org b/inbox/PROCESSED-2026-06-11-1755-from-work-from-the-work-project-2026-06-11-craig.org
deleted file mode 100644
index cbd8241..0000000
--- a/inbox/PROCESSED-2026-06-11-1755-from-work-from-the-work-project-2026-06-11-craig.org
+++ /dev/null
@@ -1,7 +0,0 @@
-#+TITLE: From the work project, 2026-06-11: Craig's guidance on triag
-#+SOURCE: from work
-#+DATE: 2026-06-11 17:55:46 -0500
-
-From the work project, 2026-06-11: Craig's guidance on triage-intake reporting, for the canonical triage-intake.org engine's Render/summary section. Sweep summaries should report DELTAS ONLY: a new invite, a new/moved/cancelled calendar event, a new message needing attention. A sweep where nothing changed renders as one line (e.g. '17:39 sweep: no changes'), never a per-source 'quiet' roll-call. His words: 'we only need to report if anything's changed when we do triage intake. did someone send me a new invite? did christine throw something on my calendar that wasn't there earlier? did someone cancel a meeting?' Failures still surface loudly per the existing engine rule (never folded into the no-change line), and the suggested-actions queue line stays. The work project is applying this immediately; please fold into the canonical engine so all projects pick it up on template sync.
-
-Addendum (same day, 17:55 CDT): Craig also ruled that Telegram dev-community group traffic (zed, GNU Emacs, Kitty, etc.) is skipped in sweep reports entirely — not even the FYI name+count line the telegram plugin's Render currently specifies — unless he specifically asks. Real DMs from known contacts still surface as Action. Please update triage-intake.telegram.org's Render section accordingly.
diff --git a/inbox/PROCESSED-2026-06-11-1823-from-.emacs.d-memory-sweep-phase-1-5-complete-for.org b/inbox/PROCESSED-2026-06-11-1823-from-.emacs.d-memory-sweep-phase-1-5-complete-for.org
deleted file mode 100644
index b7ea37c..0000000
--- a/inbox/PROCESSED-2026-06-11-1823-from-.emacs.d-memory-sweep-phase-1-5-complete-for.org
+++ /dev/null
@@ -1,5 +0,0 @@
-#+TITLE: Memory sweep (Phase 1.5) complete for dotemacs (~/.emacs.d),
-#+SOURCE: from .emacs.d
-#+DATE: 2026-06-11 18:23:38 -0500
-
-Memory sweep (Phase 1.5) complete for dotemacs (~/.emacs.d), 2026-06-11. Inventory: 7 memories. Promoted to KB: 3 (no-make-frame-in-live-daemon, proton-bridge-headless-cert-mismatch, open-images-with-imv — pushed in roam commit a915760). Stayed local: 3 (commit-flow-no-approval-gate — Craig scoped it per-project, explicitly not KB; color-sorting-research and theme-design-render-format — theme-scoped, held local at Craig's call, with the palette-columns spec possibly superseding LCCL soon). Deleted: 1 (numbered-options-with-recommendation-first — superseded by the canonical interaction.md rule).
diff --git a/inbox/PROCESSED-2026-06-11-1909-from-home-inbox-response-consolidation-and-todo.org b/inbox/PROCESSED-2026-06-11-1909-from-home-inbox-response-consolidation-and-todo.org
deleted file mode 100644
index e920dfd..0000000
--- a/inbox/PROCESSED-2026-06-11-1909-from-home-inbox-response-consolidation-and-todo.org
+++ /dev/null
@@ -1,25 +0,0 @@
-#+TITLE: Home reply: folds resuming + todo.org shape decided — single pair, kill the multi-pair fix
-#+DATE: 2026-06-11
-
-Replying to both of your 2026-06-11 handoffs (consolidation okay 1708, todo-shape heads-up 1716).
-
-* Folds resuming
-
-Confirmed received — the HOLD is lifted and Phase 3 (jr-estate) is clear to start. Thanks for the tooling sweep and the Phase 7 reference list; the list is folded into our ecosystem-pass task so the link sweep cross-checks your fixes.
-
-On your ask: the session archives are already covered by design. Redistribution step 1 merges each source's .ai/sessions/ into home's .ai/sessions/ with the area name inserted into the filename, and both pilots did exactly that (philosophy 5 files, clipper 6). Nothing for session-harvest to lose — it'll see the history as home's, area-prefixed.
-
-* todo.org shape: single pair wins
-
-Craig confirmed the single-pair shape in tonight's home session, and the reshape is already done — only clipper's import was in (an empty open section plus 3 resolved entries), so it cost a few minutes now versus a real migration after jr-estate and finances.
-
-What landed on our side:
-
-- todo.org holds one Home Open Work / Home Resolved pair. Clipper's imported tasks moved under Home Resolved, each carrying its own :MIGRATED_FROM: clipper / :MIGRATED_ON: drawer plus a :clipper: tag. The per-area level-1 sections are dissolved.
-- Spec amended: D3 (decision + amendment note), task-merge step 2 (append under the home pair, per-task provenance), restore Mode B path 2 (excision is a targeted :MIGRATED_FROM: property sweep; the manifest lists imported headings individually).
-- The clipper manifest is amended with the reshape and the individual imported headings, so its Mode-B contract stays honest.
-
-* What this decides for you
-
-- Kill the todo-cleanup.el multi-pair archive [#B] — the single-pair file works with the existing tooling unmodified, which was half the argument for the shape.
-- No staleness-pool changes needed. Imported tasks join the unified A-C pool on their own merits once re-graded at fold-time triage; area tags are just tags. The existing convention stands: [#D] imports sit outside the A-C pool by design.
diff --git a/inbox/PROCESSED-2026-06-11-1951-from-home-inbox-response-jr-estate-memory-sweep.org b/inbox/PROCESSED-2026-06-11-1951-from-home-inbox-response-jr-estate-memory-sweep.org
deleted file mode 100644
index ebaeff0..0000000
--- a/inbox/PROCESSED-2026-06-11-1951-from-home-inbox-response-jr-estate-memory-sweep.org
+++ /dev/null
@@ -1,12 +0,0 @@
-#+TITLE: jr-estate memory sweep complete — 2 promoted / 3 kept / 2 deleted (via the home fold)
-#+DATE: 2026-06-11
-
-Answering your 2026-06-10 migrate-memories handoff to jr-estate. The sweep ran inside jr-estate's fold into home (Phase 3 of the consolidation, completed tonight), since the fold's memory-merge step is the same walk.
-
-Counts, Craig-approved at fold-time triage:
-
-- Promoted 2 to ~/org/roam/agents/ (commit 45d8e6c, pushed): the forms name-with-number preference (always pair a form's id with its full name) and the PDF-editing tooling split (Xournal++ for Craig, pdftools-venv overlay edits for Claude, signatures always through Craig).
-- Kept 3 local, now in home's memory dir with jr-estate attribution: aj-fudge (who AJ is), bond-waiver-conditions, chevron-stock-holding — estate-scoped facts.
-- Deleted 2: default-email-cmail (rule-encoded in protocols.org's email table) and feedback-no-same-day-scheduling (duplicate of home's existing no-scheduling-for-today memory).
-
-jr-estate is now a home area; its future durable facts flow through home's capture-then-promote discipline. Its 44 session archives merged into home's .ai/sessions/ area-prefixed, so session-harvest's first run will see them.
diff --git a/inbox/PROCESSED-2026-06-11-2154-from-home-inbox-response-finances-memory-sweep.org b/inbox/PROCESSED-2026-06-11-2154-from-home-inbox-response-finances-memory-sweep.org
deleted file mode 100644
index a3bed46..0000000
--- a/inbox/PROCESSED-2026-06-11-2154-from-home-inbox-response-finances-memory-sweep.org
+++ /dev/null
@@ -1,8 +0,0 @@
-#+TITLE: finances memory sweep complete — 0 promoted / 1 kept / 0 deleted (via the home fold)
-#+DATE: 2026-06-11
-
-Answering your 2026-06-10 migrate-memories handoff to finances. The sweep ran inside finances' fold into home (Phase 5 of the consolidation).
-
-Counts: promoted 0; kept 1 local in home's memory dir with finances attribution (rosalea-daly-passed — contact guidance scoped to the Strata Trust SDIRA workstream, no cross-project value); deleted 0.
-
-finances is now a home area; its 24 session archives merged into home's .ai/sessions/ area-prefixed for session-harvest.
diff --git a/inbox/PROCESSED-2026-06-11-2308-from-home-lint-org-el-false-positive-mu4e-msgid.org b/inbox/PROCESSED-2026-06-11-2308-from-home-lint-org-el-false-positive-mu4e-msgid.org
deleted file mode 100644
index 585ae4d..0000000
--- a/inbox/PROCESSED-2026-06-11-2308-from-home-lint-org-el-false-positive-mu4e-msgid.org
+++ /dev/null
@@ -1,5 +0,0 @@
-#+TITLE: lint-org.el false positive: mu4e:msgid: links flag as invali
-#+SOURCE: from home
-#+DATE: 2026-06-11 23:08:59 -0500
-
-lint-org.el false positive: mu4e:msgid: links flag as invalid-fuzzy-link in batch runs. The mu4e link type is registered by mu4e at runtime in a live Emacs, so batch org-lint parses [[mu4e:msgid:...]] as a fuzzy heading ref and reports 'Unknown fuzzy location'. Eight such links in home's todo.org survived a full lint pass tonight as the only remaining judgment items — all work fine interactively. Suggested fix in lint-org.el: register the link type as a no-op before linting, e.g. (org-link-set-parameters "mu4e"), or add a suppressed-categories entry for invalid-fuzzy-link items whose target starts with a known runtime link prefix (mu4e:, possibly others like attachment:). Same pattern as the existing verbatim-asterisk suppression.
diff --git a/inbox/PROCESSED-2026-06-12-0101-from-.emacs.d-page-signal-is-broken-the-dedicated.org b/inbox/PROCESSED-2026-06-12-0101-from-.emacs.d-page-signal-is-broken-the-dedicated.org
deleted file mode 100644
index 30a680c..0000000
--- a/inbox/PROCESSED-2026-06-12-0101-from-.emacs.d-page-signal-is-broken-the-dedicated.org
+++ /dev/null
@@ -1,5 +0,0 @@
-#+TITLE: page-signal is broken: the dedicated pager account (+1504517
-#+SOURCE: from .emacs.d
-#+DATE: 2026-06-12 01:01:58 -0500
-
-page-signal is broken: the dedicated pager account (+15045173983, the Claude Pager Google Voice number registered with signal-cli) reports 'User ... is not registered' on every send, including with explicit --to. Signal appears to have deregistered the account (GV numbers get periodically re-verified). Re-registration needs Craig (captcha/SMS). Discovered 2026-06-12 when the dotemacs config-audit completion page failed; fallback used email. Wrapper: claude-templates/bin/page-signal.
diff --git a/inbox/PROCESSED-2026-06-12-0207-from-home-memory-sweep-reply-for-the-2026-06-10.org b/inbox/PROCESSED-2026-06-12-0207-from-home-memory-sweep-reply-for-the-2026-06-10.org
deleted file mode 100644
index 72e459b..0000000
--- a/inbox/PROCESSED-2026-06-12-0207-from-home-memory-sweep-reply-for-the-2026-06-10.org
+++ /dev/null
@@ -1,5 +0,0 @@
-#+TITLE: Memory-sweep reply for the 2026-06-10 migrate-memories hando
-#+SOURCE: from home
-#+DATE: 2026-06-12 02:07:49 -0500
-
-Memory-sweep reply for the 2026-06-10 migrate-memories handoff, covering elibrary, health, and kit (all three were folded into home as areas on 2026-06-11, so the sweep ran at fold time with Craig's approval; counts are promoted / kept local / deleted). elibrary: 0 / 0 / 2 — private-remote fact duplicates home's git-hosting-privacy-model memory; project-scripts convention is encoded in startup.org. health: 0 / 0 / 1 — scheduling feedback duplicates home's no-scheduling-for-today memory. kit: 1 / 0 / 2 — feedback-hand-prep-items-to-work-inbox promoted into home's memory (operative for home's wrap-up extension); no-default-today-scheduling duplicates the same home memory; no-emphasis-formatting-in-prose is rule-encoded in /voice prose mode. Nothing went to the org-roam KB — no swept fact met the durable cross-project bar that wasn't already encoded in rules or home memory. All source files preserved in the pre-consolidation memory tar. The home, documents, and remaining-area sweeps are covered by home's own session discipline going forward.
diff --git a/inbox/lint-followups.org b/inbox/lint-followups.org
new file mode 100644
index 0000000..9d2bd8f
--- /dev/null
+++ b/inbox/lint-followups.org
@@ -0,0 +1,18 @@
+* 2026-07-20 Mon — Task-review health: 1 top-level [#A]/[#B]/[#C] tasks unreviewed for >30 days (daily review may have slipped)
+
+* lint-org follow-ups — todo.org (2026-07-29)
+** TODO misplaced-heading — Possibly misplaced heading line (line 2263)
+** TODO link-to-local-file — Link to non-existent local file "working/hook-fail-open/validate-el.diff" (line 2239)
+** TODO misplaced-planning-info — Misplaced planning info line (line 2228)
+** TODO link-to-local-file — Link to non-existent local file "working/hook-fail-open/pre-commit.diff" (line 2223)
+** TODO misplaced-planning-info — Misplaced planning info line (line 2208)
+** TODO org-table-standard — table violates the org-table standard: no closing rule; missing rule between rows — wrap-org-table.el reflows it (line 314)
+** TODO task-missing-last-reviewed — task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed (line 472)
+** TODO task-missing-last-reviewed — task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed (line 475)
+** TODO task-missing-last-reviewed — task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed (line 482)
+** TODO task-missing-last-reviewed — task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed (line 490)
+** TODO task-missing-last-reviewed — task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed (line 493)
+** TODO task-missing-last-reviewed — task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed (line 502)
+** TODO task-missing-last-reviewed — task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed (line 600)
+** TODO task-missing-last-reviewed — task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed (line 759)
+** TODO task-missing-last-reviewed — task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed (line 768)
diff --git a/languages/bash/CLAUDE.md b/languages/bash/CLAUDE.md
new file mode 100644
index 0000000..2511c47
--- /dev/null
+++ b/languages/bash/CLAUDE.md
@@ -0,0 +1,71 @@
+# CLAUDE.md
+
+## Project
+
+Bash/shell project. Customize this section with your own description, layout,
+and conventions.
+
+**Typical layout:**
+- `bin/` or top-level `*.sh` — entry-point scripts
+- `lib/*.sh` — sourced function libraries (no `set -e`; the caller owns the shell)
+- `tests/*.bats` — bats-core tests beside the scripts they exercise
+
+## Build & Test Commands
+
+If the project has a Makefile, document targets here. Common pattern:
+
+```bash
+make test # run the bats suite
+make test FILE=tests/x.bats # one file
+make lint # shellcheck across the tree
+make fmt # shfmt -w (if the project adopts shfmt)
+```
+
+Direct equivalents: `bats -r tests/`, `shellcheck script.sh`,
+`shfmt -d script.sh` (diff), `shfmt -w script.sh` (write).
+
+## Language Rules
+
+See rule files in `.claude/rules/`:
+- `bash.md` — code style and patterns (strict mode, quoting, `[[ ]]`, traps)
+- `bash-testing.md` — bats conventions
+- `verification.md` — verify-before-claim-done discipline
+
+## Git Workflow
+
+Commit conventions: see `.claude/rules/commits.md` (author identity,
+no AI attribution, message format).
+
+Pre-commit hook in `githooks/` scans for secrets and runs `shellcheck` on staged
+shell files. Activate on a fresh clone with `git config core.hooksPath githooks`.
+
+## Problem-Solving Approach
+
+Investigate before fixing. When diagnosing a bug:
+1. Read the relevant script and trace what actually happens
+2. Identify the root cause, not a surface symptom
+3. Write a failing bats test that captures the correct behavior
+4. Fix, then re-run tests
+
+## Testing Discipline
+
+TDD is the default: write a failing test before any implementation. If you can't
+write the test, you don't yet understand the change. Details in
+`.claude/rules/bash-testing.md`.
+
+## Editing Discipline
+
+A PostToolUse hook runs `shellcheck` on every shell file after Edit/Write/
+MultiEdit and blocks on a violation — read the SCxxxx code and fix it (each has a
+wiki page). The hook covers `.sh`, `.bash`, and extensionless files with a shell
+shebang. Formatting (`shfmt`) is recommended but not enforced by the hook, since
+shell has no single canonical style; adopt one per project via `.editorconfig`.
+
+## What Not to Do
+
+- Don't add features beyond what was asked
+- Don't refactor surrounding code when fixing a bug
+- Don't leave expansions unquoted or use `[ ]` where `[[ ]]` fits
+- Don't add comments to code you didn't change
+- Don't commit `.env` files, credentials, or API keys — the pre-commit hook
+ catches common patterns but isn't a substitute for care
diff --git a/languages/bash/claude/hooks/validate-bash.sh b/languages/bash/claude/hooks/validate-bash.sh
new file mode 100755
index 0000000..4e75f40
--- /dev/null
+++ b/languages/bash/claude/hooks/validate-bash.sh
@@ -0,0 +1,66 @@
+#!/usr/bin/env bash
+# Validate shell files after Edit/Write/MultiEdit.
+# PostToolUse hook: receives tool-call JSON on stdin.
+#
+# On success: exit 0 silent.
+# On failure: emit JSON with hookSpecificOutput.additionalContext so Claude
+# sees a structured error in its context, THEN exit 2 to block the tool
+# pipeline. stderr still echoes the error for terminal visibility.
+#
+# Gate: shellcheck. It catches the bugs that define shell — unquoted
+# expansions, unset variables, masked exit codes — and is the high-value,
+# universally-agreed check. Formatting (shfmt) is deliberately NOT enforced
+# here: shell has no single canonical style (tabs vs spaces), so blocking on
+# it would impose a contested choice. bash.md recommends shfmt; this hook
+# enforces correctness.
+#
+# Scope: .sh and .bash files, plus extensionless files whose first line is a
+# sh/bash shebang (the CLI tools that fill a shell-heavy repo carry no
+# extension).
+
+set -u
+
+# Emit a JSON failure payload and exit 2. Arguments:
+# $1 — short failure type (e.g. "SHELLCHECK FAILED")
+# $2 — file path
+# $3 — tool output (error body)
+fail_json() {
+ local ctx
+ ctx="$(printf '%s: %s\n\n%s\n\nFix before proceeding.' "$1" "$2" "$3" \
+ | jq -Rs .)"
+ cat <<EOF
+{"hookSpecificOutput": {"hookEventName": "PostToolUse", "additionalContext": $ctx}}
+EOF
+ printf '%s: %s\n%s\n' "$1" "$2" "$3" >&2
+ exit 2
+}
+
+f="$(jq -r '.tool_input.file_path // .tool_response.filePath // empty')"
+[ -z "$f" ] && exit 0
+[ -f "$f" ] || exit 0
+
+# Is this a shell file? By extension, or by shebang when it has no extension.
+# Match on the basename, not the full path — a temp/parent dir can carry a dot
+# (e.g. validate-bash-bats.XXXX/) and misfire the "*.*" extension test.
+is_shell=0
+base="${f##*/}"
+case "$base" in
+ *.sh | *.bash) is_shell=1 ;;
+ *.*) is_shell=0 ;; # some other extension — not ours
+ *)
+ # No extension: sniff the shebang.
+ if head -1 "$f" 2>/dev/null | grep -qE '^#!.*\b(bash|sh)\b'; then
+ is_shell=1
+ fi
+ ;;
+esac
+[ "$is_shell" -eq 1 ] || exit 0
+
+# No shellcheck on this machine — nothing to validate, don't block the edit.
+command -v shellcheck >/dev/null 2>&1 || exit 0
+
+if ! out="$(shellcheck "$f" 2>&1)"; then
+ fail_json "SHELLCHECK FAILED" "$f" "$out"
+fi
+
+exit 0
diff --git a/languages/bash/claude/rules/bash-testing.md b/languages/bash/claude/rules/bash-testing.md
new file mode 100644
index 0000000..c904927
--- /dev/null
+++ b/languages/bash/claude/rules/bash-testing.md
@@ -0,0 +1,71 @@
+# Bash Testing Rules
+
+Applies to: `**/*.bats`
+
+Implements the core principles from `testing.md`. All rules there apply here —
+this file covers shell-specific patterns.
+
+## Framework: bats-core
+
+Use [bats-core](https://bats-core.readthedocs.io/) for shell tests. A test file
+is `<thing>.bats`; each test is a `@test "description" { ... }` block; a non-zero
+exit inside the block fails the test. Run a file with `bats path/to/file.bats`,
+or a tree with `bats -r tests/`.
+
+Drive the script under test with `run`: it captures `$status` (exit code),
+`$output` (combined stdout+stderr), and `$lines[]` (output split by line)
+without the failure aborting the test. Assert on those.
+
+```bash
+@test "greet: prints the name passed in" {
+ run bash "$SCRIPT" --name Ada
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"Hello, Ada"* ]]
+}
+```
+
+## Test the Real Script, Through Its Interface
+
+Run the actual script file — never copy its logic into the test. Invoke it the
+way a caller does (`run bash "$SCRIPT" <args>`, or `run "$SCRIPT"` when it's
+executable) and assert on exit status and output. A test that re-implements the
+script's logic passes even when the script breaks.
+
+For a script that sources a library of functions, source the library in `setup`
+and call the functions directly — that's the unit level; the `run` invocation is
+the integration level.
+
+## Normal, Boundary, Error — the Three Categories
+
+Cover all three from `testing.md` per script:
+
+- Normal: the expected arguments and inputs produce the expected output and a
+ zero exit.
+- Boundary: empty argument, missing optional flag, single-item vs many,
+ whitespace and unicode in inputs, a path with a space.
+- Error: missing required argument, nonexistent input file, a dependency
+ absent. Assert the exit code and that the error names the problem — not the
+ exact wording (`testing.md`'s error-behavior rule).
+
+## Isolation and Determinism
+
+- `setup()` makes a fresh `mktemp -d` per test; `teardown()` removes it. No test
+ leans on another's leftovers, and tests pass in any order.
+- Mock an external command by putting a stub earlier on `PATH`: write a small
+ script named like the command into a temp dir, `chmod +x`, and prepend that
+ dir to `PATH` for the `run`. This is how you simulate a tool being absent,
+ returning an error, or emitting canned output — without touching the network
+ or the real tool.
+- Never hardcode dates; generate them relative to `date` (see the
+ `task-review-staleness.bats` pattern in this repo for relative-date fixtures).
+- Mock at the boundary (network, the external CLI, the clock). Don't mock the
+ script's own functions — those are the work.
+
+## What Not to Do
+
+- Don't assert exact error-message prose; assert the exit code plus a value the
+ message must contain.
+- Don't share mutable state between tests through a fixed temp path.
+- Don't test that `shellcheck` or `bats` themselves work — trust the tools.
+- Don't skip the error cases because the happy path passes; the error paths are
+ where shell scripts actually break.
diff --git a/languages/bash/claude/rules/bash.md b/languages/bash/claude/rules/bash.md
new file mode 100644
index 0000000..042138a
--- /dev/null
+++ b/languages/bash/claude/rules/bash.md
@@ -0,0 +1,83 @@
+# Bash Code Rules
+
+Applies to: `**/*.sh`, `**/*.bash`, and extensionless files with a `sh`/`bash` shebang
+
+Shell-specific style and structure. Pairs with `bash-testing.md` for tests and
+the generic `verification.md` / `commits.md` rules. When in doubt, defer to
+[ShellCheck](https://www.shellcheck.net/wiki/) (every SCxxxx code has a wiki
+page explaining the fix) and Google's
+[Shell Style Guide](https://google.github.io/styleguide/shellguide.html).
+
+## ShellCheck Is the Gate, Not a Suggestion
+
+The bundle's PostToolUse hook runs `shellcheck` on every edited shell file and
+blocks on a violation; the pre-commit hook re-checks staged files. ShellCheck
+catches the bugs that define shell: unquoted expansions that word-split, unset
+variables, `[ ]` pitfalls, masked exit codes. Fix the finding rather than
+silence it. When a warning is a genuine false positive, disable it narrowly with
+a `# shellcheck disable=SCxxxx` directive on the line above and a comment saying
+why, never a file-wide blanket disable.
+
+## The Header: Strict Mode
+
+Every script starts with `#!/usr/bin/env bash` and `set -euo pipefail`:
+
+- `-e` exits on an unhandled non-zero command. Handle the expected-failure cases
+ explicitly (`cmd || true`, an `if`, a `case`) so the exit is a real error.
+- `-u` treats an unset variable as an error. Use `"${VAR:-default}"` for the
+ ones that are legitimately optional.
+- `-o pipefail` makes a pipeline fail if any stage fails, not just the last.
+
+A script meant to be *sourced* (a library) skips `set -e` — it would change the
+caller's shell. Libraries guard their own commands instead.
+
+## Quote Everything
+
+- Double-quote every expansion: `"$var"`, `"$@"`, `"${arr[@]}"`,
+ `"$(command)"`. Unquoted is the single largest source of shell bugs — a path
+ with a space becomes two arguments.
+- `"$@"` (quoted) passes arguments through untouched; `$*` and unquoted `$@`
+ word-split. Use `"$@"` unless you specifically want the joined string.
+- Loop over arrays and `find -print0 | while IFS= read -r -d ''`, never over
+ unquoted command substitution or `ls` output.
+
+## Test, Compare, Branch
+
+- Use `[[ ]]` for tests, not `[ ]` / `test`. `[[ ]]` doesn't word-split its
+ operands, supports `&&`/`||`/`=~`, and has fewer quoting traps.
+- Arithmetic goes in `(( ))` or `$(( ))`, not `[ ]` with `-eq`.
+- `$(command)`, never backticks — nests cleanly and reads better.
+- Prefer `printf` over `echo` for anything but a fixed literal string;
+ `echo` mangles values that start with `-` or contain backslashes.
+
+## Functions and Scope
+
+- Declare function-local variables with `local`. A bare assignment in a
+ function writes a global and leaks across calls.
+- `local var; var="$(cmd)"` on two lines when you need the command's exit
+ status: `local var="$(cmd)"` masks `cmd`'s exit code behind `local`'s.
+- Keep functions focused. A function that fetches, parses, and writes is three
+ functions; the test difficulty in `bash-testing.md` is the tell.
+- Put `main "$@"` at the bottom for a script with more than a couple of
+ functions, so definition order doesn't dictate execution order.
+
+## Robustness
+
+- `trap 'rm -rf "$tmpdir"' EXIT` right after creating a temp resource, so
+ cleanup runs on every exit path including errors.
+- Make a temp file or dir with `mktemp` / `mktemp -d`, never a fixed
+ `/tmp/name` (race + collision).
+- Check that a required command exists before the work: `command -v jq
+ >/dev/null || { echo "jq required" >&2; exit 1; }`.
+- Never parse `ls` output and don't `cat` a file into a pipe you could read
+ directly. Glob, or use `find`, or read the file in place.
+
+## What Not to Do
+
+- Don't leave an expansion unquoted to "save a quote" — quote it.
+- Don't use `[ ]` when `[[ ]]` is available, or backticks when `$()` is.
+- Don't silence a ShellCheck warning file-wide to clear it; fix it or disable
+ the one code with a reason.
+- Don't refactor surrounding code while fixing a bug — keep the diff scoped.
+- Don't commit credentials or API keys — the pre-commit hook catches common
+ patterns but isn't a substitute for care.
diff --git a/languages/bash/claude/settings.json b/languages/bash/claude/settings.json
new file mode 100644
index 0000000..b725603
--- /dev/null
+++ b/languages/bash/claude/settings.json
@@ -0,0 +1,68 @@
+{
+ "attribution": {
+ "commit": "",
+ "pr": ""
+ },
+ "permissions": {
+ "allow": [
+ "Bash(make)",
+ "Bash(make help)",
+ "Bash(make targets)",
+ "Bash(make test)",
+ "Bash(make test *)",
+ "Bash(make lint)",
+ "Bash(make fmt)",
+ "Bash(shellcheck *)",
+ "Bash(shfmt *)",
+ "Bash(bats)",
+ "Bash(bats *)",
+ "Bash(git status)",
+ "Bash(git status *)",
+ "Bash(git diff)",
+ "Bash(git diff *)",
+ "Bash(git log)",
+ "Bash(git log *)",
+ "Bash(git show)",
+ "Bash(git show *)",
+ "Bash(git blame *)",
+ "Bash(git branch)",
+ "Bash(git branch -v)",
+ "Bash(git branch -a)",
+ "Bash(git branch --list *)",
+ "Bash(git remote)",
+ "Bash(git remote -v)",
+ "Bash(git remote show *)",
+ "Bash(git ls-files *)",
+ "Bash(git rev-parse *)",
+ "Bash(git cat-file *)",
+ "Bash(git stash list)",
+ "Bash(git stash show *)",
+ "Bash(jq *)",
+ "Bash(date)",
+ "Bash(date *)",
+ "Bash(which *)",
+ "Bash(file *)",
+ "Bash(ls)",
+ "Bash(ls *)",
+ "Bash(wc *)",
+ "Bash(du *)",
+ "Bash(readlink *)",
+ "Bash(realpath *)",
+ "Bash(basename *)",
+ "Bash(dirname *)"
+ ]
+ },
+ "hooks": {
+ "PostToolUse": [
+ {
+ "matcher": "Edit|Write|MultiEdit",
+ "hooks": [
+ {
+ "type": "command",
+ "command": "$CLAUDE_PROJECT_DIR/.claude/hooks/validate-bash.sh"
+ }
+ ]
+ }
+ ]
+ }
+}
diff --git a/languages/bash/githooks/pre-commit b/languages/bash/githooks/pre-commit
new file mode 100755
index 0000000..1520690
--- /dev/null
+++ b/languages/bash/githooks/pre-commit
@@ -0,0 +1,76 @@
+#!/usr/bin/env bash
+# Pre-commit hook: secret scan + shellcheck on staged shell files.
+# Use `git commit --no-verify` to bypass for confirmed false positives.
+
+set -u
+
+REPO_ROOT="$(git rev-parse --show-toplevel)"
+cd "$REPO_ROOT" || exit 1
+
+# --- 1. Secret scan ---
+# Patterns for common credentials. Scans only added lines in the staged diff.
+#
+# Two passes because case-sensitivity differs. AWS keys are uppercase, sk- keys
+# lowercase, PEM headers fixed, so those match case-SENSITIVELY: under -i,
+# AKIA[0-9A-Z]{16} matches any mixed-case 20-char run, which random base64 in an
+# embedded image blob hits ~6% of the time per 100KB and blocks real commits.
+# Only the keyword=value patterns need -i.
+SECRET_PATTERNS_CS='(AKIA[0-9A-Z]{16}|sk-[a-zA-Z0-9_-]{20,}|-----BEGIN (RSA|DSA|EC|OPENSSH|PGP)( PRIVATE)?( KEY| KEY BLOCK)?-----)'
+SECRET_PATTERNS_CI='(api[_-]?key|api[_-]?secret|auth[_-]?token|secret[_-]?key|bearer[_-]?token|access[_-]?token|password)[[:space:]]*[:=][[:space:]]*["'"'"'][^"'"'"']{16,}["'"'"']'
+
+# Read the diff on its own so a git failure is distinguishable from "grep
+# matched nothing". Both end in a non-zero status, but only one of them means
+# there is nothing to scan; piping them together and swallowing the result with
+# `|| true` made a broken git look like a clean commit — the scan searched an
+# empty string, found nothing, and the secret went in.
+if ! staged_diff="$(git diff --cached -U0 --diff-filter=AM)"; then
+ echo "pre-commit: cannot read the staged diff — refusing to skip the secret scan" >&2
+ exit 1
+fi
+
+# The greps keep their `|| true`: exiting 1 on no match is their normal result.
+added_lines="$(printf '%s\n' "$staged_diff" | grep '^+' | grep -v '^+++' || true)"
+
+cs_hits="$(printf '%s\n' "$added_lines" | grep -nE "$SECRET_PATTERNS_CS" || true)"
+ci_hits="$(printf '%s\n' "$added_lines" | grep -niE "$SECRET_PATTERNS_CI" || true)"
+# awk dedupes lines both passes matched, keeping first-seen order.
+secret_hits="$(printf '%s\n%s' "$cs_hits" "$ci_hits" \
+ | grep -v '^[[:space:]]*$' | awk '!seen[$0]++' || true)"
+
+if [ -n "$secret_hits" ]; then
+ echo "pre-commit: potential secret in staged changes:" >&2
+ echo "$secret_hits" >&2
+ echo "" >&2
+ echo "Review the lines above. If this is a false positive (test fixture, documentation)," >&2
+ echo "bypass with: git commit --no-verify" >&2
+ exit 1
+fi
+
+# --- 2. shellcheck on staged .sh / .bash files ---
+# Same split as the secret scan above: a git failure must not read as "no files
+# staged", which would skip the language check silently.
+if ! staged_names="$(git diff --cached --name-only --diff-filter=AM)"; then
+ echo "pre-commit: cannot read the staged file list — refusing to skip the check" >&2
+ exit 1
+fi
+
+staged_sh="$(printf '%s\n' "$staged_names" | grep -E '\.(sh|bash)$' || true)"
+
+if [ -n "$staged_sh" ] && command -v shellcheck >/dev/null 2>&1; then
+ failed=""
+ while IFS= read -r f; do
+ [ -z "$f" ] && continue
+ [ -f "$f" ] || continue
+ if ! shellcheck "$f" >/dev/null 2>&1; then
+ failed="${failed}${f}"$'\n'
+ fi
+ done <<< "$staged_sh"
+
+ if [ -n "$failed" ]; then
+ printf 'pre-commit: shellcheck failed on staged files:\n\n%s\n' "$failed" >&2
+ echo "Run: shellcheck <file> and fix the findings, then re-stage." >&2
+ exit 1
+ fi
+fi
+
+exit 0
diff --git a/languages/bash/gitignore-add.txt b/languages/bash/gitignore-add.txt
new file mode 100644
index 0000000..899f5ba
--- /dev/null
+++ b/languages/bash/gitignore-add.txt
@@ -0,0 +1,4 @@
+# Claude Code — local tooling, delivered by install/sync, not committed
+.claude/
+CLAUDE.md
+githooks/
diff --git a/languages/bash/tests/validate-bash.bats b/languages/bash/tests/validate-bash.bats
new file mode 100644
index 0000000..9f268a1
--- /dev/null
+++ b/languages/bash/tests/validate-bash.bats
@@ -0,0 +1,96 @@
+#!/usr/bin/env bats
+#
+# Tests for languages/bash/claude/hooks/validate-bash.sh — the PostToolUse hook
+# that runs shellcheck on edited shell files and blocks on a violation.
+#
+# The hook reads tool-call JSON on stdin and extracts the file path, so each
+# test pipes a JSON payload naming a real file it wrote into a temp dir. The
+# shellcheck dependency is real (integration): clean files pass, genuinely
+# broken ones fail. Tests needing shellcheck skip when it's absent so the suite
+# stays portable.
+
+HOOK="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/claude/hooks/validate-bash.sh"
+
+setup() {
+ TEST_DIR="$(mktemp -d -t validate-bash-bats.XXXXXX)"
+}
+
+teardown() {
+ rm -rf "$TEST_DIR"
+}
+
+# Build a tool-call JSON payload naming a file_path.
+payload() {
+ printf '{"tool_input": {"file_path": "%s"}}' "$1"
+}
+
+# ---- Normal ----------------------------------------------------------
+
+@test "validate-bash: a clean .sh file passes silently (exit 0)" {
+ command -v shellcheck >/dev/null 2>&1 || skip "shellcheck not installed"
+ printf '#!/usr/bin/env bash\nset -euo pipefail\necho "ok"\n' > "$TEST_DIR/clean.sh"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/clean.sh")"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+# ---- Error -----------------------------------------------------------
+
+@test "validate-bash: a shellcheck violation blocks (exit 2, names shellcheck)" {
+ command -v shellcheck >/dev/null 2>&1 || skip "shellcheck not installed"
+ # SC2086: unquoted expansion that word-splits — a real shellcheck warning.
+ printf '#!/usr/bin/env bash\nf=$1\nrm $f\n' > "$TEST_DIR/bad.sh"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/bad.sh")"
+ [ "$status" -eq 2 ]
+ [[ "$output" == *"SHELLCHECK"* ]]
+}
+
+# ---- Boundary --------------------------------------------------------
+
+@test "validate-bash: a non-shell file is ignored (exit 0)" {
+ printf 'print("hello")\n' > "$TEST_DIR/script.py"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/script.py")"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "validate-bash: an extensionless file with a bash shebang is validated" {
+ command -v shellcheck >/dev/null 2>&1 || skip "shellcheck not installed"
+ printf '#!/usr/bin/env bash\nf=$1\nrm $f\n' > "$TEST_DIR/cli-tool"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/cli-tool")"
+ [ "$status" -eq 2 ]
+ [[ "$output" == *"SHELLCHECK"* ]]
+}
+
+@test "validate-bash: an extensionless non-shell file is ignored (exit 0)" {
+ printf 'just some text\nno shebang here\n' > "$TEST_DIR/notes"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/notes")"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "validate-bash: empty file_path is a no-op (exit 0)" {
+ run bash "$HOOK" <<< '{"tool_input": {}}'
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "validate-bash: a missing file is a no-op (exit 0)" {
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/does-not-exist.sh")"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "validate-bash: shellcheck absent does not block the edit (exit 0)" {
+ # PATH with jq + coreutils symlinked but no shellcheck → hook can't validate,
+ # must not block.
+ STUB="$TEST_DIR/bin"
+ mkdir -p "$STUB"
+ for b in bash jq head cat printf grep sed; do
+ src="$(command -v "$b" 2>/dev/null)" && ln -sf "$src" "$STUB/$b"
+ done
+ printf '#!/usr/bin/env bash\nf=$1\nrm $f\n' > "$TEST_DIR/bad.sh"
+ run env PATH="$STUB" bash "$HOOK" <<< "$(payload "$TEST_DIR/bad.sh")"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
diff --git a/languages/default-CLAUDE.md b/languages/default-CLAUDE.md
new file mode 100644
index 0000000..a5b6925
--- /dev/null
+++ b/languages/default-CLAUDE.md
@@ -0,0 +1,64 @@
+# CLAUDE.md
+
+## Project
+
+Describe this project: what it is, its layout, and its conventions. This
+default was seeded by `install-lang` because the installed bundle ships no
+language-specific CLAUDE.md — it deliberately names no language, so replace
+this section with an accurate description rather than inheriting a wrong one.
+
+**Typical layout (edit to match):**
+- entry points — the file(s) that run first
+- source directories — where the real code lives
+- tests — beside the code, or under a `tests/` tree
+
+## Build & Test Commands
+
+If the project has a Makefile, document its targets here. A common shape:
+
+```bash
+make test # run the test suite
+make lint # run the linter / formatter check
+make build # build the project
+```
+
+Otherwise, document the direct commands a contributor runs to test and build.
+
+## Language Rules
+
+Shared rules live in `.claude/rules/` (installed from `claude-rules/`):
+- `commits.md` — author identity, no AI attribution, message format
+- `testing.md` — TDD discipline and test-quality standards
+- `verification.md` — verify-before-claim-done discipline
+
+If a language bundle was installed, its own rule files (code style, testing
+conventions) sit alongside these in `.claude/rules/`.
+
+## Git Workflow
+
+Commit conventions: see `.claude/rules/commits.md`.
+
+If a `githooks/` pre-commit hook was installed, activate it on a fresh clone
+with `git config core.hooksPath githooks`.
+
+## Problem-Solving Approach
+
+Investigate before fixing. When diagnosing a bug:
+1. Read the relevant code and trace what actually happens
+2. Identify the root cause, not a surface symptom
+3. Write a failing test that captures the correct behavior
+4. Fix, then re-run tests
+
+## Testing Discipline
+
+TDD is the default: write a failing test before any implementation. If you
+can't write the test, you don't yet understand the change. Details in
+`.claude/rules/testing.md`.
+
+## What Not to Do
+
+- Don't add features beyond what was asked
+- Don't refactor surrounding code when fixing a bug
+- Don't add comments to code you didn't change
+- Don't create abstractions for one-time operations
+- Don't commit credentials or API keys
diff --git a/languages/elisp/claude/hooks/validate-el.sh b/languages/elisp/claude/hooks/validate-el.sh
index 2529fcc..870eefe 100755
--- a/languages/elisp/claude/hooks/validate-el.sh
+++ b/languages/elisp/claude/hooks/validate-el.sh
@@ -39,8 +39,6 @@ f="$(jq -r '.tool_input.file_path // .tool_response.filePath // empty')"
[ -z "$f" ] && exit 0
[ "${f##*.}" = "el" ] || exit 0
-MAX_AUTO_TEST_FILES=20 # skip if more matches than this (large test suites)
-
# --- Phase 1: syntax + byte-compile ---
case "$f" in
*/init.el|*/early-init.el)
@@ -55,6 +53,7 @@ case "$f" in
# under a tests/ subdir) so cross-project edits compile against their
# own modules, not just this project's.
if ! output="$(emacs --batch --no-site-file --no-site-lisp \
+ --eval '(setq load-prefer-newer t)' \
-L "$(dirname "$f")" \
-L "$(dirname "$f")/.." \
-L "$PROJECT_ROOT" \
@@ -95,15 +94,17 @@ case "$f" in
esac
count="${#tests[@]}"
-if [ "$count" -ge 1 ] && [ "$count" -le "$MAX_AUTO_TEST_FILES" ]; then
+if [ "$count" -ge 1 ]; then
load_args=()
for t in "${tests[@]}"; do load_args+=("-l" "$t"); done
if ! output="$(emacs --batch --no-site-file --no-site-lisp \
+ --eval '(setq load-prefer-newer t)' \
-L "$PROJECT_ROOT" \
-L "$PROJECT_ROOT/modules" \
-L "$PROJECT_ROOT/tests" \
-L "$PROJECT_ROOT/themes" \
--eval '(package-initialize)' \
+ --eval "(cd \"$PROJECT_ROOT/tests\")" \
-l ert "${load_args[@]}" \
--eval "(ert-run-tests-batch-and-exit '(not (tag :slow)))" 2>&1)"; then
# Terminal gets a compact summary (the run tally + the failing test names);
diff --git a/languages/elisp/claude/rules/elisp-testing.md b/languages/elisp/claude/rules/elisp-testing.md
index 7c3a9ef..1ac76a0 100644
--- a/languages/elisp/claude/rules/elisp-testing.md
+++ b/languages/elisp/claude/rules/elisp-testing.md
@@ -43,6 +43,8 @@ The bundle ships a coverage summary at `.claude/scripts/coverage-summary.el` and
The number to watch is the missing-file count. A module no test loads never appears in the SimpleCov report, so a line-weighted total skips it silently — the suite looks healthier than it is. The summary counts every `modules/*.el` on disk that's absent from the report as 0%, so an untested module drags the project number down where you can see it. Copy the fragment's targets into your own Makefile to adopt it; the bundle never edits your Makefile.
+This is a local-only helper by design. `.claude/scripts/` is gitignored in code projects, so `coverage-summary.el` is untracked and CI never runs `make coverage-summary` against it — it's a developer-run check, not a CI gate. A gitignored install is intentional, not a coverage gap; don't move the script to a tracked `scripts/` dir to make CI pick it up.
+
## TDD Workflow
Write the failing test first. A failing test proves you understand the change. Assume the bug is in production code until the test proves otherwise — never fix the test before proving the test is wrong.
diff --git a/languages/elisp/claude/scripts/coverage-summary.el b/languages/elisp/claude/scripts/coverage-summary.el
index eb30c66..ed7ecfc 100644
--- a/languages/elisp/claude/scripts/coverage-summary.el
+++ b/languages/elisp/claude/scripts/coverage-summary.el
@@ -11,8 +11,15 @@
;; by file rather than by line, so untested modules are visible.
;;
;; Self-contained on purpose — it ships into a project's =.claude/scripts/= and
-;; must run with nothing but stock Emacs (`json' is built in). The SimpleCov
-;; JSON shape it parses is:
+;; must run with nothing but stock Emacs (`json' is built in).
+;;
+;; Local-only helper (Craig, 2026-06-28). =.claude/scripts/= is gitignored in
+;; code projects, so this file is not tracked and CI cannot run
+;; `make coverage-summary' against it. That is intentional: it stays a
+;; developer-run helper, not shipped to a tracked =scripts/= dir and not a CI
+;; gate. A gitignored install here is the design, not a coverage gap.
+;;
+;; The SimpleCov JSON shape it parses is:
;; { <suite>: { "coverage": { <abs-path>: [null | 0 | int, ...] } } }
;; where a null entry is a non-executable line, 0 is executable-but-unhit, and
;; any positive integer is a hit. Data unions across multiple suite keys.
@@ -91,11 +98,15 @@ missing or malformed."
(defun cj/coverage-summary--source-files (source-dir project-root)
"Return *.el files directly under SOURCE-DIR, relative to PROJECT-ROOT.
-Sorted; compiled files and subdirectories are out of scope."
+Sorted. Compiled files and subdirectories are out of scope, as are generated
+package files (`*-autoloads.el', `*-pkg.el') -- a build tool writes those, no
+test covers them, and counting them as untested source skews the number."
(let ((source-dir (file-name-as-directory (expand-file-name source-dir)))
(project-root (file-name-as-directory (expand-file-name project-root))))
- (sort (mapcar (lambda (p) (file-relative-name p project-root))
- (directory-files source-dir t "\\.el\\'"))
+ (sort (seq-remove
+ (lambda (p) (string-match-p "\\(?:-autoloads\\|-pkg\\)\\.el\\'" p))
+ (mapcar (lambda (p) (file-relative-name p project-root))
+ (directory-files source-dir t "\\.el\\'")))
#'string<)))
(defun cj/coverage-summary--missing (tracked source-dir project-root)
diff --git a/languages/elisp/githooks/pre-commit b/languages/elisp/githooks/pre-commit
index 909cde2..a87bedf 100755
--- a/languages/elisp/githooks/pre-commit
+++ b/languages/elisp/githooks/pre-commit
@@ -5,15 +5,37 @@
set -u
REPO_ROOT="$(git rev-parse --show-toplevel)"
-cd "$REPO_ROOT"
+cd "$REPO_ROOT" || exit 1
# --- 1. Secret scan ---
# Patterns for common credentials. Scans only added lines in the staged diff.
-SECRET_PATTERNS='(AKIA[0-9A-Z]{16}|sk-[a-zA-Z0-9_-]{20,}|-----BEGIN (RSA|DSA|EC|OPENSSH|PGP)( PRIVATE)?( KEY| KEY BLOCK)?-----|(api[_-]?key|api[_-]?secret|auth[_-]?token|secret[_-]?key|bearer[_-]?token|access[_-]?token|password)[[:space:]]*[:=][[:space:]]*["'"'"'][^"'"'"']{16,}["'"'"'])'
+#
+# Two passes because case-sensitivity differs. AWS keys are uppercase, sk- keys
+# lowercase, PEM headers fixed, so those match case-SENSITIVELY: under -i,
+# AKIA[0-9A-Z]{16} matches any mixed-case 20-char run, which random base64 in an
+# embedded image blob hits ~6% of the time per 100KB and blocks real commits.
+# Only the keyword=value patterns need -i.
+SECRET_PATTERNS_CS='(AKIA[0-9A-Z]{16}|sk-[a-zA-Z0-9_-]{20,}|-----BEGIN (RSA|DSA|EC|OPENSSH|PGP)( PRIVATE)?( KEY| KEY BLOCK)?-----)'
+SECRET_PATTERNS_CI='(api[_-]?key|api[_-]?secret|auth[_-]?token|secret[_-]?key|bearer[_-]?token|access[_-]?token|password)[[:space:]]*[:=][[:space:]]*["'"'"'][^"'"'"']{16,}["'"'"']'
-secret_hits="$(git diff --cached -U0 --diff-filter=AM \
- | grep '^+' | grep -v '^+++' \
- | grep -iEn "$SECRET_PATTERNS" || true)"
+# Read the diff on its own so a git failure is distinguishable from "grep
+# matched nothing". Both end in a non-zero status, but only one of them means
+# there is nothing to scan; piping them together and swallowing the result with
+# `|| true` made a broken git look like a clean commit — the scan searched an
+# empty string, found nothing, and the secret went in.
+if ! staged_diff="$(git diff --cached -U0 --diff-filter=AM)"; then
+ echo "pre-commit: cannot read the staged diff — refusing to skip the secret scan" >&2
+ exit 1
+fi
+
+# The greps keep their `|| true`: exiting 1 on no match is their normal result.
+added_lines="$(printf '%s\n' "$staged_diff" | grep '^+' | grep -v '^+++' || true)"
+
+cs_hits="$(printf '%s\n' "$added_lines" | grep -nE "$SECRET_PATTERNS_CS" || true)"
+ci_hits="$(printf '%s\n' "$added_lines" | grep -niE "$SECRET_PATTERNS_CI" || true)"
+# awk dedupes lines both passes matched, keeping first-seen order.
+secret_hits="$(printf '%s\n%s' "$cs_hits" "$ci_hits" \
+ | grep -v '^[[:space:]]*$' | awk '!seen[$0]++' || true)"
if [ -n "$secret_hits" ]; then
echo "pre-commit: potential secret in staged changes:" >&2
@@ -25,7 +47,14 @@ if [ -n "$secret_hits" ]; then
fi
# --- 2. Paren check on staged .el files ---
-staged_el="$(git diff --cached --name-only --diff-filter=AM | grep '\.el$' || true)"
+# Same split as the secret scan above: a git failure must not read as "no files
+# staged", which would skip the language check silently.
+if ! staged_names="$(git diff --cached --name-only --diff-filter=AM)"; then
+ echo "pre-commit: cannot read the staged file list — refusing to skip the check" >&2
+ exit 1
+fi
+
+staged_el="$(printf '%s\n' "$staged_names" | grep '\.el$' || true)"
if [ -n "$staged_el" ]; then
paren_fail=""
diff --git a/languages/elisp/tests/test-coverage-summary.el b/languages/elisp/tests/test-coverage-summary.el
index 5be03b3..a4525db 100644
--- a/languages/elisp/tests/test-coverage-summary.el
+++ b/languages/elisp/tests/test-coverage-summary.el
@@ -109,6 +109,33 @@ is a JSON array string like \"[1, 0, null]\"."
(ert-deftest cs-file-pct-fully-covered ()
(should (= 100.0 (cj/coverage-summary--file-pct 4 4))))
+;; --- source-file scan and under-dir filtering ------------------------------
+
+(ert-deftest cs-source-files-is-non-recursive ()
+ "Only top-level *.el under SOURCE-DIR are source; files in subdirectories
+are out of scope."
+ (cs-test--with-project
+ (list :sources '(("top.el" . ";; t") ("sub/nested.el" . ";; n"))
+ :report (cs-test--report '(("top.el" . "[1]"))))
+ (let ((sources (mapcar #'file-name-nondirectory
+ (cj/coverage-summary--source-files src root))))
+ (should (member "top.el" sources))
+ (should-not (member "nested.el" sources)))))
+
+(ert-deftest cs-under-dir-filters-outside-source-and-rekeys ()
+ "Report entries outside SOURCE-DIR are dropped; survivors are keyed
+relative to PROJECT-ROOT."
+ (cs-test--with-project
+ (list :sources '(("in.el" . ";; i"))
+ :report (cs-test--report '(("in.el" . "[1, 1]")
+ ("../out.el" . "[1, 0]"))))
+ (let* ((table (cj/coverage-summary--under-dir
+ (cj/coverage-summary--parse-file report) src root))
+ (keys (let (ks) (maphash (lambda (k _v) (push k ks)) table) ks)))
+ (should (equal keys (list (file-relative-name
+ (expand-file-name "src/in.el" root) root))))
+ (should (= 1 (hash-table-count table))))))
+
;; --- missing-file detection (the kernel) -----------------------------------
(ert-deftest cs-missing-finds-ondisk-file-absent-from-report ()
@@ -135,6 +162,24 @@ is a JSON array string like \"[1, 0, null]\"."
(missing (cj/coverage-summary--missing tracked src root)))
(should (null missing)))))
+(ert-deftest cs-missing-excludes-generated-package-files ()
+ "Generated -autoloads.el / -pkg.el are not source, so a build tool writing
+them does not drag the number down; a genuinely untested source is still
+flagged (the filter is not over-broad)."
+ (cs-test--with-project
+ (list :sources '(("real.el" . ";; r") ("untested.el" . ";; u")
+ ("proj-autoloads.el" . ";; gen")
+ ("proj-pkg.el" . ";; gen"))
+ :report (cs-test--report '(("real.el" . "[1, 1]"))))
+ (let* ((table (cj/coverage-summary--under-dir
+ (cj/coverage-summary--parse-file report) src root))
+ (tracked (let (ks) (maphash (lambda (k _v) (push k ks)) table) ks))
+ (missing (mapcar #'file-name-nondirectory
+ (cj/coverage-summary--missing tracked src root))))
+ (should (member "untested.el" missing))
+ (should-not (member "proj-autoloads.el" missing))
+ (should-not (member "proj-pkg.el" missing)))))
+
;; --- project number (unit-weighted, missing as 0%) -------------------------
(ert-deftest cs-project-pct-unit-weighted-with-missing-as-zero ()
diff --git a/languages/elisp/tests/test-pre-commit-hook.bats b/languages/elisp/tests/test-pre-commit-hook.bats
new file mode 100644
index 0000000..413c71d
--- /dev/null
+++ b/languages/elisp/tests/test-pre-commit-hook.bats
@@ -0,0 +1,126 @@
+#!/usr/bin/env bats
+# Tests for githooks/pre-commit — the secret scan and paren check.
+#
+# The scan reads its input through a pipeline:
+#
+# added_lines="$(git diff --cached ... | grep '^+' | grep -v '^+++' || true)"
+#
+# `grep` exits 1 when it matches nothing, which is the ordinary case, so the
+# `|| true` has to stay. But with no `pipefail` it also swallows a failure of
+# `git diff` itself, and an empty `added_lines` makes the scan search nothing,
+# find nothing, and report clean. A gate that passes without looking is the
+# failure this file exists to pin: the fail-open test drives a broken `git diff`
+# and asserts the hook refuses rather than exiting 0.
+#
+# Each test builds a throwaway git repo in BATS_TEST_TMPDIR, so nothing touches
+# the real repository or its hooks.
+
+setup() {
+ HOOK="${BATS_TEST_DIRNAME}/../githooks/pre-commit"
+ REPO="${BATS_TEST_TMPDIR}/repo"
+ mkdir -p "$REPO"
+ cd "$REPO" || return 1
+ git init -q .
+ git config user.email t@example.com
+ git config user.name Test
+ # Split so the fixtures never appear as credential-shaped literals here.
+ AWS_TAIL="IOSFODNN7EXAMPLE"
+ WORD_TAIL="word"
+}
+
+# Put a stub `git` ahead of the real one that fails for the staged-diff call
+# and delegates everything else, so only the pipeline under test breaks.
+break_staged_diff() {
+ mkdir -p "${BATS_TEST_TMPDIR}/bin"
+ cat > "${BATS_TEST_TMPDIR}/bin/git" <<'STUB'
+#!/usr/bin/env bash
+if [ "${1:-}" = "diff" ] && [ "${2:-}" = "--cached" ] && [ "${3:-}" = "-U0" ]; then
+ echo "simulated git failure" >&2
+ exit 128
+fi
+exec /usr/bin/git "$@"
+STUB
+ chmod +x "${BATS_TEST_TMPDIR}/bin/git"
+ PATH="${BATS_TEST_TMPDIR}/bin:$PATH"
+}
+
+# ------------------------------- Normal cases -------------------------------
+
+@test "secret scan: blocks a staged AWS key" {
+ # Assembled at runtime: a literal key-shaped string in this file would trip
+ # the very hook under test on every commit that touches it, and this repo
+ # mirrors to a public remote.
+ printf 'aws = "%s"\n' "AKIA${AWS_TAIL}" > creds.txt
+ git add creds.txt
+ run "$HOOK"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"potential secret"* ]]
+}
+
+@test "secret scan: blocks a staged keyword=value password" {
+ printf '%s = "%s"\n' "pass${WORD_TAIL}" "correcthorsebatterystaple" > conf.txt
+ git add conf.txt
+ run "$HOOK"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"potential secret"* ]]
+}
+
+@test "secret scan: allows an ordinary staged file" {
+ printf 'just some prose\n' > notes.txt
+ git add notes.txt
+ run "$HOOK"
+ [ "$status" -eq 0 ]
+}
+
+# ------------------------------ Boundary cases ------------------------------
+
+@test "secret scan: allows a commit with nothing staged" {
+ run "$HOOK"
+ [ "$status" -eq 0 ]
+}
+
+@test "paren check: blocks an unbalanced staged .el file" {
+ printf '(defun broken ()\n (message "no close"\n' > bad.el
+ git add bad.el
+ run "$HOOK"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"paren check failed"* ]]
+}
+
+@test "paren check: allows a balanced staged .el file" {
+ printf '(defun fine ()\n (message "ok"))\n' > good.el
+ git add good.el
+ run "$HOOK"
+ [ "$status" -eq 0 ]
+}
+
+# -------------------------------- Error cases -------------------------------
+
+@test "secret scan: refuses to pass when the staged diff cannot be read" {
+ # The scan must not report clean after searching nothing. Without a
+ # pipefail-aware guard the broken diff yields an empty added_lines and the
+ # hook exits 0, letting a real secret through unscanned.
+ printf 'aws = "%s"\n' "AKIA${AWS_TAIL}" > creds.txt
+ git add creds.txt
+ break_staged_diff
+ run "$HOOK"
+ [ "$status" -ne 0 ]
+}
+
+@test "paren check: refuses to pass when the staged file list cannot be read" {
+ printf '(defun broken ()\n (message "no close"\n' > bad.el
+ git add bad.el
+ mkdir -p "${BATS_TEST_TMPDIR}/bin2"
+ cat > "${BATS_TEST_TMPDIR}/bin2/git" <<'STUB'
+#!/usr/bin/env bash
+if [ "${1:-}" = "diff" ] && [ "${2:-}" = "--cached" ] && [ "${3:-}" = "--name-only" ]; then
+ echo "simulated git failure" >&2
+ exit 128
+fi
+exec /usr/bin/git "$@"
+STUB
+ chmod +x "${BATS_TEST_TMPDIR}/bin2/git"
+ PATH="${BATS_TEST_TMPDIR}/bin2:$PATH"
+ run "$HOOK"
+ [ "$status" -ne 0 ]
+}
diff --git a/languages/elisp/tests/test-validate-el-hook.bats b/languages/elisp/tests/test-validate-el-hook.bats
new file mode 100644
index 0000000..d4d6f23
--- /dev/null
+++ b/languages/elisp/tests/test-validate-el-hook.bats
@@ -0,0 +1,100 @@
+#!/usr/bin/env bats
+# Tests for .claude/hooks/validate-el.sh — the auto-test runner.
+#
+# The runner used to skip entirely above MAX_AUTO_TEST_FILES=20, with no else
+# branch: nothing printed, exit 0, indistinguishable from a passing run. That
+# was live for the three largest families here (calendar-sync 63 test files,
+# music 45, ai-term 35), so every edit to those ran parens and byte-compile and
+# zero tests, silently.
+#
+# The cap was removed rather than made loud, because its premise did not hold.
+# Measured on this machine, running a whole family takes about a second:
+# ai-term 208 tests in 1.0s, music 403 in 1.7s, calendar-sync 633 in 0.9s. It
+# was also concealing a real cross-test pollution bug in calendar-sync that
+# only appears when that family runs in one process.
+#
+# These tests pin that no file count is skipped. Each builds a synthetic
+# project in BATS_TEST_TMPDIR and points CLAUDE_PROJECT_DIR at it, so nothing
+# runs against the real tree.
+
+setup() {
+ # Bundle layout: the hook ships at claude/hooks/ here and installs to
+ # .claude/hooks/ in a consuming project. This test was written against the
+ # installed layout, so re-homing it needed the path adjusted.
+ HOOK="${BATS_TEST_DIRNAME}/../claude/hooks/validate-el.sh"
+ PROJ="${BATS_TEST_TMPDIR}/proj"
+ mkdir -p "$PROJ/modules" "$PROJ/tests"
+ export CLAUDE_PROJECT_DIR="$PROJ"
+ printf '(provide (quote widget))\n' > "$PROJ/modules/widget.el"
+}
+
+# N green test files matching the widget stem.
+make_tests() {
+ local n="$1" i
+ for ((i = 1; i <= n; i++)); do
+ printf '(require (quote ert))\n(ert-deftest test-widget-%d () (should t))\n' \
+ "$i" > "$PROJ/tests/test-widget-${i}.el"
+ done
+}
+
+# One failing test file, to prove the run is real rather than merely quiet.
+make_failing_test() {
+ printf '(require (quote ert))\n(ert-deftest test-widget-bad () (should nil))\n' \
+ > "$PROJ/tests/test-widget-bad.el"
+}
+
+hook_input() {
+ printf '{"tool_input":{"file_path":"%s"}}' "$PROJ/modules/widget.el"
+}
+
+run_hook() {
+ run bash -c "$(printf '%q' "$HOOK") <<< '$(hook_input)'"
+}
+
+# ------------------------------- Normal cases -------------------------------
+
+@test "a small family runs and passes quietly" {
+ make_tests 3
+ run_hook
+ [ "$status" -eq 0 ]
+}
+
+@test "a failing test blocks, so a quiet pass means the tests really ran" {
+ make_tests 3
+ make_failing_test
+ run_hook
+ [ "$status" -eq 2 ]
+ [[ "$output" == *"TESTS FAILED"* ]]
+}
+
+# ------------------------------ Boundary cases ------------------------------
+
+@test "at the old cap of 20 files: runs" {
+ make_tests 20
+ run_hook
+ [ "$status" -eq 0 ]
+}
+
+@test "past the old cap: still runs, no longer skipped" {
+ make_tests 21
+ run_hook
+ [ "$status" -eq 0 ]
+ [[ "${output,,}" != *"skipped"* ]]
+}
+
+@test "well past the old cap: a failure in file 63 is still caught" {
+ # The regression this guards: at 63 files the runner used to skip, so a red
+ # test in a big family reported clean. calendar-sync is exactly this size.
+ make_tests 63
+ make_failing_test
+ run_hook
+ [ "$status" -eq 2 ]
+ [[ "$output" == *"TESTS FAILED"* ]]
+}
+
+# -------------------------------- Error cases -------------------------------
+
+@test "no matching tests: exits clean without running anything" {
+ run_hook
+ [ "$status" -eq 0 ]
+}
diff --git a/languages/go/githooks/pre-commit b/languages/go/githooks/pre-commit
index a3d6f3f..7d93949 100755
--- a/languages/go/githooks/pre-commit
+++ b/languages/go/githooks/pre-commit
@@ -5,15 +5,37 @@
set -u
REPO_ROOT="$(git rev-parse --show-toplevel)"
-cd "$REPO_ROOT"
+cd "$REPO_ROOT" || exit 1
# --- 1. Secret scan ---
# Patterns for common credentials. Scans only added lines in the staged diff.
-SECRET_PATTERNS='(AKIA[0-9A-Z]{16}|sk-[a-zA-Z0-9_-]{20,}|-----BEGIN (RSA|DSA|EC|OPENSSH|PGP)( PRIVATE)?( KEY| KEY BLOCK)?-----|(api[_-]?key|api[_-]?secret|auth[_-]?token|secret[_-]?key|bearer[_-]?token|access[_-]?token|password)[[:space:]]*[:=][[:space:]]*["'"'"'][^"'"'"']{16,}["'"'"'])'
+#
+# Two passes because case-sensitivity differs. AWS keys are uppercase, sk- keys
+# lowercase, PEM headers fixed, so those match case-SENSITIVELY: under -i,
+# AKIA[0-9A-Z]{16} matches any mixed-case 20-char run, which random base64 in an
+# embedded image blob hits ~6% of the time per 100KB and blocks real commits.
+# Only the keyword=value patterns need -i.
+SECRET_PATTERNS_CS='(AKIA[0-9A-Z]{16}|sk-[a-zA-Z0-9_-]{20,}|-----BEGIN (RSA|DSA|EC|OPENSSH|PGP)( PRIVATE)?( KEY| KEY BLOCK)?-----)'
+SECRET_PATTERNS_CI='(api[_-]?key|api[_-]?secret|auth[_-]?token|secret[_-]?key|bearer[_-]?token|access[_-]?token|password)[[:space:]]*[:=][[:space:]]*["'"'"'][^"'"'"']{16,}["'"'"']'
-secret_hits="$(git diff --cached -U0 --diff-filter=AM \
- | grep '^+' | grep -v '^+++' \
- | grep -iEn "$SECRET_PATTERNS" || true)"
+# Read the diff on its own so a git failure is distinguishable from "grep
+# matched nothing". Both end in a non-zero status, but only one of them means
+# there is nothing to scan; piping them together and swallowing the result with
+# `|| true` made a broken git look like a clean commit — the scan searched an
+# empty string, found nothing, and the secret went in.
+if ! staged_diff="$(git diff --cached -U0 --diff-filter=AM)"; then
+ echo "pre-commit: cannot read the staged diff — refusing to skip the secret scan" >&2
+ exit 1
+fi
+
+# The greps keep their `|| true`: exiting 1 on no match is their normal result.
+added_lines="$(printf '%s\n' "$staged_diff" | grep '^+' | grep -v '^+++' || true)"
+
+cs_hits="$(printf '%s\n' "$added_lines" | grep -nE "$SECRET_PATTERNS_CS" || true)"
+ci_hits="$(printf '%s\n' "$added_lines" | grep -niE "$SECRET_PATTERNS_CI" || true)"
+# awk dedupes lines both passes matched, keeping first-seen order.
+secret_hits="$(printf '%s\n%s' "$cs_hits" "$ci_hits" \
+ | grep -v '^[[:space:]]*$' | awk '!seen[$0]++' || true)"
if [ -n "$secret_hits" ]; then
echo "pre-commit: potential secret in staged changes:" >&2
@@ -27,8 +49,14 @@ fi
# --- 2. gofmt check on staged .go files ---
# gofmt -l lists files that aren't gofmt-clean. Skip generated and vendored
# files the same way the rest of the toolchain does.
-staged_go="$(git diff --cached --name-only --diff-filter=AM \
- | grep '\.go$' \
+# Same split as the secret scan above: a git failure must not read as "no files
+# staged", which would skip the language check silently.
+if ! staged_names="$(git diff --cached --name-only --diff-filter=AM)"; then
+ echo "pre-commit: cannot read the staged file list — refusing to skip the check" >&2
+ exit 1
+fi
+
+staged_go="$(printf '%s\n' "$staged_names" | grep '\.go$' \
| grep -vE '(^|/)vendor/' || true)"
if [ -n "$staged_go" ] && command -v gofmt >/dev/null 2>&1; then
diff --git a/languages/python/CLAUDE.md b/languages/python/CLAUDE.md
new file mode 100644
index 0000000..a2d0a82
--- /dev/null
+++ b/languages/python/CLAUDE.md
@@ -0,0 +1,80 @@
+# CLAUDE.md
+
+## Project
+
+Python project. Customize this section with your own description, layout,
+and conventions.
+
+**Typical layout:**
+- `src/<package>/` or a top-level package directory — importable code
+- `tests/` — pytest tests mirroring the package layout
+- `pyproject.toml` — dependencies, tool config (ruff, pytest, coverage)
+
+## Build & Test Commands
+
+If the project has a Makefile, document targets here. Common pattern:
+
+```bash
+make test # run the pytest suite
+make test FILE=tests/x.py # one file
+make coverage # suite + coverage report
+make lint # ruff across the tree
+make typecheck # mypy (if the project adopts it)
+make fmt # ruff format / black
+```
+
+Direct equivalents: `python3 -m pytest`, `pytest tests/test_x.py::test_name`,
+`ruff check .`, `ruff format --diff .`, `mypy src/`.
+
+## Language Rules
+
+See rule files in `.claude/rules/`:
+- `python-testing.md` — pytest conventions and fixture discipline
+- `verification.md` — verify-before-claim-done discipline
+
+## Git Workflow
+
+Commit conventions: see `.claude/rules/commits.md` (author identity,
+no AI attribution, message format).
+
+Pre-commit hook in `githooks/` scans for secrets, syntax-checks staged Python,
+and runs `ruff` when it's installed. Activate on a fresh clone with
+`git config core.hooksPath githooks`.
+
+## Problem-Solving Approach
+
+Investigate before fixing. When diagnosing a bug:
+1. Read the relevant module and trace what actually happens
+2. Identify the root cause, not a surface symptom
+3. Write a failing test that captures the correct behavior
+4. Fix, then re-run tests
+
+## Testing Discipline
+
+TDD is the default: write a failing test before any implementation. If you can't
+write the test, you don't yet understand the change. Details in
+`.claude/rules/python-testing.md`.
+
+## Editing Discipline
+
+A PostToolUse hook syntax-checks every Python file after Edit/Write/MultiEdit
+and blocks on a parse error, then runs `ruff` when it's installed. The hook
+covers `.py`, `.pyi`, and extensionless files with a python shebang.
+
+Type checking is not enforced by the hook — it needs the whole package and its
+dependencies resolved, which is a build-scale operation rather than a
+per-keystroke one. Run it via `make typecheck`.
+
+Formatting is likewise not enforced: a project picks its own line length and
+quote style, so blocking on an unconfigured default would impose a contested
+choice. Adopt one per project in `pyproject.toml`.
+
+## What Not to Do
+
+- Don't add features beyond what was asked
+- Don't refactor surrounding code when fixing a bug
+- Don't use a bare `except:` or swallow an exception without handling it
+- Don't use a mutable default argument (`def f(xs=[])`)
+- Don't add comments to code you didn't change
+- Don't commit `.env` files, credentials, or API keys — the pre-commit hook
+ catches common patterns but isn't a substitute for care
diff --git a/languages/python/claude/hooks/validate-python.sh b/languages/python/claude/hooks/validate-python.sh
new file mode 100755
index 0000000..e43ad77
--- /dev/null
+++ b/languages/python/claude/hooks/validate-python.sh
@@ -0,0 +1,95 @@
+#!/usr/bin/env bash
+# Validate Python files after Edit/Write/MultiEdit.
+# PostToolUse hook: receives tool-call JSON on stdin.
+#
+# On success: exit 0 silent.
+# On failure: emit JSON with hookSpecificOutput.additionalContext so Claude
+# sees a structured error in its context, THEN exit 2 to block the tool
+# pipeline. stderr still echoes the error for terminal visibility.
+#
+# Phase 1: syntax — python3 compiles the file. Always available wherever this
+# hook can meaningfully run, so it's the floor rather than an optional
+# gate: a file that doesn't parse is never worth passing on.
+# Phase 2: ruff — lint, when installed. Catches undefined names, unused
+# imports, and the rest of the pyflakes set. Absent ruff doesn't block
+# the edit, matching how the bash bundle treats shellcheck.
+#
+# Formatters (black, ruff format) are deliberately NOT enforced here. A project
+# picks its own line length and quote style, so blocking on an unconfigured
+# default would impose a contested choice. python.md recommends a formatter;
+# this hook enforces correctness.
+#
+# Type checking (mypy, pyright) is also out: it needs the whole package and its
+# dependencies resolved, which is a build-scale operation, not a per-keystroke
+# one. Run it via `make lint` / `make typecheck`.
+#
+# Scope: .py and .pyi files, plus extensionless files whose first line is a
+# python shebang (the CLI tools that fill a script-heavy repo carry no
+# extension).
+
+set -u
+
+# Emit a JSON failure payload and exit 2. Arguments:
+# $1 — short failure type (e.g. "PYTHON SYNTAX ERROR")
+# $2 — file path
+# $3 — tool output (error body)
+fail_json() {
+ local ctx
+ ctx="$(printf '%s: %s\n\n%s\n\nFix before proceeding.' "$1" "$2" "$3" \
+ | jq -Rs .)"
+ cat <<EOF
+{"hookSpecificOutput": {"hookEventName": "PostToolUse", "additionalContext": $ctx}}
+EOF
+ printf '%s: %s\n%s\n' "$1" "$2" "$3" >&2
+ exit 2
+}
+
+f="$(jq -r '.tool_input.file_path // .tool_response.filePath // empty')"
+[ -z "$f" ] && exit 0
+[ -f "$f" ] || exit 0
+
+# Is this a Python file? By extension, or by shebang when it has no extension.
+# Match on the basename, not the full path — a temp/parent dir can carry a dot
+# (e.g. my.project/) and misfire the "*.*" extension test.
+is_python=0
+base="${f##*/}"
+case "$base" in
+ *.py | *.pyi) is_python=1 ;;
+ *.*) is_python=0 ;; # some other extension — not ours
+ *)
+ # No extension: sniff the shebang.
+ if head -1 "$f" 2>/dev/null | grep -qE '^#!.*\bpython[0-9.]*\b'; then
+ is_python=1
+ fi
+ ;;
+esac
+[ "$is_python" -eq 1 ] || exit 0
+
+# No python3 on this machine — nothing to validate, don't block the edit.
+command -v python3 >/dev/null 2>&1 || exit 0
+
+# --- Phase 1: syntax ---
+# compile() rather than py_compile so no __pycache__ lands beside the source;
+# the hook is a checker and must not leave build artifacts in the tree.
+if ! out="$(python3 -c '
+import sys
+p = sys.argv[1]
+with open(p, "rb") as fh:
+ src = fh.read()
+try:
+ compile(src, p, "exec")
+except SyntaxError as e:
+ print(f"{e.msg} ({p}, line {e.lineno})", file=sys.stderr)
+ sys.exit(1)
+' "$f" 2>&1)"; then
+ fail_json "PYTHON SYNTAX ERROR" "$f" "$out"
+fi
+
+# --- Phase 2: lint (optional) ---
+command -v ruff >/dev/null 2>&1 || exit 0
+
+if ! out="$(ruff check "$f" 2>&1)"; then
+ fail_json "RUFF FAILED" "$f" "$out"
+fi
+
+exit 0
diff --git a/languages/python/claude/settings.json b/languages/python/claude/settings.json
new file mode 100644
index 0000000..9c6b2a9
--- /dev/null
+++ b/languages/python/claude/settings.json
@@ -0,0 +1,79 @@
+{
+ "attribution": {
+ "commit": "",
+ "pr": ""
+ },
+ "permissions": {
+ "allow": [
+ "Bash(make)",
+ "Bash(make help)",
+ "Bash(make targets)",
+ "Bash(make test)",
+ "Bash(make test *)",
+ "Bash(make lint)",
+ "Bash(make fmt)",
+ "Bash(make coverage)",
+ "Bash(make coverage-summary)",
+ "Bash(make typecheck)",
+ "Bash(pytest)",
+ "Bash(pytest *)",
+ "Bash(python3 -m pytest *)",
+ "Bash(ruff check *)",
+ "Bash(ruff format --diff *)",
+ "Bash(black --check *)",
+ "Bash(black --diff *)",
+ "Bash(mypy *)",
+ "Bash(python3 -m py_compile *)",
+ "Bash(python3 --version)",
+ "Bash(pip list)",
+ "Bash(pip show *)",
+ "Bash(git status)",
+ "Bash(git status *)",
+ "Bash(git diff)",
+ "Bash(git diff *)",
+ "Bash(git log)",
+ "Bash(git log *)",
+ "Bash(git show)",
+ "Bash(git show *)",
+ "Bash(git blame *)",
+ "Bash(git branch)",
+ "Bash(git branch -v)",
+ "Bash(git branch -a)",
+ "Bash(git branch --list *)",
+ "Bash(git remote)",
+ "Bash(git remote -v)",
+ "Bash(git remote show *)",
+ "Bash(git ls-files *)",
+ "Bash(git rev-parse *)",
+ "Bash(git cat-file *)",
+ "Bash(git stash list)",
+ "Bash(git stash show *)",
+ "Bash(jq *)",
+ "Bash(date)",
+ "Bash(date *)",
+ "Bash(which *)",
+ "Bash(file *)",
+ "Bash(ls)",
+ "Bash(ls *)",
+ "Bash(wc *)",
+ "Bash(du *)",
+ "Bash(readlink *)",
+ "Bash(realpath *)",
+ "Bash(basename *)",
+ "Bash(dirname *)"
+ ]
+ },
+ "hooks": {
+ "PostToolUse": [
+ {
+ "matcher": "Edit|Write|MultiEdit",
+ "hooks": [
+ {
+ "type": "command",
+ "command": "$CLAUDE_PROJECT_DIR/.claude/hooks/validate-python.sh"
+ }
+ ]
+ }
+ ]
+ }
+}
diff --git a/languages/python/githooks/pre-commit b/languages/python/githooks/pre-commit
new file mode 100755
index 0000000..03536db
--- /dev/null
+++ b/languages/python/githooks/pre-commit
@@ -0,0 +1,95 @@
+#!/usr/bin/env bash
+# Pre-commit hook: secret scan + syntax/lint check on staged Python files.
+# Use `git commit --no-verify` to bypass for confirmed false positives.
+
+set -u
+
+REPO_ROOT="$(git rev-parse --show-toplevel)"
+cd "$REPO_ROOT" || exit 1
+
+# --- 1. Secret scan ---
+# Patterns for common credentials. Scans only added lines in the staged diff.
+#
+# Two passes because case-sensitivity differs. AWS keys are uppercase, sk- keys
+# lowercase, PEM headers fixed, so those match case-SENSITIVELY: under -i,
+# AKIA[0-9A-Z]{16} matches any mixed-case 20-char run, which random base64 in an
+# embedded image blob hits ~6% of the time per 100KB and blocks real commits.
+# Only the keyword=value patterns need -i.
+SECRET_PATTERNS_CS='(AKIA[0-9A-Z]{16}|sk-[a-zA-Z0-9_-]{20,}|-----BEGIN (RSA|DSA|EC|OPENSSH|PGP)( PRIVATE)?( KEY| KEY BLOCK)?-----)'
+SECRET_PATTERNS_CI='(api[_-]?key|api[_-]?secret|auth[_-]?token|secret[_-]?key|bearer[_-]?token|access[_-]?token|password)[[:space:]]*[:=][[:space:]]*["'"'"'][^"'"'"']{16,}["'"'"']'
+
+# Read the diff on its own so a git failure is distinguishable from "grep
+# matched nothing". Both end in a non-zero status, but only one of them means
+# there is nothing to scan; piping them together and swallowing the result with
+# `|| true` made a broken git look like a clean commit — the scan searched an
+# empty string, found nothing, and the secret went in.
+if ! staged_diff="$(git diff --cached -U0 --diff-filter=AM)"; then
+ echo "pre-commit: cannot read the staged diff — refusing to skip the secret scan" >&2
+ exit 1
+fi
+
+# The greps keep their `|| true`: exiting 1 on no match is their normal result.
+added_lines="$(printf '%s\n' "$staged_diff" | grep '^+' | grep -v '^+++' || true)"
+
+cs_hits="$(printf '%s\n' "$added_lines" | grep -nE "$SECRET_PATTERNS_CS" || true)"
+ci_hits="$(printf '%s\n' "$added_lines" | grep -niE "$SECRET_PATTERNS_CI" || true)"
+# awk dedupes lines both passes matched, keeping first-seen order.
+secret_hits="$(printf '%s\n%s' "$cs_hits" "$ci_hits" \
+ | grep -v '^[[:space:]]*$' | awk '!seen[$0]++' || true)"
+
+if [ -n "$secret_hits" ]; then
+ echo "pre-commit: potential secret in staged changes:" >&2
+ echo "$secret_hits" >&2
+ echo "" >&2
+ echo "Review the lines above. If this is a false positive (test fixture, documentation)," >&2
+ echo "bypass with: git commit --no-verify" >&2
+ exit 1
+fi
+
+# --- 2. Syntax check on staged Python files ---
+# Same split as the secret scan above: a git failure must not read as "no files
+# staged", which would skip the language check silently.
+if ! staged_names="$(git diff --cached --name-only --diff-filter=AM)"; then
+ echo "pre-commit: cannot read the staged file list — refusing to skip the check" >&2
+ exit 1
+fi
+
+staged_py="$(printf '%s\n' "$staged_names" | grep -E '\.pyi?$' || true)"
+
+if [ -n "$staged_py" ] && command -v python3 >/dev/null 2>&1; then
+ failed=""
+ while IFS= read -r f; do
+ [ -z "$f" ] && continue
+ [ -f "$f" ] || continue
+ # compile() rather than py_compile so no __pycache__ lands in the tree.
+ if ! python3 -c 'import sys; compile(open(sys.argv[1], "rb").read(), sys.argv[1], "exec")' "$f" >/dev/null 2>&1; then
+ failed="${failed}${f}"$'\n'
+ fi
+ done <<< "$staged_py"
+
+ if [ -n "$failed" ]; then
+ printf 'pre-commit: Python syntax errors in staged files:\n\n%s\n' "$failed" >&2
+ echo "Run: python3 -m py_compile <file> to see the error, then re-stage." >&2
+ exit 1
+ fi
+fi
+
+# --- 3. ruff on staged Python files (when installed) ---
+if [ -n "$staged_py" ] && command -v ruff >/dev/null 2>&1; then
+ failed=""
+ while IFS= read -r f; do
+ [ -z "$f" ] && continue
+ [ -f "$f" ] || continue
+ if ! ruff check "$f" >/dev/null 2>&1; then
+ failed="${failed}${f}"$'\n'
+ fi
+ done <<< "$staged_py"
+
+ if [ -n "$failed" ]; then
+ printf 'pre-commit: ruff failed on staged files:\n\n%s\n' "$failed" >&2
+ echo "Run: ruff check <file> and fix the findings, then re-stage." >&2
+ exit 1
+ fi
+fi
+
+exit 0
diff --git a/languages/python/tests/pre-commit.bats b/languages/python/tests/pre-commit.bats
new file mode 100644
index 0000000..1ac82ee
--- /dev/null
+++ b/languages/python/tests/pre-commit.bats
@@ -0,0 +1,138 @@
+#!/usr/bin/env bats
+#
+# Tests for languages/python/githooks/pre-commit — the secret scan plus
+# syntax/lint gate that runs on staged Python files.
+#
+# The secret scan is the security-critical half and is language-independent, so
+# it gets the same coverage here as in the bash bundle: a real key blocks, a
+# clean diff passes, and the case-sensitivity split that keeps base64 blobs from
+# false-positiving is exercised directly.
+#
+# Each test builds a throwaway git repo, stages content, and runs the hook from
+# inside it — the hook reads `git diff --cached`, so a real index is required.
+
+HOOK="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/githooks/pre-commit"
+
+setup() {
+ TEST_DIR="$(mktemp -d -t pre-commit-py-bats.XXXXXX)"
+ cd "$TEST_DIR" || exit 1
+ git init -q .
+ git config user.email t@example.com
+ git config user.name Test
+ # A base commit so `git diff --cached` has a parent to diff against.
+ echo "seed" > seed.txt
+ git add seed.txt
+ git commit -qm seed
+}
+
+teardown() {
+ cd / || true
+ rm -rf "$TEST_DIR"
+}
+
+# ---- Normal ----------------------------------------------------------
+
+@test "pre-commit(py): a clean staged Python file passes (exit 0)" {
+ printf 'def f(x):\n return x + 1\n' > ok.py
+ git add ok.py
+ run bash "$HOOK"
+ [ "$status" -eq 0 ]
+}
+
+@test "pre-commit(py): an empty staging area passes (exit 0)" {
+ run bash "$HOOK"
+ [ "$status" -eq 0 ]
+}
+
+# ---- Error: the secret scan ------------------------------------------
+
+@test "pre-commit(py): an AWS key in a staged file blocks (exit 1)" {
+ printf 'KEY = "AKIAIOSFODNN7EXAMPLE"\n' > conf.py
+ git add conf.py
+ run bash "$HOOK"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"potential secret"* ]]
+}
+
+@test "pre-commit(py): an sk- style token blocks (exit 1)" {
+ printf 'TOKEN = "sk-abcdefghijklmnopqrstuvwxyz0123"\n' > conf.py
+ git add conf.py
+ run bash "$HOOK"
+ [ "$status" -eq 1 ]
+}
+
+@test "pre-commit(py): a quoted api_key assignment blocks (exit 1)" {
+ printf 'api_key = "abcdefghijklmnopqrstuvwxyz"\n' > conf.py
+ git add conf.py
+ run bash "$HOOK"
+ [ "$status" -eq 1 ]
+}
+
+@test "pre-commit(py): a private-key header blocks (exit 1)" {
+ printf 'PEM = """-----BEGIN RSA PRIVATE KEY-----"""\n' > conf.py
+ git add conf.py
+ run bash "$HOOK"
+ [ "$status" -eq 1 ]
+}
+
+# ---- Boundary: the case-sensitivity split ----------------------------
+
+@test "pre-commit(py): a mixed-case base64 blob does NOT false-positive" {
+ # The AWS pattern is uppercase-only by design. Under -i it would match any
+ # 20-char mixed-case run, which random base64 hits often enough to block
+ # real commits. This is the regression test for that split.
+ printf 'BLOB = "AKIAbcdefGHIJklmnOPqr0123456789abcdefGHIJ"\n' > data.py
+ git add data.py
+ run bash "$HOOK"
+ [ "$status" -eq 0 ]
+}
+
+@test "pre-commit(py): a short quoted password value does NOT block" {
+ # The keyword patterns require 16+ chars, so a placeholder stays quiet.
+ printf 'password = "short"\n' > conf.py
+ git add conf.py
+ run bash "$HOOK"
+ [ "$status" -eq 0 ]
+}
+
+@test "pre-commit(py): a secret only in a REMOVED line does not block" {
+ printf 'KEY = "AKIAIOSFODNN7EXAMPLE"\n' > conf.py
+ git add conf.py
+ git commit -qm "add key"
+ rm conf.py
+ git add -A
+ run bash "$HOOK"
+ [ "$status" -eq 0 ]
+}
+
+# ---- Error: the syntax gate ------------------------------------------
+
+@test "pre-commit(py): a staged Python syntax error blocks (exit 1)" {
+ printf 'def f(:\n return 1\n' > bad.py
+ git add bad.py
+ run bash "$HOOK"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"syntax"* ]]
+}
+
+@test "pre-commit(py): a .pyi stub with a syntax error blocks (exit 1)" {
+ printf 'def f( -> int: ...\n' > bad.pyi
+ git add bad.pyi
+ run bash "$HOOK"
+ [ "$status" -eq 1 ]
+}
+
+@test "pre-commit(py): a broken NON-Python file does not trip the syntax gate" {
+ printf 'this is (((not python\n' > notes.txt
+ git add notes.txt
+ run bash "$HOOK"
+ [ "$status" -eq 0 ]
+}
+
+@test "pre-commit(py): the syntax gate leaves no __pycache__ in the repo" {
+ printf 'def f():\n return 1\n' > ok.py
+ git add ok.py
+ run bash "$HOOK"
+ [ "$status" -eq 0 ]
+ [ ! -d __pycache__ ]
+}
diff --git a/languages/python/tests/validate-python.bats b/languages/python/tests/validate-python.bats
new file mode 100644
index 0000000..b5e4957
--- /dev/null
+++ b/languages/python/tests/validate-python.bats
@@ -0,0 +1,117 @@
+#!/usr/bin/env bats
+#
+# Tests for languages/python/claude/hooks/validate-python.sh — the PostToolUse
+# hook that syntax-checks edited Python files and blocks on a violation.
+#
+# The hook reads tool-call JSON on stdin and extracts the file path, so each
+# test pipes a JSON payload naming a real file it wrote into a temp dir.
+#
+# The syntax gate is python3's own compiler, which is present wherever the hook
+# can meaningfully run, so those tests never skip. The lint gate (ruff) is
+# optional and its tests skip when it's absent, matching the bash bundle's
+# treatment of shellcheck.
+
+HOOK="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/claude/hooks/validate-python.sh"
+
+setup() {
+ TEST_DIR="$(mktemp -d -t validate-python-bats.XXXXXX)"
+}
+
+teardown() {
+ rm -rf "$TEST_DIR"
+}
+
+payload() {
+ printf '{"tool_input": {"file_path": "%s"}}' "$1"
+}
+
+# ---- Normal ----------------------------------------------------------
+
+@test "validate-python: a clean .py file passes silently (exit 0)" {
+ printf 'def f(x):\n return x + 1\n' > "$TEST_DIR/clean.py"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/clean.py")"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "validate-python: a .pyi stub is validated too" {
+ printf 'def f(x: int) -> int: ...\n' > "$TEST_DIR/clean.pyi"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/clean.pyi")"
+ [ "$status" -eq 0 ]
+}
+
+# ---- Error -----------------------------------------------------------
+
+@test "validate-python: a syntax error blocks (exit 2, names the failure)" {
+ printf 'def f(:\n return 1\n' > "$TEST_DIR/bad.py"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/bad.py")"
+ [ "$status" -eq 2 ]
+ [[ "$output" == *"SYNTAX"* ]]
+}
+
+@test "validate-python: the block payload is valid JSON carrying the context" {
+ printf 'def f(:\n' > "$TEST_DIR/bad.py"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/bad.py")"
+ [ "$status" -eq 2 ]
+ # The first line of stdout must parse as JSON and carry the hook event name.
+ echo "$output" | head -1 | jq -e '.hookSpecificOutput.hookEventName == "PostToolUse"'
+}
+
+@test "validate-python: a ruff violation blocks when ruff is installed" {
+ command -v ruff >/dev/null 2>&1 || skip "ruff not installed"
+ # F821: reference to an undefined name — syntactically valid, lint-caught.
+ printf 'def f():\n return undefined_name\n' > "$TEST_DIR/lint.py"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/lint.py")"
+ [ "$status" -eq 2 ]
+ [[ "$output" == *"RUFF"* ]]
+}
+
+# ---- Boundary --------------------------------------------------------
+
+@test "validate-python: a non-Python file is ignored (exit 0)" {
+ printf 'not python at all (((\n' > "$TEST_DIR/notes.txt"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/notes.txt")"
+ [ "$status" -eq 0 ]
+}
+
+@test "validate-python: an extensionless file with a python shebang is validated" {
+ printf '#!/usr/bin/env python3\ndef f(:\n' > "$TEST_DIR/cli-tool"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/cli-tool")"
+ [ "$status" -eq 2 ]
+}
+
+@test "validate-python: an extensionless non-python file is ignored (exit 0)" {
+ printf '#!/usr/bin/env bash\necho hi\n' > "$TEST_DIR/shell-tool"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/shell-tool")"
+ [ "$status" -eq 0 ]
+}
+
+@test "validate-python: a dotted parent directory does not misfire the extension test" {
+ mkdir -p "$TEST_DIR/my.project"
+ printf '#!/usr/bin/env python3\ndef f(:\n' > "$TEST_DIR/my.project/cli-tool"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/my.project/cli-tool")"
+ [ "$status" -eq 2 ]
+}
+
+@test "validate-python: empty file_path is a no-op (exit 0)" {
+ run bash "$HOOK" <<< '{"tool_input": {}}'
+ [ "$status" -eq 0 ]
+}
+
+@test "validate-python: a missing file is a no-op (exit 0)" {
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/does-not-exist.py")"
+ [ "$status" -eq 0 ]
+}
+
+@test "validate-python: an empty .py file passes (valid, compiles to nothing)" {
+ : > "$TEST_DIR/empty.py"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/empty.py")"
+ [ "$status" -eq 0 ]
+}
+
+@test "validate-python: compiling leaves no __pycache__ beside the file" {
+ printf 'def f():\n return 1\n' > "$TEST_DIR/clean.py"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/clean.py")"
+ [ "$status" -eq 0 ]
+ [ ! -d "$TEST_DIR/__pycache__" ]
+}
diff --git a/languages/typescript/CLAUDE.md b/languages/typescript/CLAUDE.md
new file mode 100644
index 0000000..1794115
--- /dev/null
+++ b/languages/typescript/CLAUDE.md
@@ -0,0 +1,82 @@
+# CLAUDE.md
+
+## Project
+
+TypeScript/JavaScript project. Customize this section with your own
+description, layout, and conventions.
+
+**Typical layout:**
+- `src/` — source modules
+- `tests/` or `*.test.ts` beside the source — test files
+- `package.json` — scripts and dependencies
+- `tsconfig.json` — compiler options
+
+## Build & Test Commands
+
+If the project has a Makefile, document targets here. Common pattern:
+
+```bash
+make test # run the test suite
+make coverage # suite + coverage report
+make typecheck # tsc --noEmit across the project
+make lint # eslint
+make build # production build
+```
+
+Direct equivalents: `npm test`, `npx tsc --noEmit`, `npx eslint src/`,
+`npx prettier --check .`, `node --test`.
+
+## Language Rules
+
+See rule files in `.claude/rules/`:
+- `typescript-testing.md` — test conventions and mocking discipline
+- `verification.md` — verify-before-claim-done discipline
+
+## Git Workflow
+
+Commit conventions: see `.claude/rules/commits.md` (author identity,
+no AI attribution, message format).
+
+Pre-commit hook in `githooks/` scans for secrets and parse-checks staged TS/JS.
+Activate on a fresh clone with `git config core.hooksPath githooks`.
+
+## Problem-Solving Approach
+
+Investigate before fixing. When diagnosing a bug:
+1. Read the relevant module and trace what actually happens
+2. Identify the root cause, not a surface symptom
+3. Write a failing test that captures the correct behavior
+4. Fix, then re-run tests
+
+## Testing Discipline
+
+TDD is the default: write a failing test before any implementation. If you can't
+write the test, you don't yet understand the change. Details in
+`.claude/rules/typescript-testing.md`.
+
+## Editing Discipline
+
+A PostToolUse hook parse-checks every TS/JS file after Edit/Write/MultiEdit and
+blocks on a syntax error. It covers `.ts`, `.tsx`, `.mts`, `.cts`, `.js`,
+`.jsx`, `.mjs`, and `.cjs`.
+
+Two checkers, because one tool can't do both jobs: `node --check` for
+JavaScript, `tsc` filtered to syntax diagnostics for TypeScript. Do not
+substitute `node --check` for the TypeScript path — it ignores
+`--experimental-strip-types`, so it rejects valid TypeScript and accepts broken
+TypeScript (measured on node v26.4.0).
+
+Full type checking is not enforced by the hook: it needs the whole project graph
+and its dependencies resolved, which is a build-scale operation rather than a
+per-keystroke one. Run it via `make typecheck`. Formatting is likewise not
+enforced; adopt a style per project in the project's own config.
+
+## What Not to Do
+
+- Don't add features beyond what was asked
+- Don't refactor surrounding code when fixing a bug
+- Don't reach for `any` to silence a type error — narrow the type instead
+- Don't use `==` where `===` is meant
+- Don't add comments to code you didn't change
+- Don't commit `.env` files, credentials, or API keys — the pre-commit hook
+ catches common patterns but isn't a substitute for care
diff --git a/languages/typescript/claude/hooks/validate-typescript.sh b/languages/typescript/claude/hooks/validate-typescript.sh
new file mode 100755
index 0000000..b76f1df
--- /dev/null
+++ b/languages/typescript/claude/hooks/validate-typescript.sh
@@ -0,0 +1,94 @@
+#!/usr/bin/env bash
+# Validate TypeScript/JavaScript files after Edit/Write/MultiEdit.
+# PostToolUse hook: receives tool-call JSON on stdin.
+#
+# On success: exit 0 silent.
+# On failure: emit JSON with hookSpecificOutput.additionalContext so Claude
+# sees a structured error in its context, THEN exit 2 to block the tool
+# pipeline. stderr still echoes the error for terminal visibility.
+#
+# Gate: parseability. A file that doesn't parse is never worth passing on.
+# Full type checking is deliberately NOT enforced here — it needs the whole
+# project graph and its dependencies resolved, which is a build-scale
+# operation, not a per-keystroke one. A type error that parses cleanly passes
+# this hook; `make typecheck` / `tsc --noEmit` over the project owns it.
+#
+# Formatting (prettier) is also out: a project picks its own style, so blocking
+# on an unconfigured default would impose a contested choice.
+#
+# Two checkers, because one tool can't do both jobs:
+#
+# .js/.jsx/.mjs/.cjs → `node --check`, a straight parse.
+# .ts/.tsx/.mts/.cts → `tsc`, filtered to syntax-category diagnostics.
+#
+# `node --check` must NOT be used on TypeScript. It ignores
+# --experimental-strip-types, so it is wrong in *both* directions: it rejects
+# valid TS (an `interface` declaration reads as a syntax error) and accepts
+# broken TS (a genuinely unparseable file exits 0). Measured on node v26.4.0,
+# 2026-07-23. tsc is the only correct parser for these extensions.
+#
+# The tsc call is filtered to TS1xxx codes, which is TypeScript's syntactic
+# diagnostic range; TS2xxx and up are semantic (type) errors and are out of
+# scope by the paragraph above. Without the filter this hook would block every
+# unresolved import in a file whose dependencies aren't installed yet.
+
+set -u
+
+# Emit a JSON failure payload and exit 2. Arguments:
+# $1 — short failure type (e.g. "TYPESCRIPT SYNTAX ERROR")
+# $2 — file path
+# $3 — tool output (error body)
+fail_json() {
+ local ctx
+ ctx="$(printf '%s: %s\n\n%s\n\nFix before proceeding.' "$1" "$2" "$3" \
+ | jq -Rs .)"
+ cat <<EOF
+{"hookSpecificOutput": {"hookEventName": "PostToolUse", "additionalContext": $ctx}}
+EOF
+ printf '%s: %s\n%s\n' "$1" "$2" "$3" >&2
+ exit 2
+}
+
+f="$(jq -r '.tool_input.file_path // .tool_response.filePath // empty')"
+[ -z "$f" ] && exit 0
+[ -f "$f" ] || exit 0
+
+# Classify by extension. Match on the basename, not the full path — a parent
+# dir can carry a dot (e.g. my.project/) and confuse a path-wide match.
+kind=""
+base="${f##*/}"
+case "$base" in
+ *.ts | *.tsx | *.mts | *.cts) kind="ts" ;;
+ *.js | *.jsx | *.mjs | *.cjs) kind="js" ;;
+ *) exit 0 ;;
+esac
+
+if [ "$kind" = "js" ]; then
+ command -v node >/dev/null 2>&1 || exit 0
+ if ! out="$(node --check "$f" 2>&1)"; then
+ fail_json "JAVASCRIPT SYNTAX ERROR" "$f" "$out"
+ fi
+ exit 0
+fi
+
+# TypeScript. Prefer a project-local tsc so the project's own version decides,
+# falling back to one on PATH.
+tsc_bin=""
+if [ -x "./node_modules/.bin/tsc" ]; then
+ tsc_bin="./node_modules/.bin/tsc"
+elif command -v tsc >/dev/null 2>&1; then
+ tsc_bin="tsc"
+else
+ exit 0 # no TypeScript compiler available — don't block the edit
+fi
+
+# --moduleDetection force so a file with no import/export still parses as a
+# module rather than tripping global-scope collisions against lib types.
+out="$("$tsc_bin" --noEmit --skipLibCheck --target es2022 --moduleDetection force "$f" 2>&1 || true)"
+syntax_errors="$(printf '%s\n' "$out" | grep -E 'error TS1[0-9]{3}:' || true)"
+
+if [ -n "$syntax_errors" ]; then
+ fail_json "TYPESCRIPT SYNTAX ERROR" "$f" "$syntax_errors"
+fi
+
+exit 0
diff --git a/languages/typescript/claude/settings.json b/languages/typescript/claude/settings.json
new file mode 100644
index 0000000..f4c9211
--- /dev/null
+++ b/languages/typescript/claude/settings.json
@@ -0,0 +1,80 @@
+{
+ "attribution": {
+ "commit": "",
+ "pr": ""
+ },
+ "permissions": {
+ "allow": [
+ "Bash(make)",
+ "Bash(make help)",
+ "Bash(make targets)",
+ "Bash(make test)",
+ "Bash(make test *)",
+ "Bash(make lint)",
+ "Bash(make fmt)",
+ "Bash(make coverage)",
+ "Bash(make coverage-summary)",
+ "Bash(make typecheck)",
+ "Bash(make build)",
+ "Bash(npm test)",
+ "Bash(npm test *)",
+ "Bash(npm run *)",
+ "Bash(npm ci)",
+ "Bash(npm ls *)",
+ "Bash(node --check *)",
+ "Bash(node --test *)",
+ "Bash(node --version)",
+ "Bash(tsc --noEmit *)",
+ "Bash(npx tsc --noEmit *)",
+ "Bash(eslint *)",
+ "Bash(prettier --check *)",
+ "Bash(git status)",
+ "Bash(git status *)",
+ "Bash(git diff)",
+ "Bash(git diff *)",
+ "Bash(git log)",
+ "Bash(git log *)",
+ "Bash(git show)",
+ "Bash(git show *)",
+ "Bash(git blame *)",
+ "Bash(git branch)",
+ "Bash(git branch -v)",
+ "Bash(git branch -a)",
+ "Bash(git branch --list *)",
+ "Bash(git remote)",
+ "Bash(git remote -v)",
+ "Bash(git remote show *)",
+ "Bash(git ls-files *)",
+ "Bash(git rev-parse *)",
+ "Bash(git cat-file *)",
+ "Bash(git stash list)",
+ "Bash(git stash show *)",
+ "Bash(jq *)",
+ "Bash(date)",
+ "Bash(date *)",
+ "Bash(which *)",
+ "Bash(file *)",
+ "Bash(ls)",
+ "Bash(ls *)",
+ "Bash(wc *)",
+ "Bash(du *)",
+ "Bash(readlink *)",
+ "Bash(realpath *)",
+ "Bash(basename *)",
+ "Bash(dirname *)"
+ ]
+ },
+ "hooks": {
+ "PostToolUse": [
+ {
+ "matcher": "Edit|Write|MultiEdit",
+ "hooks": [
+ {
+ "type": "command",
+ "command": "$CLAUDE_PROJECT_DIR/.claude/hooks/validate-typescript.sh"
+ }
+ ]
+ }
+ ]
+ }
+}
diff --git a/languages/typescript/githooks/pre-commit b/languages/typescript/githooks/pre-commit
new file mode 100755
index 0000000..fd494d2
--- /dev/null
+++ b/languages/typescript/githooks/pre-commit
@@ -0,0 +1,107 @@
+#!/usr/bin/env bash
+# Pre-commit hook: secret scan + syntax check on staged TypeScript/JavaScript files.
+# Use `git commit --no-verify` to bypass for confirmed false positives.
+
+set -u
+
+REPO_ROOT="$(git rev-parse --show-toplevel)"
+cd "$REPO_ROOT" || exit 1
+
+# --- 1. Secret scan ---
+# Patterns for common credentials. Scans only added lines in the staged diff.
+#
+# Two passes because case-sensitivity differs. AWS keys are uppercase, sk- keys
+# lowercase, PEM headers fixed, so those match case-SENSITIVELY: under -i,
+# AKIA[0-9A-Z]{16} matches any mixed-case 20-char run, which random base64 in an
+# embedded image blob hits ~6% of the time per 100KB and blocks real commits.
+# Only the keyword=value patterns need -i.
+SECRET_PATTERNS_CS='(AKIA[0-9A-Z]{16}|sk-[a-zA-Z0-9_-]{20,}|-----BEGIN (RSA|DSA|EC|OPENSSH|PGP)( PRIVATE)?( KEY| KEY BLOCK)?-----)'
+SECRET_PATTERNS_CI='(api[_-]?key|api[_-]?secret|auth[_-]?token|secret[_-]?key|bearer[_-]?token|access[_-]?token|password)[[:space:]]*[:=][[:space:]]*["'"'"'][^"'"'"']{16,}["'"'"']'
+
+# Read the diff on its own so a git failure is distinguishable from "grep
+# matched nothing". Both end in a non-zero status, but only one of them means
+# there is nothing to scan; piping them together and swallowing the result with
+# `|| true` made a broken git look like a clean commit — the scan searched an
+# empty string, found nothing, and the secret went in.
+if ! staged_diff="$(git diff --cached -U0 --diff-filter=AM)"; then
+ echo "pre-commit: cannot read the staged diff — refusing to skip the secret scan" >&2
+ exit 1
+fi
+
+# The greps keep their `|| true`: exiting 1 on no match is their normal result.
+added_lines="$(printf '%s\n' "$staged_diff" | grep '^+' | grep -v '^+++' || true)"
+
+cs_hits="$(printf '%s\n' "$added_lines" | grep -nE "$SECRET_PATTERNS_CS" || true)"
+ci_hits="$(printf '%s\n' "$added_lines" | grep -niE "$SECRET_PATTERNS_CI" || true)"
+# awk dedupes lines both passes matched, keeping first-seen order.
+secret_hits="$(printf '%s\n%s' "$cs_hits" "$ci_hits" \
+ | grep -v '^[[:space:]]*$' | awk '!seen[$0]++' || true)"
+
+if [ -n "$secret_hits" ]; then
+ echo "pre-commit: potential secret in staged changes:" >&2
+ echo "$secret_hits" >&2
+ echo "" >&2
+ echo "Review the lines above. If this is a false positive (test fixture, documentation)," >&2
+ echo "bypass with: git commit --no-verify" >&2
+ exit 1
+fi
+
+# --- 2. Syntax check on staged TS/JS files ---
+# Two checkers, because one tool can't do both jobs. `node --check` ignores
+# --experimental-strip-types, so on TypeScript it is wrong in BOTH directions:
+# it rejects valid TS (an `interface` reads as a syntax error) and accepts
+# broken TS. Measured on node v26.4.0, 2026-07-23. tsc is the only correct
+# parser for .ts; node is correct and much faster for .js.
+# Same split as the secret scan above: a git failure must not read as "no files
+# staged", which would skip the language check silently.
+if ! staged_names="$(git diff --cached --name-only --diff-filter=AM)"; then
+ echo "pre-commit: cannot read the staged file list — refusing to skip the check" >&2
+ exit 1
+fi
+
+staged_js="$(printf '%s\n' "$staged_names" | grep -E '\.(js|jsx|mjs|cjs)$' || true)"
+staged_ts="$(printf '%s\n' "$staged_names" | grep -E '\.(ts|tsx|mts|cts)$' || true)"
+
+failed=""
+
+if [ -n "$staged_js" ] && command -v node >/dev/null 2>&1; then
+ while IFS= read -r f; do
+ [ -z "$f" ] && continue
+ [ -f "$f" ] || continue
+ if ! node --check "$f" >/dev/null 2>&1; then
+ failed="${failed}${f}"$'\n'
+ fi
+ done <<< "$staged_js"
+fi
+
+if [ -n "$staged_ts" ]; then
+ tsc_bin=""
+ if [ -x "./node_modules/.bin/tsc" ]; then
+ tsc_bin="./node_modules/.bin/tsc"
+ elif command -v tsc >/dev/null 2>&1; then
+ tsc_bin="tsc"
+ fi
+
+ if [ -n "$tsc_bin" ]; then
+ while IFS= read -r f; do
+ [ -z "$f" ] && continue
+ [ -f "$f" ] || continue
+ # Filter to TS1xxx, TypeScript's syntactic diagnostic range. TS2xxx and
+ # up are type errors, which need the whole project graph and are the
+ # build's job, not this hook's.
+ out="$("$tsc_bin" --noEmit --skipLibCheck --target es2022 \
+ --moduleDetection force "$f" 2>&1 || true)"
+ if printf '%s\n' "$out" | grep -qE 'error TS1[0-9]{3}:'; then
+ failed="${failed}${f}"$'\n'
+ fi
+ done <<< "$staged_ts"
+ fi
+fi
+
+if [ -n "$failed" ]; then
+ printf 'pre-commit: syntax errors in staged files:\n\n%s\n' "$failed" >&2
+ echo "Fix the parse errors above, then re-stage." >&2
+ exit 1
+fi
+
+exit 0
diff --git a/languages/typescript/tests/pre-commit.bats b/languages/typescript/tests/pre-commit.bats
new file mode 100644
index 0000000..5519baa
--- /dev/null
+++ b/languages/typescript/tests/pre-commit.bats
@@ -0,0 +1,150 @@
+#!/usr/bin/env bats
+#
+# Tests for languages/typescript/githooks/pre-commit — the secret scan plus
+# syntax/lint gate that runs on staged TS/JS files.
+#
+# The secret scan is the security-critical half and is language-independent, so
+# it gets the same coverage here as in the bash bundle: a real key blocks, a
+# clean diff passes, and the case-sensitivity split that keeps base64 blobs from
+# false-positiving is exercised directly.
+#
+# Each test builds a throwaway git repo, stages content, and runs the hook from
+# inside it — the hook reads `git diff --cached`, so a real index is required.
+
+HOOK="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/githooks/pre-commit"
+
+setup() {
+ TEST_DIR="$(mktemp -d -t pre-commit-ts-bats.XXXXXX)"
+ cd "$TEST_DIR" || exit 1
+ git init -q .
+ git config user.email t@example.com
+ git config user.name Test
+ # A base commit so `git diff --cached` has a parent to diff against.
+ echo "seed" > seed.txt
+ git add seed.txt
+ git commit -qm seed
+}
+
+teardown() {
+ cd / || true
+ rm -rf "$TEST_DIR"
+}
+
+# ---- Normal ----------------------------------------------------------
+
+@test "pre-commit(ts): a clean staged JS file passes (exit 0)" {
+ printf 'export const f = (x) => x + 1;\n' > ok.js
+ git add ok.js
+ run bash "$HOOK"
+ [ "$status" -eq 0 ]
+}
+
+@test "pre-commit(ts): an empty staging area passes (exit 0)" {
+ run bash "$HOOK"
+ [ "$status" -eq 0 ]
+}
+
+# ---- Error: the secret scan ------------------------------------------
+
+@test "pre-commit(ts): an AWS key in a staged file blocks (exit 1)" {
+ printf 'const KEY = "AKIAIOSFODNN7EXAMPLE";\n' > conf.ts
+ git add conf.ts
+ run bash "$HOOK"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"potential secret"* ]]
+}
+
+@test "pre-commit(ts): an sk- style token blocks (exit 1)" {
+ printf 'const TOKEN = "sk-abcdefghijklmnopqrstuvwxyz0123";\n' > conf.ts
+ git add conf.ts
+ run bash "$HOOK"
+ [ "$status" -eq 1 ]
+}
+
+@test "pre-commit(ts): a quoted api_key assignment blocks (exit 1)" {
+ printf 'const api_key = "abcdefghijklmnopqrstuvwxyz";\n' > conf.ts
+ git add conf.ts
+ run bash "$HOOK"
+ [ "$status" -eq 1 ]
+}
+
+@test "pre-commit(ts): a private-key header blocks (exit 1)" {
+ printf 'const PEM = "-----BEGIN RSA PRIVATE KEY-----";\n' > conf.ts
+ git add conf.ts
+ run bash "$HOOK"
+ [ "$status" -eq 1 ]
+}
+
+# ---- Boundary: the case-sensitivity split ----------------------------
+
+@test "pre-commit(ts): a mixed-case base64 blob does NOT false-positive" {
+ # The AWS pattern is uppercase-only by design. Under -i it would match any
+ # 20-char mixed-case run, which random base64 hits often enough to block
+ # real commits. This is the regression test for that split.
+ printf 'const BLOB = "AKIAbcdefGHIJklmnOPqr0123456789abcdefGHIJ";\n' > data.ts
+ git add data.ts
+ run bash "$HOOK"
+ [ "$status" -eq 0 ]
+}
+
+@test "pre-commit(ts): a short quoted password value does NOT block" {
+ # The keyword patterns require 16+ chars, so a placeholder stays quiet.
+ printf 'const password = "short";\n' > conf.ts
+ git add conf.ts
+ run bash "$HOOK"
+ [ "$status" -eq 0 ]
+}
+
+@test "pre-commit(ts): a secret only in a REMOVED line does not block" {
+ printf 'const KEY = "AKIAIOSFODNN7EXAMPLE";\n' > conf.ts
+ git add conf.ts
+ git commit -qm "add key"
+ rm conf.ts
+ git add -A
+ run bash "$HOOK"
+ [ "$status" -eq 0 ]
+}
+
+# ---- Error: the syntax gate ------------------------------------------
+
+@test "pre-commit(ts): a staged JS syntax error blocks (exit 1)" {
+ command -v node >/dev/null 2>&1 || skip "node not installed"
+ printf 'const x = ;\n' > bad.js
+ git add bad.js
+ run bash "$HOOK"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"syntax"* ]]
+}
+
+@test "pre-commit(ts): a staged TS syntax error blocks (exit 1)" {
+ command -v tsc >/dev/null 2>&1 || skip "tsc not installed"
+ printf 'export function f( {\n return 1;\n}\n' > bad.ts
+ git add bad.ts
+ run bash "$HOOK"
+ [ "$status" -eq 1 ]
+}
+
+@test "pre-commit(ts): valid TS-only syntax is NOT read as broken JS" {
+ command -v tsc >/dev/null 2>&1 || skip "tsc not installed"
+ # The regression guard for the node --check trap: `node --check` rejects
+ # valid TypeScript, so using it on .ts would block every real commit.
+ printf 'interface P { a: string }\nexport const p: P = { a: "x" };\n' > types.ts
+ git add types.ts
+ run bash "$HOOK"
+ [ "$status" -eq 0 ]
+}
+
+@test "pre-commit(ts): a TYPE error that parses does not block (out of scope)" {
+ command -v tsc >/dev/null 2>&1 || skip "tsc not installed"
+ printf 'const n: number = "nope";\nexport { n };\n' > typeerr.ts
+ git add typeerr.ts
+ run bash "$HOOK"
+ [ "$status" -eq 0 ]
+}
+
+@test "pre-commit(ts): a broken NON-TS/JS file does not trip the syntax gate" {
+ printf 'this is (((not javascript\n' > notes.txt
+ git add notes.txt
+ run bash "$HOOK"
+ [ "$status" -eq 0 ]
+}
diff --git a/languages/typescript/tests/validate-typescript.bats b/languages/typescript/tests/validate-typescript.bats
new file mode 100644
index 0000000..c5da5d4
--- /dev/null
+++ b/languages/typescript/tests/validate-typescript.bats
@@ -0,0 +1,125 @@
+#!/usr/bin/env bats
+#
+# Tests for languages/typescript/claude/hooks/validate-typescript.sh — the
+# PostToolUse hook that syntax-checks edited TS/JS files and blocks on a
+# violation.
+#
+# The hook reads tool-call JSON on stdin and extracts the file path, so each
+# test pipes a JSON payload naming a real file it wrote into a temp dir.
+#
+# The syntax gate needs node, so those tests skip when node is absent. Full
+# type checking is deliberately out of scope for the hook (it needs the whole
+# project graph), so a type error that is syntactically valid must pass.
+
+HOOK="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/claude/hooks/validate-typescript.sh"
+
+setup() {
+ TEST_DIR="$(mktemp -d -t validate-ts-bats.XXXXXX)"
+}
+
+teardown() {
+ rm -rf "$TEST_DIR"
+}
+
+payload() {
+ printf '{"tool_input": {"file_path": "%s"}}' "$1"
+}
+
+# ---- Normal ----------------------------------------------------------
+
+@test "validate-typescript: a clean .ts file passes silently (exit 0)" {
+ command -v node >/dev/null 2>&1 || skip "node not installed"
+ printf 'export function f(x: number): number {\n return x + 1;\n}\n' > "$TEST_DIR/clean.ts"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/clean.ts")"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "validate-typescript: a clean .js file passes silently (exit 0)" {
+ command -v node >/dev/null 2>&1 || skip "node not installed"
+ printf 'export function f(x) {\n return x + 1;\n}\n' > "$TEST_DIR/clean.js"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/clean.js")"
+ [ "$status" -eq 0 ]
+}
+
+@test "validate-typescript: a .tsx file is validated too" {
+ command -v node >/dev/null 2>&1 || skip "node not installed"
+ printf 'export const A = 1;\n' > "$TEST_DIR/clean.tsx"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/clean.tsx")"
+ [ "$status" -eq 0 ]
+}
+
+# ---- Error -----------------------------------------------------------
+
+@test "validate-typescript: a syntax error blocks (exit 2, names the failure)" {
+ command -v node >/dev/null 2>&1 || skip "node not installed"
+ printf 'export function f( {\n return 1;\n}\n' > "$TEST_DIR/bad.ts"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/bad.ts")"
+ [ "$status" -eq 2 ]
+ [[ "$output" == *"SYNTAX"* ]]
+}
+
+@test "validate-typescript: the block payload is valid JSON carrying the context" {
+ command -v node >/dev/null 2>&1 || skip "node not installed"
+ printf 'const x = ;\n' > "$TEST_DIR/bad.js"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/bad.js")"
+ [ "$status" -eq 2 ]
+ echo "$output" | head -1 | jq -e '.hookSpecificOutput.hookEventName == "PostToolUse"'
+}
+
+# ---- Boundary --------------------------------------------------------
+
+@test "validate-typescript: a type error that parses is NOT blocked (out of scope)" {
+ command -v node >/dev/null 2>&1 || skip "node not installed"
+ # Assigning a string to a number is a type error, not a syntax error. The
+ # hook checks parseability only; tsc over the project graph owns this.
+ printf 'const n: number = "not a number";\nexport { n };\n' > "$TEST_DIR/typeerr.ts"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/typeerr.ts")"
+ [ "$status" -eq 0 ]
+}
+
+@test "validate-typescript: TS-only syntax in a .ts file parses (not read as JS)" {
+ command -v node >/dev/null 2>&1 || skip "node not installed"
+ # Interfaces and type annotations are invalid JS. Stripping types must happen
+ # before the parse, or every real .ts file would be reported as broken.
+ printf 'interface P { a: string }\nexport const p: P = { a: "x" };\n' > "$TEST_DIR/types.ts"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/types.ts")"
+ [ "$status" -eq 0 ]
+}
+
+@test "validate-typescript: a non-TS/JS file is ignored (exit 0)" {
+ printf 'not javascript at all (((\n' > "$TEST_DIR/notes.txt"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/notes.txt")"
+ [ "$status" -eq 0 ]
+}
+
+@test "validate-typescript: a .json file is ignored (exit 0)" {
+ printf '{"a": 1}\n' > "$TEST_DIR/data.json"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/data.json")"
+ [ "$status" -eq 0 ]
+}
+
+@test "validate-typescript: empty file_path is a no-op (exit 0)" {
+ run bash "$HOOK" <<< '{"tool_input": {}}'
+ [ "$status" -eq 0 ]
+}
+
+@test "validate-typescript: a missing file is a no-op (exit 0)" {
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/does-not-exist.ts")"
+ [ "$status" -eq 0 ]
+}
+
+@test "validate-typescript: an empty .ts file passes" {
+ command -v node >/dev/null 2>&1 || skip "node not installed"
+ : > "$TEST_DIR/empty.ts"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/empty.ts")"
+ [ "$status" -eq 0 ]
+}
+
+@test "validate-typescript: a file in a dotted parent dir is still matched" {
+ command -v node >/dev/null 2>&1 || skip "node not installed"
+ mkdir -p "$TEST_DIR/my.project"
+ printf 'const x = ;\n' > "$TEST_DIR/my.project/bad.ts"
+ run bash "$HOOK" <<< "$(payload "$TEST_DIR/my.project/bad.ts")"
+ [ "$status" -eq 2 ]
+}
diff --git a/publish/SKILL.md b/publish/SKILL.md
new file mode 100644
index 0000000..d2b906c
--- /dev/null
+++ b/publish/SKILL.md
@@ -0,0 +1,471 @@
+---
+name: publish
+description: |
+ The publish flow for commits, pull requests, and PR review comments. Covers the mandatory pre-flight reconcile against upstream, the local code review gate, the draft/voice/approval gate before anything is published, conventional-commit message format, the Voice and Focus rules for commit bodies and PR prose, PR description structure, the three PR-review shapes (bundled review with inline pins, issue-thread comment, threaded reply), hook-level authorization, merge strategy, and the pre-commit checklist.
+
+ Use whenever a commit, a push, a pull request, a PR description, or a PR review comment is about to be produced — including amends, and including a one-line commit. Load it BEFORE drafting the message, not after, because the flow gates what gets written.
+
+ Do NOT use for the invariants that apply whether or not you are publishing: author identity, the no-AI-attribution ban, and the public-artifact content-scope rules all live in claude-rules/commits.md and are always loaded.
+---
+
+# Publish Flow
+
+Applies to: commits, pull requests, and PR review comments.
+
+The invariants — author identity, no AI attribution anywhere, and what must
+never appear in a team-visible artifact — are NOT in this file. They live in
+`claude-rules/commits.md`, which is always loaded, because a violation there is
+permanent and reaches other people. This file is the procedure: how the message
+gets written, reviewed, approved, and published.
+
+## Commit Message Format
+
+Commit messages follow the [Conventional Commits](https://www.conventionalcommits.org/) spec.
+
+### Structure
+
+ <type>[optional scope]: <description>
+
+ [optional body]
+
+ [optional footer(s)]
+
+### Types
+
+- `feat:` — new feature (correlates with MINOR in SemVer)
+- `fix:` — bug fix (correlates with PATCH in SemVer)
+- `refactor:` — code restructuring, no behavior change
+- `perf:` — performance improvement
+- `test:` — adding or updating tests
+- `docs:` — documentation only
+- `style:` — formatting, whitespace, missing semicolons (no code-behavior change)
+- `build:` — build system or external dependencies
+- `ci:` — CI configuration and scripts
+- `chore:` — anything else: tooling, meta, housekeeping
+
+The Conventional Commits spec doesn't mandate the type list. Add a new type only when the existing ones genuinely don't fit and the team will agree on what it means.
+
+### Scope
+
+A scope MAY follow the type, in parentheses, naming the affected area of the codebase: `feat(parser): add ability to parse arrays`. Use a single noun.
+
+### Breaking changes
+
+Either append `!` after the type or scope, or include a `BREAKING CHANGE:` footer (uppercase — required). Both at once is fine and adds detail. `!` alone is enough.
+
+ feat!: drop support for Node 6
+
+ BREAKING CHANGE: uses JavaScript features not available in Node 6.
+
+### Subject line
+
+Imperative mood. ≤72 characters. No trailing period. The full subject is `<type>[scope]: <description>` — the 72-char limit covers the whole thing.
+
+### Body
+
+Optional. Begins one blank line after the subject. Free-form, multiple paragraphs allowed. Don't hard-wrap body lines — write each paragraph and each bullet as a single logical line and let the renderer (GitHub, Linear, `git log`) soft-wrap. Hard wraps shrink the visible render width in web UIs and cause awkward mid-sentence breaks. The same soft-wrap rule applies to PR bodies.
+
+Skip the body when the subject line covers the change.
+
+### Footers
+
+Optional. One blank line after the body. One per line. Format: `Token: value` or `Token #value` — the git trailer convention. The token uses `-` in place of whitespace (e.g. `Reviewed-by`, `Refs`, `Acked-by`). `BREAKING CHANGE:` is the one token allowed to contain a space, and `BREAKING-CHANGE:` is treated as a synonym.
+
+### How to write the message
+
+Write commit messages as if you're explaining the change to someone debugging a failure six months from now. Focus on what changed and why, not the play-by-play of how you typed it. Short imperative summaries like "Validate input before processing" age better than diary-style notes.
+
+The body, when you need it, is where context belongs — the constraint, bug, or tradeoff that forced the change. Over time the body becomes a lightweight decision log, which is more valuable than perfectly formatted messages.
+
+Commit messages describe what changed and why, not the process that produced the change. Don't reference code review, linting, test runs, or other workflow steps in the body (e.g. "from local review," "review surfaced," "flagged by reviewer"). Reviewers and future archaeologists want the what and the why. How you got there belongs in the PR discussion, not the commit.
+
+### Examples
+
+**Subject only:**
+
+ docs: correct spelling of CHANGELOG
+
+**With scope:**
+
+ feat(lang): add Polish language
+
+**With body and footer:**
+
+ fix: prevent racing of requests
+
+ Introduce a request id and a reference to the latest request. Dismiss incoming responses other than from the latest request.
+
+ Remove timeouts which were used to mitigate the racing issue but are obsolete now.
+
+ Refs: #123
+
+**Breaking change with `!`:**
+
+ feat(api)!: send an email to the customer when a product is shipped
+
+**Breaking change in footer:**
+
+ feat: allow provided config object to extend other configs
+
+ BREAKING CHANGE: `extends` key in config file is now used for extending other config files.
+
+## Voice and Focus
+
+Applies to commit bodies, PR descriptions, and PR comments (review replies, follow-up notes, thread responses).
+
+**Write as if to a colleague.** The reader is a teammate who'll see this in `git log`, a PR feed, or a Linear thread. "I" is allowed where natural. Don't sound abstract — name the file, the function, the constraint, the symptom. Press-release voice ("This change improves...") and committee voice ("It is recommended that...") both come out. The message has to read like one engineer talking to another, not like a generated artifact.
+
+**No felt-experience narration.** Don't tell the reader how the change will feel or how often you'll use it. Phrases like "I'll feel this every time I commit", "this will be a relief", "I'm excited about" — these read as performance, not communication. State what changed and let the reader decide what to do with it.
+
+**Don't noun-ify verbs.** "The ask", "a learn", "a reveal", "the spend", "a build" — use the real noun: "the request", "the lesson", "the finding", "the budget", "the system". Verb-as-noun reads as corporate-speak and makes the sentence feel performed.
+
+**No sentence fragments in prose.** Every prose sentence needs a subject and a verb. "Two changes." or "Fix incoming." or "Body as decision log." read as bullet-list shorthand even when they're standing alone in a paragraph. Bullets and headings can be fragments — prose sentences cannot.
+
+**"I" is the author, not the user.** First person is for what *I* did or decided in this commit ("I dropped the legacy fallback because..."). It's not for describing how the software or rule behaves for whoever uses it next. "The dialog only opens if I ask" is wrong when the rule is read by someone else — that "I" becomes ambiguous. Use third-person or passive for behavior: "opens on request", "opens when asked", "opens when the user invokes it". Code and systems are the actor; "I" stays for decisions.
+
+**First person where it fits.** When the subject is you or a decision you made, use "I" ("I added X", "I kept the parameter as `Any` because..."). When the subject is a team decision or shared rationale, "we" fits. When another author's prior work is the subject, name them ("Kostya's PR #116 did X"). Third-person constructions like "This PR introduces X" or "This change restores Y" read as press-release self-narration. The commit *is* the change, so don't announce it. Code and systems can stay third-person when they're the actor ("the guard rejects...", "the serializer returns...") — first person is for describing what you did or decided, not for narrating how the code behaves.
+
+**Brief. Terse is preferred.** A one-sentence body beats a paragraph saying the same thing. If the subject line covers it, skip the body entirely. Cut every clause that restates what the diff or the PR card already shows. Length is not a proxy for care. Rhetorical padding ("worth noting", "it's important to understand") always comes out; keep what a reader will actually use.
+
+**Follow-up approvals stay terse.** A re-review that just confirms prior CHANGES_REQUESTED feedback got addressed should be `Approved.` and nothing more. The fixes are visible in the diff and in the prior review thread, so restating them adds noise. The first round of substantive review gets a real comment. Subsequent sign-offs after fixes do not. Counts as a trivial one-liner under the Step 2 exception, so the draft-file flow can be skipped.
+
+**Kind.** PR comments and review replies are directed at a specific person. Acknowledge them when it fits ("thanks for the review") without pouring it on. When you disagree or push back, frame it as your read rather than a correction ("I think...", "my read was...", "did you mean X?"). Leave room for the other person to have seen something you didn't. A polite question beats a defensive explanation. Kindness is free and makes the next review cheaper.
+
+Focus on what was wrong and what was corrected. Not the mechanics.
+Readers skimming `git log` or a PR want the before-state, the
+after-state, and the reason. They don't need a TypeScript-variance
+lesson, a compiler-inference walkthrough, or a trip through an API's
+internals. Keep the "why" to one sentence unless a subtle invariant
+genuinely needs more.
+
+Don't stack technical terms. A sentence that chains three or more type
+signatures, API names, or compiler concepts reads as a jargon wall.
+Break it into shorter sentences and translate to reader-facing
+language. "The mock returns `Promise<Mission>`, so the resolver's
+argument is `Mission`, not `unknown`" beats the full inference chain
+that produces that signature. Keep the terms a reader will grep for,
+drop the ones that name compiler internals.
+
+
+Different artifact types carry different content. Don't duplicate.
+
+**PR descriptions:** four sections, in order.
+
+1. **Problem** — what's wrong, with enough detail that a teammate can
+ recognize the same failure mode in their own work.
+2. **Fix** — what changed.
+3. **Why this fixes it** — causal link, one or two sentences.
+4. **How it was tested** — skip for proposals, specs, or discussions;
+ required for shipped fixes.
+
+The PR is the technical artifact. It carries the detail.
+
+If the project's publishing overlay defines a ticket system, see it for
+ticket-body conventions (a ticket body is typically just the Problem and
+Fix, with the causal why and test verification left to the PR).
+
+**PR review comments** are conversational and don't follow this
+structure — they follow the Voice and Focus rules above.
+
+Verbose preambles, motivational language, and context unrelated to the
+problem belong out. Same conciseness pressure as commit-message bodies.
+
+
+## Review and Publish
+
+Commits and PRs are team-visible, permanent, and hard to amend once shared
+(especially after push or after a reviewer has replied). Before executing
+`git commit` or `gh pr create`, the change must pass a local code review
+*and* the message must be reviewed by the user. The flow has three steps, in
+order.
+
+### Step 0: pre-flight reconcile (mandatory)
+
+Before reviewing the diff, fetch from the remote and reconcile against the
+upstream of the current branch. Reconciliation can change the working state
+when a rebase brings in upstream commits that touch staged files, and that
+would invalidate Step 1's review. Handling drift first means the review and
+the commit message describe the post-reconcile state.
+
+1. Fetch all remotes:
+
+ git fetch --all --prune
+
+2. If the current branch has no upstream (new branch, never pushed), skip
+ to Step 1 — there's nothing to reconcile against, and the first push
+ sets the upstream.
+
+3. Otherwise, check divergence against `@{u}`:
+
+ git rev-list --left-right --count @{u}...HEAD
+
+ Output is `<behind>\t<ahead>`. Decide based on the pair:
+
+ - **0 behind, anything ahead** — no-op. Continue to Step 1.
+ - **Behind only, clean tree** — fast-forward: `git merge --ff-only @{u}`.
+ - **Behind only, dirty tree** — surface to the user. Don't auto-stash or
+ auto-merge. Offer to commit or stash first, or skip the reconcile and
+ proceed knowing the push may need attention later.
+ - **Diverged (behind AND ahead)** — surface to the user. Ask whether to
+ rebase the local commits onto upstream (default for feature branches),
+ merge the upstream branch in (rare; preserves both lines), or skip and
+ proceed with the divergence. Don't auto-rebase.
+
+4. **PR flow only.** Also fetch the base branch (usually `main`) and check
+ whether the feature branch's base is behind. Surface this informationally;
+ don't auto-rebase the feature branch without asking. The "X commits
+ behind base" badge on the PR is a follow-up decision, not a reason to
+ block publish.
+
+The startup workflow's `git fetch --all --prune` doesn't substitute for
+Step 0. Upstream can advance during a long session, especially across
+machines or with teammates pushing in parallel. Run Step 0 every time the
+publish flow starts.
+
+### Step 1: adversarial review by an isolated reviewer (mandatory)
+
+The review runs in a **subagent**, never inline, and it runs on **every**
+commit. The author does not review their own work.
+
+**Why isolation, not just review.** A self-review checks the diff against the
+author's own model of what the diff should do. It cannot check the model. The
+errors that survive a self-review are the ones that were never visible in the
+diff — a scope inherited from whoever reported the problem, a blast radius
+estimated instead of measured, a fix that is correct for the case the author
+had in mind and wrong for the one they never considered. Only a reviewer that
+does not hold the author's model catches those, so the isolation is the point
+and the adversarial stance is the method.
+
+**Dispatch contract.** Spawn the reviewer via the Agent tool and give it these
+three things, the third whenever one exists:
+
+1. **The diff** — `git diff --cached` for a commit, the branch diff for a PR.
+2. **The claim** — one line from the author stating what the change does. Write
+ it before dispatching. This is the thing under test: the reviewer's job is
+ to check the diff against the claim.
+3. **The requirement source, when one exists** — the ticket, plan, ADR, or task
+ body the work was done against. Pass it verbatim.
+
+Withhold everything else: the conversation, the exploration, the dead ends, and
+above all the author's reasoning for why the change is right. Those are what
+transmit the author's model, which is what the reviewer exists to not have. A
+reviewer given the rationale reviews the rationale.
+
+**Why the requirement source is not withheld.** A ticket or plan is not the
+author's model of the change — it is the independent record of what was asked,
+written before the work and usually by someone else. It is the only artifact
+that can contradict the author's one-line claim. Withhold it and the claim
+becomes self-certifying: the reviewer checks the diff against a sentence the
+author wrote, which cannot surface scope creep or a missing requirement. That
+also strands `review-code`'s Intent-vs-Delivery criterion, which is skipped
+outright when no intent context is supplied and is the one criterion aimed at
+the inherited-scope error this whole gate exists to catch.
+
+Invoke the review with `/review-code --staged` (commit), `/review-code` (branch
+diff against the `main` merge-base), or `/review-code <PR#>` (someone else's
+PR), and tell it to run its adversarial pass.
+
+**Adversarial, with substantiation.** The reviewer is prompted to *refute* the
+change rather than to bless it. But an agent told to attack will manufacture
+findings to satisfy the instruction, so the stance carries a floor: a finding
+that cannot be substantiated against the diff is not a finding and must be
+dropped. `review-code`'s confidence filter and false-positive filter are what
+enforce that floor — adversarial raises the appetite for looking, never the
+tolerance for a weak claim.
+
+**Scope is every commit; the reviewer decides triviality, not the author.**
+There is no "trivial enough to skip" exemption. A floor written in terms of
+"small" or "mechanical" puts the judgment back with the author, whose judgment
+is the thing being checked. Dispatch always, and let `review-code`'s own Phase 0
+eligibility gate return fast on a whitespace-only diff, an obvious revert, or an
+already-reviewed SHA. A cheap spawn on a trivial commit is the price of the
+author never getting to rule on their own diff.
+
+**Verdict, and the re-review loop.** The reviewer returns one of four outcomes,
+and all four are defined exits:
+
+- **Approve** — the gate is satisfied. Proceed to Step 2.
+- **Skipped** — `review-code`'s Phase 0 found the diff ineligible (whitespace
+ only, an obvious revert, an already-reviewed SHA). This **satisfies the gate**
+ and the flow proceeds. A skip is a reviewer's ruling, which is the point; what
+ is forbidden is the *author* ruling their own diff too trivial to look at.
+- **Request Changes** — blocking findings stand. Enter the loop below.
+- **Needs Discussion** — the reviewer has a disagreement it cannot settle from
+ the diff: an architectural objection, a question about whether the change
+ should exist at all. This **stops immediately and goes to the user**; it does
+ not enter the loop. Routing it to the loop would answer "should we do this?"
+ with "fix these findings," which is the wrong question and burns rounds on a
+ disagreement no amount of editing resolves. Unattended callers park it exactly
+ as they park a bound hit.
+
+Approval is the reviewer's to give; the author never declares their own change
+clean.
+
+That set is closed. `review-code` emits Approve, Request Changes, or Needs
+Discussion, and its Phase 0 emits Skipped; every one has a defined exit above. A
+verdict outside those four means the reviewer went off-contract — surface it
+rather than interpreting it.
+
+**The loop turns on blocking findings, not on the verdict token.** It ends when
+no Critical or Important finding stands. A Minor-only result is not grounds for
+another round: fix it or don't, but do not spend a round on it, and do not
+escalate to the user over one. A reviewer holding only Minor findings should
+return Approve and say what it left.
+
+On Request Changes:
+
+1. Surface **all** findings — Critical, Important, Minor. Critical and Important
+ block; Minor is shown and does not block.
+2. Fix the blocking findings.
+3. **Re-review, and keep re-reviewing until the reviewer approves.** A fix is a
+ new change and gets the same scrutiny as the original. Fixing under review
+ pressure is exactly when a regression gets introduced, so an unreviewed fix
+ is the hole this loop closes.
+
+**Continue the same reviewer, don't spawn a fresh one.** Send the updated diff
+back to the existing reviewer (`SendMessage` with its agent ID). It holds its own
+findings, so it can confirm each one is actually addressed. A fresh reviewer each
+round cannot tell "addressed" from "never existed", re-litigates settled points,
+and drifts to a new set of findings every round, which never converges.
+
+The continued reviewer must **re-verify each finding against the new diff**, not
+against the author's description of the fix. "I fixed it" is a claim, and taking
+it at face value is how a review round becomes a rubber stamp.
+
+**Bounds, so the loop terminates.** Two conditions end it early and hand the
+decision to the user:
+
+- **Three rounds without approval** (the initial review plus two re-reviews).
+ This matches the two-fix-attempts limit in `subagents.md`: past that, the
+ problem is usually the approach rather than the diff.
+- **A finding recurs after being reported fixed.** That is oscillation — the
+ fix for one finding reintroducing another — and another round will not
+ resolve it. Stop on the first recurrence rather than spending the remaining
+ rounds.
+
+In both cases, stop and surface: the standing findings, what was tried, and the
+decision needed.
+
+**Override.** The user can bypass the block with an explicit "proceed anyway" (or
+equivalent). The user is also the adjudicator when the author believes a finding
+is wrong: say so with the reasoning and let the user rule. Do not resolve a
+disagreement with the reviewer by overruling it silently. Without an explicit
+override, do not proceed to Step 2.
+
+**When the Agent tool is unavailable.** Per `subagents.md`, don't block: run the
+review in the main thread, but hold it to the same contract — review against the
+stated claim, refute rather than bless, substantiate every finding, loop on fixes
+until clean, same bounds. State plainly that the review was not isolated, because
+a self-review under an adversarial prompt is weaker evidence and the user should
+know which one they got.
+
+### Step 2: draft, review, publish
+
+**Voice patterns and the approval gate are two independent decisions.** Don't bundle them.
+
+*Voice patterns are always personal for publish artifacts.* Commit messages, PR titles + bodies, and PR review comments all go out under the user's name, so they always run through `/voice personal` (the full pattern walk — general + Craig's-voice + the artifact-mechanics patterns: first-person rewrite, public-artifact scope flag, praise/correction asymmetry, finding stems), regardless of whether `.ai/` is tracked. These three are personal-voice artifacts by definition — the skill's personal mode exists for exactly them. Pattern #39 (public-artifact scope flag) matters *most* on team-visible artifacts, so it must never be skipped on a PR comment or PR body. There is no "general-voice mode" for publish artifacts.
+
+*The approval gate turns on whether anyone else reads the history.* The gate
+exists so Craig sees the exact words that go out under his name. Skipping it
+trades that for velocity, and that trade is only worth making where the repo is
+genuinely shared with other people.
+
+The old signal for this was whether `.ai/` is tracked, used as a proxy for
+"team repo." It was the wrong proxy and it failed in the direction that
+matters: rulesets, home, and work all track `.ai/` — rulesets as a committed
+mirror, the others because the project history *is* the project — while all
+three are Craig's private single-user repos. The rule as written skipped the
+gate on his three most-used projects.
+
+Check the remote host instead, which is what actually distinguishes them:
+
+```
+git remote -v 2>/dev/null | grep -v 'cjennings\.net' | head -1
+```
+
+- **No output** — every remote is on `cjennings.net`, so the repo is Craig's
+ own and nobody else reads the log. **Gate applies**: write to `/tmp`, run
+ `/voice personal`, print inline, ask approve / request changes / open in
+ editor, and publish only on explicit approval.
+- **Any output** — a remote on a host someone else can read (GitHub, a GHE
+ instance, a team server). **Gate skipped for velocity**: write to `/tmp`, run
+ `/voice personal`, print inline, publish immediately.
+- **No remote at all** — a local-only repo. Gate applies; there is no
+ velocity argument without a reader.
+
+As of 2026-07-27 every project resolves to gate-applies, because every remote
+is `cjennings.net`. That is the correct answer, not a bug: it matches how the
+flow has actually been run.
+
+Either way the draft runs through `/voice personal` first. The subflows below describe the full gated path. For the gate-skipped path, run the same `/voice personal` pass, then collapse the "Ask: approve, request changes, or open in editor" step — the draft prints inline and the publish step runs immediately afterward.
+
+**For commit messages:**
+
+1. Write the proposed message to `/tmp/commit-<short-slug>.md`.
+2. Run `/voice personal` on the file. Always. The skill walks its full pattern list covering signs of AI writing, universal good-writing rules (Strunk & White, Orwell, Plain English, Garner), and Craig's voice patterns (first-person rewrite, semicolons → periods/commas, contractions, sentence-split on conjunctions, felt-experience cut, sentence-fragment rewrite, terse cut for rhetorical padding, no-emphasis-formatting, public-artifact scope flag, praise/correction asymmetry, finding stems). The commit subject line stays imperative per Conventional Commits — `/voice personal` rewrites the body, not the subject. Skip the pass for purely mechanical commits (a chore version bump, a typo fix) where the subject alone carries the message.
+3. Print the final draft inline in the terminal. Every line, exactly as it'll be committed. No truncation, no summary. State that the skill ran (e.g. "/voice personal — full pattern walk"). If pattern #39 (public-artifact scope) flagged anything, surface those warnings; the user resolves them manually.
+4. Ask: approve, request changes, or open in editor. Wait for an explicit answer. Do not open the file in `emacsclient` (or any editor) by default — print first, edit only if asked.
+ - **Approve** → commit with `git commit -F /tmp/commit-<short-slug>.md`.
+ - **Request changes** → make them, re-run `/voice personal`, re-print inline, ask again.
+ - **Open in editor** → only if the user asks. `emacsclient -n /tmp/commit-<short-slug>.md`. After the editor closes, re-read the file, re-print the contents inline, and ask again.
+
+**For PR descriptions, and for PR review comments and replies:** read
+`references/pull-requests.md` in this skill directory. It carries the PR
+description shape, the three review shapes (bundled review with inline pins,
+issue-thread comment, threaded reply), the `gh api` calls, and the
+publishing-overlay hooks. A plain commit needs none of it, so it is kept out of
+this file — load it when the artifact is a PR.
+
+**Approve does not authorize a merge.** Reviewing a PR never authorizes merging it. Anything in `## Merge Strategy` below applies only to merges *you* are about to perform on your own branches — and even then, the merge needs its own explicit user confirmation per the rules there. A project's publishing overlay may add a team merge practice (e.g. approve-then-author-merges, where the review notification hands the merge decision to the PR author); that's an overlay concern, not a global one.
+
+**Exception:** trivial one-liners the user dictated verbatim in the
+conversation (e.g. "commit this as `chore: bump version`", "reply just
+'thanks for the review'") can skip the draft-file step in Step 2.
+Step 1's review still runs on every commit — this carve-out is about the
+draft-file step, not the review. Phase 0 is what rules a trivial diff out, and
+its Skipped result satisfies the gate. An acknowledgment-only PR reply commits
+nothing, so there is no diff to review.
+
+**Single-skill gate.** Each of the three subflows above runs `/voice personal` before printing the draft — the full pattern walk covering AI-writing signs, universal good-writing rules, Craig's voice patterns, and the artifact-mechanics patterns (first-person rewrite, public-artifact scope flag, praise/correction asymmetry, finding stems). Publish artifacts (commits, PR titles + bodies, PR review comments) always use personal mode; the `.ai/`-tracking check at the top of Step 2 decides only whether the approval gate fires, not which patterns run. Running the skill is mandatory; the printed draft must have been through it. When the user asks mid-flow for "the voice pass" on an in-progress draft, that means re-run the full pattern walk — not a subset. Always state that the skill ran when announcing the printed draft (e.g. "/voice personal — full pattern walk"). Skipping the pass without flagging it is a defect. The terse/omit-needless-words cut (pattern #38) is the *last* thing the skill does before the draft is printed: read each sentence and cut it in half, keeping only what changes meaning. The draft the user first sees must already be terse — if they have to ask for an Orwell pass after seeing it, the pass was skipped.
+
+**If `/voice` is unavailable.** The skill should be installed (it ships with rulesets), but a fresh or partial environment may not have it. Don't let that block the publish, and don't skip the discipline silently. Walk the same patterns inline — they're documented in the skill, and the publish flow already names which ones matter (first-person rewrite, semicolons → periods/commas, contractions, sentence-split, felt-experience cut, fragment rewrite, terse cut, the pattern #39 public-artifact scope flag, plus the AI-writing and good-writing passes). Then state that the skill was unavailable and the pass was applied by hand (e.g. "/voice unavailable — patterns walked inline"). The gate is the pattern walk, not the tooling; the skill is the convenient way to run it, not the only way. Flag the missing skill so it gets installed.
+
+### Hook-level authorization
+
+The Step 1 code review plus the Step 2 user approval together constitute the
+authorization gate for the publish action. No separate hook-level approval
+prompt is needed on `git commit`, `gh pr create`, `git push`, or their
+variants once Step 2 has been approved. If a hook is configured, rely on the
+flow above to be the source of truth; do not treat the hook as a second
+independent gate.
+
+## Merge Strategy
+
+- *Squash-merge is the default* for feature branches. It avoids carrying
+ WIP and fix-up commits into the target branch history and produces one
+ logical change per merge.
+- State the planned merge approach (squash, rebase, or merge commit) and
+ the target branch *before* pushing or merging. Wait for explicit user
+ confirmation before `git push`, `gh pr merge`, or any equivalent. The
+ Review and Publish flow above approves the *content*; merge strategy is
+ a separate decision that needs its own confirmation.
+- *Pre-push reconcile.* Right before `git push`, do one more
+ `git fetch <remote> <branch>` and verify the local branch is still
+ ahead-only against its upstream. If something landed between Step 0 and
+ push (review and draft together can take several minutes, and another
+ machine or teammate may push in that window), surface and resolve before
+ the push command runs. Catching drift here is cheaper than recovering
+ from a failed non-fast-forward push under publish-step pressure.
+- Override the squash default only when there's a concrete reason: a
+ clean per-commit review history the user has explicitly asked for, a
+ multi-commit semantic narrative the team values, etc. Squash is the
+ safe default; document why when deviating.
+
+## Before Committing
+
+1. Check author identity: `git log -1 --format='%an <%ae>'` — should be the user.
+2. Scan the message for AI-attribution language (including emojis and footers), and on a public or shared-remote repo for tooling-path enumeration — prose that lists `CLAUDE.md`, `.claude/`, `.ai/`, `todo.org`, `notes.org`, or `session-context`. Name the category, not the paths. Exempt: a commit whose change is one of those files, and private single-user repos.
+3. Review the diff — only intended changes staged; no unrelated files.
+4. Confirm staged files belong in the repo: nothing that the project's policy keeps untracked (the personal-tooling set in gitignore-mode projects), and in repos with a canonical/mirror split, the edit is on the canonical side — a mirror-only edit gets reverted by the next sync.
+5. Run the full test suite and linters as their own step, read the result, and commit only on zero failures — never chain the run into the commit command (see `verification.md`).
+
diff --git a/publish/references/pull-requests.md b/publish/references/pull-requests.md
new file mode 100644
index 0000000..9ac4ded
--- /dev/null
+++ b/publish/references/pull-requests.md
@@ -0,0 +1,105 @@
+# Pull requests and PR review comments
+
+Loaded from the `publish` skill when the artifact is a PR description or a PR
+review comment. A plain commit never needs any of this, which is why it lives
+here rather than in SKILL.md.
+
+Steps 0 and 1 of the publish flow (pre-flight reconcile, local code review) and
+the `/voice personal` pass still apply — see SKILL.md. This file carries only
+what is specific to PRs.
+
+**For PR descriptions:**
+
+1. Write the title as line 1 and the body below it to `/tmp/pr-<slug>.md`. **Title format:** the conventional-commit subject (`refactor: remove dead if-count-is-not-None check in admin`). If the project defines a publishing overlay with a ticket system, follow it for the ticket suffix in the title and the cross-link line in the body (see the overlay).
+2. Run `/voice personal` on the file. The PR title stays imperative per Conventional Commits — `/voice personal` rewrites the body, not the title.
+3. Print the final draft inline in the terminal. Title on line 1, blank line, then body — exactly as it'll be posted. State that the skill ran. Surface any pattern #39 (public-artifact scope) warnings.
+4. Ask: approve, request changes, or open in editor. Wait for an explicit answer. Do not open the file in `emacsclient` (or any editor) by default.
+ - **Approve** → continue to step 5.
+ - **Request changes** → make them, re-run `/voice personal`, re-print inline, ask again.
+ - **Open in editor** → only if the user asks. `emacsclient -n /tmp/pr-<ticket-or-slug>.md`. After the editor closes, re-read the file, re-print inline, ask again.
+5. Split the file on the first blank line and pass the title and body to `gh pr create --title "..." --body "$(tail -n +3 <file>)"` (or a heredoc) so formatting is preserved. Add `--reviewer <user[,user...]>` in the same call when you already know who should review.
+6. Request reviewers on the new PR if you didn't pass `--reviewer` at create time. Use `gh pr edit <N> --add-reviewer <user>`. If the repo has a `CODEOWNERS` file, GitHub auto-suggests based on touched paths. Still issue the explicit request so the reviewer gets notified. Pick reviewers per the team's convention for the area touched (often documented in the per-repo `CLAUDE.md`). For follow-up PRs, consider tagging the parent PR's author if their context would help. PRs without a human reviewer request stall — "checks passed" is not a substitute for review.
+7. **Project publishing overlay (if present).** If the project defines a publishing overlay — a `publishing-<team>.md` rule loaded from its `.claude/rules/` — run its post-create steps now: ticket cross-linking, ticket-state moves, and any other tracker integration it specifies. A project with no overlay skips this; the PR is already open and reviewers are requested, which is the complete universal flow.
+
+**For PR review comments and replies (review verdicts, threaded discussion, follow-up notes on someone else's PR or your own):**
+
+Pick the shape first. Most reviews are Shape 1.
+
+- **Shape 1 — Single review** (verdict + summary body + 0+ inline pins). The default for any post that carries a verdict (`APPROVE`, `REQUEST_CHANGES`, `COMMENT`), even when the verdict has no line-specific findings. One `gh api` call posts the summary, every inline pin, and the verdict together. review notification fires once for `APPROVE` or `REQUEST_CHANGES`.
+- **Shape 2 — Issue-thread comment** (no verdict). General PR discussion, not a review. No inline pins. No review notification.
+- **Shape 3 — Reply on an existing inline thread**. Responding to a specific prior reviewer comment. Threads under that comment. No review notification.
+
+**Inline threshold for Shape 1.** Any finding that names a `path:line` belongs as an inline comment pinned to that line. Cross-cutting observations (verdict rationale, "third PR with the same pattern", overall test-coverage gaps that don't pin to one place) stay in the summary body. There's no "fold one inline into the summary" exception — a single line-specific finding still goes inline.
+
+**Shape 1: Single review (bundled summary + inline)**
+
+1. Identify findings, split into **inline-eligible** (each names a specific `path:line`) and **summary-only** (cross-cutting). Decide the verdict.
+
+2. Write one concatenated draft to `/tmp/pr-<N>-review.md` with explicit separators:
+
+ ```
+ === SUMMARY ===
+ <verdict summary body>
+
+ === INLINE path=frontend/src/foo.tsx line=440 ===
+ <inline body 1>
+
+ === INLINE path=frontend/src/bar.tsx line=137 ===
+ <inline body 2>
+ ```
+
+ The separator format is exactly `=== SUMMARY ===` and `=== INLINE path=<path> line=<n> ===`. The summary block is mandatory even for verdict-only reviews. Inline blocks are zero-or-more.
+
+3. Run `/voice personal` on the file once. The skill walks its full pattern list across every block at the same time. The separators stay intact because they aren't prose.
+
+4. Print the final draft inline in the terminal. Every block — the summary body AND the full prose of every inline comment — exactly as it'll be posted, with its separator header. Print the inline in full; never describe it in place of printing it ("I'd pair it with one inline on…"). Craig approves the exact words that post under his name, so the exact words must be on screen. State that the skill ran (e.g. "/voice personal — full pattern walk across summary + 3 inline"). Surface any pattern #39 warnings.
+
+5. Ask: approve, request changes, or open in editor. Wait for an explicit answer. Do not open the file in `emacsclient` (or any editor) by default.
+ - **Approve** → continue to step 6.
+ - **Request changes** → make them, re-run `/voice personal` on the whole file, re-print inline, ask again.
+ - **Open in editor** → only if the user asks. `emacsclient -n /tmp/pr-<N>-review.md`. After the editor closes, re-read, re-print inline, ask again.
+
+6. Split the file on the separator lines and post in **a single** `gh api` call:
+
+ ```
+ gh api repos/<owner>/<repo>/pulls/<N>/reviews \
+ --hostname <ghe-host-or-omit> \
+ -F event=REQUEST_CHANGES \
+ -F body="<summary block>" \
+ -F "comments[][path]=<path1>" \
+ -F "comments[][line]=<line1>" \
+ -F "comments[][body]=<inline 1>" \
+ -F "comments[][path]=<path2>" \
+ -F "comments[][line]=<line2>" \
+ -F "comments[][body]=<inline 2>"
+ ```
+
+ `event` is one of `APPROVE`, `REQUEST_CHANGES`, `COMMENT`. The `comments[]` array can be empty for verdicts with zero line-specific findings — the call still uses the same endpoint. Pass `--hostname` for non-`github.com` hosts (a project's publishing overlay names its host when it's a GitHub Enterprise instance).
+
+7. Verify the review landed. `gh api repos/<owner>/<repo>/pulls/<N>/reviews --hostname ...` returns the latest review with bundled inlines. Confirm `state` matches the verdict and the inline count matches what was posted.
+
+8. **Project review-notification overlay (if present).** If the project defines a publishing overlay with a review-notification step (e.g. a Slack ping to the PR author), run it now — but only for `APPROVE` and `REQUEST_CHANGES` verdicts. The overlay owns the channel, the message format, the author-mention lookup, and the threading. A project with no overlay skips notification entirely. `COMMENT` verdicts and Shapes 2-3 below never notify, overlay or not.
+
+**Shape 2: Issue-thread comment (no verdict)**
+
+Use when the post is informal discussion that shouldn't appear as a review verdict (e.g. "I'd like to discuss the X approach before you continue").
+
+1. Write the proposed comment to `/tmp/pr-<N>-comment.md`.
+2. Run `/voice personal`.
+3. Print inline, ask approve/changes/edit, gate as in Shape 1 step 5.
+4. Post: `gh pr comment <N> --body-file /tmp/pr-<N>-comment.md`.
+5. Verify: `gh api repos/<owner>/<repo>/issues/<N>/comments`.
+6. No review notification.
+
+**Shape 3: Reply on an existing inline thread**
+
+Use when responding to a specific prior reviewer comment.
+
+1. Find the parent comment ID: `gh api repos/<owner>/<repo>/pulls/<N>/comments`.
+2. Write the reply to `/tmp/pr-<N>-reply-<comment-id>.md`.
+3. Run `/voice personal`.
+4. Print inline, ask approve/changes/edit, gate as in Shape 1 step 5.
+5. Post: `gh api repos/<owner>/<repo>/pulls/<N>/comments -F in_reply_to=<comment-id> -F body="$(cat /tmp/pr-<N>-reply-<comment-id>.md)"`.
+6. Verify in the same `comments` list.
+7. No review notification.
+
diff --git a/review-code/SKILL.md b/review-code/SKILL.md
index ec08a9d..559cee8 100644
--- a/review-code/SKILL.md
+++ b/review-code/SKILL.md
@@ -33,7 +33,31 @@ When intent context is given, the review grades "does this match what was asked?
## Execution Model
-For substantive reviews on large diffs: **dispatch the perspective passes as parallel sub-agents** via the Agent tool. Each sub-agent starts with a clean context window — the reviewer shouldn't inherit the implementer's mental model. For small single-commit tweaks, run inline.
+Two levels of dispatch, and conflating them is the mistake to avoid.
+
+**Level one — who runs this skill.** When invoked from the `publish` flow's Step 1 gate, this skill is already running inside an isolated reviewer subagent that the publish flow spawned. That isolation is not optional and not this skill's call: the author never reviews their own change. If you are reading this in the same context that wrote the diff, the publish flow was not followed.
+
+**Level two — how the perspectives run inside it.** For substantive reviews on large diffs, dispatch the Phase 2 perspective passes as parallel sub-agents. For small single-commit tweaks, run them inline. This is a cost decision about fan-out *within* the review and never a licence to skip level one.
+
+### The adversarial contract (publish-flow reviews)
+
+When running as the publish flow's reviewer, operate under these terms:
+
+- **You have three inputs, and the boundary is deliberate.** The diff; a one-line claim from the author about what it does; and, when one exists, the requirement source it was built against (ticket, plan, ADR, task body). You were *not* given the conversation, the author's reasoning, or the alternatives they rejected, because those transmit the author's model of the change and your value is in not holding it. Do not ask for them.
+- **The requirement source is not leakage — use it.** It was written before the work and usually by someone else, so it is the one input that can contradict the author's claim about their own diff. Without it, the claim is self-certifying and you can only check the diff against a sentence the author wrote. It is what makes the Intent-vs-Delivery criterion below live rather than skipped, and scope creep and missing requirements are exactly what it catches.
+- **Review the diff against the claim.** Does it do what the claim says? What does it do that the claim does not mention? What breaks that the claim assumes is fine?
+- **Try to refute the change, not to bless it.** Assume there is something wrong and go looking. The default posture is skepticism.
+- **A finding you cannot substantiate against the diff is not a finding.** Drop it. An agent told to attack will manufacture findings to satisfy the instruction, and a manufactured finding costs the author a round of work and teaches them to discount the next review. The confidence filter and false-positive filter in Phase 4 are what hold this line: adversarial raises how hard you look, never how weak a claim you will ship.
+- **Verify the premise before judging the fix.** When the diff claims to fix a bug, confirm the bug is real and reproduces in the pre-change code. A fix for a misdiagnosed problem passes a diff-shaped review and is still wrong.
+
+### Re-review mode (continued rounds)
+
+The publish flow sends the updated diff back to the *same* reviewer rather than spawning a fresh one, so a continued round has your prior findings in context. On each round:
+
+1. **Re-verify every prior blocking finding against the new diff.** Not against the author's description of the fix. "I fixed it" is a claim like any other, and accepting it is how a re-review becomes a rubber stamp.
+2. **Review the fix as a new change.** A fix written under review pressure is prime ground for a regression, so it earns the same scrutiny as the original diff, including on lines the fix touched incidentally.
+3. **Say when a prior finding recurs.** If something reported fixed is back, name it as a recurrence rather than filing it fresh. The publish flow treats recurrence as oscillation and stops the loop, which is the right outcome — another round will not converge.
+4. **Do not invent new findings to justify another round.** If the blocking findings are addressed and nothing new is substantiated, approve. Approval is the expected end state, not a failure to find something.
## Phase 0 — Eligibility Gate
@@ -261,7 +285,7 @@ Do **not** flag any of these as issues:
- **Issues explicitly silenced in code** (e.g., `# type: ignore[...]` with a reason, lint ignore comments) unless the silencing is unjustified
- **Intentional changes** in functionality clearly related to the PR's stated goal
- **Changes in unmodified lines** (real issues in files the PR touches but on lines it doesn't change)
-- **Framework behavior being tested** — see `testing.md` anti-patterns
+- **Framework behavior being tested** — see the `testing-standards` skill, anti-patterns
### Severity Categorization
@@ -273,9 +297,9 @@ Remaining issues get tagged:
## Phase 5 — Output
-**Terminal display — plain text only.** Everything this skill echoes to Craig in the chat terminal — the structured report, the per-criterion table, the verdict, and any draft summary or inline awaiting approval — must be rendered as plain text. No bold (`**`), no backtick code spans, no markdown tables, no headings-as-markup. They render as reverse video in his terminal and are hard to read. Write `file.py:42` as bare text, identifiers and `test:`-style prefixes unquoted, the criterion audit as a plain dashed list rather than a table, severity tiers as plain labels. This applies only to the terminal echo. The review actually posted to GitHub (the `gh api` body and inline blocks) keeps normal markdown — GitHub renders it correctly, so draft the posted artifact in markdown and strip the markup only when mirroring it into the chat. (Same rule as `interaction.md`'s no-reverse-video constraint; repeated here because the violation happens at exactly this print step.)
+**Terminal display — plain text only.** Everything this skill echoes to Craig in the chat terminal — the structured report, the per-criterion table, the verdict, the draft summary, and the full prose of every inline pin awaiting approval (the exact words, never a description of them) — must be rendered as plain text. No bold (`**`), no backtick code spans, no markdown tables, no headings-as-markup. They render as reverse video in his terminal and are hard to read. Write `file.py:42` as bare text, identifiers and `test:`-style prefixes unquoted, the criterion audit as a plain dashed list rather than a table, severity tiers as plain labels. This applies only to the terminal echo. The review actually posted to GitHub (the `gh api` body and inline blocks) keeps normal markdown — GitHub renders it correctly, so draft the posted artifact in markdown and strip the markup only when mirroring it into the chat. (Same rule as `interaction.md`'s no-reverse-video constraint; repeated here because the violation happens at exactly this print step.)
-Before printing any approve/request-changes summary for posting, run the praise/correction gate (see Posted Summary Voice, and `/voice` personal #40): scan the summary and cut every clause that describes or justifies a good change — keep praise plus verdict only. Then confirm each finding and change-request states its why, gently and briefly. This is a mechanical pass, not a judgment call.
+Before posting, print the full summary body AND every inline comment exactly as it will post — never the summary alone with an inline merely described ("I'd pair it with one inline on..."). Craig approves the exact words that post under his name, so the exact words must be on screen. Then run the praise/correction gate (see Posted Summary Voice, and `/voice` personal #40): scan the summary and cut every praise clause and every clause that describes or justifies a good change — an approve summary is the substantive pointer plus the verdict, no praise, not even a bare positive. Then confirm each finding and change-request states its why, gently and briefly. This is a mechanical pass, not a judgment call.
```markdown
# Code Review — <PR title / branch name / SHA range>
@@ -437,7 +461,7 @@ None.
## Hand-Off
-- **Critical** → must be addressed before merge; author fixes, re-review via `/review-code` on the updated SHA
+- **Critical** → must be addressed before merge; author fixes, then the updated diff comes back to *this* reviewer for another round (publish flow, Step 1), not to a fresh one and not to the author's own judgment
- **Important** → fix, or deliberately defer with an ADR (run `/arch-decide`)
- **Minor** → follow-up issues or a cleanup PR
- **Intent-vs-Delivery gaps** → either file tickets for the missing pieces or update the plan to reflect reality
@@ -446,21 +470,24 @@ None.
The summary body and the inline pins work as a pair: scannable verdict on top, full coaching conversation in the pins. Read this section paired with Inline Comment Voice below — the summary is terse precisely because the inlines carry the teaching weight.
-The structured report above stays local. When the verdict is posted as a GitHub review (per `commits.md` Step 2 Shape 1), keep the summary body terse — one long sentence or a few short ones is plenty. Vary the phrasing run-to-run so consecutive reviews don't read templated. Voice: an encouraging senior dev who doesn't like to talk; positive feedback is short, blunt, and lands cleanly.
+The structured report above stays local. When the verdict is posted as a GitHub review (per the `publish` skill, Step 2 Shape 1), keep the summary body terse — one long sentence or a few short ones is plenty. Vary the phrasing run-to-run so consecutive reviews don't read templated. Voice: an encouraging senior dev who doesn't like to talk. Lead with the substantive pointer — the design note or blocker that's pinned inline — and close with the verdict; the summary carries no praise clause.
-Name the good thing and stop: do not explain *why* it's good. The author made the change and already knows the rationale, so justifying the praise reads as sycophantic. "Clean migration off the window globals, tests cover the right edges" lands; appending "...because there are no stragglers and the provider, mocks, and Normal/Boundary/Error cases are all covered" turns a compliment into padding. Elaboration is for findings (something is wrong, here's the failure mode and the fix), never for compliments.
+The summary body carries no praise — not a named good thing, not a bare positive. The author made the change and already knows its merits, so a compliment in a terse summary reads as filler or sycophancy. If a genuine positive is worth surfacing, it goes as a single inline pin on the relevant line (see below), never the summary body. Elaboration in the summary is for the substantive pointer and for findings — what's wrong, the failure mode, the fix — never for compliments.
-This holds for re-review approvals too. A re-review confirming requested changes is just "Approving." Mechanical rule: an approve summary is the verdict plus at most a bare positive ("Clean.", "Solid fix."). It must contain no clause that says what the change does or why it works. "The hoist to App fixes the crash, and the new test locks it in" is the banned pattern — it describes and justifies the change on an approve. If a clause references the code's behavior, cut it.
+This holds for re-review approvals too. A re-review confirming requested changes is just "Approving." Mechanical rule: an approve summary is the substantive pointer (the inline design note, if any) followed by the verdict — no praise, not even a bare positive. Lead with the pointer, close with the verdict: "One design note inline, not a blocker. Approving." An approve with nothing to flag is just "Approving." No clause may say what the change does or why it works, and none may compliment it — "Clean.", "Solid fix.", "The hoist to App fixes the crash and the new test locks it in" are all cut. If a clause references the code's behavior or praises it, cut it.
-The asymmetry: praise gets no why, but a finding, change-request, or inline coaching note *always* gets the why. Behavior only changes when the reason lands, so a correction that just says what to fix without saying why teaches nothing. Deliver that why gently and briefly, the way a kind coach would, never as a verdict from on high. The praise-strips / correction-explains split is enforced as `/voice` personal pattern #40, which every posted review summary passes through.
+The asymmetry: the summary drops praise entirely, but a finding, change-request, or inline coaching note *always* gets the why. Behavior only changes when the reason lands, so a correction that just says what to fix without saying why teaches nothing. Deliver that why gently and briefly, the way a kind coach would, never as a verdict from on high. The drop-praise / correction-explains split is enforced as `/voice` personal pattern #40, which every posted review summary passes through.
Good:
-- "Nice, clean, good coverage. One small naming point inline. Approving."
-- "Clean shape, tests cover the right edges. Approving."
-- "Solid. One blocker inline — see the auth gap. Request changes."
+- "One small naming point inline, not a blocker. Approving."
+- "One edge-case gap noted inline, minor. Approving."
+- "One blocker inline — see the auth gap. Request changes."
+- "Approving." (nothing to flag)
-Bad (chatty, padded, marketing-adjacent):
+Bad (chatty, padded, or any praise on an approve):
- "Great work overall! This is a really clean addition. The OneToOne relationship behaves as expected, the migration is correctly dependent on 0028, CI is green across all backend/frontend checks, and the tests cover Normal/Boundary/Error cases. One small naming nit inline — fine to roll into a follow-up."
+- "Clean, good coverage. One naming point inline. Approving." — the leading praise is now cut; lead with the pointer instead.
+- "Solid fix. Approving." — a bare positive is still praise; drop it.
If specific praise lands somewhere, surface it as a single inline comment on the relevant line, not in the summary body. The summary stays scannable; the inline pins carry the specifics.
diff --git a/scripts/audit.sh b/scripts/audit.sh
index 86eeb76..a7b8089 100755
--- a/scripts/audit.sh
+++ b/scripts/audit.sh
@@ -144,10 +144,14 @@ for proj in "${projects[@]}"; do
# - the rulesets repo itself (canonical .ai/ lives at the repo root, not under a project)
# - the canonical-source subdir (rulesets/claude-templates/.ai/ is the source, not a target)
# - the legacy standalone claude-templates checkout (frozen during fold transition)
+ # - retired projects under any .retired/ ancestor (~/projects/.retired/<name>) —
+ # shelved, never a live sync target; nested one level deeper than live projects,
+ # so the maxdepth-3 find reaches them and they must be excluded explicitly
case "$proj" in
"$REPO") continue ;;
"$REPO/claude-templates") continue ;;
"$HOME/projects/claude-templates") continue ;;
+ */.retired/*) continue ;;
esac
# Display path: strip $HOME prefix to ~/, otherwise leave alone.
diff --git a/scripts/install-ai.sh b/scripts/install-ai.sh
index 7007eed..8c04e22 100755
--- a/scripts/install-ai.sh
+++ b/scripts/install-ai.sh
@@ -141,6 +141,13 @@ today="$(date +%Y-%m-%d)"
sed "s|\[Project Name\]|$project_name|g; s|\[Date\]|$today|g" \
"$CANONICAL/notes.org" > "$project/.ai/notes.org"
+# Seed AGENTS.md — the runtime-neutral agent entry file (thin pointer at
+# protocols.org, rules, and /name resolution) for Codex-style harnesses.
+# Seed-only, like CLAUDE.md: project-owned after bootstrap, never overwritten.
+if [ ! -e "$project/AGENTS.md" ]; then
+ cp "$REPO/claude-templates/AGENTS.md" "$project/AGENTS.md"
+fi
+
# Tracking setup.
case "$track_mode" in
track)
@@ -172,6 +179,24 @@ case "$track_mode" in
;;
esac
+# Ignore temp/ in BOTH track and gitignore modes (not the "not-a-git-repo"
+# case). temp/ is the home for ephemeral, disposable, regenerable artifacts —
+# throwaway regardless of whether the project tracks its .ai/ tooling, so it
+# rides neither the track .gitkeep step nor the gitignore-only tooling block.
+# working/ is deliberately NOT ignored: it's the tracked home of in-progress
+# work, version-controlled from creation (see working-files.md). Idempotent:
+# accepts either the unanchored `temp/` or anchored `/temp/` form.
+if [ -n "$track_mode" ]; then
+ gi="$project/.gitignore"
+ if ! { [ -f "$gi" ] && { grep -qFx 'temp/' "$gi" || grep -qFx '/temp/' "$gi"; }; }; then
+ {
+ [ -s "$gi" ] && echo ""
+ echo "# Ephemeral working artifacts (throwaway; see working-files.md)"
+ echo "temp/"
+ } >> "$gi"
+ fi
+fi
+
# Banner.
echo
echo "Done."
diff --git a/scripts/install-lang.sh b/scripts/install-lang.sh
index 0fc9ea8..3aaa76e 100755
--- a/scripts/install-lang.sh
+++ b/scripts/install-lang.sh
@@ -33,14 +33,87 @@ fi
# Resolve to absolute path
PROJECT="$(cd "$PROJECT" && pwd)"
+# 0. Cross-bundle collision guard.
+#
+# Several bundles ship a file at the same path. gitignore-add.txt is appended
+# and deduped, and CLAUDE.md is seed-only, so both compose across bundles. Three
+# do not: claude/settings.json and githooks/* are installed with `cp -rT`
+# (always overwrite), and coverage-makefile.txt is seeded under a fixed name. So
+# installing a second bundle replaced the first's hook wiring and pre-commit
+# while printing [ok], so a project could lose its paren check or secret scan
+# and read the output as success. Refuse instead, naming what would go.
+#
+# Detection is by rule fingerprint, matching sync-language-bundle.sh: a project
+# has bundle X iff one of X's own rule files is in .claude/rules/. No marker
+# file, so this works on installs that predate the guard.
+project_has_bundle() {
+ local b="$1" rf
+ for rf in "$REPO_ROOT/languages/$b/claude/rules"/*.md; do
+ [ -f "$rf" ] || continue
+ [ -f "$PROJECT/.claude/rules/$(basename "$rf")" ] && return 0
+ done
+ return 1
+}
+
+# Files $1's bundle and $LANG both ship, and that install would overwrite.
+shared_overwritten_files() {
+ local other="$1" rel
+ for rel in claude/settings.json coverage-makefile.txt; do
+ [ -f "$SRC/$rel" ] && [ -f "$REPO_ROOT/languages/$other/$rel" ] \
+ && echo " $(printf '%s' "$rel" | sed 's|^claude/|.claude/|')"
+ done
+ if [ -d "$SRC/githooks" ] && [ -d "$REPO_ROOT/languages/$other/githooks" ]; then
+ for rel in "$SRC/githooks"/*; do
+ [ -f "$rel" ] || continue
+ [ -f "$REPO_ROOT/languages/$other/githooks/$(basename "$rel")" ] \
+ && echo " githooks/$(basename "$rel")"
+ done
+ fi
+}
+
+if [ "$FORCE" != "1" ]; then
+ collisions=""
+ for other_dir in "$REPO_ROOT/languages"/*/; do
+ other="$(basename "$other_dir")"
+ [ "$other" = "$LANG" ] && continue
+ [ -d "$other_dir/claude/rules" ] || continue
+ project_has_bundle "$other" || continue
+ files="$(shared_overwritten_files "$other")"
+ [ -n "$files" ] && collisions="${collisions}The '$other' bundle is already installed here. Installing '$LANG' would replace:
+${files}
+"
+ done
+
+ if [ -n "$collisions" ]; then
+ {
+ echo "ERROR: bundle collision. Refusing to install '$LANG' into $PROJECT"
+ echo
+ printf '%s' "$collisions"
+ echo "Both bundles ship these files, and installing overwrites rather than merges,"
+ echo "so the bundle already here would silently lose them."
+ echo
+ echo "Whether a project can carry two bundles at once is an open question."
+ echo "Until it's settled, install one bundle per project."
+ echo
+ echo "To override: re-run with FORCE=1. That also re-seeds CLAUDE.md from the"
+ echo "bundle template, which overwrites any project-specific edits to it."
+ } >&2
+ exit 1
+ fi
+fi
+
echo "Installing '$LANG' ruleset into $PROJECT"
-# 1. Generic rules from claude-rules/ (shared across all languages)
-if [ -d "$REPO_ROOT/claude-rules" ]; then
- mkdir -p "$PROJECT/.claude/rules"
- cp "$REPO_ROOT/claude-rules"/*.md "$PROJECT/.claude/rules/" 2>/dev/null || true
- count=$(ls -1 "$REPO_ROOT/claude-rules"/*.md 2>/dev/null | wc -l)
- echo " [ok] .claude/rules/ — $count generic rule(s) from claude-rules/"
+# 1. Generic rules are NOT copied here. They install once at ~/.claude/rules/
+# via `make install` and load in every session on the machine. Copying them per
+# project loaded them twice, and project rules outrank user-level ones — so a
+# stale project copy quietly overrode the fresh global rule. The bundle owns its
+# own language rules only; sync-language-bundle.sh sweeps copies left by earlier
+# installs. A machine that wants the generic rules runs `make install`.
+mkdir -p "$PROJECT/.claude/rules"
+if [ ! -d "$HOME/.claude/rules" ]; then
+ echo " [!!] ~/.claude/rules/ is missing — run 'make install' so the generic"
+ echo " rules are available; this bundle installs language rules only."
fi
# 2. .claude/ — language-specific rules, hooks, settings (authoritative, always overwrite)
@@ -66,13 +139,26 @@ if [ -d "$SRC/githooks" ]; then
fi
fi
-# 3. CLAUDE.md — seed on first install, don't overwrite unless FORCE=1
+# 3. CLAUDE.md — seed on first install, don't overwrite unless FORCE=1.
+# Prefer the bundle's own template; fall back to the language-neutral
+# default so a bundle that ships none still seeds an accurate, non-
+# mislabeling header instead of nothing. The default names no language,
+# so multi-bundle and wrong-bundle installs don't inherit a false header.
+CLAUDE_SRC=""
+CLAUDE_KIND=""
if [ -f "$SRC/CLAUDE.md" ]; then
+ CLAUDE_SRC="$SRC/CLAUDE.md"
+ CLAUDE_KIND="$LANG"
+elif [ -f "$REPO_ROOT/languages/default-CLAUDE.md" ]; then
+ CLAUDE_SRC="$REPO_ROOT/languages/default-CLAUDE.md"
+ CLAUDE_KIND="language-neutral default"
+fi
+if [ -n "$CLAUDE_SRC" ]; then
if [ -f "$PROJECT/CLAUDE.md" ] && [ "$FORCE" != "1" ]; then
echo " [skip] CLAUDE.md already exists (use FORCE=1 to overwrite)"
else
- cp "$SRC/CLAUDE.md" "$PROJECT/CLAUDE.md"
- echo " [ok] CLAUDE.md installed"
+ cp "$CLAUDE_SRC" "$PROJECT/CLAUDE.md"
+ echo " [ok] CLAUDE.md installed ($CLAUDE_KIND)"
fi
fi
@@ -113,5 +199,30 @@ if [ -f "$SRC/gitignore-add.txt" ]; then
fi
fi
+# --- Bundle completeness check ---
+# Every component copy above is guarded by a plain existence test, so a bundle
+# missing one installs silently and reports success. That is how the python and
+# typescript bundles shipped for nearly two months with no pre-commit hook, and
+# therefore no credential scan, on every project that installed them: nothing
+# ever said the component was absent.
+#
+# This can't be fixed by never missing a component — someone adding the seventh
+# bundle will miss one too. It's fixed by the install saying so. Warn, never
+# block: a partial bundle is still worth installing, and turning this into a
+# failure would just teach people to skip the installer.
+missing=""
+[ -f "$SRC/githooks/pre-commit" ] || missing="${missing} - githooks/pre-commit (secret scan on commit)"$'\n'
+[ -f "$SRC/claude/settings.json" ] || missing="${missing} - claude/settings.json (permissions + PostToolUse hook wiring)"$'\n'
+ls "$SRC"/claude/hooks/*.sh >/dev/null 2>&1 || missing="${missing} - claude/hooks/*.sh (validate-on-edit hook)"$'\n'
+[ -f "$SRC/CLAUDE.md" ] || missing="${missing} - CLAUDE.md (seed project instructions)"$'\n'
+
+if [ -n "$missing" ]; then
+ echo ""
+ echo "WARNING: the '$LANG' bundle is incomplete. Not installed, because the bundle doesn't ship them:" >&2
+ printf '%s' "$missing" >&2
+ echo "The install above succeeded; these components are simply absent upstream." >&2
+ echo "Add them under languages/$LANG/ in rulesets, then re-run this install." >&2
+fi
+
echo ""
echo "Install complete."
diff --git a/scripts/lint.sh b/scripts/lint.sh
index ae30aa5..ca6abbd 100755
--- a/scripts/lint.sh
+++ b/scripts/lint.sh
@@ -21,10 +21,22 @@ warn() {
errors=$((errors + 1))
}
+# Print a rule file's body with any leading YAML frontmatter stripped, so the
+# structural checks below see the Markdown regardless of whether the file
+# carries a `paths:` block. Claude Code reads the frontmatter; the heading check
+# should not care that it is there.
+md_body() {
+ awk 'NR==1 && $0=="---" {fm=1; next}
+ fm && $0=="---" {fm=0; next}
+ !fm {print}' "$1"
+}
+
check_md_heading() {
local f="$1"
[ -f "$f" ] || return 0
- if ! head -1 "$f" | grep -q '^# '; then
+ # First non-blank line, so a blank separator after frontmatter doesn't read as
+ # a missing heading.
+ if ! md_body "$f" | grep -m1 -v '^[[:space:]]*$' | grep -q '^# '; then
warn "$f — missing top-level heading"
fi
}
@@ -37,6 +49,27 @@ check_md_applies_to() {
fi
}
+# A rule whose prose declares a file-type scope must also carry `paths:`
+# frontmatter, or Claude Code loads it into every session regardless of what the
+# prose says. Three rules declared a narrow scope this way and were loaded
+# universally for as long as they shipped, because the declaration lived only in
+# a line the loader never reads. The prose is for the human; the frontmatter is
+# what actually scopes the load. Keep them saying the same thing.
+check_md_paths_frontmatter() {
+ local f="$1" applies
+ [ -f "$f" ] || return 0
+ applies=$(grep -m1 '^Applies to:' "$f" 2>/dev/null) || return 0
+ # A scope naming a concrete extension (`**/*.org`, `**/*.el`) is path-scopable.
+ # A bare `**/*` is genuinely universal and wants no frontmatter.
+ case "$applies" in
+ *'**/*.'*)
+ if ! head -1 "$f" | grep -q '^---$'; then
+ warn "$f — declares a file-type scope in prose but has no 'paths:' frontmatter; it loads in every session"
+ fi
+ ;;
+ esac
+}
+
check_hook() {
local f="$1"
[ -f "$f" ] || return 0
@@ -81,6 +114,7 @@ for f in claude-rules/*.md; do
[ -f "$f" ] || continue
check_md_heading "$f"
check_md_applies_to "$f"
+ check_md_paths_frontmatter "$f"
done
# Per-language rule files
@@ -99,6 +133,9 @@ for claude_md in languages/*/CLAUDE.md; do
check_md_heading "$claude_md"
done
+# Language-neutral default CLAUDE.md (install-lang's fallback when a bundle ships none)
+[ -f languages/default-CLAUDE.md ] && check_md_heading languages/default-CLAUDE.md
+
# Hook scripts
for h in languages/*/claude/hooks/*.sh languages/*/githooks/*; do
[ -f "$h" ] || continue
@@ -111,6 +148,14 @@ for s in scripts/*.sh; do
check_hook "$s"
done
+# Scripts `make install` symlinks onto PATH. Extensionless by convention, so
+# they need their own glob — scripts/*.sh above never matched them, which left
+# the most exposed shell in the repo as the only shell with no gate over it.
+for s in claude-templates/bin/*; do
+ [ -f "$s" ] || continue
+ check_hook "$s"
+done
+
# Markdown link validation across rules and skills
for f in claude-rules/*.md */SKILL.md; do
[ -f "$f" ] || continue
diff --git a/scripts/remove.sh b/scripts/remove.sh
index 3d8b7e4..3d8b7e4 100644..100755
--- a/scripts/remove.sh
+++ b/scripts/remove.sh
diff --git a/scripts/roam-sync.sh b/scripts/roam-sync.sh
index 55422ec..ef43c8f 100755
--- a/scripts/roam-sync.sh
+++ b/scripts/roam-sync.sh
@@ -3,8 +3,13 @@
#
# Commit any local changes, rebase onto the remote, push. Run by the
# roam-sync systemd user timer (scripts/systemd/) every 15 minutes so
-# Craig's hand edits travel without a manual git step. Agents don't need
-# this — they pull/commit/push inline per claude-rules/knowledge-base.md.
+# Craig's hand edits travel without a manual git step. This script is the
+# roam repo's only committer (the 2026-06-24 one-git-owner rule): the tree
+# is chronically dirty from live captures, so a second committer risks
+# sweeping an in-flight capture into a stray commit. Agents edit the working
+# tree under the roam-write lock, then trigger this unit (systemctl --user
+# start roam-sync.service) instead of committing themselves — see
+# claude-rules/knowledge-base.md and the inbox workflow's roam mode.
#
# On a rebase conflict: abort the rebase (never leave the repo mid-rebase
# for a timer to mangle), keep the local commit, exit 1 so the failure is
diff --git a/scripts/signal-receive.sh b/scripts/signal-receive.sh
new file mode 100755
index 0000000..8c3ef01
--- /dev/null
+++ b/scripts/signal-receive.sh
@@ -0,0 +1,51 @@
+#!/usr/bin/env bash
+# signal-receive.sh — drain the Signal pager account's inbound queue.
+#
+# The pager identity (+15045173983) lives on velox (primary) and any linked
+# device (ratio). The Signal protocol expects a registered account to receive
+# regularly; when it goes quiet, signal-cli prints a staleness warning
+# ("Messages have been last received N days ago") and the account drifts toward
+# an unhealthy state. This script pulls anything queued and exits, keeping the
+# account warm. Run by the signal-receive systemd user timer (scripts/systemd/)
+# on every machine holding the account — the roam-sync-shaped fix for the
+# receive-staleness caveat on the pager channel.
+#
+# Draining the queue is also what surfaces Craig's replies: a reply he types in
+# Signal arrives as a data message here, so a warm account is a prerequisite for
+# the read-replies half of the pager (see docs/design/…-signal-pager-runbook.org).
+#
+# The units stow via the shared dotfiles `common` package, so the timer may land
+# on a machine that does not hold the account. That is fine: the script no-ops
+# cleanly (exit 0) when the account is not registered locally, rather than
+# erroring every cadence.
+#
+# Usage: signal-receive.sh [account] [timeout-seconds]
+# account defaults to the pager identity below
+# timeout-seconds seconds to wait for new messages (default 10)
+#
+# Exit: 0 on a clean receive (including "nothing queued" and "account not on this
+# machine"), non-zero only if signal-cli itself errors, so a real failure is
+# visible in `systemctl --user status signal-receive`.
+
+set -euo pipefail
+
+account="${1:-+15045173983}"
+timeout="${2:-10}"
+
+if ! command -v signal-cli >/dev/null 2>&1; then
+ echo "signal-receive: signal-cli not on PATH — nothing to do" >&2
+ exit 0
+fi
+
+# No-op cleanly where the account isn't registered (a machine that stows the
+# common units but isn't a pager device). Only a genuine receive error should
+# surface as a failure.
+if ! signal-cli listAccounts 2>/dev/null | grep -q "$account"; then
+ echo "signal-receive: $account not registered on this machine — nothing to do" >&2
+ exit 0
+fi
+
+# `receive` prints envelopes to stdout and returns 0 once the queue drains or
+# the timeout elapses. --send-read-receipts tells Craig's phone his replies were
+# read by the pager, matching normal Signal behavior.
+exec signal-cli -a "$account" receive --timeout "$timeout" --send-read-receipts
diff --git a/scripts/sweep-gitignore-tooling.sh b/scripts/sweep-gitignore-tooling.sh
index 63fc066..7194d6c 100755
--- a/scripts/sweep-gitignore-tooling.sh
+++ b/scripts/sweep-gitignore-tooling.sh
@@ -10,13 +10,19 @@
#
# For each AI project (a directory with .ai/protocols.org) under the search
# roots, if it's a git checkout in gitignore mode (.ai/ already appears in its
-# .gitignore), ensure .ai/, .claude/, CLAUDE.md, and AGENTS.md are all ignored.
-# Append only the missing lines, so a re-run is a no-op.
+# .gitignore, in either the unanchored `.ai/` or anchored `/.ai/` form), ensure
+# .ai/, .claude/, CLAUDE.md, and AGENTS.md are all ignored. Append only the
+# missing lines, in whichever style the file already uses, so a re-run is a
+# no-op.
#
# Track-mode projects (.ai/ NOT in .gitignore) are skipped by design: they
# track their tooling on purpose — team repos sharing config with teammates who
# don't run rulesets, or private-remote personal repos where the history IS the
-# project.
+# project. But a track-mode project whose tracked tooling is reachable from a
+# non-cjennings.net remote gets a loud WARN: per convention the tooling set is
+# gitignored anywhere the repo can reach a public host, and a server-side
+# mirror hook can publish even a "private" remote (the 2026-06-30 .emacs.d
+# exposure rode exactly that).
#
# A line added here only stops *future* commits. If a target path is already
# tracked, the ignore has no effect until it's untracked; the sweep warns so
@@ -30,6 +36,14 @@ set -euo pipefail
IGNORE_SET=('.ai/' '.claude/' 'CLAUDE.md' 'AGENTS.md')
+# A pattern counts as present in either the unanchored (`.ai/`) or anchored
+# (`/.ai/`) form — both ignore the root-level path; treating them as different
+# is what silently skipped anchored-style projects.
+has_ignore() {
+ local pat="$1" gi="$2"
+ grep -qFx "$pat" "$gi" || grep -qFx "/$pat" "$gi"
+}
+
dry_run=0
roots=()
for arg in "$@"; do
@@ -72,16 +86,41 @@ for project in "${projects[@]}"; do
continue
fi
- # Gitignore mode iff .ai/ is already ignored. Otherwise track-mode: leave it.
- if [ ! -f "$gi" ] || ! grep -qFx '.ai/' "$gi"; then
- echo "skip $name — track-mode (.ai/ not gitignored)"
+ # Gitignore mode iff .ai/ is already ignored (either style). Otherwise
+ # track-mode: leave the .gitignore alone, but warn when tracked tooling can
+ # reach a non-cjennings.net remote — a track-mode repo on a public host (or
+ # behind an invisible server-side mirror) is the exposure the convention
+ # exists to prevent.
+ if [ ! -f "$gi" ] || ! has_ignore '.ai/' "$gi"; then
+ tracked_tooling=()
+ for pat in "${IGNORE_SET[@]}"; do
+ path="${pat%/}"
+ if git -C "$project" ls-files --error-unmatch "$path" >/dev/null 2>&1; then
+ tracked_tooling+=("$path")
+ fi
+ done
+ # Private = the cjennings.net server, whether addressed by FQDN or by the
+ # bare `cjennings` ssh-config alias (git@cjennings:repo.git).
+ public_remote="$(git -C "$project" remote -v 2>/dev/null \
+ | awk '{print $2}' | grep -vE '(@|://)cjennings(\.net)?[:/]' | sort -u | head -1 || true)"
+ if [ "${#tracked_tooling[@]}" -gt 0 ] && [ -n "$public_remote" ]; then
+ echo "skip $name — track-mode (.ai/ not gitignored)"
+ echo " WARN $name: tracked tooling (${tracked_tooling[*]}) is publicly reachable via $public_remote — gitignore the set and 'git -C $project rm --cached -r <path>' unless this is a deliberate team-shared config"
+ else
+ echo "skip $name — track-mode (.ai/ not gitignored)"
+ fi
skipped=$((skipped + 1))
continue
fi
+ # Append in the style the file already uses: anchored if its .ai/ marker
+ # line is the anchored form.
+ prefix=""
+ grep -qFx '/.ai/' "$gi" && prefix="/"
+
needed=()
for pat in "${IGNORE_SET[@]}"; do
- grep -qFx "$pat" "$gi" || needed+=("$pat")
+ has_ignore "$pat" "$gi" || needed+=("${prefix}${pat}")
done
if [ "${#needed[@]}" -eq 0 ]; then
@@ -103,16 +142,48 @@ for project in "${projects[@]}"; do
swept=$((swept + 1))
# Warn on any newly-ignored path that's already tracked — the ignore won't
- # untrack it.
+ # untrack it. Strip the anchored prefix before asking git: the pattern
+ # `/CLAUDE.md` is the repo-relative path `CLAUDE.md`.
for pat in "${needed[@]}"; do
- path="${pat%/}"
+ path="${pat#/}"
+ path="${path%/}"
if git -C "$project" ls-files --error-unmatch "$path" >/dev/null 2>&1; then
echo " WARN $name: $path is currently tracked — 'git -C $project rm --cached -r $path' to untrack"
fi
done
done
+# temp/ backfill — mode-independent. Unlike the personal-tooling set above
+# (gitignore-mode only, track-mode deliberately skipped), temp/ holds ephemeral
+# artifacts in every project regardless of whether it tracks its .ai/, so both
+# track- and gitignore-mode projects get it. A separate pass, not an IGNORE_SET
+# member — folding it into that set would miss every track-mode project, which
+# is exactly the set that needs it. working/ is never ignored: it's the tracked
+# home of in-progress work.
+temp_swept=0
+for project in "${projects[@]}"; do
+ [ -d "$project/.git" ] || continue
+ name="$(basename "$project")"
+ gi="$project/.gitignore"
+
+ if [ -f "$gi" ] && { grep -qFx 'temp/' "$gi" || grep -qFx '/temp/' "$gi"; }; then
+ continue
+ fi
+
+ if [ "$dry_run" -eq 1 ]; then
+ echo "DRY $name — would add: temp/"
+ else
+ {
+ [ -s "$gi" ] && echo ""
+ echo "# Ephemeral working artifacts (throwaway; see working-files.md; swept $(date +%Y-%m-%d))"
+ echo "temp/"
+ } >> "$gi"
+ echo "temp $name — added: temp/"
+ fi
+ temp_swept=$((temp_swept + 1))
+done
+
echo
-echo "Summary: $swept swept, $complete already complete, $skipped skipped (of ${#projects[@]} projects)."
+echo "Summary: $swept swept, $complete already complete, $skipped skipped (of ${#projects[@]} projects); $temp_swept temp/ backfilled."
[ "$dry_run" -eq 1 ] && echo "(dry-run — no files written)"
exit 0
diff --git a/scripts/sync-language-bundle.sh b/scripts/sync-language-bundle.sh
index 45f8259..fe922af 100755
--- a/scripts/sync-language-bundle.sh
+++ b/scripts/sync-language-bundle.sh
@@ -105,7 +105,7 @@ inbox_drop() {
# shared generic rules, its hooks/githooks, and surfaces settings.json.
process_bundle() {
local kind="$1" src="${2%/}"
- local name rules rf f rel manual_before
+ local name rules rf f rel manual_before base
name="$(basename "$src")"
rules="$src/claude/rules"
[ -d "$rules" ] || return 0
@@ -126,12 +126,27 @@ process_bundle() {
HEADER_DONE=0
manual_before=$MANUAL
- # AUTO-FIX: language bundles carry the shared generic rules; team overlays
- # carry only their own rule(s).
+ # SWEEP: generic rules are installed once at ~/.claude/rules/ by `make
+ # install` and load in every session. Language bundles used to copy them into
+ # each project too, which made Claude Code load them twice — and project rules
+ # take priority over user-level ones, so a stale project copy silently
+ # overrode the fresh global rule until the next startup healed it. The bundle
+ # now owns only its own language rules, and sweeps the duplicates it shipped
+ # before.
+ #
+ # The sweep only fires when the global rule is actually present to take over.
+ # On a machine mid-bootstrap, or one where `make install` has not run, the
+ # project copy is the only copy and removing it would leave no rule at all.
if [ "$kind" = language ]; then
for f in "$GENERIC_RULES"/*.md; do
[ -f "$f" ] || continue
- fix "$f" "$PROJECT/.claude/rules/$(basename "$f")"
+ base="$(basename "$f")"
+ [ -f "$PROJECT/.claude/rules/$base" ] || continue
+ [ -f "$HOME/.claude/rules/$base" ] || continue
+ rm -f "$PROJECT/.claude/rules/$base"
+ ensure_header
+ OUT+=" swept .claude/rules/$base (duplicate of ~/.claude/rules/)"$'\n'
+ FIXED=$((FIXED + 1))
done
fi
for f in "$rules"/*.md; do
diff --git a/scripts/systemd/signal-receive.service b/scripts/systemd/signal-receive.service
new file mode 100644
index 0000000..d63a5c8
--- /dev/null
+++ b/scripts/systemd/signal-receive.service
@@ -0,0 +1,10 @@
+# Signal pager receive — keep the pager account warm on every machine holding it.
+# Stowed via the dotfiles `common` package, so it lands on both daily drivers;
+# the script no-ops cleanly where the account isn't registered. Enable per machine:
+# systemctl --user daemon-reload && systemctl --user enable --now signal-receive.timer
+[Unit]
+Description=Drain the Signal pager account inbound queue (keeps it warm)
+
+[Service]
+Type=oneshot
+ExecStart=%h/code/rulesets/scripts/signal-receive.sh
diff --git a/scripts/systemd/signal-receive.timer b/scripts/systemd/signal-receive.timer
new file mode 100644
index 0000000..3fdaee0
--- /dev/null
+++ b/scripts/systemd/signal-receive.timer
@@ -0,0 +1,10 @@
+[Unit]
+Description=Keep the Signal pager account warm every 15 minutes
+
+[Timer]
+OnBootSec=3min
+OnUnitActiveSec=15min
+RandomizedDelaySec=60
+
+[Install]
+WantedBy=timers.target
diff --git a/scripts/tests/agent-text.bats b/scripts/tests/agent-text.bats
new file mode 100644
index 0000000..e5d80fc
--- /dev/null
+++ b/scripts/tests/agent-text.bats
@@ -0,0 +1,78 @@
+#!/usr/bin/env bats
+# agent-text — the runtime-neutral Signal phone messenger ("text me"). Reaches
+# Craig over Signal from any machine on the tailnet: sends directly wherever the
+# account is registered locally (velox, or any linked device), and ssh-relays to
+# velox from a machine that doesn't hold the account. These tests stub
+# ssh/signal-cli on PATH to verify command construction without a network or a
+# phone. The signal-cli stub answers `listAccounts` to control which branch the
+# dispatch takes. The final test covers the deprecated agent-page shim.
+
+setup() {
+ REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)"
+ PAGE="$REPO_ROOT/claude-templates/bin/agent-text"
+ SHIM="$REPO_ROOT/claude-templates/bin/agent-page"
+ STUBS="$(mktemp -d)"
+ LOG="$STUBS/calls.log"
+ cat > "$STUBS/ssh" <<EOF
+#!/bin/bash
+echo "ssh \$*" >> "$LOG"
+exit 0
+EOF
+ # HAS_ACCOUNT controls the listAccounts answer: "1" → the account is
+ # local (send direct), unset/"0" → not local (relay).
+ cat > "$STUBS/signal-cli" <<EOF
+#!/bin/bash
+if [ "\$1" = "listAccounts" ]; then
+ [ "\${HAS_ACCOUNT:-0}" = "1" ] && echo "Number: +15045173983"
+ exit 0
+fi
+echo "signal-cli \$*" >> "$LOG"
+exit 0
+EOF
+ chmod +x "$STUBS/ssh" "$STUBS/signal-cli"
+}
+
+teardown() {
+ rm -rf "$STUBS"
+}
+
+@test "no message exits 2 with usage" {
+ run bash "$PAGE"
+ [ "$status" -eq 2 ]
+ [[ "$output" == *"usage"* ]]
+}
+
+@test "relays through ssh to velox when the account is not local" {
+ HAS_ACCOUNT=0 PATH="$STUBS:$PATH" run bash "$PAGE" build finished
+ [ "$status" -eq 0 ]
+ grep -q "^ssh .*velox" "$LOG"
+ grep -q "15045173983" "$LOG"
+ grep -q "b1b5601e-6126-47f8-afaa-0a59f5188fde" "$LOG"
+ # printf %q escapes the space, so the relayed message reads build\ finished.
+ grep -qF 'build\ finished' "$LOG"
+}
+
+@test "sends directly (no ssh) when the pager account is registered locally" {
+ HAS_ACCOUNT=1 PATH="$STUBS:$PATH" run bash "$PAGE" hello
+ [ "$status" -eq 0 ]
+ grep -q "^signal-cli -a +15045173983 send " "$LOG"
+ ! grep -q "^ssh " "$LOG"
+}
+
+@test "a failed relay reports the desktop fallback and propagates failure" {
+ cat > "$STUBS/ssh" <<'EOF'
+#!/bin/bash
+exit 255
+EOF
+ chmod +x "$STUBS/ssh"
+ HAS_ACCOUNT=0 PATH="$STUBS:$PATH" run bash "$PAGE" urgent thing
+ [ "$status" -ne 0 ]
+ [[ "$output" == *"notify"* ]]
+}
+
+@test "the deprecated agent-page shim delegates to agent-text" {
+ HAS_ACCOUNT=1 PATH="$STUBS:$PATH" run bash "$SHIM" via shim
+ [ "$status" -eq 0 ]
+ # Reaches the same direct-send path as agent-text.
+ grep -q "^signal-cli -a +15045173983 send " "$LOG"
+}
diff --git a/scripts/tests/ai-launcher-characterization.bats b/scripts/tests/ai-launcher-characterization.bats
new file mode 100644
index 0000000..5b93ff3
--- /dev/null
+++ b/scripts/tests/ai-launcher-characterization.bats
@@ -0,0 +1,365 @@
+#!/usr/bin/env bats
+# Characterization tests for the ai launcher's internal functions.
+#
+# These pin the launcher's CURRENT behavior (record-not-spec, per testing.md)
+# before the hardening refactor, so the extraction of pure cores and the
+# footgun fixes can proceed against a green net. The pure/near-pure functions
+# are exercised by sourcing bin/ai (the run-vs-sourced guard keeps dispatch
+# off) and calling them directly; the tmux-coupled pipeline is exercised
+# against a throwaway tmux server on a PRIVATE socket (TMUX_TMPDIR under the
+# test tmpdir, TMUX unset) so nothing can reach Craig's live 'ai' session.
+
+# The launcher stores candidates as literal "~/..." display strings and expands
+# them downstream; the assertions below compare against those literal tildes on
+# purpose, so SC2088 (tilde-in-quotes) does not apply. SC2030/SC2031 fire on
+# the shared `candidates` global because bats runs each @test in its own
+# subshell — the array is set and read within that same subshell, so the
+# warning is a false positive here.
+# shellcheck disable=SC2088,SC2030,SC2031
+
+setup() {
+ REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)"
+ AI="$REPO_ROOT/claude-templates/bin/ai"
+ WORK="$(mktemp -d)"
+ # Isolate every tmux call in this file onto a private server.
+ export TMUX_TMPDIR="$WORK"
+ unset TMUX
+ # Source for direct function access. The guard skips main() when sourced.
+ # shellcheck disable=SC1090
+ source "$AI"
+}
+
+teardown() {
+ tmux kill-server 2>/dev/null || true
+ rm -rf "$WORK"
+}
+
+# --- git repo helpers ---------------------------------------------------------
+
+# A committed repo with no remote/upstream. Auto-maintenance off so background
+# git can't race the teardown rm -rf (the rename-ai flake, fixed 2026-07-19).
+_mk_repo() {
+ local d="$1"
+ git init -q "$d"
+ git -C "$d" config user.email t@example.com
+ git -C "$d" config user.name tester
+ git -C "$d" config commit.gpgsign false
+ git -C "$d" config gc.auto 0
+ git -C "$d" config maintenance.auto false
+ git -C "$d" commit -q --allow-empty -m init
+}
+
+# A repo with a bare origin and a tracked branch, in sync at one commit.
+_mk_repo_upstream() {
+ local d="$1" remote="$1.remote"
+ git init -q --bare "$remote"
+ _mk_repo "$d"
+ git -C "$d" remote add origin "$remote"
+ git -C "$d" push -q -u origin HEAD
+}
+
+# --- usage() ------------------------------------------------------------------
+
+@test "usage: -h prints the help banner and exits 0" {
+ run bash "$AI" -h
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"Usage:"* ]]
+ [[ "$output" == *"--attach"* ]]
+ [[ "$output" == *"Single-project mode"* ]]
+}
+
+# --- maybe_add_candidate() ----------------------------------------------------
+
+@test "maybe_add_candidate: a HOME-rooted project dir is added as ~/rel" {
+ HOME="$WORK/home"
+ mkdir -p "$HOME/code/proj/.ai"
+ touch "$HOME/code/proj/.ai/protocols.org"
+ candidates=()
+ maybe_add_candidate "$HOME/code/proj"
+ [ "${#candidates[@]}" -eq 1 ]
+ [ "${candidates[0]}" = "~/code/proj" ]
+}
+
+@test "maybe_add_candidate: a dir without .ai/protocols.org is filtered out" {
+ HOME="$WORK/home"
+ mkdir -p "$HOME/code/plain"
+ candidates=()
+ maybe_add_candidate "$HOME/code/plain" || true
+ [ "${#candidates[@]}" -eq 0 ]
+}
+
+@test "maybe_add_candidate: a non-HOME dir keeps its full path after ~/ (current quirk)" {
+ HOME="$WORK/home"
+ mkdir -p "$WORK/outside/.ai"
+ touch "$WORK/outside/.ai/protocols.org"
+ candidates=()
+ maybe_add_candidate "$WORK/outside"
+ # $HOME/ prefix doesn't match, so the path is left whole behind the literal ~/.
+ [ "${candidates[0]}" = "~/$WORK/outside" ]
+}
+
+# --- git_status_indicator() ---------------------------------------------------
+
+@test "git_status_indicator: a non-git dir yields no annotation" {
+ mkdir -p "$WORK/plain"
+ run git_status_indicator "$WORK/plain"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "git_status_indicator: clean repo with no upstream reads (no upstream)" {
+ _mk_repo "$WORK/r"
+ run git_status_indicator "$WORK/r"
+ [ "$output" = " (no upstream)" ]
+}
+
+@test "git_status_indicator: an untracked file with no upstream reads (no upstream dirty)" {
+ _mk_repo "$WORK/r"
+ touch "$WORK/r/scratch"
+ run git_status_indicator "$WORK/r"
+ [ "$output" = " (no upstream dirty)" ]
+}
+
+@test "git_status_indicator: inbox-only delivery is visible but not called dirty" {
+ _mk_repo "$WORK/r"
+ mkdir -p "$WORK/r/inbox"
+ touch "$WORK/r/inbox/from-home.org"
+ run git_status_indicator "$WORK/r"
+ [ "$output" = " (no upstream inbox)" ]
+}
+
+@test "git_status_indicator: in-sync tracked repo reads (✓)" {
+ _mk_repo_upstream "$WORK/r"
+ run git_status_indicator "$WORK/r"
+ [ "$output" = " (✓)" ]
+}
+
+@test "git_status_indicator: one unpushed commit reads (↑1)" {
+ _mk_repo_upstream "$WORK/r"
+ git -C "$WORK/r" commit -q --allow-empty -m ahead
+ run git_status_indicator "$WORK/r"
+ [ "$output" = " (↑1)" ]
+}
+
+@test "git_status_indicator: one commit behind upstream reads (↓1)" {
+ _mk_repo_upstream "$WORK/r"
+ git -C "$WORK/r" commit -q --allow-empty -m b
+ git -C "$WORK/r" push -q origin HEAD
+ git -C "$WORK/r" reset -q --hard HEAD~1
+ run git_status_indicator "$WORK/r"
+ [ "$output" = " (↓1)" ]
+}
+
+@test "auto_pull_if_clean fast-forwards with an untracked inbox delivery" {
+ _mk_repo_upstream "$WORK/r"
+ git clone -q "$WORK/r.remote" "$WORK/writer"
+ git -C "$WORK/writer" config user.email t@example.com
+ git -C "$WORK/writer" config user.name tester
+ git -C "$WORK/writer" commit -q --allow-empty -m remote
+ git -C "$WORK/writer" push -q
+ git -C "$WORK/r" fetch -q
+ mkdir -p "$WORK/r/inbox"
+ printf 'handoff\n' >"$WORK/r/inbox/from-home.org"
+
+ run auto_pull_if_clean "$WORK/r"
+ [ "$status" -eq 0 ]
+ [ "$(git -C "$WORK/r" rev-parse HEAD)" = "$(git -C "$WORK/r" rev-parse '@{u}')" ]
+ [ -f "$WORK/r/inbox/from-home.org" ]
+}
+
+@test "auto_pull_if_clean refuses a non-inbox untracked file" {
+ _mk_repo_upstream "$WORK/r"
+ git clone -q "$WORK/r.remote" "$WORK/writer"
+ git -C "$WORK/writer" config user.email t@example.com
+ git -C "$WORK/writer" config user.name tester
+ git -C "$WORK/writer" commit -q --allow-empty -m remote
+ git -C "$WORK/writer" push -q
+ git -C "$WORK/r" fetch -q
+ printf 'scratch\n' >"$WORK/r/scratch"
+ before="$(git -C "$WORK/r" rev-parse HEAD)"
+
+ run auto_pull_if_clean "$WORK/r"
+ [ "$status" -eq 0 ]
+ [ "$(git -C "$WORK/r" rev-parse HEAD)" = "$before" ]
+}
+
+# --- annotate_candidates() ----------------------------------------------------
+
+@test "annotate_candidates: appends each candidate's status suffix" {
+ _mk_repo_upstream "$WORK/home/code/clean"
+ mkdir -p "$WORK/home/code/plain"
+ HOME="$WORK/home"
+ candidates=("~/code/clean" "~/code/plain")
+ annotate_candidates
+ [ "${candidates[0]}" = "~/code/clean (✓)" ]
+ # A non-git candidate gets an empty suffix (git_status_indicator returns "").
+ [ "${candidates[1]}" = "~/code/plain" ]
+}
+
+# --- read_selections() --------------------------------------------------------
+
+@test "read_selections: strips the ' (annotation)' suffix from each line" {
+ read_selections "$(printf '~/code/a (✓)\n~/projects/b (↑1 dirty)')"
+ [ "${#selected[@]}" -eq 2 ]
+ [ "${selected[0]}" = "~/code/a" ]
+ [ "${selected[1]}" = "~/projects/b" ]
+}
+
+@test "read_selections: a line with no annotation is kept verbatim" {
+ read_selections "~/code/plain"
+ [ "${selected[0]}" = "~/code/plain" ]
+}
+
+@test "read_selections: empty input yields a single empty element (current quirk)" {
+ read_selections ""
+ [ "${#selected[@]}" -eq 1 ]
+ [ -z "${selected[0]}" ]
+}
+
+# --- build_candidates() -------------------------------------------------------
+
+@test "build_candidates: discovers .ai projects under HOME, filters, sorts" {
+ HOME="$WORK/home"
+ mkdir -p "$HOME/.emacs.d/.ai" "$HOME/code/beta/.ai" "$HOME/code/alpha/.ai" \
+ "$HOME/code/plain" "$HOME/projects/gamma/.ai"
+ touch "$HOME/.emacs.d/.ai/protocols.org" \
+ "$HOME/code/beta/.ai/protocols.org" \
+ "$HOME/code/alpha/.ai/protocols.org" \
+ "$HOME/projects/gamma/.ai/protocols.org"
+ # '|| true' suspends bats's errexit so the function runs to completion as it
+ # does in production (bin/ai runs without set -e). Its non-zero exit is
+ # incidental — it inherits the last maybe_add_candidate's filter result,
+ # which callers ignore; see the file's top-of-script NOTE.
+ build_candidates || true
+ # .emacs.d first (probed first), then ~/code sorted, then ~/projects.
+ [ "${candidates[0]}" = "~/.emacs.d" ]
+ [ "${candidates[1]}" = "~/code/alpha" ]
+ [ "${candidates[2]}" = "~/code/beta" ]
+ [ "${candidates[3]}" = "~/projects/gamma" ]
+ [ "${#candidates[@]}" -eq 4 ]
+}
+
+@test "build_candidates: a HOME with no .ai projects yields an empty list" {
+ HOME="$WORK/empty-home"
+ mkdir -p "$HOME/code/plain"
+ build_candidates || true
+ [ "${#candidates[@]}" -eq 0 ]
+}
+
+# --- functional: tmux-coupled pipeline (private socket) -----------------------
+
+@test "functional create_window: adds a named window and returns its id" {
+ export AGENT_CMD="true"
+ tmux new-session -d -s ai -n base -c "$WORK"
+ run create_window "$WORK" "proj-x"
+ [ "$status" -eq 0 ]
+ [ -n "$output" ]
+ tmux list-windows -t ai -F '#{window_name}' | grep -qx proj-x
+}
+
+@test "functional find_window_id: returns the id for a name, empty for a miss" {
+ tmux new-session -d -s ai -n only -c "$WORK"
+ run find_window_id only
+ [ -n "$output" ]
+ run find_window_id nope
+ [ -z "$output" ]
+}
+
+@test "functional sort_windows: others alpha-first, then projects alpha" {
+ HOME="$WORK/home"
+ mkdir -p "$HOME/code/alpha/.ai" "$HOME/code/beta/.ai"
+ touch "$HOME/code/alpha/.ai/protocols.org" "$HOME/code/beta/.ai/protocols.org"
+ tmux new-session -d -s ai -n zzz-other -c "$WORK"
+ tmux new-window -t ai -n beta
+ tmux new-window -t ai -n alpha
+ tmux new-window -t ai -n aaa-other
+ run sort_windows
+ [ "$status" -eq 0 ]
+ order="$(tmux list-windows -t ai -F '#{window_name}' | paste -sd, -)"
+ [ "$order" = "aaa-other,zzz-other,alpha,beta" ]
+}
+
+@test "functional attach_mode: no session prints an error and exits 1" {
+ run bash "$AI" --attach
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"no 'ai' session"* ]]
+}
+
+# --- extracted pure cores -----------------------------------------------------
+
+@test "_git_prep_action: no upstream is always 'none'" {
+ run _git_prep_action 0 0 0 5
+ [ "$output" = none ]
+}
+
+@test "_git_prep_action: clean and purely behind is 'pull'" {
+ run _git_prep_action 1 0 0 3
+ [ "$output" = pull ]
+}
+
+@test "_git_prep_action: in sync with upstream is 'none'" {
+ run _git_prep_action 1 0 0 0
+ [ "$output" = none ]
+}
+
+@test "_git_prep_action: ahead is 'report', never 'pull'" {
+ run _git_prep_action 1 0 2 0
+ [ "$output" = report ]
+}
+
+@test "_git_prep_action: dirty is 'report', never 'pull'" {
+ run _git_prep_action 1 1 0 0
+ [ "$output" = report ]
+}
+
+@test "_git_prep_action: behind AND dirty is 'report' — a dirty repo is never auto-pulled" {
+ run _git_prep_action 1 1 0 3
+ [ "$output" = report ]
+}
+
+@test "_git_prep_action: diverged (ahead and behind) is 'report'" {
+ run _git_prep_action 1 0 2 3
+ [ "$output" = report ]
+}
+
+@test "_order_windows: others alpha, then projects alpha" {
+ local listing names out
+ listing="$(printf 'beta\t@1\nzzz-other\t@2\nalpha\t@3\naaa-other\t@4')"
+ names="$(printf 'alpha\nbeta')"
+ out="$(printf '%s\n' "$listing" | _order_windows "$names" | cut -f1 | paste -sd, -)"
+ [ "$out" = "aaa-other,zzz-other,alpha,beta" ]
+}
+
+@test "_order_windows: all-projects input keeps only the projects, sorted" {
+ local out
+ out="$(printf 'beta\t@1\nalpha\t@2\n' | _order_windows "$(printf 'alpha\nbeta')" | cut -f1 | paste -sd, -)"
+ [ "$out" = "alpha,beta" ]
+}
+
+@test "_order_windows: empty input yields empty output" {
+ run _order_windows "$(printf 'alpha\nbeta')" <<<""
+ [ -z "$output" ]
+}
+
+@test "_order_windows: a name matches only as a whole line, not as a substring" {
+ # 'alph' must not match project 'alpha' (grep -qxF is a full-line match).
+ local out
+ out="$(printf 'alph\t@1\n' | _order_windows "$(printf 'alpha')" | cut -f1)"
+ # 'alph' is not a project, so it lands in others, still present.
+ [ "$out" = "alph" ]
+}
+
+@test "_match_window_id: returns the id for a name, empty for a miss" {
+ local listing out
+ listing="$(printf 'one\t@1\ntwo\t@2\n')"
+ out="$(printf '%s\n' "$listing" | _match_window_id two)"
+ [ "$out" = "@2" ]
+ out="$(printf '%s\n' "$listing" | _match_window_id nope)"
+ [ -z "$out" ]
+}
+
+@test "_match_window_id: first match wins and stops" {
+ local out
+ out="$(printf 'dup\t@1\ndup\t@2\n' | _match_window_id dup)"
+ [ "$out" = "@1" ]
+}
diff --git a/scripts/tests/ai-launcher-runtime.bats b/scripts/tests/ai-launcher-runtime.bats
new file mode 100644
index 0000000..3f0ad69
--- /dev/null
+++ b/scripts/tests/ai-launcher-runtime.bats
@@ -0,0 +1,97 @@
+#!/usr/bin/env bats
+# The ai launcher's --runtime flag: pick the agent CLI (claude, codex) that
+# a project window launches. Part of the generic-agent-runtime arc — the
+# tmux-side counterpart of .emacs.d's ai-term multi-LLM handoff. The
+# --print-launch mode exists for exactly these tests: it prints the launch
+# command a real run would send to the pane, without touching tmux or fzf.
+
+setup() {
+ REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)"
+ AI="$REPO_ROOT/claude-templates/bin/ai"
+ PROJ="$(mktemp -d)"
+ mkdir -p "$PROJ/.ai"
+ touch "$PROJ/.ai/protocols.org"
+ STUB_BIN="$(mktemp -d)"
+}
+
+teardown() {
+ rm -rf "$PROJ" "$STUB_BIN"
+}
+
+@test "default runtime launches claude" {
+ run bash "$AI" --print-launch "$PROJ"
+ [ "$status" -eq 0 ]
+ [[ "$output" == claude\ * ]]
+ [[ "$output" == *"protocols.org"* ]]
+}
+
+@test "--runtime codex launches codex with the same opening line" {
+ run bash "$AI" --runtime codex --print-launch "$PROJ"
+ [ "$status" -eq 0 ]
+ [[ "$output" == codex\ * ]]
+ [[ "$output" == *"protocols.org"* ]]
+ [[ "$output" == *"$(basename "$PROJ")"* ]]
+}
+
+@test "AI_RUNTIME env selects the runtime without the flag" {
+ AI_RUNTIME=codex run bash "$AI" --print-launch "$PROJ"
+ [ "$status" -eq 0 ]
+ [[ "$output" == codex\ * ]]
+}
+
+@test "--runtime local launches codex --oss over ollama with the default local model" {
+ run bash "$AI" --runtime local --print-launch "$PROJ"
+ [ "$status" -eq 0 ]
+ [[ "$output" == "codex --oss --local-provider=ollama -m gpt-oss:120b "* ]]
+ [[ "$output" == *"protocols.org"* ]]
+}
+
+@test "AI_LOCAL_MODEL overrides the local runtime's model" {
+ AI_LOCAL_MODEL=qwen3-coder:30b run bash "$AI" --runtime local --print-launch "$PROJ"
+ [ "$status" -eq 0 ]
+ [[ "$output" == "codex --oss --local-provider=ollama -m qwen3-coder:30b "* ]]
+}
+
+@test "an unknown runtime errors and names the valid ones" {
+ run bash "$AI" --runtime frobnitz --print-launch "$PROJ"
+ [ "$status" -eq 2 ]
+ [[ "$output" == *"claude"* ]]
+ [[ "$output" == *"codex"* ]]
+}
+
+@test "--print-launch refuses a non-project directory" {
+ run bash "$AI" --print-launch "$BATS_TEST_TMPDIR"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"protocols.org"* ]]
+}
+
+@test "--print-runtimes lists claude, codex, and one line per ollama model" {
+ cat > "$STUB_BIN/ollama" <<'STUB'
+#!/bin/bash
+if [ "$1" = "list" ]; then
+ printf 'NAME ID SIZE MODIFIED\n'
+ printf 'gpt-oss:120b aaa 65 GB 1 hour ago\n'
+ printf 'qwen3-coder:30b bbb 18 GB 1 hour ago\n'
+fi
+STUB
+ chmod +x "$STUB_BIN/ollama"
+ PATH="$STUB_BIN:$PATH" run bash "$AI" --print-runtimes
+ [ "$status" -eq 0 ]
+ [ "${lines[0]%% *}" = "claude" ]
+ [[ "$output" == *"codex"* ]]
+ [[ "$output" == *"local:gpt-oss:120b"* ]]
+ [[ "$output" == *"local:qwen3-coder:30b"* ]]
+}
+
+@test "a dead ollama server just drops the local lines" {
+ cat > "$STUB_BIN/ollama" <<'STUB'
+#!/bin/bash
+echo "Error: could not connect to a running Ollama instance" >&2
+exit 1
+STUB
+ chmod +x "$STUB_BIN/ollama"
+ PATH="$STUB_BIN:$PATH" run bash "$AI" --print-runtimes
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"claude"* ]]
+ [[ "$output" != *"local:"* ]]
+}
diff --git a/scripts/tests/ai-wrap-teardown-hook.bats b/scripts/tests/ai-wrap-teardown-hook.bats
new file mode 100644
index 0000000..12ac941
--- /dev/null
+++ b/scripts/tests/ai-wrap-teardown-hook.bats
@@ -0,0 +1,211 @@
+#!/usr/bin/env bats
+# hooks/ai-wrap-teardown.sh — Stop hook that tears down the ai-term session
+# (or powers off) after a wrap-up, gated on a sentinel wrap-it-up drops. On a
+# normal stop (no sentinel) it is a silent no-op. The emacsclient call is
+# stubbed here so the test records the elisp form without a live daemon.
+
+setup() {
+ REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)"
+ SCRIPT="$REPO_ROOT/hooks/ai-wrap-teardown.sh"
+ DISARM="$REPO_ROOT/hooks/session-start-disarm.sh"
+ GATE="$REPO_ROOT/claude-templates/bin/git-worktree-gate"
+ TMPDIR_T="$(mktemp -d)"
+ PROJ="proj-$$-$BATS_TEST_NUMBER" # unique so /tmp sentinels don't collide
+ CWD="$TMPDIR_T/$PROJ"
+ mkdir -p "$CWD"
+ git init -q "$CWD"
+ git -C "$CWD" config user.email test@example.com
+ git -C "$CWD" config user.name tester
+ printf 'base\n' >"$CWD/tracked"
+ git -C "$CWD" add tracked
+ git -C "$CWD" commit -qm init
+ bash "$GATE" certify "$CWD"
+ TEARDOWN_SENTINEL="/tmp/ai-wrap-teardown-${PROJ}"
+ SHUTDOWN_SENTINEL="/tmp/ai-wrap-shutdown-${PROJ}"
+
+ # Stub emacsclient on PATH: record the elisp form it was called with.
+ BIN="$TMPDIR_T/bin"
+ mkdir -p "$BIN"
+ EC_LOG="$TMPDIR_T/emacsclient.log"
+ cat >"$BIN/emacsclient" <<EOF
+#!/usr/bin/env bash
+# args: -e <form>
+shift # drop -e
+printf '%s\n' "\$1" >> "$EC_LOG"
+EOF
+ chmod +x "$BIN/emacsclient"
+}
+
+teardown() {
+ rm -rf "$TMPDIR_T"
+ rm -f "$TEARDOWN_SENTINEL" "$SHUTDOWN_SENTINEL"
+}
+
+run_hook() {
+ # invoke with the stubbed emacsclient on PATH, feeding Stop-hook JSON
+ printf '{"cwd":"%s","hook_event_name":"Stop"}' "$CWD" \
+ | GIT_WORKTREE_GATE="$GATE" PATH="$BIN:$PATH" bash "$SCRIPT"
+}
+
+@test "no sentinel: silent no-op, emacsclient never called" {
+ run run_hook
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+ [ ! -f "$EC_LOG" ]
+}
+
+@test "teardown sentinel: calls cj/ai-term-quit with the project basename" {
+ : > "$TEARDOWN_SENTINEL"
+ run run_hook
+ [ "$status" -eq 0 ]
+ grep -q "cj/ai-term-quit \"$PROJ\"" "$EC_LOG"
+}
+
+@test "teardown sentinel is removed after firing" {
+ : > "$TEARDOWN_SENTINEL"
+ run run_hook
+ [ "$status" -eq 0 ]
+ [ ! -f "$TEARDOWN_SENTINEL" ]
+}
+
+@test "shutdown sentinel: calls cj/ai-term-shutdown-countdown" {
+ : > "$SHUTDOWN_SENTINEL"
+ run run_hook
+ [ "$status" -eq 0 ]
+ grep -q "cj/ai-term-shutdown-countdown" "$EC_LOG"
+}
+
+@test "shutdown supersedes teardown when both sentinels exist" {
+ : > "$TEARDOWN_SENTINEL"
+ : > "$SHUTDOWN_SENTINEL"
+ run run_hook
+ [ "$status" -eq 0 ]
+ grep -q "cj/ai-term-shutdown-countdown" "$EC_LOG"
+ ! grep -q "cj/ai-term-quit" "$EC_LOG"
+ [ ! -f "$TEARDOWN_SENTINEL" ]
+ [ ! -f "$SHUTDOWN_SENTINEL" ]
+}
+
+@test "emacsclient absent: clears the sentinel and exits 0 (graceful)" {
+ : > "$TEARDOWN_SENTINEL"
+ status=0
+ output="$(printf '{"cwd":"%s","hook_event_name":"Stop"}' "$CWD" \
+ | GIT_WORKTREE_GATE="$GATE" PATH="/usr/bin:/bin" bash "$SCRIPT")" || status=$?
+ [ "$status" -eq 0 ]
+ [ ! -f "$TEARDOWN_SENTINEL" ]
+}
+
+@test "falls back to PWD basename when cwd is absent from JSON" {
+ # No cwd key: hook uses $PWD. Run from CWD so basename resolves to PROJ.
+ : > "$TEARDOWN_SENTINEL"
+ run env "GIT_WORKTREE_GATE=$GATE" "PATH=$BIN:$PATH" \
+ bash -c "cd '$CWD' && printf '{}' | bash '$SCRIPT'"
+ [ "$status" -eq 0 ]
+ grep -q "cj/ai-term-quit \"$PROJ\"" "$EC_LOG"
+}
+
+@test "emits no stderr noise on a normal stop" {
+ err="$(printf '{"cwd":"%s","hook_event_name":"Stop"}' "$CWD" \
+ | GIT_WORKTREE_GATE="$GATE" PATH="$BIN:$PATH" bash "$SCRIPT" 2>&1 >/dev/null)"
+ [ -z "$err" ]
+}
+
+@test "dirty worktree blocks teardown, preserves sentinel, and names the path" {
+ printf 'late\n' >>"$CWD/tracked"
+ : >"$TEARDOWN_SENTINEL"
+ run run_hook
+ [ "$status" -eq 0 ]
+ echo "$output" | jq -e '.decision == "block"'
+ echo "$output" | jq -e '.reason | test("tracked")'
+ [ -f "$TEARDOWN_SENTINEL" ]
+ [ ! -f "$EC_LOG" ]
+}
+
+@test "changed HEAD after certification blocks teardown" {
+ git -C "$CWD" commit -q --allow-empty -m later
+ : >"$TEARDOWN_SENTINEL"
+ run run_hook
+ echo "$output" | jq -e '.reason | test("HEAD changed")'
+ [ -f "$TEARDOWN_SENTINEL" ]
+ [ ! -f "$EC_LOG" ]
+}
+
+@test "missing certificate blocks teardown" {
+ rm -f "$(git -C "$CWD" rev-parse --absolute-git-dir)/ai-wrap-clean"
+ : >"$TEARDOWN_SENTINEL"
+ run run_hook
+ echo "$output" | jq -e '.reason | test("no clean-tree certificate")'
+ [ -f "$TEARDOWN_SENTINEL" ]
+}
+
+@test "Codex payload receives continue false and a stop reason" {
+ printf 'late\n' >>"$CWD/tracked"
+ : >"$TEARDOWN_SENTINEL"
+ run bash -c \
+ "printf '{\"cwd\":\"$CWD\",\"hook_event_name\":\"Stop\",\"model\":\"gpt-test\"}' \
+ | GIT_WORKTREE_GATE='$GATE' PATH='$BIN:$PATH' bash '$SCRIPT'"
+ [ "$status" -eq 0 ]
+ echo "$output" | jq -e '.continue == false'
+ echo "$output" | jq -e '.stopReason | test("tracked")'
+ [ -f "$TEARDOWN_SENTINEL" ]
+}
+
+@test "successful teardown consumes the clean-tree certificate" {
+ : >"$TEARDOWN_SENTINEL"
+ cert="$(git -C "$CWD" rev-parse --absolute-git-dir)/ai-wrap-clean"
+ [ -f "$cert" ]
+ run run_hook
+ [ "$status" -eq 0 ]
+ [ ! -f "$cert" ]
+}
+
+# --- Cross-session leakage (the 2026-07-27 work-session kill) ---------------
+#
+# A sentinel is deliberately preserved when certification fails, so a blocked
+# wrap can retry within the same session once the tree is clean. But nothing
+# bounded that to the session: the sentinel outlived it and detonated in
+# whatever session next happened to have a clean tree. work's 11:37 wrap left
+# one armed; the 13:20 session committed during startup, went clean, and was
+# torn down mid-work. archsetup's had been armed for two days on a live
+# terminal.
+#
+# A new session means the wrap that armed the sentinel is gone, so its pending
+# teardown is moot. session-start-disarm.sh clears it.
+
+@test "session-start disarm: removes a teardown sentinel left by a prior session" {
+ : > "$TEARDOWN_SENTINEL"
+ printf '{"cwd":"%s","hook_event_name":"SessionStart"}' "$CWD" \
+ | bash "$DISARM"
+ [ ! -f "$TEARDOWN_SENTINEL" ]
+}
+
+@test "session-start disarm: removes a shutdown sentinel too" {
+ : > "$SHUTDOWN_SENTINEL"
+ printf '{"cwd":"%s","hook_event_name":"SessionStart"}' "$CWD" \
+ | bash "$DISARM"
+ [ ! -f "$SHUTDOWN_SENTINEL" ]
+}
+
+@test "session-start disarm: silent and exit 0 when nothing is armed" {
+ run bash -c "printf '{\"cwd\":\"$CWD\",\"hook_event_name\":\"SessionStart\"}' | bash '$DISARM'"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "session-start disarm: only clears this project, not another's" {
+ other="/tmp/ai-wrap-teardown-someotherproj"
+ : > "$TEARDOWN_SENTINEL"
+ : > "$other"
+ printf '{"cwd":"%s","hook_event_name":"SessionStart"}' "$CWD" | bash "$DISARM"
+ [ ! -f "$TEARDOWN_SENTINEL" ]
+ [ -f "$other" ]
+ rm -f "$other"
+}
+
+@test "within-session retry still works: sentinel survives a blocked stop" {
+ : > "$TEARDOWN_SENTINEL"
+ echo dirt > "$CWD/untracked.txt"
+ run run_hook
+ [ -f "$TEARDOWN_SENTINEL" ]
+ rm -f "$CWD/untracked.txt"
+}
diff --git a/scripts/tests/audit.bats b/scripts/tests/audit.bats
index 3df69c9..c6123e1 100644
--- a/scripts/tests/audit.bats
+++ b/scripts/tests/audit.bats
@@ -40,9 +40,28 @@ git_init_with_ai_tracked() {
local proj_dir="$1"
(cd "$proj_dir" \
&& git init -q \
+ && git config maintenance.auto false \
+ && git config gc.auto 0 \
&& git add -A \
&& git -c user.email=test@test -c user.name=test commit -q -m initial)
}
+# maintenance.auto/gc.auto are off because `git commit` otherwise spawns
+# `git maintenance run --auto --quiet --detach` (traced on git 2.55). The commit
+# returns while that detached process is still writing .git/objects/pack, so
+# teardown's `rm -rf` intermittently failed with "Directory not empty" and bats
+# reported a passing test as failed. Measured at ~8% of runs. Killing the
+# background writer is the fix; a retry loop in teardown would only hide it.
+#
+# What triggers it: the fixture rsyncs ~170 workflow/script files before
+# committing, and maintenance's loose-objects task fires at a default threshold
+# of 100. (Not gc.auto's 6700 — that never fires here, which is why disabling
+# gc.auto alone was never the explanation.) maintenance.auto false is what
+# suppresses it on 2.55; gc.auto 0 is the portable guard for older git, where
+# commit calls gc --auto directly and maintenance.auto does not exist. Either
+# is sufficient on its own here.
+#
+# Other bats suites that git-init fixtures are currently safe only by staying
+# under that 100-object threshold — an implicit dependency, not a guarantee.
@test "audit: clean projects report ok with exit 0" {
scaffold_synced_ai "$TEST_HOME/code/alpha"
@@ -113,6 +132,23 @@ git_init_with_ai_tracked() {
! grep -q "# uncommitted" "$TEST_HOME/code/alpha/.ai/protocols.org"
}
+@test "audit: retired projects are excluded entirely" {
+ # ~/projects/.retired/<name> holds shelved projects. The audit must
+ # never sync into them — they're not live targets. A drifted retired
+ # project must not appear in the output or the counts.
+ scaffold_synced_ai "$TEST_HOME/projects/.retired/oldproj"
+ echo "# drift marker" >> "$TEST_HOME/projects/.retired/oldproj/.ai/protocols.org"
+ scaffold_synced_ai "$TEST_HOME/code/alpha"
+
+ run bash "$AUDIT" --no-doctor
+
+ # The retired project is invisible to the audit, in any state.
+ [[ "$output" != *".retired/oldproj"* ]]
+ # The live project is still audited and clean.
+ [[ "$output" == *"ok ~/code/alpha"* ]]
+ [[ "$output" == *"Summary: 1 ok, 0 drift, 0 skipped, 0 failed"* ]]
+}
+
@test "audit: loop continues past .ai/-missing failure" {
# Edge case 2 from todo.org:1766. The defensive [ ! -d "$proj/.ai" ]
# branch fires when a discovered .ai/ disappears between find and
diff --git a/scripts/tests/before-close-queue.bats b/scripts/tests/before-close-queue.bats
new file mode 100644
index 0000000..9c968f7
--- /dev/null
+++ b/scripts/tests/before-close-queue.bats
@@ -0,0 +1,44 @@
+#!/usr/bin/env bats
+# The "Colloquialisms and Expansions" convention and its "the list" before-close
+# queue must stay wired into the synced template: the protocols.org reference
+# section, and the wrap-it-up.org Step 1 sub-step that drains the queue before
+# the Summary. Guards against either being dropped in a future edit or sync.
+
+setup() {
+ REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)"
+ PROTO="$REPO_ROOT/claude-templates/.ai/protocols.org"
+ WRAP="$REPO_ROOT/claude-templates/.ai/workflows/wrap-it-up.org"
+ PROTO_MIRROR="$REPO_ROOT/.ai/protocols.org"
+ WRAP_MIRROR="$REPO_ROOT/.ai/workflows/wrap-it-up.org"
+}
+
+@test "protocols.org documents the Colloquialisms and Expansions convention" {
+ grep -qF '* Colloquialisms and Expansions' "$PROTO"
+ grep -q 'the list' "$PROTO"
+ grep -q 'Before-Close Queue' "$PROTO"
+ grep -qF 'tell <project>' "$PROTO"
+ grep -q 'inbox-send' "$PROTO"
+}
+
+@test "the colloquialisms reference scopes the queue to the session anchor" {
+ grep -q 'session-context.org' "$PROTO"
+ # A must-outlive item is a todo.org task, not a list item.
+ grep -q 'must outlive the session' "$PROTO"
+}
+
+@test "wrap-it-up Step 1 works the Before-Close Queue before the Summary" {
+ grep -qF 'Work the Before-Close Queue' "$WRAP"
+ grep -q 'oldest-first' "$WRAP"
+ grep -q 'silent no-op' "$WRAP"
+ # The queue must be worked before the Summary is written, so its edits ride
+ # this wrap's commit. Assert it sits ahead of the first Summary sub-step.
+ local queue_line kb_line
+ queue_line=$(grep -n 'Work the Before-Close Queue' "$WRAP" | head -1 | cut -d: -f1)
+ kb_line=$(grep -n 'Early KB reflection' "$WRAP" | head -1 | cut -d: -f1)
+ [ "$queue_line" -lt "$kb_line" ]
+}
+
+@test "the convention is mirrored to the committed .ai copy" {
+ diff -q "$PROTO" "$PROTO_MIRROR"
+ diff -q "$WRAP" "$WRAP_MIRROR"
+}
diff --git a/scripts/tests/git-worktree-gate.bats b/scripts/tests/git-worktree-gate.bats
new file mode 100755
index 0000000..7d0d8a1
--- /dev/null
+++ b/scripts/tests/git-worktree-gate.bats
@@ -0,0 +1,146 @@
+#!/usr/bin/env bats
+
+setup() {
+ REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)"
+ GATE="$REPO_ROOT/claude-templates/bin/git-worktree-gate"
+ WORK="$(mktemp -d)"
+ REPO="$WORK/repo"
+ git init -q "$REPO"
+ git -C "$REPO" config user.email test@example.com
+ git -C "$REPO" config user.name tester
+ printf 'base\n' >"$REPO/tracked"
+ git -C "$REPO" add tracked
+ git -C "$REPO" commit -qm init
+}
+
+teardown() {
+ rm -rf "$WORK"
+}
+
+@test "strict accepts a completely clean worktree" {
+ run bash "$GATE" strict "$REPO"
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "strict rejects unstaged, staged, and untracked changes with paths" {
+ printf 'changed\n' >>"$REPO/tracked"
+ printf 'new\n' >"$REPO/staged"
+ git -C "$REPO" add staged
+ printf 'loose\n' >"$REPO/loose"
+
+ run bash "$GATE" strict "$REPO"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"wrap blocked"* ]]
+ [[ "$output" == *"tracked"* ]]
+ [[ "$output" == *"staged"* ]]
+ [[ "$output" == *"loose"* ]]
+}
+
+@test "sync-safe permits untracked inbox deliveries and strict still rejects them" {
+ mkdir -p "$REPO/inbox/nested"
+ printf 'handoff\n' >"$REPO/inbox/nested/from-home.org"
+
+ run bash "$GATE" sync-safe "$REPO"
+ [ "$status" -eq 0 ]
+
+ run bash "$GATE" strict "$REPO"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"inbox/nested/from-home.org"* ]]
+}
+
+@test "sync-safe rejects tracked changes even inside inbox" {
+ mkdir -p "$REPO/inbox"
+ printf 'tracked\n' >"$REPO/inbox/tracked.org"
+ git -C "$REPO" add inbox/tracked.org
+ git -C "$REPO" commit -qm inbox
+ printf 'changed\n' >>"$REPO/inbox/tracked.org"
+
+ run bash "$GATE" sync-safe "$REPO"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"inbox/tracked.org"* ]]
+}
+
+@test "sync-safe rejects untracked files outside inbox" {
+ printf 'scratch\n' >"$REPO/scratch"
+ run bash "$GATE" sync-safe "$REPO"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"scratch"* ]]
+}
+
+@test "ignored files do not block either policy" {
+ printf 'cache/\n' >"$REPO/.gitignore"
+ git -C "$REPO" add .gitignore
+ git -C "$REPO" commit -qm ignore
+ mkdir -p "$REPO/cache"
+ printf 'generated\n' >"$REPO/cache/output"
+
+ run bash "$GATE" strict "$REPO"
+ [ "$status" -eq 0 ]
+ run bash "$GATE" sync-safe "$REPO"
+ [ "$status" -eq 0 ]
+}
+
+@test "reports unusual filenames without losing the entry" {
+ odd=$'line break\nname'
+ printf 'odd\n' >"$REPO/$odd"
+ run bash "$GATE" strict "$REPO"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"line"* ]]
+ [[ "$output" == *"name"* ]]
+}
+
+@test "certify and verify bind a clean worktree to its current HEAD" {
+ run bash "$GATE" certify "$REPO"
+ [ "$status" -eq 0 ]
+ run bash "$GATE" verify "$REPO"
+ [ "$status" -eq 0 ]
+
+ git -C "$REPO" commit -q --allow-empty -m later
+ run bash "$GATE" verify "$REPO"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"HEAD changed"* ]]
+}
+
+@test "verify rejects changes made after certification" {
+ run bash "$GATE" certify "$REPO"
+ [ "$status" -eq 0 ]
+ printf 'late\n' >>"$REPO/tracked"
+ run bash "$GATE" verify "$REPO"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"tracked"* ]]
+}
+
+@test "dirty submodule blocks strict and sync-safe" {
+ CHILD="$WORK/child"
+ git init -q "$CHILD"
+ git -C "$CHILD" config user.email test@example.com
+ git -C "$CHILD" config user.name tester
+ printf 'child\n' >"$CHILD/file"
+ git -C "$CHILD" add file
+ git -C "$CHILD" commit -qm init
+ git -C "$REPO" -c protocol.file.allow=always submodule add -q "$CHILD" sub
+ git -C "$REPO" commit -qam submodule
+ printf 'dirty\n' >>"$REPO/sub/file"
+
+ run bash "$GATE" strict "$REPO"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"sub"* ]]
+ run bash "$GATE" sync-safe "$REPO"
+ [ "$status" -eq 1 ]
+}
+
+@test "a low-level git status failure blocks instead of looking clean" {
+ printf 'not an index\n' >"$WORK/bad-index"
+ run env GIT_INDEX_FILE="$WORK/bad-index" bash "$GATE" strict "$REPO"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"git status failed"* ]]
+}
+
+@test "an in-progress sequencer operation blocks a clean-looking tree" {
+ gitdir="$(git -C "$REPO" rev-parse --absolute-git-dir)"
+ mkdir -p "$gitdir/sequencer"
+ run bash "$GATE" strict "$REPO"
+ [ "$status" -eq 1 ]
+ [[ "$output" == *"sequencer"* ]]
+}
diff --git a/scripts/tests/inbox-boundary-check-hook.bats b/scripts/tests/inbox-boundary-check-hook.bats
new file mode 100644
index 0000000..58668a5
--- /dev/null
+++ b/scripts/tests/inbox-boundary-check-hook.bats
@@ -0,0 +1,83 @@
+#!/usr/bin/env bats
+# hooks/inbox-boundary-check.sh — Stop hook that soft-nudges the agent to
+# process pending inbox/ handoffs before yielding. Blocks the stop ONCE (emits
+# a block decision + reason) when inbox-status reports pending items; on the
+# harness re-entry (stop_hook_active: true) it steps aside so an unprocessable
+# item or a mid-task pause never wedges. Self-skips where there's no inbox/ or
+# no inbox-status. The real inbox-status is copied into the test project so the
+# hook runs against real code, not a stub.
+
+setup() {
+ REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)"
+ SCRIPT="$REPO_ROOT/hooks/inbox-boundary-check.sh"
+ INBOX_STATUS="$REPO_ROOT/claude-templates/.ai/scripts/inbox-status"
+ TMPDIR_T="$(mktemp -d)"
+ CWD="$TMPDIR_T/proj"
+ mkdir -p "$CWD/.ai/scripts"
+ cp "$INBOX_STATUS" "$CWD/.ai/scripts/inbox-status"
+ chmod +x "$CWD/.ai/scripts/inbox-status"
+}
+
+teardown() {
+ rm -rf "$TMPDIR_T"
+}
+
+# Feed Stop-hook JSON on stdin. $1 = stop_hook_active (default false).
+run_hook() {
+ local active="${1:-false}"
+ printf '{"cwd":"%s","hook_event_name":"Stop","stop_hook_active":%s}' \
+ "$CWD" "$active" | bash "$SCRIPT"
+}
+
+@test "pending handoffs block the stop with a count in the reason" {
+ mkdir -p "$CWD/inbox"
+ printf 'x\n' >"$CWD/inbox/2026-07-19-from-home-thing.org"
+ printf 'y\n' >"$CWD/inbox/2026-07-19-from-work-other.org"
+ run run_hook
+ [ "$status" -eq 0 ]
+ # Valid JSON with a block decision.
+ echo "$output" | jq -e '.decision == "block"'
+ echo "$output" | jq -e '.reason | test("2 pending")'
+ echo "$output" | jq -e '.reason | test("inbox.org")'
+}
+
+@test "a clean inbox (only artifacts) emits nothing" {
+ mkdir -p "$CWD/inbox"
+ touch "$CWD/inbox/.gitkeep"
+ printf 'done\n' >"$CWD/inbox/PROCESSED-2026-07-19-old.org"
+ run run_hook
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "soft-nudge: stop_hook_active=true steps aside even with pending items" {
+ mkdir -p "$CWD/inbox"
+ printf 'x\n' >"$CWD/inbox/2026-07-19-from-home-thing.org"
+ run run_hook true
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "no inbox/ directory is a silent no-op" {
+ run run_hook
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "inbox-status absent degrades to a silent no-op" {
+ mkdir -p "$CWD/inbox"
+ printf 'x\n' >"$CWD/inbox/2026-07-19-from-home-thing.org"
+ rm -f "$CWD/.ai/scripts/inbox-status"
+ # Also ensure nothing named inbox-status is on PATH for this run.
+ run env PATH="/usr/bin:/bin" bash -c '
+ printf "{\"cwd\":\"'"$CWD"'\",\"hook_event_name\":\"Stop\"}" | bash "'"$SCRIPT"'"'
+ [ "$status" -eq 0 ]
+ [ -z "$output" ]
+}
+
+@test "the emitted reason names the project for context" {
+ mkdir -p "$CWD/inbox"
+ printf 'x\n' >"$CWD/inbox/2026-07-19-from-home-thing.org"
+ run run_hook
+ echo "$output" | jq -e '.reason | test("proj")'
+}
diff --git a/scripts/tests/install-agents-entry.bats b/scripts/tests/install-agents-entry.bats
new file mode 100644
index 0000000..03343a6
--- /dev/null
+++ b/scripts/tests/install-agents-entry.bats
@@ -0,0 +1,51 @@
+#!/usr/bin/env bats
+# make install must link the runtime-neutral agent entry file (AGENTS.md)
+# into CODEX_DIR so Codex-style harnesses bootstrap from the same
+# protocols/rules/skills the Claude side reads. The thin-pointer shape and
+# the decision trail live in docs/design/2026-07-13-runtime-portability-
+# inventories.org and the generic-agent-runtime task.
+
+setup() {
+ REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)"
+ TMPHOME="$(mktemp -d)"
+}
+
+teardown() {
+ rm -rf "$TMPHOME"
+}
+
+run_install() {
+ make -C "$REPO_ROOT" install \
+ SKILLS_DIR="$TMPHOME/skills" \
+ RULES_DIR="$TMPHOME/rules" \
+ HOOKS_DIR="$TMPHOME/hooks" \
+ CLAUDE_DIR="$TMPHOME/claude" \
+ CODEX_DIR="$TMPHOME/codex" \
+ LOCAL_BIN="$TMPHOME/bin"
+}
+
+@test "install links AGENTS.md into CODEX_DIR" {
+ run run_install
+ [ "$status" -eq 0 ]
+ [ -L "$TMPHOME/codex/AGENTS.md" ]
+ grep -q "protocols.org" "$TMPHOME/codex/AGENTS.md"
+}
+
+@test "install is idempotent on the agent entry (second run skips)" {
+ run run_install
+ [ "$status" -eq 0 ]
+ run run_install
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"skip AGENTS.md (already linked)"* ]]
+ [ -L "$TMPHOME/codex/AGENTS.md" ]
+}
+
+@test "install warns on a non-symlink AGENTS.md collision and leaves it alone" {
+ mkdir -p "$TMPHOME/codex"
+ echo "hand-written entry" > "$TMPHOME/codex/AGENTS.md"
+ run run_install
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"WARN AGENTS.md exists and is not a symlink"* ]]
+ [ ! -L "$TMPHOME/codex/AGENTS.md" ]
+ grep -q "hand-written entry" "$TMPHOME/codex/AGENTS.md"
+}
diff --git a/scripts/tests/install-ai.bats b/scripts/tests/install-ai.bats
index 8e91770..1549184 100644
--- a/scripts/tests/install-ai.bats
+++ b/scripts/tests/install-ai.bats
@@ -149,3 +149,90 @@ EOF
[ -d "$TEST_HOME/code/pickme/.ai" ]
[ ! -d "$TEST_HOME/code/skipme/.ai" ]
}
+
+@test "install-ai: seeds AGENTS.md at the project root" {
+ mkdir -p "$TEST_HOME/code/fresh"
+ (cd "$TEST_HOME/code/fresh" && git init -q)
+
+ run bash "$INSTALL_AI" --gitignore "$TEST_HOME/code/fresh"
+
+ [ "$status" -eq 0 ]
+ [ -f "$TEST_HOME/code/fresh/AGENTS.md" ]
+ grep -q "protocols.org" "$TEST_HOME/code/fresh/AGENTS.md"
+}
+
+@test "install-ai: never overwrites an existing AGENTS.md" {
+ mkdir -p "$TEST_HOME/code/fresh"
+ (cd "$TEST_HOME/code/fresh" && git init -q)
+ echo "project-owned entry file" > "$TEST_HOME/code/fresh/AGENTS.md"
+
+ run bash "$INSTALL_AI" --gitignore "$TEST_HOME/code/fresh"
+
+ [ "$status" -eq 0 ]
+ grep -q "project-owned entry file" "$TEST_HOME/code/fresh/AGENTS.md"
+ ! grep -q "protocols.org" "$TEST_HOME/code/fresh/AGENTS.md"
+}
+
+# --- claude-templates/bin/install-ai launcher --------------------------------
+#
+# The launcher is the PATH-facing front door (make install symlinks it into
+# ~/.local/bin/install-ai, same loop as `ai` and `agent-text`). It resolves its
+# own real path through the symlink and execs scripts/install-ai.sh, so the
+# repo-root computation survives being invoked as a symlink from anywhere.
+
+LAUNCHER="$REAL_REPO/claude-templates/bin/install-ai"
+
+@test "install-ai launcher: exists and is executable" {
+ [ -x "$LAUNCHER" ]
+}
+
+@test "install-ai launcher: --help reaches install-ai.sh through the launcher" {
+ run bash "$LAUNCHER" --help
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"Bootstrap .ai/"* ]]
+}
+
+@test "install-ai launcher: works when invoked as a symlink from another dir" {
+ # Reproduces the ~/.local/bin symlink: a link elsewhere must still resolve
+ # the repo root, not break on dirname of the link path.
+ ln -s "$LAUNCHER" "$TEST_HOME/install-ai"
+ run bash "$TEST_HOME/install-ai" --help
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"Bootstrap .ai/"* ]]
+}
+
+# --- temp/ ephemeral-artifacts ignore (working/ tracked-from-creation ruling) ---
+
+@test "install-ai --gitignore: ignores temp/ but never working/" {
+ mkdir -p "$TEST_HOME/code/fresh"
+ (cd "$TEST_HOME/code/fresh" && git init -q)
+
+ run bash "$INSTALL_AI" --gitignore "$TEST_HOME/code/fresh"
+
+ [ "$status" -eq 0 ]
+ grep -qFx "temp/" "$TEST_HOME/code/fresh/.gitignore"
+ # working/ is the tracked home of in-progress work — it must never be ignored.
+ ! grep -qEx "/?working/?" "$TEST_HOME/code/fresh/.gitignore"
+}
+
+@test "install-ai --track: ignores temp/ even in track mode" {
+ mkdir -p "$TEST_HOME/code/tracked"
+ (cd "$TEST_HOME/code/tracked" && git init -q)
+
+ run bash "$INSTALL_AI" --track "$TEST_HOME/code/tracked"
+
+ [ "$status" -eq 0 ]
+ # temp/ is ephemeral regardless of whether the project tracks its .ai/ tooling.
+ grep -qFx "temp/" "$TEST_HOME/code/tracked/.gitignore"
+}
+
+@test "install-ai: temp/ ignore is idempotent (not re-added)" {
+ mkdir -p "$TEST_HOME/code/fresh"
+ (cd "$TEST_HOME/code/fresh" && git init -q)
+ printf 'temp/\n' > "$TEST_HOME/code/fresh/.gitignore"
+
+ run bash "$INSTALL_AI" --gitignore "$TEST_HOME/code/fresh"
+
+ [ "$status" -eq 0 ]
+ [ "$(grep -cFx 'temp/' "$TEST_HOME/code/fresh/.gitignore")" -eq 1 ]
+}
diff --git a/scripts/tests/install-hooks-link.bats b/scripts/tests/install-hooks-link.bats
index 80ac8dd..5368781 100644
--- a/scripts/tests/install-hooks-link.bats
+++ b/scripts/tests/install-hooks-link.bats
@@ -20,6 +20,7 @@ run_install() {
RULES_DIR="$TMPHOME/rules" \
HOOKS_DIR="$TMPHOME/hooks" \
CLAUDE_DIR="$TMPHOME/claude" \
+ CODEX_DIR="$TMPHOME/codex" \
LOCAL_BIN="$TMPHOME/bin"
}
@@ -28,6 +29,20 @@ run_install() {
[ "$status" -eq 0 ]
[ -L "$TMPHOME/hooks/session-clear-resume.sh" ]
[ -L "$TMPHOME/hooks/precompact-priorities.sh" ]
+ [ -L "$TMPHOME/hooks/rulesets-write-boundary.py" ]
+}
+
+@test "install links Codex Stop-hook configuration" {
+ run run_install
+ [ "$status" -eq 0 ]
+ [ -L "$TMPHOME/codex/hooks.json" ]
+ grep -q "ai-wrap-teardown.sh" "$TMPHOME/codex/hooks.json"
+}
+
+@test "install links the shared Git worktree gate beside the launcher" {
+ run run_install
+ [ "$status" -eq 0 ]
+ [ -L "$TMPHOME/bin/git-worktree-gate" ]
}
@test "install does not link opt-in hooks" {
diff --git a/scripts/tests/install-lang-collision.bats b/scripts/tests/install-lang-collision.bats
new file mode 100644
index 0000000..e03e136
--- /dev/null
+++ b/scripts/tests/install-lang-collision.bats
@@ -0,0 +1,143 @@
+#!/usr/bin/env bats
+# Tests for install-lang's cross-bundle collision guard.
+#
+# Several bundles ship files at the same path. gitignore-add.txt merges
+# (appended, deduped) and CLAUDE.md is seed-only, so both compose across
+# bundles. Three do not:
+#
+# claude/settings.json all 5 bundles — cp -rT, silently overwritten
+# githooks/pre-commit all 5 bundles — cp -rT, silently overwritten
+# coverage-makefile.txt 4 bundles — [skip]ped, fragment dropped
+#
+# The first two read "elisp, bash, go" until 2026-07-23, when python and
+# typescript gained the components they had been missing. The consequence is
+# that no two shipping bundles compose any more; see the polyglot task.
+#
+# Installing a second bundle used to replace the first's settings.json and
+# pre-commit while printing [ok], so a project could lose its paren check or
+# secret scan and read the output as success. The guard refuses instead, naming
+# what would be replaced. FORCE=1 still overrides.
+
+INSTALL="${BATS_TEST_DIRNAME}/../install-lang.sh"
+
+setup() {
+ PROJ="$(mktemp -d)"
+ git init -q "$PROJ"
+}
+
+teardown() {
+ [ -n "${PROJ:-}" ] && rm -rf "$PROJ"
+}
+
+# ---- Normal: single-bundle installs are unaffected ----
+
+@test "install-lang: a fresh single-bundle install succeeds" {
+ run bash "$INSTALL" elisp "$PROJ"
+ [ "$status" -eq 0 ]
+ [ -f "$PROJ/.claude/settings.json" ]
+ grep -q 'validate-el.sh' "$PROJ/.claude/settings.json"
+}
+
+@test "install-lang: reinstalling the SAME bundle is idempotent, not a collision" {
+ bash "$INSTALL" elisp "$PROJ"
+ run bash "$INSTALL" elisp "$PROJ"
+ [ "$status" -eq 0 ]
+ [[ "$output" != *"collision"* ]]
+ grep -q 'validate-el.sh' "$PROJ/.claude/settings.json"
+}
+
+# ---- The guard: a second, different bundle must not silently replace ----
+
+@test "install-lang: a second bundle sharing settings.json and githooks is refused" {
+ bash "$INSTALL" elisp "$PROJ"
+ run bash "$INSTALL" bash "$PROJ"
+ [ "$status" -ne 0 ] || { echo "second bundle installed without refusal"; return 1; }
+ [[ "$output" == *"elisp"* ]] || { echo "refusal does not name the existing bundle"; return 1; }
+}
+
+@test "install-lang: the refusal names each file that would be replaced" {
+ bash "$INSTALL" elisp "$PROJ"
+ run bash "$INSTALL" bash "$PROJ"
+ [[ "$output" == *"settings.json"* ]] || { echo "refusal omits settings.json"; return 1; }
+ [[ "$output" == *"pre-commit"* ]] || { echo "refusal omits githooks/pre-commit"; return 1; }
+}
+
+@test "install-lang: a refused install leaves the first bundle intact" {
+ bash "$INSTALL" elisp "$PROJ"
+ bash "$INSTALL" bash "$PROJ" || true
+ grep -q 'validate-el.sh' "$PROJ/.claude/settings.json" \
+ || { echo "elisp settings.json was replaced despite refusal"; return 1; }
+ grep -q 'check-parens' "$PROJ/githooks/pre-commit" \
+ || { echo "elisp pre-commit was replaced despite refusal"; return 1; }
+}
+
+@test "install-lang: bundles colliding only on coverage-makefile.txt are refused too" {
+ # Named for the pre-2026-07-23 reason: python and typescript shipped no
+ # settings.json or githooks, so the coverage fragment was their only overlap
+ # and the second one's used to be silently dropped. They now overlap on all
+ # three, so this exercises the multi-file refusal path — the coverage
+ # fragment must still be named among them.
+ bash "$INSTALL" python "$PROJ"
+ run bash "$INSTALL" typescript "$PROJ"
+ [ "$status" -ne 0 ] || { echo "typescript installed over python's coverage fragment"; return 1; }
+ [[ "$output" == *"coverage-makefile.txt"* ]]
+}
+
+@test "install-lang: two bundles that share no overwritten file install together" {
+ # The guard must stay out of the way when nothing overlaps. This is the path
+ # where a false refusal would be easiest to introduce: the bundle IS
+ # detected, and only the empty file-list stops it.
+ #
+ # This used to be tested with bash + python, which composed because python
+ # shipped no settings.json and no githooks. That was the incomplete-bundle
+ # bug (fixed 2026-07-23), not a design property — so no pair of *shipping*
+ # bundles is non-colliding any more, and the case needs a synthetic bundle.
+ # See the polyglot task in todo.org: whether every pair now colliding is
+ # acceptable is an open question, but the guard's own no-false-refusal
+ # behavior is not, and stays pinned here.
+ fake="${BATS_TEST_DIRNAME}/../../languages/zz-test-rulesonly"
+ mkdir -p "$fake/claude/rules"
+ printf '# rule\n' > "$fake/claude/rules/zz-testing.md"
+
+ bash "$INSTALL" bash "$PROJ"
+ run bash "$INSTALL" zz-test-rulesonly "$PROJ"
+ rm -rf "$fake"
+
+ [ "$status" -eq 0 ] || { echo "guard falsely refused a non-colliding pair: $output"; return 1; }
+ # bash's config survives untouched.
+ grep -q 'validate-bash.sh' "$PROJ/.claude/settings.json"
+ [ -f "$PROJ/.claude/rules/bash.md" ] && [ -f "$PROJ/.claude/rules/zz-testing.md" ]
+}
+
+@test "install-lang: completing python made it collide with bash (documents the tradeoff)" {
+ # Pins the consequence of the 2026-07-23 bundle completion so it can't drift
+ # back unnoticed: python now ships settings.json + githooks/pre-commit, so
+ # bash + python is a genuine overwrite conflict and the guard refuses it.
+ # Whether that's the right trade is Craig's open call; that it IS the current
+ # behavior is what this test records.
+ bash "$INSTALL" bash "$PROJ"
+ run bash "$INSTALL" python "$PROJ"
+ [ "$status" -ne 0 ]
+ [[ "$output" == *"collision"* ]]
+}
+
+# ---- The escape hatch ----
+
+@test "install-lang: FORCE=1 overrides the collision guard" {
+ bash "$INSTALL" elisp "$PROJ"
+ run bash "$INSTALL" bash "$PROJ" 1
+ [ "$status" -eq 0 ] || { echo "FORCE=1 did not override: $output"; return 1; }
+ grep -q 'validate-bash.sh' "$PROJ/.claude/settings.json"
+}
+
+@test "install-lang: the refusal points at FORCE=1 and warns it re-seeds CLAUDE.md" {
+ bash "$INSTALL" elisp "$PROJ"
+ run bash "$INSTALL" bash "$PROJ"
+ # Assert the refusal fired first: the pre-existing "[skip] CLAUDE.md already
+ # exists (use FORCE=1 to overwrite)" line mentions both strings on its own, so
+ # without this the test passes against an unguarded install.
+ [ "$status" -ne 0 ] || { echo "no refusal fired"; return 1; }
+ refusal="$(printf '%s\n' "$output" | grep -v '^ \[skip\]')"
+ [[ "$refusal" == *"FORCE=1"* ]] || { echo "refusal does not name the override"; return 1; }
+ [[ "$refusal" == *"CLAUDE.md"* ]] || { echo "refusal does not warn about the CLAUDE.md re-seed"; return 1; }
+}
diff --git a/scripts/tests/install-lang-completeness.bats b/scripts/tests/install-lang-completeness.bats
new file mode 100644
index 0000000..42832dc
--- /dev/null
+++ b/scripts/tests/install-lang-completeness.bats
@@ -0,0 +1,93 @@
+#!/usr/bin/env bats
+#
+# Tests for install-lang.sh's bundle-completeness warning.
+#
+# Background: the python and typescript bundles shipped for nearly two months
+# with no githooks/pre-commit, so any project installing them got no
+# credential scan on commit. install-lang guarded each component copy with a
+# plain `[ -d ... ]`, so a missing component was indistinguishable from a
+# complete install — it printed nothing and exited 0.
+#
+# The fix is not "never miss a component" (a person will), it's "say so when
+# you do". These tests pin that: a complete bundle installs quietly, an
+# incomplete one names exactly what's absent, and neither case fails the
+# install — a warning must not become a new way to block work.
+
+REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)"
+INSTALL="$REPO_ROOT/scripts/install-lang.sh"
+
+setup() {
+ TEST_DIR="$(mktemp -d -t install-lang-bats.XXXXXX)"
+ PROJECT="$TEST_DIR/proj"
+ mkdir -p "$PROJECT"
+ git init -q "$PROJECT"
+}
+
+teardown() {
+ rm -rf "$TEST_DIR"
+}
+
+# ---- Normal: a complete bundle is quiet ------------------------------
+
+@test "install-lang: a complete bundle warns about nothing" {
+ run env LANG_ARG=bash bash "$INSTALL" bash "$PROJECT"
+ [ "$status" -eq 0 ]
+ [[ "$output" != *"incomplete"* ]]
+}
+
+@test "install-lang: python is now a complete bundle" {
+ run bash "$INSTALL" python "$PROJECT"
+ [ "$status" -eq 0 ]
+ [[ "$output" != *"incomplete"* ]]
+ [ -f "$PROJECT/githooks/pre-commit" ]
+ [ -f "$PROJECT/.claude/settings.json" ]
+ [ -f "$PROJECT/.claude/hooks/validate-python.sh" ]
+}
+
+@test "install-lang: typescript is now a complete bundle" {
+ run bash "$INSTALL" typescript "$PROJECT"
+ [ "$status" -eq 0 ]
+ [[ "$output" != *"incomplete"* ]]
+ [ -f "$PROJECT/githooks/pre-commit" ]
+ [ -f "$PROJECT/.claude/settings.json" ]
+ [ -f "$PROJECT/.claude/hooks/validate-typescript.sh" ]
+}
+
+@test "install-lang: the installed pre-commit is executable" {
+ run bash "$INSTALL" python "$PROJECT"
+ [ "$status" -eq 0 ]
+ [ -x "$PROJECT/githooks/pre-commit" ]
+}
+
+# ---- Error: an incomplete bundle announces itself --------------------
+
+@test "install-lang: a bundle missing githooks/ warns and names it" {
+ fake="$REPO_ROOT/languages/zz-test-partial"
+ mkdir -p "$fake/claude/rules"
+ printf '# rule\n' > "$fake/claude/rules/zz.md"
+ run bash "$INSTALL" zz-test-partial "$PROJECT"
+ rm -rf "$fake"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"incomplete"* ]]
+ [[ "$output" == *"githooks/pre-commit"* ]]
+}
+
+@test "install-lang: the warning names every missing component, not just the first" {
+ fake="$REPO_ROOT/languages/zz-test-partial"
+ mkdir -p "$fake/claude/rules"
+ printf '# rule\n' > "$fake/claude/rules/zz.md"
+ run bash "$INSTALL" zz-test-partial "$PROJECT"
+ rm -rf "$fake"
+ [[ "$output" == *"githooks/pre-commit"* ]]
+ [[ "$output" == *"settings.json"* ]]
+}
+
+@test "install-lang: an incomplete bundle still installs (warn, never block)" {
+ fake="$REPO_ROOT/languages/zz-test-partial"
+ mkdir -p "$fake/claude/rules"
+ printf '# rule\n' > "$fake/claude/rules/zz.md"
+ run bash "$INSTALL" zz-test-partial "$PROJECT"
+ rm -rf "$fake"
+ [ "$status" -eq 0 ]
+ [ -f "$PROJECT/.claude/rules/zz.md" ]
+}
diff --git a/scripts/tests/install-lang.bats b/scripts/tests/install-lang.bats
index ecfbe01..c00b915 100644
--- a/scripts/tests/install-lang.bats
+++ b/scripts/tests/install-lang.bats
@@ -79,6 +79,76 @@ teardown() {
grep -qxF "coverage/" "$PROJECT/.gitignore"
}
+@test "install-lang: seeds the language-neutral default CLAUDE.md when the bundle ships none" {
+ # Every shipping bundle now carries its own CLAUDE.md (python and typescript
+ # gained theirs 2026-07-23), so the fallback needs a synthetic bundle to
+ # exercise. Keep testing it: the fallback is what stops a bundle added later,
+ # before its CLAUDE.md is written, from inheriting another language's header.
+ fake="$REAL_REPO/languages/zz-test-noclaude"
+ mkdir -p "$fake/claude/rules"
+ printf '# rule\n' > "$fake/claude/rules/zz.md"
+ run bash "$INSTALL_LANG" zz-test-noclaude "$PROJECT"
+ rm -rf "$fake"
+
+ [ "$status" -eq 0 ]
+ [ -f "$PROJECT/CLAUDE.md" ]
+ # The default names no language, so it can't mislabel a project the way
+ # inheriting elisp's "Elisp project" header did.
+ ! grep -qi "Elisp project" "$PROJECT/CLAUDE.md"
+ grep -qF "names no language" "$PROJECT/CLAUDE.md"
+ [[ "$output" == *"language-neutral default"* ]]
+}
+
+@test "install-lang python: seeds the bundle's own CLAUDE.md, not the default" {
+ run bash "$INSTALL_LANG" python "$PROJECT"
+
+ [ "$status" -eq 0 ]
+ grep -qF "Python project." "$PROJECT/CLAUDE.md"
+ [[ "$output" == *"CLAUDE.md installed (python)"* ]]
+}
+
+@test "install-lang typescript: seeds the bundle's own CLAUDE.md, not the default" {
+ run bash "$INSTALL_LANG" typescript "$PROJECT"
+
+ [ "$status" -eq 0 ]
+ grep -qF "TypeScript/JavaScript project." "$PROJECT/CLAUDE.md"
+ [[ "$output" == *"CLAUDE.md installed (typescript)"* ]]
+}
+
+@test "install-lang elisp: seeds the bundle's own CLAUDE.md, not the default" {
+ run bash "$INSTALL_LANG" elisp "$PROJECT"
+
+ [ "$status" -eq 0 ]
+ grep -qF "Elisp project." "$PROJECT/CLAUDE.md"
+ [[ "$output" == *"CLAUDE.md installed (elisp)"* ]]
+}
+
+@test "install-lang python: does not overwrite an existing CLAUDE.md without FORCE" {
+ echo "MY OWN CLAUDE" > "$PROJECT/CLAUDE.md"
+ run bash "$INSTALL_LANG" python "$PROJECT"
+
+ [ "$status" -eq 0 ]
+ grep -qxF "MY OWN CLAUDE" "$PROJECT/CLAUDE.md"
+}
+
+@test "install-lang bash: full bundle lands (rules, hook, settings, githook, CLAUDE.md)" {
+ run bash "$INSTALL_LANG" bash "$PROJECT"
+
+ [ "$status" -eq 0 ]
+ # Language + testing rules — the bundle's sync fingerprint
+ [ -f "$PROJECT/.claude/rules/bash.md" ]
+ [ -f "$PROJECT/.claude/rules/bash-testing.md" ]
+ # PostToolUse validate hook, executable and wired into settings
+ [ -x "$PROJECT/.claude/hooks/validate-bash.sh" ]
+ grep -qF "validate-bash.sh" "$PROJECT/.claude/settings.json"
+ # Pre-commit githook
+ [ -x "$PROJECT/githooks/pre-commit" ]
+ # The bundle ships its own CLAUDE.md, so it wins over the neutral default
+ grep -qF "Bash/shell project" "$PROJECT/CLAUDE.md"
+ # Gitignore footprint
+ grep -qxF ".claude/" "$PROJECT/.gitignore"
+}
+
@test "install-lang go: full bundle lands (rules, hook, settings, githook, CLAUDE.md, coverage)" {
run bash "$INSTALL_LANG" go "$PROJECT"
@@ -100,3 +170,32 @@ teardown() {
grep -qxF ".claude/" "$PROJECT/.gitignore"
grep -qxF "cover.out" "$PROJECT/.gitignore"
}
+
+@test "install-lang python: full bundle lands (rules, hook, settings, githook, CLAUDE.md, coverage)" {
+ run bash "$INSTALL_LANG" python "$PROJECT"
+
+ [ "$status" -eq 0 ]
+ [ -f "$PROJECT/.claude/rules/python-testing.md" ]
+ # PostToolUse validate hook, executable and wired into settings
+ [ -x "$PROJECT/.claude/hooks/validate-python.sh" ]
+ grep -qF "validate-python.sh" "$PROJECT/.claude/settings.json"
+ # Pre-commit githook — the secret scan. Absent until 2026-07-23.
+ [ -x "$PROJECT/githooks/pre-commit" ]
+ grep -qF "potential secret" "$PROJECT/githooks/pre-commit"
+ # Coverage slice
+ [ -f "$PROJECT/.claude/scripts/coverage-summary.py" ]
+ grep -qxF ".claude/" "$PROJECT/.gitignore"
+}
+
+@test "install-lang typescript: full bundle lands (rules, hook, settings, githook, CLAUDE.md, coverage)" {
+ run bash "$INSTALL_LANG" typescript "$PROJECT"
+
+ [ "$status" -eq 0 ]
+ [ -f "$PROJECT/.claude/rules/typescript-testing.md" ]
+ [ -x "$PROJECT/.claude/hooks/validate-typescript.sh" ]
+ grep -qF "validate-typescript.sh" "$PROJECT/.claude/settings.json"
+ [ -x "$PROJECT/githooks/pre-commit" ]
+ grep -qF "potential secret" "$PROJECT/githooks/pre-commit"
+ [ -f "$PROJECT/.claude/scripts/coverage-summary.js" ]
+ grep -qxF ".claude/" "$PROJECT/.gitignore"
+}
diff --git a/scripts/tests/lint-coverage.bats b/scripts/tests/lint-coverage.bats
new file mode 100644
index 0000000..130212d
--- /dev/null
+++ b/scripts/tests/lint-coverage.bats
@@ -0,0 +1,54 @@
+#!/usr/bin/env bats
+#
+# Coverage tests for scripts/lint.sh.
+#
+# lint.sh sweeps scripts/*.sh, languages/*/claude/hooks/*.sh, and
+# languages/*/githooks/* through check_hook (shebang present, executable bit
+# set). It never touched claude-templates/bin/ — the four scripts `make install`
+# symlinks into ~/.local/bin, so the most exposed shell in the repo was the only
+# shell with no gate over it (found 2026-07-24).
+#
+# These tests pin the coverage itself rather than the current cleanliness: they
+# plant a deliberately broken file in each swept location and assert lint.sh
+# complains. A location that stops being swept fails here.
+
+REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)"
+LINT="$REPO_ROOT/scripts/lint.sh"
+
+teardown() {
+ [ -n "${PLANTED:-}" ] && rm -f "$PLANTED"
+ PLANTED=""
+}
+
+@test "lint: a bin/ script missing its shebang is flagged" {
+ PLANTED="$REPO_ROOT/claude-templates/bin/zz-test-noshebang"
+ printf 'echo hi\n' > "$PLANTED"
+ chmod +x "$PLANTED"
+ run bash "$LINT"
+ [[ "$output" == *"zz-test-noshebang"* ]] || {
+ echo "lint.sh did not report the planted bin/ script — that path is unswept"
+ echo "$output"
+ return 1
+ }
+}
+
+@test "lint: a bin/ script that is not executable is flagged" {
+ PLANTED="$REPO_ROOT/claude-templates/bin/zz-test-noexec"
+ printf '#!/usr/bin/env bash\necho hi\n' > "$PLANTED"
+ chmod -x "$PLANTED"
+ run bash "$LINT"
+ [[ "$output" == *"zz-test-noexec"* ]]
+}
+
+@test "lint: the existing scripts/ sweep still works (guards the regression)" {
+ PLANTED="$REPO_ROOT/scripts/zz-test-noshebang.sh"
+ printf 'echo hi\n' > "$PLANTED"
+ chmod +x "$PLANTED"
+ run bash "$LINT"
+ [[ "$output" == *"zz-test-noshebang.sh"* ]]
+}
+
+@test "lint: the real tree passes (no planted file)" {
+ run bash "$LINT"
+ [ "$status" -eq 0 ]
+}
diff --git a/scripts/tests/pre-commit-secret-scan.bats b/scripts/tests/pre-commit-secret-scan.bats
new file mode 100644
index 0000000..4647556
--- /dev/null
+++ b/scripts/tests/pre-commit-secret-scan.bats
@@ -0,0 +1,197 @@
+#!/usr/bin/env bats
+# Tests for the secret-scan block shared by the elisp, bash, and go pre-commit
+# hooks. The block greps added lines in the staged diff for credential
+# patterns; a hit blocks the commit (exit 1), a clean scan falls through to the
+# variant's language check (exit 0).
+#
+# Every case stages a .txt file, so the language checks that follow the scan
+# (check-parens, shellcheck, gofmt) all skip and the scan is what's under test.
+#
+# The two boundary cases exist because of a live false-positive in a downstream
+# project: an embedded PNG sprite data URI blocked a real commit and forced
+# --no-verify. Root cause was `grep -iE` applying case-insensitivity to the
+# fixed-case AWS token AKIA[0-9A-Z]{16}, so any mixed-case 20-char run inside a
+# random base64 blob matched. Measured at ~6% of 100KB blobs; case-sensitive
+# matching drops it to 0 across ~10MB.
+
+# Discovered, never enumerated. This list read "elisp bash go" while python and
+# typescript also shipped pre-commit hooks, so every "in every variant" test
+# below silently skipped two bundles from the day they were added — the same
+# enumerate-instead-of-discover failure these tests exist to catch. A new bundle
+# is now covered the moment it has a hook.
+VARIANTS="$(cd "${BATS_TEST_DIRNAME}/../../languages" && \
+ for d in */githooks/pre-commit; do [ -f "$d" ] && printf '%s ' "${d%%/*}"; done)"
+
+setup() {
+ REPO="$(mktemp -d)"
+ cd "$REPO" || return 1
+ git init -q .
+ git config user.email t@example.com
+ git config user.name Test
+}
+
+teardown() {
+ cd /tmp || true
+ [ -n "${REPO:-}" ] && rm -rf "$REPO"
+}
+
+# Stage $2 as the content of file $1 (default staged.txt).
+stage() {
+ local file="${2:-staged.txt}"
+ printf '%s\n' "$1" > "$file"
+ git add "$file"
+}
+
+# Run a variant's hook in the temp repo. $1 = variant name.
+run_hook() {
+ bash "${BATS_TEST_DIRNAME}/../../languages/$1/githooks/pre-commit"
+}
+
+# ---- Normal: real secrets still block, clean content still passes ----
+
+@test "secret-scan: clean content passes in every variant" {
+ stage 'const greeting = "hello world";'
+ for v in $VARIANTS; do
+ run run_hook "$v"
+ [ "$status" -eq 0 ] || { echo "$v blocked clean content: $output"; return 1; }
+ done
+}
+
+@test "secret-scan: a real uppercase AWS access key blocks in every variant" {
+ stage 'aws_key = "AKIAIOSFODNN7EXAMPLE"'
+ for v in $VARIANTS; do
+ run run_hook "$v"
+ [ "$status" -eq 1 ] || { echo "$v missed an AWS key"; return 1; }
+ [[ "$output" == *"potential secret"* ]]
+ done
+}
+
+@test "secret-scan: a keyword=value credential blocks in every variant" {
+ stage 'api_key: "sk_live_9f3b2a7c1e4d8f0a6b5"'
+ for v in $VARIANTS; do
+ run run_hook "$v"
+ [ "$status" -eq 1 ] || { echo "$v missed an api_key assignment"; return 1; }
+ done
+}
+
+@test "secret-scan: a PEM private-key header blocks in every variant" {
+ stage '-----BEGIN RSA PRIVATE KEY-----'
+ for v in $VARIANTS; do
+ run run_hook "$v"
+ [ "$status" -eq 1 ] || { echo "$v missed a PEM header"; return 1; }
+ done
+}
+
+# ---- Boundary: the case-sensitivity fix ----
+
+@test "secret-scan: a lowercase akia-like run does not block (AWS keys are uppercase)" {
+ # Under `grep -iE` this matched the AKIA token and blocked a legitimate commit.
+ stage 'const blob = "akiaiosfodnn7examplexyz";'
+ for v in $VARIANTS; do
+ run run_hook "$v"
+ [ "$status" -eq 0 ] || { echo "$v false-positived on a lowercase run: $output"; return 1; }
+ done
+}
+
+@test "secret-scan: an embedded base64 data URI carrying a mixed-case akia run does not block" {
+ # The live failure: a sprite blob whose random base64 contained a mixed-case
+ # 20-char run. Case-sensitive matching is what clears it.
+ stage 'const SPRITE = "data:image/png;base64,iVBORw0KGgoAkIaIOSFODNN7ExAMPLEqQmCC";'
+ for v in $VARIANTS; do
+ run run_hook "$v"
+ [ "$status" -eq 0 ] || { echo "$v false-positived on a sprite data URI: $output"; return 1; }
+ done
+}
+
+# ---- Boundary: the scan must not go blind on data-URI lines ----
+
+@test "secret-scan: a real credential sharing a line with a base64 data URI still blocks" {
+ # Minified bundles put a whole file on one line, so a data URI and a real key
+ # can share it. Skipping any line containing ';base64,' would hide the key.
+ stage 'const S="data:image/png;base64,iVBORw0KGgoAAAANS";const c={api_key:"sk_live_9f3b2a7c1e4d8f0a6b5"};'
+ for v in $VARIANTS; do
+ run run_hook "$v"
+ [ "$status" -eq 1 ] || { echo "$v went blind on a data-URI line and missed the key"; return 1; }
+ done
+}
+
+@test "secret-scan: a line matching both passes is reported once, not twice" {
+ # The scan runs a case-sensitive and a case-insensitive pass. A line carrying
+ # both an AWS key and a keyword=value credential hits both; reporting it twice
+ # reads as two separate leaks.
+ stage 'api_key = "AKIAIOSFODNN7EXAMPLE_padding"'
+ for v in $VARIANTS; do
+ run run_hook "$v"
+ [ "$status" -eq 1 ]
+ hits="$(printf '%s\n' "$output" | grep -c 'AKIAIOSFODNN7EXAMPLE_padding')"
+ [ "$hits" -eq 1 ] || { echo "$v reported the line $hits times, want 1"; return 1; }
+ done
+}
+
+# ---- Error / edge ----
+
+@test "secret-scan: an empty staged diff passes" {
+ for v in $VARIANTS; do
+ run run_hook "$v"
+ [ "$status" -eq 0 ] || { echo "$v failed on an empty diff: $output"; return 1; }
+ done
+}
+
+@test "secret-scan: a secret only on a removed line does not block" {
+ # The scan reads added lines. Deleting a key should never block the deletion.
+ stage 'aws_key = "AKIAIOSFODNN7EXAMPLE"'
+ git commit -qm "seed" --no-verify
+ printf 'clean\n' > staged.txt
+ git add staged.txt
+ for v in $VARIANTS; do
+ run run_hook "$v"
+ [ "$status" -eq 0 ] || { echo "$v blocked a removal: $output"; return 1; }
+ done
+}
+
+# ---- Fail-closed: a broken git must never read as "nothing to scan" ----
+
+# Put a stub `git` ahead of the real one that fails only the staged-diff call
+# and delegates everything else, so just the pipeline under test breaks.
+break_git() {
+ mkdir -p "$REPO/bin"
+ cat > "$REPO/bin/git" <<'STUB'
+#!/usr/bin/env bash
+if [ "${1:-}" = "diff" ] && [ "${2:-}" = "--cached" ]; then
+ echo "simulated git failure" >&2
+ exit 128
+fi
+exec /usr/bin/git "$@"
+STUB
+ chmod +x "$REPO/bin/git"
+}
+
+@test "secret-scan: a broken git refuses rather than passing blind, in every variant" {
+ # The scan built its input as `git diff ... | grep ... || true`. With no
+ # pipefail, a git failure yielded an empty string, so the scan searched
+ # nothing, found nothing, and reported clean with a real secret staged.
+ # Found by .emacs.d in elisp 2026-07-24; all five variants had it.
+ stage 'aws_key = "AKIAIOSFODNN7EXAMPLE"'
+ break_git
+ for v in $VARIANTS; do
+ run env PATH="$REPO/bin:$PATH" bash \
+ "${BATS_TEST_DIRNAME}/../../languages/$v/githooks/pre-commit"
+ [ "$status" -ne 0 ] || {
+ echo "$v FAILED OPEN: exited 0 with a secret staged and git broken"
+ return 1
+ }
+ done
+}
+
+@test "secret-scan: the refusal says why, in every variant" {
+ stage 'aws_key = "AKIAIOSFODNN7EXAMPLE"'
+ break_git
+ for v in $VARIANTS; do
+ run env PATH="$REPO/bin:$PATH" bash \
+ "${BATS_TEST_DIRNAME}/../../languages/$v/githooks/pre-commit"
+ [[ "$output" == *"cannot read"* ]] || {
+ echo "$v refused without naming the cause: $output"
+ return 1
+ }
+ done
+}
diff --git a/scripts/tests/rename-ai-artifact.bats b/scripts/tests/rename-ai-artifact.bats
index f00c92f..ea7f36e 100644
--- a/scripts/tests/rename-ai-artifact.bats
+++ b/scripts/tests/rename-ai-artifact.bats
@@ -27,7 +27,17 @@ setup() {
printf 'Old session mentioning foo and foo.org — this is history.\n' > "$base/sessions/2026-01-01-old.org"
done
printf 'See foo.org and foo-helper.py. Also foobar.org stays.\n' > "$REPO/notes.org"
- ( cd "$REPO" && git init -q && git add -A && git -c user.email=t@t -c user.name=t commit -qm init )
+ # Disable background auto-maintenance/gc before any git command can arm it:
+ # otherwise a post-commit `git maintenance run --auto` writes into .git after
+ # the test body and races teardown's `rm -rf`, which then fails intermittently
+ # with ".git: Directory not empty". Diagnose-not-mask: kill the writer, don't
+ # retry the rm.
+ ( cd "$REPO" \
+ && git init -q \
+ && git config gc.auto 0 \
+ && git config maintenance.auto false \
+ && git add -A \
+ && git -c user.email=t@t -c user.name=t commit -qm init )
}
teardown() {
diff --git a/scripts/tests/signal-receive.bats b/scripts/tests/signal-receive.bats
new file mode 100644
index 0000000..8fe4d67
--- /dev/null
+++ b/scripts/tests/signal-receive.bats
@@ -0,0 +1,66 @@
+#!/usr/bin/env bats
+# signal-receive.sh — drains the Signal pager account's inbound queue to keep it
+# warm (the roam-sync-shaped fix for receive-staleness) and to surface Craig's
+# replies. Run on velox by the signal-receive systemd timer. These tests stub
+# signal-cli on PATH to verify command construction without a network or the
+# real account.
+
+setup() {
+ REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)"
+ RECV="$REPO_ROOT/scripts/signal-receive.sh"
+ STUBS="$(mktemp -d)"
+ LOG="$STUBS/calls.log"
+ # HAS_ACCOUNT (default 1) controls the listAccounts answer so the local-account
+ # guard can be exercised: "1" lists both test accounts as present, "0" lists
+ # none (so the guard no-ops).
+ cat > "$STUBS/signal-cli" <<EOF
+#!/bin/bash
+if [ "\$1" = "listAccounts" ]; then
+ if [ "\${HAS_ACCOUNT:-1}" = "1" ]; then
+ echo "Number: +15045173983"
+ echo "Number: +19995550000"
+ fi
+ exit 0
+fi
+echo "signal-cli \$*" >> "$LOG"
+exit 0
+EOF
+ chmod +x "$STUBS/signal-cli"
+}
+
+teardown() {
+ rm -rf "$STUBS"
+}
+
+@test "defaults to the pager account, a 10s timeout, and read receipts" {
+ PATH="$STUBS:$PATH" run bash "$RECV"
+ [ "$status" -eq 0 ]
+ grep -q -- "-a +15045173983 receive --timeout 10 --send-read-receipts" "$LOG"
+}
+
+@test "honors an explicit account and timeout" {
+ PATH="$STUBS:$PATH" run bash "$RECV" +19995550000 25
+ [ "$status" -eq 0 ]
+ grep -q -- "-a +19995550000 receive --timeout 25" "$LOG"
+}
+
+@test "no-ops (exit 0) when the account is not registered on this machine" {
+ # Guards the runbook claim that the timer no-ops on a common-package machine
+ # that stows the units but holds no pager account.
+ HAS_ACCOUNT=0 PATH="$STUBS:$PATH" run bash "$RECV"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"not registered"* ]]
+ # The receive must not have run.
+ ! grep -q "receive" "$LOG"
+}
+
+@test "no-ops (exit 0) when signal-cli is not on PATH" {
+ # Empty stub dir with no signal-cli; echo/exit/command are bash builtins, so
+ # the script still runs and hits its signal-cli-absent branch, which no-ops.
+ rm -f "$STUBS/signal-cli"
+ # Invoke bash by absolute path so `run` finds it regardless of the empty
+ # PATH the script itself sees.
+ PATH="$STUBS" run "$(command -v bash)" "$RECV"
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"signal-cli"* ]]
+}
diff --git a/scripts/tests/sweep-gitignore-tooling.bats b/scripts/tests/sweep-gitignore-tooling.bats
index a28087e..240c3be 100644
--- a/scripts/tests/sweep-gitignore-tooling.bats
+++ b/scripts/tests/sweep-gitignore-tooling.bats
@@ -109,3 +109,125 @@ make_project() {
[ "$status" -eq 0 ]
[[ "$output" == *"not a git checkout"* ]]
}
+
+@test "sweep: anchored /.ai/ is recognized as gitignore-mode, appends anchored" {
+ make_project anchored $'/.ai/\n'
+
+ run bash "$SWEEP" "$ROOT"
+
+ [ "$status" -eq 0 ]
+ [[ "$output" != *"anchored — track-mode"* ]]
+ grep -qFx "/.claude/" "$ROOT/anchored/.gitignore"
+ grep -qFx "/CLAUDE.md" "$ROOT/anchored/.gitignore"
+ grep -qFx "/AGENTS.md" "$ROOT/anchored/.gitignore"
+}
+
+@test "sweep: anchored partial project gets only the missing lines" {
+ make_project anchoredpartial $'/.ai/\n/.claude/\n'
+
+ run bash "$SWEEP" "$ROOT"
+
+ [ "$status" -eq 0 ]
+ # /.claude/ already present in anchored form — not re-added in either form.
+ [ "$(grep -cFx '/.claude/' "$ROOT/anchoredpartial/.gitignore")" -eq 1 ]
+ ! grep -qFx ".claude/" "$ROOT/anchoredpartial/.gitignore"
+ grep -qFx "/CLAUDE.md" "$ROOT/anchoredpartial/.gitignore"
+ grep -qFx "/AGENTS.md" "$ROOT/anchoredpartial/.gitignore"
+}
+
+@test "sweep: anchored gitignore-mode is idempotent" {
+ make_project anchored2 $'/.ai/\n'
+ bash "$SWEEP" "$ROOT" >/dev/null
+
+ run bash "$SWEEP" "$ROOT"
+
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"already complete"* ]]
+ [ "$(grep -cFx '/.claude/' "$ROOT/anchored2/.gitignore")" -eq 1 ]
+}
+
+@test "sweep: track-mode with tracked tooling and a non-cjennings.net remote warns" {
+ make_project publictrack $'out/\n'
+ echo "# project rules" > "$ROOT/publictrack/CLAUDE.md"
+ (cd "$ROOT/publictrack" \
+ && git add CLAUDE.md \
+ && git -c user.email=t@t -c user.name=t commit -qm seed \
+ && git remote add origin git@github.com:someone/publictrack.git)
+
+ run bash "$SWEEP" "$ROOT"
+
+ [ "$status" -eq 0 ]
+ [[ "$output" == *"WARN"* ]]
+ [[ "$output" == *"publicly reachable"* ]]
+ # Still track-mode: nothing written to its .gitignore.
+ ! grep -qFx ".claude/" "$ROOT/publictrack/.gitignore"
+}
+
+@test "sweep: track-mode with tracked tooling on a cjennings.net remote stays quiet" {
+ make_project privatetrack $'out/\n'
+ echo "# project rules" > "$ROOT/privatetrack/CLAUDE.md"
+ (cd "$ROOT/privatetrack" \
+ && git add CLAUDE.md \
+ && git -c user.email=t@t -c user.name=t commit -qm seed \
+ && git remote add origin git@cjennings.net:privatetrack.git)
+
+ run bash "$SWEEP" "$ROOT"
+
+ [ "$status" -eq 0 ]
+ [[ "$output" != *"publicly reachable"* ]]
+}
+
+@test "sweep: the bare cjennings ssh-alias remote counts as private too" {
+ make_project aliastrack $'out/\n'
+ echo "# project rules" > "$ROOT/aliastrack/CLAUDE.md"
+ (cd "$ROOT/aliastrack" \
+ && git add CLAUDE.md \
+ && git -c user.email=t@t -c user.name=t commit -qm seed \
+ && git remote add origin git@cjennings:aliastrack.git)
+
+ run bash "$SWEEP" "$ROOT"
+
+ [ "$status" -eq 0 ]
+ [[ "$output" != *"publicly reachable"* ]]
+}
+
+# --- temp/ ephemeral-artifacts backfill (mode-independent) ---
+
+@test "sweep: adds temp/ to a gitignore-mode project" {
+ make_project gimode $'.ai/\n'
+
+ run bash "$SWEEP" "$ROOT"
+
+ [ "$status" -eq 0 ]
+ grep -qFx "temp/" "$ROOT/gimode/.gitignore"
+}
+
+@test "sweep: adds temp/ to a TRACK-mode project too (temp/ is mode-independent)" {
+ make_project trackmode $'# build\nout/\n'
+
+ run bash "$SWEEP" "$ROOT"
+
+ [ "$status" -eq 0 ]
+ # The tooling set is skipped in track mode, but temp/ is not.
+ grep -qFx "temp/" "$ROOT/trackmode/.gitignore"
+ ! grep -qFx ".ai/" "$ROOT/trackmode/.gitignore"
+}
+
+@test "sweep: never adds working/" {
+ make_project gimode $'.ai/\n'
+
+ run bash "$SWEEP" "$ROOT"
+
+ [ "$status" -eq 0 ]
+ ! grep -qEx "/?working/?" "$ROOT/gimode/.gitignore"
+}
+
+@test "sweep: temp/ backfill is idempotent" {
+ make_project gimode $'.ai/\ntemp/\n'
+ bash "$SWEEP" "$ROOT" >/dev/null
+
+ run bash "$SWEEP" "$ROOT"
+
+ [ "$status" -eq 0 ]
+ [ "$(grep -cFx 'temp/' "$ROOT/gimode/.gitignore")" -eq 1 ]
+}
diff --git a/scripts/tests/sync-language-bundle.bats b/scripts/tests/sync-language-bundle.bats
index 1871444..0eb3ae2 100644
--- a/scripts/tests/sync-language-bundle.bats
+++ b/scripts/tests/sync-language-bundle.bats
@@ -19,7 +19,20 @@ teardown() {
rm -rf "$PROJ"
}
-# Mirror install-lang.sh: copy the bundle's files into a synthetic project.
+# Mirror what a CURRENT install-lang.sh leaves: language rules only, no copies
+# of the generic rules (those live once at ~/.claude/rules/).
+install_bundle_current() {
+ install_bundle "$1" "$2"
+ local f
+ for f in "$REAL_REPO/claude-rules"/*.md; do
+ [ -f "$f" ] || continue
+ rm -f "$2/.claude/rules/$(basename "$f")"
+ done
+}
+
+# Mirror what an OLDER install-lang.sh left behind: language rules PLUS copies
+# of every generic rule. This is the state the sweep exists to clean up, and
+# real projects are still in it until their next startup.
install_bundle() {
local lang="$1" proj="$2"
mkdir -p "$proj/.claude/rules"
@@ -67,14 +80,14 @@ install_team_overlay() {
}
@test "sync: clean elisp bundle is a quiet no-op (exit 0)" {
- install_bundle elisp "$PROJ"
+ install_bundle_current elisp "$PROJ"
run bash "$SCRIPT" "$PROJ"
[ "$status" -eq 0 ]
[ -z "$output" ]
}
@test "sync: absent CLAUDE.md is not flagged as drift (seed-only/project-owned)" {
- install_bundle elisp "$PROJ" # helper never seeds CLAUDE.md
+ install_bundle_current elisp "$PROJ" # helper never seeds CLAUDE.md
[ ! -f "$PROJ/CLAUDE.md" ]
run bash "$SCRIPT" "$PROJ"
[ "$status" -eq 0 ]
@@ -94,13 +107,17 @@ install_team_overlay() {
matches_canonical ".claude/rules/elisp.md" "$REAL_REPO/languages/elisp/claude/rules/elisp.md"
}
-@test "sync: drifted generic rule is auto-fixed and restored" {
+# Generic rules are no longer auto-fixed in place: they are swept, because the
+# global copy at ~/.claude/rules/ is the one that loads. A drifted project copy
+# is not repaired, it is removed — which is the stronger fix, since the drifted
+# copy outranked the global rule while it existed.
+@test "sync: a drifted generic rule copy is swept, not repaired" {
install_bundle elisp "$PROJ"
echo "junk" >> "$PROJ/.claude/rules/commits.md"
run bash "$SCRIPT" "$PROJ"
[ "$status" -eq 0 ]
- [[ "$output" == *".claude/rules/commits.md"* ]]
- matches_canonical ".claude/rules/commits.md" "$REAL_REPO/claude-rules/commits.md"
+ [[ "$output" == *"swept"* ]]
+ [ ! -f "$PROJ/.claude/rules/commits.md" ]
}
@test "sync: missing rule is re-copied" {
@@ -276,3 +293,49 @@ install_team_overlay() {
[ ! -f "$PROJ/.claude/rules/commits.md" ]
[ ! -f "$PROJ/.claude/rules/testing.md" ]
}
+
+# --- generic-rule de-duplication -------------------------------------------
+#
+# Generic rules live at ~/.claude/rules/ (symlinked by `make install`) and load
+# in every session. Copying them into each project as well made Claude Code
+# load them twice, and project copies take priority — so a stale project copy
+# silently overrode the fresh global one. The bundle now ships only its own
+# language rules and sweeps the duplicates it previously installed.
+
+@test "sync: sweeps generic rule copies that duplicate the global set" {
+ install_bundle python "$PROJ"
+ [ -f "$PROJ/.claude/rules/commits.md" ]
+ HOME_RULES="$(mktemp -d)" ; mkdir -p "$HOME_RULES/.claude/rules"
+ cp "$REAL_REPO/claude-rules"/*.md "$HOME_RULES/.claude/rules/"
+ run env HOME="$HOME_RULES" bash "$SCRIPT" "$PROJ"
+ [ "$status" -eq 0 ]
+ [ ! -f "$PROJ/.claude/rules/commits.md" ]
+ [ ! -f "$PROJ/.claude/rules/todo-format.md" ]
+}
+
+@test "sync: keeps the language bundle's own rules while sweeping generics" {
+ install_bundle python "$PROJ"
+ HOME_RULES="$(mktemp -d)" ; mkdir -p "$HOME_RULES/.claude/rules"
+ cp "$REAL_REPO/claude-rules"/*.md "$HOME_RULES/.claude/rules/"
+ run env HOME="$HOME_RULES" bash "$SCRIPT" "$PROJ"
+ [ "$status" -eq 0 ]
+ [ -f "$PROJ/.claude/rules/python-testing.md" ]
+}
+
+@test "sync: keeps a project-owned overlay rule the bundle does not own" {
+ install_bundle python "$PROJ"
+ printf '# Publishing\n\nApplies to: `**/*`\n' > "$PROJ/.claude/rules/publishing.md"
+ HOME_RULES="$(mktemp -d)" ; mkdir -p "$HOME_RULES/.claude/rules"
+ cp "$REAL_REPO/claude-rules"/*.md "$HOME_RULES/.claude/rules/"
+ run env HOME="$HOME_RULES" bash "$SCRIPT" "$PROJ"
+ [ "$status" -eq 0 ]
+ [ -f "$PROJ/.claude/rules/publishing.md" ]
+}
+
+@test "sync: does NOT sweep when the global rule is absent (nothing takes over)" {
+ install_bundle python "$PROJ"
+ HOME_RULES="$(mktemp -d)" # no ~/.claude/rules/ at all
+ run env HOME="$HOME_RULES" bash "$SCRIPT" "$PROJ"
+ [ "$status" -eq 0 ]
+ [ -f "$PROJ/.claude/rules/commits.md" ]
+}
diff --git a/testing-standards/SKILL.md b/testing-standards/SKILL.md
new file mode 100644
index 0000000..1f16528
--- /dev/null
+++ b/testing-standards/SKILL.md
@@ -0,0 +1,391 @@
+---
+name: testing-standards
+description: |
+ The full testing standard: characterization tests for untested legacy code, the Normal/Boundary/Error case detail, combinatorial and property-based and mutation testing, test organization and the pyramid, integration-test rules, naming conventions, test-quality rules (independence, determinism, performance, mocking boundaries, signs of overmocking, testing framework-heavy code, never inlining production code, asserting error behavior not error text), the refactor-when-tests-are-hard principle, coverage targets, the TDD discipline table, the spike exception, and the anti-pattern list.
+
+ Use when writing or reviewing tests, deciding how to test something, hardening untested code, or judging whether a test suite is adequate.
+
+ Do NOT use for the standing directive itself — that TDD is the default and that every unit needs Normal, Boundary, and Error cases lives in claude-rules/testing.md and is always loaded. Also see the add-tests skill for the guided coverage workflow and pairwise-tests for the combinatorial matrix generator.
+---
+
+# Testing Standards — the detail
+
+Applies to test code and to decisions about how to test.
+
+The standing directive is NOT here. That TDD is the default, and that every
+unit needs Normal, Boundary, and Error cases, lives in `claude-rules/testing.md`
+and is always loaded, because it has to fire before any code gets written and
+nothing else would summon it. This file is everything needed once you are
+actually writing the tests.
+
+### Understand Before You Test
+
+Before writing tests, invest time in understanding the code:
+
+1. **Explore the codebase** — Read the module under test, its callers, and its dependencies. Understand the data flow end to end.
+2. **Identify the root cause** — If fixing a bug, trace the problem to its origin. Don't test (or fix) surface symptoms when the real issue is deeper in the call chain.
+3. **Reason through edge cases** — Consider boundary conditions, error states, concurrent access, and interactions with adjacent modules. Your tests should cover what could actually go wrong, not just the obvious happy path.
+
+### Adding Tests to Existing Untested Code
+
+When working in a codebase without tests:
+
+1. Write a **characterization test** that captures current behavior before making changes
+2. Use the characterization test as a safety net while refactoring
+3. Then follow normal TDD for the new change
+
+A characterization test asserts what the code *actually does* right now, not
+what it *should* do. Write it by running the code against a fixed input,
+reading the exact value or effect it currently produces, and asserting that
+value — Feathers' recipe is to assert something you know is wrong, run it, and
+paste the real value out of the failure. You don't need to know the correct
+answer to write one; you record the observed one. That's what makes it
+mechanical enough to bring a large untested surface under test without
+re-deriving each unit's spec.
+
+**Characterize with the same Normal/Boundary/Error set as any unit** (the three
+categories below), not one happy-path capture per function. On a characterization
+test the negative and boundary cases are the ones that find bugs: untested legacy
+code is weakest exactly at the empty input, the malformed value, the missing
+upstream, and pinning what it *currently* does there writes the wrong behavior
+down in black and white, where it becomes a bug you can see and decide on. When a
+pinned case turns out to be a bug rather than behavior worth preserving, that one
+test graduates from "record current" to "assert correct" and you fix the code.
+The happy-path case is the regression net; the negative and boundary cases are
+the audit.
+
+Bugs that live *inside* a unit are caught by this three-category set; bugs in how
+units compose — ordering, shared state handed between them — are invisible to any
+per-unit test and need a functional/integration test over the composed path (see
+Integration Tests below and the pyramid).
+
+### 1. Normal Cases (Happy Path)
+- Standard inputs and expected use cases
+- Common workflows and default configurations
+- Typical data volumes
+
+### 2. Boundary Cases
+- Minimum/maximum values (0, 1, -1, MAX_INT)
+- Empty vs null vs undefined (language-appropriate)
+- Single-element collections
+- Unicode and internationalization (emoji, RTL text, combining characters)
+- Very long strings, deeply nested structures
+- Timezone boundaries (midnight, DST transitions)
+- Date edge cases (leap years, month boundaries)
+
+### 3. Error Cases
+- Invalid inputs and type mismatches
+- Network failures and timeouts
+- Missing required parameters
+- Permission denied scenarios
+- Resource exhaustion
+- Malformed data
+
+## Combinatorial Coverage
+
+For functions with 3+ parameters that each take multiple values (feature-flag
+combinations, config matrices, permission/role interactions, multi-field
+form validation, API parameter spaces), the exhaustive test count explodes
+(M^N) while 3-5 ad-hoc cases miss pair interactions. Use **pairwise /
+combinatorial testing** — generate a minimal matrix that hits every 2-way
+combination of parameter values. Empirically catches 60-90% of combinatorial
+bugs with 80-99% fewer tests.
+
+Invoke `/pairwise-tests` on the offending function; continue using `/add-tests`
+and the Normal/Boundary/Error discipline for the rest. The two approaches
+complement: pairwise covers parameter *interactions*; category discipline
+covers each parameter's individual edge space.
+
+Skip pairwise when: the function has 1-2 parameters (just write the cases),
+the context requires *provably* exhaustive coverage (regulated systems — document
+in an ADR), or the testing target is non-parametric (single happy path,
+performance regression, a specific error).
+
+## Escalation Beyond Category and Pairwise
+
+The Normal/Boundary/Error categories and the pairwise matrix are the default
+discipline. Two further techniques escalate beyond them — reach for them when
+the default leaves a gap, not on every unit.
+
+### Property-Based Testing
+
+When an invariant holds across a broad input domain — round-trips
+(`decode(encode(x)) == x`), idempotence (`f(f(x)) == f(x)`), ordering
+invariants (output is always sorted), or any "output always satisfies X" —
+generate inputs and assert the property instead of enumerating cases. The
+generator explores corners you wouldn't think to write by hand, and a
+failing case shrinks to a minimal reproducer. Use the standard tool for the
+language (Hypothesis for Python, fast-check for JS, proptest for Rust).
+State the property as the test name and let the framework supply the inputs.
+
+Reach for this when the behavior is a law over a domain rather than a fixed
+set of examples. Keep category-discipline cases for the specific edges that
+must always hold; the property test covers the space between them.
+
+### Mutation Testing
+
+When line coverage is high but you suspect the assertions are thin — tests
+that execute the code without checking its output, or that pass with a
+function body replaced by a stub — use mutation testing to measure whether
+the suite actually kills injected faults. The tool flips conditionals, swaps
+operators, and deletes statements, then reruns the suite; a surviving mutant
+is a fault the tests didn't catch. Use mutmut or cosmic-ray for Python,
+Stryker for JS. High line coverage with a low mutation score means weak
+assertions, not a tested codebase.
+
+Reach for this on critical logic where coverage looks reassuring but you
+want evidence the tests would fail on a regression. It's a diagnostic, not a
+gate on every change — mutation runs are slow.
+
+## Test Organization
+
+Typical layout:
+
+```
+tests/
+ unit/ # One test file per source file
+ integration/ # Multi-component workflows
+ e2e/ # Full system tests
+```
+
+Per-language files may adjust this (e.g. Elisp collates ERT tests into
+`tests/test-<module>*.el` without subdirectories).
+
+### Testing Pyramid
+
+Rough proportions for most projects:
+- Unit tests: 70-80% (fast, isolated, granular)
+- Integration tests: 15-25% (component interactions, real dependencies)
+- E2E tests: 5-10% (full system, slowest)
+
+Don't duplicate coverage: if unit tests fully exercise a function's logic,
+integration tests should focus on *how* components interact — not repeat the
+function's case coverage.
+
+## Integration Tests
+
+Integration tests exercise multiple components together. Two rules:
+
+**The docstring names every component integrated** and marks which are real vs
+mocked. Integration failures are harder to pinpoint than unit failures;
+enumerating the participants up front tells you where to start looking.
+
+Example:
+
+```
+def test_integration_refund_during_sync_updates_ledger_atomically():
+ """Refund processed mid-sync updates order and ledger in one transaction.
+
+ Components integrated:
+ - OrderService.refund (entry point)
+ - PaymentGateway.reverse (MOCKED — returns success)
+ - Ledger.credit (real)
+ - db.transaction (real)
+
+ Validates:
+ - Refund rolls back if ledger write fails
+ - Both tables updated or neither
+ """
+```
+
+**Write an integration test when** multiple components must work together,
+state crosses function boundaries, or edge cases combine. **Don't** when
+single-function behavior suffices, or when mocking would erase the interaction
+you meant to test.
+
+## Naming Convention
+
+- Unit: `test_<module>_<function>_<scenario>_<expected>`
+- Integration: `test_integration_<workflow>_<scenario>_<outcome>`
+
+Examples:
+- `test_cart_apply_discount_expired_coupon_raises_error`
+- `test_integration_order_sync_network_timeout_retries_three_times`
+
+Languages that prefer camelCase, kebab-case, or other conventions keep the
+structure but use their idiom. Consistency within a project matters more than
+the specific case choice.
+
+## Test Quality
+
+### Independence
+- No shared mutable state between tests
+- Each test runs successfully in isolation
+- Explicit setup and teardown
+
+### Determinism
+- Never hardcode dates or times — generate them relative to `now()`
+- No reliance on test execution order
+- No flaky network calls in unit tests
+- Time/clock-mocking helpers must avoid two recurring failure modes:
+ - *Infinite recursion.* The helper must not call the primitive it's
+ replacing. If the mock for `now()` calls `now()`, the test stack
+ overflows. Compute the mock value from a fixed source (a captured
+ instant, an injected fake clock).
+ - *Scope-shadowing without reach.* A mock that only exists inside
+ the test function won't affect production code that reads the
+ symbol through its canonical path. Replace the symbol at its
+ definition site (monkey-patch the module attribute in Python,
+ redefine the global in Lisp, swap the package-level binding in
+ Go, replace the named export in JavaScript) — or inject a fake
+ via dependency-inversion. Don't lean on scope-shadowing
+ primitives (Lisp `let`, Python local rebind, JS shadowed `let`)
+ that fence the mock to the test's lexical scope; production code
+ won't see them and the test passes against the real clock.
+
+### Performance
+- Unit tests: <100ms each
+- Integration tests: <1s each
+- E2E tests: <10s each
+- Mark slow tests with appropriate decorators/tags
+
+### Mocking Boundaries
+Mock external dependencies at the system boundary:
+- Network calls (HTTP, gRPC, WebSocket)
+- File I/O and cloud storage
+- Time and dates
+- Third-party service clients
+
+Never mock:
+- The code under test
+- Internal domain logic
+- Framework behavior (ORM queries, middleware, hooks, buffer primitives)
+
+### Signs of Overmocking
+
+Ask yourself:
+
+- Would this test still pass if I replaced the function body with `raise NotImplementedError` (or equivalent)? If yes, the mocks are doing the work — you're testing mocks, not code.
+- Is the mock more complex than the function being tested? Smell.
+- Am I mocking internal string / parsing / decoding helpers? Those aren't boundaries — they're the work.
+- Does the test break when I refactor without changing behavior? Good tests survive refactors; overmocked ones couple to implementation.
+
+When tests demand heavy internal mocking, the fix isn't better mocks — it's
+restructuring the code (see *If Tests Are Hard to Write* below).
+
+### Testing Code That Uses Frameworks
+
+When a function mostly delegates to framework or library code, test *your*
+integration logic:
+- ✓ "I call the library with the right arguments in the right context"
+- ✓ "I handle its return value correctly"
+- ✗ "The library works in 50 scenarios" — trust it; it has its own tests
+
+For polyglot behavior (e.g., comment handling across C/Java/Go/JS), test 2-3
+representative modes thoroughly plus a minimal smoke test in the others.
+Exhaustive permutations are diminishing returns.
+
+### Test Real Code, Not Copies
+
+Never inline or copy production code into test files. Always `require`/`import`
+the module under test. Copied code passes even when production breaks — the
+bug hides behind the duplicate.
+
+Mock dependencies at their boundary; exercise the real function body.
+
+### Error Behavior, Not Error Text
+
+Test that errors occur with the right type; don't assert exact wording:
+- ✓ Right exception type (`pytest.raises(ValueError)`, `(should-error ... :type 'user-error)`)
+- ✓ Regex on values the message *must* contain (e.g., the offending filename)
+- ✗ `assert str(e) == "File 'foo' not found"` — breaks when prose changes even though behavior is unchanged
+
+Production code should emit clear, contextual errors. Tests verify the
+behavior (raised, caught, returned nil) and values that must appear — not the
+prose.
+
+## If Tests Are Hard to Write, Refactor the Code
+
+If a test needs extensive mocking of internal helpers, elaborate fixture
+scaffolding, or mocks that recreate the function's own logic, the production
+code needs restructuring — not the test.
+
+Signals:
+- Deep nesting (callbacks inside callbacks)
+- Long functions doing multiple things ("fetch AND parse AND decode AND save")
+- Tests that mock internal string / parsing / I/O helpers
+- Tests that break on refactors with no behavior change
+
+Fix: extract focused helpers (one responsibility each), test each in isolation
+with real inputs, compose them in a thin outer function. Several small unit
+tests plus one composition test beats one monster test behind a wall of mocks.
+
+When the untestable function is legacy code you're hardening, this extraction
+**is** the hardening — not a detour around it. A function whose boundary or
+error case can't be exercised without mocking the world (a shell function that
+calls `tmux`/`git` directly, a handler that reaches straight into I/O) can't be
+characterized, so you can't refactor it safely and you can't pin its edge
+behavior. Extracting the pure decision logic into a helper that takes plain
+inputs and returns a plain result makes that logic characterizable with the full
+Normal/Boundary/Error set; the I/O calls become a thin wrapper you cover once
+with a single composition test. "It needs too much mocking to test" is therefore
+never a reason to skip the boundary and error cases — it's the signal to reshape
+the function so those cases are writable.
+
+## Coverage Targets
+
+- Business logic and domain services: **90%+**
+- API endpoints and views: **80%+**
+- UI components: **70%+**
+- Utilities and helpers: **90%+**
+- Overall project minimum: **80%+**
+
+New code must not decrease coverage. PRs that lower coverage require justification.
+
+## TDD Discipline
+
+TDD is non-negotiable. These are the rationalizations agents use to skip it — don't fall for them:
+
+| Excuse | Why It's Wrong |
+|--------|----------------|
+| "This is too simple to need a test" | Simple code breaks too. The test takes 30 seconds. Write it. |
+| "I'll add tests after the implementation" | You won't, and even if you do, they'll test what you wrote rather than what was needed. Test-after validates implementation, not behavior. |
+| "Let me just get it working first" | That's not TDD. If you can't write a failing test, you don't understand the requirement yet. |
+| "This is just a refactor" | Refactors without tests are guesses. Write a characterization test first, then refactor while it stays green. |
+| "I'm only changing one line" | One-line changes cause production outages. Write a test that covers the line you're changing. |
+| "The existing code has no tests" | Start with a characterization test. Don't make the problem worse. |
+| "This is demo/prototype code" | Demos build habits. Untested demo code becomes untested production code. |
+| "I need to spike first" | Spikes are fine — under the protocol below. Throw the spike away, then write the first failing test before productionizing. |
+
+If you catch yourself thinking any of these, stop and write the test.
+
+### The Spike Exception (Disciplined)
+
+TDD stays the default. The one sanctioned way to write code before a test is
+a spike — exploratory code that answers "is this approach even viable?" when
+you can't yet write a meaningful failing test because the shape of the
+solution is unknown. A spike is disciplined only when all three hold:
+
+1. **Timebox it.** Set a limit before starting (an hour, an afternoon) and
+ stop when it's up. An open-ended spike is just untested implementation
+ wearing a different name.
+2. **Do not commit spike code.** The spike is a learning artifact, not a
+ deliverable. It never enters the branch history. Keep it in a scratch
+ file or a throwaway worktree.
+3. **Throw the spike away, then start with a failing test.** Once the spike
+ has answered the viability question, delete it. Write the first failing
+ test against the now-understood behavior, then productionize under normal
+ Red/Green/Refactor. The production code is written test-first even though
+ the exploration wasn't — you don't promote the spike into production by
+ bolting tests on after.
+
+The spike buys understanding, not code. If you find yourself keeping the
+spike because rewriting it feels wasteful, the timebox was too long or the
+problem was tractable enough to TDD from the start.
+
+## Anti-Patterns (Do Not Do)
+
+- Hardcoded dates or timestamps (they rot)
+- Testing implementation details instead of behavior
+- Mocking the thing you're testing
+- Mocking internal helpers (string ops, parsing, decoding) — those are the work
+- Inlining production code into test files — always `require` / `import` the real module
+- Asserting exact error-message text instead of type + key values
+- Shared mutable state between tests
+- Non-deterministic tests (random without seed, network in unit tests)
+- Testing framework behavior instead of your code
+- Ignoring or skipping failing tests without a tracking issue
+
+## Content scope
+
+Test code, fixtures, docstrings, and comments are checked into the repo and visible to the team. They must follow the *Content scope for public artifacts* rule in [`commits.md`](commits.md): no local paths, no private repo names, no personal tooling references.
diff --git a/todo.org b/todo.org
index 8eb0b02..9d47566 100644
--- a/todo.org
+++ b/todo.org
@@ -30,23 +30,522 @@ Optional *effort and autonomy tags* — orthogonal to type, both can apply on th
- =:quick:= — likely to take ≤30 minutes from start through verification.
- =:solo:= — Claude can complete the work end to end, including verification, without input from Craig.
+Optional *dependency tags* — cross-project, both plain tags with the which-project detail in the task body (per =todo-format.md=):
+
+- =:blocked:= — the task can't advance until another project delivers the work named in its body. =open-tasks.org= pulls =:blocked:= tasks out of the cascade and surfaces them on their own. Distinct from =VERIFY= (which waits on Craig).
+- =:blocker:= — this task owes work that's blocking another project (named in its body). =open-tasks.org= surfaces =:blocker:= tasks first, since clearing one unblocks the other project.
+
Tags are assigned and refreshed by =task-audit=; =task-review= keeps them honest in passing.
* Rulesets Open Work
-** DOING [#B] Wrap-up inbox/transcript routing to destination projects :feature:spec:
+** DOING [#B] Hostile subagent review before every agent commit :feature:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-28
+:END:
+From the roam inbox, 2026-07-28: "code reviews must occur before every commit an agent does, and they should be hostile reviews from a subagent without the agent's context."
+
+Two asks, and only the second is new. The publish flow already mandates a review before every commit (Step 1). What changes is *who reviews*: today the reviewing agent is the one that wrote the change, so it inherits the author's mental model, and =review-code= only *suggests* subagent dispatch, and only "for substantive reviews on large diffs".
+
+Decisions settled with Craig, 2026-07-28, and shipped:
+- *Scope* — every commit. The reviewer's own Phase 0 rules a diff trivial and returns Skipped, which satisfies the gate; the author never rules on their own diff.
+- *Stance* — "adversarial", not "hostile" (Craig's call). An agent told to attack manufactures findings, so the stance carries a substantiation floor: a finding not substantiated against the diff is dropped.
+- *What the reviewer gets* — the diff, a one-line claim of what it does, and the requirement source (ticket, plan, task body) where one exists. Withheld: the conversation, the exploration, the author's rationale. The requirement source stays *in* because it is the only artifact that can contradict the author's claim; withholding it makes the claim self-certifying.
+- *Loop* — re-review until the reviewer approves, turning on blocking findings rather than the verdict token. Bounded at three rounds, and stopped early on a finding that recurs after being reported fixed. Both bounds hand the decision to Craig; the unattended callers park instead.
+- *Adjudication* — Craig, never the author overruling the reviewer.
+- *Home* — the =publish= skill Step 1, with =review-code= carrying the adversarial contract and re-review mode.
+- *The =subagents.md= tension* — resolved with an Isolation Override section: the size heuristics assume the main thread could do the task equally well, and they lapse when its own context is what makes its answer untrustworthy.
+
+Remaining: nothing on the design. The change shipped in this session.
+
+** TODO [#B] wrap-org-table splits logical rows, and lint drives it :bug:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-29
+:END:
+Reported by work 2026-07-28 against =arch-00-deepsat-platform-spec-draft.org=. Reproduced here.
+
+*** Verified
+
+Each of these is a measurement, re-run under adversarial review. The analysis I built on top of them was wrong three times, so this section is deliberately separated from the open questions below.
+
+- *The defect.* =wrap-org-table.el= turns one logical table row into two or more, by writing a rule between its continuation lines. Content survives; structure does not.
+- *Root cause.* =wot--continuation-group-p= (=wrap-org-table.el:168=) requires every line past the first to carry at least one empty cell. When a row overflows in every column its continuation line is fully populated, the predicate rejects the group, and =wot--logical-rows= appends each physical line as its own row (=:197=).
+- *Controlled A/B.* Rules present in both, continuation line's middle cell the only variable: blank merges, populated splits.
+- *Idempotence is broken.* Running the tool twice on its own correct output corrupts it. Pass 1 emits a properly rule-delimited three-line row; pass 2 splits it into three rows. The docstring at =:203-204= asserts the opposite, and =wot-reformat-is-idempotent= passes because its fixture overflows only one column.
+- *lint doesn't just miss it, it causes it.* =lint-org.el:424= calls the same predicate. Given the tool's own correct output — a rule after every logical row — lint reports "missing rule between rows — wrap-org-table.el reflows it". Nothing is missing. Follow that advice and the tool splits the row; lint then reports the result 0 mechanical, 0 judgment. Control: a conformant table whose continuation keeps an empty cell returns 0 and 0. So the loop is not two independent green lights, it is the linter manufacturing a false violation and certifying the damage it caused.
+- *A second path.* With no hlines at all, =wot--logical-rows= short-circuits (=:184=) before the predicate is reached, so every physical line becomes a row.
+- *Nothing invokes the tool unattended.* =todo-cleanup.el= names it in comments only; the entry-script guard from the 2026-07-09 incident holds.
+- *rulesets is exposed*: =todo.org= carries a four-row attachment-sanitization table with no rules between rows. Don't reflow it until the tool is fixed — reflowing is the trigger.
+
+*** Two fixes that were verified to work
+
+- *Conformant round-trip.* Replacing the predicate body with =(> (length group) 1)= — pure rule-delimited grouping, no emitter marker, no new field — makes the corrupting pass-1 output round-trip cleanly.
+- *Telling a continuation group from two real rows at runtime.* Provenance isn't available (=wot-reformat-table-string= receives a string), but the test is cheap: merge the group, re-wrap at the allocated widths, compare to the group as given. A continuation group reproduces itself under merge-then-wrap; two genuinely distinct short rows don't, because merged they fit on one line.
+
+Start with a red test from the double-run repro — it needs no unusual input and falsifies the docstring and the passing idempotence test together. Emission is already correct (=:222-224= emits one rule per element of =rows=), so the repair is entirely in how =rows= is computed.
+
+*Prefer the round-trip check to any detector.* work implemented the predicate-based detection I recommended and demonstrated it cannot discriminate at any threshold (2026-07-29, worked examples from their repo). Run bare it flags 144 files — essentially every table — because an ordinary header-plus-body table is one multi-line fully-populated group and rejecting it is the predicate working correctly. Add a per-row-ruled precondition and it cuts to 9, but at 9 it still mixes a genuinely wrapped row (=arch-09:40=, a header row whose second line continues the sentence) with two distinct rows legitimately sharing a rule (their =todo.org:117=, where splitting is the *right* answer). Same structural signature, opposite correct verdicts. The difference is whether one line continues the other as prose, which is semantics and not in the parse.
+
+So =lint-org.el:424= cannot simply inherit the repair, and a checker keyed on the predicate would train people to ignore it. The idempotence property is the checkable one: reflow twice and diff. It sidesteps detection entirely, because round-tripping the tool's own output correctly is true by construction and needs nobody to decide what a group means.
+
+*** Open questions
+
+- *What should no-hline input do?* My "refuse to reflow" instruction was wrong: adding rules to a ruleless table is the tool's main job, and =test-wrap-org-table.el:147= asserts exactly that. The real danger is narrower — a no-hline table where some line carries an empty cell, so a physical line might be a continuation. Needs a decision on refuse, ask, or heuristic.
+- *Which path bit work?* One question settles it: did the =arch-00= Document Status table have rules between its rows before the reflow? I inferred "probably secondary" from a file-level scan, which can't resolve a table-level incident.
+- *How exposed is work?* Partly resolved by their own implementation, 2026-07-29. *Four secondary-path files are exact.* The primary-path number is 9 under a per-row-ruled precondition, which they correctly label a floor on a population they cannot cleanly define rather than a measurement — the precondition also drops the minimal =a_full= fixture, which has only two groups. My earlier "24 secondary-path files" certification was not sound: their original signature also matches every correctly-reflowed table, so it never partitioned.
+- *Does the grade go up?* Currently Major x most users frequently = P2 = [#B] (any table you reflow that overflows every column, which is what a width violation sends you to the tool to fix). The lint-drives-it finding arrived after that grading and may lift it.
+
+*** Provenance
+
+work reverted their table and left it over-budget at 134 columns, correctly: an over-wide table is cosmetic, a table that says something else is not. I hit this same failure on 2026-07-27, wrote "reflowed a table into a worse shape and lint-org then passed it" into a session summary, and never filed it — which is why work met it three weeks later. Their framing beats the apology: a correct observation recorded and then read as fine is the same failure as a green check on a corrupted table.
+
+Process note for whoever picks this up. This task bounded out of its review loop at three rounds. Every finding across all three landed on inference, never on a measurement — the reproductions held throughout while the reasoning built on them was refuted repeatedly. Trust the Verified section; re-derive anything else.
+
+
+** TODO [#B] Synced workflows link outside the synced set with ../../ :bug:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-28
+:END:
+Seven link sites across four synced workflows, three distinct targets, all escaping the =.ai/= boundary into rulesets repo-root paths. From a consuming project's =.ai/workflows/=, =../../= is that project's root, where none of these exist:
+
+- → =../../claude-rules/todo-format.md= (five sites): =open-tasks.org:163=, =task-audit.org:84=, =task-review.org:60=, =task-review.org:64=, =task-review.org:99=
+- → =../../docs/design/task-review.org= (one site): =task-review.org:11=
+- → =../../flush/SKILL.md= (one site): =suspend.org:22=
+
+Verified dead in both home and =.emacs.d=; they resolve only in rulesets, which is why nobody noticed.
+
+Grading: Minor severity (a documented reference an agent can't follow, workaround is to search) x every user every time (every consuming project, every sync) = P2 = [#B]. Same grade as the =references/= link this came from, and the same mechanism — a synced file linking a path the sync doesn't deliver — with seven sites instead of one.
+
+Fix direction, two halves. Rewrite the seven as prose references naming the file, the same move taken for the credential paths: a link that resolves in one repo shouldn't be a link in a file that ships to two dozen. Then close the gate, or the eighth arrives unnoticed.
+
+The gate already exists and is nearly right. =scripts/lint.sh='s =check_md_links= was written for this exact class — its comment says so ("Validate cross-references to =claude-rules/= — the install-layout problem"). It misses these for two concrete reasons: it matches only markdown link syntax (=grep -oE '\[[^]]*\]\([^)]+\)'=), so org =[[file:...]]= links are invisible to it, and its driver loop only feeds it =claude-rules/*.md= and the language rule files, never =.ai/workflows/*.org=. Extending it on both axes is a smaller and more durable fix than a manual sweep.
+
+Found by the adversarial reviewer on the =references/= fix, 2026-07-28, as the sibling class the original report missed.
+
+** TODO [#C] start-work Phase 7 still summarizes the old publish flow :chore:solo:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-28
+:END:
+=.claude/commands/start-work.md:336-339= hands off with "Follow =commits.md= exactly" and "Run =/review-code --staged= before each commit" — no isolated reviewer, no re-review loop. It was already stale before the 2026-07-28 review change, because =commits.md= moved the publish flow into the =publish= skill in an earlier commit, so the pointer names a file that no longer holds the flow.
+
+Grading: Cosmetic severity (a stale summary beside a correct canonical, and start-work is attended so the escalation target exists) x most users frequently = P3 = [#C].
+
+Fix is to point Phase 7 at the =publish= skill rather than restate the flow, which is what let it drift in the first place. Worth a sweep for other files that restate the flow instead of pointing at it.
+
+Found by the adversarial reviewer during the 2026-07-28 review-flow change, and correctly filed rather than fixed there: the line was untouched by that diff.
+
+** TODO [#C] Two lint defects at the template source :chore:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-28
+:END:
+Both verified at the rulesets source, so every project seeded from them inherits the defect.
+
+=retrospectives/PRINCIPLES.org:38= violates the org-table standard (no closing rule). =lint-org= flags it as =org-table-standard=, and =wrap-org-table.el= reflows it mechanically. This half is purely tool-driven.
+
+=protocols.org= lints at 8 mechanical + 19 judgment =misplaced-heading= hits, all from Markdown =**bold**= in an org file, which org reads as a level-2 heading when it starts a line. 48 bold spans total, 14 of them line-initial. This half is not mechanical: converting to org =*bold*= is 48 edits, and =lint-org --fix= would rewrite the line-initial ones without knowing they are emphasis rather than headings.
+
+Grading: Cosmetic severity x every user every time = P3 = [#C]. It is most of the linter's noise on protocols.org, which is the real cost — noise that trains the reader to skip the report.
+
+Not =:solo:= — both files are synced templates.
+
+Source: winvm link-integrity pass, 2026-07-28.
+
+** DOING [#B] Finish context-engineering rightsizing :refactor:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-27
+:END:
+Started 2026-07-27 from three Anthropic posts Craig supplied. Always-loaded rules went from ~57,800 tokens to ~28,949, plus 13,461 path-scoped. Shipped: =paths:= frontmatter on the three file-type rules, the per-project generic-rule de-duplication, =commits.md= split into a 2,804-token invariant core plus the =publish= skill, =testing.md= split into a 347-word directive plus the =testing-standards= skill, the approval-gate signal fixed from =.ai/=-tracking to remote host, and the first-person directive.
+
+*What remains is Craig's decisions, not execution.* Each of these needs him:
+
+1. *C1 — =verification.md= (3,388 tok).* The Opus 5 guide says explicit verification instructions cause over-verification and should be removed. His standing direction is never guess, always check. My read: the honesty core (don't claim a green suite you didn't run) stays and shortens, the process injection (green baseline, suite-as-its-own-step) moves to the publish skill. His call — and it's the rule closest to a preference he's stated outright.
+2. *=interaction.md= (3,828 tok, now the largest).* Carries genuinely universal output constraints (no popup menus, no reverse-video markup) plus the new peer-reasoning contract. Splitting it means deciding which parts must fire on every turn.
+3. *The TDD rationalization table.* Moved to =testing-standards= rather than cut. The posts argue that kind of over-argument is counterproductive on current models. Deleting his defense against me skipping TDD is his call.
+4. *D3 — the gate separation.* Which approval gates are preference (he wants to see what goes out under his name) versus guardrail (written when the worst case was worse). They read identically in the files; only he can tell them apart. Highest-leverage input remaining.
+
+*Do first:* the three docs in =working/context-engineering-rightsizing/= are one commit behind — they were corrected at d74d98d, before the two splits and the gate fix. Reconcile before deciding anything from them.
+
+Risk on the record: =testing.md='s margin is thinner than =commits.md='s. If =testing-standards= fails to trigger mid-test-writing the mocking-boundary rules are lost — a quality regression, visible in review, but a real bet where =commits.md='s moved half was purely procedural.
+
+** TODO [#B] Recurring-loop mechanics as a shared rule :feature:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-27
+:END:
+Work proposes promoting the pattern behind its 2026-07-27 auto-mode triage-intake into a standing rule covering all recurring agent tasks: fixed interval via =CronCreate= rather than dynamic self-pacing, a subagent per firing so raw scan output never reaches the orchestrator, silence with no heartbeat when subagent-backed, and accumulate-don't-mutate between closes with explicit "close the X" / "stop the X" verbs. Likely touches =triage-intake.org= auto mode, =inbox.org= monitor mode, and a new short rule in =claude-rules/= so individual workflows point at one definition of cadence, isolation, and silence.
+
+Points 1, 2, and 4 mostly promote existing practice. Point 3 changes documented behavior and needs a real decision. Three findings from the skeptical review, all to resolve before this ships:
+
+1. *Point 1 is too absolute.* Fixed interval is the right default, but not the right universal. A loop waiting on unpredictable external state (a CI run, a deploy queue) should pace to how fast that state actually changes, which is what dynamic scheduling exists for. Write it as "default to a fixed interval; use dynamic pacing only when polling external state whose timing you can't predict."
+2. *Point 2 collides with a standing instruction.* Craig's harness prompt says not to spawn subagents unless he requested it. Making subagent-per-firing the standing pattern needs that reconciled explicitly — the honest reading is that asking for the loop *is* the request, and the rule should say so rather than leaving two instructions to fight. The isolation argument itself is sound and matches the Opus 5 guidance (delegate for genuinely independent, sizeable work).
+3. *Point 3 has a silent-failure hole.* Removing the heartbeat means a loop that died looks exactly like a loop quietly finding nothing. "The subagent completing is proof it ran" only holds if a *failed* subagent still surfaces. Either keep a rare heartbeat (hourly, not per-fire) or specify that failure always breaks silence. Suppressing success is fine; suppressing failure is not.
+
+Also underspecified: what counts as "signal" for a subagent-backed loop. And the silent-until-signal spec (=docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=, IMPLEMENTED) documents the per-fire heartbeat, so it needs a superseding history line rather than a silent contradiction.
+
+*** 2026-07-27 Mon @ 16:55 Work accepted all three findings and sharpened point 3
+
+Work agreed without pushback and corrected my either/or on point 3, which was too weak. Both mechanisms are needed, not one: a failed or hung subagent breaking silence covers a *scan-level* failure, but only a periodic heartbeat catches the *scheduler* dying — because in that case nothing spawns at all and there is no failure to report. Heartbeat rare (hourly), not per-fire.
+
+The live consequence makes it urgent rather than theoretical: =CronCreate= auto-expires recurring jobs after seven days, so any fully-silent loop is *guaranteed* to die quietly. Work's auto-triage loop is running under exactly that contract right now.
+
+Work also added the reason that makes point 2's carve-out principled and should go in the rule text: the subagent exists to keep N sources' worth of raw scan output out of the orchestrator across a multi-hour loop, not to parallelize.
+
+Craig's call on point 3 is pending; work is surfacing it and will send the answer.
+
+Source: work handoff 2026-07-27 14:51, reply 16:55.
+
+** TODO [#B] Sentry triage split — work mail and messengers in, personal mail out :bug:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-27
+:END:
+Craig's 2026-07-27 correction supersedes his 2026-07-21 ruling: sentry should scan DeepSat work mail and every active messenger source, excluding only personal email. =sentry.org= currently excludes all mail and all messengers (overview line and pass 3), so work's first 2026-07-27 fire reported a quiet pass while a Hayk DM and a Kostya PR-review request sat unread. Work's live at-prompt carries the override in the meantime; the canonical workflow, the source-activation probe, and the tests still encode the old policy.
+
+Grading: Major severity (the pass reports "nothing" while real work items sit — the failure is silent, which is what makes it worse than a visible miss) × most users, frequently (every fire in any project declaring a work-mail or messenger source) = P2 = =[#B]=.
+
+The blocker is not wording. The current rule excludes by *category* — "mail", "messenger" — and the new rule is a *work-vs-personal* split that category cannot express. Shipping general plugins are =cmail=, =personal-gmail=, =personal-calendar=, =telegram=, =github-prs=; there is no general work-mail plugin, so work's source is project-specific. Naming the personal plugins in a denylist fails open the moment a new personal source is added. Decide the classification mechanism first — the durable shape is a per-plugin eligibility declaration inside each plugin file, so a new plugin classifies itself and the probe reads the declaration rather than pattern-matching a name.
+
+Scope once decided: the =sentry.org= overview line and pass 3, the activation probe's "survives that exclusion" test, and the pass-3 tests.
+
+Source: work handoff 2026-07-27 06:21.
+
+** TODO [#B] Repository publish-lock for two sessions sharing one clone :feature:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-27
+:END:
+Craig approved this in archsetup on 2026-07-26 after weighing the competing approaches recorded there. Two agent sessions in one clone share the working tree *and* =.git/index=, so one can sweep the other's hunks into a commit; a pathspec commit doesn't help, because =git commit -- <path>= reads the current working-tree state for that path. The approved design: one repository-scoped publish lock keyed on the real Git common-directory path (so worktrees sharing a clone contend, and unrelated clones with the same basename don't), acquired before any publish-flow operation that can mutate the shared index and held across reconcile, staging, =/review-code --staged=, message approval, and =git commit=. Ownership tracks =AI_AGENT_ID=, then a stable harness id. The reviewed staged-tree fingerprint (=git write-tree= after the review) is re-compared immediately before commit; a lost lock, a vanished lock, or a changed fingerprint forces reacquire and a fresh staged review. Ordinary edits stay concurrent — the lock serializes only the publish mutation window.
+
+Skeptical review — the design is sound and the acceptance checks are testable. Three gaps to close during the build:
+
+1. *TTL sizing across a human wait.* The lock is held across the commit-message approval gate, which is an unbounded human pause. "Refresh on conversational re-entry" covers an agent that keeps working; it does not cover Craig stepping away for an hour mid-review. Size the staleness horizon for that, or refresh on a timer while the gate is open — sentry's TTL is nowhere near long enough.
+2. *Blocked-session behavior is unspecified.* The proposal says a second session cannot stage or commit through the guarded flow, but not what it should then do — wait on =--wait=, defer and report, or stop and surface. Pick one and test it.
+3. *Size.* This is a real build (key derivation, owner support in =agent-lock=, the fingerprint gate, the review-restart path, and the acceptance battery), not a wording change.
+
+Canonical touchpoints: =claude-rules/commits.md= (add the lock to the pre-commit flow; state explicitly that an approval waiver never waives the staged review, since the review is the gate that reads the hunks entering the commit), =claude-templates/.ai/scripts/agent-lock= (backward-compatible caller ownership — existing sentry and roam-write callers must stay green), and =claude-templates/.ai/scripts/tests/agent-lock.bats=.
+
+Non-goals from the approval: no per-session =GIT_INDEX_FILE=, no worktree requirement for live stow-managed dotfiles, no locking of ordinary edits, no replacement for remote pre-push reconciliation.
+
+Source: archsetup handoff 2026-07-26 10:55.
+
+** VERIFY [#B] Parked: reusable Claude-to-Codex MCP registry sync :spec:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-25
+:END:
+Two independent 2026-07-25 handoffs (work and home) report the same successful local migration: expose Claude's global =~/.claude.json= MCP registrations to Codex without duplicating secrets. Stdio definitions launch through a mode-0700 =claude-mcp-for-codex SERVER_NAME= wrapper that reads the mode-0600 Claude config at process start; Streamable HTTP entries remain native Codex entries; the loopback-only legacy Slack SSE endpoint bridges through =mcp-remote=. Both senders verified initialize + tools/list, not merely =codex mcp list=.
+
+Recommendation: accept the need, but spec it before adding an installer. "Every project" is misleading because these are machine-global registries; the reusable unit is a host-level, idempotent reconciler. It must:
+
+1. Preserve Codex-only and plugin-provided entries and distinguish active global registrations from inactive marketplace templates and broken project-local definitions.
+2. Inventory and report without printing environment values, secret-bearing arguments, tokens, or auth material.
+3. Back up and update atomically; preserve secure file modes; handle malformed input, name collisions, upstream removals, missing dependencies, and repeated runs.
+4. Classify stdio, Streamable HTTP, and legacy SSE explicitly. Never register SSE as HTTP, and allow cleartext only for loopback or an equivalently trusted private endpoint.
+5. Health-check each locally executable server with initialize + tools/list and report only redacted names/counts. Verify both machines and document restart/new-session plus OAuth reauthentication behavior.
+6. Keep the Figma token finding separate: rotate it and move it out of command-line arguments without ever recording its value.
+
+The proposed Claude-memory migration auditor is related only by runtime portability and should be a separate task/spec. It should classify guidance, useful context, duplicates, stale/ephemeral material, and secret-bearing denylist entries; it must not bulk-copy generated memory files.
+
+No prepared diff: this is a shared configuration design decision, and the current local wrappers/configs remain machine-owned evidence rather than canonical source. Say "spec the MCP registry sync" to start the decisions walk.
+
+** VERIFY [#B] A SCAN FAILED source must not advance its sentinel (engine, not the plugin)
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-24
+:END:
+Promoted to top-level 2026-07-28 when its parent closed — it is a separate engine question and would have been buried under a DONE parent.
+
+Secondary finding from the 2026-07-24 handoff, flagged for your judgment because it's *engine* behavior in =triage-intake.org=, not the telegram plugin. work reported that after a SCAN FAILED, "the marker still advanced" — and telegram is =:ANCHOR: none=, so something advanced a cursor it shouldn't have. The compounding harm: a source that reports SCAN FAILED but advances its window means the next sweep believes it already covered that window, so the blind-sweep hole persists across sweeps rather than self-healing. Not touched .emacs.d-side; needs a look at the engine's per-source last-run/anchor advance.
+
+** TODO [#B] cj-remove-block still can delete the WRONG cj block :bug:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-24
+:END:
+The 2026-07-24 range fix (17f5d48) closed the *over-deletion* case: a range spanning two blocks is now refused. It did NOT close the underlying class. The validator proves the range is a well-formed single block; it never proves it is the block that was *scanned*.
+
+An adversarial reviewer demonstrated this on the fixed code: passing the second block's range while intending the first deletes the second, exit 0, no warning. Drift by exactly one whole block still slips through — which is the failure the validator exists to prevent.
+
+Grading: Major severity (silent deletion of the wrong annotation in Craig's org files, no warning, success exit) x rare edge case (needs drift landing exactly on another well-formed block) = P3 = [#C]... graded up to [#B] because it is the *same* failure the fix was believed to have closed, so the current state carries false confidence.
+
+Fix direction needs a decision, which is why this is not :solo:. The range alone cannot identify the intended block, so the caller must assert something about content — a hash of the expected block, or the expected first body line, passed alongside the range and verified before deletion. That changes the CLI contract and the calling skill, so it is a design call.
+
+** TODO [#D] Atomic-write residuals in cj-remove-block :bug:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-24
+:END:
+Three narrower gaps a reviewer found in the atomic write, all verified, none fixed in the 2026-07-24 repair round:
+
+1. =shutil.copymode= runs before =write_text=, and writing clears setuid/setgid — a 2755 file comes back 755.
+2. ACLs are not carried across, and copying the ACL mask into the group bits can *widen* group permission on a file carrying a named ACL entry.
+3. Hard links are broken the same way symlinks were: =os.replace= gives the path a new inode, so a second hard link keeps stale content. Introduced by the atomic write itself (17f5d48), which did not exist on main.
+4. No =fsync= before =os.replace=, so the "never a partial" guarantee holds against process failure but not a system crash. Two lines to close.
+
+Grading: Minor severity (each needs an unusual file mode, an ACL, a hard link, or a crash mid-write) x rare edge case = P4 = [#D].
+
+** TODO [#D] Test suites leak backup files into shared /tmp :test:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-24
+:END:
+The elisp todo-cleanup suite writes roughly 81 backup files into the shared temp dir per =make test= run, named after randomized fixtures (=tc-test-XXXXXX.org.before-todo-cleanup.*=). 1598 had accumulated by 2026-07-24. Not data loss and not production-named, so nothing masquerades as a real backup and nothing is deleted — this is clutter that grows every run.
+
+The python sibling was fixed with an autouse fixture (f850ad6) that sets =TMPDIR= and overrides =tempfile.tempdir=. ERT has no autouse equivalent, so the elisp fix wants a load-time rebind of =temporary-file-directory= plus a =kill-emacs-hook= cleanup.
+
+Also unaddressed: =test-lint-org.el= scans a hardcoded =/tmp=, and =lint-org.el= hardcodes =/tmp/= in =lo--backup=, so that pair cannot be isolated the same way without touching the script. And =lint-org.el= carries the same second-resolution backup-overwrite flaw that todo-cleanup and cj-remove-block were fixed for.
+
+Grading: Minor severity (clutter in a directory cleared on reboot) x every user, every time (every suite run) = P2... graded [#D] on the no-harm read.
+
+** VERIFY [#B] Should rulesets run shellcheck on its own shell?
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-24
+:END:
+Surfaced by the lint-coverage finding above, and a bigger question than that task should decide.
+
+rulesets ships shellcheck enforcement to *other* projects — the bash bundle's =githooks/pre-commit= scans staged shell, and =validate-bash.sh= blocks an edit that fails it. rulesets runs neither on itself. Its own =githooks/pre-commit= calls =sync-check.sh= and nothing else; =lint.sh='s =check_hook= validates only shebang and exec bit; =make test= has no shellcheck step. So the repo asks consumers for a standard it does not apply to its own shell.
+
+Why this needs your call rather than an overnight fix: turning shellcheck on today would surface the findings I dispositioned as false positives during this session's sweeps — =SC2094= on =install-ai.sh= and =sweep-gitignore-tooling.sh= (append-with-stat, not a read-write race), =SC2088= on =doctor.sh= and =audit.sh= (tilde in a *display* string, not a path), =SC2164= on =lint.sh= and =status.sh=. Enabling the gate means either fixing those or adding =# shellcheck disable== directives with justifications, and which of those you want is a preference, not a fact.
+
+Options: (1) shellcheck in =make lint= as a warning, (2) in =githooks/pre-commit= as a hard gate matching what consumers get, (3) leave it, on the grounds that the repo's shell is small and reviewed. My lean is 2, since the asymmetry is the odd part — but it is your repo's bar to set.
+
+** TODO [#C] Attachment filenames from email are only partly sanitized :bug:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-24
+:END:
+Both attachment writers derive on-disk filenames from attacker-controlled input (the =filename= an email declares for its attachment), and both sanitize incompletely — in *different* ways, so neither covers the other's gaps.
+
+=eml-view-and-extract-attachments.py= runs the name through =_clean_for_filename= but interpolates the *extension* raw: =name, ext = os.path.splitext(...)= then =f"{basename}-ATTACH-{clean_name}{ext}"=. =gmail-fetch-attachments.py='s =safe_filename= handles path separators and leading =..= and nothing else.
+
+Verified 2026-07-24 by probing both with the same adversarial set:
+
+| input | eml-view result | gmail-fetch result |
+|------------------------+------------------------------------+--------------------|
+| =report.pdf; rm -rf ~= | keeps =; rm -rf ~= in the filename | keeps it |
+| =x.p\ndf= | keeps a literal newline | keeps it |
+| =a.= + 300 chars | 314-char filename | unbounded |
+| =../../../etc/passwd= | neutralized | neutralized |
+
+What this is NOT, checked so it isn't over-graded later: not remote code execution (files are written through Python =open=, never a shell) and not path traversal (=os.path.splitext= only returns an extension when the last dot follows the last separator, so =ext= can never contain a slash — the traversal case above is neutralized in both). The real harms are narrower: a newline in a filename breaks any downstream tool that reads the output directory as a newline-delimited list, and an unbounded extension exceeds the 255-byte filesystem limit so a crafted attachment aborts the extraction.
+
+Grading: Major severity (untrusted input reaches a filesystem name unsanitized, and the newline case breaks real tooling today) x rare edge case (needs an attachment name with unusual characters *after* the last dot, which ordinary mail doesn't produce) = P3 = [#C].
+
+Fix direction: sanitize the extension with the same rules as the name, cap the *whole* filename rather than just the stem, and reject or replace control characters in both scripts. The open question is where the sanitizer lives — see the VERIFY below.
+
+** VERIFY [#B] One sanitizer or two for the attachment filename fix?
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-24
+:END:
+Deferred from pass 12 on 2026-07-24: the bug above is well-specified, but *where* the shared sanitizer lives is a design call with real tradeoffs, and the unattended loop has no one to ask.
+
+The two scripts are standalone stdlib-only CLI tools with kebab-case names, so neither is importable as a normal module. Three options, none obviously right:
+
+1. *A shared helper module* in =.ai/scripts/=. Cleanest single source of truth, but it becomes another synced template file, and the kebab-named callers need =importlib= gymnastics to reach it — the same trick =route_recommend.py= already uses to load =inbox-send.py=, so there is precedent.
+2. *Duplicate a small sanitizer in each script.* Zero import machinery and each tool stays self-contained, which is the current house style for these scripts. Costs drift, and sibling drift is exactly the defect class that produced three separate findings this session.
+3. *Fix only the demonstrated gaps in place* (extension sanitizing in one, length and control chars in both) without unifying. Smallest diff, leaves the two sanitizers permanently different.
+
+I lean 1 given how much drift has bitten tonight, but it changes the shape of =.ai/scripts/= and that is Craig's call, not an overnight one.
+
+** VERIFY [#B] Parked: question-capture pattern — ask async, answer live (from archsetup, Craig's idea)
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-24
+:END:
+What arrived: your own idea, relayed via archsetup's roam inbox. Drop a question (not a build task) into the roam inbox as a capture; the agent retrieves it during a sentry/inbox pass but does NOT auto-answer; it holds the question and answers it at the next live conversation, closing on your acknowledgement. Distinct from VERIFY: a VERIFY blocks the agent on your input, here the agent owes you the answer and you owe only the ack. The instance that spawned it: "why does the cursor not appear over the desktop when the world-clock wallpaper is on?", answered live in archsetup.
+
+Recommendation: adopt, as a small spec rather than a direct build. The value is real — a captured question has no home today (not a task, not quite a VERIFY), and decoupling async-capture from sync-answer stops the agent burning cycles guessing at something you'd rather discuss. It clears the value gate on your authorship. But the shape carries decisions I shouldn't guess, which is why it parks rather than lands:
+
+1. How is a question item identified — an explicit marker (a =:question:= filetag, a =Q:= prefix) or phrasing (ends in "?")? Phrasing is fuzzy (a real task can contain a question); a marker is reliable. My lean: explicit marker.
+2. Where does the "answer owed" state live between capture and the live session? Options: the roam item stays put with an answered-pending tag, or it routes to the owning project as an "answer owed" item that startup surfaces. My lean: surface at startup, since that's the next live moment.
+3. Does it warrant a new keyword (the inverse of VERIFY — agent-owes-answer) or a tagged VERIFY variant? The proposal draws the VERIFY distinction sharply enough that a distinct marker may be cleaner.
+
+Prepared artifact: none — this is a rule-design decision, not a mechanical edit, so there's no diff to stage. The proposal and the concrete instance are in [[file:working/question-capture-pattern/proposal-from-archsetup.org]].
+Say "spec the question-capture pattern" (answering 1-3, or leaving them to the spec's decisions walk) and it becomes a spec-create.
+
+** VERIFY [#A] Parked: account-binding guard for the personal-gmail triage plugin (from home)
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-23
+:END:
+What arrived: home hit a real cross-account hazard on 2026-07-23 — a sentry triage fire used =mcp__claude_ai_Gmail= (bound to the DeepSat work account, name gives no hint) instead of =google-docs-personal=, pulled 201 unread work messages, and would have run every hygiene action against the wrong inbox. The fix adds one guard block to the =triage-intake.personal-gmail.org= Scan phase: confirm a sample result's =to:= is =craigmartinjennings@gmail.com= before classifying, and on mismatch or unavailability fall back to the local mu mirror rather than another MCP.
+
+Recommendation: accept as-is. The diff is a single clean addition between the promotions-filter warning and the maxResults-cap warning — nothing else in the plugin changes. The bug is reproduced and dated, the account→tool mapping was verified by home 2026-07-23 (google-docs-personal=personal, google-docs-work + claude_ai_Gmail=work-bound), and the mu-mirror fallback is the right escape (it's account-fixed by maildir, not by an ambiguous MCP name). I agree with home's judgment that cmail needs no analogous note — it's a bridge script fixed to c@cjennings.net by construction, no tool-binding ambiguity.
+
+Graded [#A] on severity alone, per the todo-format security/privacy carve-out: acting on the wrong person's mailbox is a privacy exposure, and one occurrence with work mail misrouted through a personal-hygiene close is a showstopper regardless of frequency.
+
+Prepared diff: [[file:working/triage-account-guard/proposed.diff]] — apply is mechanical (home's file becomes canonical) on your go. Companion note: [[file:working/triage-account-guard/companion-note-from-home.org]].
+Say "approve the parked account guard" (or adjust / reject) and it gets applied.
+
+** VERIFY [#B] Parked: term-translation-density voice pattern #48 (from work)
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-23
+:END:
+What arrived: you proposed a new voice pattern in the work session — a sentence forcing the reader to translate more than one specialized term is too dense and gets rewritten. You supplied the rule text and a worked before/after (the ViT/VLM/detect-then-contextualize email sentence), and delegated number, placement, and mode-tag to me.
+
+Recommendation: accept, as #48, general mode. Your reasoning holds — it's a universal clarity rule in the Orwell/Plain English family, distinct from #7 (which words) and #30 (fragments), so it reads for anyone's prose, not just Craig's. The diff is prepared and verified: both files consistent, #47 intact, all mode enumerations and counts updated.
+
+The one wrinkle I resolved, worth a glance before you approve: general mode was defined as exactly "patterns #1-31," a contiguous block. A general pattern at #48 breaks that, so the prepared diff updates every "general walks #1-31" line to "#1-31 and #48" and notes the number is an artifact of when it was added, not a scope signal. The alternative was to scope it [prose · personal] to keep general clean, but that costs the pattern its reach over third-party prose, which is exactly where jargon density bites. I went with your general-mode call.
+
+Prepared diffs: [[file:working/voice-term-density/skill.diff]] and [[file:working/voice-term-density/profile.diff]] — apply is mechanical on your go.
+Say "approve the parked term-density pattern" (or adjust / reject) and it gets applied.
+
+** VERIFY [#B] Parked: auto-empty working/ when a task closes (from work)
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-23
+:END:
+What arrived: your roam capture relayed via work — every project carries working/ and temp/, and when a task completes its working/ files are automatically ("via a soft hook if possible") archived or moved to temp/. The sender flagged that most of this may already exist and asked for a check before treating it as net-new.
+
+Recommendation: accept the intent, reject the mechanism. Three of the four asks already shipped — working/ tracked from creation and temp/ gitignored (2026-07-20), and temp/ cleared at wrap (today, 10ea44b, from this same capture). Only the auto-empty is new, and I'd not build it as described.
+
+Two reasons. Moving a closed task's artifacts to temp/ destroys them, because temp/ is cleared at the next wrap, while working-files.md says those artifacts get renamed individually and filed flat into assets/ with meaningful names. And filing is deliberately a judgment step: it forces a review of each artifact's permanent value, which is what keeps assets/ searchable and catches the ones not worth keeping. Automating it removes exactly that review. Tonight is the live case — I closed four parked VERIFY tasks and filed their working/ dirs by hand, checking for inbound links and confirming the content stayed recoverable in git first. An automatic sweep would have destroyed four prepared diffs at the next wrap.
+
+The real gap is narrower: orphan detection exists only inside a sentry fire (pass 6), so a project that never runs sentry never gets the nudge. The prepared diff adds that detection to wrap-up as a report-only step, which serves the intent without the destruction. Verified against this repo's live tree.
+
+Prepared diff: [[file:working/working-dir-orphan-check/proposed.diff]] — apply is mechanical on your go.
+Say "approve the parked orphan check" (or adjust / reject) and it gets applied.
+
+** TODO [#B] No two language bundles compose any more — polyglot is now refused :feature:spec:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-23
+:END:
+Completing the python and typescript bundles (the [#A] above, fixed 2026-07-23) had a consequence worth deciding on deliberately: all five bundles now ship =claude/settings.json= and =githooks/pre-commit=, so *every* pair collides and =install-lang= refuses the second without =FORCE=1=.
+
+This reopens a decision that was closed on a premise that turned out to be a bug. On 2026-07-17 the call was "polyglot: case-by-case, no option-2 machinery — bundles already compose; only coverage-makefile.txt collides, and it's a manual paste." Bundles appeared to compose only because python and typescript were missing the two files everything else collides on. The composition was the defect, not a design property.
+
+Live impact: =clock-panel= is a python + typescript project. It can install one bundle's hooks or the other's, not both. Today it has neither, so nothing regressed underneath it, but the polyglot path it would have wanted is now closed. =work= is python-only and unaffected.
+
+The real question is what a polyglot project should get. Both files are single-owner by construction: =settings.json= would need its =PostToolUse= arrays merged, and =pre-commit= would need each language's checks concatenated behind one secret scan (which is already identical across all five). Neither merge is hard; the reason it was never built is that nobody had a project that needed it. Now one does.
+
+Options, roughly: (1) build the merge — settings arrays union, pre-commit composes per-language phases; (2) split the shared half out, so the secret scan is one file every bundle sources and only the language phase is per-bundle; (3) leave =FORCE=1= as the polyglot answer and document that the last install wins; (4) do nothing, since only one project is affected. Option 2 is the one that would also have prevented the [#A] — a shared secret scan can't go missing from a bundle that never had its own copy.
+
+Not [#A] because nothing is currently broken: no project is running with a bundle it lost. Not [#D] because a live project wants it and the decision is now forced rather than hypothetical.
+
+Pinned by =scripts/tests/install-lang-collision.bats= so the trade can't drift back unnoticed.
+
+** TODO [#C] Two Signal-channel gaps: unbounded send, second account never received :bug:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-23
+:END:
+Both found 2026-07-23 reading =claude-templates/bin/agent-text= and =scripts/signal-receive.sh=. Shellcheck is clean on all four =claude-templates/bin/= scripts; these are logic gaps, not style.
+
+1. *The direct send has no timeout.* =agent-text='s relay path passes =-o ConnectTimeout=10= to ssh, but the local-account path calls =signal-cli send= with no bound at all. A stalled network or a lock contended by the receive timer hangs the call indefinitely. That matters more than it looks: agent-text is invoked *by agents*, so an unbounded call blocks the calling turn with no output, and the caller can't distinguish a hang from a slow send. The receive path already takes ~16s of wall clock every cadence, so the two contending on the same account is not hypothetical. Fix: wrap the direct send in =timeout=, matching the relay's bound, and let the existing non-zero path print the desktop-fallback message.
+
+2. *ratio's second Signal account is never received.* =signal-cli listAccounts= on ratio warns "Messages have been last received 8 days ago". Established by elimination, not inference: ratio holds two accounts (=+15045173983= pager, =+15103169357= Craig's personal), =signal-receive.sh= hardcodes the pager as its default and takes no others, and the timer's journal shows it draining the pager cleanly every 15 minutes (last run 2 minutes before the warning appeared). So the 8-day-stale account is the personal one, and nothing on the machine keeps it warm.
+
+ Honest limit on item 2: the staleness is verified, the *harm* is not. That registration is described in the session history as note-to-self with no push, so it may be fine to leave cold, and Signal's own tolerance for a quiet linked account isn't something I established. Filed so the question gets answered rather than rediscovered. The script already accepts an account argument, so if it does matter the fix is a second timer instance rather than a code change — which makes it a dotfiles handoff (that repo owns the unit) with rulesets owning only the script.
+
+Grading: Minor severity for both (one is a latent hang in a tool with a documented fallback, the other a warning on an account nothing currently depends on) x most users, frequently (the stale warning is a standing condition on this host; the hang needs a stall) = P3 = [#C].
+
+** TODO Manual testing and validation
+*** Sentry — entry gates fire with Craig present
+What we're verifying: the interactive entry gates stop for the right states and start the loop only on a clean, green baseline.
+- On rulesets (ratio), with a clean tree and green suite, say "start sentry hourly".
+- Confirm it reads =:COMMIT_AUTONOMY: yes= and doesn't decline.
+#+begin_src sh :results output
+# Dirty the tree on purpose, then re-arm — the dirty-tree gate should stop and offer finish/stash/rollback.
+echo "# scratch" >> README.org
+git diff --stat
+#+end_src
+- Re-run "start sentry"; confirm it surfaces the dirty README and the three numbered options rather than starting.
+- Discard the scratch edit (=git checkout -- README.org=), re-arm, and confirm it reconciles, creates =sentry/$(date +%F)-$(uname -n)=, and arms the loop.
+Expected: sentry declines on the dirty tree with numbered options, and on a clean+green tree it creates the host-suffixed branch and starts the hourly loop.
+*** 2026-07-20 Mon @ 11:22:56 -0500 Sentry — one fire runs end to end, verified in live trial
+Verified across the 8-fire live trial on =sentry/2026-07-19-ratio= (ratio, 2026-07-20). Each fire wrote one digest block with a ran-or-skipped line per pass (no silent gaps), parked judgment/destructive items under the =* Sentry approval queue= heading with their exact commands (never executed unattended), committed writing passes to the branch, left the tracked tree clean, and freed the single-runner lock between fires. Fires 1-2 productive (2 handoffs processed + 1 parked, 3 design considerations filed); fires 3-8 clean idle. Digest lives in the session anchor.
+*** Sentry — wrap-up guard and stop-sentry
+What we're verifying: wrap-up refuses while sentry is live, and stop-sentry cleanly ends the loop.
+- With sentry live, say "wrap it up"; confirm it refuses with "sentry is active — say 'stop sentry' first."
+- Say "stop sentry"; confirm the loop cancels and it offers the branch disposition (merge now / leave named) and the approval-queue walk.
+Expected: wrap-up blocks on the held single-runner lock; stop-sentry cancels the loop and walks branch + queue disposition.
+*** Sentry — morning branch review
+What we're verifying: the morning teardown reviews and merges cleanly, nothing reached main unpushed.
+- =git log main..sentry/<date>-<host>= and read the diff.
+- Squash-merge what's wanted, delete the branch, revert any stale Emacs buffers.
+- Confirm main was never pushed to during the night and the approval queue was walked.
+Expected: the night's work is reviewable as one branch, merges by choice, and main stayed clean throughout.
+*** Silent-until-signal — empty fire heartbeats, real item full turn
+What we're verifying: an in-session monitor fire collapses to a one-line heartbeat when nothing changed, and still does the full turn on a real item. Covers sentry, auto triage-intake, and auto inbox-zero (spec: docs/specs/2026-07-20-silent-until-signal-monitors-spec.org).
+- Run a sentry fire with nothing pending. Expected: one line, =sentry at HH:MM: nothing=, no per-pass digest block, tree clean.
+- Run a sentry fire after planting an inbox handoff. Expected: the full per-pass digest block, the handoff processed or queued.
+- Start "auto triage-intake" and let an empty sweep run. Expected: one line, =triage intake at HH:MM: nothing=, no three-section output.
+- Plant a real change in a triage source, let the next sweep run. Expected: the full three-section output (deltas / unacked / timestamp).
+- Start "auto inbox zero" with an empty roam inbox, let a cycle run. Expected: one line, =inbox zero at HH:MM: nothing=.
+- Add a roam inbox item, let the next cycle run. Expected: the item summarized, filed, and queued.
+Expected: every empty fire is a single labelled heartbeat; every fire with real work does its full turn unchanged.
+*** Triage source activation — declared sources pulled, undeclared dormant
+What we're verifying: triage-intake pulls only the sources a project declares (spec: docs/specs/2026-07-20-triage-source-activation-spec.org).
+- In rulesets (no =:TRIAGE_SOURCES:=, no project plugin), run "triage intake". Expected: Phase 0 announces every general plugin as "inactive (undeclared)", loads zero sources, and the sweep no-ops.
+- In a project with =:TRIAGE_SOURCES: personal-gmail cmail=, run "triage intake". Expected: Phase 0 loads personal-gmail and cmail as active general sources, names the rest inactive, and scans only the two.
+- In a project with a project-specific plugin (=.ai/project-workflows/triage-intake.*.org=) and no declaration, run "triage intake". Expected: the project plugin is active by presence; general plugins stay inactive.
+- Under sentry, confirm pass 3 (triage) probe-skips in a project with no active source, and runs where one is declared.
+Expected: undeclared general plugins never scan; declared ones and project-specific ones do; sentry's triage pass activates only where a source is active.
+
+** TODO [#D] Document polyglot coverage-makefile namespacing :chore:
+The one manual step the polyglot decision (2026-07-20) left: when a project installs two bundles that both ship =coverage-makefile.txt= (elisp/go/python/typescript), their =coverage:= / =coverage-summary:= Makefile targets duplicate. Document the fix — rename to per-language =coverage-<lang>:= targets and add a =coverage:= aggregate that depends on them — where a polyglot user would meet it (install-lang docs, or a short note in the bundle README / languages/ overview). Low urgency: clock-panel is polyglot today and unaffected because it has no assembled Makefile. Do it when the coverage-makefile fragments are next actually pasted, or opportunistically.
+
+** TODO [#D] Cross-host agent coordination — a lock across ratio and velox :feature:
+Foundational primitive several deferred features wait on. =agent-lock= serializes agents on ONE host (its locks live on tmpfs, =$XDG_RUNTIME_DIR/agent-locks/=), so nothing coordinates the two daily drivers. When ratio and velox both act on shared state at once, there's no lock to stop them — the roam-write lock and sentry's single-runner lock are all host-local. Design a cross-machine lock (a lock node committed to a shared repo, a tailnet lock service, or a designated-primary election) that the host-local locks escalate to when the operation touches cross-machine-shared state (roam, a sibling-freshness push). Take up when cross-host contention proves real in practice, not speculatively.
+
+Blocks (the shared-prerequisite half of the sentry cluster):
+- [[file:todo.org::*Cross-host roam conflict surfacing][Cross-host roam conflict surfacing]].
+- [[file:todo.org::*Sentry vNext passes][Sentry vNext passes]] items #2 (system-health) and #3 (sibling-freshness) — both need ratio/velox not to run the pass simultaneously.
+
+** TODO [#D] Cross-host roam conflict surfacing :feature:
+Deferred from the sentry spec (Decision: host-local lock only). The
+roam-write lock serializes same-host agents; a true cross-host race still
+forks a sync-conflict file, surfaced only by roam-sync's rebase abort.
+Investigate making the conflict loud (roam-sync retry-once, or a
+cross-host lock node in the repo) when it proves frequent in practice.
+Shared prerequisite: [[file:todo.org::*Cross-host agent coordination][Cross-host agent coordination]] (the missing cross-machine lock).
+
+** TODO [#D] KB lesson-detection heuristic :spec:
+Deferred from the sentry spec review (2026-07-14): sentry's KB promotion pass was cut from v1 because "promote recent durable lessons" had no defined lesson source, and an unattended judgment pass writing to the shared KB needs one that doesn't spam or go silent. Design the detection heuristic — candidate source: Session Log entries and memory-dir changes since the last fire, judged against knowledge-base.md's inclusion bar — and define how precision gets verified before the pass ships in a sentry vNext. Blocks re-adding the pass.
+
+** TODO [#D] Unattended /schedule cron contract — no-session variant :feature:
+The design problem shared by two deferred features: a =/schedule= cron pass that fires with *no live session* needs its own contract, distinct from the interactive =/loop= shape. The open questions are the same for both consumers — mutation rights (read-only vs may-mutate =todo.org= / =~/org/roam/inbox.org=), how a find surfaces asynchronously when Craig isn't at the session, how dedup state persists across runs that don't share a session, and what session/auth context a cron run carries (the MCP-auth wall: a detached run can't reach Gmail/Slack/Linear, the same constraint the silent-until-signal spec turned on).
+
+Two consumers, one contract:
+- *Sentry* (deferred from the sentry spec Non-Goal): v1 is interactive, a live session Craig starts, /loop-driven. A cron variant firing with no session is out of v1 scope.
+- *Auto inbox zero* (vNext from the inbox-consolidation spec, Codex finding 1; [[id:a7fe2a10-dfa8-4ba3-a11a-e7b1288b7573][spec]]): v1 =auto inbox zero= is the interactive /loop check that waits for Craig's yes. Its "design after v1 consolidation lands" precondition cleared 2026-06-28 (the inbox engine consolidation 24ca58d and monitor-inbox loop edb545d shipped), so this is actionable backlog.
+
+Design the one contract; both features consume it. Merged 2026-07-20 from the separate "Sentry unattended /schedule variant" and "Fully-unattended scheduled inbox check" tasks — same no-session design problem.
+
+** TODO [#D] Research: MCP for device locations shared with you :feature:
+From .emacs.d (2026-07-20, rulesets-owned research). Is there a Google Maps MCP (or similar) that reports the locations of devices sharing their location with you? If none exists, research how hard it would be to build one. (Google's location-sharing has no official public API; likely needs investigation of unofficial routes or a different provider.)
+
+** TODO [#C] Sentry vNext passes — from live-trial design input :feature:spec:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-25
+:END:
+Three considerations captured by .emacs.d's own hand-run sentry trial (routed via the shared roam inbox, 2026-07-20), for folding into the sentry workflow. Design input, needs deliberation — not applied to the shared workflow unattended.
+
+1. Bug/enhancement logging pass. Only where the project owns a codebase: file bugs by default, enhancements only if asked; review-and-accept/decline during the morning reconciliation. Live-tested — .emacs.d ran exactly this by hand and logged findings to its session anchor. Strongest candidate of the three; overlaps the deferred KB lesson-detection pass ([[file:todo.org::*KB lesson-detection heuristic][KB lesson-detection heuristic]]).
+2. System-health pass. Decide which checks are safe unattended, and how to keep ratio and velox from running it simultaneously (a cross-driver lock/coordination question — agent-lock is host-local tmpfs, so it doesn't coordinate across machines).
+3. Sibling-machine freshness pass. Keep the other daily driver (velox/ratio) up to date during sentry (see daily-drivers.md; same cross-driver coordination question as #2).
+
+Take up with the sentry Living Document refinements once the trial has quiet nights behind it. Passes #2 and #3 are blocked on [[file:todo.org::*Cross-host agent coordination][Cross-host agent coordination]] (the missing ratio↔velox lock). Related: [[file:todo.org::*Unattended /schedule cron contract][Unattended /schedule cron contract]] (the no-session variant).
+
+** TODO [#B] Extend ui-prototyping rule with build-to-prototype :feature:
:PROPERTIES:
-:CREATED: [2026-06-13 Sat]
-:LAST_REVIEWED: 2026-06-13
+:CREATED: [2026-07-11 Sat]
+:LAST_REVIEWED: 2026-07-25
:END:
-Optional wrap-up step that surfaces filed keepers belonging to another project, recommends a destination, and batch-moves them into that project's =todo.org= Open Work section (transcript filing deferred to vNext). All six decisions resolved (Reading B: the router acts on session-filed keepers, separate from the inbox gate and from defer-and-stage). Spec ready for review.
+.emacs.d proposal (2026-07-11, Craig-approved promotion), extending the ui-prototyping rule shipped tonight (=claude-rules/ui-prototyping.md=, 53f6ce6). The base rule covers research → prototypes → iterate → decisions-backed-by-a-prototype, but not the "build to the prototype" half. Add: after prototyping, fold what settled back into the spec and make the *build target* the prototype rather than the original spec text — the built feature should match the prototype, and any deviation is documented in a "Prototype & deviations" addendum section the build keeps current. Wire into four touch points: =brainstorm= Phase 3 (a UI design isn't "accepted" until it's been through the prototype loop; Next Steps say "build to the prototype"), =spec-create= (emit the deviations-addendum section for UI specs), =spec-response= (a UI spec decomposes into a prototype loop first, then build-to-prototype tasks), =start-work= (its verify phase drives the UI end-to-end; the bar becomes "matches the prototype," deviations logged). Worked example: the takuzu Emacs game — colored tiles read as all-black in the real terminal frame (dark faces + GUI-only box cursor), caught by a prototype loop on the first screenshot where build-to-spec would have shipped an unusable board. Multi-asset synced-rule change, so review-gated and needs a focused session; decide brainstorm-vs-lifecycle placement with Craig. Source: =inbox/PROCESSED-2026-07-11-0222-from-.emacs.d-ui-prototype-rule-proposal.org=.
-Spec: [[file:docs/design/wrapup-routing-spec.org]]. Source proposal: [[file:docs/design/2026-06-13-wrapup-inbox-transcript-routing-proposal.org]] (archsetup handoff 2026-06-13). Next: =spec-review=.
+** TODO [#C] Roam-only startup for .ai projects — investigate :spec:
+:PROPERTIES:
+:CREATED: [2026-07-11 Sat]
+:LAST_REVIEWED: 2026-07-25
+:END:
+Open question from the roam inbox (2026-07-11): could the startup sequence for .ai projects use org-roam as the single store instead of local files? Potential gain is near-guaranteed shared information across projects — lessons and proven techniques on a common thread, scannable across similar projects. It would need a way to isolate a project from the rest. Weigh the pros/cons, the risk, and whether it's worth it before any build. Exploratory, no commitment yet.
-** TODO [#B] Helper-instance support — concurrent same-project Claude :feature:spec:
+** TODO [#C] Keep WIP from blocking the template sync gate :feature:
+:PROPERTIES:
+:CREATED: [2026-07-11 Sat]
+:LAST_REVIEWED: 2026-07-25
+:END:
+From the roam inbox (2026-07-11): work in progress in one project shouldn't stop the sync gate. Idea: keep all diffs/changes in a =working/= directory and exclude it (and its subdirectories) from the sync gate. Many projects run at once, so their WIP files need to be grouped. Also add a per-project count of when the gate tripped, tracked as a metric to investigate. Distinct from the 2026-07-02 policy (untracked/gitignored changes already pass — this is about *tracked* WIP under =working/=). Verify how the gate detects dirtiness today before designing.
+
+** TODO [#C] KB orphan-node review pass :chore:
+:PROPERTIES:
+:CREATED: [2026-07-01 Wed]
+:LAST_REVIEWED: 2026-07-25
+:END:
+The 2026-07-01 kb-hygiene report listed 42 agent KB nodes with no inbound id: links (of 53 agent nodes; 0 conflicts, no duplicate titles). Orphan-ness alone isn't a defect — agent nodes are found by rg, not only by links — but a periodic pass is worth doing: prune nodes that aged out, merge near-duplicates, add id: links where clusters exist. Regenerate the list with the kb-hygiene script rather than trusting the snapshot. Propose deletions/merges to Craig before applying (auto-cleanup allowed only for :agent:-tagged nodes after approval, per knowledge-base.md).
+
+** TODO [#B] Helper-agent instance support — concurrent same-project Claude :feature:spec:
:PROPERTIES:
:CREATED: [2026-06-11 Thu]
-:LAST_REVIEWED: 2026-06-12
+:LAST_REVIEWED: 2026-07-25
:END:
SPEC REVIEWED 2026-06-12: [[file:docs/design/2026-05-28-generic-agent-runtime-spec-review.org][Codex review]] now rates Phase 1.5 =Ready with caveats=. Before any build, keep the Emacs integration as a cross-project handoff to =~/.emacs.d=, preserve the three-ring gate (bats → sandbox drills → pilot project), and do not let startup/helper changes reach synced template paths until the live drills pass.
@@ -64,111 +563,217 @@ Implement Phase 1.5 of the generic-agent-runtime spec ([[file:docs/design/2026-0
Independent of the spec's phases 2-6 (runtime-neutral refactor), which stay gated on their own go/no-go.
-** DOING [#C] Check that memories are sync'd across machines via git :spec:
+*** 2026-06-15 progress — detection + contract landed (inert); wiring is what's left
+RESUME NOTE. Before picking this back up as the big item, re-read what it is, does, and why first — don't dive straight into the next slice. The orientation, in one breath: this lets Craig open a second Claude in the *same project* while a primary session is running (a second terminal, to look something up or make a small task edit) without the two sessions clobbering each other's files. The whole risk it manages is two Edit-tool writers on one shared file (todo.org, notes.org) losing each other's writes, last-write-wins, silently. The design's answer: a singleton primary keeps the unisolated session-context.org; helpers get their own session-context.d/<id>.org, make only scoped single-heading edits to shared files, and never touch git. Full why-and-how in the spec ([[file:docs/design/2026-05-28-generic-agent-runtime-spec.org]], "Concurrent same-project agents" amendment) and the contract ([[file:.ai/workflows/helper-mode.org]]). A helper is NOT a subagent — subagents are for bounded dispatched lookups; this is for interactive parallel work Craig drives himself.
+
+Done so far (shipped, pushed, inert until wiring routes to them):
+- agent-roster (commit f8bdf30): the detection primitive. pgrep -x claude → /proc cwd → keep in-project → drop own ancestry. Exit 0 alone / 1 others / 2 unavailable. Injectable boundary (ROSTER_PGREP/PROC/SELF_PID), 11 bats, live-verified against 4 real sessions. [[file:.ai/scripts/agent-roster]].
+- helper-mode.org (commit 0b681dc): the canonical contract — read/write tiers, four data-integrity rules, light startup, helper wrap-up. Triggerless INDEX entry, protocols.org pointer.
+- Already shipped earlier: the AI_AGENT_ID + session-context.d/ split, and (2026-06-14) the epoch-on-the-tail id convention.
+
+What's next — the WIRING, all behind the spec's three-ring gate (bats → sandbox drills → live pilot), none of it sync'd to live template paths until the two-session drill passes:
+- Startup roster-detection branch: roster runs first; not-alone routes to helper-mode.org, alone keeps crashed-vs-fresh anchor logic. (Edits startup.org — synced, gated.)
+- wrap-it-up.org helper branch (archive own file, skip hygiene+commit; orphaned-helper lifts the git ban).
+- ai --helper launcher: roster → assign+export id → launch with helper opener. Plus the Emacs surface (ai-term.el) via an .emacs.d cross-project handoff.
+- Hygiene-pass live-helper gate: todo-cleanup.el / lint-org.el / wrap-org-table.el check session-context.d/ and pause+ask on live helper files.
+- todo-cleanup.el backup-to-/tmp backstop (lint-org and wrap-org-table already conform).
+- Manual validation drills with Craig (the live two-session test + the corruption drill).
+
+Stand up a drill rig before the gated work; build against it, don't touch synced paths until the live drill passes.
+
+DEPENDENCY QUESTION (Craig, 2026-06-15, resolved 2026-06-24 — see below): doesn't helper-instance support depend on generic agent runtime support? Starting point: the spec frames this work as Phase 1.5, "Independent of the spec's phases 2-6 (runtime-neutral refactor), which stay gated on their own go/no-go," and the body claims it sits only on the already-shipped session-context split. The separate =Generic agent runtime support — Codex spec v0= task (#C, below) is that phases-2-6 arc. So the spec's stated answer is "no, 1.5 is independent" — but confirm that's actually true for every wiring slice (does ai --helper, the roster branch, or helper-mode routing secretly assume any runtime-manifest / multi-runtime machinery from 2-6?), or whether helper-instance should be sequenced after, or merged into, the generic-runtime task.
+
+*** 2026-06-24 Wed @ 00:30:32 -0400 RESOLVED — independent, unblocked (keyword VERIFY → TODO)
+Craig's call (2026-06-24): helper-instance is independent of the generic-runtime refactor and builds on its own. The shipped pieces and the remaining wiring are all shared-file concurrency-safety (two Edit writers, one file), orthogonal to which LLM runtime launches — none of it assumes the runtime-manifest / multi-runtime machinery of phases 2-6. One caveat: =ai --helper= overlaps the launcher refactor the generic-runtime arc plans, so keep that launcher change small and contained so the later refactor doesn't fight it. Now a buildable [#B] task behind its own three-ring gate (bats → sandbox drills → live pilot).
+
+** TODO [#B] Wrap-up routing — manual end-to-end validation :test:
:PROPERTIES:
-:LAST_REVIEWED: 2026-06-12
+:LAST_REVIEWED: 2026-07-25
:END:
-v1 implemented end-to-end 2026-06-10 (Phases 0-4 below, no-approvals batch). Remaining before DONE: the manual testing and validation child, plus the other personal machines' one-time clone + timer setup (archsetup handoff).
-*** 2026-05-14 Thu @ 19:14:11 -0500 Investigate current memory storage
+What we're verifying: a real keeper routes through a live wrap and the destination actually files it. The task-routing build shipped IMPLEMENTED 2026-07-04 (spec [[id:00b47414-2213-4a99-be35-48ceb266fc08][wrapup-routing]]); this confirms it works end to end across a real cross-project wrap. A failed check promotes to a bug.
+- In a project session, let process-inbox file a handoff whose home is a different project; confirm the local task carries =:ROUTE_CANDIDATE: <dest>=.
+- Run wrap-it-up; at the router sub-step, confirm the candidate is surfaced with the right destination + confidence, then choose "go".
+- Confirm a =from-<thisproject>= handoff landed in the destination's =inbox/= and the keeper was removed from the local =todo.org=.
+- Open the destination project; confirm its startup/process-inbox files the handoff into its =todo.org= per its own conventions.
+Expected: the task ends up in the destination's =todo.org=, gone from the source, with no foreign =todo.org= written directly. Not =:solo:= — needs a real cross-project wrap and the destination's next session.
-Memory files live at
-[[file:/home/cjennings/.claude/projects/-home-cjennings-code-rulesets/memory/][~/.claude/projects/-home-cjennings-code-rulesets/memory/]]
-— four files including =MEMORY.md= and three individual entries
-(=feedback_never_guess.md=, =project_ai_scripts_canonical_source.md=,
-=reference_pdftools_venv.md=). The directory is a plain unmanaged dir
-(no symlink, no enclosing git checkout). Neither
-[[file:/home/cjennings/.claude/][~/.claude/]] itself nor any subtree
-containing the project-memory dirs is tracked in
-[[file:/home/cjennings/code/archsetup/][archsetup]] or
-[[file:/home/cjennings/code/rulesets/][rulesets]]. Without a symlink
-into a stowed or tracked location, memory files don't survive a new
-machine setup or a dotfiles restore.
+** TODO [#D] Wrap-up routing — transcript filing (vNext) :feature:no-sync:
+File a meeting recording into the destination =assets/= per =working-files.md=, batch go/skip mirroring the task router. Gated on the source-location decision (spec D4). Spec: [[id:00b47414-2213-4a99-be35-48ceb266fc08][wrapup-routing-spec.org]] (Phase 5).
-Proposed setup: stow =~/.claude/projects= →
-=archsetup/dotfiles/common/.claude/projects/= (path doesn't exist yet
-— it's the target location pending VERIFY).
-Create the destination in archsetup, move existing per-project
-=projects/<encoded-cwd>/memory/= dirs there, run =stow= to link, then
-commit + push archsetup. After that, every machine running =stow=
-picks up the same memory tree.
+** TODO [#C] Multiple agent-source improvements :spec:
+:PROPERTIES:
+:CREATED: [2026-06-23 Tue]
+:LAST_REVIEWED: 2026-07-13
+:END:
+Make the tooling agent-agnostic instead of Claude-specific. Three threads from Craig (roam 2026-06-23): (1) give the agent a name so workflows don't say "Claude" everywhere — a non-Claude agent (Codex) reading "you are Claude" gets confused; evaluate whether naming resolves the confusion or whether other spots also leak Claude-specificity. (2) Pull agent-neutral content out of Anthropic-specific files (=CLAUDE.md=) into a shared source that each agent's own entry file points to, so Codex (which runs more literal) reads the same rules; or link =CLAUDE.md= and the Codex equivalent to one source. Have Codex review the workflows for literal-reading wording gaps. (3) Send =.emacs.d= a note (inbox-send) to let =ai-term= launch Claude / Codex / a local ollama LLM, switchable seamlessly at session start. Spec-shaped — needs design before build. From the roam inbox 2026-06-23 (deferred from the 2026-06-21 session).
-*** 2026-05-23 Sat @ 16:12:48 -0500 Decided: dedicated private repo, not stow
-Worked through dotfiles → rulesets → dedicated repo. Dropped stow/dotfiles (machine config, wrong cadence) and rulesets (it's pulled first in every session, so memory edits would dirty its tree and skip the startup =git pull --ff-only=). Chose a dedicated private repo on cjennings.net: storage is unified there while recall stays per-project (the encoded-cwd subdirs), since pooling recall would hurt relevance and risk work-private facts surfacing in personal-project artifacts.
+*** 2026-06-24 Wed @ 00:21:20 -0400 Partial — agent-neutral wording sweep + thread-3 note landed
+Thread 2's wording half shipped in 6ad0442 (=refactor(rules): use agent-neutral language in shared rules=): agent-as-actor phrasing replaced with "the agent" across interaction.md, cross-project.md, triggers.md, working-files.md. Thread 3's note reached =.emacs.d=, whose 2026-06-23 inbox FYI confirms it received and filed the "multi-LLM support" ai-term handoff. Remaining and still TODO: thread 1 (give the agent a name), and thread 2's structural half (extract agent-neutral content into a shared source with a Codex entry-file pointer, then have Codex review the workflows for literal-reading gaps).
-*** 2026-05-23 Sat @ 16:12:48 -0500 Shipped: claude-memory.git + folded symlinks
-Created bare =git@cjennings.net:claude-memory.git=, cloned to =~/.claude-memory= (later deleted in the reversal below), moved all 7 per-project =memory/= dirs in (54 files; work has 40) and replaced each live =~/.claude/projects/<enc>/memory= with a folded dir-symlink so new memory lands in the clone and a push syncs it. Added =link-claude-memory.sh= (idempotent — recreates the symlinks on a new machine after clone) + README. Private repo, never GitHub (carries work/DeepSat memory). Initial import pushed (=f496370=).
+*** 2026-07-13 Mon @ 16:04:04 -0500 Thread 2's entry-file half landed via the runtime-portability build
+The Codex entry-file pointer now exists: =claude-templates/AGENTS.md= (thin pointer at protocols.org + rules + /name resolution), linked to =~/.codex/AGENTS.md= by =make install= and seeded per-project by =install-ai.sh= (see the generic-agent-runtime parent's 2026-07-13 children). Remaining here: thread 1 (agent naming — the entry file's "you are this project's agent" phrasing is a start, not the whole answer) and the Codex literal-reading review of workflows.
-*** 2026-05-24 Sun @ 01:53:35 -0500 Reversed the migration — back to unmanaged per-project memory
-Cancelled the follow-up brainstorm and undid the dedicated-repo migration at Craig's call. Moved all 7 memory dirs back to =~/.claude/projects/<enc>/memory/= (content preserved), deleted the =~/.claude-memory= clone, and deleted the bare =claude-memory.git= on the server. Memory is back to its original at-risk state, so the task reopens at [#C] pending a direction. The brainstorm landed on a two-tier idea for whenever this resumes: promote general lessons into a rulesets-tracked file symlinked into =~/.claude/rules/= (loaded into every project natively, one repo), and keep project-specific memory under each project's own =.ai/memory/= (committed where =.ai/= is tracked, at-risk where it's gitignored). Not implemented.
+** TODO [#C] Flashcard tooling improvements :feature:
+:PROPERTIES:
+:CREATED: [2026-06-28 Sun]
+:LAST_REVIEWED: 2026-07-13
+:END:
+Three flashcard-tooling tasks that all edit =flashcard-to-anki.py= and/or =flashcard-stats.py=, grouped so they get built together instead of colliding on the same files (prior sessions flagged the conflict risk). The Anki =#+TITLE= deck-name fix already landed (commit 060a938), so any preserved pre-fix script copy gets re-derived against the current canonical, never copied wholesale. The three children each ship independently.
-*** 2026-06-05 Fri @ 05:57:35 -0500 Pivot: adopt the existing org-roam KB as the shared agent substrate
-Pressure-tested the two-tier idea, then Craig redirected: a shared org-roam knowledge base any project can read and write makes this simpler. Ground truth verified: =~/sync/org/roam/= already exists (484 org files, curated since 2023, Syncthing-synced, not git). So cross-machine sync is already solved, and the task stops being "build a memory-sync system" and becomes "point agents at the KB that already syncs." The dedicated-repo and two-tier approaches are both superseded for the storage+sync half.
+*** 2026-07-19 Sun @ 18:40:36 -0500 Built apkg-to-orgdrill.py, the inverse converter
+Commit a143679. Reads an Anki =.apkg= (stdlib =zipfile= + =sqlite3=, no genanki) and emits org-drill =.org= in the house shape: deck → =#+TITLE=, note Front → =** heading :drill:=, Back HTML → body (=<br>=→newlines, entities unescaped amp-last, answer =<hr>= stripped), tag → =* section=, fresh =:ID:= per card. 17 tests incl. the round-trip through =flashcard-to-anki.py='s own parse(); verified end-to-end against a real genanki apkg. Grounded the schema by generating + inspecting a real apkg first. Speedrun task 1 of 2.
-Wrote a one-page spec: [[file:docs/agent-knowledge-base-spec.org][agent-knowledge-base-spec.org]] (originally docs/design/2026-06-05-org-roam-knowledge-base-spec.org; superseded by the 2026-06-10 spec-create rewrite at the new path). Five decisions, mechanics recommended: (1) KB is a queried substrate accessed as files (ripgrep + follow =[[id:]]= by grep), not via the org-roam package; (2) capture in harness memory, promote durable facts into the KB (same cadence as the pattern catalog) — resolves the at-risk problem since the valuable knowledge moves to the synced KB; (3) a =claude-rules/knowledge-base.md= pointer rule carries path/query/write-schema/boundary; (4) write schema = roam-valid node + =:agent:= filetag so agent notes stay distinguishable and index on the next =org-roam-db-sync=. The rules layer (=claude-rules/=, =CLAUDE.md=) is untouched — the KB replaces the memory tier, not the rules tier.
+*** TODO [#C] flashcard-stats refutation / claim-prompt mode :feature:
+:PROPERTIES:
+:CREATED: [2026-06-22 Mon]
+:LAST_REVIEWED: 2026-06-28
+:END:
+A refutation card (heading is a bare false claim, body is the rebuttal) is valid org-drill but trips two BLOCKING =flashcard-stats.py= checks as false positives: non-prompt-heading (a declarative claim has no =?= or imperative verb) and answer-leakage (claim words reappear in the rebuttal). =flashcard-sync='s gate then blocks the whole deck.
-*** 2026-06-10 Wed @ 14:29:20 -0500 Spec ratified — write boundary is option C; rewritten to spec-create format
-Craig answered via cj annotations in the spec (2026-06-10): DECISION 5 is option C (read-shared, write-scoped — work agents never write the KB). Syncthing does replicate ~/sync/ to a work machine and Craig is fine with how C handles it. Node granularity: per-fact nodes. Write review: agent writes land freely in the KB only — explicitly not permission to post to email, Linear, or any public channel without review and consent. The spec was rewritten into the spec-create format at [[file:docs/agent-knowledge-base-spec.org][agent-knowledge-base-spec.org]] (old draft removed). Implementation explicitly held pending Craig's go-ahead; one decision still open (D7, next VERIFY).
+Design (Craig, 2026-06-28, supersedes the two proposed options): make the exemption *generic*, not refutation-specific — more card kinds like this will come. When the org header declares the relevant info, the gate honors it rather than blocking. So a general file-level header keyword (a card-kind / check-exemption declaration) tells =flashcard-stats.py= which checks not to apply, instead of a hardcoded =#+DECK_KIND: refutation= keyword or a per-card =:claim:= tag. Document the mechanism in =flashcard-review.org= and add tests (a header-declared exemption file passes despite declarative headings + claim/answer overlap). Edits =flashcard-stats.py= — coordinate with the multi-tag reconcile, same file. Proposal: [[file:docs/design/2026-06-21-flashcard-stats-refutation-proposal.org][proposal]] (its two-option fix is superseded by this generic header approach). Backlog. From home 2026-06-21.
-*** 2026-06-10 Wed @ 14:35:40 -0500 Spec review — not ready
-Review written at docs/agent-knowledge-base-spec-review.org (deleted on disposition completion; content summarized in the spec's Review dispositions). Rubric: =Not ready=. Blockers: resolve D7 (keep vs retire harness memory) and define the executable personal/work/unknown write-boundary classifier plus work-side write/refusal destination. Medium notes: use concrete ripgrep commands that exclude =*.sync-conflict-*= files, and define seed-node approval/rollback.
+*** 2026-07-19 Sun @ 18:41:00 -0500 Reconciled the flashcard multi-tag tooling into canonical
+Commit a14e43b. Broadened =CARD_RE= in both =flashcard-to-anki.py= and =flashcard-stats.py= so a heading is a card when =drill= is among its tag block (=:fundamental:drill:=), bounded card bodies by any L1/L2 heading (=HEADING_RE=), and added =--tag-filter= (subset emit) + =--guid-salt= (distinct GUID space) to to-anki, plus a drill-membership guard to stats. Re-derived against the current canonical rather than copying the preserved 2026-06-17 files, which predated the =#+TITLE= fix (060a938) and would have reverted it. parse()'s 3rd element is now the Anki-tag list; existing parse tests moved to that contract, new tests cover multi-tag / tag-filter / guid-salt / stats count, and a =flashcard-sync.bats= case guards the count. The real 465/100 check needs the work deck; verified end-to-end on a synthetic deck (2 cards → 1 with =--tag-filter fundamental=) through real genanki. Speedrun task 2 of 2.
-*** 2026-06-10 Wed @ 14:44:00 -0500 D7 resolved — keep harness memory as the capture layer
-Craig ratified "keep" in chat (2026-06-10). Harness memory stays the ephemeral, auto-recalled capture layer; the KB holds promoted durable facts; Phase 3's wrap-up promotion cadence is mandatory. Spec D7 flipped to accepted; D2 stands as written.
+** TODO [#C] Agent-KB / memory-sync — work + unknown-project write refusal :test:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-13
+:END:
+Residual manual validation from the memory-sync task (closed 2026-07-04, implementation IMPLEMENTED). Two checks need live sessions in other project contexts:
+- In the work project, a durable-storage request produces no KB write and the refusal report names the fact.
+- In an unknown project (outside =~/code/=, =~/projects/=, =~/.emacs.d=), the agent refuses or asks rather than guessing.
+Expected: both refusal checks behave per the spec; any miss promotes to a bug. Not =:solo:= — needs sessions in the work and an unknown project.
-*** 2026-06-10 Wed @ 14:44:00 -0500 Project classification defined — work-root denylist, unknown refuses
-Resolved in the spec-response pass: =knowledge-base.md= carries an explicit work-root denylist (initially =~/projects/work=) as the source of truth. Personal = under a known project parent (=~/code/=, =~/projects/=, =~/.emacs.d=) and not denylisted → KB writes allowed. Work or unknown → no KB write; the agent reports the refusal with a one-line redacted summary of the fact. v1 adds no new work-side store — work projects keep their existing project-tree conventions. See the "Project classification and write routing" section of [[file:docs/agent-knowledge-base-spec.org][the spec]]. Denylist completeness is the one open caveat (next VERIFY).
+** TODO [#C] Token-rotation helper for =@a-bonus/google-docs-mcp= OAuth refresh :feature:quick:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-13
+:END:
-*** 2026-06-10 Wed @ 14:44:00 -0500 Codex review incorporated — spec ready with caveats
-Spec-response pass processed the 2026-06-10 Codex review with D7 = keep as a pre-agreed input. Both blockers cleared (D7 accepted; classification/write-routing section added). Mediums accepted: canonical rg commands with conflict-file exclusion, Phase 2 seed-node approval/rollback mechanics, Makefile no-change note, Testing/Verification section. Three recommendations modified, none rejected — see the spec's Review dispositions. Review file deleted per the workflow. Rubric: ready with caveats (denylist confirmation). Implementation tasks broken out below; implementation itself awaits Craig's go.
+When a Google refresh token gets revoked (re-grant scopes, removed Connected App, account password reset), recovery is currently manual: run =npx -y @a-bonus/google-docs-mcp= with the right env, follow the URL in a browser, kill the process, base64-encode the new =token.json=, decrypt =secrets.env.gpg=, replace the var, re-encrypt. A small =mcp/refresh-google-docs-token.sh <profile>= would chain that into one command.
-*** 2026-06-10 Wed @ 17:29:37 -0500 Work-root denylist confirmed — ~/projects/work only
-Craig confirmed (2026-06-10, in chat): the denylist is just =~/projects/work=. Archangel is not work-scoped. The spec's one caveat clears — status now ready. Phase 1 is unblocked, but implementation still awaits Craig's explicit go.
+*** Sketch
-*** 2026-06-10 Wed @ 17:57:08 -0500 Spec amended — D8 git transport + migration/metrics/docs/maintenance folds
-Craig's five design questions answered and folded into the spec, and D8 ratified (Shape A): the KB moves out of the =~/sync/org= Syncthing share into its own git repo on cjennings.net, with an =agents/= subdirectory for agent writes, a systemd auto-sync timer for Craig's edits, opt-in-by-clone replication (work machine doesn't clone), and the phone staying on the on-demand =~/sync/phone= pattern. Folded in: inclusion criteria + a Phase 1.5 guided memory sweep, a Success metrics section with a 30-day checkpoint, the seed node redefined as the KB's own documentation, and Phase 4 maintenance automation. Phases renumbered 0-4; tasks below updated. Implementation still held.
+#+begin_src bash
+# usage: mcp/refresh-google-docs-token.sh personal
+profile="$1"
+gpg -d ... | grep -v "GOOGLE_DOCS_${profile^^}_TOKEN_B64" > /tmp/secrets.env.tmp
+GOOGLE_MCP_PROFILE="$profile" npx -y @a-bonus/google-docs-mcp &
+xdg-open <captured-url>
+# wait for ~/.config/google-docs-mcp/$profile/token.json to land
+kill %1
+echo "GOOGLE_DOCS_${profile^^}_TOKEN_B64=$(base64 -w0 ~/.config/google-docs-mcp/$profile/token.json)" >> /tmp/secrets.env.tmp
+gpg -c --cipher-algo AES256 -o mcp/secrets.env.gpg.new /tmp/secrets.env.tmp
+mv mcp/secrets.env.gpg.new mcp/secrets.env.gpg
+rm /tmp/secrets.env.tmp
+#+end_src
-*** 2026-06-10 Wed @ 18:21:33 -0500 Phase 0 done — roam migrated to git
-Backed up (~/roam-backup-2026-06-10.tar.gz), copied to =~/org/roam=, 63 conflict files deleted (424 org files), git repo with origin =git@cjennings.net:roam.git= (initial commit 515693d), old location replaced with a transition symlink. Emacs =roam-dir= updated in user-constants.el + live-reloaded (db rebuilt, 416 nodes); handoff to .emacs.d for the commit. =roam-sync.sh= (6 bats green) on a 15-min systemd user timer, installed + enabled + round-trip verified. Old-path references repointed (protocols task-list pointer, journal workflow, notes template). archsetup handoff covers dotfiles adoption + other-machine clones. rulesets commit fcf554a.
+The flow tonight worked but took a handful of manual steps. One script collapses it.
-*** 2026-06-10 Wed @ 18:21:33 -0500 Phase 1 done — knowledge-base.md rule live
-=claude-rules/knowledge-base.md= written (path, git discipline, query commands, agents/ write schema, denylist + refusal contract, inclusion criteria, capture-then-promote). =make install= linked it machine-wide; verified the link, a known-note query, and conflict-glob exclusion with a planted file. Commit d071f1f.
+Decision (Craig, 2026-05-31): *hold until a token rotation is imminent.* The OAuth re-grant is a browser step that can't be triggered without revoking a live token, so the script can't be verified in isolation. Not marked =:solo:= — when a token actually needs rotating, write and verify in one pass (solo at that point).
-*** 2026-06-10 Wed @ 18:21:33 -0500 Phase 1.5 done — rulesets swept, 10 projects broadcast
-Rulesets' 6 memories classified: 3 promoted as =agents/= nodes (notify-attention pattern, pdftools venv, gpg-agent SSH TTL trap), 2 kept local (rule-encoded in verification.md / interaction.md), 1 kept + de-staled (ai-scripts-canonical updated for the claude-templates subtree fold). Sweep handoff broadcast to the 10 other memory-bearing projects (archsetup, org-drill, pearl, .emacs.d, elibrary, finances, health, home, jr-estate, kit); work skipped by the boundary; the orphaned =linear-emacs= memory dir (project retired, likely pearl's predecessor) noted for Craig.
+** TODO [#D] Generic agent runtime support — Codex spec v0 :spec:design:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-06-28
+:END:
+Codex drafted a v0 design doc for making rulesets runtime-neutral rather than Claude-Code-specific. Motivating cases: offline operation with a local LLM, and two LLMs running in the same project at the same time without trampling each other's session-context.
-*** 2026-06-10 Wed @ 18:21:33 -0500 Phase 2 done — seed/doc node written and indexed
-=agents/20260610181640-how-the-agent-knowledge-base-works.org= written: the KB's user-facing guide (what agents do, how it syncs, finding/pruning agent content, the rule pointer). Index verified programmatically: =org-roam-node-from-title-or-alias= resolves it with tags (agent reference); node count 416 → 420. Craig's visual check remains in the manual-testing child.
+Spec at [[file:docs/design/2026-05-28-generic-agent-runtime-spec.org]] (moved here from inbox on intake).
-*** 2026-06-10 Wed @ 18:21:33 -0500 Phase 3 done — wrap-up promotes + records the KB receipt
-wrap-it-up.org Step 1 gains the promotion check (inclusion-criteria bar) and the mandatory "KB: promoted N / consulted yes-no" Summary line; validation checklist enforces it. Mirror synced, integrity OK (44), parse OK. Commit 242b95e.
+Immediate correctness issue Codex flagged: the singleton .ai/session-context.org is unsafe under simultaneous agents. Codex recommends starting with Phase 1 only — add AI_AGENT_ID + session-context.d/<id>.org without renaming the rest.
-*** 2026-06-11 Thu @ 19:26:26 -0500 .emacs.d memory sweep complete (first broadcast response)
-First of the 10 broadcast projects to report Phase 1.5 done (handoff 18:23). Inventory 7: promoted 3 to KB (no-make-frame-in-live-daemon, proton-bridge-headless-cert-mismatch, open-images-with-imv — roam commit a915760), kept 3 local at Craig's call (commit-flow-no-approval-gate per-project-scoped; two theme-scoped ones possibly superseded by the palette-columns spec), deleted 1 (superseded by canonical interaction.md rule). 9 projects' sweeps outstanding.
+Broader refactor proposes runtimes/ adapter manifests, generic install commands, language-bundle split (common/ + runtimes/<runtime>/), launcher refactor, local model service via llama.cpp/ollama. Big surface area, six phases.
-*** 2026-06-12 Fri @ 02:25:12 -0500 Five more sweeps complete via the home folds
-Overnight handoffs from home closed five more broadcast targets, each swept at fold-time triage with Craig's approval: jr-estate 2 promoted (forms name-with-number, PDF-editing tooling split; roam 45d8e6c) / 3 kept with area attribution / 2 deleted as rule-encoded or duplicate; finances 0/1/0 (rosalea-daly contact fact kept local); elibrary 0/0/2, health 0/0/1, kit 1/0/2 (hand-prep-items-to-work-inbox promoted into home's memory; the rest duplicated rules or home memories). Nothing from these five met the KB bar that wasn't already encoded. All folded projects' session archives merged area-prefixed into home's .ai/sessions/, so session-harvest's first run sees them. Home covers its own and remaining areas' sweeps through ongoing discipline; still pending from the broadcast: archsetup and work.
+2026-06-12 spec review complete: [[file:docs/design/2026-05-28-generic-agent-runtime-spec-review.org][Codex review]] rubric for the whole spec is =Not ready=. Phase 1 is already shipped, and Phase 1.5 is tracked separately as the helper-instance task. Before any phases 2-5 implementation, decide whether to commit to the larger arc and answer the blocker decisions: generic instruction-file strategy, default local runtime/server, first supported local editing CLI, adapter scope, and compatibility behavior for existing =CLAUDE.md= / =.claude/= projects.
-*** TODO Agent KB — manual testing and validation :test:
-What we're verifying: the v1 acceptance surface that needs Craig's eyes or a live cross-project session. Run after Phases 0-2 land.
-- Seed node appears in org-roam (autosync) and in the =rg '#\+filetags:.*:agent:'= inventory.
-- In the work project, a durable-storage request produces no write in the KB and the refusal report names the fact.
-- In an unknown project (outside =~/code/=, =~/projects/=, =~/.emacs.d=), the agent refuses or asks rather than guessing.
-- After Phase 0: an edit made on one machine appears on another within the auto-sync timer interval, no new sync-conflict files appear, and the work machine has no KB clone.
-Expected: all four behave per the spec; any miss promotes to a bug task. (Agent-runnable checks — make install link, rg finds a known note, conflict-file exclusion — are verified inside Phases 0-2.)
+*** 2026-06-10 Wed @ 14:13:55 -0500 Noted Phase 1 already shipped; narrowed scope to the phases 2-6 decision
+Phase 1 (the correctness fix) is live: protocols.org documents the AI_AGENT_ID-scoped session-context path (=.ai/session-context.d/<id>.org=) and =.ai/scripts/session-context-path= resolves it. The singleton race Codex flagged is closed. What remains is the spec review plus a go/no-go on the broader runtime-neutral refactor: runtimes/ adapter manifests, generic install commands, language-bundle split, launcher refactor, local model service.
+
+*** 2026-06-11 Thu @ 19:26:26 -0500 Spec amended with the helper-instance slice; implementation split out
+Craig's motivating case (a second Claude in the same project for lookups and safe task updates) was under-specified in v0 — it had identity and message targeting but no spawn mechanics and no write-safety contract for the shared files the session-context split doesn't isolate. Added the "Concurrent same-project agents (helper instances)" section (subagent boundary, identity/spawn via =ai --helper=, the tiered read/write contract, light startup, helper wrap-up) and Phase 1.5 to the migration plan. Implementation filed as its own [#B] task ("Helper-instance support"); this task stays scoped to the phases 2-6 go/no-go.
+
+*** 2026-06-12 Fri @ 02:09:10 -0500 Independent spec review complete
+Codex ran the spec-review workflow. Outcome: the combined spec is =Not ready= because phases 2-5 still require product decisions and current external-runtime/model verification. Phase 1.5 can proceed only as the already-split helper task, with rollout/manual-validation caveats accepted and no accidental template-wide release before sandbox/pilot drills pass. Review file: [[file:docs/design/2026-05-28-generic-agent-runtime-spec-review.org]].
+
+*** 2026-06-12 Fri @ 02:39:38 -0500 Second review after response pass
+Codex re-ran spec-review after the dispositions were folded in. Outcome by arc: Phase 1.5 helper instances =Ready with caveats=; phases 2-5 remain =Not ready= behind the explicit decisions/reverification gate. No new blocking findings for the helper slice. Review file updated in place: [[file:docs/design/2026-05-28-generic-agent-runtime-spec-review.org]].
+
+*** 2026-07-13 Mon @ 13:26:57 -0500 Gap assessment decomposed into child tasks
+Craig asked what's left to run ChatGPT or a local LLM as the agent. Assessment: the =.ai/= layer (protocols, workflows, scripts, inbox, todo, session anchors) is already runtime-neutral — plain org + bash, and a Codex session has run in it (2026-06-13). The Claude-bound remainder decomposed into the child tasks below; each overlapping spec-blocker decision is named in its body. The phases 2-5 go/no-go above still gates any big build, but several children are useful standalone.
+
+*** 2026-07-13 Mon @ 16:04:04 -0500 Instruction bootstrap built — thin-pointer AGENTS.md, both install paths
+Craig picked the thin-pointer shape (Decisions: option 1, definitively). Shipped TDD (bats red → green): canonical =claude-templates/AGENTS.md= (you-are-this-project's-agent + read protocols.org + rules locations + the /name resolution rule from the skill-parity finding + degrade-per-fallback, never skip gates); =make install= links it to =~/.codex/AGENTS.md= (new CODEX_DIR stanza, house skip/WARN/link idiom, covered by NEW =scripts/tests/install-agents-entry.bats=, 3 tests); =install-ai.sh= seeds a project-owned copy at bootstrap, never overwriting (+2 tests in install-ai.bats); rulesets root gets a tracked symlink as dogfood. Resolves the spec's "generic instruction-file strategy" blocker. velox picks up the global link automatically — startup Phase A.0 runs =make install= every session. Existing projects get seeded on demand (no auto-sweep in v1).
+
+*** 2026-07-13 Mon @ 14:52:38 -0500 Skill parity resolved — one resolution rule, no per-skill matrix
+Analysis in [[file:docs/design/2026-07-13-runtime-portability-inventories.org]] (Skill and command parity section). All 29 artifacts (11 skills + 18 commands) are markdown bodies; a single resolution sentence in the bootstrap entry file ("a /name reference resolves to the skill/command file — read and follow it") makes the library portable to any file-reading harness. Auto-invocation degrades to by-name invocation (the publish flow already invokes by name), native slash registration is an optional install nicety, flush is explicitly excluded (session-plumbing child owns it). Folds into the instruction-bootstrap child's build.
+
+*** 2026-07-13 Mon @ 13:34:17 -0500 Hook parity inventoried — only two hooks carry real porting work
+Full mapping in [[file:docs/design/2026-07-13-runtime-portability-inventories.org]]. AskUserQuestion deny is moot off Claude (no popup tool to deny); PostToolUse validators survive via the bundles' git pre-commit hooks; clear-resume folds into the session-plumbing child; session-title is cosmetic. Real gaps: PreCompact priority-save (prose downgrade) and Stop wrap-teardown (Codex notify / manual elsewhere) — decisions in the VERIFY below.
+
+*** 2026-07-13 Mon @ 23:21:08 -0500 Interactive agent picker — bare =ai= asks which brain first
+Craig's ask: the fzf flow should offer the LLMs before the projects. Bare =ai= now runs a runtime picker (claude first so Enter-Enter keeps the old muscle memory; codex labeled ChatGPT; one =local:<model>= line per ollama model, live-queried with a 3s timeout — a dead server just drops the local lines), then the familiar annotated project multi-select. =--runtime= / =AI_RUNTIME= skip the pick; single-dir mode unchanged. New =--print-runtimes= test seam; 9/9 launcher bats, suite 441/0. This substantially delivers the "easy lightweight way to change agents / agentically democratic" roam ask — the launcher-hardening task keeps the deeper refactor.
+
+*** 2026-07-13 Mon @ 23:04:21 -0500 Local runtime wired — ai --runtime local runs codex --oss over ollama
+The reserved lane went live the same day: =local= maps to =codex --oss --local-provider=ollama -m $AI_LOCAL_MODEL= (default gpt-oss:120b; qwen3-coder:30b via the env var). Verified end to end: =codex exec --oss --local-provider=ollama -m gpt-oss:120b= returned a correct completion through the local model (~7K tokens). The explicit provider flag avoids config.toml dependence (a root-level =oss_provider= key would also work; appending to the file lands in the last TOML table and silently does nothing — learned live). Dependency check keys on AGENT_BIN (codex) now that AGENT_CMD carries flags. This partially answers the spec's "first supported local editing CLI" blocker: codex-over-ollama is the de-facto v1. The model-floor child's live trial (a takuzu session under =ai --runtime local=) is the next step.
-*** 2026-06-10 Wed @ 18:21:33 -0500 Phase 4 done — monthly hygiene automation live
-=scripts/kb-hygiene.sh= (6 bats green, shellcheck clean, read-only by design) inventories =:agent:= nodes, flags orphans / duplicate titles / conflict files, and writes an org report into the rulesets inbox; =roam-hygiene.timer= (monthly, Persistent) installed + enabled. Live run against the real KB verified (4 agent nodes, 428 files, 0 conflicts). Conditional vNext stays in the spec's scope tiers: a =/promote= command if the wrap-up prompt proves insufficient, an =:agent:inbox:= staging tag if free writes prove too noisy. Commit b014095.
+*** 2026-07-13 Mon @ 16:39:41 -0500 Launcher runtime flag built — ai --runtime claude|codex
+Shipped TDD (new =scripts/tests/ai-launcher-runtime.bats=, 6 tests red→green): =--runtime= flag + =AI_RUNTIME= env on =bin/ai=, runtime→CLI mapping (claude default, codex 1:1 — both take the opening line as a positional prompt), runtime-aware dependency check, =local= reserved with a clear not-wired-yet error (pending the model-floor evaluation), and a =--print-launch= mode as the test seam (prints the pane launch command, no tmux/fzf). Kept small per the 2026-06-24 helper-task caveat — dispatch pre-parse only, no launcher restructure. The Emacs-side equivalent (ai-term multi-LLM) remains .emacs.d's June handoff. Live smoke: correct commands for both runtimes against real projects.
-** TODO [#C] Morning ops orchestrator pilot — read-only :feature:
+*** 2026-07-13 Mon @ 17:05:00 -0500 Roam capture folded — the agent-switching ask
+Two roam items (2026-07-13) asked for lightweight agent selection (claude, qwen, chatgpt, ollama — "agentically democratic"). Partly shipped the same day: =--runtime claude|codex= (04c3b29; codex is the ChatGPT-side CLI). The ollama/qwen choice is the reserved =local= runtime, landing with the model-floor eval below. An interactive runtime picker rides the new launcher-hardening task ([#C] :refactor:solo:, filed from the same capture).
+
+*** 2026-07-13 Mon @ 16:41:14 -0500 Session plumbing assessed — no build needed
+Analysis in [[file:docs/design/2026-07-13-runtime-portability-inventories.org]] (Session plumbing section). session-context-path is runtime-aware by design; the anchor cycle is plain files any agent drives identically; self-inject.sh is harness-agnostic (only its PAYLOAD is Claude-shaped — a codex auto-flush is a payload variant whose second injected line carries the resume instruction itself, no hook needed); session-clear-resume.sh stays Claude-only and the payload variant obsoletes it off Claude. Wire the codex variant the day a codex session wants it.
+
+*** 2026-07-13 Mon @ 13:34:17 -0500 Memory story confirmed — KB is already the cross-agent store
+Details in [[file:docs/design/2026-07-13-runtime-portability-inventories.org]]. Auto-memory stays Claude-owned; non-Claude agents use the org-roam KB + file artifacts (todo/notes/session anchors), all runtime-neutral already. One wording gap: knowledge-base.md's capture-then-promote names harness memory as the capture layer — a one-sentence addition (session log as the capture layer for agents without harness memory) closes it, pending approval in the VERIFY below.
+
+*** 2026-07-13 Mon @ 13:34:17 -0500 MCP portability checked — portable except paging
+Details in [[file:docs/design/2026-07-13-runtime-portability-inventories.org]]. Nine locally-configured servers (linear, notion, figma, slack-deepsat, google-calendar, google-docs x2, drawio, google-keep) port mechanically to any MCP-speaking harness. claude.ai-managed connectors (Gmail + claude.ai Calendar/Drive) don't travel but are redundant with local servers / cmail-action. The real finding: signal-mcp is NOT locally configured anywhere — the protocols.org paging path is claude.ai-side only, so off Claude there is no page channel. The Signal-pager [#C] task's signal-cli runbook is the fix; add runtime-portability as motivation there (VERIFY below).
+
+*** 2026-07-13 Mon @ 14:40:00 -0500 Four inventory decisions — Craig approved all four
+Recorded in [[file:docs/design/2026-07-13-runtime-portability-inventories.org][the inventories doc]]: prose-only PreCompact downgrade off Claude; Stop-teardown via Codex notify where available, manual elsewhere; runtime-portability note added to the Signal-pager task; capture-layer sentence added to knowledge-base.md (non-Claude runtimes capture into the session log and promote from there).
+
+*** TODO Local model floor evaluation
+The workflows assume long-context instruction-following (startup's multi-file read; the commits.md publish chain). Establish the minimum viable local tier (likely strong-70B+/MoE, 100k+ context), and what compensations a weaker model needs: shortened protocols, more checklist gates, more hook-level enforcement. Feeds the spec's "default local runtime/server" and "first supported local editing CLI" blocker decisions.
+
+Environment inventory done 2026-07-13 (KB node "Local LLM inference inventory — daily drivers"): ratio is the inference box — Strix Halo iGPU + 125 GiB unified RAM. velox is out of scope (Iris Xe, 60 GiB, no ollama).
+
+Environment BUILT and verified, evening of 2026-07-13: ollama 0.31.2 as a systemd service, Vulkan/RADV backend forced via systemd override (ROCm loaded a 61 GB model at <25 MB/s — 40+ min, unusable; Vulkan loads it in 44 s), models swapped to the MoE shape this platform wants: gpt-oss:120b (44 s load, 37.9 tok/s, 100% GPU) and qwen3-coder:30b (5 s load, 76 tok/s). All four legacy dense models dropped; the orphaned ~/.ollama user store purged. Gotcha for the eval harness: pass num_ctx 8-32K — gpt-oss's native 128K default balloons the load. Remaining work is now purely the evaluation: drive a scripted startup read + a publish-flow transcript through gpt-oss:120b (and qwen3-coder as the fast comparator), grade instruction-following against the protocol stack, and answer the spec's default-local-runtime + first-supported-CLI blockers.
+
+** TODO [#C] Docs-lifecycle convention — manual validation :test:
:PROPERTIES:
-:CREATED: [2026-06-11 Thu]
-:LAST_REVIEWED: 2026-06-11
+:LAST_REVIEWED: 2026-07-13
:END:
-A scheduled headless morning run chaining the existing pieces: startup checks, the triage-intake scan, a system health check — producing the prep doc plus a report and a notify ping, with all remediation propose-only. Staged adoption from the 2026-06-11 insights report's "Self-Healing Daily Ops Orchestrator": read-only first; promote individual routine remediations to auto only after each has a track record. Known blockers to design around: headless MCP auth (interactively-authenticated servers are absent in cron runs) and the consent boundary (triage Phase D, anything destructive).
+The human-eyes half of the docs-lifecycle acceptance surface. The convention shipped IMPLEMENTED 2026-07-04 (spec [[id:80b0787b-4a60-4c82-8a16-b383d3e3c8f2][docs-lifecycle]]); these checks confirm the human-facing behavior. A failed check promotes to a bug.
+
+*** Startup nudge appears and clears
+What we're verifying: the Phase A probe + Phase C nudge fire exactly once per project.
+- Open a session in a project with an unsorted docs/design (before its sort) and read the startup output.
+Expected: one line offering "run spec-sort"; after the pilot stamps :LAST_SPEC_SORT:, the next session shows nothing.
-** TODO [#C] Build =create-documentation= skill for high-quality project/product docs :feature:
+*** Moved-spec links click through in Emacs
+What we're verifying: the pilot's relink pass left todo.org and docs links working for the human reader, not just the residue grep.
+- After the Phase 3 pilot, open todo.org in Emacs and click three links that point at moved specs (including one from a dated log entry).
+Expected: each opens the spec at its new docs/specs/ path.
+
+** TODO [#D] Docs lifecycle vNext — org-agenda spec-status view :feature:no-sync:
+Once specs carry lifecycle TODO keywords under =docs/specs/=, add a custom org-agenda view that lists =DRAFT= / =READY= / =DOING= / terminal specs by status. Deferred from [[id:80b0787b-4a60-4c82-8a16-b383d3e3c8f2][the docs-lifecycle spec]]; not part of v1 because the grep board is sufficient until the status headings exist.
+
+** TODO [#C] Wrap-it-up summary mode — keep or cut :feature:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-13
+:END:
+From Craig via the roam inbox (2026-07-02, routed by archsetup). Teardown-by-default already shipped (bare "wrap it up" closes the window; "with summary" keeps it). Craig's follow-on: "maybe we cut the summary altogether. help me think through when I'd want a summary and how I would recognize it before confirming and then having it close." Run that think-through with him (brainstorm-shaped, not solo), then adjust wrap-it-up.org's Step 6 + trigger phrases to the outcome.
+
+** TODO [#D] Warn-only pre-commit hook for tooling-path enumeration :feature:
:PROPERTIES:
-:LAST_REVIEWED: 2026-06-12
+:CREATED: [2026-06-22 Mon]
+:END:
+Optional enforcement teeth for the no-attribution / no-tooling-artifacts tightening landed 2026-06-22 (commit 91217d9), which is documentation-only. A warn-only (not blocking) pre-commit hook could scan the commit subject + body for tooling-path enumeration (=CLAUDE.md=, =.claude/=, =.ai/=, =todo.org=, =notes.org=, =session-context=) and AI-attribution language, with the two exemptions baked in: a commit whose change IS one of those files, and private single-user repos. Must warn, not block — a rigid grep false-positives on legit subject mentions. Deferred: Craig chose docs-only for now.
+
+** TODO [#D] Build =create-documentation= skill for high-quality project/product docs :feature:
+:PROPERTIES:
+:LAST_REVIEWED: 2026-06-15
:END:
Create a Claude skill named =create-documentation= that can plan, write,
@@ -803,9 +1408,9 @@ The skill should reject:
public/library/API docs: =llms.txt= or markdown export is valuable, but normal
human navigation remains primary.
-** TODO [#C] Build /research-writer — clean-room synthesis for research-backed long-form :feature:
+** TODO [#D] Build /research-writer — clean-room synthesis for research-backed long-form :feature:
:PROPERTIES:
-:LAST_REVIEWED: 2026-06-12
+:LAST_REVIEWED: 2026-06-15
:END:
Gap in current rulesets: between =brainstorm= (idea refinement → design doc)
@@ -862,7 +1467,7 @@ Upstream reference (do not vendor): ComposioHQ/awesome-claude-skills
** TODO [#D] Revisit =c4-*= rename if a second notation skill ships :chore:
:PROPERTIES:
-:LAST_REVIEWED: 2026-06-10
+:LAST_REVIEWED: 2026-06-15
:END:
Current naming keeps =c4-analyze= and =c4-diagram= as-is (framework prefix
@@ -1006,1701 +1611,723 @@ having a skill to generate or check OV-1-shaped artifacts. Don't build
speculatively — defense-specific notations are narrow enough that each
skill should be driven by a concrete contract need, not aspiration.
-** TODO [#C] Token-rotation helper for =@a-bonus/google-docs-mcp= OAuth refresh :feature:quick:
+* Rulesets Resolved
+** CANCELLED [#C] ntfy phone channel as general two-way agent-comms :feature:spec:
+CLOSED: [2026-07-13 Mon]
:PROPERTIES:
-:LAST_REVIEWED: 2026-06-12
+:CREATED: [2026-06-20 Sat]
+:LAST_REVIEWED: 2026-06-24
:END:
-
-When a Google refresh token gets revoked (re-grant scopes, removed Connected App, account password reset), recovery is currently manual: run =npx -y @a-bonus/google-docs-mcp= with the right env, follow the URL in a browser, kill the process, base64-encode the new =token.json=, decrypt =secrets.env.gpg=, replace the var, re-encrypt. A small =mcp/refresh-google-docs-token.sh <profile>= would chain that into one command.
-
-*** Sketch
-
-#+begin_src bash
-# usage: mcp/refresh-google-docs-token.sh personal
-profile="$1"
-gpg -d ... | grep -v "GOOGLE_DOCS_${profile^^}_TOKEN_B64" > /tmp/secrets.env.tmp
-GOOGLE_MCP_PROFILE="$profile" npx -y @a-bonus/google-docs-mcp &
-xdg-open <captured-url>
-# wait for ~/.config/google-docs-mcp/$profile/token.json to land
-kill %1
-echo "GOOGLE_DOCS_${profile^^}_TOKEN_B64=$(base64 -w0 ~/.config/google-docs-mcp/$profile/token.json)" >> /tmp/secrets.env.tmp
-gpg -c --cipher-algo AES256 -o mcp/secrets.env.gpg.new /tmp/secrets.env.tmp
-mv mcp/secrets.env.gpg.new mcp/secrets.env.gpg
-rm /tmp/secrets.env.tmp
-#+end_src
-
-The flow tonight worked but took a handful of manual steps. One script collapses it.
-
-Decision (Craig, 2026-05-31): *hold until a token rotation is imminent.* The OAuth re-grant is a browser step that can't be triggered without revoking a live token, so the script can't be verified in isolation. Not marked =:solo:= — when a token actually needs rotating, write and verify in one pass (solo at that point).
-
-** TODO [#C] Generic agent runtime support — Codex spec v0 :spec:design:
+Killed at the 2026-07-13 task review: home retired and tore down the ntfy channel on 2026-07-04, so this proposal's transport no longer exists. Its living successors are the Signal pager ([#B] task above — one identity on velox, runbook pending) and agent-page (shipped 2026-07-13), which cover the send half; two-way (read-replies) rides the Signal runbook.
+Proposal from the home project (2026-06-17): promote the self-hosted ntfy-over-Tailscale phone channel it built and verified on ratio into a general two-way agent-comms tool rulesets owns. Full proposal: [[file:docs/design/2026-06-17-ntfy-agent-comms-proposal.org]] (as-built runbook stays in the home project at =working/phone-notifications/spec.org=). What rulesets would decide: canonicalize =phone-notify= (send) plus a new =phone-recv= (check-since) as synced bin scripts; the per-machine config/secret convention (token in =~/.config/phone-notify/config= chmod 600 today, vs GPG-encrypted in dotfiles); a reference =ntfy-inbound-handler= plus systemd user-unit for event-driven delivery (Tier A subscriber routes inbound to inbox/notify, Tier B inbound spawns an agent session, Tier C notify a live session — harness research); approval-button workflows for the commits.md gates when Craig is away from the desk (tap-to-approve, the high-value concrete use); and the relationship to the retired cross-agent-comms scripts (ntfy may be the transport they lacked). Worked via =spec-create=. Blocks the triage-intake phone-push task below.
+** DONE [#B] Org-table helpers corrupt example blocks :bug:solo:
+CLOSED: [2026-07-14 Tue]
:PROPERTIES:
-:LAST_REVIEWED: 2026-06-12
+:CREATED: [2026-07-11 Sat]
+:LAST_REVIEWED: 2026-07-13
:END:
-Codex drafted a v0 design doc for making rulesets runtime-neutral rather than Claude-Code-specific. Motivating cases: offline operation with a local LLM, and two LLMs running in the same project at the same time without trampling each other's session-context.
-
-Spec at [[file:docs/design/2026-05-28-generic-agent-runtime-spec.org]] (moved here from inbox on intake).
-
-Immediate correctness issue Codex flagged: the singleton .ai/session-context.org is unsafe under simultaneous agents. Codex recommends starting with Phase 1 only — add AI_AGENT_ID + session-context.d/<id>.org without renaming the rest.
+Fixed in 951b6fc, test-first. Both defects landed as filed (block-type-aware scanning in both helpers; lint-org CLI report-only by default, writes behind --fix), plus the true corruption path found during the work: wrap-org-table's load-time CLI dispatch fired on lint-org's require and reformatted lint-org's file arguments. Entry-script guard added. The named regression test (example block byte-identical) is in the suite.
+=wrap-org-table.el= and =lint-org.el= both scan for =/^\s*|/= lines and rewrite them as org tables without skipping =#+begin_example=/=src=/=quote= regions, so ASCII art using pipe characters gets mangled into bordered tables. Reproduced 2026-07-09 in the work project against an architecture doc with a pipe/=v= flow diagram; a plain indented block became a table with =|---|= rules between every line. Two separable defects: (1) table detection is line-based — both helpers should use =org-element-at-point= / =org-in-block-p= to skip example/src/quote/verse blocks; (2) =lint-org.el= mutates its input on disk with no confirmation — passing five files reformatted all five (one by 1949 lines). A linter must report, not write; put the reformat behind an explicit =--fix= flag.
-Broader refactor proposes runtimes/ adapter manifests, generic install commands, language-bundle split (common/ + runtimes/<runtime>/), launcher refactor, local model service via llama.cpp/ollama. Big surface area, six phases.
+Grading (severity × frequency, per todo-format.md): Critical severity (silent org data loss; recoverable here only because the content was git-staged) × rare-edge-case frequency (fires only when a mixed table+example file is passed to the helper) = P2 = [#B].
-2026-06-12 spec review complete: [[file:docs/design/2026-05-28-generic-agent-runtime-spec-review.org][Codex review]] rubric for the whole spec is =Not ready=. Phase 1 is already shipped, and Phase 1.5 is tracked separately as the helper-instance task. Before any phases 2-5 implementation, decide whether to commit to the larger arc and answer the blocker decisions: generic instruction-file strategy, default local runtime/server, first supported local editing CLI, adapter scope, and compatibility behavior for existing =CLAUDE.md= / =.claude/= projects.
-
-*** 2026-06-10 Wed @ 14:13:55 -0500 Noted Phase 1 already shipped; narrowed scope to the phases 2-6 decision
-Phase 1 (the correctness fix) is live: protocols.org documents the AI_AGENT_ID-scoped session-context path (=.ai/session-context.d/<id>.org=) and =.ai/scripts/session-context-path= resolves it. The singleton race Codex flagged is closed. What remains is the spec review plus a go/no-go on the broader runtime-neutral refactor: runtimes/ adapter manifests, generic install commands, language-bundle split, launcher refactor, local model service.
-
-*** 2026-06-11 Thu @ 19:26:26 -0500 Spec amended with the helper-instance slice; implementation split out
-Craig's motivating case (a second Claude in the same project for lookups and safe task updates) was under-specified in v0 — it had identity and message targeting but no spawn mechanics and no write-safety contract for the shared files the session-context split doesn't isolate. Added the "Concurrent same-project agents (helper instances)" section (subagent boundary, identity/spawn via =ai --helper=, the tiered read/write contract, light startup, helper wrap-up) and Phase 1.5 to the migration plan. Implementation filed as its own [#B] task ("Helper-instance support"); this task stays scoped to the phases 2-6 go/no-go.
-
-*** 2026-06-12 Fri @ 02:09:10 -0500 Independent spec review complete
-Codex ran the spec-review workflow. Outcome: the combined spec is =Not ready= because phases 2-5 still require product decisions and current external-runtime/model verification. Phase 1.5 can proceed only as the already-split helper task, with rollout/manual-validation caveats accepted and no accidental template-wide release before sandbox/pilot drills pass. Review file: [[file:docs/design/2026-05-28-generic-agent-runtime-spec-review.org]].
-
-*** 2026-06-12 Fri @ 02:39:38 -0500 Second review after response pass
-Codex re-ran spec-review after the dispositions were folded in. Outcome by arc: Phase 1.5 helper instances =Ready with caveats=; phases 2-5 remain =Not ready= behind the explicit decisions/reverification gate. No new blocking findings for the helper slice. Review file updated in place: [[file:docs/design/2026-05-28-generic-agent-runtime-spec-review.org]].
-
-** DONE [#C] Session title hostname-project, no space :feature:quick:
-CLOSED: [2026-06-13 Sat]
+Regression test: run =wrap-org-table.el= against a file containing a =#+begin_example= block whose lines start with =|= and assert the block is byte-identical afterward. Source: work handoff 2026-07-09 (=inbox/2026-07-09-1341-from-work-bug-data-loss-wrap-org-table-el-and.org=).
+** DONE [#B] Sentry workflow — build from spec :feature:
+CLOSED: [2026-07-19 Sun]
:PROPERTIES:
-:CREATED: [2026-06-13 Sat]
-:LAST_REVIEWED: 2026-06-13
+:SPEC_ID: f6c51f27-d7a2-4b63-9ff9-5ba005a66dfb
:END:
-Routed from the roam global inbox via inbox-zero 2026-06-13. The SessionStart hook (=hooks/session-title.sh=) emitted =<host> <project>= with a space; Craig wanted =<host>-<project>= with a hyphen and no space. Changed the =sessionTitle= join to ="$host-$project"= plus the header comments, and updated the three =session-title-hook.bats= expectations (test-first; 6/6 green).
-
-* Rulesets Resolved
-** DONE [#C] Fix =cj-scan= false positives on cj fences nested inside other =#+begin_*= blocks :bug:
-CLOSED: [2026-05-15 Fri]
-
-=cj-scan.py= was matching =#+begin_src cj:= / =#+end_src= line-by-line
-without awareness of enclosing block scopes. A cj fence embedded inside a
-=#+begin_example= block (typically when documenting what the =<cj= yasnippet
-emits) or inside =#+begin_src snippet= (the yasnippet definition itself) was
-misclassified as a live cj annotation. Surfaced from a /respond-to-cj-comments
-run against the dotemacs =todo.org= that reported two false positives in the
-=<cj= yasnippet documentation.
-
-Fix: track an active =wrapper_type= state. When the scanner sees =#+begin_<type>=
-(for any =<type>= other than =cj:= via the more-specific cj-open regex, which
-is checked first), it enters a wrapper state where every line is treated as
-content until the matching =#+end_<type>= closer fires. Inside a wrapper, cj
-fence patterns and legacy inline =cj:= lines are both suppressed.
-
-Tests: added =TestCjScanNestedFencesIgnored= (6 tests) to
-=claude-templates/.ai/scripts/tests/test_cj_scan.py= covering nesting inside
-=#+begin_example=, =#+begin_src <other-lang>=, and =#+begin_quote=, plus
-regression guards that a wrapper closes cleanly (a subsequent real cj fence
-is still detected) and that an unclosed wrapper doesn't silently swallow
-later content into false-positive cj blocks.
-
-Full =make test-scripts= equivalent (=python3 -m pytest=): 302 passed, 1
-skipped, 0 failures.
-
-** DONE [#A] Add =make doctor= — verify ~/.claude/ matches repo + settings.json :feature:
-
-A drift detector that scans =~/.claude/= and reports anything inconsistent with what the repo expects. Single-command answer to "is my machine consistent with rulesets?"
-
-*** Why this matters
-
-A 2026-05-06 sweep found =~/.claude/hooks/= didn't exist on this machine even though =settings.json= referenced =~/.claude/hooks/precompact-priorities.sh= as a PreCompact hook. Compaction would have silently failed to invoke the hook. The fix was =make install-hooks=, but the breakage was invisible until I happened to grep for it. =make doctor= run regularly (or even as part of session start) would catch this kind of drift in seconds instead of after the fact.
-
-*** Checks
-
-- Every entry in =settings.json= ="hooks"= block points at a file that exists.
-- Every entry in =enabledPlugins= has a matching install under =~/.claude/plugins/data/=.
-- Every skill in =$(SKILLS)= has a working symlink at =~/.claude/skills/<name>=.
-- Every rule in =$(RULES)= has a working symlink at =~/.claude/rules/<name>=.
-- Every default hook has a symlink at =~/.claude/hooks/<name>= (warn-only — opt-out is legitimate).
-- =settings.json= and =.mcp.json= symlinks resolve to the rulesets versions.
-- =mcp/install.py= state matches =claude mcp list= (every server in =servers.json= is registered).
-- No dangling symlinks anywhere under =~/.claude/=.
-
-*** Output
-
-One line per check: =ok= / =WARN= / =FAIL=. Final summary: =N ok, M warnings, K failures=. Exit non-zero on any failure so it can ride a pre-flight check.
-
-** DONE [#A] Build =voice= skill — combine =humanizer= with universal + personal style passes :feature:
-
-Combine =humanizer= with universal good-writing passes (Strunk & White, Orwell, Plain English) and the personal-style passes from =commits.md=. Two modes — =general= for arbitrary writing, =personal= for commits/PRs/comments — share a foundation and diverge on register.
-
-Built and shipped 2026-05-07: =voice/SKILL.md= with 39 numbered patterns walked sequentially. Patterns 1-25 carried over from humanizer, 26-31 are universal good-writing additions, 32-39 are personal-only. Migrated three callers (=commits.md=, =respond-to-cj-comments.md=, =start-work.md=). Removed the standalone =humanizer= skill since voice supersedes it.
-
-*** Why this matters
-
-Three transformations want to run together for personal-mode artifacts (commits, PR titles + bodies, PR comments) but lived in three places: =humanizer= as a skill, S&W-style universal rules nowhere (applied ad-hoc), and the personal-style passes as prose steps in =commits.md= that got re-applied by hand each time. Costs: (1) the "I forgot pass (e)" failure mode — skipping a pass without flagging is a defect but happens in practice. (2) No single-call invocation of the full transform. (3) General-mode writing (research notes, philosophy, history) got only humanizer with no universal-prose pass at all. Combining brings them under one skill with one invocation.
-
-*** Design
-
-Two modes:
-
-- *general* (default) — for arbitrary writing not bound for commit/PR/comment publishing (research notes, philosophy/history essays, emails, README prose). Runs:
- - humanizer (current behavior — strip AI-generated-writing fingerprints)
- - tier-1 universal passes (canonical good-writing rules)
- - the 2 personal-style passes that have no register conflict (jargon-fragment rewrite, noun-ified verbs)
-
-- *personal* — for commits, PR titles + bodies, PR comments. Runs general PLUS:
- - 8 personal-only passes (first-person rewrite, semicolons, contractions, sentence-split, felt-experience, sentence fragments, terse cut, public-artifact scope check)
-
-The 8 personal-only passes are explicitly *not* in general mode. They conflict with academic / literary / philosophical register. Forcing first-person on a Foucault essay or stripping felt-experience from a journal entry would damage the writing.
-
-*** Tier 1 universals (v1)
-
-From Strunk & White, Orwell's "Politics and the English Language", Plain English Campaign, and Garner's Modern English Usage. Each is a detection-pattern + rewrite-rule pair, mechanical enough to apply consistently across runs.
-
-- *Omit needless words* — curated phrase list (=the fact that= → =that=/=because=, =in order to= → =to=, =at this point in time= → =now=, =due to the fact that= → =because=, =for the purpose of= → =to=, =in spite of= → =despite=, etc.)
-- *Long word → short word* — Plain English wordlist (~150 entries: =utilize=→=use=, =commence=→=start=, =terminate=→=end=, =facilitate=→=help=, =demonstrate=→=show=, =sufficient=→=enough=, =prior to=→=before=, =subsequent to=→=after=, =in the event that=→=if=, =a great deal of=→=much=)
-- *Active over passive voice* — detect "to be + past-participle" patterns. Suggestion-only in v1 (auto-rewrite is risky in technical contexts where passive is appropriate); graduate to auto-rewrite for unambiguous cases in v2.
-- *Comma splices* — detect independent clauses joined only by comma; rewrite to period or semicolon-then-period.
-- *Cliché flag* — small curated list (=at the end of the day=, =moving forward=, =going forward=, =at this juncture=, =circle back=, =low-hanging fruit=, =deep dive=, =leverage= as verb).
-
-*** Tier 2 universals (v2)
-
-- *Positive over negative form* (S&W) — =not unlike= → =like=, =do not fail to= → =remember to=, =did not pay any attention= → =ignored=
-- *Garner-style word-pair corrections* — comprise/compose, less/fewer, that/which (restrictive vs nonrestrictive), affect/effect, principal/principle
-- *Parallelism in lists* — detect mismatched grammar in bullet items
-- *Tense consistency* — flag mid-paragraph tense shifts
-- *Acronym definition on first use* — detect uppercase tokens used before being expanded
-
-*** Tier 3 (v3, may not land)
-
-- *Concrete-over-abstract* preference
-- *Emphatic word at sentence end* (S&W rule 18)
-- *Vary sentence length / rhythm*
-- *Reading-grade-level scoring* (Hemingway-style)
-
-*** Personal-style pass placement
-
-| # | Pass | Mode | Why |
-|----+-------------------------------------+-------------------------------------+-------------------------------------|
-| 1 | First-person voice rewrite | personal only | Forces "I" voice; wrong for |
-| | | | academic prose where third-person |
-| | | | and "we" are conventional |
-|----+-------------------------------------+-------------------------------------+-------------------------------------|
-| 2 | Jargon-fragment → complete sentence | both | Universal clarity, no genre |
-| | | | conflict |
-|----+-------------------------------------+-------------------------------------+-------------------------------------|
-| 3 | Semicolon → period/comma | personal only | Semicolons are conventional in |
-| | | | long-form / academic prose |
-|----+-------------------------------------+-------------------------------------+-------------------------------------|
-| 4 | Contractions ("it's", "don't") | personal only | Academic and formal writing |
-| | | | typically avoids contractions |
-|----+-------------------------------------+-------------------------------------+-------------------------------------|
-| 5 | Sentence split on conjunctions | personal only | Foucault, Hegel, Adorno |
-| | | | deliberately use long compound |
-| | | | sentences |
-|----+-------------------------------------+-------------------------------------+-------------------------------------|
-| 6 | Felt-experience narration ("I'll | personal only | Personal essays *use* |
-| | feel this every time") | | felt-experience as content |
-|----+-------------------------------------+-------------------------------------+-------------------------------------|
-| 7 | Noun-ified verbs ("the ask", "a | both | Targets corporate-speak with |
-| | learn", "the spend") | | curated wordlist; doesn't catch |
-| | | | philosophical nominalizations like |
-| | | | "the becoming" |
-|----+-------------------------------------+-------------------------------------+-------------------------------------|
-| 8 | Sentence fragments → complete (in | personal only | Fragments are valid stylistic |
-| | prose) | | devices in literary prose |
-|----+-------------------------------------+-------------------------------------+-------------------------------------|
-| 9 | Terse cut (rhetorical padding: | personal only | Tier 1 omit-needless-words covers |
-| | "worth noting", "it's important to | | the worst offenders universally; |
-| | understand") | | aggressive cut conflicts with |
-| | | | academic register |
-|----+-------------------------------------+-------------------------------------+-------------------------------------|
-| 10 | Public-artifact scope check (local | personal only — *flag-only*, no | Operational/safety check, not |
-| | paths, private repos, personal | auto-rewrite | stylistic; auto-masking risks |
-| | tooling) | | silently editing meaningful text |
-|----+-------------------------------------+-------------------------------------+-------------------------------------|
-
-*** Inclusive-language pass — explicitly excluded
-
-Considered and rejected. Conflicts with planned writing on philosophy/history topics (Foucault on sexuality and gender, history of slavery in New Orleans). Wordlist substitutions would override deliberate vocabulary choices in those genres.
-
-*** V1 scope
-
-- [ ] Skill at =~/code/rulesets/voice/= with =SKILL.md=
-- [ ] Frontmatter with positive triggers (commit, PR, comment, "humanize", "voice pass") and negative triggers (code, structured data, plain bullet lists)
-- [X] Mode invocation: default = =general= when invoked bare; =personal= invoked explicitly by publish-context callers
-- [X] humanizer content migrated from =humanizer/= → =voice/=
-- [X] Tier 1 universal passes implemented (5 patterns: #26-30, plus #31 noun-ified verbs as a universal personal addition)
-- [X] 2 personal passes that run in both modes (#30 jargon-fragment, #31 noun-ified verbs)
-- [X] 8 personal passes that run in personal mode only (#32 first-person, #33 semicolons, #34 contractions, #35 sentence-split, #36 felt-experience, #37 fragments, #38 terse cut, #39 scope check)
-- [X] Each pass = detection-pattern + rewrite-rule pair (#39 is detection + flag-only)
-- [X] Total v1 pattern count: 31 in general mode (humanizer's 25 + 4 tier-1 + 2 universal personal); +8 personal-only = 39 in personal mode
-- [X] Update =commits.md= to invoke =/voice personal= instead of "run =humanizer= and apply five passes manually"
-- [X] Remove the existing =humanizer/= skill (no callers outside this repo, all migrated)
-- [X] =make doctor= still passes
-- [X] =make lint= clean
-
-*** v2 (deferred)
-
-- [ ] Tier 2 universals (positive form, word-pair corrections, parallelism, tense consistency, acronym definition)
-- [ ] Per-pass severity flags for Tier 1 active-voice (suggestion-only when actor is implicit; auto-rewrite when actor is named)
-- [ ] Reporting mode: list which passes fired and which were no-ops
-
-*** v3 (aspirational, may not land)
-
-- [ ] Tier 3 (concrete-over-abstract, emphatic-word position, sentence-length variation, reading-grade scoring)
-- [ ] Progressive disclosure split: =voice/SKILL.md= orchestrator + =voice/passes/<pass-name>.md= per pass with worked examples
-
-*** Migration (resolved)
-
-Decision: deleted =humanizer/= entirely. Three callers (=commits.md=, =respond-to-cj-comments.md=, =start-work.md=) all updated to invoke =/voice= directly. No alias needed since nothing outside the repo invoked humanizer.
-
-*** Naming alternatives considered
-
-- =voice= — chosen. Captures both modes; broad enough.
-- =polish= — descriptive of multi-pass nature; less prescriptive about whose voice.
-- =house-style= — signals "this is the house style"; appropriate for personal repo.
-- =commit-voice= — too narrow (passes apply to research notes, emails, etc. in general mode).
-- =humanize= (extending current) — undersells the universal + personal additions.
-
-*** Open questions before implementation
-
-Resolved during implementation:
-- Default mode when =/voice= is invoked bare: =general=. Personal-context callers (=commits.md= publish flow, =respond-to-cj-comments.md=) invoke =/voice personal= explicitly. Avoids accidentally first-person-ifying research notes.
-- Reporting: skill prints "Summary of changes" listing which patterns fired (audit value).
-- Public-artifact scope check (#39): flag-only, user resolves manually. Blocking would frustrate on legitimate path mentions.
-- Tier 1 active-voice detection: suggestion-only in v1. Auto-rewrite for unambiguous cases deferred to v2.
-
-** DONE [#B] Add =--archive-done= mode to =.ai/scripts/todo-cleanup.el= :feature:
-
-Opt-in mode that moves every level-2 subtree whose TODO state is DONE or CANCELLED out of the "Open Work" section and into the "Resolved" section of the same org file, subtree intact.
-
-- *Section matching.* Key on a top-level heading containing "Open Work" and one containing "Resolved" — that pairing is the only naming consistent across projects (=Work Open Work= / =Work Resolved= here; bare =Open Work= / =Resolved= elsewhere). Require exactly one match for each; otherwise skip with a clear message, no crash.
-- *Modes.* =--check= previews and writes nothing, same as the existing hygiene pass. Idempotent. Not run by default in the wrap-up flow — archiving is consequential, so it stays opt-in: =emacs --batch -q -l todo-cleanup.el --archive-done FILE=.
-- *Edge cases.* Source or target section missing; subtree at EOF; nested DONE subtree under an open parent stays put (only level-2 entries move); nothing to move → clean no-op.
-- *Tests.* TDD with ERT — the project's first elisp tests. Fixtures (synthetic) under =.ai/scripts/tests/=; run via =make test= (rulesets) or =make test-scripts= (claude-templates), which run pytest + every =tests/test-*.el= ERT suite. Cases: one DONE level-2 moves; multiple; CANCELLED also moves; structural (no-state) headings don't move; nested DONE under an open parent stays; level-2 DONE with open level-3 children moves intact; subtree at EOF; missing source/target section; ambiguous "Resolved"; lowercase headings; nothing-to-do; idempotency; =--check= preview + its idempotency; realistic-sample integration.
-
-Origin: came up while scrubbing a project's todo.org on 2026-05-11 — moving a big completed PROJECT subtree (plus a few smaller ones) into the Resolved section by hand was the cue to build a reusable tool.
-
-Built and shipped 2026-05-11: =--archive-done= added to =.ai/scripts/todo-cleanup.el= test-first; 13-test ERT suite (=tests/test-todo-cleanup.el=) + realistic synthetic fixture (=tests/fixtures/todo-sample.org=), wired into =make test= / =make test-scripts= alongside pytest. The CLI dispatch moved into =tc-main= behind a guard so the suite can =require= the file without firing it. Section matching is case-insensitive and tolerates the =<Project> Open Work= / =<Project> Resolved= naming variants. Opt-in only — not wired into the wrap-up flow. Source of truth is =~/projects/claude-templates/=; rsync'd into this repo.
-
-** DONE [#B] Encode follow-up filing rules into =/start-work=
-CLOSED: [2026-05-15 Fri]
-
-Phase 4 step 5 of =/start-work= ("refactor audit") says any candidate that isn't fix-now must land in one of three buckets: fold-into-related-commit, separate =refactor:= commit, or "file a ticket or todo.org entry." The third disposition doesn't say *where* — which leaves the orchestrator picking a location ad-hoc. Result: follow-ups buried under children of an epic parent get orphaned when the parent closes, or follow-ups for standalone tasks scatter across the file with no convention.
-
-Proposed placement rule (already memorized for this project as =feedback_followups_as_siblings.md=, generalizing):
-
-- *Epic-style parent task* (level-2 with multiple level-3 children) → follow-ups file as level-2 *siblings* of the parent. Stays visible after parent closure.
-- *Standalone task* (level-2 with no children, or a level-3 inside another structure) → follow-up files as a new level-2 top-level entry in the same =* Open Work= section. Don't nest under the originating task.
-
-Both cases: include a "Triggered by: <date> <task or commit>" line so a future reader sees what surfaced it.
-
-Update =.claude/commands/start-work.md= Phase 4 step 5's "Disposition for each candidate" section to spell this out. Update any cross-references in =commits.md= or other files that touch the discipline.
-
-Triggered by: 2026-05-15 fold-epic session — Craig flagged the gap mid-flight after I'd surfaced a follow-up but hadn't filed it.
-** DONE [#A] Consolidate =.ai/= template infrastructure (fold + audit + install-ai + ratio) :feature:
-CLOSED: [2026-05-15 Fri]
-
-End-state: one repo (=rulesets=) is the single source of truth for =.ai/= template content. =make audit= verifies and applies drift across every =.ai/=-using project on the machine. =make install-ai= bootstraps new projects. Same setup propagated to ratio so both machines run the same way.
-
-Today (2026-05-15) the canonical-source rule got violated again: rulesets commit =372fb76= added a wrap-up subsection to =rulesets= without going through =claude-templates= first, and the next session's startup rsync was about to silently undo it. Two-repo coordination is the root cause; fold solves it.
-
-Build order: fold first (others depend on the new canonical path), then audit + install-ai in parallel, then test, then propagate to ratio.
-
-*** DONE [#A] Fold =claude-templates= into rulesets
-CLOSED: [2026-05-15 Fri]
-
-Two repos, one source of truth. =~/projects/claude-templates/= is the canonical =.ai/= template that gets rsync'd into every project at session start. Keeping it standalone means a second =git pull= in startup Phase A.0, a second remote to push to at wrap-up, and a split history any time a change touches both. Folding it into =rulesets/claude-templates/= gives one repo to clone on a fresh machine and one place to edit templates.
-
-**** Open design choices
-
-- *History.* =git subtree add --prefix=claude-templates ~/projects/claude-templates main= preserves the 84-commit history under the new prefix. Plain content copy (=cp -a= + =git add=) is simpler but loses history. Either is fine since the standalone repo stays archived on =cjennings.net=.
-- *Layout.* =rulesets/claude-templates/= mirrors the old repo name and sits next to =claude-rules/= cleanly. Alternative: absorb =.ai/= directly under a different name (=rulesets/.ai-template/= or similar). First option is clearer.
-- *bin/ai.* The standalone Makefile symlinks =$HOME/.local/bin/ai → bin/ai=. After the move, fold that into rulesets' Makefile as another install target.
-
-**** Mechanical steps
-
-1. Subtree-merge or copy =~/projects/claude-templates/= into =rulesets/claude-templates/=.
-2. Update 3 references in rulesets:
- - =.ai/protocols.org= line 163 — pointer in the "Let's run/do the X workflow" section.
- - =.ai/workflows/cross-agent-comms.org= line 8 — promotion-target path.
- - =.ai/workflows/startup.org= lines 22, 96-98 — Phase A.0 pull + Phase A rsync sources.
-3. Update Phase A.0 of =startup.org= to pull rulesets instead of claude-templates. Inside rulesets sessions, the existing project-repo pull already covers it. Outside rulesets (every other project's session), Phase A.0 needs an explicit =git pull= on =~/code/rulesets/= before the rsync — otherwise the templates will be stale.
-4. Replace =~/projects/claude-templates/= with a symlink to =~/code/rulesets/claude-templates/= for transition continuity.
-5. After every active project has had one session start (and rsync'd the new =startup.org=), drop the symlink and archive =cjennings.net:git/claude-templates.git=.
-
-**** Bootstrap gap
-
-Every project on the machine has a =.ai/workflows/startup.org= that rsyncs from =~/projects/claude-templates/=. Until each project's startup.org gets refreshed (which happens via the rsync itself), the old path needs to keep resolving. The symlink at step 4 is the bridge: old paths resolve into the new location, the rsync delivers the updated startup.org, next session uses the new path directly.
-
-*** DONE [#A] Add =make audit= — drift detector across all =.ai/=-using projects
-CLOSED: [2026-05-15 Fri]
-
-Companion to =make doctor= (single-machine scope, checks =~/.claude/=). =audit= is cross-project scope: walks every directory on the machine that has a =.ai/=, diffs the synced template files against the canonical source, and reports drift. =--apply= flag rsyncs the drift into the project's working tree (no auto-commit). Catches stale projects without forcing a session start in each one.
-
-**** Open design choices
-
-- *Scope.* Template-sync drift is the useful flavor: for each project, diff =.ai/protocols.org=, =.ai/workflows/=, =.ai/scripts/= against the canonical source.
-- *Source path.* Post-fold: =~/code/rulesets/claude-templates/.ai/=. Build =audit= against the new path from day one.
-- *Project discovery.* Walk =~/code/=, =~/projects/=, =~/.emacs.d/= up to depth 3 for any directory containing =.ai/=. Skip the canonical source itself.
-- *Default mode is report-only.* =--apply= triggers rsync; =--force= overrides the dirty-skip safety.
-
-**** Per-project flow (designed 2026-05-15)
-
-For each discovered project, in order:
-
-1. Verify =.ai/= exists (path probe). If missing → =FAIL=, skip, continue loop.
-2. Detect git tracking via =git check-ignore .ai/= → =tracked= or =gitignored=.
-3. Verify no uncommitted =.ai/= changes (=git status --porcelain .ai/=). Dirty → =WARN=, skip rsync unless =--force=.
-4. Verify content matches canonical via three =rsync -a --dry-run --itemize-changes= calls (=protocols.org=, =workflows/=, =scripts/=). Zero items = clean.
-5. Action (=--apply= only, drift detected): three =rsync -a [--delete]= calls.
-6. Verify rsync converged (re-run the dry-runs; zero now).
-7. Verify working-tree state after rsync (tracked projects). Report deltas. Do not auto-commit.
-8. Verify no unpushed =.ai/= commits (=git log @{u}..HEAD -- .ai/=). Informational only.
-
-**** Output format (mirrors =doctor=)
-
-#+begin_example
-Claude-templates source:
- ok rulesets/claude-templates is current (origin/main)
-
-Per-project .ai/ drift:
- ok ~/projects/work
- applied ~/projects/homelab 3 files changed
- skipped ~/code/winvm uncommitted .ai/ (use --force)
- ok ~/projects/clipper
-
-Summary: 18 ok, 3 applied, 1 skipped, 0 failed
-#+end_example
-
-Exit code: =0= if all clean, no skips, no failures. =1= otherwise.
-
-**** Why not extend =make doctor= instead
-
-=doctor= has a clean meaning today: "is this machine's =~/.claude/= consistent with rulesets?" Mixing in cross-project =.ai/= drift muddies the exit code. Keep them separate. =audit= can optionally invoke =doctor= as its last check since both ask "did the symlinks keep up with the source?". A future =make all-checks= can wrap both.
-
-*** DONE [#A] Add =make install-ai PROJECT=<path>= — bootstrap =.ai/= in a fresh project
-CLOSED: [2026-05-15 Fri]
-
-Separate target from =audit= because operating on projects that lack =.ai/= is a distinct action. The absence might be intentional, so =audit= skips them. Bootstrap is explicit opt-in.
-
-**** Flow
-
-1. Refuse if =.ai/= already exists in =PROJECT=. Message: "already installed; use =make audit --apply= to update."
-2. Verify =PROJECT= is a git checkout (warn if not — works without git, loses some lifecycle benefits).
-3. Create =PROJECT/.ai/= directory.
-4. Rsync canonical content: =protocols.org=, =workflows/=, =scripts/= (same three rsyncs as =audit=).
-5. Seed =PROJECT/.ai/notes.org= from a canonical template with project-name placeholder.
-6. Create empty =PROJECT/.ai/sessions/= (with =.gitkeep= for tracked projects).
-7. Track or gitignore =.ai/=? Default: ask. Flag: =--track= / =--gitignore=.
-8. Print next-steps banner: =make install-lang LANG=<lang> PROJECT=<path>=; open Claude Code in the project.
-
-**** Symmetry with existing install targets
-
-#+begin_example
-make install-lang LANG=python PROJECT=/path # language bundle (existing)
-make install-ai PROJECT=/path # .ai/ template (new)
-make install-lang # no args → fzf-pick
-make install-ai # no args → fzf-pick from
- # ~/projects/* + ~/code/* dirs
- # without an existing .ai/
-#+end_example
-
-*** DONE [#A] Test plan for audit + install-ai before propagating to ratio
-CLOSED: [2026-05-15 Fri]
-
-Test against the current state of this machine before pushing changes to ratio.
-
-**** =make audit= tests
-
-1. Dry-run report only (no =--apply=). Should show: claude-templates current; per-project drift; correct =ok=/=drift= classifications; summary line and exit code match.
-2. After the fold lands, every project should be reported as drift (their =startup.org= still points at the old path). Run =--apply= → rsync converges. Re-run audit → all =ok=.
-3. Manually edit one =.ai/workflows/foo.org= in a tracked project. Re-run audit → should report =skipped: uncommitted .ai/=. Run =--apply --force= → rsync clobbers the edit. Verify the edit is gone.
-4. Manually delete one =.ai/= dir. Re-run audit → =FAIL: .ai/ missing=. Loop continues.
-5. Idempotency: =--apply= twice in a row converges to all =ok= on the second pass.
-
-**** =make install-ai= tests
-
-1. Create =/tmp/test-fresh-project= as a git repo. Run =make install-ai PROJECT=/tmp/test-fresh-project=. Verify =.ai/= structure matches canonical, =notes.org= has placeholder, =sessions/= exists.
-2. Run =make install-ai PROJECT=/tmp/test-fresh-project= again → should refuse (=.ai/= already exists).
-3. Open Claude Code in the new project. Startup workflow runs cleanly (Phase A.0 + Phase A rsync should be a no-op since the install just ran).
-4. fzf form: =make install-ai= with no args. Lists candidate dirs (=~/projects/*=, =~/code/*= without =.ai/=).
-
-**** Pass criteria
-
-- =audit= behavior matches the per-project flow spec for every classification path.
-- =install-ai= produces a project indistinguishable from one that's been running sessions for a while.
-- =make doctor= still passes 36/0/0 after all the work.
-- =make test= (pytest + ERT) passes.
-
-*** DONE [#A] Migrate projects on ratio (second machine)
-CLOSED: [2026-05-15 Fri]
-
-After local fold + audit + install-ai are working, propagate to ratio.
-
-**** Steps
-
-1. On ratio: =git -C ~/code/rulesets pull= — picks up the folded =claude-templates/= subdir and updated =Makefile= targets.
-2. On ratio: archive or =mv= the standalone =~/projects/claude-templates/= aside, replace with symlink to =~/code/rulesets/claude-templates/= (same bridge mechanic as local).
-3. On ratio: =make audit= → see drift across ratio's projects.
-4. On ratio: =make audit --apply= → rsync into each tracked/gitignored project. Surface projects with uncommitted =.ai/= drift for manual handling.
-5. On ratio: =make doctor= → catch any =~/.claude/= install drift (likely some, since ratio hasn't seen recent rulesets updates).
-6. Verify by opening Claude Code in a few ratio projects. Startup should be a no-op or near-zero rsync.
-
-**** Known unknowns
-
-- Ratio may have its own project list overlapping with this machine's but not identical. =audit= discovers projects via the walk, so this is automatic.
-- Ratio might have uncommitted =.ai/= work in some projects that this machine doesn't. =audit= surfaces them; handle case-by-case.
-- If anything goes wrong, ratio's archived =~/projects/claude-templates/= is the safety net — restore the symlink target and re-run audit.
-
-**** Adjacent: cross-machine memory sync
-
-The =[#A] DOING= memory-sync investigation (todo.org:10) is adjacent. Both involve "make my Claude setup portable across machines." Coordinate so the memory-sync stow approach (if approved) doesn't conflict with this fold's symlink mechanics.
-** DONE [#B] Document startup pull-ordering rule in protocols.org
-CLOSED: [2026-05-15 Fri]
-
-Phase A.0 of =startup.org= now pulls rulesets ff-only before the project repo
-(shipped 2026-05-15 as part of the claude-templates fold — after the subtree
-merge, there's no separate claude-templates pull, just rulesets-then-project).
-The protocols.org paragraph stating the ordering and "resolve any issues
-before proceeding" rule shipped 2026-05-15 in the =** Startup Pull Ordering=
-subsection under =IMPORTANT - MUST DO=.
-** DONE [#A] Build =/lint-org= skill + wrap-up integration
-CLOSED: [2026-05-14 Thu]
-
-Spec: [[file:.ai/specs/lint-org-skill-spec.md]]
-
-A two-mode skill (=interactive=, =mechanical-only=) that runs =org-lint=,
-auto-fixes safe categories (item-number, missing-language-in-src-block,
-misplaced-planning-info, markdown-bold → single-asterisk), and walks judgment
-items (broken local-file links, invalid fuzzy links, verbatim-asterisk false
-positives, suspicious-language blocks) inline.
-
-Wrap-up integration: =wrap-it-up.org= invokes
-=/lint-org todo.org --mode=mechanical-only= after the existing
-=todo-cleanup.el --archive-done= pass. Judgment items defer to a
-carry-forward file that the next morning's daily-prep merges in, so
-wrap-up never blocks on a judgment call.
-
-Baseline that motivated this: the 2026-05-14 manual pass took =todo.org=
-from 55 → 1 lint warnings across two commits (=0d10458= signal,
-=9ad5b30= cosmetic). A nightly mechanical sweep keeps the count near
-zero forever — each day's drift is small.
-** DONE [#C] Test harness for =make audit= + =make install-ai= edge cases :test:
-CLOSED: [2026-05-15 Fri]
-
-Three edge cases from the fold-epic test plan were not exercised because they're destructive on real projects:
-
-- =audit --force= clobbers uncommitted =.ai/= work — needs a project with intentionally dirty =.ai/= to verify the override path.
-- =audit= reports =FAIL= when =.ai/= is missing — needs a project where the directory was deleted to verify the loop continues past the failure.
-- =install-ai= fzf-pick form (no =PROJECT= arg) — needs interactive testing.
-
-Build a self-contained test harness under =.ai/scripts/tests/= that spins up =/tmp/audit-test-projects/= with a known matrix of project states (clean, dirty, missing =.ai/=, pristine, etc.), runs the audit + install-ai targets against it, and asserts expected outputs. The harness should clean up after itself.
-
-Pattern reference: bats or shell-based assertions (similar to the elisp ERT suites for =todo-cleanup= and =lint-org=, but for shell scripts).
-
-Triggered by: 2026-05-15 fold-epic, child 4 test plan; commits =94782ee= (audit) + =d364cf2= (install-ai).
-** DONE [#A] wrap it up mentions github, which isn't the remote for many projects. :chore:
-CLOSED: [2026-05-16 Sat]
-For many of them, git.cjennings.net mirrors to github.com, and github.com isn't the remote.
-For many others, git.cjennings.net is the remote with no mirror.
-Remove or replace the reference to github.com
-** DONE [#B] Phase A startup blind to =claude-templates/inbox/= post-fold :bug:fold:
-CLOSED: [2026-05-19 Tue]
-
-Resolved on inspection: the bug is moot in current state. =inbox-send.py='s discovery scans =~/code/*= and =~/projects/*= single-level only, so =claude-templates/= (two levels under =~/code/=) is never a routable target; the 2026-05-15 incident was a one-time manual workaround because =rulesets/inbox/= didn't exist yet, and that root inbox was added in =470085f=. =claude-templates/inbox/= was removed 2026-05-15 and is no longer on disk.
-
-Phase A's inbox check at =startup.org:107= runs =\ls -la inbox/= against the project root. Post-fold, the canonical's inbox sits inside the subtree at =claude-templates/inbox/= and never gets scanned. A 2026-05-15 cross-project handoff from a dotemacs session dropped a record there; the next rulesets session (this one) missed it at startup entirely. Picked up only when the working-tree drift surfaced during the publish flow.
-
-Fix: extend Phase A's discovery to also scan =claude-templates/inbox/= when the canonical lives in-repo (i.e., when =claude-templates/.ai/= exists alongside =./.ai/=). The Phase B/C inbox-processing flow already handles per-file routing once a file is surfaced; the gap is only in discovery.
-
-Adjacent question worth answering at the same time: should cross-project handoffs file into =./inbox/= at the project root (matching what Phase A already scans), or stay in =claude-templates/inbox/= and rely on the discovery fix? The =inbox-send= script's target-project logic is the place to settle that.
-
-Triggered by: 2026-05-15 evening session, surfaced when committing the test-harness work.
-** DONE [#A] Implement task-review daily-habit per spec
-CLOSED: [2026-05-20 Wed]
-:PROPERTIES:
-:LAST_REVIEWED: 2026-05-20
-:END:
-Spec: [[file:docs/design/task-review.org]]
-
-Retires =wrap-it-up.org='s date-coverage scan and replaces it with a daily list-hygiene review (N=7 oldest-unreviewed top-level =[#A]= / =[#B]= / =[#C]= tasks per session, ~12-day rotation). Built as a pure Claude workflow — Shape B, no elisp; see the spec's Revision section for why the elisp approach was dropped.
-
-Status:
-1. [X] =task-review-staleness.sh= + bats (count + =--list= modes).
-2. [X] =wrap-it-up.org= health check (threshold 30).
-3. [-] =task-review.el= — dropped (Shape B is a pure workflow, not an Emacs mode).
-4. [X] New =task-review.org= workflow + INDEX entry (the existing listing workflow was renamed to =open-tasks.org= to free the name).
-5. [X] Startup nudge in template =startup.org= (threshold 7), not the project-only startup-extras layer.
-6. [X] Smoke test against live =todo.org= — first cycle run 2026-05-20 (7 tasks reviewed: 3 re-grades, 1 cancellation, 1 bump-and-tag).
-
-Triggered by: 2026-05-16 brainstorm on retiring the date-coverage scan.
-** CANCELLED [#B] Build =ov-1= skill for DoDAF OV-1 (High-Level Operational Concept Graphic)
-CLOSED: [2026-05-20 Wed]
-
-Cancelled during the 2026-05-20 task review.
-
-Triggered by SOFWeek (May 2026, Tampa) — DeepSat attending; DoD attendees
-may ask for architecture diagrams. OV-1 is the universal informal
-currency in DoD briefings ("show me the architecture" → OV-1 by default).
-
-Priority upgrades to =[#A]= if Craig confirms scenario 2 below (personal
-load-bearing need at the event); stays =[#B]= or drops to =[#C]= if
-scenario 1 (team already covers it, future asset only).
-
-*** Prior art (searched 2026-04-19)
-
-No existing Claude Code skill exists for DoDAF / OV-1 / SV-1 / SysML.
-
-- =anthropics/skills= — 17 skills, zero DoDAF/SysML/defense coverage.
-- =awesome-claude-code= list — zero hits for DoDAF/OV-1/SysML/UAF.
-- =mfsgr/sysml2dodaf= — empty repo (0 stars, no code). Vapor.
-- =HowardKao-1130/mini-NEXEN= — broad SE methodology skill that
- name-drops DoDAF as a trigger keyword; no artifact generation. 0 stars.
-- =gaphor/gaphor= (Apache-2.0, 2.2k stars) — mature UML/SysML GUI
- modeler. Not a skill; not a pipeline. Useful reference only.
-
-Nearest prior art to lean on when building:
-- DoDAF 2.02 Viewpoints & Models reference (dodcio.defense.gov) —
- canonical OV-1 exemplars. Embed 3-5 layouts as skill =references/=.
-- Pattern from existing =c4-diagram= skill — same shape (prose → diagram
- spec), swap the viewpoint vocabulary to DoDAF.
-- PlantUML for SV-1 (when that skill comes later); Mermaid or draw.io
- XML for OV-1 lightweight visuals.
-
-*** Build scope (when triggered)
-
-*In scope:*
-- Input: prose description of a system + its operational context.
-- Output: structured OV-1 *spec* — performers, external actors (other
- systems, forces, adversaries), relationships (data/control flows),
- narrative captions, classification marking, legend requirements.
-- DoDAF 2.02 completeness checklist as a quality gate — verify the
- produced spec contains every element a correct OV-1 requires.
-- Optional lightweight visual: draw.io XML or Mermaid approximation for
- quick review; NOT a finished rendering.
-
-*Out of scope:*
-- Icon libraries, pictorial assets, finished PowerPoint export. OV-1
- final art belongs to a designer or Craig in Visio/PowerPoint; the
- skill's job is the spec and the check, not the slide.
-- SV-1, SV-2, UAF, IDEF1X, other viewpoints. Build only when a
- concrete need triggers each.
-
-Estimate: 4-6 hours.
-
-*** Craig's investigation before kickoff
-
-1. Does DeepSat's systems-engineering or marketing team already have an
- OV-1 (or the equivalent briefing artifact) for SOFWeek?
-2. If yes (scenario 1) — skill is a future asset, not event-load-bearing.
- Ship after SOFWeek. Priority drops to =[#C]=.
-3. If no, or if the scenario is "Craig may need to produce/iterate an
- OV-1 on the fly during the event" (scenario 2) — skill is load-bearing
- for the event. Priority upgrades to =[#A]=; build before SOFWeek.
-4. Confirm the classification level the skill needs to handle
- (unclassified-only? or FOUO markings? affects the classification
- block in the spec).
-5. Confirm the target rendering format DeepSat uses for OV-1
- deliverables (PowerPoint slide? Cameo? Visio? affects whether the
- skill emits draw.io XML vs Mermaid vs pure structured spec).
-
-*** Related
-
-See also the DoD-specific notations section under the later TODO
-(=c4-*= rename revisit) — OV-1 is flagged there as the highest-value
-starting point across the DoD notation landscape (SysML, DoDAF/UAF,
-IDEF1X). This entry is the execution plan for that starting point.
-** DONE [#A] Split team-specific publishing rules out of commits.md :commits:
-CLOSED: [2026-05-22 Fri]
-Shipped 3cb467e. Moved the DeepSat publishing steps (Linear ticket-state, the Slack notification protocol + channel ID, the GHE host, the team merge norm, the Linear ticket-body structure) out of the global =claude-rules/commits.md= into =teams/deepsat/claude/rules/publishing.md=. The global file keeps the universal skeleton and uses seams ("run the project's publishing overlay here if present") like startup-extras. Added =install-team= (targeted per-project copy, keyed on PROJECT, never globally symlinked) and generalized =sync-language-bundle.sh= to keep team overlays fresh at startup (3 new bats; make test green).
-
-Remaining deploy step (cross-project, surfaced to Craig): install the overlay into the DeepSat work project — =make install-team TEAM=deepsat PROJECT=<deepsat-path>= — so it actually loads there.
-** DONE [#A] Define a /voice-unavailable fallback in the commits.md publish flow :commits:
-CLOSED: [2026-05-22 Fri]
-Added an "If =/voice= is unavailable" paragraph to the Single-skill gate in =commits.md=: walk the same patterns inline (the flow already names which matter), state the skill was unavailable and the pass was applied by hand ("/voice unavailable — patterns walked inline"), and flag the missing skill for install. The gate is the pattern walk, not the tooling. The original "=humanizer= unavailable" framing was moot (humanizer → /voice).
-** DONE [#A] wrap-it-up Step 3.5 assumes GitHub-family remote :chore:quick:
-CLOSED: [2026-05-22 Fri]
-:PROPERTIES:
-:LAST_REVIEWED: 2026-05-20
-:END:
-Documented the assumption inline at =wrap-it-up.org= Step 3.5 (chose the lightweight path over a provider-agnostic rewrite): the =gh= lookup expects a GitHub-family host, holds today via DeepSat on GHE, flagged for update if a future Linear project lands on GitLab/Gitea/Bitbucket.
-Triggered by: 2026-05-16 wrap-it-up github.com cleanup (audit of the same file).
-
-Step 3.5 (Linear ticket-state hygiene) at =wrap-it-up.org:207= says "the project's GitHub remote — use =gh pr list ...=". Currently fine in practice: the step is Linear-gated, and the only Linear-using project is DeepSat (on =deepsat.ghe.com=, a GitHub-family host where =gh= works). Would break if a future Linear-using project lived on a non-GitHub host (gitlab, gitea, bitbucket). Either drop the GitHub-family assumption (provider-agnostic lookup, harder) or document the assumption explicitly so future projects know the step needs an update if they don't fit.
-** DONE [#C] Review pass: tighten skills and rulesets after 2026-05-04 audit
-CLOSED: [2026-05-22 Fri]
-:PROPERTIES:
-:LAST_REVIEWED: 2026-05-20
-:END:
-All 55 grouped-index items dispositioned (2026-05-22): ~49 edited across skills, commands, rule files, hooks, and the two playwright skills; several came out moot post-audit (humanizer→voice, skills→commands, typescript ruleset added); the two commits.md items shipped as the team-overlay split + /voice fallback. Freshness-checked each item against current reality before editing.
-
-Source notes used in this pass:
-- C4 official docs: C4 is notation-independent; System Context and Container
- diagrams are enough for most teams; every diagram needs title, key/legend,
- explicit element types, and audience-appropriate abstraction.
- [[https://c4model.com/diagrams][C4 diagrams]],
- [[https://c4model.com/diagrams/notation][C4 notation]],
- [[https://c4model.com/abstractions/component][C4 component]]
-- arc42 docs: quality requirements need measurable scenarios; section 10
- should reference top quality goals and capture lesser quality requirements
- with specific measures. [[https://docs.arc42.org/section-10/][arc42 section 10]],
- [[https://quality.arc42.org/articles/specify-quality-requirements][specifying quality requirements]]
-- ADR references: ADRs capture one justified architecturally significant
- decision and its rationale; Nygard's original guidance emphasizes short,
- numbered, repository-stored records and superseding rather than rewriting old
- decisions. [[https://adr.github.io/][adr.github.io]],
- [[https://cognitect.com/blog/2011/11/15/documenting-architecture-decisions][Nygard ADR article]]
-- Playwright docs: prefer user-visible locators and web assertions; locators
- auto-wait and retry; =networkidle= is discouraged for testing readiness.
- [[https://playwright.dev/docs/best-practices][Playwright best practices]],
- [[https://playwright.dev/docs/locators][Playwright locators]],
- [[https://playwright.dev/docs/next/api/class-page][Playwright page API]]
-- OWASP references: Top 10 2021 includes Broken Access Control,
- Cryptographic Failures, Injection, Insecure Design, Security
- Misconfiguration, Vulnerable and Outdated Components, Identification and
- Authentication Failures, Software and Data Integrity Failures, Security
- Logging and Monitoring Failures, and SSRF; WSTG adds a broader testing map
- across configuration, identity, authn/z, sessions, input validation, error
- handling, cryptography, business logic, client-side, and API testing.
- [[https://owasp.org/Top10/2021/][OWASP Top 10 2021]],
- [[https://owasp.org/www-project-web-security-testing-guide/latest/4-Web_Application_Security_Testing/][OWASP WSTG]]
-- V2MOM references: Salesforce calls the last M "Measures" and emphasizes a
- simple alignment document with prioritized Methods, explicit Obstacles, and
- measurable outcomes. [[https://trailhead.salesforce.com/content/learn/modules/selfmotivation/get-focused-with-your-personal-v2mom][Salesforce Trailhead personal V2MOM]],
- [[https://www.salesforce.com/blog/?p=12][Salesforce V2MOM alignment]]
-- Prompt research: the cited Meincke paper is titled "Call Me A Jerk:
- Persuading AI to Comply with Objectionable Requests"; its scope is
- persuasion increasing compliance with objectionable requests, not a general
- proof that persuasion framing improves prompt quality.
- [[https://papers.ssrn.com/sol3/papers.cfm?abstract_id=5357179][SSRN paper]]
-- Combinatorial testing references: NIST supports t-way combinatorial testing
- and notes pairwise is one covering strength, with higher-strength arrays
- useful for failures requiring more interacting factors.
- [[https://www.nist.gov/publications/practical-combinatorial-testing-beyond-pairwise][NIST beyond pairwise]],
- [[https://www.nist.gov/publications/combinatorial-software-testing][NIST combinatorial testing]]
-
-*** Grouped index (for batching by area)
-
-Each item below is a one-line summary of a sub-TODO further down. Tick the box when the matching sub-TODO is moved to =DONE=. Items are grouped by area so they can be batched (e.g., "do all Playwright items in one session").
-
-**** Browser testing
-- [X] [#A] =playwright-js=: locator/assertion-first guidance (replace raw CSS, =networkidle=)
-- [X] [#B] =playwright-js= + =playwright-py=: reconcile headless/visible defaults
-- [X] [#B] =playwright-js= + =playwright-py=: remove emoji console markers from examples
-
-**** Frontend / UI
-- [X] [#B] =frontend-design=: WCAG 2.2 alignment, accessibility non-optional
-- [X] [#B] =frontend-design=: harmonize aesthetic guidance with anti-pattern rules
-
-**** Security
-- [X] [#A] =security-check=: OWASP 2021 + WSTG coverage
-- [X] [#B] =security-check=: tooling and offline/network caveats
-
-**** Combinatorial testing
-- [X] [#B] =pairwise-tests=: t-way escalation guidance beyond pairwise
-- [X] [#B] =pairwise-tests=: clarify negative value syntax + generator availability
-
-**** V2MOM
-- [X] [#A] =create-v2mom=: rename Metrics → Measures (Salesforce alignment)
-- [X] [#B] =create-v2mom=: prevent task migration from turning V2MOM into a backlog
-- [X] [#B] =create-v2mom=: mitigation/owner fields for Obstacles
-
-**** Prompt engineering
-- [X] [#A] =prompt-engineering=: correct/narrow Meincke citation
-- [X] [#B] =prompt-engineering=: eval-harness requirement for production prompts
-
-**** Codify
-- [X] [#B] =codify=: stale-entry review + privacy checks before writing project =CLAUDE.md=
-
-**** Code review
-- [X] [#A] =review-code=: resolve local-verification vs CI boundary
-- [X] [#B] =review-code=: =CLAUDE.md= citation scope for public artifacts
-- [X] [#B] =review-code=: relax three-strengths rule for tiny/failing diffs
-
-**** PR / review responses
-- [X] [#A] =respond-to-review=: remove review-process language from commit messages
-- [X] [#B] =respond-to-review=: use unresolved threads + resolution state
-- [X] [#B] =respond-to-cj-comments=: drop personal absolute paths from public-writing (moot — already clean)
-- [X] [#B] =respond-to-cj-comments=: fallback when =humanizer= or =emacsclient= unavailable (moot — superseded by /voice + VERIFY pattern)
-
-**** Branch workflow
-- [X] [#A] =finish-branch=: fix base-branch detection
-- [X] [#B] =finish-branch=: worktree-aware pull/merge safety
-- [X] [#B] =start-work=: tool-availability + ceremony-scaling rules
-- [X] [#B] =start-work=: claim-before-justify rollback risk
-
-**** Tests / TDD
-- [X] [#B] =add-tests=: fix missing =typescript-testing.md= reference or add ruleset (moot — ruleset now exists)
-- [X] [#B] =add-tests=: explicit exceptions to "all three categories per function"
-
-**** Debugging / RCA
-- [X] [#B] =debug=: capture environment + recent-change context before hypotheses
-- [X] [#B] =root-cause-trace=: constrain defense-in-depth to trust boundaries
-- [X] [#B] =five-whys=: require evidence + counterfactual validation per why
-
-**** Brainstorming
-- [X] [#B] =brainstorm=: timebox + research/source rules for high-stakes designs
-
-**** Architecture
-- [X] [#B] =arch-decide=: timeless examples, drop unverifiable claims
-- [X] [#B] =arch-decide=: standardize statuses + immutability language
-- [X] [#B] =arch-design=: threat modeling + privacy/compliance as first-class inputs
-- [X] [#B] =arch-design=: separate paradigms from tactical patterns
-- [X] [#B] =arch-document=: arc42/Q42 quality scenarios
-- [X] [#B] =arch-document=: staleness + ownership metadata for generated docs
-- [X] [#B] =arch-evaluate=: confidence levels for framework-agnostic findings
-- [X] [#B] =arch-evaluate=: report skipped tool checks explicitly
-
-**** C4 modeling
-- [X] [#A] =c4-analyze= + =c4-diagram=: notation/output fallback (not draw.io-only)
-- [X] [#B] =c4-analyze= + =c4-diagram=: clarify abstraction boundaries
-
-**** Global rules
-- [X] [#B] =commits.md=: split DeepSat/Linear/Slack-specific from global rules → promoted to a top-level task (deferred for Craig)
-- [X] [#A] =commits.md= + publish flows: =humanizer=-unavailable fallback → promoted to a top-level task (deferred; humanizer premise moot)
-- [X] [#B] =verification.md=: explicit "unable to verify" reporting standard
-- [X] [#B] =testing.md=: property-based + mutation testing as escalation paths
-- [X] [#B] =testing.md=: soften absolute TDD with explicit spike protocol
-- [X] [#B] =subagents.md=: capability/availability + cost checks
-
-**** Languages
-- [X] [#A] =python-testing.md=: revisit in-memory SQLite guidance
-- [X] [#B] =python-testing.md=: separate "never mock ORM" from unit-test boundaries
-- [X] [#B] =elisp.md=: drop tool-specific advice
-- [X] [#B] =elisp-testing.md=: batch-mode + native-comp caveats
-
-**** Hooks
-- [X] [#A] =hooks/README.md=: include =destructive-bash-confirm.py= in install/settings snippets
-- [X] [#A] =hooks/git-commit-confirm.py= + =gh-pr-create-confirm.py=: inspect message/body files referenced by =-F= / =--body-file=
-- [X] [#B] =hooks/destructive-bash-confirm.py=: shell-aware command parsing (not regex)
-
-*** 2026-05-22 Fri @ 15:47:10 -0500 Made playwright guidance locator/assertion-first, dropped networkidle-as-readiness
-
-Rewrote the readiness guidance in both =playwright-js/SKILL.md= and =playwright-py/SKILL.md=: reconnaissance now waits for a visible app landmark via a web assertion or locator (=expect(...).toBeVisible()= / =get_by_role(...).wait_for()=), not =networkidle= (which Playwright discourages). Updated the login/form examples to =getByLabel=/=getByRole= + web assertions, the API_REFERENCE.md waiting section, and =lib/helpers.js= defaults (=waitForPageReady= now defaults to =load= and prefers a caller-supplied landmark; =authenticate= races the success indicator over a =load= navigation). node --check passes.
-
-*** 2026-05-22 Fri @ 14:23:02 -0500 Added headed/headless decision tables to both playwright skills
-
-Added matching purpose-based decision tables to =playwright-js/SKILL.md= (was "always visible") and =playwright-py/SKILL.md= Best Practices (was "always headless"). Each names its own default and points at the other skill, so the difference is deliberate, not a habit-flip: headed for interactive debugging, headless for CI/pytest. Also softened the absolutist "Always launch... headless" comment in the py example.
-
-*** 2026-05-22 Fri @ 15:47:10 -0500 Removed emoji console markers from the playwright skills
-
-Replaced every emoji status marker with a plain ASCII prefix across =playwright-js/= (run.js, lib/helpers.js, SKILL.md) and =playwright-py/= (SKILL.md, examples/*.py): 📦/⚡/📄/📥/🎭/🚀/📋/✅/❌/🔍/📸/✓/✗ → =[setup]=/=[run]=/=[ok]=/=[error]=/=[fail]= etc. Post-change emoji grep is clean (excluding node_modules); node --check and py_compile pass.
-
-*** 2026-05-22 Fri @ 14:35:16 -0500 Made accessibility a non-optional WCAG 2.2 gate in frontend-design
-
-Added an "Accessibility Gate (required before handoff)" section to =frontend-design/SKILL.md= covering keyboard operation, focus visibility, focus-not-obscured (2.2), target size (2.2), contrast, reduced motion, labels, and semantic structure — a baseline for all frontend work, not just interactive components. Rewrote the Build/Review phases to build accessibly as you go and clear the gate before handoff, and bumped =references/accessibility.md= from WCAG 2.1 to 2.2 with backing detail for the new criteria.
-
-*** 2026-05-22 Fri @ 14:35:16 -0500 Added a "creative but bounded" section to frontend-design
-
-Added a subsection under Frontend Aesthetics framing the bold/maximalist directions as tools, not obligations: domain fit, readability first, responsive stability, and no decorative effect that degrades the workflow. Reconciles rather than contradicts the maximalist encouragement (maximalism stays on the table as deliberate usable density), and ties the readability bullet to the new accessibility gate.
-
-*** 2026-05-22 Fri @ 14:35:16 -0500 Updated security-check to OWASP Top 10 2021 + WSTG mapping
-
-Replaced the older six-category list in =.claude/commands/security-check.md= with the full Top 10 2021 set, each finding mapped to a 2021 category or WSTG area. Added the four missing categories (Insecure Design, Software and Data Integrity Failures, Security Logging and Monitoring Failures, SSRF) plus explicit checks for object/function-level authorization, SSRF on URL-fetch paths, update/plugin/dependency integrity, and logging/monitoring gaps.
-
-*** 2026-05-22 Fri @ 14:35:16 -0500 Added scanner tooling + network caveats to security-check
-
-Added an optional configured-scanners step (=gitleaks=/=trufflehog= secrets, =semgrep= source patterns, OSV scanner, lockfile-diff review) that supplements the manual scans, plus a network caveat: dependency audits that can't run (offline, tool absent, DB unreachable) must report "not run" naming the tool and reason, never read as a pass. Carried that into the no-issues summary.
-
-*** 2026-05-22 Fri @ 14:35:16 -0500 Added t-way escalation guidance to pairwise-tests
-
-Added an "Escalating Beyond Pairwise (t-way)" subsection: start with pairwise across the whole space, then escalate specific high-risk clusters to 3-way+ when history, safety, security, or domain coupling says a fault needs more than two interacting factors. Lists escalation triggers and shows the sub-model order syntax (={ A, B, C } @ 3=) vs a blanket =/o:3= bump, stressing targeted not uniform escalation. Cites NIST combinatorial-testing work.
-
-*** 2026-05-22 Fri @ 14:35:16 -0500 Clarified PICT ~ syntax + honest generator-availability path in pairwise-tests
-
-Added a "~ prefix" explanation (PICT marker tagging a value as negative/invalid, not an arithmetic operator; PICT pairs negatives with valid values once and strips the marker before the SUT) and a stop-at-the-model rule: if neither the =pict= binary nor =pypict= is present, produce the model and stop rather than hand-writing a table and passing it off as PICT output.
-
-*** 2026-05-22 Fri @ 14:43:17 -0500 Renamed Metrics → Measures throughout create-v2mom
-
-Full rename across =.claude/commands/create-v2mom.md= (acronym expansions, Phase 7 heading, the "Measures must be measurable" principle, exit criteria, review questions, red flags, examples) to match Salesforce's official term. Kept the "vanity metrics" idiom intact — it's the anti-pattern term, not a section reference.
-
-*** 2026-05-22 Fri @ 14:43:17 -0500 Split strategy from execution in create-v2mom task migration
-
-Rewrote Phase 8 (and tightened Phase 5.5): tasks stay in the backlog grouped by method, and each method gains a one-line link to where its tasks live, instead of transplanting the task tree into the V2MOM. Strategy (V2MOM) and execution (backlog) are now explicitly separate sources of truth, keeping the V2MOM concise.
-
-*** 2026-05-22 Fri @ 14:43:17 -0500 Made create-v2mom obstacles operational (mitigation/owner/cadence)
-
-Phase 6 now captures, per obstacle: name, manifestation, stakes, mitigation, owner, and review cadence — with a worked example per domain (health/finance/software), a "good obstacle" characteristic, a Phase 9 review question, and a red flag for candid-but-not-operational obstacles. An obstacle without a countermove is now flagged as an observation, not a plan.
-
-*** 2026-05-22 Fri @ 14:43:17 -0500 Corrected and narrowed the Meincke citation in prompt-engineering
-
-Fixed the title to "Call Me A Jerk: Persuading AI to Comply with Objectionable Requests" (SSRN abstract_id=5357179) in all three spots (frontmatter, Seven Principles intro, References). Reframed the ~33%→72% result as what it is — a prompt-safety caution that persuasion raises compliance with objectionable requests — explicitly not evidence that persuasion framing improves engineering prompt quality. Kept the seven principles as a tone vocabulary.
-
-*** 2026-05-22 Fri @ 14:43:17 -0500 Added an eval-harness requirement to prompt-engineering critique mode
-
-Added critique step 7 + a checklist line: for fragile or reusable/production prompts, write 3-5 adversarial/edge inputs, run both the old and new prompt against each, and record the behavioral delta. A throwaway prompt can ship on the rewrite alone; a discipline/reused/production one can't. Without examples, "the rewrite is better" is an assertion, not a result.
-
-*** 2026-05-22 Fri @ 14:43:17 -0500 Added mandatory stale-entry + privacy pre-write checks to codify
-
-Added a "Mandatory pre-write checks" block at the top of Phase 3 (Write) in =.claude/commands/codify.md=: a stale-entry scan (update/remove no-longer-true entries in place, don't append contradictions around them) and a privacy/leak check carrying both questions verbatim — "safe if the project were public?" and "belongs in private memory instead?" — routing private content to auto-memory. Gates, not background guidance.
-
-*** 2026-05-22 Fri @ 14:06:41 -0500 Scoped review-code's CI-trust rule to reviewing, not shipping
-
-Expanded the False-Positive Filter bullet in =review-code/SKILL.md=: "trust CI, don't run builds" applies to reading a diff, not producing one. A pre-commit/pre-push flow still owes the local verification =verification.md= requires (run the suite or state "not run because..."). Closes the apparent contradiction with =verification.md= / =finish-branch=.
-
-*** 2026-05-22 Fri @ 14:06:41 -0500 Added private-vs-public CLAUDE.md citation modes to review-code
-
-Expanded the Content scope section in =review-code/SKILL.md= with two modes: a private/internal review cites =CLAUDE.md= directly; a public/team review translates the rule into the engineering reason it encodes and doesn't name the rules file (a teammate can act on the reason, not on a file they can't reach). Same principle =commits.md= states for personal tooling in public artifacts.
-
-*** 2026-05-22 Fri @ 13:48:14 -0500 Relaxed review-code "three strengths" to up-to-three-or-none
-
-Changed all three "three minimum" spots in =review-code/SKILL.md= (Strengths section, Critical Rules DO list, Anti-Patterns) to "up to three specific; say none found on a tiny or weak diff." Reframed the old "No Strengths section" anti-pattern as "Skipping strengths out of laziness" so a substantive diff still demands them while a weak one can honestly report nothing notable. Landed alongside Craig's adjacent edit telling reviewers not to explain why a strength is good (sycophantic padding).
-
-*** 2026-05-22 Fri @ 14:12:24 -0500 Removed review-process language from respond-to-review commit guidance
-
-Replaced the =fix: Address review — [description]= example (and the matching description-line phrasing) in =.claude/commands/respond-to-review.md= with "name the actual fix (=fix: validate export filename=), not the review that prompted it." Killed the non-ASCII dash and the process-in-commit pattern that conflicted with =commits.md=.
-
-*** 2026-05-22 Fri @ 14:12:24 -0500 Made respond-to-review fetch unresolved threads + resolve after verification
-
-Rewrote section 1 (Gather) in =.claude/commands/respond-to-review.md= to pull =reviewThreads= via =gh api graphql= with =isResolved=, skipping already-resolved threads so settled feedback isn't re-processed; top-level conversation comments still come from REST. Added a section-4 step: reply and resolve a thread only after the fix is verified, never before.
-
-*** 2026-05-22 Fri @ 14:12:24 -0500 Verified respond-to-cj-comments no longer embeds an absolute path (moot)
-
-Already resolved by a prior migration: =grep= for =/home/= and =/Users/= in =.claude/commands/respond-to-cj-comments.md= returns nothing. The public-writing section refers to the rules by name, not by local path. No edit needed.
-
-*** 2026-05-22 Fri @ 14:12:24 -0500 Closed respond-to-cj-comments humanizer/emacsclient fallback (largely moot)
-
-Overtaken by two later changes: =/humanizer= was replaced by =/voice personal= (no =/humanizer= invocation remains), and the mandatory =emacsclient= summary-open was replaced by the in-place VERIFY-task pattern (workflow line ~262, Craig's 2026-05-12 standing instruction). Only a stale descriptive phrase remained — tidied "humanizer's signs of AI writing" to "the signs of AI writing." The original fresh-environment-fallback concern no longer applies as written.
-
-*** 2026-05-22 Fri @ 14:51:37 -0500 Fixed finish-branch base-branch detection
-
-Rewrote Phase 2: resolve the base *branch name* in priority order (open PR's =baseRefName=, then =git symbolic-ref --short refs/remotes/origin/HEAD= stripped, then ask), and compute the merge-base *SHA* separately only where a commit range is needed. Made the branch-name-vs-merge-base distinction explicit, since the old command returned a SHA where a branch name was needed.
-
-*** 2026-05-22 Fri @ 14:51:37 -0500 Made finish-branch merge safer + worktree-aware
-
-Added pre-flight checks to Option 1 (Merge Locally): dirty-tree refusal with no auto-stash, protected-branch awareness, upstream-gated =git pull --ff-only=, and merge-commit-vs-rebase as a team-policy choice instead of a hardcoded =--no-ff=. Replaced the fragile =git worktree list | grep <branch>= detection with a =git rev-parse --git-dir= vs =--git-common-dir= comparison plus =git worktree list --porcelain= for the path.
-
-*** 2026-05-22 Fri @ 14:51:37 -0500 Added tool-availability + ceremony-scale paths to start-work
-
-Added a "Tool availability" section (graceful degradation when Linear MCP / =gh= / =/voice= / Playwright are missing — do what's available, surface what isn't, don't block) and a "Ceremony scale" section (trivial / small / standard tiers so a two-line fix skips ticket+branch+gates unless asked). The =humanizer= reference in the original item is moot — the file already uses =/voice= throughout.
-
-*** 2026-05-22 Fri @ 14:51:37 -0500 Resolved start-work claim-before-justify rollback risk
-
-Split the claim by tracker type: personal todo.org claims defer to after the Justify gate (a killed task needs no rollback), while team trackers (Linear/GitHub) still claim first to signal intent but record prior state (status, assignee, label) so the Phase 2 rollback restores exactly it. Updated the per-tracker rollback steps and the matching anti-pattern.
-
-*** 2026-05-22 Fri @ 14:28:41 -0500 Verified add-tests typescript-testing.md reference resolves (moot)
-
-Resolved since the audit: =languages/typescript/claude/rules/typescript-testing.md= now exists, and =add-tests/SKILL.md:68= references it by bare filename, the same way it references =python-testing.md= (both get copied into a project's =.claude/rules/=). The "missing file" premise no longer holds. No edit needed.
-
-*** 2026-05-22 Fri @ 14:28:41 -0500 Added a category-exception protocol to add-tests
-
-Added an exception note to step 7 (proposal) in =add-tests/SKILL.md=: pure adapters, generated code, tiny pass-through wrappers, and framework glue may skip a category that would only re-test the framework, but the skip must be stated and justified in the plan and the behavior covered at integration/E2E level — never a silent omission. Step 12 (write) now points back to "honor documented category exceptions."
-
-*** 2026-05-22 Fri @ 14:25:37 -0500 Added environment + recent-change capture to debug Phase 1
-
-Added a fourth Phase-1 step in =debug/SKILL.md=: record versions, feature-flag/config state, dataset/fixture, seed/clock, concurrency, and recent commits/config-infra changes. Noted that intermittent bugs usually live in environment/state transitions (and "what changed recently" is often the fastest route), while a deterministic local bug only needs a one-liner. Updated the phase's closing recap to include the context.
-
-*** 2026-05-22 Fri @ 14:25:37 -0500 Constrained root-cause-trace defense-in-depth to boundaries
-
-Rewrote step b in =root-cause-trace/SKILL.md=: instead of "add a check at each layer that could have caught it," add one only at a layer that owns a boundary or invariant — ingress/trust, persistence, invariant-owning service, final render. Added the explicit rule that a pass-through function owning neither shouldn't get a duplicate null check (validation spam). Recast the three example layers as the boundary types.
-
-*** 2026-05-22 Fri @ 14:25:37 -0500 Required evidence + counterfactual per why in five-whys
-
-Expanded step 2 in =five-whys/SKILL.md=: each link now owes an evidence field (a log/commit/metric/config you can point to) and a counterfactual check (remove this cause — does the symptom above plausibly not happen?). Framed the counterfactual as the main guard against monocausal storytelling, and updated the worked example to show both fields.
-
-*** 2026-05-22 Fri @ 15:51:59 -0500 Added timebox + fresh-sources rules to brainstorm
-
-Phase 1 gained a "Timebox the dialogue" rule (aim for the one-sentence restatement in ~5-8 questions, then move on and park the rest as open questions). Phase 2 gained "Ground high-stakes claims in fresh sources" (check load-bearing claims about markets/regulations/tools/vendors/APIs against a current source; mark unverified ones as assumptions). The design-doc skeleton gained an "## Assumptions" section that distinguishes researched facts (with source) from assumptions (to confirm before building).
-
-*** 2026-05-22 Fri @ 14:59:32 -0500 Made arch-decide examples timeless + required citations
-
-Dated the MongoDB multi-document-transaction example (scoped to 2024-01) with a backing reference, and added a "Cite, don't assert" Do: every concrete technical claim about a tool/version/platform carries a link, doc, version, or "checked YYYY-MM" date, or gets a domain-neutral placeholder — so unsourced "X can't do Y" doesn't rot into stale fact.
-
-*** 2026-05-22 Fri @ 14:59:32 -0500 Standardized arch-decide ADR statuses + immutability rule
-
-Declared a canonical five-status set (Proposed, Accepted, Rejected, Deprecated, Superseded) with an explicit "no synonyms" line, and spelled out the immutability rule in the Don'ts: an accepted ADR's body is frozen, only status/link metadata changes, a changed decision gets a new superseding ADR and the old one stays as the historical record.
-
-*** 2026-05-22 Fri @ 14:59:32 -0500 Added Trust/Data/Compliance phase to arch-design
-
-Added a new Phase 4 (Trust, Data, and Compliance) before the paradigm shortlist: trust boundaries, data classification, abuse/misuse cases, privacy constraints, compliance evidence, and operational ownership — surfaced early so the architecture is drawn around them, not retrofitted by a downstream =security-check=. Threaded into the workflow list, brief template (new §6), review checklist, and anti-patterns.
-
-*** 2026-05-22 Fri @ 14:59:32 -0500 Split paradigms from tactical patterns in arch-design
-
-Split Phase 5's single mixed table into Step 1 (pick one paradigm: monolith/microservices/layered/event-driven/serverless/pipeline/space-based) and Step 2 (compose tactical patterns: DDD, hexagonal, CQRS, event sourcing — several or none, often per-module), with composition examples and an anti-pattern against treating DDD/CQRS as alternatives to a paradigm. Recommendation + brief now name a paradigm plus composed patterns.
-
-*** 2026-05-22 Fri @ 14:59:32 -0500 Expanded arch-document quality scenarios to the Q42 six-part template
-
-Replaced §10's thin "Under [condition]..." template with the arc42/Q42 six-part structure (source, stimulus, environment, artifact, response, response measure), each glossed, with the cart-checkout example rewritten across all six parts. A one-line prose form stays acceptable once all six parts are recoverable.
-
-*** 2026-05-22 Fri @ 14:59:32 -0500 Added staleness/ownership metadata to arch-document output
-
-Added a per-section metadata block (owner, generated-against SHA + date, review cadence, "stale-when" conditions) as an HTML-comment header plus a visible Doc-status note, with field-fill guidance, and a whole-document Doc Status table replacing the README's "Last Updated" stub. Wired into the review checklist and an "Undated docs" anti-pattern.
-
-*** 2026-05-22 Fri @ 14:59:32 -0500 Added confidence levels to arch-evaluate findings
-
-Added a "Confidence and Provenance" subsection: every framework-agnostic finding carries High/Medium/Low + how it was determined, with a required "Not fully checked because..." note when scale, runtime imports, reflection, or dynamic dispatch cap certainty. Updated the example findings and review checklist; a finding with no note now asserts a full read.
-
-*** 2026-05-22 Fri @ 14:59:32 -0500 Made arch-evaluate report skipped tool checks explicitly
-
-Replaced "skip silently" with explicit reporting: for each detected language whose tool isn't configured or can't run, emit an Info "tool not configured / not run" finding (with an example) so the audit shows what was and wasn't verified. A check that didn't run no longer reads as a pass. Updated workflow step 4 and the review checklist.
-
-*** 2026-05-22 Fri @ 14:51:37 -0500 Added notation/output fallback to c4-analyze + c4-diagram
-
-Both commands now treat C4 as notation-independent: a "Choosing a notation" section (draw.io XML, Structurizr DSL, Mermaid with native C4 types, PlantUML/C4-PlantUML) and a headless fallback that emits a text notation (Mermaid or Structurizr DSL) and skips PNG-export/desktop-open when =drawio= or a GUI is absent, rather than failing. draw.io is now one option, not the only one.
-
-*** 2026-05-22 Fri @ 14:51:37 -0500 Clarified C4 abstraction boundaries in c4-analyze + c4-diagram
-
-Added an "Abstraction boundaries" section to both: a Container is a separately deployable/runnable unit (not synonymous with a Docker container — a SPA or managed DB counts), a Component lives inside one Container and isn't separately deployable. Added a 4e "Verify single abstraction level" check that walks every element and relationship to confirm it stays at the diagram's level, notation-independent.
-
-*** 2026-05-22 Fri @ 15:10:35 -0500 Added "When You Cannot Verify" standard to verification.md
-
-Added a section requiring, when a verification command can't run, a four-part report: command attempted, why it couldn't run, risk left unverified, and the smallest next command for the user. States the principle that a check that didn't run is never reported as a pass — "unable to verify" is a required honest outcome, not silence. Placed after Red Flags.
-
-*** 2026-05-22 Fri @ 15:10:35 -0500 Added property-based + mutation testing escalation to testing.md
-
-Added an "Escalation Beyond Category and Pairwise" section: property-based testing for invariants over a broad input domain (round-trips, idempotence, ordering — Hypothesis/fast-check/proptest) and mutation testing for when high line coverage hides thin assertions (mutmut/cosmic-ray/Stryker). Both framed as escalation paths to reach for on a gap, not gates on every unit.
-
-*** 2026-05-22 Fri @ 15:10:35 -0500 Added a disciplined spike protocol to testing.md
-
-Formalized the existing "I need to spike first" excuse-table row into a "Spike Exception (Disciplined)" subsection under TDD Discipline: TDD stays the default, but a spike is sanctioned when all three hold — timeboxed, spike code not committed, and the first failing test written before productionizing the discovered approach. Built on the existing row rather than contradicting it.
-
-*** 2026-05-22 Fri @ 15:10:35 -0500 Added pre-dispatch availability + cost checks to subagents.md
-
-Added a "Pre-Dispatch Checks" section with two gates: Availability (no Agent capability → do the work in the main thread under the same scope/constraints/output discipline the contract would enforce) and Cost (when writing the full contract costs more than the task, do it inline). Cross-references the existing "Don't Subagent At All" section and "Subagenting trivial work" anti-pattern rather than duplicating.
-
-*** 2026-05-22 Fri @ 15:06:04 -0500 Revised python-testing SQLite guidance toward production-like DBs
-
-Replaced "prefer in-memory SQLite for speed" with: run ORM/query tests against a production-like DB (same engine as prod, often containerized), since SQLite diverges from Postgres/MySQL on query semantics, constraints, transactions, JSON, time zones, and indexes (a test can pass on SQLite and fail in prod). SQLite stays only for pure unit tests with no DB-semantics dependency.
-
-*** 2026-05-22 Fri @ 15:06:04 -0500 Clarified python-testing ORM-mocking boundary
-
-Changed the "never mock" bullet from "ORM queries" to "ORM internals (querysets, sessions, model internals)" and added a paragraph: domain services use real model methods/validation, but a thin orchestration unit can inject a fake at a deliberate data-access port (a repository/interface the code owns). That's still mocking at a boundary, not at ORM internals.
-
-*** 2026-05-22 Fri @ 15:06:04 -0500 Made elisp.md editing advice tool-agnostic
-
-Rephrased the "prefer Write over repeated Edits" bullet around intent: land nontrivial Elisp as one cohesive change rather than dribbling it in over tiny partial edits (which accumulate paren mismatches), and run paren-balance + byte-compile checks immediately after, whatever editing mechanism the environment uses.
-
-*** 2026-05-22 Fri @ 15:06:04 -0500 Added batch-mode + native-comp caveats to elisp-testing.md
-
-Added three sections: Batch-Mode Reproducibility (=emacs --batch= as source of truth, no interactive-session state, no blocking prompts, deterministic), Isolating Emacs State (temp =user-emacs-directory=, explicit load-path, declared deps only, with an unwind-protect sandbox example), and Byte-Compile/Native-Comp Warnings (=byte-compile-error-on-warn=, native-comp gated on =native-comp-available-p= and kept opt-in/version-aware).
-
-*** 2026-05-22 Fri @ 15:16:22 -0500 Synced hooks/README install snippets with the destructive hook (opt-in)
-
-Brought the README's manual-install and settings-JSON snippets in line with the canonical =hooks/settings-snippet.json= (which already wires all three) and the Makefile's opt-in design: added the destructive-bash-confirm.py symlink as an opt-in step, added its settings entry, and reworded the note to say all three are no-op-safe but the destructive gate is opt-in (=make install-hooks= excludes it by default — link manually before relying on the snippet entry).
-
-*** 2026-05-22 Fri @ 15:35:06 -0500 Hooks now scan file-backed commit/PR messages
-
-Added =read_referenced_file()= to =_common.py= (safe local read: missing/oversize/non-UTF-8 → None) and wired it in: =git-commit-confirm.py= =extract_commit_message= now handles =-F=/=--file=/=--file===<path>= (reads + scans the file, falls through to UNPARSEABLE → asks if unreadable), and =gh-pr-create-confirm.py= reads =--body-file= content instead of a placeholder. Attribution scanning now sees the real committed/posted text. Built a pytest harness (=hooks/tests/=, importlib-by-path loader for the hyphen-named hooks) and wired =hooks/tests= into =make test=. 54 hook tests pass; full suite green.
-
-*** 2026-05-22 Fri @ 15:35:06 -0500 Rewrote destructive-bash rm parsing on shlex
-
-=detect_rm_rf= now tokenizes with =shlex.split= instead of a whitespace split, so quoted/spaced paths and combined/separate/reordered flags (=-rf=, =-r -f=, =-fr=, =--recursive=/=--force=) all parse. Fails toward asking — returns a sentinel that still fires the modal — on unbalanced quotes or when a forced recursive rm coexists with a compound/pipeline/substitution/redirect construct. Documented the supported/unsupported shell constructs in the docstrings, and extended the dangerous-path banner to =$HOME=-prefixed and wildcard targets. Covered by 25 new tests. (Pre-existing, out-of-scope: path-prefixed =rm= like =/bin/rm= still isn't matched.)
-** DONE [#B] Add =make remove= for interactive ruleset removal via fzf
-CLOSED: [2026-05-22 Fri]
-Shipped: =scripts/remove.sh= (three modes — =--list=, =--remove-selected= reading stdin, and the default fzf-multi interactive flow) + =make remove= target + =scripts/tests/remove.bats= (5 cases). Lists only symlinks resolving into the repo (foreign links left alone); rm's picked links while leaving repo sources untouched; reports-and-continues on a missing target; quiet no-op on empty selection. shellcheck clean, make test green. Dropped the stale =bridge= entry per the note below.
-
-Add a Makefile target that lists every currently-installed ruleset entry
-and lets me pick one or more to remove via fzf. Granular alternative to
-=make uninstall= (removes everything) and =make uninstall-hooks= (removes
-only hooks).
-
-*** Why this matters
-
-Tearing down a single skill, rule, hook, or config file currently means
-either running =make uninstall= and re-installing what I want to keep,
-or =rm=ing the symlink directly and remembering the exact path. Both are
-friction. An interactive picker lets me filter, multi-select with Tab,
-and confirm with Enter — the typical fzf flow. Costs about 3-5 seconds
-per teardown instead of 15+ seconds of "what's the exact name?".
-
-*** Design
-
-The recipe builds a tab-separated list of every currently-installed item,
-categorized by type, and pipes it to =fzf --multi=. The user filters,
-marks with Tab, and confirms with Enter. The recipe parses the selections
-and =rm=s the matching symlinks.
-
-#+begin_example
- skill debug
- rule commits.md
- hook destructive-bash-confirm.py
- config settings.json
- commands commands
- bridge claude-rules
-#+end_example
-
-Each line is =<kind>\t<name>=. The recipe maps =<kind>= to the right path:
-
-- =skill= → =$(SKILLS_DIR)/<name>=
-- =rule= → =$(RULES_DIR)/<name>=
-- =hook= → =$(HOOKS_DIR)/<name>=
-- =config= → =$(CLAUDE_DIR)/<name>=
-- =commands= → =$(CLAUDE_DIR)/commands=
-- =bridge= → =$(SKILLS_DIR)/claude-rules=
-
-Source files in =rulesets/= stay untouched. =make install= re-creates the
-removed links if needed (the install loop is idempotent).
-
-*** Edge cases
-
-- Esc instead of Enter → empty selection → clean exit, no removal.
-- Filter to nothing then Enter → same as Esc.
-- Selected item already gone → =rm= fails visibly, processing continues
- on the rest.
-- =fzf= not installed → fail fast with a clear error (matches the pattern
- used by =install-lang=).
-
-*** Possible extensions
-
-- Parallel =make pick-install= target that lists not-yet-installed items
- and installs the chosen ones. Symmetric UX, same fzf flow.
-- Confirmation prompt when more than N items selected (defense against
- accidental select-all).
-- =--source= flag that also runs =git rm= against the rulesets source for
- the selected item. Probably bad idea — too easy to lose work.
-- The =bridge → $(SKILLS_DIR)/claude-rules= entry above is stale — the
- bridge symlink got removed in a later commit. Drop that bullet when the
- recipe lands.
-** DONE [#B] Document the =mcp/= install pipeline in =mcp/README.org=
-CLOSED: [2026-05-22 Fri]
-Wrote =mcp/README.org= covering everything in the "what to cover" list: the file layout (tracked vs gitignored), the secrets-bundle shape (plain =${VAR}= secrets + base64-bundled OAuth artifacts, AES256 symmetric =gpg -c=), the install flow (decrypt → materialize keys/token caches at mode 600 → expand → register unregistered, idempotent), the http/sse-vs-stdio transport split, token rotation when a Google refresh token is revoked, and adding a new server. Grounded in a read of the actual =install.py= + =servers.json=.
-
-=mcp/= has =install.py=, =servers.json=, =secrets.env.gpg=, =gcp-oauth.keys.json= (gitignored, regenerated at install). No README. Coming back to this in three months I'll re-discover how the bundle is structured, what =install.py= does, and how to rotate tokens. Saving that re-discovery is the whole point.
-
-*** What to cover
-
-- Layout: what each file is, which are tracked vs gitignored.
-- Secrets bundle shape: how vars are listed in =secrets.env=, the symmetric-encryption pattern (=gpg -c --cipher-algo AES256=), the base64-bundled OAuth artifacts (=GCP_OAUTH_KEYS_JSON_B64=, =GOOGLE_DOCS_PERSONAL_TOKEN_B64=, =GOOGLE_DOCS_WORK_TOKEN_B64=).
-- Install flow: =make install-mcp= → =install.py= decrypts, writes the keys file and Google Docs token caches at mode 600, expands =${VAR}= in =servers.json=, calls =claude mcp add --scope user= for unregistered servers. Idempotent.
-- Token rotation: when a refresh token gets revoked, the recovery flow (re-auth on one machine, re-bundle, recommit).
-- Adding a new server: edit =servers.json=, add any new =${VAR}= placeholders to the bundle, re-encrypt.
-- The OAuth dance for HTTP-transport servers (linear, notion) versus stdio (google-docs-*) — different paths, different gotchas.
-** DONE [#C] Add =make uninstall-mcp= + =mcp/install.py --check= for symmetry :feature:solo:quick:
-CLOSED: [2026-05-28 Thu]
+Built the sentry supervisor workflow from the spec
+([[file:docs/specs/2026-07-14-sentry-workflow-spec.org][sentry workflow spec]], now IMPLEMENTED). Four phases, each committed and
+pushed in no-approvals + auto-flush mode; full suite green throughout. The overnight
+live trial is handed to Craig as a manual-testing task (below); its findings file as
+follow-ups.
+*** 2026-07-19 Sun @ 04:52:00 -0500 Built the agent-lock helper + 18 bats tests
+=.ai/scripts/agent-lock= (canonical =claude-templates/.ai/scripts/=, mirror synced):
+mkdir-atomic acquire, PID/host/ISO-timestamp metadata, mtime-based staleness reclaim
+(atomic-rename claim so two acquirers can't double-acquire — caught by the pre-commit
+review), heartbeat refresh, acquire/release/status/path subcommands, XDG_RUNTIME_DIR
+home with =~/.cache= fallback. Commit a8b6cf4.
+*** 2026-07-19 Sun @ 04:56:00 -0500 Built the sentry.org engine + INDEX entry
+=.ai/workflows/sentry.org= (mirror synced): :COMMIT_AUTONOMY: entry ticket, the
+interactive entry gates, ff-only reconcile, =sentry/<date>-<host>= branch mechanics,
+the ten-pass probe→work→session-context→commit runner, digest + morning-approval
+queue, skip-not-degrade safety, spine-excluded dirty checks + fire-end digest commit,
+multi-day stall notify, the stop-sentry operation. All 10 decisions and 12 findings
+reflected. Commit ccc9c26.
+*** 2026-07-19 Sun @ 05:00:00 -0500 Wired the roam writers + wrap-up guard
+knowledge-base.md and inbox.org core §5 acquire the roam-write lock and edit-plus-
+trigger (roam-sync stays sole committer); roam-sync.sh header updated to match;
+wrap-it-up.org gained a Step 0 active-sentry guard; triage-intake.org notes its
+sentry-pass role. Graceful degradation when agent-lock is absent. Commit c6383e9.
+*** 2026-07-19 Sun @ 05:04:00 -0500 Verified suite green + flipped spec to IMPLEMENTED
+=make test= green at HEAD (pytest 393, ERT + bats all pass, exit 0). Flipped the spec
+keyword DOING → IMPLEMENTED with a dated history line and mirrored the Metadata Status.
+Filed the overnight live trial as a structured manual-testing task.
+** DONE [#C] ai launcher hardening — bug hunt + refactor pass :refactor:solo:
+CLOSED: [2026-07-19 Sun]
:PROPERTIES:
-:LAST_REVIEWED: 2026-05-28
+:CREATED: [2026-07-13 Mon]
+:LAST_REVIEWED: 2026-07-19
:END:
+Resolved across two commits (113e8d8 net, 2b619f1 refactor). Brought the 17 uncovered functions under characterization tests (launcher tests 9 → 42), extracted the git/tmux decision logic into four pure cores (_git_prep_action, _order_windows, _match_window_id, _git_is_dirty) each with a Normal/Boundary/Error set, and dispositioned the footgun audit + /refactor pass. Objective floor met: shellcheck clean, shfmt -i 2 -ci consistent, make test green before and after, live black-box smoke correct for claude and codex. Honest limit: attach_session and the full end-to-end of single/multi/fetch stay partially covered (their terminal step attaches to tmux or blocks on fzf, can't run headless). Their decision logic was extracted into the netted cores. The interactive runtime picker stayed out of scope (a design call, filed on the generic-agent-runtime parent).
+Harden =claude-templates/bin/ai= (the agent-session launcher, 540 lines, 22 functions). Origin: roam inbox 2026-07-13, phrased open-endedly ("find bugs until none visible, refactor until nothing worthwhile remains"). Rescoped 2026-07-19 with measurable acceptance criteria per =todo-format.md='s "Making an open-ended task measurable," which is what makes it =:solo:=. The four moves:
-Currently the MCP install pipeline only flows one direction. No way to remove rulesets-managed MCP servers in one command. No way to ask "what's the drift between =servers.json= and =claude mcp list=" without eyeballing.
+1. *Bound the surface.* The 22 functions are the done-set. *Covered* (behaviorally, via =scripts/tests/ai-launcher-runtime.bats=, 9 tests over the runtime path): =resolve_agent_cmd=, =build_runtime_choices=, =pick_runtime=, =build_instructions=, the print modes. *Uncovered* (the 17 to bring under test): =usage=, =check_deps=, =attach_session=, =create_window=, =maybe_add_candidate=, =build_candidates=, =fetch_candidates=, =git_status_indicator=, =annotate_candidates=, =auto_pull_if_clean=, =read_selections=, =sort_windows=, =find_window_id=, =prep_git_single=, =attach_mode=, =single_mode=, =multi_mode=, =print_launch_mode=.
-*** =make uninstall-mcp=
+2. *Net the behavior.* Characterization tests (Normal/Boundary/Error per unit, per =testing.md=) over the uncovered surface. The pure/near-pure ones take the category set directly: =git_status_indicator=, =maybe_add_candidate= (dedup), =annotate_candidates= (formatting), =read_selections= (selection parse), =usage=. The =tmux=/=git=-coupled ones (=sort_windows='s ordering, =create_window=, =attach_session=, =find_window_id=, =prep_git_single=, =auto_pull_if_clean=) get their pure decision logic extracted into helpers that take plain inputs and return plain results — that extraction *is* the hardening — with the I/O calls left as thin wrappers.
-Iterate over =servers.json=, run =claude mcp remove <name> -s user= for each. Ignore "not registered" errors. Idempotent.
+3. *Disposition every finding.* (a) A bash-footgun audit per function, each cell fixed / n-a / filed: unquoted expansions + word-splitting, =set -euo pipefail= gaps and where errexit is intentionally off, subshell state loss, exit-code propagation, ordering/races in =sort_windows= + window creation, and the git-prep error paths (=prep_git_single= / =auto_pull_if_clean= on a dirty tree, detached HEAD, no upstream). (b) A =/refactor= pass, each finding applied (tests green) or declined with a one-line reason.
-*** =mcp/install.py --check=
+4. *Objective floor.* =shellcheck= clean, =shfmt=-consistent, =make test= green before and after, and every uncovered function above has its characterization set (a per-function checklist — =kcov= isn't installed; install it if a single coverage number is wanted). Plus ~3 functional tests over the launch pipelines (=single_mode=, =multi_mode=, =attach_mode=) against a throwaway =tmux= session, for the composition bugs no per-function unit can see.
-Dry-run mode. Decrypt secrets, but instead of registering, print the drift report:
+*Qualifying answer:* a dispositioned report — surface split covered/uncovered, tests before → after, =shellcheck=/=shfmt= result, the footgun matrix fully dispositioned, the =/refactor= findings fully dispositioned, all green. Not "no bugs remain" (unprovable) — "every enumerated path passes its characterization set and clears the audit."
-- Servers in =servers.json= not in =claude mcp list= → =MISSING=
-- Servers in =claude mcp list= not in =servers.json= → =EXTRA=
-- Servers in both → =ok=
-
-Useful for diagnosing connection failures and for the eventual =make doctor= integration.
-** DONE [#C] Update =README.org= with MCP install pipeline section :chore:solo:quick:
-CLOSED: [2026-05-28 Thu]
+*Out of this task's =:solo:= scope:* an interactive runtime picker. It's a feature carrying a design/preference call (does Craig want it, what shape), which is deliberation, not hardening — file it separately if wanted. The runtime-selection arc (claude/codex shipped; ollama/qwen pending the model-floor eval) stays on the generic-agent-runtime parent.
+** DONE [#C] Put install-ai on PATH, launchable as =install-ai= :chore:quick:solo:
+CLOSED: [2026-07-18 Sat]
:PROPERTIES:
-:LAST_REVIEWED: 2026-05-28
+:CREATED: [2026-07-11 Sat]
+:LAST_REVIEWED: 2026-07-13
:END:
+From the roam inbox (2026-07-11): make install-ai launchable as =install-ai= (no =.sh=) from PATH. dotfiles needs a copy that stays in sync with the rulesets canonical — decide whether the startup script-sync already covers it or a dedicated mechanism is needed.
-=README.org= covers global install, per-project language bundles, and design principles, but doesn't mention =make install-mcp= or the =mcp/= directory. Add a short section after "Per-project language bundles" describing the user-scope MCP install pattern (decrypt → expand → register) and pointing at the eventual =mcp/README.org=.
-** DONE [#C] Consolidate =claude-templates/Makefile= after fold :chore:quick:solo:
-CLOSED: [2026-05-28 Thu]
+Resolved: added =claude-templates/bin/install-ai=, a thin launcher that resolves its own path through the symlink chain and execs =scripts/install-ai.sh=. =make install='s existing bin loop symlinks it into =~/.local/bin/install-ai= (same mechanism as =ai= and =agent-page=), so no dedicated sync and no dotfiles copy — the symlink always points at the canonical. 3 launcher bats added (incl. symlink-invocation resolution). Verified live: =install-ai --help= runs from PATH.
+** DONE [#C] coverage-summary.el documented as a local-only helper :chore:quick:solo:
+CLOSED: [2026-07-18 Sat]
:PROPERTIES:
-:LAST_REVIEWED: 2026-05-28
+:CREATED: [2026-06-22 Mon]
+:LAST_REVIEWED: 2026-07-13
:END:
-
-Sibling follow-up from the fold child (2026-05-15). After the subtree merge, =rulesets/claude-templates/Makefile= still has its standalone =install= / =uninstall= / =list= / =test-scripts= targets. The =install= target's =bin/ai= logic is now duplicated in =rulesets/Makefile=. Both work; the redundancy is harmless but worth cleaning up.
-
-Options:
-- *Delete* =claude-templates/Makefile= entirely — forces all install through rulesets root. Cleaner.
-- *Strip down* to just =test-scripts= — the one piece not redundant with =rulesets/Makefile=.
-- *Leave it* — slight redundancy, no functional harm.
-
-Triggered by: 2026-05-15 fold session's refactor audit (commit =2d645fc=).
-** DONE [#C] Run =--archive-done= sweep at start of =open-tasks.org= Phase A :chore:quick:solo:
-CLOSED: [2026-05-28 Thu]
+The elisp bundle installs =coverage-summary.el= into =.claude/scripts/=, gitignored in code projects, so CI can't run =make coverage-summary= against it. Decision (Craig, 2026-06-28): keep it in =.claude/scripts/= and document it as a local-only helper — don't ship it to a tracked =scripts/= dir, don't expect CI to run it. Remaining work (docs only, no move): state the local-only status in the script's header comment and wherever =make coverage-summary= is described, so the gitignored install reads as intentional rather than a gap. Note: emacs-wttrin rewrote its copy's header to claim a tracked =scripts/= home, which now contradicts this decision and should be reverted on their side. Surfaced 2026-06-21 during the coverage-summary autoloads bugfix (commit fb86736).
+
+Resolved: documented the local-only status in the =coverage-summary.el= commentary header and in =elisp-testing.md='s "Measuring it" section — the gitignored install now reads as intentional, not a coverage gap. Sent emacs-wttrin a handoff to revert its contradicting header claim.
+** DONE [#B] todo-cleanup.el dated-seal archiving :feature:solo:
+CLOSED: [2026-07-18 Sat]
+Redefine =--archive-done= aging from a 7-day roll-into-one-file model to a
+one-month retention with dated seals. Craig ratified the design in the work
+project 2026-07-17; origin handoff preserved at
+[[file:docs/design/2026-07-17-todo-cleanup-dated-seal-proposal.md]].
+
+Changes to =.ai/scripts/todo-cleanup.el= (canonical in
+=claude-templates/.ai/scripts/=):
+- =tc-archive-retain-days= default 7 → 31.
+- Aging predicate: archive a level-2 DONE/CANCELLED subtree when CLOSED is
+ older than the retain window OR CLOSED is unparseable. The unparseable case
+ already archives per the docstring; make it explicit in the contract.
+- New seal step (flag, e.g. =--seal=): rename the working =tc-archive-file=
+ (=task-archive.org=) → =resolved-YYYY-MM-DD.org= beside it; next aging
+ recreates a fresh working file. Auto-seal-at-quarter vs manual flag is an
+ open call — default manual (the per-project rotation task covers cadence).
+- Update the file-aging tests (=test-todo-cleanup.el= lines 362, 378, 491,
+ 513, 518) for the new retain default and the seal/rename behavior. The
+ gitignore-inheritance logic (=tc--ensure-archive-gitignored=) applies to both
+ names unchanged.
+
+TDD, canonical-then-mirror per the sync-check invariant. Home's planning-line
+strip proposal (filed alongside) also touches todo-cleanup.el =--convert-subtasks=
+— the two can be built as one batch.
+** DONE [#B] Strip stale planning lines on dated completion + lint backstop :feature:solo:
+CLOSED: [2026-07-18 Sat]
+Two linked fixes so a closed sub-task can't keep polluting the org agenda. Origin:
+home handoff 2026-07-17; design preserved at
+[[file:docs/design/2026-07-17-dated-log-planning-line-strip-proposal.md]]. A dated-log
+heading (no TODO keyword) that still carries an active =SCHEDULED:= renders as
+weeks-overdue on the agenda forever — invisible to a keyword scan and surviving
+=--archive-done=. Root cause: the completion rewrite strips keyword/priority/tags
+but nothing strips the planning line, and an interactive org close only stamps
+=CLOSED:=.
+
+1. =claude-rules/todo-format.md= (canonical rule) — in the "=***= and deeper —
+ rewrite to a dated event-log entry" section, add a step: remove any
+ =SCHEDULED:=/=DEADLINE:= line when rewriting to the dated form. Apply the same
+ to the VERIFY dated-completion path.
+2. =.ai/scripts/lint-org.el= (canonical in =claude-templates/.ai/scripts/=) — new
+ checker =dated-log-heading-active-timestamp=: flag any dated-log heading (the
+ =YYYY-MM-DD ... @ ...= form, no TODO keyword) carrying an active =<...>=
+ SCHEDULED or DEADLINE. The mechanical backstop for #1, mirroring
+ =subtask-done-not-dated=.
+3. =todo-cleanup.el= =--convert-subtasks= — drop the planning line alongside its
+ existing CLOSED-timestamp pull.
+
+Build notes: the checker's regex must key on "no TODO keyword" so it never flags a
+live TODO that legitimately carries a SCHEDULED. TDD, canonical-then-mirror. Batches
+with the dated-seal task above (both touch todo-cleanup.el).
+** DONE [#B] Enforce the task-boundary inbox check via a hook :feature:
+CLOSED: [2026-07-19 Sun]
+Resolved with the soft-nudge design (94e54f6). New =hooks/inbox-boundary-check.sh= Stop hook blocks the yield once + injects the pending count when =inbox-status -q= exits 1, steps aside on the harness re-entry (=stop_hook_active=) so a mid-task pause never wedges, self-skips on no-inbox/no-inbox-status/clean. Wired ahead of =ai-wrap-teardown= in =.claude/settings.json= (which the live =~/.claude/settings.json= symlinks to) + the snippet; glob-installed by =make install-hooks=. 6 bats (pending/clean/re-entry/no-inbox/absent-status/project-name). protocols.org Inbox Monitoring Cadence now notes the enforcement. Live-verified on ratio (clean inbox no-ops, pending fixture blocks); velox synced + hook linked. The UserPromptSubmit visibility complement stayed unbuilt (optional in the design) — file separately if wanted.
+Today the "check =inbox/= at every task boundary" rule (protocols.org, "Inbox
+Monitoring Cadence") is prose-only — present in all 27 projects' synced
+protocols.org, but nothing mechanically enforces it, so it holds only as well as
+the agent's adherence. Convert it to a hook so a pending handoff can't slip past a
+turn unseen. Design worked out with Craig 2026-07-18.
+
+*** The one decision to settle at kickoff
+Hard-block vs soft-nudge (this is why the task isn't =:solo:= — it's a preference
+call). My recommendation: *soft-nudge*. Answer this first, then the rest is
+mechanical.
+- *Hard-block*: the Stop hook blocks EVERY yield while items are pending, forcing
+ the agent to process before it can return control. Strongest guarantee, but it
+ also fires when the agent pauses mid-task to ask Craig a clarifying question
+ (that pause is also a Stop), pushing inbox processing ahead of the question.
+- *Soft-nudge*: use the =stop_hook_active= flag to inject the reason once per turn,
+ then let the turn end if the agent chooses not to act. Stronger than today's
+ prose rule, no clarifying-question hijack.
+
+*** Why the Stop event
+The harness has no "task boundary" event. But the rule's own definition — "after
+finishing a unit of work, before reporting back or asking what's next" — maps onto
+the Stop event (agent finishing its turn, about to yield). Every "reporting back"
+is a Stop. PostToolUse is wrong granularity (fires per tool call); UserPromptSubmit
+is the START of the next task, useful only as a complement (below).
+
+*** Mechanism
+A Stop hook returns =\{"decision":"block","reason":"N pending handoffs — process
+per inbox.org before yielding."\}= when =inbox-status -q= exits 1. The harness then
+keeps the agent going with that reason injected instead of returning to Craig. The
+loop self-terminates: once items are dispositioned (deleted or renamed
+=PROCESSED-=, which =inbox-status= excludes), the next Stop exits 0 and the turn
+ends. The harness passes =stop_hook_active: true= on the re-entry — check it to
+nudge once rather than wedge on an item the agent genuinely can't process.
+
+*** The complement (optional, pair with the Stop hook)
+A =UserPromptSubmit= hook that runs =inbox-status= and injects a one-line "N
+pending handoffs" note into context at the start of Craig's next instruction.
+Non-blocking; guarantees the agent sees pending items the moment a new task starts.
+Catches the "arrived while away" case from the other direction. Can ship alone
+(pure visibility) or alongside the Stop hook. Recommended: ship both.
+
+*** Placement + files
+- Global, in =~/.claude/settings.json= (the =Stop= array already holds
+ =ai-wrap-teardown.sh= — add a second entry, or a combined script).
+- Ship the script from =claude-templates/.claude/hooks/= (e.g.
+ =inbox-boundary-check.sh=); =make install-hooks= links it; wire it in the tracked
+ =settings.json= so it travels. Mirror the reference pattern in
+ =hooks/ai-wrap-teardown.sh= — reads =cwd= from stdin JSON via =jq=, basenames to
+ the project.
+- Self-skips where inapplicable: =inbox-status= exits 2 with no =inbox/= dir, so
+ the hook no-ops in any project without an inbox. One hook, every project, no
+ config.
+
+*** Verify
+- bats around the hook script (=hooks/tests/=): exit-1 inbox → block JSON emitted;
+ exit-0 → no output; =stop_hook_active: true= input → single-nudge path; no
+ =inbox/= → clean no-op. Follow the existing hook-test pattern.
+- Live: drop a test handoff, confirm the agent is pushed to process it at turn end.
+
+*** Honest limitation to carry forward
+No hook can perfectly tell "a unit of work finished" from "the agent paused for any
+other reason" — the harness exposes only "the turn is ending," not "a task is
+ending." The design leans on those two being usually the same in Craig's workflow.
+That's the tradeoff hard-block vs soft-nudge is really about.
+** DONE [#B] "Colloquialisms and Expansions" + "the list" before-close-queue convention :feature:
+CLOSED: [2026-07-19 Sun]
+Resolved (approved in the speedrun). New =* Colloquialisms and Expansions= section in =protocols.org= documents both shorthands: "put X on the list" → append to a session-scoped Before-Close Queue (=* Before-Close Queue= heading in the session anchor, resets on archive, todo.org for must-outlive items); "tell <project> <msg>" → =inbox-send=. =wrap-it-up.org= Step 1 gained a "Work the Before-Close Queue (before the Summary)" sub-step so queued work rides the wrap commit, unfinished items surfaced in the valediction. Design calls: reference in protocols.org not per-project notes.org (synced = shared norm); queue in the session anchor as home did; wrap step at the front of Step 1, not a new half-step (keeps the "Steps 1-5" framing). 4 documentation-integrity bats (=before-close-queue.bats=). Canonical + mirror synced. The UserPromptSubmit-style variant wasn't in scope.
+Home proposes two linked cross-project norms; Craig recommends rolling them out.
+Origin: home handoff 2026-07-18, design preserved at
+[[file:docs/design/2026-07-18-colloquialisms-and-the-list-proposal.md]]. Needs Craig's
+adoption decision + a small design pass before implementation — that's why it's
+filed, not applied.
+
+The two parts:
+1. *"the list" = a before-close FIFO queue.* "Put X on the list" / "add X to the
+ list" appends X to a session-scoped Before-Close Queue, worked oldest-first at
+ wrap-up before teardown, unfinished items surfaced rather than dropped. Resets
+ when the session anchor archives. Anything that must outlive the session is a
+ =todo.org= task instead.
+2. *"Colloquialisms and Expansions" shorthand dictionary.* A per-project (or
+ shared) map of Craig shorthand → expansion the agent applies without asking.
+ Seed entries: "the list" → the queue above; "tell <project> <msg>" → drop the
+ message in that project's inbox via =inbox-send= (already the sanctioned
+ handoff, so this entry is near-documentation).
+
+Durable wiring (why rulesets, not home-local): (a) a colloquialisms reference in
+the template — a =protocols.org= section or a shipped reference file; (b) a
+=wrap-it-up.org= step that processes the Before-Close Queue before teardown.
+=wrap-it-up.org= is a synced rulesets-owned workflow, so home can't wire (b)
+durably from downstream — it stubbed the norm via its local =notes.org= for now.
+
+Design pass to settle first: where the colloquialisms reference lives (protocols.org
+section vs a new shipped reference file); where the queue itself lives (home used a
+=* Before-Close Queue= heading in =session-context.org=); and the exact wrap-it-up
+insertion point (before the teardown/valediction, alongside the roam-inbox
+sub-step). Then it's a synced-file change (canonical-then-mirror) + a test that the
+wrap step drains the queue.
+** DONE [#B] working/ tracked-from-creation + gitignored temp/ :feature:
+CLOSED: [2026-07-20 Mon]
+Craig's ruling relayed from .emacs.d (2026-07-19): working/ is version-controlled from creation (not excluded until graduation); ephemeral artifacts go in a gitignored temp/ or /tmp; graduation reorganizes durable artifacts into permanent homes rather than marking when they become durable.
+
+Implemented (Shape A) during the morning sentry review, 2026-07-20:
+- =claude-rules/working-files.md=: added "working/ Is Version-Controlled From Creation" section (tracked-from-creation, graduation-is-a-move, temp/ for ephemeral).
+- =.ai/protocols.org= (canonical + mirror): one-paragraph mirror in the Working-Files Convention section.
+- =scripts/install-ai.sh=: emits a =temp/= ignore block in BOTH track and gitignore modes; working/ never ignored. Idempotent.
+- =scripts/sweep-gitignore-tooling.sh=: separate mode-independent temp/ backfill pass (a distinct loop, not an IGNORE_SET member — that would skip track-mode projects, the set that most needs it).
+- Tests: 3 new install-ai bats + 4 new sweep bats (temp/ in both modes, never working/, idempotent). Suite green, 384 bats ok.
+
+Finding confirmed at implementation: the canonical machinery already never ignored working/ (tooling set is only =.ai/ .claude/ CLAUDE.md AGENTS.md=), so .emacs.d's =/working/= ignore was a purely local deviation. The staging proposal dir was removed after shipping; its content lives in the feat commit and this body.
+** DONE [#C] Polyglot projects — supported, or refused? :spec:
+CLOSED: [2026-07-20 Mon]
+DECISION (2026-07-20, scouting with Craig): *case-by-case, and it already composes — no option-2 machinery.* The evidence: bundle contents split into namespaced/additive files (rules =<lang>.md=, =validate-<lang>.sh= hooks, =coverage-summary.<ext>= scripts, appended =gitignore-add.txt=) that compose cleanly, and exactly three colliding files — =claude/settings.json= and =githooks/pre-commit= (full bundles only: bash/elisp/go) and =coverage-makefile.txt= (elisp/go/python/typescript). Of the three, only =coverage-makefile.txt= occurs in the fleet, and clock-panel (the one real polyglot, python+typescript) proves it's benign: both bundles installed, no breakage, because that fragment is a hand-pasted Makefile block nobody pasted twice. So: keep the install-lang collision guard (it blocks the destructive full+full settings/githooks clobber, which no project actually hits), and document the =coverage-<lang>:= + =coverage:= aggregate namespacing as the one manual step when going polyglot (filed below). No two-full-co-equal-bundle project exists, so the settings.json/githooks merge is unwarranted. Follow-up doc: [[file:todo.org::*Document polyglot coverage-makefile namespacing][Document polyglot coverage-makefile namespacing]].
+
+Do we support more than one language bundle per project? The honest answer today
+is "partly, by accident." The collision guard added 2026-07-16 refuses a
+*colliding* second bundle rather than silently replacing the first's config, but
+a non-overlapping pair still installs fine: bash ships =settings.json= +
+githooks and no coverage fragment, python ships only a coverage fragment, so
+=bash= + =python= composes cleanly today and yields a real polyglot project with
+both rule sets. So the line isn't polyglot-vs-not, it's overlap-vs-not — and
+nobody chose that line, it fell out of which bundle happens to ship what. Origin:
+home's report after scaffolding clock-panel with python + typescript,
+[[file:docs/design/2026-07-16-polyglot-bundle-collision.txt][docs/design/2026-07-16-polyglot-bundle-collision.txt]].
+
+Pair this with the subproject scouting below — it's the same question in a
+different costume ("which projects would actually be polyglot, and why"), so
+they should be one conversation.
+
+The three options, in the order they'd be weighed:
+
+1. *Unsupported, explicitly.* Keep the guard as the answer. Cheapest, and
+ matches how little polyglot exists (one project, clock-panel).
+2. *Supported.* Needs per-bundle filenames, a merged =settings.json= (the hooks
+ arrays compose rather than clobber), composed githooks, and namespaced
+ Makefile targets with a =coverage= aggregate. This is the real work.
+3. *Case-by-case.* Support the pairs that come up, refuse the rest.
+
+What the decision needs to know:
+
+- *The target-name collision is the deeper half* (home's point, and it's right).
+ Every bundle's fragment defines =coverage:= and =coverage-summary:=, so even
+ with both files present a polyglot project can't paste both into one Makefile.
+ Renaming files doesn't fix it.
+- *Only three of five shared filenames actually collide.* =gitignore-add.txt=
+ (5 bundles) appends deduped and composes. =CLAUDE.md= (3) is seed-only, and
+ its fallback comment shows multi-bundle was already considered there.
+ =claude/settings.json= (3), =githooks/*= (3), and =coverage-makefile.txt= (4)
+ are the real ones.
+- *=FORCE=1= is a poor escape hatch* (home's catch): it also re-seeds
+ =CLAUDE.md=, which is destructive on a customized project. If polyglot
+ becomes supported, the override wants to be its own flag.
+** DONE [#C] Subproject pattern — promote to claude-rules? :spec:
+CLOSED: [2026-07-20 Mon]
+DECISION (2026-07-20, scouting with Craig): *don't promote — keep it local at home.* The scouting confirmed N=1: across all 27 =.ai= scopes, home (9 subprojects, all from the single 2026-06-11 fold) is the only real user. The nearest neighbors aren't the pattern — rulesets folds claude-templates in as a git *subtree* (different mechanism, not shared-=.ai/=-scope), archsetup's dotfiles/ likewise; every other project is a focused single package with no subproject structure or need. Promoting a 282-line convention into the always-on claude-rules layer for one project fails the thin-always-on principle (the precedent is patterns.md at 29 lines + docs-lifecycle.md's depth-in-a-spec). Keep the convention as home's local instance. Revisit only if a genuine second case appears, and then as a thin pointer + a spec, never 282 lines always-on.
+
+home proposes promoting its subproject pattern (a former standalone project
+folded into a parent, living as a self-contained subdir sharing the parent's
+=.ai/= scope) into the rules layer: vocabulary, the read-first
+=<subproject>/<subproject>-brief.org= convention, the parent-vs-subproject
+content criterion ("one fact, one home"), and create/archive criteria.
+Proposal + home's full instance:
+[[file:docs/design/2026-07-15-subproject-pattern-proposal.org][proposal]],
+[[file:docs/design/2026-07-15-subprojects-convention-home-instance.org][home's convention doc]].
+
+Deferred 2026-07-16 rather than promoted. *Craig's framing:* he wants to scout
+which projects would actually get subprojects, and why, before we shape a rule.
+If he hasn't done that scouting by the time this comes up, offer to do it
+together — brainstorm the candidates, then explore the reasons behind each. That
+evidence decides it: either we drop the pattern, or we know enough to adjust it
+so it's effective. Don't shape the rule before the scouting.
+
+*Review findings from the 2026-07-16 pass* (the inputs the decision needs):
+
+- *N=1.* home is the only project with subprojects, across all 27 =.ai= scopes;
+ its nine all came from the single 2026-06-11 fold. This is the fact the
+ scouting tests.
+- *Placement contradicts the proposal's own principle.* =claude-rules/*.md=
+ loads into every session of every project. home's doc argues the always-on
+ layer is "a tax paid whether or not it's relevant today" and depth belongs
+ "one open away". At 282 lines the doc would be the third-largest rule and add
+ ~11% to the always-on layer, so every .emacs.d / takuzu / chime session would
+ carry a one-project convention.
+- *Precedent for the shape:* =patterns.md= (29 lines, explicit "don't carry the
+ catalog in context") and =docs-lifecycle.md= (75 lines, depth in a spec).
+ Thin rule + on-demand depth is the established answer.
+- *Dangling reference:* the doc cites =claude-rules/git-hosting-privacy-model=
+ as authority for its shared-scope-safety criterion. No such file exists — the
+ real content is the gitignore-vs-track and public-reachability decision in
+ =protocols.org=. Fix before any promotion.
+- *Instance vs rule:* the metrics, self-improvement log, kill criteria, rollout
+ dates, and adoption table are home's instance, not rule content.
+** DONE [#B] Triage source activation — per-project source declaration :feature:spec:
+CLOSED: [2026-07-20 Mon]
+Spec: [[file:docs/specs/2026-07-20-triage-source-activation-spec.org][docs/specs/2026-07-20-triage-source-activation-spec.org]] (IMPLEMENTED, ID af73ef0b-cd1d-46f1-9e1d-62695733a4de). Craig approved after his read, both open decisions resolved via cj comments; built same session.
+
+From the sentry live trial (2026-07-20): triage-intake self-activated in every project because the general (personal-account) plugins are template-synced everywhere, so sentry's pass-3 probe misfired and would pull Craig's personal inboxes into whatever project the fire ran in. Fixed with an activation gate in triage-intake Phase 0 — general plugins gate on a per-project =:TRIAGE_SOURCES:= declaration; project-specific plugins stay active by presence. Applies to interactive and unattended alike. sentry pass-3 probe reads the same signal. Migration handoffs sent to home + work. Supersedes Fire 1's narrower probe-only approval-queue item.
+** DONE [#B] Silent-until-signal for in-session monitor loops :feature:spec:
+CLOSED: [2026-07-20 Mon]
+Spec: [[file:docs/specs/2026-07-20-silent-until-signal-monitors-spec.org][docs/specs/2026-07-20-silent-until-signal-monitors-spec.org]] (IMPLEMENTED, ID af592bd6-d3e6-47e2-8804-2a287b4d9303). Craig approved without changes 2026-07-20; all 5 phases shipped that day. Sentry, auto triage-intake, and auto inbox-zero all collapse an empty fire to =<workflow> at HH:MM: nothing=; a manual-testing entry covers the live-loop verification.
+
+Craig-approved proposal from .emacs.d (2026-07-20), demonstrated by the sentry live trial (fires 3-8 were walls of no-op lines). Reframed live from a watcher *mechanism* to a *policy*: an in-session monitor fire detects first, and on an empty check collapses to one labelled heartbeat line (=<workflow> at HH:MM: nothing=) instead of a full turn; only a real item earns the full surface-and-judge turn. Keeping detection in-session dissolves the MCP-auth split (triage's Gmail/Slack/Linear need session auth), so it applies uniformly to sentry, auto triage-intake, and auto inbox-zero. No external watcher, no new seen-list. Five build phases in the spec. Phase 1 (sentry quiet-fire heartbeat) shipped 2026-07-20 ahead of the full READY gate at Craig's direction — a quiet fire now collapses to =sentry at HH:MM: nothing=. Phases 2-4 (auto triage-intake, auto inbox-zero, shared-policy home) + Phase 5 verification pending Craig's spec read. Overlaps the sentry cluster ([[file:todo.org::*Sentry vNext passes][Sentry vNext passes]], [[file:todo.org::*Triage source activation][Triage source activation]]).
+** DONE [#C] Apply suspend.org detach-on-suspend change to canonical :feature:
+CLOSED: [2026-07-20 Mon]
+Craig's change relayed from archsetup (2026-07-20): add Step 6 to suspend.org — detach the tmux client (=tmux detach-client -s "$sess"=) as the final action of every suspend, so a suspended session parks in the re-attachable set instead of cluttering the alt-space rotation. Also reworded the neighbors bullet and the "does NOT do" teardown bullet to draw the detach-vs-teardown line.
+
+Applied to canonical 2026-07-20 (verified the diff was exactly the intended change, canonical + mirror synced, lint clean bar one pre-existing flush/SKILL.md link). Replied to archsetup that its local stopgap is now canonical.
+** DONE [#B] Document (and own) the Signal pager :feature:spec:
+CLOSED: [2026-07-20 Mon]
:PROPERTIES:
-:CREATED: [2026-05-28 Thu]
-:LAST_REVIEWED: 2026-05-28
+:CREATED: [2026-07-11 Sat]
+:LAST_REVIEWED: 2026-07-13
:END:
+home retired the ntfy phone-notification channel (phone-notify/phone-recv, self-hosted ntfy on ratio) on 2026-07-04 in favor of paging over Signal, and tore ntfy down. The Signal pager it replaced ntfy with is undocumented: no pager script in =~/.local/bin=, and =notify= doesn't reference Signal. What exists on ratio: signal-cli 0.14.5, account 404211. Deliverable: a documented Signal pager (send + read-replies), the signal-cli setup/account notes, and the sync path — the Signal equivalent of the retired ntfy runbook. Cross-machine tooling, so canonical home + docs belong in rulesets.
-From pearl handoff 2026-05-28. =open-tasks.org= Next Mode reads =* Project Open Work= and skips =* Project Resolved= correctly, but a level-2 task that completed during a session sits as =** DONE= under Open Work until something archives it. Between cleanups, a freshly-DONE task can surface as a "what's next" candidate.
+RECONCILE FIRST: =protocols.org= "Paging Craig" already documents an agent-paging path via the *signal-mcp* tool (=send_message_to_user=, pager account +15045173983, Craig's UUID =b1b5601e-…=, verified 2026-06-30). home's handoff is about a *different* mechanism (signal-cli / account 404211, home's own paging). Decide whether these are one channel or two, and whether the signal-cli side needs its own runbook or should route through signal-mcp. Source: home handoff 2026-07-04 (=inbox/2026-07-04-1302-from-home-task-for-rulesets-document-and-decide.org=). Successor to the 2026-06-17 two-way-comms proposal.
-Proposed fix: as the first step of =open-tasks.org= Phase A, run =emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done todo.org=, then read =todo.org=. The cleanup tool already exists; this is wiring it into the workflow.
+*** 2026-07-13 Mon @ 05:16:37 -0500 Folded home's ownership ack + current-state report into the reconcile scope
+home confirmed (2026-07-11 reply) rulesets owns this task and the two-paths reconciliation is the right first step. New facts for the reconcile: from a home session on 2026-07-09, signal-mcp was NOT connected, and the local signal-cli is registered as Craig's own number — =send --note-to-self= returns a message id but produces no phone push. So home currently has no working ad-hoc page channel at all; whatever the runbook lands on must give home a live path.
-Cost: a few hundred ms at the start of every "what's next" invocation. Win: recommendations never include DONE work.
+*** 2026-07-13 Mon @ 14:40:00 -0500 Added runtime-portability as a second motivation (Craig approved)
+The MCP portability inventory ([[file:docs/design/2026-07-13-runtime-portability-inventories.org]]) found signal-mcp exists only claude.ai-side — no local config anywhere — so a non-Claude agent (Codex-style or local LLM) has no paging path at all. The signal-cli runbook this task produces is therefore also the runtime-neutral page channel, not just home's replacement for ntfy.
-Optional refinement: gate behind a check for read-only / dry-run mode if that's ever introduced. The default invocation archives.
-** DONE [#C] Triage Codex enhancement backlog :spec:
-CLOSED: [2026-05-28 Thu]
-:PROPERTIES:
-:CREATED: [2026-05-28 Thu]
-:LAST_REVIEWED: 2026-05-28
-:END:
-
-Triaged interactively 2026-05-28. Disposition table for all 14 items lives at [[file:docs/design/2026-05-28-rulesets-enhancement-backlog.org][2026-05-28-rulesets-enhancement-backlog.org]] under "Triage Dispositions": 3 accepted (filed below as TODOs), 3 pilot/scope-limited (filed below), 2 marked as conventions rather than tracked tasks, 6 rejected with rationale. Items #1 and #2 already had homes (#16 and the Phase-1 codex TODO).
-** DONE [#C] Canonical/mirror drift detection via pre-commit hook or =make sync-check= :feature:quick:solo:
-CLOSED: [2026-05-28 Thu]
-:PROPERTIES:
-:CREATED: [2026-05-28 Thu]
-:LAST_REVIEWED: 2026-05-28
-:END:
-
-From the codex enhancement backlog (item #7), reframed: don't dedupe the dual source — the canonical-in-=claude-templates/= + mirror-in-=.ai/= pattern is a feature (other projects rsync from the canonical; the mirror lets rulesets-as-a-project have a working copy). The real pain is sync-discipline overhead — every workflow edit needs both copies updated, and forgetting one leaves the next startup's rsync to surface the drift.
+*** 2026-07-13 Mon @ 18:30:19 -0500 agent-page shipped — every project now knows both channels
+Craig's call: make paging universal and rename it the agent pager. NEW =claude-templates/bin/agent-page= (runs signal-cli directly on velox, ssh-relays from anywhere else, desktop-fallback hint on failure; 4 bats tests; live-verified from ratio through the real script). protocols.org "Paging Craig" rewritten around the two channels (notify desktop + agent-page phone; signal-mcp demoted to a velox-local nicety); page-me.org gained the phone section + fire-both guidance; work-the-backlog's end-of-set page names both surfaces; INDEX updated. Every project inherits via the startup sync + make install. Remaining here: the runbook proper, the receive timer, ssh-only vs linked-device.
-Scope: write a small =scripts/sync-check.sh= (or fold into the existing Makefile) that diffs =claude-templates/.ai/workflows/= against =.ai/workflows/=, exits non-zero on drift. Wire as a pre-commit hook (=githooks/pre-commit= or equivalent) so the discipline is enforced before publish, not at the next startup. =make sync-check= as a manual entry point.
+*** 2026-07-13 Mon @ 18:13:07 -0500 RECONCILED — one channel, on velox; live page verified end to end
+The two-paths question is answered: there is ONE pager identity, +15045173983, registered in velox's signal-cli (account file 465310) — and signal-mcp is a locally-configured MCP server in velox's global ~/.claude.json (not claude.ai-side as the 14:40 entry inferred; it's just invisible from ratio, which is why home and this session couldn't find it). ratio's signal-cli holds only Craig's personal number (note-to-self, no push). Verified live today: =ssh velox 'signal-cli -a +15045173983 send -m … b1b5601e-6126-47f8-afaa-0a59f5188fde'= buzzed Craig's phone — his remembered CLI page was this same account on velox. Reliability findings for the runbook: both accounts throw receive-staleness warnings (velox 40 days, ratio 26; the signal protocol wants regular receives — a systemd receive timer on velox is the roam-sync-shaped fix), and the channel requires velox to be up. Remaining deliverables sharpened: the runbook (send + read-replies + receive timer + account notes); decide ssh-over-tailnet-only vs registering ratio as a linked device of the pager account; update protocols.org "Paging Craig" (it names signal-mcp as the only supported path — true only on velox; the ssh recipe is the cross-machine path) — shared-asset edit, own review pass. Interim recipe sent to home so it's unblocked today.
-Verification: introduce a deliberate diff, commit, hook should block. Restore parity, hook should pass.
-** DONE [#C] Add =make status= — compose audit + doctor + open-task count :feature:quick:solo:
-CLOSED: [2026-05-28 Thu]
+*** 2026-07-20 Mon @ 15:56:56 -0500 Runbook shipped; ratio linked as a device; receive timer on both machines
+All four remaining deliverables landed. (1) Runbook: [[file:docs/design/2026-07-20-signal-pager-runbook.org]] — send, read-replies, receive timer, signal-cli account/setup notes, the resolved topology decision. (2) Receive timer: =scripts/signal-receive.sh= + =scripts/systemd/signal-receive.{service,timer}= (roam-sync-shaped, 15-min cadence, 3 bats), stowed via dotfiles =common=, enabled + verified on ratio; a manual drain also cleared the 47-day staleness live. (3) Topology decision (Craig, option 2): register daily drivers as linked devices rather than ssh-relay-only — ratio linked as Device 2 "ratio-pager", direct send verified, =agent-page= generalized from a velox-only check to "any machine holding the account sends directly, else relay" (bats updated). (4) protocols.org "Paging Craig": verified accurate, no edit — already describes both channels and the caveats, and its generic runbook pointer is correct to leave un-pathed since it syncs into every project. Remaining one-time step: enable the timer on velox after it pulls dotfiles.
+** DONE [#C] triage-intake auto mode — push signal sweeps to phone via agent-text :feature:solo:
+CLOSED: [2026-07-20 Mon]
:PROPERTIES:
-:CREATED: [2026-05-28 Thu]
-:LAST_REVIEWED: 2026-05-28
+:CREATED: [2026-06-20 Sat]
+:LAST_REVIEWED: 2026-07-13
:END:
+Send half shipped 2026-07-20. Folded a "Phone delivery" subsection into canonical =triage-intake.org= auto mode: a full-three-section sweep pushes to Craig's phone over Signal via =agent-text=, with a pointer from "End-of-sweep output" and a Living Document note. Signal-only by Craig's 2026-07-20 ruling — a quiet sweep's =nothing= heartbeat never reaches the phone, so silent-until-signal governs the phone channel too and the in-session heartbeat stays as proof-of-life. Falls back to inline when =agent-text= is absent.
-From the codex enhancement backlog (item #12), scope-limited: =make status= only. Reject the rest of #12 (=make sync= duplicates the existing sync flow; =make health= wraps existing checks without adding signal; =make bootstrap-project= duplicates =install-ai= + =install-lang=).
+Reply-polling half (the old =phone-recv=) deferred to the reply-correlation follow-up: with the Signal account linked on more than one device a reply fans out to every device and neither knows which page it answers, so auto mode pushes but does not poll until that's resolved. The recv wiring is owned by that spec, not this task.
-Scope: one Makefile target that prints a compact summary of:
+Origin: the work project's 2026-06-18 fold-in request; preserved bundle [[file:docs/design/2026-06-18-triage-intake-phone-push-note.org][note]] + [[file:docs/design/2026-06-18-triage-intake-phone-push-workflow.org][edited workflow]]. Transport re-pointed off retired ntfy → agent-text (renamed from agent-page 2026-07-20).
+** CANCELLED [#D] Fully-unattended scheduled inbox check (/schedule cron pass) :feature:
+CLOSED: [2026-07-20 Mon]
+Merged 2026-07-20 into [[file:todo.org::*Unattended /schedule cron contract][Unattended /schedule cron contract — no-session variant]]. Same no-session design problem as the sentry /schedule variant (mutation rights, async surfacing, cross-run dedup, cron auth context); its inbox-specific context was absorbed there.
+** CANCELLED [#B] lint-org resolves file: links against cwd, not the linted file :bug:solo:
+CLOSED: [2026-07-23 Thu]
+Not a defect. I filed this on a bad comparison and retracted it the same night.
-- Install audit state (clean / drift, calling =make audit=).
-- Machine-global doctor state (calling =make doctor=).
-- Open-task count (top-level entries in =todo.org= under =* Rulesets Open Work=).
-- Inbox count (files in =inbox/= excluding =.gitkeep= and =PROCESSED-= prefixes).
-- Git working-tree status (clean / dirty, ahead/behind upstream).
+The claimed evidence was that linting =claude-templates/.ai/notes.org= from the repo root reports =protocols.org= and =workflows/first-session.org= missing, while linting it from its own directory reports neither. The first half of that was never run. The findings came from a =/tmp= copy of the template made while preparing the smoke diff, and in =/tmp= those two siblings genuinely are missing, so org-lint was right. I compared two different files and read the difference as a cwd bug.
-Output should be roughly 10 lines, scannable in one glance. Composes the existing checks; no new logic except the summary formatting.
-** DONE [#C] Iteration-history backfill for spec-review and spec-response :docs:followup:
-CLOSED: [2026-05-28 Thu]
+Retested three ways: the real file from the repo root gives zero link findings, the real file from its own directory gives zero, and only the =/tmp= copy gives two. A direct probe confirms =find-file-noselect= already sets =default-directory= to the linted file's directory, so the mechanism I proposed could not have been the cause either.
+
+Worth keeping from the episode: =link-to-local-file= is org-lint's own checker, not one lint-org.el implements, so a real fix here would mean pre- or post-filtering upstream output rather than editing a local checker.
+** DONE [#A] Python and TypeScript bundles ship no secret-scan hook :bug:
+CLOSED: [2026-07-23 Thu]
:PROPERTIES:
-:LAST_REVIEWED: 2026-05-28
+:LAST_REVIEWED: 2026-07-23
:END:
-Source: org-drill inbox 2026-05-28.
-
-Once the in-flight WIP lands (the requirement that specs carry a bottom =Review and iteration history= section, with iteration / date / contributor / role / what / why / artifacts), backfill the two workflow files themselves using rulesets' session history as evidence.
+Fixed 2026-07-23. Both bundles now ship all four components: =githooks/pre-commit= (the secret scan, shared verbatim with the other bundles, plus a language-appropriate syntax gate), =claude/hooks/validate-python.sh= / =validate-typescript.sh=, =claude/settings.json= wiring the PostToolUse hook, and a seed =CLAUDE.md=. 53 new tests (13 + 13 hook, 13 + 14 pre-commit), suite green.
-Files to update:
-- =claude-templates/.ai/workflows/spec-review.org=
-- =claude-templates/.ai/workflows/spec-response.org=
+Two findings from the build worth keeping. First, =node --check= must never be used on TypeScript: it ignores =--experimental-strip-types=, so it rejects valid TS (an =interface= reads as a syntax error) and accepts broken TS. Measured on node v26.4.0. The hook uses =tsc= filtered to TS1xxx (syntactic) diagnostics instead, which also keeps type errors out of scope — those need the whole project graph. Second, the install now warns when a bundle lacks a documented component, which is the half that stops this recurring: nobody will remember the seventh bundle either, so the installer says so.
-Investigation: search =.ai/sessions/=, =.ai/notes.org=, inbox archive, and git log for mentions of these workflow docs. Identify review/response/design iterations, dates, and contributors (including agents where known: Claude Code, Codex, local models). Distinguish high-confidence history (commits, dated session entries) from inferred (chat-only context). Recommend whether enough evidence exists to populate the section, and draft the entries if so.
+Left undone deliberately: re-running =make install-lang= on =work= and =clock-panel=. Both are other projects, so that's a cross-project action for Craig. See the polyglot task below — =clock-panel= is python + typescript and can no longer install both.
+The =python= and =typescript= language bundles carry rules, a coverage script, and a gitignore fragment. They carry no =githooks/pre-commit=, no =claude/hooks/=, no =claude/settings.json=, and no =CLAUDE.md=. The =bash=, =elisp=, and =go= bundles carry all four.
-Dependency: spec-review.org and spec-response.org have uncommitted edits in flight. Wait for those to land before writing to the files. The read-only research portion (search sessions, identify iterations, draft entries to a scratch file) can run in parallel without conflict.
-** DONE [#B] Startup Phase A rsync propagates dirty rulesets WIP into downstream projects :feature:
-CLOSED: [2026-05-30 Sat]
-:PROPERTIES:
-:CREATED: [2026-05-29 Fri]
-:LAST_REVIEWED: 2026-05-29
-:END:
-Fixed via option 1 (skip-when-dirty), scoped to the synced source paths: startup.org Phase A now guards the protocols/workflows/scripts rsyncs behind a =git status --porcelain= check on =claude-templates/.ai/{protocols.org,workflows/,scripts/}=, skipping the sync when any are dirty. The propagation anomaly (cross-project-broadcast.org / page-signal.org not reaching jr-estate) was a timeline artifact: both files were added in 664bf01 on 2026-05-29, after jr-estate's Phase A rsync had already run — correct behavior, not a bug.
+The pre-commit hook is the secret scanner. So a project installing the Python or TypeScript bundle gets no credential scan on commit, no validate-on-edit hook, and no settings — while README's "Bundle structure" section documents all four as what each bundle follows, and notes.org describes the bundles as "rules + hooks + settings".
-From jr-estate handoff 2026-05-29. When rulesets has uncommitted WIP at the moment a downstream project starts a session, Phase A.0 reports "dirty, skipping pull" and proceeds. Phase A's =rsync -a --delete= then runs against the dirty rulesets working tree and copies the WIP state into the downstream project's =.ai/workflows/= and =.ai/scripts/=. The downstream project's =git status= then shows drift the user did not author. Two bad recovery paths: commit the drift as "chore: sync .ai tooling from templates" (creates fake commit history about template state) or leave it dirty (noisy wrap-ups, pressure to commit anyway).
+Live as of 2026-07-23, verified by scanning every project carrying =.claude/rules/=: =work= (python) and =clock-panel= (python + typescript) both have no =githooks/= and no =.claude/settings.json=. The four elisp projects all have both. =work= is the one that matters — a work repo is where a leaked credential is most costly and most likely to reach a company remote.
-Three options proposed in the handoff:
-1. *Skip-when-dirty.* Make Phase A's workflows/ and scripts/ rsync no-op when Phase A.0 reports rulesets dirty. Simplest defense.
-2. *Clean-files-only.* Restrict the rsync to files git considers unmodified in rulesets. Untracked files in rulesets do not propagate. Most precise.
-3. *Clean-ref-based.* Cache the last-known-clean state as a git tag or ref and rsync from that ref rather than the working tree. Most decoupled, also the most infrastructure.
+Not a design choice. Both bundles were added 2026-05-31; =go='s githooks landed 2026-06-02 and =bash='s 2026-06-23, so the hook rollout swept the two bundles added *after* these and skipped these. =install-lang.sh= guards its copy with =[ -d "$SRC/githooks" ]=, so the install succeeds silently and reports nothing missing — which is why this stayed invisible for nearly two months.
-Recommendation (mine): option 1. The downstream impact of skipping a sync once is small (the next session with rulesets clean catches up), and the implementation is one =if [ "$dirty" -eq 0 ]= guard around the existing rsync block. Option 2 adds shellout complexity per file; option 3 requires tagging discipline that has no other reason to exist.
+Grading: Major severity (a documented security control absent, silently, with the install reporting success) x every user, every time (every install of either bundle, standing on 2 of the 6 bundle-using projects) = P1 = [#A].
-The original handoff also noted a related anomaly: even with =--delete=, two files that DO exist in rulesets canonical (=cross-project-broadcast.org=, =page-signal.org=) did NOT propagate to jr-estate. Worth confirming whether that was a transient rsync issue or evidence of a deeper Phase A bug. Could be ordering: those files were added to rulesets AFTER the jr-estate Phase A rsync ran, in which case the behavior is correct and the report is misreading the timeline.
-
-Source: =inbox/2026-05-29-0832-from-jr-estate-investigate-startup-rsync-carried-dirty.org= (processed and deleted).
-** DONE [#B] Codex Phase 1 — AI_AGENT_ID + session-context.d/<id>.org :feature:
-CLOSED: [2026-05-30 Sat]
+Fix direction: port =githooks/pre-commit= to both bundles, adapting the language-specific half (the secret scan is common; =gofmt=/=shellcheck= becomes a formatter/linter check per language), add =claude/hooks/validate-*.sh= and =claude/settings.json=, and seed =CLAUDE.md=. Then make the gap loud: =install-lang= should warn when a bundle lacks a component the README documents, so the next partial bundle announces itself instead of installing quietly. Re-run =make install-lang= on =work= and =clock-panel= afterward.
+** DONE [#C] Two language bundles' pre-commit hooks lost the cd guard :bug:quick:solo:
+CLOSED: [2026-07-23 Thu]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-23
+:END:
+Fixed 2026-07-23. =go= and =elisp= now read =cd "$REPO_ROOT" || exit 1=, matching =bash=; the two new bundles were written with the guard. All five siblings agree, all five shellcheck clean.
+=languages/bash/githooks/pre-commit= line 8 reads =cd "$REPO_ROOT" || exit 1=. The =go= and =elisp= copies of the same line read a bare =cd "$REPO_ROOT"=. Three siblings of one file, one hardened and two not — the guard landed in bash and never propagated.
+
+Low impact, stated honestly: git chdirs to the working-tree root before running a hook, so if =cd= fails the cwd is already correct and the checks still run against the right tree. Neither script sets =-e=, so a failure wouldn't abort them either. It takes a =GIT_WORK_TREE= or bare-repo edge case for =rev-parse --show-toplevel= to disagree with the cwd at all.
+
+Grading: Minor severity (belt-and-braces guard, no silent-pass path found — the checks run regardless) x rare edge case (needs a git configuration that makes toplevel differ from the hook's cwd) = P4... graded [#C] rather than [#D] because it's a two-character fix in a security-relevant gate and the drift pattern is the real signal: a fix landed in one bundle copy and stopped there, which is the same shape as the [#A] above.
+** DONE [#B] Parked: zero markup in chat output, fences included (from org-drill)
+CLOSED: [2026-07-23 Thu]
+Craig approved 2026-07-23. Applied to =claude-rules/interaction.md=: the fenced-code-block carve-out is gone, replaced with a zero-markup-always statement citing his 2026-05-30 direction. The rule's factual claim stays honest — fences don't invert the way inline spans do, so the text says they read as markup he didn't ask for rather than inventing a rendering problem, and points at plain indented text or a named file path as the way to hand over something copyable. It was the only copy of the carve-out.
+** DONE [#B] Parked: sentry Living Document updates from two dogfood runs (from takuzu + archangel)
+CLOSED: [2026-07-23 Thu]
+Craig approved 2026-07-23. Four folds applied to =sentry.org=: =--archive-done= touches =.gitignore= on its first run so an "org-only" pass can still produce a commit, a mirror-only project's quiet fires leave no commits so the anchor's heartbeat list is the only record, randomized property sweeps are good quiet-fire work, and the task-audit pass splits into a mechanical hourly subset plus a nightly judgment half. The fifth finding (make the bug hunt official) had already landed as pass 11, so it became a corroboration note. Takuzu's lint-org finding stayed a filed bug task rather than a workflow edit, since it's a defect not a design change.
+** DONE [#B] Parked: clear four lint flags in the notes.org template (from smoke)
+CLOSED: [2026-07-23 Thu]
+Craig approved 2026-07-23. Applied to =claude-templates/.ai/notes.org= and synced to the mirror. The template now lints with zero mechanical and zero judgment findings, down from four. Two fixes beyond what smoke reported: two more column-0 bold lines, and the real cause of the block flags — a literal =** Feature Name= inside the example block that org parses as a heading, cleared by comma-escaping that one line. Worth remembering: those heading flags were mechanical, not judgment, so =lint-org --fix= in any project would have rewritten the template locally and drifted it from canonical.
+** DONE [#B] Parked: clear temp/ during wrap-up teardown (from your roam capture)
+CLOSED: [2026-07-23 Thu]
+Craig approved 2026-07-23. Added =*** Clear temp/= to =wrap-it-up.org= Step 3, after the archive pass, and synced to the mirror. Two guards: confirm before deleting anything that reads as in-progress rather than throwaway (that belongs in =working/=, so move it there), and skip entirely where =temp/= isn't gitignored, since that means the project uses the directory for something else. This closes the last open clause of his 2026-07-20 working/temp capture.
+** DONE [#B] inbox-send leaves a phantom empty handoff in the target inbox :bug:solo:
+CLOSED: [2026-07-23 Thu]
:PROPERTIES:
-:CREATED: [2026-05-28 Thu]
-:LAST_REVIEWED: 2026-05-28
+:LAST_REVIEWED: 2026-07-23
:END:
-Shipped backward-compatibly. New =.ai/scripts/session-context-path= helper resolves the active path from =AI_AGENT_ID=: unset → the legacy =.ai/session-context.org= singleton (one-agent default unchanged, per the spec's compatibility rule), set → =.ai/session-context.d/<sanitized-id>.org=. startup.org's existence check and wrap-it-up.org's rename now resolve through the helper (with a singleton fallback for older checkouts); wrap folds the agent id into the archive name. protocols.org documents the rule. Verified: 5 bats cases + a two-agent simulation showing distinct paths per id. Larger runtime-neutral arc (runtimes/ manifests, launcher refactor) stays parked under the parent spec.
+Fixed 2026-07-23 (speedrun, 0f91a8e). Both send paths write to a .inbox-send-* temp sibling and os.replace it into place, so a mid-write failure leaves no phantom; utf-8 pinned on the write and the roots read; inbox-status skips the in-flight temp. 5 new tests (atomic write, no-partial-on-failure, no-temp-on-success, both paths) + 1 inbox-status bats. Review found and fixed one Important: the temp was discoverable by inbox-status during the write window.
+=inbox-send.py= writes straight to the destination path in the target project's =inbox/=. =Path.write_text= opens with mode =w=, which creates and truncates before any content is written, so *any* failure between opening and finishing leaves a zero-byte =.org= file sitting in another project's inbox.
-Lifted from the broader codex runtime spec ([[file:docs/design/2026-05-28-generic-agent-runtime-spec.org]]) as the immediate-correctness slice independent of the larger arc. The singleton =.ai/session-context.org= is unsafe under simultaneous agents — two LLMs running in the same project at the same time would overwrite each other's session state.
+That file is not inert. =inbox-status= counts it as a pending handoff, so it trips the receiving project's =inbox-boundary-check= Stop hook and blocks a turn there. The receiving agent then has to resolve a handoff with no content and no sender context, which it cannot do from the file. Meanwhile the sender saw an error and will most likely retry, so the target gets a second file too.
-Scope: introduce an =AI_AGENT_ID= environment variable and split the single =session-context.org= into a per-agent =session-context.d/<id>.org= directory. No other phases of the runtime refactor are in this task — keep the surface small, fix the race, ship.
+Reproduced 2026-07-23 end to end: a send whose text contains non-ASCII under =LC_ALL=C PYTHONUTF8=0 PYTHONCOERCECLOCALE=0= fails on the ASCII codec, leaves a zero-byte file in the destination inbox, and =inbox-status= in that project then reports it as pending.
-Touches: =.ai/protocols.org= (rename rule + recovery anchor), =.ai/workflows/startup.org= (Phase A check), wrap-up workflow (rename target), per-project session record discoverability.
+The encoding case is one trigger, not the defect. =write_text= and =read_text= are both called with no =encoding= argument, so they follow the locale; passing =encoding="utf-8"= closes that trigger. The defect underneath is the non-atomic write, which a full disk, a revoked permission, or an interrupted process reaches just as easily.
-Verification: simulate two agents sharing a project (separate AI_AGENT_ID values) and confirm session-context writes land in distinct files without interleaving.
+Grading: Major severity (a phantom handoff in a *different* project's inbox, unresolvable from its own content, that blocks a turn there via the boundary hook) x some users, sometimes (any mid-write failure, of which the locale case is only the one reproduced) = P2 = [#B].
-Parent: see [[Generic agent runtime support — Codex spec v0]] above for the larger arc this is sliced from.
-** DONE [#C] Decide on category-3 rule copies in the deepsat tree :chore:quick:solo:
-CLOSED: [2026-05-31 Sun]
+Fix direction: write to a temp file in the destination directory and =os.replace= into place, so the inbox only ever sees a complete file. Pin =encoding="utf-8"= on both =write_text= and the =read_text= in =resolve_roots=. Same treatment for =send_file=, whose =shutil.copy2= has the same shape. Test by forcing a mid-write failure and asserting the destination directory is unchanged.
+** DONE [#C] Two smaller inbox-send defects: uncaught traceback, duplicate roots :bug:quick:solo:
+CLOSED: [2026-07-23 Thu]
:PROPERTIES:
-:LAST_REVIEWED: 2026-05-28
+:LAST_REVIEWED: 2026-07-23
:END:
-Diffed 2026-05-31. Both copies (coding-rulesets vendored + orchestration_dashboard_mvp) are byte-identical to each other and stale against canonical: =testing.md= 221 lines behind with 5 lines unique to the copies (older wording or a small team tweak), =verification.md= 40 behind with nothing unique. Same older vendored version in both spots. Left untouched per the A1 decision — team-owned, and canonicalizing would create a cross-repo dependency on the private rulesets (the orchestration_dashboard_mvp pair is team-visible from Vrezh's PR thread). No files modified.
+Fixed 2026-07-23 (speedrun, a053e9d). main now catches OSError, so an unreadable source gives the clean inbox-send: error rather than a traceback. discover_projects dedupes on the resolved path. 2 tests.
+Both verified 2026-07-23, both in =.ai/scripts/inbox-send.py=, both low-harm. Grouped because they're the same file and the same fix session.
-While symlinking personal-project =.claude/rules/= mirrors to the rulesets canonical on 2026-05-07, two locations didn't fit the "personal mirror → symlink" pattern and were left untouched pending judgment:
+1. *Uncaught =PermissionError=.* =main= catches =(ValueError, FileNotFoundError)= around the send, but =send_file= reaches =shutil.copy2=, which raises =PermissionError= on an unreadable source. That is an =OSError=, not caught, so the user gets a raw Python traceback instead of the clean =inbox-send: <message>= error every other failure path produces. Reproduced with a =chmod 000= source. No partial file is left in this case. Fix: catch =OSError= alongside the other two.
-- =~/projects/work/deepsat/code/coding-rulesets/claude-rules/{testing,verification}.md= — looks like a vendored team-shared copy.
-- =~/projects/work/deepsat/code/orchestration_dashboard_mvp/.claude/rules/{testing,verification}.md= — could be project-specific overrides.
+2. *Duplicate roots list a project twice.* =discover_projects= appends without deduping, so a roots config naming both a parent and one of its children (=/x= and =/x/proj=) lists =proj= at two different numeric indices. Reproduced via =INBOX_SEND_ROOTS=. Harmless today — both indices resolve to the same project, so no message goes to the wrong place, and Craig's current =~/.claude/inbox-roots.txt= has no overlap. Fix: dedupe on =resolve()= before returning.
-For each: read the file, diff against the rulesets canonical, decide whether it's an intentional diverge (leave alone), stale (sync content), or should canonicalize (replace with symlink and accept the cross-repo dependency). The orchestration_dashboard_mvp pair is the project where Vrezh's PR review surfaced this whole thread, so any decision there has team-visibility implications.
-
-Decision (Craig, 2026-05-31): *leave team-tree copies alone.* Personal rulesets does not reach into team repos — canonicalizing would create a cross-repo dependency on the private rulesets, and the orchestration_dashboard_mvp copy is team-visible. This makes the task solo: diff each copy against canonical, record whether it's identical / drifted / overridden in the disposition, and close as "left alone (team-owned)" without modifying the team-tree files.
-** DONE [#C] Audit language-specific rule files for cross-project duplication :chore:solo:
-CLOSED: [2026-05-31 Sun]
+Grading: Minor severity (one is cosmetic output noise, the other an ugly but accurate failure message; neither loses or misroutes a message) x rare edge case (an unreadable source file, or a roots config with an overlap) = P4... graded up to [#C] rather than [#D] because both fixes are one line each and sit in a script every project depends on for cross-project messaging.
+** DONE [#C] lint-org todo-format checkers fire on spec files :bug:solo:
+CLOSED: [2026-07-24 Fri]
:PROPERTIES:
-:LAST_REVIEWED: 2026-05-28
+:LAST_REVIEWED: 2026-07-23
:END:
-Audited 2026-05-31. Findings: in sync with canonical (=languages/<lang>/claude/rules/=) — work =python-testing.md=, deepsat =typescript-testing.md=, =.emacs.d= =elisp-testing.md= + =elisp.md=. Drifted — =gloss= and =chime= (byte-identical to each other): =elisp-testing.md= 44 lines behind (canonical added Batch-Mode Reproducibility + Isolating Emacs State; zero lines unique to the copies), =elisp.md= one line behind (canonical expanded the edit-cohesively guidance). No project-specific additions anywhere — every copy is either current or purely stale.
+Fixed 2026-07-24 (speedrun, c38bab9). A lo--spec-file-p path guard skips all five todo-format-family checkers on files under docs/specs/. Verified: the nine repo specs go from 100 todo-format findings to zero, todo.org keeps its full checker set. 7 ERT tests.
+=level2-done-without-closed= and =level-2-dated-header= encode =todo.org= completion conventions, but they run against every org file, including =docs/specs/=. A spec's Decisions section legitimately uses =** DONE <decision>= with no =CLOSED:= cookie, and its Review-and-iteration-history section legitimately uses =** <dated> — <who> — <role>= headings. Neither is a completion defect there.
-Disposition: *leave them project-local* (the task's own option). The language-rule copies in code projects are the bundle's deliberate copy-and-sync model, not the symlink pattern the generic rules (commits/testing/verification/subagents) use in personal doc-projects. =sync-language-bundle.sh= auto-fixes drifted bundle rules on each startup, so gloss/chime self-heal the moment those projects next boot — no canonicalize/symlink needed, and symlinking would fight the bundle model. Did not reach into work/deepsat/gloss/chime/.emacs.d from here (cross-project boundary; team copies left alone per the 2026-05-31 category-3 decision).
+Reproduced 2026-07-23 across rulesets' own specs: 100 findings over 7 of the 9 files in =docs/specs/= (docs-lifecycle 27, sentry-workflow 25, autonomous-batch 11, wrapup-routing 11, inbox-consolidation 10, agent-kb 8, encourage-kb 8; the two 2026-07-20 specs are clean). Every project carrying =docs/specs/= inherits the same noise on every sweep. Reported independently by takuzu's first sentry dogfood run.
-The four canonical rules (=commits=, =testing=, =verification=, =subagents=) are now symlinked across the five personal-project mirrors as of 2026-05-07. But several language-specific rule files exist in multiple project mirrors and may be duplicated or drifted:
+Grading: Minor severity (judgment-kind output, so nothing mutates; the cost is noise that trains the reader to skim) x every user, every time (fires on every sweep in every project with specs) = P2... graded down to [#C] because the two inputs disagree: the frequency row is genuinely universal, but Minor severity with zero mutation risk and a known cause reads as P3. Recorded so the read can be argued.
-- =python-testing.md= in =~/projects/work/.claude/rules/=
-- =typescript-testing.md= in =~/projects/work/deepsat/code/.claude/rules/=
-- =elisp-testing.md= and =elisp.md= in =~/.emacs.d/=, =~/code/gloss/=, =~/code/chime/=
+Fix direction: scope the todo-format-family checkers away from =docs/specs/= by path. Correction (2026-07-23 speedrun): the earlier note that these are org-lint's own checkers was wrong — =lo--check-level2-dated-headers= (line 413) and =lo--check-level2-done-without-closed= (line 507) are lint-org.el's own functions, so the fix is a direct local guard, not upstream filtering.
-The Elisp pair is the most suspicious — three repos using essentially the same rules. Audit: diff these across the projects, check for drift, then decide whether to canonicalize them under =~/code/rulesets/claude-rules/languages/<lang>/= and symlink, or leave them as project-local.
-** DONE [#C] Refactor =daily-prep.org= to delegate to =triage-intake.org= for the triage section :chore:solo:
-CLOSED: [2026-05-31 Sun]
+Decision (Craig, 2026-07-23 speedrun pre-flight): scope *all four* todo-format-family checkers, not just the two that fired — the two named plus =subtask-done-not-dated= and =dated-log-heading-active-timestamp=, which would misfire the same way on a spec's dated review-history headings. Path-based (any file under a =docs/specs/= segment), matching the docs-lifecycle canon that specs live there. Takuzu's spec-create alternative fixes new specs only and leaves the nine existing ones noisy, so path-scoping is the general fix.
+** DONE [#C] notes.org template trips four lint-org flags in every project :bug:quick:solo:
+CLOSED: [2026-07-24 Fri]
:PROPERTIES:
-:LAST_REVIEWED: 2026-05-28
+:LAST_REVIEWED: 2026-07-23
:END:
-Collapsed Phase 3's inline source scans (sub-steps 3b email / 3c mark-read / 3d Slack / 3e Linear / 3f PRs / 3g dedup, ~280 lines) into four: 3b runs the triage-intake engine, 3c surfaces today's reactive items as Day's Priorities thin links, 3d re-sorts by urgency, 3e writes the audit footer from the engine's coverage. Source coverage carries via the engine's Phase 0 two-dir glob (general + .ai/project-workflows/ plugins), so the work account's Gmail/Slack/Linear/GHE plugins still get scanned. Adapted the downstream refs (Prep Doc Structure rule, Heads-up FYI source, Recommended Approach Pattern reframed as engine-applied), removed the orphaned Linear-digest note, added a Living Document entry. Verified: workflow-integrity clean (no dangling script refs), sync-check clean, full suite green. daily-prep.org went 825 → 576 lines.
+Already satisfied. The canonical claude-templates/.ai/notes.org was fixed in 10ea44b (2026-07-23) when Craig approved the smoke proposal — the two column-0 bold lines rephrased, the example block's marker comma-escaped. Verified 2026-07-24: the template lints with zero mechanical and zero judgment findings, down from four. This TODO and the applied smoke VERIFY were duplicate work items for the same fix; closing DONE citing the commit rather than filing a no-op VERIFY, since the end-state is verifiable now with nothing to ask.
+The synced =claude-templates/.ai/notes.org= carries two boilerplate lines that open with markdown-style bold at column 0 (=**Session history is NOT in this file.**=, =**For protocols and conventions, see:**=). Org parses a line-initial =**= as a level-2 heading, so =misplaced-heading= flags both, and the =#+begin_example= block in the Pending Decisions instructions trips =invalid-block= twice. Every project inherits all four on every sweep.
-=daily-prep.org= still does its own inline triage (Gmail × 3 accounts, Slack, Linear, GHE PRs, calendars) as part of the full prep flow. =triage-intake.org= is now a source-agnostic engine that loads =triage-intake.<source>.org= plugins (refactored 2026-05-26), so daily-prep could call the engine and consume its synthesis instead of duplicating the source-scan logic. That DRYs up a large workflow and keeps both flows in sync when sources change — a source change now lives in one plugin that both flows pick up.
+Reproduced 2026-07-23. Confirms smoke's proposal. Additional finding from the same run: these two are =mechanical-fixed=, not judgment, so =lint-org --fix= silently rewrites the template's =**bold**= to org =*bold*=. The rewrite is correct org, but it lands on a synced template, so whichever project runs =--fix= first creates drift against canonical.
-Scope:
-- Identify the sections in =daily-prep.org= that do the inline triage (the email / Slack / Linear / PR / calendar fan-out, plus the "Sources checked: ..." footer at the top of each generated prep doc).
-- Replace those sections with "run the =triage-intake.org= engine" and adapt the downstream sections (Heads-up, Day's Priorities, Carry-forwards) to read the engine's synthesis output rather than the inline scan results.
-- Verify the generated prep doc still has the same shape (Heads-up + Day's Priorities + Carry-forwards + Sources checked).
-- Reconcile source coverage: daily-prep's inline triage scans work accounts (3 Gmail, Slack, Linear, GHE PRs) that are project-specific plugins under =.ai/project-workflows/=, not general plugins. The delegation must ensure the engine loads those project plugins (Phase 0 globs both dirs) so nothing daily-prep currently scans drops out.
+Grading: Minor severity (cosmetic noise; the mechanical rewrite is correct org and harmless in isolation) x every user, every time (every project, every sweep) = P2... same disagreement as the checker task above; graded [#C] on the no-real-harm read.
-Origin: came up while authoring =triage-intake.org= on 2026-05-11; body refreshed after the engine/plugin refactor on 2026-05-26.
-** DONE [#C] Templatize =make coverage-summary= into the language bundles (Elisp pilot) :feature:solo:
-CLOSED: [2026-05-31 Sun]
+Fix direction: rephrase both lines in the canonical template so no line starts with =**= (a list dash, or move the bold off column 0), and comma-escape the example block's own markers. One canonical fix clears it everywhere.
+** DONE [#B] cj-remove-block silently deletes content between two cj blocks :bug:solo:
+CLOSED: [2026-07-24 Fri]
:PROPERTIES:
-:LAST_REVIEWED: 2026-05-28
+:LAST_REVIEWED: 2026-07-24
:END:
+Fixed 2026-07-24 (sentry fire 2, 17f5d48). The range validation now looks inside the range and refuses when a second fence appears before the end. Verified it doesn't over-tighten: indented blocks, interior blank lines, legacy single-line annotations, and prose mentioning a fence mid-line all still validate. Second defect fixed in the same commit: remove_range now backs up to /tmp (matching lint-org.el) and writes atomically through a temp sibling, utf-8 pinned. 6 new tests.
+=looks_like_cj_range= validates only that the *first* line of the range opens a =#+begin_src cj:= fence and the *last* line closes with =#+end_src=. It never checks that the range holds exactly one block. A range spanning one block's opening fence to a *later* block's closing fence passes validation, and =remove_range= then deletes everything between — real prose, headings, whole tasks — silently, exit 0.
-Done 2026-05-31 (Elisp pilot, the scoped milestone): ported the kernel into the elisp bundle as a self-contained =languages/elisp/claude/scripts/coverage-summary.el= (no coverage-core dependency), proven end-to-end against the real dotemacs SimpleCov report (93 tracked, 27 untested modules surfaced, project number 66.4%). The missing-file-as-0% + unit-weighted number is the kernel. Delivery: the script ships under =.claude/scripts/= (gitignored, auto-fixed on drift by =sync-language-bundle.sh=); =languages/elisp/coverage-makefile.txt= holds the project-owned Makefile fragment, seeded at project root by =install-lang.sh= and dropped into =.ai/inbox/= by sync when that convention exists. Tests: 12 ERT (=languages/elisp/tests/=, wired into =make test=), 5 new sync bats, 2 new install-lang bats. The fan-out to Python/Go/TS is the follow-up below.
+That is precisely the failure the validation exists to prevent. Its docstring says it "protects against accidentally trimming the wrong block when line numbers drift between a cj-scan call and a remove call", and drift is the normal operating mode: =respond-to-cj-comments= edits the file as it processes each item, and a file being processed for cj comments generally holds several of them.
-Borrow dotemacs's =make coverage-summary= into the language bundles. After =make coverage= writes a coverage file, =coverage-summary= prints per-unit covered/total with percentages, a unit-weighted project number, and a list of source files present on disk but missing from the coverage report.
+Reproduced 2026-07-24 on a fixture with two cj blocks separated by real content. =looks_like_cj_range(lines, 2, 10)= returned =ok=True=, and the removal reduced a 10-line file to its first line. A heading and two content lines were destroyed with no warning and a zero exit.
-*The kernel — the only part worth building.* Weight the project number by file/module rather than by line, and count a source file absent from the report as 0% instead of omitting it. A module no test imports just doesn't appear in coverage.py or nyc output, so it silently fails to drag the number down. That missing-file detection is the value; everything else (per-file table, total) the built-in reporters already print, so don't reimplement those.
+Grading: Major severity (silent destruction of real content in the file that holds Craig's tasks and notes, with no warning and a success exit; git recovers only to the last commit, so intra-session work is lost) x some users, sometimes (needs drift plus multiple cj blocks, which together are the skill's normal operating mode) = P2 = [#B].
-*Scope Elisp-first.* Port the proven dotemacs version into the elisp bundle, prove the pattern end-to-end, then fan out. Don't open all four bundles at once.
+Fix direction: =looks_like_cj_range= must confirm the range contains exactly one block — scan lines start..end and reject when any =#+end_src= appears before the final line, or when any second =#+begin_src cj:= appears after the first. Test with the two-block fixture above asserting the validation refuses.
-*Delivery (settled 2026-05-25).* Two rulesets-owned pieces per language:
-- The summary *script* ships in the bundle under =.claude/= (inside the now-gitignored tooling footprint), copied in on install and auto-fixed on drift by =sync-language-bundle.sh=, never committed by the project.
-- One *text file per language* holding the Makefile fragment (the =coverage-summary= target plus its =coverage= prerequisite) and a block recommending how to set up coverage for that language. The bundle never edits the project's own Makefile.
- - *New project:* install copies that file in for the project to own.
- - *Existing project:* sync drops the fragment into the project's =inbox/= rather than touching its Makefile — the project adopts it deliberately.
+Second defect, same file, same fix session: =remove_range= writes the mutated org file with a bare =path.write_text=, which truncates the target on open, and it takes no backup first. A mid-write failure leaves Craig's =todo.org= truncated. =lint-org.el= — the other tool that mutates these files — copies a backup to =/tmp/<basename>.before-lint-pass.<timestamp>= before touching anything. cj-remove-block should match that: back up first, then write atomically via a temp file and =os.replace=, the same shape shipped for =inbox-send= in 0f91a8e. Pin =encoding="utf-8"= on the read and write while there.
+** DONE [#D] route_recommend downgrades strong to weak on a duplicate basename :bug:quick:solo:
+CLOSED: [2026-07-24 Fri]
+:PROPERTIES:
+:LAST_REVIEWED: 2026-07-24
+:END:
+Fixed 2026-07-24 (sentry fire 2, 1b0f284). The dedupe went into recommend rather than discover_destination_names as the task proposed, so every caller of the pure core is protected, not just the CLI path. Identical names collapse; genuine ambiguity between two different projects still downgrades to weak, pinned by a test. 3 new tests.
+=discover_destination_names= collapses discovered projects to bare basenames (=[p.name for p in ...]=). Two projects sharing a basename across roots (=~/code/notes= and =~/projects/notes=) therefore appear twice in the candidate list, both literal-match the same item, and =recommend= reads =len(strong) > 1= as an ambiguous tie — downgrading a correct strong match to weak.
-*Prerequisite caveat.* The summary presumes a coverage harness exists (undercover, coverage.py, nyc, =go cover=). Several bundles may have no =make coverage= yet, so for those this task implies adding the harness first — or the per-language file documents it as a prereq.
+Reproduced 2026-07-24 by direct probe: =recommend("fix the notes thing", ["notes", "other"])= returns =('notes', 'strong')=, while the same call with =["notes", "notes", "other"]= returns =('notes', 'weak')=.
-Per-language parser (the script is ~40 lines over each tool's output):
-- Elisp: undercover SimpleCov JSON (=.coverage/simplecov.json=) — dotemacs/auto-dim scripts already parse this.
-- Go: =go test -coverprofile=cover.out=; parse =cover.out= (simple text), or lean on =go tool cover -func=.
-- Python: =coverage json= per-file JSON, or lean on =coverage report=.
-- TypeScript/JS: nyc/Istanbul =coverage-final.json= / json-summary.
+Latent, not live: the current project set is 27 projects with 27 distinct basenames, so no collision exists today. The destination stays correct either way; only the confidence tier is wrong, which costs an unnecessary routing prompt rather than a misroute.
-Reference (dotemacs): =scripts/coverage-summary.el=, =modules/coverage-core.el=, and the =coverage= / =coverage-summary= Makefile targets.
+Grading: Minor severity (right destination, wrong tier, cost is one extra prompt) x rare edge case (needs a basename collision across roots, which doesn't currently exist) = P4 = [#D].
-Origin: handoff from the .emacs.d session, 2026-05-25.
-** DONE [#C] Fan out coverage-summary across all language bundles :feature:
-CLOSED: [2026-05-31 Sun]
+Fix direction: dedupe names in =discover_destination_names= (=list(dict.fromkeys(names))=, order-preserving). A duplicate basename is one addressable name as far as routing goes, since =inbox-send='s =find_target= resolves a name to its first match anyway. Add a test with a duplicated candidate asserting the tier stays strong.
+** DONE [#C] audit.bats has a flaky teardown that produces false suite reds :bug:test:solo:
+CLOSED: [2026-07-24 Fri]
:PROPERTIES:
-:CREATED: [2026-05-31 Sun]
+:LAST_REVIEWED: 2026-07-24
:END:
+Fixed 2026-07-24 (sentry fire 3, 7f45d4b). The fixture now sets =maintenance.auto false= and =gc.auto 0= before staging.
+
+Cause, traced not guessed: =git commit= spawns =git maintenance run --auto --quiet --detach= on git 2.55. The commit returns while that detached process is still writing a pack, and teardown's =rm -rf= races it. Every failed run left a =tmp_pack_*= behind, which is what pointed at it.
+
+*My original lead in this task was wrong* and worth recording as such. It blamed =gc.auto='s loose-object threshold; the object counts kill that outright (a fixture holds five objects against a default threshold of 6700). The right knob on modern git is =maintenance.auto=, and the mechanism is the =--detach=, not the threshold. Filing the theory as an explicitly-labelled lead rather than a finding is what kept it from being implemented as fact.
-Done 2026-05-31: coverage-summary now ships in all four bundles. Elisp pilot, then Python, Go, and TypeScript. Each parses its tool's report (SimpleCov / coverage.py JSON / Go cover.out / Istanbul json-summary), counts on-disk source files absent from the report as 0%, and file-weights the project number. The plumbing proved generic: =install-lang.sh= seeds the project-owned =coverage-makefile.txt= and ships the script into the gitignored =.claude/scripts/=; =make test= discovers ERT (=test-*.el=), pytest (=test_*.py=), =go test= (=*_test.go=), and =node --test= (=*.test.js=) under =languages/*/tests/=, each guarded on its toolchain. TypeScript and Go scripts were dogfooded (Go against a live profile, TS against the CLI); Python and TS weren't run against a live coverage tool (coverage.py / nyc not installed) — proven against faithful fixtures matching each tool's stable schema.
+Validation: 20 consecutive runs clean with no leftover fixture directories, against a baseline of roughly one failure in eight. Stated honestly, 20 clean runs alone would be about 7% likely by luck at that rate, so the statistics confirm rather than prove — the trace is the evidence. Disabled the background writer rather than retrying the delete, since a retry loop hides a live process instead of removing it.
+=scripts/tests/audit.bats= test 4 ("tracked project with dirty .ai/ is skipped") intermittently fails in its =teardown=, not its assertions: =rm -rf "$TEST_HOME"= exits non-zero with =rm: cannot remove '/tmp/audit-bats.XXXX/code/alpha/.git/objects': Directory not empty=. The test body passes; only the cleanup fails, and bats reports the whole test as failed.
-Remaining follow-ups (not blockers):
-- Go is a coverage-only slice — =languages/go/= has no rule file, so =sync-language-bundle.sh= can't fingerprint it and won't sync-maintain the script. Build out the real Go bundle (=go.md= / =go-testing.md= + =CLAUDE.md=) to close that.
-- First real adopters of the Python and TS scripts should sanity-check against a live =coverage json= / nyc =coverage-summary.json= run.
+Verified intermittent 2026-07-24: it failed twice during a sentry fire (once in a full =make test=, once running the file alone), then passed three consecutive runs immediately afterward with an identical tree. That intermittency is the defect — it makes =make test= return a false red.
-Original notes retained below for the next person.
+This matters more than a normal flaky test because the green suite is load-bearing for sentry: it gates entry, and the fire-end conditional run gates whether a night's commits are trusted. A spurious red there either blocks a fire or flags a clean night for morning review.
-The Elisp pilot proved the pattern; Python and Go followed. The plumbing is generic: =install-lang.sh= seeds the fragment, and =make test= now discovers ERT (=test-*.el=), pytest (=test_*.py=), and =go test= (=*_test.go=) under =languages/*/tests/=. TypeScript is the last one.
+Grading: Minor severity (a teardown-only failure, no production code implicated, and the assertions themselves pass) x some users, sometimes (fired twice in roughly six runs tonight) = P3 = [#C].
-- TypeScript/JS: nyc/Istanbul =coverage-final.json= / =coverage-summary.json=. Same kernel: file-weighted project number, on-disk =*.ts=/=*.js= absent from the report counted as 0%. nyc prints its own table, so the script focuses on the missing-file list and the number. Needs a vitest/jest (or =node --test=) discovery path in =make test=, mirroring the go-test block.
+Lead, not a verified cause: =audit.bats= line 42 runs =git init -q= in each fixture project and sets no =gc.auto=, while =audit.sh= runs five git commands against them. Git can fork background maintenance (=gc --auto=) that keeps writing into =.git/objects= after the foreground command returns, which would race the teardown's =rm -rf=. That fits the symptom exactly but is untested. Confirm before fixing.
-Notes for the next person, from the Python + Go runs:
-- Python: parses coverage.py's =files[path].summary.{covered_lines,num_statements}= (stable since coverage 5.x), resolves report paths against the report's parent dir. Proven against a synthetic report, not a live =coverage json= run (coverage.py wasn't installed). Sanity-check against a real one.
-- Go: =languages/go/= is a coverage-only slice with no rule file, so =sync-language-bundle.sh= can't fingerprint it (detection keys on a bundle's own =.claude/rules/*.md=). The script is delivered by =make install-lang LANG=go= but is not sync-maintained until the Go bundle gets a real rule file + =CLAUDE.md=. Building out that bundle is the natural companion task. Also: modern =go test ./...= already lists every module package in the profile at 0%, so the missing-file list is usually empty for in-module code; it earns its keep on build-tagged files and dirs outside =./...=.
-** DONE [#C] Enumerate implementation tasks in =spec-review.org= Phase 6 :feature:solo:
-CLOSED: [2026-05-31 Sun]
+Fix direction (once the cause is confirmed): set =git config gc.auto 0= (and =maintenance.auto false=) in the fixture setup so no background writer exists. Failing that, make teardown resilient rather than papering over it — a retry loop hides a real writer instead of removing it, so prefer killing the writer.
+** DONE [#C] todo-cleanup rewrites todo.org with no backup, unlike its siblings :bug:solo:
+CLOSED: [2026-07-24 Fri]
:PROPERTIES:
-:CREATED: [2026-05-28 Thu]
-:LAST_REVIEWED: 2026-05-28
+:LAST_REVIEWED: 2026-07-24
:END:
-Added a Phase 6 step that lifts the spec's =Implementation phases= into a drop-in =todo.org= block (one =[#B]= per phase + a test-surface entry mirroring =Acceptance criteria=); a spec lacking phase decomposition raises that as a finding instead. Added Exit Criterion 6 and a review-history entry. Pure workflow-doc change.
+Fixed 2026-07-24 (sentry fire 4, 0686784). Copies to =/tmp/<basename>.before-todo-cleanup.<stamp>= before the first mutation, matching lint-org.el's convention, once per invocation and skipped under =--check=. 2 ERT tests. The two non-findings recorded in this task's original body stand: the archive move was already fail-safe, and missing-file behavior is unchanged (exit 255, nothing created) — both re-verified against the pre-change version.
+=todo-cleanup.el= rewrites =todo.org= in place (hygiene fixes, =--convert-subtasks=, =--archive-done=, =--sync-child-priority=) and leaves no copy behind. The two sibling tools that mutate the same files both do: =lint-org.el= copies to =/tmp/<basename>.before-lint-pass.<stamp>= and =wrap-org-table.el= does the equivalent. =cj-remove-block= joined them in 17f5d48. todo-cleanup is the outlier, and it is the one that runs most often — wrap-up calls it, and every sentry fire's pass 4 calls it three times (four runs tonight alone).
-From pearl handoff 2026-05-28. =spec-review.org= Phase 6 currently says "log deferred work to =todo.org=: v1 implementation = [#B] ... vNext/someday = [#D]." That covers deferred and v1 in passing but doesn't lift the spec's =Implementation phases= section into a drop-in =todo.org= block.
+Verified 2026-07-24: after a real =--convert-subtasks= mutation on a fixture, the directory holds only =todo.org=. Emacs's own backup mechanism does not fire under =--batch -q=, so there is genuinely no undo short of git, which recovers only to the last commit and loses intra-session work.
-Proposed addition to Phase 6: a structured step that reads the spec's =Implementation phases= section and produces a =[#B] TODO= entry per phase (subject line, tags, one-line body, pointer back to spec), plus a final entry for the test surface (unit / integration / e2e / manual-verify mirroring the spec's =Acceptance criteria= when present). Emit under a new section "Implementation tasks (drop-in for todo.org)" in the review file. Format follows =todo-format.md= (terse heading, body holds context, tags on heading).
+Two things this is NOT, both checked so they don't get re-investigated:
-Three wins: handoff is one paste not a re-read; forces specs to be implementable in pieces (a spec without a phase decomposition fails this step, surfacing the shape problem); closes the loop on =Acceptance criteria= as manual-verify entries.
+- The archive move is *fail-safe*, not lossy. It deletes subtrees from the buffer, writes the archive file, and only saves =todo.org= at the very end, so an archive-write failure aborts before the save. Verified by making the archive directory unwritable: exit 255, =todo.org= byte-identical, content intact. My initial hypothesis that a mid-move failure could lose a subtree from both files was wrong.
+- No error swallowing on the mutation path. The only =ignore-errors= in the file wrap =call-process "git"=, not any write.
-If the spec lacks an =Implementation phases= section, the step is the prompt to ask the author to add one before =Ready=.
-** DONE [#C] Add =.aiignore= for agent inventory exclusions :chore:solo:
-CLOSED: [2026-05-31 Sun]
-:PROPERTIES:
-:CREATED: [2026-05-28 Thu]
-:LAST_REVIEWED: 2026-05-28
-:END:
-Shipped a gitignore-syntax =.aiignore= at the rulesets root (deps, build output, language caches, editor cruft, token artifacts, lockfiles-as-agent-read-skip) and documented the convention + defaults + lockfile policy in protocols.org ("Recursive Reads"). Per Craig's scope call (2026-05-31): did NOT wire audit.sh / diff-lang.sh / sync-language-bundle.sh — they do targeted finds over .ai/.claude/bundle dirs, never naive whole-tree walks, so honoring .aiignore there would be dead code. Script-side honoring belongs in a future catalog/inventory tool if one ships; the real consumer today is agent recursive reads (the protocols guidance).
+So this is a hardening gap rather than an active bug: a future defect in a mechanical rewriter would have no undo.
-From the codex enhancement backlog (item #8). Filesystem scans by agents and helper scripts pick up =node_modules=, =__pycache__=, =.pytest_cache=, lockfiles, generated OAuth artifacts, and test caches, even when those are gitignored. Token waste during exploration and skewed project summaries.
+Grading: Minor severity (no known active defect; the exposure is that any future one is unrecoverable within a session) x every user, every time (it runs on every wrap and every sentry fire) = P2 = [#C].
-Scope: add a shared =.aiignore= file (or =rulesets-ignore.json= if a more structured format helps) listing default exclusions. Teach the scripts that walk the project (=audit.sh=, =diff-lang.sh=, =sync-language-bundle.sh=, future =catalog= work if any) to honor it. Document in =protocols.org= so agents know to consult it before naive recursive reads.
-
-Keep the lockfile policy explicit: ignored when a local skill dependency cache, tracked when reproducibility matters.
-** DONE [#C] Workflow test harness — drift + integrity tests :feature:solo:
-CLOSED: [2026-05-31 Sun]
+Fix direction: back up before the first mutation, matching =lint-org.el='s convention exactly — =/tmp/<basename>.before-todo-cleanup.<YYYYMMDD-HHMMSS>=. One copy per invocation, not per pass, and skip it under =--check= (which writes nothing). Test by asserting the backup exists and holds the pre-edit content after a real mutation.
+** DONE [#C] claude-templates/bin/ gets no lint coverage at all :bug:quick:solo:
+CLOSED: [2026-07-24 Fri]
:PROPERTIES:
-:CREATED: [2026-05-28 Thu]
-:LAST_REVIEWED: 2026-05-28
+:LAST_REVIEWED: 2026-07-24
:END:
+Fixed 2026-07-24 (sentry fire 8, f91feef). lint.sh now sweeps =claude-templates/bin/*= through =check_hook=, matching the extensionless-file shape =languages/*/githooks/*= already uses. Added =scripts/tests/lint-coverage.bats=, which pins the *coverage* rather than current cleanliness: it plants a broken file in each swept location and asserts lint.sh complains, so a location that silently stops being swept fails the suite. The broader question this surfaced — whether rulesets should run shellcheck on its own shell at all — is the VERIFY below and stays Craig's call.
+=scripts/lint.sh= sweeps =scripts/*.sh=, =languages/*/claude/hooks/*.sh=, and =languages/*/githooks/*= through =check_hook= (shebang present, executable bit set). It never touches =claude-templates/bin/=. Verified: zero references to that path in the file.
-From the codex enhancement backlog (item #10). Startup's drift check catches index-vs-directory mismatches but not deeper integrity: a workflow that references a script that's been renamed, a plugin whose parent engine has been deleted, a required section missing from a newly-added workflow.
+Those four scripts — =ai=, =agent-text=, =agent-page=, =install-ai= — are the ones =make install= symlinks into =~/.local/bin=, so they run on Craig's PATH on every machine. They are the *most* exposed shell in the repo and the only shell with no gate over it.
-Scope: add =scripts/tests/workflow-integrity.bats= (or pytest equivalent) verifying:
+All four are clean today (shebangs present, mode 755, shellcheck-clean when run by hand), so nothing is broken. This is a missing gate, not an active defect: the =ai= launcher was hardened to 42 tests recently and that cleanliness is not enforced going forward.
-- Every =.org= file in =.ai/workflows/= is either indexed in =INDEX.org= or classifiable as a source plugin under an indexed engine.
-- Every indexed workflow file actually exists.
-- Every =file:= or shell-command reference inside a workflow to a script under =.ai/scripts/= or =scripts/= resolves to an existing file.
-- Every source plugin maps to a parent workflow that exists and is indexed.
-- Required sections (Overview, When to Use, the workflow's main phases) are present in each workflow.
-- Workflow trigger phrases are unique enough to route — no two workflows claim the same exact trigger.
+Grading: Minor severity (nothing broken now; the exposure is a future regression in a PATH-installed script going uncaught) x every user, every time (every =make lint= silently skips them) = P2 = [#C].
-Wire into =make test=. Run on the canonical =claude-templates/.ai/workflows/= as the source of truth.
-** DONE [#C] Token-tier pilot on largest workflows :feature:solo:
-CLOSED: [2026-05-31 Sun]
+Fix: add =claude-templates/bin/*= to the =check_hook= loop, the same shape =languages/*/githooks/*= already uses for extensionless files. A no-op today by design — it passes immediately — which is exactly what a guard should do.
+** DONE [#A] Applied: the secret-scan pre-commit fails open in ALL FIVE bundles (from .emacs.d)
:PROPERTIES:
-:CREATED: [2026-05-28 Thu]
-:LAST_REVIEWED: 2026-05-28
+:LAST_REVIEWED: 2026-07-24
:END:
+CLOSED: [2026-07-24 Fri]
+Applied 2026-07-24 across all five bundles (11 sites: the secret-scan input in each, plus each bundle's staged-file list, typescript having two). Verified on all five: refuses when the diff can't be read, still blocks a real staged secret, still passes a clean commit. Adopted .emacs.d's elisp bats suite (8 tests) and extended the repo-level cross-bundle suite with two fail-closed assertions.
-Done 2026-05-31: restructured both =startup.org= and =triage-intake.org= into the four-lane structure (Summary / Execution / Reference / History), preserving every existing instruction. triage-intake's reorder ran through a content-preservation guard (the multiset of content lines is unchanged; only heading depth and lane grouping moved). workflow-integrity, sync-check, and the full test suite pass.
+Separate defect found while wiring that up: the cross-bundle suite's VARIANTS list read "elisp bash go" while python and typescript also shipped hooks, so every "in every variant" assertion had silently skipped two bundles since I added them. VARIANTS is now discovered from the tree rather than enumerated — the same failure class the tests exist to catch.
-From the codex enhancement backlog (item #5), scope-limited to a pilot rather than a universal template change.
+What arrived: .emacs.d found that =languages/elisp/githooks/pre-commit= builds its scan input as =added_lines="$(git diff --cached -U0 ... | grep '^+' | grep -v '^+++' || true)"=. With no pipefail and =|| true= swallowing everything, any git failure yields an empty string, so the scan searches nothing, finds nothing, reports clean, and the commit proceeds with the secret in it. The staged-file list feeding the paren check has the identical hole.
-Apply a standardized section structure to the largest workflow files first — =startup.org= and =triage-intake.org= are the prime candidates. Sections:
+*Verified independently, and it is worse than reported.* I reproduced the fail-open with a stub git that fails only the staged-diff call: exit 0 with an AWS-shaped key staged. Then I checked the other bundles, which the sender did not: *bash, go, python, and typescript all carry the same pattern and all fail open the same way.* Confirmed live on all five. Two of those (python, typescript) are hooks I wrote on 2026-07-24 by copying the bash one, so I propagated the defect while closing a different gap.
-- *Summary* / *Quick Contract* — one-screen purpose and outputs.
-- *Execution* — the steps an agent must follow.
-- *Reference* — examples, edge cases, rationale, old decisions.
-- *History* / *Design Notes* — durable context not needed every run.
+Graded [#A] on severity alone, per the todo-format security carve-out: a credential-scanning gate that reports clean without having looked is a showstopper regardless of how rarely git fails.
-Decision (Craig, 2026-05-31): *approved the four-lane structure (Summary/Execution/Reference/History) and the scope — restructure both =startup.org= and =triage-intake.org= now.* Makes the task solo: apply the lanes to both, preserving every existing instruction (reorganize, don't rewrite), verify the workflows still read coherently and the drift/integrity checks pass.
+The sender's fix is correct and I verified all three axes on it: refuses to proceed when the diff cannot be read (exit 1), still blocks a real staged secret (exit 1), still passes a clean commit (exit 0). It splits the git read from the greps so a git failure aborts while "grep matched nothing" stays the ordinary case.
-Teach startup/routing to read =Summary= only at routing time, then =Execution= only for the selected workflow. Other sections become opt-in.
+Decision needed: the fix as sent covers elisp only. It should be applied to all five bundles, which is my scope expansion rather than the sender's proposal — hence a VERIFY rather than a silent apply. Also unresolved: where the two attached bats suites live, since the hooks are rulesets-owned and the tests currently sit in the consuming project.
-After the pilot, evaluate: did the savings show up in real session token use? Did the structure constrain the workflow expressiveness too much? If yes to savings and no to constraint, expand to the next-largest workflows. If not, document why and stop. Don't templatize universally — shorter workflows don't need tiering.
-** DONE [#B] Add Signal MCP server (rymurr/signal-mcp) :feature:
-CLOSED: [2026-06-02 Tue]
+Prepared: [[file:working/hook-fail-open/pre-commit.diff]], plus =test-pre-commit-hook.bats= (8 tests) in the same dir.
+** DONE [#B] Applied: remove the validate-el auto-test cap (from .emacs.d, Craig's call)
:PROPERTIES:
-:CREATED: [2026-05-29 Fri]
-:LAST_REVIEWED: 2026-05-29
+:LAST_REVIEWED: 2026-07-24
:END:
-Done 2026-06-02. Registered signal-cli to the Google Voice pager account, added the signal-mcp entry to servers.json, installed via make install-mcp (claude mcp list shows it connected), and documented the signal-cli + GV dependency in mcp/README.org. The GV-registration dependency this task flagged is resolved. Shipped in cfaff12 (page-signal routing) and this commit (README).
-
-Install [[https://github.com/rymurr/signal-mcp][rymurr/signal-mcp]] so Claude can call =send_message_to_user=, =send_message_to_group=, and =receive_message= natively rather than shelling out to the =page-signal= wrapper. Python, MCP framework, depends on =signal-cli= being configured locally.
-
-Two-way capability is the differentiator over the CLI: =receive_message= lets the agent listen for replies on the phone, enabling page-as-confirm flows, "should I proceed?" loops over Signal, and structured Q&A across devices.
+CLOSED: [2026-07-24 Fri]
+Applied 2026-07-24. Cap and the unreachable notice helper removed; the gate is now count -ge 1. Adopted the rewritten bats suite (6 tests), with its hook path corrected from the installed .claude/hooks/ layout to the bundle's claude/hooks/ — the re-homing mismatch the sender flagged.
-*** Dependency
+What arrived: a superseding handoff. The first proposed a loud notice when the test count exceeds =MAX_AUTO_TEST_FILES=20= (above the cap the block was skipped, nothing printed, exit 0 — indistinguishable from a pass). Craig chose in the .emacs.d session to remove the cap instead.
-This depends on the Google Voice account being registered with =signal-cli= first. Sending from Craig's primary number to itself doesn't notify (Signal treats it as one account on linked devices). The MCP server takes =--user-id= at startup, one account per instance, so it has to point at the GV account, with the primary as the per-send recipient.
+The reasoning is the valuable part. Measured, a whole family runs in under two seconds (calendar-sync 63 files / 633 tests / 0.9s), so the cap bought nothing. Worse, it was *concealing* a real cross-test pollution bug: calendar-sync exits 1 when its 63 files run in one process, because one test marks a calendar as syncing and never resets it, and a sibling file's test then hits the stale guard. Invisible from both directions — =make test= runs each file in its own Emacs, and the hook skipped the family for being over the cap.
-If GV registration is still pending when this task runs, block here and surface that.
+Verified the superseding file drops the cap entirely (zero references, gate is now =count -ge 1=) and removes the now-unreachable notice helper.
-*** Implementation
+Since Craig already made this call, the remaining decision is only adoption scope: removing the cap means other consuming projects run every stem-matched test file per edit, and one may go red on first use by surfacing pollution that per-file runs hid. The sender flags that as intended.
-- =mcp/servers.json= — add =signal-mcp= entry under stdio transport (=command=, =args=, optional =env= for the user-id pointer).
-- =mcp/README.org= — document the signal-cli + GV-registration dependency and the user-id pattern.
-- =mcp/secrets.env.gpg= — only if the MCP server's user-id needs to be encrypted (probably not; the GV number isn't a secret beyond being personal).
-- Verify: =make install-mcp= followed by =make check-mcp= shows =signal-mcp ok=; smoke-test via a Claude tool call sending a message + waiting on =receive_message=.
-
-*** Why this matters
-
-=page-signal= is the fast path (a hook, a script, a make recipe can call it without an MCP round-trip). The MCP server is the smart path. When Claude wants to send and then *react to the reply*, the CLI can't do that — only the MCP server can. The two complement each other; this task adds the second half.
-** DONE [#C] task-review pass at end of task-audit :chore:solo:
-CLOSED: [2026-06-02 Tue]
-:PROPERTIES:
-:LAST_REVIEWED: 2026-06-02
-:END:
-Have the =task-audit= workflow chain a =task-review= pass as its final phase, so a freshly-audited list also gets the lighter staleness/honesty sweep without a second invocation. The legend already notes the division of labor — task-audit assigns and refreshes tags, task-review keeps them honest in passing — so running task-review at the tail of task-audit closes the loop in one pass. Edit =claude-templates/.ai/workflows/task-audit.org= (and the synced mirror) to add the final phase; check whether =open-tasks.org= already invokes task-review so the chaining stays consistent.
-** DONE [#C] lint-followups drift — reconcile-on-write + audit dead-link reaping :feature:solo:
-CLOSED: [2026-06-02 Tue]
+Prepared: [[file:working/hook-fail-open/validate-el.diff]], plus =test-validate-el-hook.bats= (6 tests, two of which stage a deliberately failing test so a quiet pass proves the run happened).
+** DONE [#C] lint-org invalid-block false-positives inside example/src blocks :bug:solo:
+CLOSED: [2026-07-24 Fri]
:PROPERTIES:
-:LAST_REVIEWED: 2026-06-02
+:LAST_REVIEWED: 2026-07-24
:END:
-From an .emacs.d handoff (2026-06-02): running task-audit against a large todo.org proved several =.ai/lint-followups.org= entries stale (four dead-link flags pointed at docs that now exist; three near-duplicate dated lint runs had piled up). Two fixes, scoped separately.
+Fixed 2026-07-24, test-first. =lo--matched-block-regions= scans lines for correctly paired blocks under org's real rule — once a block is open, only its own =#+end_TYPE= closes it — and =lo--handle-item= drops an =invalid-block= finding whose line falls in one, delimiters included (org-lint reports at the delimiters themselves). Line-scanning rather than asking org is the point: org's parser is what mis-reads these blocks. 4 ERT tests: the heading-in-example case, a src block holding a literal =#+end_example=, a genuinely unterminated block that must still report, and a file with one of each proving the suppression is per-block not per-file.
-1. =lint-org= workflow/script (the real fix): reconcile-on-write. Before appending a run, drop entries whose finding no longer reproduces (dead link now resolves, flagged block/timestamp now clean) and dedupe against the prior run instead of re-logging. Key entries by content/finding rather than line number, so they survive edits to the target file (line numbers go stale immediately).
-2. =task-audit.org= (small, narrow): in the Phase C link-hygiene step, when fixing/verifying a =file:= link, also reap any matching dead-link entry in the project's lint-followups file so the two artifacts don't drift. Scope explicitly to dead-link entries — do NOT pull general lint cleanup into the audit; that mixes two concerns and slows the audit.
-** DONE [#C] start-work Justify gate: explicit "reasons not to do this" item :feature:quick:solo:
-CLOSED: [2026-06-02 Tue]
-:PROPERTIES:
-:LAST_REVIEWED: 2026-06-02
-:END:
-From a work handoff (2026-06-02, surfaced running /start-work on a clean low-risk refactor). The Phase 2 Justify gate has "Downsides" and "Alternatives considered" but no forced devil's-advocate verdict on "should we even do this?" Add a "top reasons not to do this" item: surface the top three objections if any exist; when none rise to a real objection, state one line instead of manufacturing three (e.g. "Nothing material argues against this; no reason to defer or drop it"). Building the case against the work before committing is cheapest exactly at this gate, which is its purpose. Edit the start-work skill's Justify-gate phase.
-** DONE [#C] start-work Approach gate: spec-needed check :feature:quick:solo:
-CLOSED: [2026-06-02 Tue]
-:PROPERTIES:
-:LAST_REVIEWED: 2026-06-02
-:END:
-From Craig (2026-06-02). The Approach phase should consider whether the work needs a spec when one doesn't already exist. For a big task, this isn't a silent skip — the pre-confirmation summary must explicitly report why a spec isn't needed, so the decision is visible and challengeable at the gate rather than assumed. Small tasks can pass without comment. Edit the start-work skill's Approach-gate phase to add the spec-needed consideration and the big-task report-why-not requirement.
-** DONE [#B] Cross-project pattern catalog :spec:thinking:
-CLOSED: [2026-06-05 Fri]
-:PROPERTIES:
-:LAST_REVIEWED: 2026-06-02
-:END:
+Verified against home's fixture end to end: =/home/cjennings/projects/home/.ai/notes.org= produced exactly the two reported findings (lines 386, 398) under the pre-change script and zero under the new one, file untouched. Full suite green.
-From pearl handoffs [[file:docs/design/2026-05-27-pattern-catalog-pearl-notes.org][2026-05-27]] + [[file:docs/design/2026-05-28-pattern-catalog-no-empty-input.org][2026-05-28 follow-up]].
+Left alone deliberately: the =,**= comma-escape in =claude-templates/.ai/notes.org= line 53. The task noted the fix "lets that escape be reverted," but the escape is the documented org convention for a literal =**= inside a verbatim block, so reverting it would trade correct org for no gain now that the finding is suppressed either way.
-Meta-question: how do good patterns travel from project A to project B? Pearl shipped three worked examples worth capturing — one-prompt picker with typed prefix (pearl-pick-source), magit-transient state buttons, and "no empty input as meaningful" (none-sentinel as first candidate). Each is a small principle with wide surface area; without a catalog, every project re-derives them from scratch.
+home reported it, and I verified it: an =#+begin_example= block whose body contains a line beginning =** = (or any heading-shaped line) makes =invalid-block= flag *both* delimiters as "Possible incomplete block", even though the block is correctly paired. The checker reads the heading line inside the verbatim body as a structural break and loses the open block.
-Open design questions before any implementation:
-- Catalog format — structured (one pattern per file with frontmatter) vs free-form doc
-- Surfacing mechanism — agent-driven (model spots opportunity) vs human-driven (Craig grep-searches)
-- Anti-patterns included or only what worked
-- Intake cadence — every time one lands, or batch review
-- Home — rulesets repo (agent visibility) vs Linear doc vs per-project cross-links
+Reproduced 2026-07-24 on a three-line example block: 2 judgment findings, both false. This is a docs-file-common shape — any org file documenting org syntax inside an example block hits it (home's notes.org PENDING DECISIONS section does).
-Pearl recommends a one-page spec (problem + design + open questions + acceptance) before implementation. Pearl available to come back for spec-review iterations.
+*This is the root cause behind a workaround I already shipped.* During fire 1 last night I comma-escaped exactly this =** Feature Name= line in =claude-templates/.ai/notes.org= to silence the two findings. That fixed the symptom in one file; the checker bug it worked around is still live and recurs everywhere. Fixing the checker lets that escape be reverted.
-*** 2026-05-28 Thu @ 08:12:55 -0500 Pearl shipped patterns 4-6, filed alongside the prior two
-Three more pearl handoffs landed and were filed during this audit. Filed: [[file:docs/design/2026-05-28-pattern-catalog-prompt-labels-and-defaults.org][prompt-labels-and-defaults]] (patterns 4-5: label-matches-behavior, default-most-common with friction-proportional-to-consequence) and [[file:docs/design/2026-05-28-pattern-catalog-prompt-collapse.org][prompt-collapse]] (pattern 6: collapse N orthogonal prompts into one enriched prompt). The catalog's evidence base is now four pearl notes in =docs/design/= covering six patterns plus the synthesizing principle Pearl articulated — "choices on screen, accurately labeled, ordered by what the user most often wants, friction sized to the cost of being wrong."
+Grading: Minor severity (judgment output, nothing mutates; pure noise that trains the reader to skim) x most users, frequently (every org file documenting org syntax in a verbatim block) = P3 = [#C].
-*** 2026-06-05 Fri @ 00:47:59 -0500 Spec approved as written — all 5 decisions + 3 open questions accepted
-Craig approved the spec ([[file:docs/design/2026-06-02-pattern-catalog-spec.org][2026-06-02-pattern-catalog-spec.org]]) as written. Confirmed: one file per pattern with frontmatter; home =patterns/= in rulesets; thin =claude-rules/patterns.md= pointer, agent-driven; anti-patterns as a per-pattern field; capture-on-landing/promote-on-review intake. Open questions resolved to the spec's leans: directory name =patterns/=; concrete-now, generalize-on-second-use; manual promote flow first, no =/pattern= skill yet. Built as =.org= files with =#+KEYWORD= frontmatter (Craig's call over the initial =.md= draft); the =claude-rules/patterns.md= pointer stays =.md= since the rules layer and the Makefile glob require it.
+Fix direction (per home): while inside a =begin_example= / =begin_src= block, skip structural parsing of the body until the matching =#+end_= line — verbatim blocks contain no headings, timestamps, or delimiters by definition. The same class hits a src block containing =#+end_example= as literal text. Tagged :solo: because the fix and its test are mechanical.
-*** 2026-06-05 Fri @ 00:47:59 -0500 Built the catalog — 6 seed patterns + pointer + README
-Created =patterns/= with the six seed patterns (one-prompt-picker-typed-prefix, transient-state-buttons, no-empty-input-as-meaningful, label-matches-behavior, default-most-common-friction-proportional, collapse-orthogonal-prompts), each carrying the frontmatter contract (name/principle/problem/tags/source/examples) plus Problem/Do/Anti-pattern/Applicability/Related sections. =patterns/README.org= states the root principle, the frontmatter contract, and the intake cadence. =claude-rules/patterns.md= is the agent-facing pointer, auto-installed via the Makefile RULES glob. Sourced from the four pearl notes in =docs/design/=.
-** CANCELLED [#C] Try Skill Seekers on a real DeepSat docs-briefing need :chore:
-CLOSED: [2026-06-10 Wed]
+*** 2026-07-24 Fri @ 20:10:00 -0500 Open question settled — invalid-block is org-lint's, so the fix is a filter
+Home answered it and I re-verified both halves here: =grep invalid-block= over =lint-org.el= returns nothing, and =org-lint--checkers= enumerates =invalid-block= in batch Emacs alongside =link-to-local-file=, =invalid-babel-call-block=, and =missing-language-in-src-block=. So this is the same shape as the =link-to-local-file= episode: suppress an =invalid-block= finding whose line falls inside a verbatim block, on our side of org-lint's output. No local checker to edit.
+
+Regression fixture, ready to use: home's own =.ai/notes.org= PENDING DECISIONS example block (lines 386-398) holds an unescaped =** Feature Name or Topic= and trips both delimiters — findings at lines 386 and 398. Home is deliberately leaving its copy unescaped so it stays a fixture, and asked to be pinged when the filter lands so it can re-run. The filter should take those two findings to zero without touching the file.
+** DONE [#C] sentry.org calls one loop cycle a "fire" :chore:solo:
+CLOSED: [2026-07-28 Tue]
:PROPERTIES:
-:LAST_REVIEWED: 2026-05-28
+:LAST_REVIEWED: 2026-07-28
:END:
+Done 2026-07-28. 72 noun-sense instances in =sentry.org= became "cycle"; the four verb-sense uses stayed. Three more lived outside the file — =wrap-it-up.org= ("a crashed fire"), =todo-cleanup.el= and its test ("every sentry fire") — all naming a sentry cycle, so they moved too. home's proposal was scoped to =sentry.org= alone, so the leak would have split the vocabulary across files.
-=Skill Seekers= ([[https://github.com/yusufkaraaslan/Skill_Seekers]]) is a Python
-CLI + MCP server that ingests 18 source types (docs sites, PDFs, GitHub
-repos, YouTube videos, Confluence, Notion, OpenAPI specs, etc.) and
-exports to 20+ AI targets including Claude skills. MIT licensed, 12.9k
-stars, active as of 2026-04-12.
+Craig read home's "nine fires" as nine emergencies and went looking for what was burning (2026-07-28): "I assume you mean nine crises, not nine loop cycles and I begin to get scared." The term reaches him directly — digest headings render as =** Fire 11 — 08:32 CDT= in the anchor he reads every morning. 35 instances in =sentry.org=.
-*Evaluated: 2026-04-19 — not adopted for rulesets.* Generates
-*reference-style* skills (encyclopedic dumps of scraped source material),
-not *operational* skills (opinionated how-we-do-things content). Doesn't
-fit the rulesets curation pattern.
+Grading: Minor severity (a user-facing artifact that miscommunicates, no data loss) x most users frequently (every digest, on every project running sentry) = P3 = [#C].
-*Next-trigger experiment (this TODO):* the next time a DeepSat task needs
-Claude briefed deeply on a specific library, API, or docs site — try:
-#+begin_src bash
-pip install skill-seekers
-skill-seekers create <url> --target claude
-#+end_src
-Measure output quality vs hand-curated briefing. If usable, consider
-installing as a persistent tool. If output is bloated / under-structured,
-discard and stick with hand briefing.
+*Take the problem, not home's proposed term.* home proposed "pass", reasoning that it already lives in the file's vocabulary. That is exactly what disqualifies it. =sentry.org= already uses "pass" as a precise numbered noun: "the pass list", "the Pass Runner", "eleven finding/hygiene passes", "pass 12" for the implementation pass. There are exactly eleven hygiene passes, so home's proposed digest heading =** Pass 11= collides with an existing real referent. Renaming would trade a term Craig misreads as urgent for one that is genuinely ambiguous.
-*Candidate first experiments (pick one from an actual need, don't invent):*
-- A Django ORM reference skill scoped to the version DeepSat pins
-- An OpenAPI-to-skill conversion for a partner-vendor API
-- A React hooks reference skill for the frontend team's current patterns
-- A specific AWS service's docs (e.g. GovCloud-flavored)
+Counter-proposal: *cycle*. It appears zero times in =sentry.org=, so there is nothing to collide with. It is also Craig's own word from the very quote that surfaced this ("not nine loop cycles"). =** Cycle 11 — 08:32 CDT= reads cleanly. Checked and rejected: "sweep" (already used for hygiene and property sweeps) and "run" (already used as a noun, "first live run").
-*Patterns worth borrowing into rulesets even without adopting the tool:*
-- Enhancement-via-agent pipeline (scrape raw → LLM pass → structured
- SKILL.md). Applicable if we ever build internal-docs-to-skill tooling.
-- Multi-target export abstraction (one knowledge extraction → many output
- formats). Clean design for any future multi-AI-tool workflow.
+Keep the verb sense of fire throughout ("the notify fires", "the path never fires") — only the noun meaning one loop cycle changes. Past session anchors are historical records and stay as written.
-*Concerns to verify on actual use:*
-- =LICENSE= has an unfilled =[Your Name/Username]= placeholder (MIT is
- unambiguous, but sloppy for a 12k-star project)
-- Default branch is =development=, not =main= — pin with care
-- Heavy commercialization signals (website at skillseekersweb.com,
- Trendshift promo, branded badges) — license might shift later; watch
-- Companion =skill-seekers-configs= community repo has only 8 stars
- despite main's 12.9k — ecosystem thinner than headline adoption
-** DONE [#C] Promote meeting-prep to a template workflow :feature:solo:
-CLOSED: [2026-06-10 Wed]
-:PROPERTIES:
-:LAST_REVIEWED: 2026-06-10
-:END:
-meeting-prep lives in the work project's =project-workflows/= and is general-purpose — it builds a per-meeting prep doc — but its body carries project-specific references: =deepsat/assets/= transcript paths, Linear as the tracker, =knowledge.org=. Promoting to =claude-templates= means generalizing those to project-neutral terms (the project's transcript home, the project's tracker), adding it plus its =meeting-prep.pre-wire.org= supporting doc to the =.ai/= mirror and INDEX.org, and a workflow-integrity pass. Once promoted, the daily-prep 5-Day Look-Ahead's conditional "where the project has one" reference can become a direct link.
+Craig approved "cycle" on 2026-07-28. That settles the only judgment the task carried, so it is =:solo:= now: the surface is the 35 noun-sense instances in =sentry.org=, and the completion check is objective — zero noun-sense "fire" left, verb sense untouched, lint clean, suite green, canonical and mirror in sync.
-Out of the 2026-06-10 daily-prep handoff from the work project.
-** DONE [#C] Build Craig's writing voice profile from real corpora :spec:
-CLOSED: [2026-06-10 Wed]
+Source: home handoff, 2026-07-28.
+** DONE [#B] references/ is linked from protocols.org but never synced :bug:
+CLOSED: [2026-07-28 Tue]
:PROPERTIES:
-:CREATED: [2026-05-29 Fri]
-:LAST_REVIEWED: 2026-05-29
+:LAST_REVIEWED: 2026-07-28
:END:
-Shipped across 2026-05-29 → 2026-06-10. =voice/references/voice-profile.org= is the canonical paired file: Phases 1-2 corpora measured (commit bodies 128k words + email/PR/review registers), all 45 patterns carry entries with basis and history, and every reconciliation delta landed in =voice/SKILL.md= (#13/#33 self-discipline reframing, #7 soft flag, new corpus-derived #43-#45). Extension corpora (Slack, long-form, syntactic fragment detection) deliberately not pursued.
-
-Build a grounded profile of Craig's actual writing voice by mining the corpora he's produced over time. The =voice/SKILL.md= patterns today are observation-derived (em-dash zero-tolerance, semicolon → period, contractions kept, sentence-fragment rewrite, felt-experience cut, etc.). Some are spot-on; others are intuition. A real corpus pass would tell us which patterns are genuinely Craig's voice and which were guesses, plus surface idioms, sentence structures, and vocabulary the current ruleset misses.
+Craig picked option 1 on 2026-07-28: drop the link, point at the calendar workflows instead. The four of them sync already and carry the MCP tool names, both account ids, the gcalcli fallback, and the conflict-check discipline — everything the reference was cited for.
-*** Sources to mine
+The adversarial review then returned =Needs Discussion= and widened the fix twice, both correctly:
-- *Email* — sent folders across all three accounts (=gmail=, =dmail/DeepSat=, =cmail/Proton=). Filter to Craig-authored (not forwards or replies-just-quoting). Separate work voice (=dmail=) from personal voice (=gmail=, =cmail=) since they're likely distinct registers.
-- *Commit messages* — =git log --author= across his repos. Captures terse-imperative voice.
-- *PR descriptions and review comments* — same corpora. More deliberate prose than commits.
-- *Org files he authored* — =notes.org=, todo bodies he typed, design docs in =docs/design/=, journal entries. Heavier on first-person voice than emails.
-- *Slack/messages* — DeepSat work slack, family group, friends. Casual register.
-- *Long-form artifacts* — résumé, proposals, white papers, blog posts (if any).
+- My first replacement said credentials "live in the rulesets repo" without naming a file. The only calendar-named document there was =calendar-reference.org=, whose three credential paths have all been dead since the OAuth keys moved into the encrypted MCP bundle in May. So the prose sent a reader to a stale file, which fails more quietly than the dead link it replaced. Now names =mcp/README.org=, the real authority, verified to document =gcp-oauth.keys.json= as gitignored and regenerated at install.
+- =calendar-reference.org= was left orphaned by the link removal: zero live inbound references, two copies, and =references/= is not in =sync-check.sh='s gate (=paths=(protocols.org workflows scripts)=), so the copies could drift silently. Deleted both, and the empty =references/= directories with them. Its operational content is fully covered by the four workflows and its credential paths were all stale, so nothing live was lost.
-Skip session-context files, which are Claude-co-written and would muddy the signal.
+The review also found the same defect class at seven other sites, filed separately.
-*** Output
+One fact died with the file, dropped deliberately rather than by accident: that the Google Cloud app runs in production mode, so tokens don't expire after seven days. It's checkable in the console, and it was the last live line in a document whose other credential facts had all gone stale.
-- =voice/references/voice-profile.org= (or =.md=) — the canonical reference doc:
- - Vocabulary tendencies (preferred verbs, avoided cliché classes, technical-vs-plain word choice).
- - Sentence structures (typical length, conjunction patterns, parenthetical use).
- - Punctuation patterns (em-dash actual frequency, semicolon vs period split, contraction rate).
- - Register markers (signs of formal vs casual mode, work vs personal).
- - Idioms and recurring phrasings.
- - "Anti-patterns" — phrasings Craig consistently avoids that show up in AI-generated prose.
-- Updated =voice/SKILL.md= patterns grounded in evidence rather than intuition. Patterns that the corpus confirms get strengthened; patterns the corpus contradicts get rewritten or removed.
+=protocols.org:273= links to =references/calendar-reference.org=, but startup.org's rsync copies only =protocols.org=, =workflows/=, and =scripts/=. So the link is dead in every consuming project. Confirmed: the file exists at =claude-templates/.ai/references/=, and neither home nor =.emacs.d= has a =.ai/references/= at all.
-Each finding should cite at least two evidence samples from the corpora so the basis for a rule is reviewable.
+Grading: Minor severity (a documented reference an agent can't follow, workaround is to search or ask) x every user every time (every consuming project, every sync) = P2 = [#B].
-*** Approach
+Pinned 2026-07-28 for Craig's decision. Analysis below is complete; only the choice is open.
-Phase 1 (corpus assembly) — pull the relevant slices: sent-mail dumps, =git log --author --no-merges --pretty=format:'%B'=, =gh pr list --author= bodies, org-file extracts. Strip headers, replies-quoted blocks, signatures. Land in =voice/corpus/= (gitignored if the project's =.ai/= is gitignored, tracked if private repo with private remote).
+*Revised recommendation: drop the link, don't add the sync.* My first read was add =references/= to the rsync. Looking at what the directory actually holds reversed it:
-Phase 2 (analysis) — pass over the corpus with focused queries: distribution of em-dashes per 1000 words, semicolon count, contraction frequency by register, sentence-length histogram, top-N adjectives/adverbs, etc. Subagent dispatch fits here.
+- =references/= holds exactly one file, and =protocols.org:273= is its only citation anywhere in the tree.
+- The four calendar workflows (=add-=, =edit-=, =delete-=, =read-calendar-event(s).org=) already sync to every project, and already carry the operational detail: the MCP server name, both account IDs, the gcalcli fallback, conflict-checking. None of them cites =calendar-reference.org=; they are self-sufficient.
+- What the reference uniquely adds is credential *file locations* (an OAuth keys path, a GPG-encrypted gcalcli secret) — no secret values. Re-auth is a rare operation Craig performs himself, not something an agent in another project needs a pointer to.
-Phase 3 (draft profile) — write =voice-profile.org= with findings + evidence. Surface contradictions with the current ruleset.
+So adding a synced directory carrying =--delete= semantics, to deliver one mostly-redundant file, is a poor trade. It also cuts against the rightsizing work: =protocols.org= is read into context every session, and a dead link is noise in a file being slimmed.
-Phase 4 (reconcile with voice/SKILL.md) — present the deltas to Craig. Each delta is one of: confirm existing rule with evidence, strengthen rule, weaken rule, add new pattern, remove unsupported pattern. Apply approved deltas.
+Preferred fix: replace the link with a one-line pointer to the calendar workflows, which travel and are current. The reference file stays rulesets-only for the credential locations.
-*** Privacy
+Alternatives if Craig prefers: add =references/= to the rsync (the =--delete= hazard is not present today — work is the only project with a =.ai/references/= and its copy is byte-identical), or fold the credential locations into the calendar workflows and then drop the link (most complete, but spreads local absolute paths across four more synced files).
-Email and Slack content is private. The corpus must NOT enter any commit unless rulesets stays on the private cjennings.net remote (which it does today). If a future move to a public remote is on the table, the corpus and any direct quotes have to go before that happens. The profile doc itself can stay (it's analysis, not raw content), but cite by pattern not by verbatim quote.
+Not =:solo:= — it changes a synced template, which needs Craig's approval by the inbox rule, and the sync-vs-drop choice is his.
-*** Why this matters
-
-The voice skill earns its place when Craig sees the rewrite and recognizes it as his own voice rather than a "clean" AI voice that approximates him. Today the skill catches common AI tells (em-dashes, semicolons, the felt-experience tic), which is useful. Corpus-grounding would make it catch the absence of *Craig-specific positive traits* — the phrasings he actually reaches for — not just the AI traits he doesn't.
-
-Likely improves =/voice personal= output quality on PR bodies, commit messages, and email drafts. Compound interest over the long run.
-** DONE [#C] Wide org-table handling — helper/lint/standard :spec:
-CLOSED: [2026-06-11 Thu]
-:PROPERTIES:
-:LAST_REVIEWED: 2026-06-11
-:END:
-The org-table standard keeps project-doc tables <=120 cols with multi-line wrapped cells and a rule between rows, but nothing enforces it and hand-wrapping a wide cell into multi-row form is tedious and error-prone. Decide among: (a) a helper that auto-wraps a wide table into multi-row cells at a target width, (b) a lint check that flags tables over the width budget, (c) tighten the written standard with a worked before/after example. Likely some combination. A worked before/after example exists in a work-project prep doc (a 6-col table reformatted by hand to a 4-col multi-row-cell version), to be reproduced generically when this lands.
-
-Out of a work-project handoff 2026-06-09.
-
-Resolution 2026-06-11: all three shipped. (c) The standard, generalized from the work project's notes.org local copy, is now claude-rules/org-tables.md (globally loaded; render-width semantics — links measure at their visible label, never split a link) with the worked wrapped-table example. (a) .ai/scripts/wrap-org-table.el reflows tables mechanically: render-width measurement, link-atomic tokenizing, column shrink-to-floor allocation, continuation rows, rules between logical rows; idempotent (rule-delimited continuation groups merge back before re-wrapping); 23 ERT tests. (b) lint-org.el gained an org-table-standard judgment check (width overruns, missing rules; conformant wrapped tables not false-flagged); 5 new ERT tests, 32 total. Verified end-to-end on a demo file: 150-col table reflowed to budget, idempotent second pass, lint clean on the result.
-** DONE [#C] SessionStart-on-clear hook for auto-resume :feature:
-CLOSED: [2026-06-11 Thu]
-:PROPERTIES:
-:LAST_REVIEWED: 2026-06-11
-:END:
-Add a SessionStart hook (matcher: clear) in settings.json that auto-injects "read .ai/session-context.org and resume if present, else run startup.org". Today /flush prompts the user to /clear and the next session relies on the model re-reading session-context; the hook makes resume automatic on /clear. Keep full startup.org for genuine fresh starts (new day, other machine, been away). Likely lands as claude-templates workflow notes plus the hook in settings.json.
-
-The checkpoint+resume halves already shipped as /flush. This is the remaining automation piece. Out of a work-project handoff 2026-06-09 (process tooling, belongs in rulesets not the work project).
-
-Resolution 2026-06-11: the hook itself had already shipped 2026-06-02 (hooks/session-clear-resume.sh + the SessionStart clear entry in the tracked settings.json — this task duplicated it). What was actually broken: make install didn't cover hooks, so the symlink never reached machines that hadn't run make install-hooks by hand, and the hook errored silently on every /clear. Fixed by folding default-hook linking into make install (startup's Phase A.0 now propagates hooks machine-wide), with bats coverage in scripts/tests/install-hooks-link.bats. Both hook branches verified on ratio; the live /clear fire is a one-keystroke manual test.
-*** TODO Manual testing and validation :test:
-**** /clear mid-session resumes from the anchor
-What we're verifying: the SessionStart(clear) hook fires and the fresh context resumes instead of cold-starting.
-- In any project session with a live .ai/session-context.org (this rulesets session qualifies), type /clear
-- Send any short message (the injected context loads but the model waits for your next keystroke)
-Expected: the reply starts with "flushed." on its own line, restates the Active Goal and immediate Next Step, and does NOT run the startup workflow.
-** 2026-06-12 Fri @ 02:56:58 -0500 New personal projects are home regroupings — no mechanism needed
-Craig's call (2026-06-12): new personal projects will live in home, and there's no project-creation mechanism to build — he'll be working in home and simply decide to group some things differently. Nothing to do.
-
-Concurrence, verified: no template doc directs new personal work into ~/projects (first-session.org, install-ai.sh, and the README carry no such guidance; the only ~/projects references are discovery-root scans, which home and work still need). The situation as it stands: a new personal "project" is an area dir plus tasks inside home's existing =.ai/= machinery, no bootstrap step; =first-session.org= remains the bootstrap for standalone code projects in ~/code, unchanged and correct; "launch finances"-style trigger phrases for folded names degrade politely to the no-match candidate list, worth work only if real friction shows up.
-
-** DONE [#C] Build =/update-skills= skill for keeping forks in sync with upstream :feature:
-CLOSED: [2026-06-11 Thu]
-:PROPERTIES:
-:LAST_REVIEWED: 2026-06-10
-:END:
-
-The rulesets repo has a growing set of forks (=arch-decide= from
-wshobson/agents, =playwright-js= from lackeyjb/playwright-skill, =playwright-py=
-from anthropics/skills/webapp-testing). Over time, upstream releases fixes,
-new templates, or scope expansions that we'd want to pull in without losing
-our local modifications. A skill should handle this deliberately rather than
-by manual re-cloning.
-
-Shipped 2026-06-11: [[file:.claude/commands/update-skills.md][/update-skills command]] + [[file:scripts/update-skills.py][helper script]] (17 bats tests) + three bootstrapped manifests under [[file:upstreams/][upstreams/]]. The first real upstream drift will exercise the interactive per-file/per-hunk flow end to end; the merge mechanics are covered by the test suite.
-
-*** 2026-06-11 Thu @ 17:05:28 -0500 Specification written as the shipped artifacts
-The command doc ([[file:.claude/commands/update-skills.md][update-skills.md]]) carries the user-facing spec: discovery, classification statuses, the per-file confirmation and per-hunk conflict flow, mark-synced semantics, and the missing-baseline fallback. The script's module docstring specifies the manifest schema. Two deviations from the 2026-05-16 design, with reasons: manifests live centrally at =upstreams/<name>/= instead of per-skill =.skill-upstream= dotfile dirs (arch-decide became two flat files in =commands/= and can't carry one — a =files= rename map covers it); baselines were seeded from the 2026-06-11 upstream HEADs since the true fork-point commits are unrecoverable, so pre-existing local modifications classify as =local-only= going forward.
-
-*** 2026-05-16 Sat @ 01:14:20 -0500 original goals and decisions
-**** Design decisions (agreed)
-
-- *Upstream tracking:* per-fork manifest =.skill-upstream= (YAML or JSON):
- - =url= (GitHub URL)
- - =ref= (branch or tag)
- - =subpath= (path inside the upstream repo when it's a monorepo)
- - =last_synced_commit= (updated on successful sync)
-- *Local modifications:* 3-way merge. Requires a pristine baseline snapshot of
- the upstream-at-time-of-fork. Store under =.skill-upstream/baseline/= or
- similar; committed to the rulesets repo so the merge base is reproducible.
-- *Apply changes:* skill edits files directly with per-file confirmation.
-- *Conflict policy:* per-hunk prompt inside the skill. When a 3-way merge
- produces a conflict, the skill walks each conflicting hunk and asks Craig:
- keep-local / take-upstream / both / skip. Editor-independent; works on
- machines where Emacs isn't available. Fallback when baseline is missing
- or corrupt (can't run 3-way merge): write =.local=, =.upstream=,
- =.baseline= files side-by-side and surface as manual review.
-
-**** V1 Scope
-
-- [ ] Skill at =~/code/rulesets/update-skills/=
-- [ ] Discovery: scan sibling skill dirs for =.skill-upstream= manifests
-- [ ] Helper script (bash or python) to:
- - Clone each upstream at =ref= shallowly into =/tmp/=
- - Compare current skill state vs latest upstream vs stored baseline
- - Classify each file: =unchanged= / =upstream-only= / =local-only= / =both-changed=
- - For =both-changed=: run =git merge-file --stdout <local> <baseline> <upstream>=;
- if clean, write result directly; if conflicts, parse the conflict-marker
- output and feed each hunk into the per-hunk prompt loop
-- [ ] Per-hunk prompt loop:
- - Show base / local / upstream side-by-side for each conflicting hunk
- - Ask: keep-local / take-upstream / both (concatenate) / skip (leave marker)
- - Assemble resolved hunks into the final file content
-- [ ] Per-fork summary output with file-level classification table
-- [ ] Per-file confirmation flow (yes / no / show-diff) BEFORE per-hunk loop
-- [ ] On successful sync: update =last_synced_commit= in the manifest
-- [ ] =--dry-run= to preview without writing
-
-**** V2+ (deferred)
-
-- [ ] Track upstream *releases* (tags) not just branches, so skill can propose
- "upgrade from v1.2 to v1.3" with release notes pulled in
-- [ ] Generate patch files as an alternative apply method (for users who prefer
- =git apply= / =patch= over in-place edits)
-- [ ] Non-interactive mode (=--non-interactive= / CI): skip conflict resolution,
- emit side-by-side files for later manual review
-- [ ] Auto-run on a schedule via Claude Code background agent
-- [ ] Summary of aggregate upstream activity across all forks (which forks have
- upstream changes waiting, which don't)
-- [ ] Optional editor integration: on machines with Emacs, offer
- =M-x smerge-ediff= as an alternate path for users who prefer ediff over
- per-hunk prompts
-
-**** Initial forks to enumerate (for manifest bootstrap)
-
-- [ ] =arch-decide= → =wshobson/agents= :: =plugins/documentation-generation/skills/architecture-decision-records= :: MIT
-- [ ] =playwright-js= → =lackeyjb/playwright-skill= :: =skills/playwright-skill= :: MIT
-- [ ] =playwright-py= → =anthropics/skills= :: =skills/webapp-testing= :: Apache-2.0
-
-**** Open questions
-
-- [ ] What happens when upstream *renames* a file we fork? Skill would see
- "file gone from upstream, still present locally" — drop, keep, or prompt?
-- [ ] What happens when upstream splits into multiple forks (e.g., a plugin
- reshuffles its structure)? Probably out of scope for v1; manual migration.
-- [ ] Rate-limit / offline mode: if GitHub is unreachable, should skill fail
- or degrade gracefully? Likely degrade; print warning per fork.
-
-** DONE [#C] Monthly session-harvest workflow :feature:
-CLOSED: [2026-06-11 Thu]
+Source: winvm link-integrity pass, 2026-07-28.
+** DONE [#B] Parked: telegram source treats "down" as launch, not SCAN FAILED (from .emacs.d)
+CLOSED: [2026-07-28 Tue]
:PROPERTIES:
-:CREATED: [2026-06-11 Thu]
-:LAST_REVIEWED: 2026-06-11
+:LAST_REVIEWED: 2026-07-24
:END:
-A monthly pass over recent =.ai/sessions/= summaries across projects proposing promotion candidates: patterns for the catalog, durable facts for the KB, rule refinements, workflow learnings. Sibling cadence to the roam-hygiene timer; a workflow run on schedule, not a standing agent. From the 2026-06-11 insights report's "Canonical-Aware Knowledge & Workflow Curator" — the capture/promote machinery exists (pattern catalog, /codify, KB); this adds the mining cadence.
+Applied 2026-07-28, merged with the segfault root-cause fix that arrived the same morning rather than applied alone.
-Shipped 2026-06-11 as [[file:.ai/workflows/session-harvest.org][session-harvest.org]] (template + INDEX entry): five phases, four promotion lanes, /codify-grade gates + work-confidentiality scrub, =:LAST_HARVEST:= marker in notes.org, and the KB receipt-line metrics readout for the ~2026-07-10 checkpoint. Window filter reads session-filename date prefixes (mtime proved unreliable in a live test). First run due ~2026-07-11.
+Merging was necessary, not tidiness. The parked proposed file still carried the bad =(telega--loadChats 'main)= call at its own lines 52 and 122, so applying it as-is would have shipped a file that fixed the wording defect while preserving the call that kills the server. Its third hunk also added prose citing "tdlib segfaults in native mode (SEGFAULT gotcha below)" — pointing at the section the new handoff rewrites to say those deaths were our own bad argument, not tdlib memory corruption.
-** CANCELLED [#B] todo-cleanup.el per-area Open Work / Resolved pairs :feature:
-CLOSED: [2026-06-11 Thu]
-=--archive-done= assumes exactly one level-1 "Open Work" and one "Resolved" heading per todo.org. Home's consolidated file briefly carried per-area pairs and the pass skipped. Filed from home's 2026-06-11 addendum, then held the same evening when Craig flagged that he expected a single pair.
-
-Cancelled 2026-06-11: Craig confirmed the decision — one todo queue with a single Open Work / Resolved pair. Home reshapes its consolidated file to that form, and the existing single-pair tooling works unmodified. No code change needed.
-
-** CANCELLED [#D] todo-cleanup =--archive-done= reports 0 moves while moving subtrees :bug:
-CLOSED: [2026-06-12 Fri]
-:PROPERTIES:
-:CREATED: [2026-06-12 Fri]
-:END:
-Observed at the 2026-06-12 wrap: the pass relocated closed subtrees from Open Work to Resolved while printing "todo-cleanup --archive-done: 0 subtree(s) moved".
+What landed: all three parked hunks (the down-is-launch directive, the SCAN-FAILED-only-after-launch-attempted rewording, the =(setq telega-use-docker t)= restored to the Step 1 code block), plus both corrected =loadChats= call sites and the rewritten gotcha. The native-mode prose was reconciled in two places so it no longer leans on the refuted story: the Step 1 comment now states plainly that docker mode and the loadChats bug are separate concerns (the deaths happened *in* docker mode, so docker mode is neither a defense against it nor evidence for it), and the Quick Reference line says "crashed in native mode (2026-06-09)" instead of "segfaults", with the same disambiguation.
-CANCELLED 2026-06-12 — cannot reproduce. =todo-cleanup.el= is unchanged since the wrap that logged this, and =tc-archived= is incremented inline with each move and read straight in the report, so no move can go uncounted. Running the exact pre-archive state (=b6d286f:todo.org=) through the tool reports the right count (3 moved, all listed). The "0 moved" was a correct second-run report: =open-tasks.org= Phase A runs =--archive-done= after wrap-it-up already archived, so the second pass finds nothing to move and prints 0 next to the first pass's git diff. Not a code defect.
+Verified: both live call sites use the TL object; the two remaining ='main= occurrences are inside the gotcha prose describing the bug. lint-org clean, mirror synced, suite green.
diff --git a/voice/SKILL.md b/voice/SKILL.md
index 4690b97..cdf7874 100644
--- a/voice/SKILL.md
+++ b/voice/SKILL.md
@@ -1,7 +1,7 @@
---
name: voice
description: |
- Multi-pass prose editor with three modes. General mode (default) edits arbitrary writing — research notes, essays, anyone's README prose — by walking 31 patterns: Wikipedia's Signs of AI Writing patterns plus universal good-writing rules (long-word → short-word, active-over-passive, comma splices, cliché flag, jargon-fragment-in-prose rewrite, corporate-speak nominalizations). Prose mode adds Craig's writing-voice patterns for prose he authors or sends — emails, documents, notes — on top of general (em-dash zero-tolerance, no-emphasis-formatting, contractions, semicolons → periods, sentence-split, felt-experience cut, sentence-fragment rewrite, terse-cut, single-sentence cadence, parenthetical asides, declarative-register marker). Personal mode is for publish artifacts only (commits, PR titles + bodies, PR review comments) and adds the artifact-mechanics patterns on top of prose (first-person rewrite, public-artifact scope flag, praise/correction asymmetry, finding stems) plus per-artifact terseness budgets. Total 45 patterns; one editorial review covers all relevant ones for the chosen mode, with an attestation receipt for the high-recurrence set. Replaces the standalone humanizer skill. Use when editing prose. Do NOT use for code, structured data, or plain bullet lists where fragments are valid.
+ Multi-pass prose editor with three modes. General mode (default) edits arbitrary writing — research notes, essays, anyone's README prose — by walking 31 patterns: Wikipedia's Signs of AI Writing patterns plus universal good-writing rules (long-word → short-word, active-over-passive, comma splices, cliché flag, jargon-fragment-in-prose rewrite, corporate-speak nominalizations). Prose mode adds Craig's writing-voice patterns for prose he authors or sends — emails, documents, notes — on top of general (em-dash zero-tolerance, no-emphasis-formatting, contractions, semicolons → periods, sentence-split, felt-experience cut, sentence-fragment rewrite, terse-cut, single-sentence cadence, parenthetical asides, declarative-register marker). Personal mode is for publish artifacts only (commits, PR titles + bodies, PR review comments) and adds the artifact-mechanics patterns on top of prose (first-person rewrite, public-artifact scope flag, praise/correction asymmetry, finding stems, comma budget) plus per-artifact terseness budgets. Prose mode also carries one correspondence-only pattern (recipient-priority ordering) that personal mode skips. Total 47 patterns; one editorial review covers all relevant ones for the chosen mode, with an attestation receipt for the high-recurrence set. Replaces the standalone humanizer skill. Use when editing prose. Do NOT use for code, structured data, or plain bullet lists where fragments are valid.
allowed-tools:
- Read
- Write
@@ -31,8 +31,8 @@ This skill is split across two files by design.
Three modes determine which patterns to walk. They nest: prose is general plus Craig's writing-voice patterns; personal is prose plus the artifact-mechanics patterns.
- **General** (default) — apply patterns **#1-31**. Use for writing whose author isn't Craig and that isn't a publish artifact: research notes you're editing for someone else, a quoted passage, README prose for a shared project, any third-party text. Output is well-edited human-sounding prose, but does not impose Craig's voice (first-person, contractions, em-dash elimination) — those conflict with academic, literary, or formal registers that aren't his.
-- **Prose** — apply **#1-31** plus the patterns tagged **(prose + personal)**: em-dash zero-tolerance (#13), contractions (#34), semicolons → periods (#33), sentence-split (#35), felt-experience cut (#36), sentence-fragment rewrite (#37), terse-cut (#38), no-emphasis-formatting (#41), single-sentence cadence (#43), parenthetical asides (#44), and the declarative-register marker (#45). Use for prose Craig authors or sends in his own voice that isn't a publish artifact: emails, documents he writes or hands to someone, working notes, journal entries. This is the mode that finally applies his actual writing voice to the documents he most wants it on. It skips the artifact-mechanics patterns (#32, #39, #40, #42) — those assume a commit or PR and misfire on free prose (a document is legitimately third-person; a journal has no public-scope concern; praise/correction asymmetry and finding stems are PR-review rules).
-- **Personal** — apply all **#1-45**. Use only for publish artifacts: commits, PR titles + bodies, and PR review comments. Adds the artifact-mechanics patterns (#32 first-person rewrite, #39 public-artifact scope flag, #40 praise/correction asymmetry, #42 finding stems) on top of everything prose mode walks.
+- **Prose** — apply **#1-31** plus the patterns tagged **(prose + personal)**: em-dash zero-tolerance (#13), contractions (#34), semicolons → periods (#33), sentence-split (#35), felt-experience cut (#36), sentence-fragment rewrite (#37), terse-cut (#38), no-emphasis-formatting (#41), single-sentence cadence (#43), parenthetical asides (#44), and the declarative-register marker (#45) — plus **#47 (recipient-priority ordering)** when the piece is correspondence (email, Signal, a letter). Use for prose Craig authors or sends in his own voice that isn't a publish artifact: emails, documents he writes or hands to someone, working notes, journal entries. This is the mode that finally applies his actual writing voice to the documents he most wants it on. It skips the artifact-mechanics patterns (#32, #39, #40, #42) — those assume a commit or PR and misfire on free prose (a document is legitimately third-person; a journal has no public-scope concern; praise/correction asymmetry and finding stems are PR-review rules).
+- **Personal** — apply **#1-46**. Use only for publish artifacts: commits, PR titles + bodies, and PR review comments. Adds the artifact-mechanics patterns (#32 first-person rewrite, #39 public-artifact scope flag, #40 praise/correction asymmetry, #42 finding stems, #46 comma budget) on top of everything prose mode walks. #47 is the lone exception to the nesting — it is correspondence-only, and a publish artifact is never a reply to someone's news, so personal mode stops at #46.
If invoked without a mode argument, default to general. Prose mode is invoked explicitly with `/voice prose` (emails, authored documents). Personal-context callers (`commits.md` publish flow, `respond-to-cj-comments.md`) invoke `/voice personal`.
@@ -44,16 +44,16 @@ Terse is a budget, not an adjective. Each publish-artifact type has a target sha
|----------|--------|
| Commit body | Skip entirely when the subject line carries the change. Otherwise short paragraphs: the constraint, bug, or tradeoff. No play-by-play. |
| PR description | Problem / Fix / Why / Testing, each section tight. |
-| PR review summary | One long sentence or a few short ones. Verdict closes it. Verdict formulas ("Approving.", "Requesting changes.") are valid sentences here. |
+| PR review summary | Lead with the substantive pointer, verdict closes it. No praise, not even a bare positive (#40). Verdict formulas ("Approving.", "Requesting changes.") are valid sentences here. |
| Inline pin (finding) | ~4 sentences in stems shape (#42): where the bug is, the fix, why it's better. |
-| Praise comment | One sentence naming what's good. Nothing else (#40). |
+| Praise comment (inline only) | One sentence naming what's good. Nothing else (#40). Never in the summary body. |
| Follow-up approval after prior feedback was addressed | Exactly "Approved." |
## Your Task
When given text to edit:
-1. **Identify which patterns apply** — Scan for the patterns numbered below. General mode walks #1-31 only. Prose mode adds the patterns tagged **(prose + personal)**. Personal mode adds those *and* the ones tagged **(personal only)** — i.e. all 45.
+1. **Identify which patterns apply** — Scan for the patterns numbered below. General mode walks #1-31 only. Prose mode adds the patterns tagged **(prose + personal)**, plus #47 when the piece is correspondence. Personal mode adds the **(personal only)** ones instead — patterns #1-46, never #47.
2. **Rewrite problematic sections** — Replace each detected pattern with its rewrite.
3. **Preserve meaning** — Keep the core message intact.
4. **Maintain voice** — Match the intended tone (formal, casual, technical, academic, literary).
@@ -296,11 +296,11 @@ See `voice/references/voice-profile.org` §31 for problem, basis, examples, and
## Craig's Voice (prose + personal modes)
-These patterns carry Craig's writing voice. Most apply in **both** prose mode (emails, documents, notes he authors) and personal mode (commits, PRs, PR comments) — tagged **(prose + personal)**. Four are publish-artifact-specific — tagged **(personal only)** — because they assume a commit or PR and misfire on free prose: #32 (first-person rewrite) wrongly imposes "I did X" voice on a document that's legitimately third-person, #39 (public-artifact scope flag) has nothing to guard in a private journal, and #40 (praise/correction asymmetry) and #42 (finding stems) are PR-review rules. General mode skips all of them — it edits text that isn't Craig's, where contractions, em-dash elimination, and first-person would conflict with academic, literary, or formal registers.
+These patterns carry Craig's writing voice. Most apply in **both** prose mode (emails, documents, notes he authors) and personal mode (commits, PRs, PR comments) — tagged **(prose + personal)**. Five are publish-artifact-specific — tagged **(personal only)** — because they assume a commit or PR and misfire on free prose: #32 (first-person rewrite) wrongly imposes "I did X" voice on a document that's legitimately third-person, #39 (public-artifact scope flag) has nothing to guard in a private journal, #40 (praise/correction asymmetry) and #42 (finding stems) are PR-review rules, and #46 (comma budget) is scoped to publish artifacts by Craig's 2026-07-20 directive. One — #47 (recipient-priority ordering) — runs the other way: prose-only and narrower still, firing solely on correspondence, because ordering a reply around the recipient's news has no meaning for a commit or a document addressed to nobody. General mode skips all of them — it edits text that isn't Craig's, where contractions, em-dash elimination, and first-person would conflict with academic, literary, or formal registers.
### 32. First-Person Voice Rewrite [personal]
-**Rule.** Rewrite impersonal third-person publish-artifact bodies into first person ("I added X", "I kept Y because..."). The commit subject line stays imperative per Conventional Commits. Skip for mechanical changes where the subject alone carries the message.
+**Rule.** Rewrite impersonal third-person publish-artifact bodies into first person ("I added X", "I kept Y because..."). **The "I" is Craig**, who is the author of record and whose name the artifact goes out under — not an agent narrating work done on his behalf. Cut any construction that writes an agent in as a separate party ("Craig asked me to", "I filed this for Craig", "needs Craig's decision"); an open decision is his own, written as "I haven't decided whether…". The commit subject line stays imperative per Conventional Commits. Skip for mechanical changes where the subject alone carries the message.
See `voice/references/voice-profile.org` §32 for problem, basis, examples, and history.
@@ -348,7 +348,7 @@ See `voice/references/voice-profile.org` §39 for problem, basis, examples, and
### 40. Praise vs Correction Asymmetry [personal]
-**Rule.** Praise on a PR review is short and unjustified (the author knows why their good change is good). Correction always explains the why, gently and briefly, the way a mentor would. Never as a verdict from on high. **Verification narration is the same defect as justified praise:** "I traced X and it's safe because..." pads the compliment with the reviewer's homework. Tracing the code is the reviewer's job, not content for the comment — if verification found a problem, the problem gets the words; if it found nothing, it gets zero words.
+**Rule.** Praise on a PR review is short and unjustified (the author knows why their good change is good). Correction always explains the why, gently and briefly, the way a mentor would. Never as a verdict from on high. **Verification narration is the same defect as justified praise:** "I traced X and it's safe because..." pads the compliment with the reviewer's homework. Tracing the code is the reviewer's job, not content for the comment — if verification found a problem, the problem gets the words; if it found nothing, it gets zero words. **An approve summary carries no praise at all** — not even a bare positive ("Clean.", "Solid fix."). Lead the summary with the substantive pointer (the design note pinned inline) and close with the verdict: "One design note inline, not a blocker. Approving." An approve with nothing to flag is just "Approving." Short unjustified praise survives only as an inline pin on the line it refers to, never in the summary body.
See `voice/references/voice-profile.org` §40 for problem, basis, examples, and history.
@@ -366,7 +366,7 @@ See `voice/references/voice-profile.org` §42 for problem, basis, examples, and
### 43. Single-Sentence Paragraph Cadence Is a Feature [prose · personal]
-**Rule.** A one-sentence paragraph is a finished thought, not a fragment. Break paragraphs after one complete thought when the next thought shifts angle, even if both are short. Never merge short paragraphs into multi-sentence ones in a "clean prose" pass (corpus: 41-74% of Craig's paragraphs are exactly one sentence, depending on register).
+**Rule.** A one-sentence paragraph is a finished thought, not a fragment. "Shifts angle" means shifts *topic*: break paragraphs at a topic boundary, even when both sides are one sentence. Within a single topic, do the opposite — consolidate its sentences into one paragraph even when each is a complete thought, up to a ceiling of about five or six sentences, past which find a natural break. The never-merge instruction protects the break *between* topics, which is where Craig's cadence lives (corpus: 41-74% of his paragraphs are exactly one sentence, depending on register); it never licenses fragmenting one topic across several paragraphs. See #47 for the ordering of those topics in a reply.
See `voice/references/voice-profile.org` §43 for problem, basis, examples, and history.
@@ -382,10 +382,22 @@ See `voice/references/voice-profile.org` §44 for problem, basis, examples, and
See `voice/references/voice-profile.org` §45 for problem, basis, examples, and history.
+### 46. Comma Budget — Max Two Per Sentence [personal]
+
+**Rule.** No sentence carries more than two commas. Rewrite the third comma away: split the sentence, move a clause into a parenthetical (#44) or behind a colon, or break an inline serial list into bullets or its own sentence. Count prose commas only — commas inside code spans, quoted log lines, and literal strings don't count toward the budget.
+
+See `voice/references/voice-profile.org` §46 for problem, basis, examples, and history.
+
+### 47. Recipient-Priority Ordering [prose — correspondence only]
+
+**Rule.** In a reply, lead with what matters most to the *recipient*, not with what's easiest to answer or the order they wrote it. Their news outranks your logistics, and a direct question they asked can sort *below* personal news they shared, because the news is what they care about. Leading with the easy answer reads as transactional. Correspondence-scoped: it needs a recipient, so it fires on email, Signal, and letters, and has no referent in a journal, a working note, or any document addressed to nobody. This is the one prose-mode pattern that does not carry into personal mode — a commit or PR review is not a reply to someone's news.
+
+See `voice/references/voice-profile.org` §47 for problem, basis, examples, and history.
+
## Process
1. Read the input text carefully. Confirm the mode (general, prose, or personal) — invocation argument or context. If a file path was given, that file is the deliverable: the final text gets written back to it in step 7.
-2. Walk patterns 1-31 in general mode; add the (prose + personal) patterns in prose mode; walk all 45 patterns in personal mode.
+2. Walk patterns 1-31 in general mode; add the (prose + personal) patterns in prose mode, plus #47 when the piece is correspondence; walk patterns 1-46 in personal mode.
3. For each pattern, scan the text. If a match is found, rewrite it according to the pattern's rule. Patterns #39 and #45 emit flags without rewriting.
4. After walking all patterns, ensure the revised text:
- Sounds natural when read aloud
@@ -395,7 +407,7 @@ See `voice/references/voice-profile.org` §45 for problem, basis, examples, and
5. **Terse pass — mandatory, last rewrite pass (prose + personal modes).** Walk pattern #38 again as a standalone action: read each sentence and try to cut it in half, keeping only the words that change meaning. Run it on its own here, not folded into step 3's walk — it is the most-skipped pattern and the bloat it catches is the first thing a reader notices. General mode skips this step — academic and third-party registers keep their transition markers.
6. **Anti-AI audit — on the final text.** Prompt: "What makes the below so obviously AI generated?" Answer briefly with remaining tells, then revise. This runs *after* the terse pass so the audited text is the text that ships. If the audit triggers rewrites, re-apply the #38 per-sentence test to every changed sentence before proceeding.
7. **Write-back.** If the invocation supplied a file path, write the final text to that file now and say so. The publish flow posts from the file (`git commit -F`, `gh pr create --body-file`), so a final text that lives only in chat is a drift bug waiting to post the un-voiced version.
-8. **Attestation block (prose + personal modes).** The high-recurrence patterns — the ones with a documented failure history — each get one explicit line: pattern, checked, match or no match, action taken. Current high-recurrence set: **#13 (em-dash), #37 (fragments), #38 (terse), #40 (praise asymmetry), #42 (finding stems)**. This is a receipt, not a summary: a pattern with no match still gets its line. When a pattern in this set fails in the wild despite the receipt, escalate it the way #38 was escalated; when one holds clean for a long stretch, it can rotate out.
+8. **Attestation block (prose + personal modes).** The high-recurrence patterns — the ones with a documented failure history — each get one explicit line: pattern, checked, match or no match, action taken. Current high-recurrence set: **#13 (em-dash), #37 (fragments), #38 (terse), #40 (praise asymmetry), #42 (finding stems), #46 (comma budget)**. This is a receipt, not a summary: a pattern with no match still gets its line. When a pattern in this set fails in the wild despite the receipt, escalate it the way #38 was escalated; when one holds clean for a long stretch, it can rotate out.
9. Present the final version per the Output Format below.
## Output Format
@@ -480,7 +492,9 @@ This skill draws from:
- Orwell, *Politics and the English Language* — patterns #26 (short over long), #27 (active over passive), #29 (cliché).
- Plain English Campaign — pattern #26 (Plain English wordlist).
- Garner, *Modern English Usage* — pattern #26 (word-pair preferences).
-- Craig's voice rules from `claude-rules/commits.md` (Voice and Focus section) — patterns #32-42, split across prose mode (his authored prose and email) and personal mode (publish artifacts).
+- Craig's voice rules from the `publish` skill (Voice and Focus section) — patterns #32-42, split across prose mode (his authored prose and email) and personal mode (publish artifacts).
+- Craig's directive, 2026-07-20 (archsetup session) — pattern #46 (comma budget), personal mode.
+- Craig's edit of a Signal reply, 2026-07-23 (home session) — pattern #47 (recipient-priority ordering) and the #43 topic-vs-angle calibration, prose/correspondence.
- Corpus measurement (2026-05-29 phases 1-2, documented in the profile) — patterns #43-45 and the calibration notes on #7, #13, #33.
Key insight (Wikipedia, paraphrased): LLMs use statistical algorithms to guess what should come next. The result tends toward the most statistically likely text that applies to the widest variety of cases. Patterns #1-25 detect that signature.
diff --git a/voice/references/voice-profile.org b/voice/references/voice-profile.org
index 088f0eb..3dfddf2 100644
--- a/voice/references/voice-profile.org
+++ b/voice/references/voice-profile.org
@@ -1308,9 +1308,9 @@ Local absolute paths (=/home/<user>/...=, =/Users/<user>/...=), private repo nam
Personal mode only. General and prose skip because the rule assumes a PR review context.
*** Rule
-Praise on a PR review is short and unjustified (the author knows why their good change is good). Correction always explains the why, gently and briefly, the way a mentor would, never as a verdict from on high. Keep it brief either way.
+Praise on a PR review is short and unjustified (the author knows why their good change is good), and it survives only as an inline pin on the line it refers to. Correction always explains the why, gently and briefly, the way a mentor would, never as a verdict from on high. Keep it brief either way.
-On an approve summary: praise plus verdict, nothing else. Cut any clause that describes or justifies the change. "Clean fix on the stacking bug, the tri-state is the right level to solve it at, and the tests cover the edges. Approving." becomes "Clean fix on the stacking bug. Approving." If a clause references what the code does or why it works, delete it.
+On an approve summary: no praise at all, not even a bare positive ("Clean.", "Solid fix."). Lead with the substantive pointer — the design note pinned inline — and close with the verdict; an approve with nothing to flag is just "Approving." "Clean fix on the stacking bug, the tri-state is the right level to solve it at, and the tests cover the edges. Approving." becomes "One design note inline, not a blocker. Approving." (or just "Approving." with nothing to flag). Cut any clause that describes, justifies, or compliments the change — if a clause references what the code does, why it works, or how good it is, delete it.
On a finding or change-request: always give the why, gently and briefly. Not "Move this to a helper." but "I'd pull this into one helper — three copies of the same rule means the next change has to touch all three, and missing one brings the bug back."
@@ -1329,9 +1329,11 @@ Nice clean migration, the provider mocks and the Normal/Boundary/Error cases are
*** After
#+begin_example
-Clean migration. Approving. One note inline: I'd rename `x` to `provider` — it reads as a generic placeholder and the next person won't know it's the resolved provider without tracing it.
+One naming note inline, not a blocker. Approving.
#+end_example
+The rename rationale (`x` reads as a generic placeholder; the next person won't know it's the resolved provider without tracing it) lives in the inline pin, not the summary — the summary points, the pin teaches.
+
*** Before (verification narration)
#+begin_example
All three fixes look right. I traced useMapActions and the unmount cleanup is safe because the hook returns a memoized object, and the provider wraps the whole app so neither call site lands on the no-op path.
@@ -1339,16 +1341,19 @@ All three fixes look right. I traced useMapActions and the unmount cleanup is sa
*** After
#+begin_example
-All three fixes are clean and well-aimed.
+Approving.
#+end_example
+Nothing to flag, so the summary is the bare verdict. The old "All three fixes are clean and well-aimed" is itself praise, and praise is now cut from the approve summary entirely.
+
*** Detection
-In a PR review summary or comment: a praise clause that explains why the good thing is good, a praise clause followed by the verification work that supports it, or a finding or change-request that states what to fix without saying why.
+In a PR review summary or comment: any praise on an approve summary (including a bare positive), a praise clause that explains why the good thing is good, a praise clause followed by the verification work that supports it, or a finding or change-request that states what to fix without saying why.
*** History
- Original SKILL.md entry: praise-versus-correction asymmetry for PR review.
- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
- 2026-06-10: verification-narration variant added after the third recurrence — a review draft praised a fix and then narrated the verification supporting the praise (the #236 draft). Added to the SKILL.md rule line and the high-recurrence attestation set. Craig's call, from the work-project session.
+- 2026-07-11: bare-positive carve-out removed. An approve summary now carries no praise at all, not even "Clean." / "Solid fix." — lead with the substantive pointer, close with the verdict. Craig's ruling from a DeepSat review session (approved "One design note inline, not a blocker. Approving."). Same change applied to review-code's Posted Summary Voice and commits.md Shape 1.
** §41 No Emphasis Formatting
@@ -1439,11 +1444,13 @@ In a PR review finding: a sentence carrying more than one claim (chained through
Prose and personal modes. General mode skips because third-party registers legitimately prefer multi-sentence paragraphs.
*** Rule
-A one-sentence paragraph is a finished thought, not a fragment. Break paragraphs after one complete thought when the next thought shifts angle, even if both are short. Never merge short paragraphs into multi-sentence ones in a "clean prose" pass.
+A one-sentence paragraph is a finished thought, not a fragment. "Shifts angle" means shifts *topic*: break paragraphs at a topic boundary, even when both sides are one sentence. Within a single topic, consolidate its sentences into one paragraph even when each is complete, up to a ceiling of about five or six sentences, past which find a natural break. The never-merge instruction protects the break *between* topics; it never licenses fragmenting one topic across several paragraphs.
*** Problem
Most prose-style guides advise multi-sentence paragraphs, so a generic cleanup pass merges Craig's short paragraphs and erases a distinctive feature of his voice. This is a protective pattern: it guards an existing trait rather than correcting a defect.
+The 2026-07-23 boundary refinement addresses the opposite failure, discovered the same day: reading "angle" at *sentence* granularity, so that three sentences all about one topic (a baking run, a mixer, tortillas) got split into three paragraphs as if each were a new angle. That fragments one topic and reads as a checklist rather than a person talking. Angle means topic. #43 governs the break between topics; consolidation fills in what happens within one, which the original rule never specified. The two are one rule seen from both sides, not a rule in tension with #47.
+
*** Basis
Corpus-measured (2026-05-29). Single-sentence-paragraph rate: git commits 41.1%, personal email 57.4%, work email 44.5%, PR descriptions 74.4%, PR review comments 50.0%. Between 41% and 74% of Craig's paragraphs are exactly one sentence, depending on register.
@@ -1465,6 +1472,7 @@ An edit pass that merged short paragraphs, or a draft whose paragraphs each stac
*** History
- 2026-05-29: surfaced by the Phase 1-2 corpus measurement as a "worth adding" trait; filed as suggested delta 4.
- 2026-06-10: promoted from the suggested-deltas list into a numbered pattern. Craig's call, from the work-project session.
+- 2026-07-23: boundary refined (angle means topic; within-topic consolidation to a ~5-6 sentence ceiling; never-merge reframed as across-topic protection). From a home session drafting a Signal reply, where the original rule was misread at sentence granularity. Paired with the new §47.
** §44 Parenthetical Asides Are Part of the Voice
@@ -1527,3 +1535,79 @@ A question mark in a draft in Craig's voice. Flag it; keep genuine questions to
*** History
- 2026-05-29: surfaced by the Phase 1-2 corpus measurement as a register marker; filed as suggested delta 6.
- 2026-06-10: promoted from the suggested-deltas list into a numbered advisory pattern. Craig's call, from the work-project session.
+
+** §46 Comma Budget — Max Two Per Sentence
+
+*** Modes
+Personal mode only. Prose and general modes skip.
+
+*** Rule
+No sentence carries more than two commas. Rewrite the third comma away: split the sentence, move a clause into a parenthetical (§44) or behind a colon, or break an inline serial list into bullets or its own sentence. Count prose commas only — commas inside code spans, quoted log lines, and literal strings (version numbers, paths) don't count toward the budget.
+
+*** Problem
+Three or more commas in one sentence almost always mark stacked clauses or an inline list doing a paragraph's work. The sentence reads fine to its author and lands as a pileup on the reader. The comma count is a mechanical proxy the walk can enforce, where "don't stack clauses" is prose advice that gets skipped.
+
+*** Basis
+Craig's directive, 2026-07-20 (archsetup session, while gating a Hyprland issue draft): "no more than two commas per sentence. we should add that to the /voice personal pass."
+
+*** Before (spec-sheet line with three commas, from the draft that prompted the rule)
+#+begin_example
+System: Arch Linux, kernel 6.18.25-lts, AMD Strix Halo (Radeon 8060S), no plugins loaded.
+#+end_example
+
+*** After
+#+begin_example
+System: Arch Linux, kernel 6.18.25-lts. GPU: AMD Strix Halo (Radeon 8060S). No plugins loaded.
+#+end_example
+
+*** Detection
+Count commas per sentence on the final text. A sentence at three or more gets restructured, not trimmed to exactly the budget — the third comma is the symptom, the stacked structure is the target.
+
+*** History
+- 2026-07-20: added at Craig's direction from the archsetup session. Scoped to personal mode; broaden to prose only if he asks. Added to the attestation high-recurrence set at birth — a mechanical count is cheap to receipt, and new discipline fails silently without one.
+
+** §47 Recipient-Priority Ordering
+
+*** Modes
+Prose mode, and only when the piece is correspondence (email, Signal, a letter). It needs a recipient, so it has no referent in a journal, a working note, or any document addressed to nobody, and it does not carry into personal mode — a commit or PR review is not a reply to someone's news. General mode skips it with the rest of Craig's voice patterns. This is the one pattern narrower than a whole mode, and the only prose pattern personal mode does not also walk.
+
+*** Rule
+In a reply, lead with what matters most to the recipient, not with what's easiest to answer or the order they wrote it. Their news outranks your logistics. A direct question they asked can sort below personal news they shared, because the news is what they care about.
+
+*** Problem
+The easy draft answers the explicit question first and orders the rest as it arrived. That reads as transactional — logistics before the person. Ordering by what the recipient cares about is what makes a reply read as one person talking to another rather than a ticket being closed. Nothing else in the skill governs the *order* of a reply's contents; the other patterns act within a paragraph or a sentence.
+
+*** Basis
+Craig's edit of a Signal reply to his sister, 2026-07-23 (home session). His framing: "start with what would be the most important things to her."
+
+*** Before (first draft — opens with the only explicit question, cooking split across three paragraphs)
+#+begin_example
+Yes, I do subscribe to MasterClass — happy to share what I've watched.
+
+That's amazing about the sourdough. English muffins from scratch is no joke.
+
+The home-roasted deli meat sounds incredible.
+
+A stand mixer would make the bread a lot easier — worth it if you're baking this much.
+
+And 30 pounds — that's huge. So happy for you.
+#+end_example
+
+*** After (Craig's order — weight first, one cooking paragraph, then the question, then the close)
+#+begin_example
+Thirty-plus pounds — that is huge, and I'm so happy for you. That's real work.
+
+And the cooking. Sourdough, English muffins, tortillas, home-roasted deli meat from scratch — that's a whole kitchen you've built, and a stand mixer would make the bread much easier if you're baking at this volume, so I say go for it. I want to hear how the tortillas come out.
+
+Yes, I subscribe to MasterClass — I'll send you what I've been watching.
+
+I miss you and I love you. Send me a few times that work for a call.
+#+end_example
+
+The cooking paragraph runs seven sentences, a hair over the §43 ceiling. Craig called it an exception rather than re-cut a message that had already gone out. The guard is the rule; this paragraph is one sentence over it; both facts stay in the record, because a real example at the boundary teaches it better than a clean one.
+
+*** Detection
+A reply whose opening answers a logistical or yes/no question while the recipient's substantive news sits lower. Reorder so the news they'd most want acknowledged leads.
+
+*** History
+- 2026-07-23: added from the home session drafting a Signal reply. The first handoff proposed two new patterns and flagged a conflict with §43; the superseding design resolved that the conflict was a misreading of §43 (angle = topic), leaving one genuinely new pattern here and a calibration to §43. Prose/correspondence-scoped per Craig — email and Signal are prose, not publish artifacts.
diff --git a/working/context-engineering-rightsizing/metrics.org b/working/context-engineering-rightsizing/metrics.org
new file mode 100644
index 0000000..e7ff189
--- /dev/null
+++ b/working/context-engineering-rightsizing/metrics.org
@@ -0,0 +1,269 @@
+#+TITLE: Context-Engineering Rightsizing — Metrics and Stop Conditions
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-27
+
+* Who is holding the instrument
+
+Most of what follows would be measured by me, about changes to my own
+instructions, in a direction I have an obvious interest in. Self-reported
+compliance is unreliable, and it is unreliable in a predictable direction: I
+will under-report misses I didn't notice, because not noticing is the failure.
+
+That is the argument for weighting mechanical detectors over my judgment
+wherever both exist. =lint-org= does not have a stake. Neither does a token
+count, a =git diff=, or you seeing a window on the wrong workspace. Where a
+claim can only be assessed by my self-report, it is marked below as judgment
+rather than measurement, and it should be discounted accordingly.
+
+* Part 1 — Which claims are testable
+
+Each post makes separable claims. Some can be tested here cheaply, some need an
+eval harness we don't have, and inventing a metric for the second group would be
+worse than admitting it.
+
+** Testable now, with existing instrumentation
+
+*** T1 — Report-everything beats pre-filtering (Opus 5 guide)
+
+*Claim.* A review told to be conservative reports less; reporting everything and
+filtering separately surfaces more real issues.
+
+*Test.* A true A/B, runnable today with no waiting. Take three past diffs with
+known outcomes. Run =/review-code= under current rules and under the
+report-all-then-filter version. Compare findings that survive verification.
+
+*Metric.* Real findings per pass, and false-positive rate. The claim holds if
+report-all finds more real issues without the false-positive rate rising
+enough to drown them.
+
+*Why this one first.* It is the only claim in the three posts we can settle in
+an afternoon instead of a week.
+
+*** T2 — Progressive disclosure preserves behavior (context-engineering post)
+
+*Claim.* Moving guidance out of always-loaded context into on-demand loading
+doesn't degrade adherence.
+
+*ANSWERED 2026-07-27 for the deterministic half.* =/context= in a live work
+session lists 17 generic rules under Memory files, with the three path-scoped
+ones absent. Path-scoping is a glob match rather than a model judgment, so it
+either fires or doesn't, and it fires. That half needs no trial and no miss-rate
+metric.
+
+*Still open for the semantic half.* Rules whose condition can't be a glob
+("when a commit is in play") route to skills, where triggering *is* a model
+judgment. Everything in Part 2 applies there and only there.
+
+*A caution the answer doesn't cover.* Path-scoping fires when Claude *reads* a
+matching file. A session that writes an org file without reading one first
+never triggers =todo-format.md=. Edit requires a prior read, so edits to
+existing files are safe; creating a new org file from scratch is the gap. Worth
+watching rather than blocking on.
+
+*** T3a — The token baseline was wrong by 45% [SETTLED]
+
+Word counts converted at a guessed ~1.3 tokens per word. The live number is
+2.28. =commits.md= is 12,800 tokens, not the ~7,000 estimated, and
+=claude-rules/= was ~57,800 tokens per session before today rather than the
+~33,000 implied.
+
+The lesson is narrower than "measure better." I had a real measurement
+available the whole time — =/context= reports per-file token counts — and used
+an estimate instead because the estimate was easier to compute from inside the
+repo. Reach for the instrument that reports the actual quantity.
+
+*** T3 — Deliverables run long without explicit calibration (Opus 5 guide)
+
+*Claim.* Files written to disk are longer than the task needs.
+
+*Test.* Measure what already exists. Word counts of the last twenty session
+archives, the specs in =docs/specs/=, and =todo.org= task bodies. Then add a
+length-calibration instruction and measure the next ten.
+
+*Metric.* Median words per artifact, before and after. Paired with a judgment
+call on whether anything useful was lost, which is the part I can't measure.
+
+*Baseline worth taking now,* since it costs one command and the before-number
+disappears the moment we change anything.
+
+*** T4 — Lower effort holds quality [WITHDRAWN 2026-07-27]
+
+Dropped with P4. The goal is output quality first, and this claim trades
+quality for cost, so testing it would answer a question we've decided not to
+act on either way.
+
+*** T4 (original text, retained for the record)
+
+*Claim.* =low= and =medium= produce strong quality at a fraction of the tokens.
+
+*Test.* Sentry fires hourly and does the same passes each time. Run a week at
+default, a week at =medium=, on the same repo state where possible.
+
+*Metric.* Tokens per fire, and findings per fire. The claim holds if findings
+per fire holds within noise while token cost drops materially.
+
+*Confound to respect.* Sentry's input changes night to night, so findings-per-
+fire is noisy. Two weeks is probably the floor for a readable signal, and the
+result will still be suggestive rather than conclusive.
+
+*** T5 — Duplicate and conflicting instructions cost something (both posts)
+
+*Claim.* Overlapping guidance across surfaces makes the model work harder to
+reconcile.
+
+*Test.* Partially measurable. The duplication itself is countable — Phase 5's
+three axes. Whether removing it improves anything is not measurable without
+evals.
+
+*Metric.* Count of rules stated on more than one surface, driven to zero. That
+measures the cleanup, not the benefit. Honest framing: we're removing a known
+cost, not demonstrating a gain.
+
+** Not testable here — judgment calls, and they should be labelled as such
+
+*** J1 — Examples constrain the exploration space
+
+The post asserts this and I have no way to test it. Removing the examples from
+=commits.md= and observing "things seem fine" is not evidence. This one gets
+adopted on the post's authority or not at all, and I'd lean toward *not*
+stripping examples that encode your taste, since the same post says skills
+should encode exactly that.
+
+*** J2 — Removing verification instructions loses no quality
+
+The claim underneath C1. Testing it properly means running the same tasks with
+and without =verification.md= and comparing error escape rates, which needs a
+task suite we don't have. What we *can* measure is the cost side — tool calls
+and tokens spent on verification per task — but not what it prevents.
+
+Asymmetric risk: the cost of over-verification is tokens, and the cost of
+under-verification is a false completion claim reaching you. Those are not
+equally bad, which argues for keeping the honesty core regardless of what a
+test would show.
+
+*** J3 — Unknowns-discovery reduces rework
+
+Long-horizon and confounded by everything else. Adopt on judgment.
+
+*** J4 — Rich references beat prose descriptions
+
+=ui-prototyping.md= already assumes it and it has worked. That's one project's
+experience, not a measurement, but it's the evidence we have.
+
+* Part 2 — Pilot go/no-go
+
+** The denominator problem
+
+The obvious metric is a miss rate, and the obvious trap is that a rule nobody
+exercised shows zero misses and reads as a pass. Every result below is
+therefore reported as *misses per exposure*, and a rule with too few exposures
+returns no verdict rather than a passing one.
+
+*Exposure* means a session did work in the rule's domain: edited an org table,
+touched a spec's lifecycle, displayed a keymap, captured a window, ran a UI
+spec.
+
+*Minimum exposures before a rule's result counts: 3.* Below that, extend the
+window or swap the rule for one that gets exercised more.
+
+** Primary metric
+
+Misses per exposure, per rule, where a miss is the guidance not being applied
+when it should have been.
+
+| Result per rule | Verdict |
+|----------------------------+----------------------------------------|
+| 0 misses in 3+ exposures | Pass |
+|----------------------------+----------------------------------------|
+| 1 miss, detector caught it | Pass with note; record the miss |
+|----------------------------+----------------------------------------|
+| 2 or more misses | Turn back that rule |
+|----------------------------+----------------------------------------|
+| Miss reached a commit | Turn back now, don't wait for the week |
+|----------------------------+----------------------------------------|
+| Miss reached a project | Turn back now, and see abandon triggers|
+|----------------------------+----------------------------------------|
+| Under 3 exposures | No verdict; extend or swap the rule |
+|----------------------------+----------------------------------------|
+
+** Secondary metrics
+
+- *Token delta.* Measured directly. Expected around 3,000 words. This is the
+ only guaranteed benefit, so if the primary metric fails, we know exactly what
+ we were buying and can decide it wasn't worth it.
+- *Correction cost.* When a miss happens, how long to fix. A misformatted table
+ is seconds. This is what distinguishes an acceptable miss rate from an
+ unacceptable one, and it is why the threshold tightens in Phase 4.
+- *False-trigger rate.* A skill loading when it isn't needed spends tokens for
+ nothing. Worth watching, not worth blocking on.
+
+** Turn back versus abandon
+
+Two different actions, and conflating them would over-react to a single bad
+rule.
+
+*Turn back a rule* — revert that one file to always-loaded. Triggered by 2+
+misses on that rule, or any miss reaching a commit. The rest of the pilot
+continues. Cost: one revert.
+
+*Abandon the rollout* — stop at Phase 0 and keep the current architecture.
+Triggered by any of:
+
+- Three or more of the six rules turn back.
+- A miss reaches another project through the sync.
+- The pattern of misses shows the skill index isn't the fix, meaning D2 was the
+ wrong lever and there's no obvious next one.
+
+*Proceed to Phase 3* requires: at least four of six rules pass, no miss reached
+a commit, and the token drop landed near expectation.
+
+** Threshold scales with blast radius
+
+The pilot tolerates one caught miss per rule because the worst case is a badly
+formatted table. Phase 4 moves =commits.md=, where the worst case is an
+unattributed commit or a leaked path in a public artifact.
+
+*Phase 4 threshold is zero.* One miss on an attribution, scope, or grading rule
+turns that rule back immediately, with no pass-with-note tier. Stated now so
+it isn't negotiated later under the pressure of wanting the phase to succeed.
+
+* Part 3 — Baselines to capture before anything changes
+
+Cheap now, impossible to reconstruct later:
+
+1. Always-loaded word count per surface. Captured: 32,123 total.
+2. Median word count of the last twenty session archives, the specs, and open
+ task bodies. For T3.
+3. Sentry tokens per fire and findings per fire, from the existing metrics
+ file. For T4.
+4. Findings per pass from the last several =/review-code= runs. For T1.
+
+Item 1 is done. The rest are one session's work and should happen before
+Phase 0 changes the review skill, since that change contaminates item 4.
+
+* Instrument reliability — the day's second lesson
+
+The plan weights mechanical detectors over self-report because they have no
+stake. Two failed on 2026-07-27, both reporting success while doing damage:
+=wrap-org-table.el= reflowed a table into a worse shape and =lint-org= then
+certified it clean, and the wrap-teardown hook consumed a two-hour-old sentinel
+and killed a live work session that had done nothing wrong.
+
+Neither invalidates the preference for mechanical detectors, which is still
+right. Both narrow the claim: a detector has no stake, but it can be
+confidently wrong, and a green check from a guard that never looked at the
+thing is indistinguishable from a green check that did. When a detector clears
+a moved rule, the useful question is whether it actually evaluated it, not just
+whether it reported clean.
+
+* What this can and cannot tell us
+
+It can tell us whether on-demand loading fires reliably here, whether
+report-everything finds more real bugs, whether artifacts shrink usefully, and
+what lower effort costs in findings.
+
+It cannot tell us whether examples constrain, whether removing verification
+instructions is safe, or whether unknowns-discovery reduces rework. Those stay
+judgment calls made on the posts' authority and your read, and they should be
+labelled that way in whatever we write down afterward — so that a future session
+doesn't mistake an adopted opinion for a tested result.
diff --git a/working/context-engineering-rightsizing/proposals.org b/working/context-engineering-rightsizing/proposals.org
new file mode 100644
index 0000000..1300a7b
--- /dev/null
+++ b/working/context-engineering-rightsizing/proposals.org
@@ -0,0 +1,350 @@
+#+TITLE: Context-Engineering Rightsizing — Proposals
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-27
+
+* Status
+
+Proposals only. Nothing here is applied. Source: three Anthropic posts Craig
+supplied 2026-07-27 — the Claude 5 context-engineering post (2026-07-24), the
+Opus 5 prompting guide, and the Fable 5 field guide (2026-07-06).
+
+* Goal
+
+Output quality and results first. Token reduction is a real goal and worth
+having, but it is the second one, and where the two conflict quality wins.
+Craig's framing, 2026-07-27: the concern is agents having the freedom to
+produce at their highest capacity, unbound by guardrails that constrain them or
+work against them.
+
+That reordering matters more than it sounds. Anthropic's 80% figure was a
+*finding* — they cut and quality held — not a target. Aimed at directly it
+optimizes the thing we don't care about, and P4 below is the proposal that
+falls to it.
+
+* The measurement (corrected 2026-07-27 from live =/context=)
+
+The original figures in this document were word counts converted at a guessed
+ratio, and they understated by about 45%. A =/context= run in a live work
+session gave the real numbers:
+
+| Surface | Tokens |
+|------------------------------------+--------|
+| =claude-rules/= before this session | 57,800 |
+|------------------------------------+--------|
+| =claude-rules/= now (17 files) | 44,410 |
+|------------------------------------+--------|
+| Path-scoped out (3 files) | 13,390 |
+|------------------------------------+--------|
+
+The real ratio is 2.28 tokens per word, not the ~1.3 assumed. =commits.md=
+alone is 12,800 tokens, not the ~7,000 estimated. Every earlier figure here
+should be read as low.
+
+** Two loading paths, not one
+
+The original framing called 32,123 words "always-loaded" and was wrong to
+lump them. They arrive by different mechanisms:
+
+- *Memory files* — =claude-rules/= (via the =~/.claude/rules/= symlinks),
+ =CLAUDE.md=, project =.claude/rules/=, and auto-memory's =MEMORY.md=. Loaded
+ at session start by the harness. This is the 44,410 above, plus CLAUDE.md and
+ project rules.
+- *Read during startup* — =protocols.org= and the workflow files. These never
+ appear under Memory files in =/context=; the startup workflow reads them, so
+ they land in Messages. Still a real per-session cost, but a different lever:
+ they shrink by editing the workflow, not by scoping a rule.
+
+Conflating the two made =protocols.org= look like it competed with
+=commits.md= for the same fix. It doesn't.
+
+** What the harness says on its own
+
+=/context= ends with its own suggestion: prune =commits.md= (12.8k),
+=testing.md= (6.3k), and =MEMORY.md= (5.5k). That is the Phase 4 target list,
+arrived at independently. Worth treating as corroboration rather than
+coincidence.
+
+* Proposals
+
+Ranked by value. Each carries my confidence and what I think the real risk is.
+
+** P1 — Progressive disclosure for =claude-rules/= [high value, highest risk]
+
+*Change.* Split the twenty rule files into two tiers.
+
+- *Always-loaded core* (target under 3,000 words): the genuine invariants that
+ must fire without being summoned. The no-AI-attribution rule, the
+ cross-project boundary stop, the no-popup-menus and no-reverse-video output
+ constraints, the =date=-before-timestamps rule, and a short index naming
+ which skill covers what.
+- *On-demand tier*: everything procedural, converted to skills whose
+ descriptions trigger them. =commits.md= becomes a publish skill that loads
+ when a commit or PR is in play. =todo-format.md= loads when an org todo file
+ is touched. =testing.md=, =working-files.md=, =docs-lifecycle.md=,
+ =org-tables.md=, =keybinding-display.md= likewise.
+
+*Why.* This is the post's central move, and the token math is the argument.
+
+*The real risk, stated plainly.* A rule that isn't loaded can't fire. Skill
+triggering is probabilistic in a way that always-resident text isn't. The
+failure mode is silent: a session commits without the voice pass because the
+publish skill didn't trigger, and nothing announces the miss. That's the same
+silent-failure shape as the two probe defects fixed this morning.
+
+*Mitigation.* Anything whose violation is expensive and hard to reverse stays
+in the always-loaded core, whatever its length. The split is by *blast radius*,
+not by word count. And the migration goes one file at a time with a live trial,
+not as a single cutover.
+
+*Confidence.* High that the direction is right. Medium on where exactly each
+line falls — that's a judgment call per rule, and worth walking together.
+
+** P2 — Stop pre-filtering review findings [high value, low risk]
+
+*Change.* =review-code/SKILL.md:251= says "Drop Low-confidence issues before
+the final report." Line 434 repeats it. Replace with: report every finding
+carrying an explicit confidence label, then filter in a named second pass.
+
+*Why.* The Opus 5 guide is specific about this: "If your review prompt says
+'only report high-severity issues' or 'be conservative,' the model may follow
+that instruction literally and report less; ask it to report everything and
+filter in a separate pass instead." The guide also reports that on this model
+the extra findings are mostly real rather than false positives, which is the
+premise the drop-rule was written against.
+
+*Confidence.* High. This is the most directly-actionable finding in the three
+posts, and it names the exact pattern the skill implements.
+
+** P3 — Add the unknowns-discovery practices [medium value, low risk]
+
+The field guide describes eight practices. The sweep found these absent:
+=unknown unknowns= framing (0 files), =implementation notes= (0), =quiz= (0).
+Present but thin: =blind spot= (1 file), =interview= (2), =pitch= (1).
+=brainstorm= appears in 5 and =references= is well covered by
+=ui-prototyping.md=, which already runs ahead of the post.
+
+*Change.* Add a blind-spot-pass practice and an interview practice, and an
+implementation-notes convention for long builds (a scratch file logging
+deviations from plan, which then feeds the retrospective). Add the quiz pattern
+to the review or wrap surface.
+
+*Placement matters.* These go in the on-demand tier from P1, never the
+always-loaded core — otherwise this proposal fights the one above it.
+
+*Confidence.* Medium-high on the practices being useful. Lower on the quiz,
+which may not fit how you actually work.
+
+** P4 — Effort calibration [DROPPED 2026-07-27 — trades quality for cost]
+
+*Change.* Sentry, work-the-backlog, and the no-approvals speedrun run many
+passes at default effort. The Opus 5 guide says to use =low= and =medium=
+liberally as the primary cost control wherever quality holds, stepping up only
+for demanding work.
+
+*Why.* Sentry fires hourly. Effort is the lever with the largest cost
+multiple, and most sentry passes are mechanical sweeps.
+
+*Dropped.* Lowering effort buys tokens by spending quality, which is exactly
+backwards under the goal above. Revisit only if a specific pass proves
+genuinely mechanical and a cost problem shows up on its own.
+
+** P5 — Positive framing over prohibition [PROMOTED — now the top quality lever]
+
+*Change.* =commits.md= carries 41 prohibition markers (NEVER / DO NOT /
+MANDATORY / CRITICAL) across 5,561 words. =todo-format.md= 21,
+=testing.md= 19. Rewrite the ones that aren't hard invariants as positive
+descriptions of the wanted behavior.
+
+*Why.* The Opus 5 guide: "Positive examples of the communication style you want
+tend to be more effective than instructions about what not to do."
+
+*Keep as prohibitions:* the AI-attribution rule and anything else where the
+worst case is genuinely unacceptable. Those earn their emphasis.
+
+*Promoted 2026-07-27.* Ranked low when the score was token savings. Under a
+quality-first goal this is the proposal that most directly targets guardrails
+working against good output, which is the stated concern. It still rides along
+with the Phase 4 edits rather than running as its own campaign, because every
+file it touches is a file those phases open anyway.
+
+** P6 — Deduplicate the two always-loaded surfaces [medium value, low risk]
+
+=protocols.org= restates rules that also live in =claude-rules/=: the
+cross-project boundary, the working-files convention, the AI-attribution ban,
+inbox cadence. Both are loaded every session, so each duplicated rule is paid
+for twice, and the two copies can drift apart.
+
+*Change.* One home per rule. =protocols.org= keeps the pointer, the rule file
+keeps the content — or the reverse, but not both.
+
+*Confidence.* High on the duplication being real, medium on which surface
+should own each rule.
+
+* Conflicts — your call, not mine to inherit
+
+Two places where a post contradicts something this system arrived at
+deliberately. I'm flagging rather than adopting.
+
+** C1 — =verification.md= versus the over-verification warning
+
+The Opus 5 guide says: "If your prompt contains explicit verification
+instructions ('include a final verification step for any non-trivial task',
+'use a subagent to verify'), remove them: instructions like these cause
+over-verification on Claude Opus 5, and removing them reduces wasted tokens
+with no loss in quality."
+
+=verification.md= is 1,486 always-loaded words of exactly that shape. But the
+two aren't the same thing, and the distinction decides the answer:
+
+- Its *honesty core* — don't claim tests pass without running them, "unable to
+ verify" is a required outcome, replace beliefs with evidence — is about
+ truthful reporting, not about adding verification steps. The guide doesn't
+ argue against it.
+- Its *process injection* — green baseline before starting, full suite as its
+ own step before every commit — is the shape the guide names.
+
+*My read:* keep the honesty core, shorten it, and let the process injection
+move into the publish skill where it fires only when publishing. *My
+confidence: medium*, and this is the one I'd most want you to overrule if it
+feels wrong. It's also the rule closest to your standing "never guess, always
+check" direction, so the guide's advice and your stated preference genuinely
+pull against each other here.
+
+** C2 — =subagents.md= versus the delegation warning
+
+The guide says "do not use subagents to verify or double-check your own work."
+=subagents.md= has a review-gate cadence and a rule to dispatch a *fix*
+subagent rather than repairing in the orchestrator's context.
+
+*My read:* the fix-subagent rule is about context pollution, not verification,
+so it survives. The review-gate cadence is closer to the flagged pattern and
+wants a look. Most of =subagents.md= already matches the guide's advice — it
+argues against spawning for small work and against letting the agent pick its
+own scope, which is what the guide asks for. *Confidence: medium-high.*
+
+* One thing the posts would change about this document
+
+Both the context-engineering post and the field guide argue that a rich
+reference beats a prose description — an HTML artifact, a test suite, source
+code. This proposal document is prose in org, which is your reading format and
+the right call for a decision doc. Worth noting the tension rather than
+silently ignoring it: for the *next* artifact in this line of work, an HTML
+comparison of the before and after rule tree would likely beat another org
+file.
+
+* Scope proposal for the consistency sweep
+
+The audit behind this document was targeted, not exhaustive — I checked the
+claims the posts made and measured the surface. A real inconsistency sweep over
+20 rule files, 47 workflows, and the skills is its own pass.
+
+Proposed scope, in order:
+
+1. *Contradictions between always-loaded surfaces* — the P6 duplication set,
+ read side by side for drift rather than just counted.
+2. *Stale facts* — assertions about tools, paths, and behavior that were true
+ when written. The spot check found the =agent-page= to =agent-text= rename
+ correctly handled, so this may be in better shape than expected.
+3. *Instructions that contradict each other across files* — the failure the
+ context-engineering post opens with. This is the expensive one and the most
+ valuable.
+
+Sizing: item 1 is an afternoon, item 2 is mechanical, item 3 is the real work.
+
+* From your side of the desk
+
+Everything above treats these files as the agent's context, to be rightsized.
+That was the smaller question. Read as *your prompts* — the map you hand every
+project — the finding is different, and it's the one worth acting on.
+
+** The bottleneck this system was built for has moved
+
+Counting the workflows by what they're for: 41 are execution and hygiene
+(publish, task grading, inbox routing, session archiving, calendar, email,
+sync), 6 are discovery and design (the spec trio, retrospective, code-quality,
+readability-audit). Roughly seven to one.
+
+That ratio was correct when the risk was the model doing things wrong. The
+field guide's claim is that the risk moved: "Claude Fable is the first model
+where I find the quality of the work is bottlenecked by my ability to clarify
+its unknowns." If that's true here too, the system is heavily invested in the
+half of the problem that got easier and thin on the half that didn't.
+
+Not an argument to delete the execution machinery. It's load-bearing, and
+hygiene that runs itself is exactly what you want automated. The argument is
+that *the growth has all been on one side*, and the next increment of quality
+probably comes from the other one.
+
+** The instructions don't practice what they demand
+
+Three concrete cases, all checkable:
+
+- =commits.md= says "Brief. Terse is preferred. A one-sentence body beats a
+ paragraph saying the same thing." It is 5,561 words, the longest file in the
+ set.
+- =interaction.md= bans bold and code spans in chat output because they render
+ as reverse video. The rule files carry 591 bold markers.
+- =testing.md= mandates TDD as "non-negotiable" and follows with a table of
+ eight rationalizations to refuse. That's the repeat-yourself-and-overconstrain
+ shape the context-engineering post retired, applied to a rule the model no
+ longer needs argued into.
+
+This matters beyond tidiness. An instruction whose own form contradicts its
+content is a mixed signal of exactly the kind the post opens with — the
+model spends effort reconciling "be terse" against a source that isn't.
+
+** Over-specification has a cost you're paying, not me
+
+The field guide: "If you are too specific, Claude will follow your instructions
+even when a pivot may be more appropriate. If you are too vague, Claude will
+often make choices and assumptions based on industry best practices that may
+not be a fit."
+
+The workflows are phase-numbered and gate-heavy. For repeatable operations —
+wrap-up, publish, inbox — that precision is right and should stay. For the
+creative surfaces it's a straitjacket: a spec workflow that prescribes its own
+phases forecloses the pivot that a brainstorm was supposed to surface. The
+system doesn't currently distinguish those two classes, and it should. Same
+file format, opposite optimal specificity.
+
+** Which gates are guardrails and which are preference
+
+The approval gates (publish, inbox shared-asset, spec) were written when the
+worst case was an agent shipping something wrong. Some of them you'd want
+regardless, because you want to be in the loop on what goes out under your
+name. Those are preference and should never be cut.
+
+Others are pure guardrail, and the context-engineering post's argument applies:
+"these constraints were once needed to avoid worst case scenarios, we have
+since found we can delete many of them."
+
+I can't tell the two apart from the files — they read identically. Only you
+know which gates you'd keep if the agent were perfectly reliable. That
+separation is the highest-leverage thing you could tell me, and it's not
+something I should guess at.
+
+** One map, many territories
+
+Every project loads the same 32,000 words. The finances project loads the TDD
+mandate and the full publish machinery. =.emacs.d= loads the org-table
+standard. Some of that is right (the universal rules genuinely are universal)
+and some is the map/territory mismatch the field guide names.
+
+The language-bundle mechanism already solves this for =languages/=. Nothing
+equivalent exists for the rules layer.
+
+** What's already right
+
+Worth saying, because the list above is all deficit:
+
+- =ui-prototyping.md= is the field guide's brainstorm-and-prototype practice,
+ written down before the post and in more detail than the post gives it.
+- The spec spine (decisions plus implementation phases) is the guide's
+ implementation-plan pattern.
+- =session-context.org= is the implementation-notes pattern, and it already
+ captures deviations and dead ends.
+- =retrospective.org= exists.
+
+So the practices are partly here. What's missing is that they're framed as
+*process compliance* rather than as *unknowns discovery*, and none of them fire
+at the moment the guide says they pay off — before scope is set.
diff --git a/working/context-engineering-rightsizing/rollout.org b/working/context-engineering-rightsizing/rollout.org
new file mode 100644
index 0000000..d641ce7
--- /dev/null
+++ b/working/context-engineering-rightsizing/rollout.org
@@ -0,0 +1,371 @@
+#+TITLE: Context-Engineering Rightsizing — Rollout Schedule
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-07-27
+
+* Why this is phased rather than done in one pass
+
+The change is a bet that on-demand loading fires as reliably as always-resident
+text. That bet is cheap to test and expensive to assume. Every phase below
+either produces evidence or spends evidence already earned. Nothing
+load-bearing moves before the mechanism has been watched working.
+
+The second reason is blast radius. Everything here rides the template sync into
+every project on its next startup, so a bad phase is not contained to this
+repo. The early phases are chosen so a failure is visible and harmless.
+
+* A finding that changes the plan
+
+The harness system prompt already carries most of what the Opus 5 guide
+recommends adding. Its task-scope block, its correction-narration block, and
+its subagent-delegation cap are present nearly verbatim. The
+context-engineering post's replacement comment guidance ("write code that reads
+like the surrounding code") is present as the post's own new wording.
+
+Two consequences:
+
+1. *Do not "apply the posts" by adding their suggested prompt blocks.* They are
+ already live. Adding them to =claude-rules/= would create exactly the
+ duplicate-and-conflict problem the first post opens with, while making the
+ token count worse.
+2. *There is a third deduplication axis.* The proposals named =protocols.org=
+ against =claude-rules/=. There is also =claude-rules/= against the harness
+ system prompt, and that one is invisible from inside the repo. Any rule that
+ restates harness guidance is pure cost.
+
+This is why the posts' value here is subtractive, not additive.
+
+* Status — 2026-07-27, end of first working session
+
+Phase 0 and the deterministic half of Phase 1 are done and verified in a live
+session. The pilot's central question is answered, which changes what remains.
+
+** Shipped
+
+- =paths:= frontmatter on the three rules that already declared a file-type
+ scope in prose (=todo-format=, =org-tables=, =emacs=), plus a =lint.sh=
+ checker that catches the prose/frontmatter mismatch, plus a heading check
+ taught to skip frontmatter. Commit 0adcb1a.
+- Generic rules no longer ship per project. =install-lang= stopped copying them
+ and =sync-language-bundle= sweeps what earlier installs left, guarded on the
+ global rule existing. Swept 20 files each from work and =.emacs.d=. Commit
+ 7ea1d7b.
+- The live session anchor is gitignored, so rulesets stops reporting
+ sync-blocked for the whole of every session. Same commit.
+
+** Verified in a live work session
+
+=/context= lists 17 generic rules under Memory files. =todo-format.md=,
+=org-tables.md=, and =emacs.md= are absent, and only =python-testing.md= and
+=publishing.md= come from the project's own rules directory.
+
+So *path-scoping works at user level* and *the de-duplication holds*. Both were
+open questions this morning.
+
+** What that changes
+
+Path-scoping is a glob match, not a model judgment. It is deterministic, so the
+silent-miss risk the whole pilot was designed around does not apply to it. That
+splits the remaining work in two:
+
+1. *Path-scopable* — any rule whose scope is a file type or directory. Ships
+ immediately, no trial, no detectors. =docs-lifecycle.md= is the obvious next
+ one (=docs/**=), and parts of =working-files.md= may qualify.
+2. *Semantic* — rules whose condition can't be written as a glob ("any spec
+ with a non-trivial UI", "when a commit is in play"). These still need the
+ skills route, and they are the only place the pilot's detectors and stop
+ conditions apply.
+
+=commits.md= is the case that matters: 12,800 tokens, the single largest item,
+and almost all of it is publish machinery that only applies when a commit is in
+play. That is a task scope rather than a path scope, so it is the skills route
+and the real test of the risky tier.
+
+** Correction carried from the live numbers
+
+Earlier phases in this document quote word counts converted at a guessed ratio
+and understate by about 45%. The real ratio is 2.28 tokens per word. Read the
+targets below as token figures needing that correction, and see proposals.org
+for the corrected table.
+
+** A pattern worth designing around
+
+Two mechanical guards failed in the same day, both mine, both reporting success
+while doing damage: =wrap-org-table.el= reflowed a table into a worse shape and
+=lint-org= then certified it clean, and the wrap-teardown hook consumed a
+two-hour-old sentinel and killed a live work session.
+
+This plan leans on mechanical detectors precisely because they have no stake in
+the outcome. Both incidents say that is necessary but not sufficient — a
+detector can be confidently wrong. Whatever Phase 3 decides, the verification
+step should include "did the guard's own claim get checked," not just "did the
+guard report clean."
+
+* Phase 0 — Free wins [DONE except /doctor]
+
+*Scope.* Three items that interact with nothing.
+
+1. Fix the review-finding pre-filter (P2). =review-code/SKILL.md= lines 251 and
+ 434 tell the reviewer to drop low-confidence findings before reporting.
+ Replace with report-everything-labelled, filter in a named second pass.
+2. Run =/doctor=. The context-engineering post says Anthropic shipped these
+ practices as a command that rightsizes skills and =CLAUDE.md=. Its output is
+ free evidence, and it may disagree with this plan, which is worth knowing
+ before executing it.
+3. Record the harness-overlap finding above where it will be seen at the moment
+ it matters — a note in the rules index, not buried in this document.
+
+*Reasoning.* None of these depend on the pilot's outcome, and item 2 could
+change the plan.
+
+*Decision needed:* none.
+
+*Success criteria.* Review skill reports with confidence labels and a separate
+filter step. =/doctor= output read and reconciled against this schedule.
+
+*Rollback.* Single revert; nothing downstream depends on it.
+
+* Phase 1 — The pilot migration [deterministic half DONE; semantic half pending]
+
+*Scope.* Six rule files move from always-loaded to on-demand skills. Roughly
+3,000 words, about 12% of the rules surface.
+
+| File | Words | How a silent miss would be caught |
+|-----------------------+-------+-------------------------------------------|
+| =org-tables.md= | 464 | =lint-org= checker =org-table-standard= |
+|-----------------------+-------+-------------------------------------------|
+| =docs-lifecycle.md= | 582 | spec status-board grep; =lint-org= checkers |
+|-----------------------+-------+-------------------------------------------|
+| =ui-prototyping.md= | 696 | =spec-review= verifies the process ran |
+|-----------------------+-------+-------------------------------------------|
+| =keybinding-display.md= | 505 | you see the wrong format immediately |
+|-----------------------+-------+-------------------------------------------|
+| =desktop-capture.md= | 458 | a window lands on your active workspace |
+|-----------------------+-------+-------------------------------------------|
+| =patterns.md= | 291 | already only a pointer; nothing to miss |
+|-----------------------+-------+-------------------------------------------|
+
+*Reasoning — the selection rule matters more than the list.* These were not
+picked for being small or cheap. They were picked because *a failure to fire is
+detectable*. Four have a mechanical checker or workflow gate that catches the
+miss; two produce an error you see within seconds. That is what makes the pilot
+an experiment rather than a hope.
+
+=daily-drivers.md= and =emacs.md= were considered and held back. Both are
+low-risk in content, but a miss on either surfaces slowly — as drift on the
+other machine, or as a stale daemon — so neither would tell us anything within
+the trial window.
+
+*Decisions needed.*
+
+- *D1 — Confirm the pilot set.* Six files as listed, or trim further. My
+ recommendation is the six: fewer than that and the trial may not exercise the
+ mechanism enough to learn from.
+- *D2 — Does the always-loaded core carry a skill index?* A one-line-per-skill
+ list naming what exists and when it applies. It costs perhaps 200 words and
+ should materially improve trigger reliability, since the model can see that a
+ rule exists even when its content isn't loaded. My recommendation is yes, and
+ the pilot is the right place to test whether the index is what does the work.
+
+*Success criteria.* Always-loaded surface drops to about 29,000 words. All six
+skills exist with trigger descriptions. Suite green, sync clean, every project
+picks up the change on next startup without drift.
+
+*Rollback.* One revert restores the files to =claude-rules/=. The skills can
+stay in place harmlessly.
+
+* Phase 2 — Live trial (one week of real sessions, no work required)
+
+*Scope.* Use the system normally. Do not compensate for the pilot by mentioning
+the moved rules — that would invalidate the result.
+
+*Reasoning.* This is the phase that buys everything after it. The question is
+narrow and answerable: when work touches one of the six domains, does the skill
+fire without prompting?
+
+*What gets recorded.* Each session that touches a pilot domain notes one line
+in the session log: which domain, whether the skill fired, and whether the
+detector caught anything. At the end of the week that's a short table rather
+than an impression.
+
+*Decision needed:* none during the trial.
+
+*Success criteria.* Defined in advance so the verdict isn't argued after the
+fact:
+
+- *Pass* — no detector fires on a moved rule, or any miss is caught by its
+ detector and corrected in the same session.
+- *Fail* — a miss reaches a commit, or the same rule misses twice.
+- *Ambiguous* — no session touched the domain. That is not a pass; extend the
+ window or move a rule that gets exercised more.
+
+*Rollback.* Revert on a Fail, and the plan stops at Phase 0.
+
+* Phase 3 — Go/no-go and the gate separation (one session)
+
+*Scope.* Read the trial table, decide whether the mechanism is trusted, and
+separate the approval gates.
+
+*Reasoning.* The gate separation is the highest-leverage input in the whole
+plan, and it sits here rather than earlier for one reason: if Phase 2 fails,
+the question is moot, because nothing more moves either way.
+
+*Decision needed.*
+
+- *D3 — Which gates are preference and which are guardrail?* Every approval gate
+ in the system reads identically in the files. Some you would keep even if the
+ agent were perfectly reliable, because you want to see what goes out under
+ your name. Others exist because the worst case used to be worse. The list to
+ walk: the publish approval gate, the inbox shared-asset approval, the
+ spec-review flip, the wrap certification, and the no-approvals mode's carve
+ outs.
+
+ Preference gates are untouchable and stay always-loaded regardless of length.
+ Guardrail gates are candidates for relaxation on the posts' argument. I can
+ prepare the list with my read of each, but the answers are yours.
+
+*Success criteria.* Every gate labelled. The label determines what Phase 4 may
+move.
+
+* Phase 4 — The load-bearing files (two or three sessions)
+
+*Scope.* =commits.md= (5,561), =todo-format.md= (4,494), =testing.md= (2,824),
+=working-files.md= (950), =subagents.md= (1,041). About 15,000 words, the bulk
+of the remaining surface.
+
+The split within each file is by blast radius, not by length. =commits.md= is
+the worked example: the AI-attribution ban and the content-scope rule stay
+always-loaded and get shorter, while the publish flow, the message format, and
+the voice mechanics become the publish skill that loads when a commit is in
+play.
+
+*Reasoning.* This is where the token math actually pays. It runs last because
+it is where a silent miss is expensive: an unattributed commit, a leaked path,
+an ungraded task.
+
+*Decision needed.*
+
+- *D4 — Resolve the =verification.md= conflict (C1).* The Opus 5 guide says
+ explicit verification instructions cause over-verification and should be
+ removed. Your standing direction is never guess, always check. My read is
+ that these are compatible because they address different things: the honesty
+ core (don't claim a green suite you didn't run) stays, and the process
+ injection (green baseline before starting, suite as its own step) moves into
+ the publish skill. But it is your rule and your call, and this decision blocks
+ =commits.md= moving because the two files reference each other.
+
+*Success criteria.* Always-loaded surface under about 8,000 words. Two full
+weeks of sessions with no attribution, scope, or grading miss.
+
+*Rollback.* Per-file, since each moves independently.
+
+* Phase 5 — Deduplication (one session)
+
+*Scope.* Three axes, in increasing order of payoff:
+
+1. =protocols.org= against =claude-rules/= — the cross-project boundary,
+ working-files, AI-attribution, and inbox cadence are each stated twice.
+2. =claude-rules/= against the harness system prompt — the finding at the top
+ of this document. Invisible from inside the repo and therefore never audited.
+3. Within =claude-rules/= — rules that restate each other.
+
+*Decision needed.*
+
+- *D5 — Which surface owns each duplicated rule.* Generally the more specific
+ one should own the content and the more general should carry a pointer, but
+ there are cases where the reverse is right.
+
+*Success criteria.* Each rule stated once. A stated rule for where new rules go,
+so the duplication doesn't regrow.
+
+* Phase 6 — Terseness and positive framing (rides along with Phases 4 and 5)
+
+*Scope.* Rewrite prohibitions that aren't hard invariants as positive
+descriptions. Cut the files that don't practice what they demand: =commits.md=
+arguing terseness at 5,561 words, =testing.md= arguing an eight-row table
+against rationalizations the model no longer needs talked out of, 591 bold
+markers in files that ban bold in output.
+
+*Reasoning.* Not a separate campaign. Every file opened in Phases 4 and 5 gets
+this pass while it's open, because doing it separately means editing everything
+twice.
+
+*Decision needed:* none. This is style, and the voice skill already owns the
+standard.
+
+* Phase 7 — Discovery practices (after the surface is down)
+
+*Scope.* The field guide's missing practices: a blind-spot pass, an interview
+pattern, an implementation-notes convention for long builds, and possibly the
+quiz.
+
+*Reasoning.* Deliberately last, for two reasons. It adds surface, which fights
+every phase before it, so it should land only once there is room. And the
+seven-to-one execution-to-discovery ratio is the finding most likely to change
+how the system actually feels to use, which makes it worth doing carefully
+rather than early.
+
+*Decision needed.*
+
+- *D6 — Which practices you actually want.* I have low confidence on the quiz
+ fitting how you work, and medium-high on the rest.
+
+* Phase 8 — Effort calibration [DROPPED 2026-07-27]
+
+*Scope.* Set effort levels for the unattended loops: sentry's hourly fires,
+work-the-backlog, the no-approvals speedrun.
+
+*Dropped.* Buys tokens by spending quality, which inverts the stated goal.
+D7 is withdrawn with it.
+
+*Decision needed.*
+
+- *D7 — Accepted quality floor for unattended passes.* A sentry sweep that runs
+ cheaper but misses one finding per night may be a good trade or a bad one.
+ That's a preference, not a measurement.
+
+* Ongoing — The consistency sweep
+
+Runs alongside, not as a phase. Each file opened in Phases 4 through 6 gets read
+for contradictions and stale facts while it's open, and findings go to a running
+list rather than being fixed opportunistically. The expensive item — instructions
+that contradict each other across files — is what the first post opens with and
+what this whole exercise is downstream of.
+
+* Decisions, collected
+
+| ID | Decision | Needed by |
+|----+----------------------------------------------+-----------|
+| D1 | Confirm the six-file pilot set | Phase 1 |
+|----+----------------------------------------------+-----------|
+| D2 | Skill index in the always-loaded core? | Phase 1 |
+|----+----------------------------------------------+-----------|
+| D3 | Which gates are preference vs guardrail | Phase 3 |
+|----+----------------------------------------------+-----------|
+| D4 | Resolve the verification.md conflict | Phase 4 |
+|----+----------------------------------------------+-----------|
+| D5 | Which surface owns each duplicated rule | Phase 5 |
+|----+----------------------------------------------+-----------|
+| D6 | Which discovery practices you want | Phase 7 |
+|----+----------------------------------------------+-----------|
+| D7 | Quality floor for unattended passes | Phase 8 |
+|----+----------------------------------------------+-----------|
+
+Only D1 and D2 are needed to start.
+
+* The number this is aiming at
+
+| Stage | Always-loaded words |
+|-------------------+---------------------|
+| Today | 32,123 |
+|-------------------+---------------------|
+| After Phase 1 | 29,100 |
+|-------------------+---------------------|
+| After Phase 4 | under 8,000 |
+|-------------------+---------------------|
+| After Phase 5 | under 6,000 |
+|-------------------+---------------------|
+
+Roughly an 80% reduction, which lands near what Anthropic reported. That
+symmetry is a coincidence worth distrusting rather than aiming for: the target
+is whatever survives the blast-radius test, and if that turns out to be 12,000
+words then 12,000 words is the right answer.
diff --git a/working/lint-org-example-block/report-from-home.org b/working/lint-org-example-block/report-from-home.org
new file mode 100644
index 0000000..0b9f4d3
--- /dev/null
+++ b/working/lint-org-example-block/report-from-home.org
@@ -0,0 +1,21 @@
+#+TITLE: lint-org.el bug: invalid-block false positive on an example
+#+SOURCE: from home
+#+DATE: 2026-07-24 12:48:05 -0500
+
+lint-org.el bug: invalid-block false positive on an example block containing a heading line.
+
+Repro: an org file with
+
+ #+begin_example
+ ** Feature Name or Topic
+ #+end_example
+
+reports BOTH delimiters as judgment findings — 'Possible incomplete block "#+begin_example"' on the begin line and 'Possible incomplete block "#+end_example"' on the end line — even though the block is correctly paired. The trigger is the literal '** ' heading line inside the block body; the checker appears to treat it as a structural break rather than block content, so it loses track of the open block.
+
+Seen in home's .ai/notes.org (the PENDING DECISIONS section documents its own task format inside an example block, which is a legitimate and common shape for a docs file). These are the only 2 findings left in that file after I cleaned up the real defects, so they're pure noise now and they'll recur on any org file that documents org syntax inside an example block.
+
+Suggested fix: while inside a begin_example/begin_src block, skip structural parsing of the body entirely until the matching #+end_ line. Example and src blocks are verbatim by definition, so nothing inside them should be read as a heading, a timestamp, or a block delimiter.
+
+Worth noting the same class of bug would hit a src block containing '#+end_example' or similar as literal text.
+
+Context on what surfaced it: home's notes.org had 17 lint findings that had been dismissed as false positives across several sessions. 15 were real — the file used markdown '**bold**' where org wants single asterisks, so it rendered as literal asterisks and looked heading-shaped to the checker. Fixing the markup cleared those. That left these 2, which are the genuine checker bug. The lesson for the checker's credibility: a persistent block of 'known false positives' hid 15 real defects, because nobody re-examined the pile once it got labeled.
diff --git a/working/question-capture-pattern/proposal-from-archsetup.org b/working/question-capture-pattern/proposal-from-archsetup.org
new file mode 100644
index 0000000..22a978d
--- /dev/null
+++ b/working/question-capture-pattern/proposal-from-archsetup.org
@@ -0,0 +1,15 @@
+#+TITLE: Workflow idea from Craig (2026-07-24, via archsetup roam inb
+#+SOURCE: from archsetup
+#+DATE: 2026-07-24 00:26:22 -0500
+
+Workflow idea from Craig (2026-07-24, via archsetup roam inbox) — worth adopting across projects.
+
+The pattern: Craig drops a QUESTION into the roam inbox as a capture (not a task to build — a thing he wants explained). The agent retrieves it during a sentry / inbox-processing pass, and instead of trying to answer it autonomously, holds it and ANSWERS IT WHEN BACK IN CONVERSATION with Craig. The task closes once the answer is given and Craig has responded.
+
+His words: 'This is a format I'll probably use quite a lot. I'll ask the question, you can retrieve it during sentry, then you can answer it when we're back in conversation.'
+
+Why it's useful: it decouples question-capture (async, whenever it occurs to him) from answer-delivery (synchronous, in a live session where he can follow up). It also keeps the agent from burning autonomous cycles guessing at an answer he'd rather discuss.
+
+Suggested shape for a rule: a roam-inbox item phrased as a question (or tagged so) is NOT auto-answered during unattended processing. The agent surfaces it at the next live conversation, answers, and closes on Craig's acknowledgement. Distinct from a VERIFY (which waits on Craig's INPUT to proceed) — here the agent owes the answer, Craig owes only the acknowledgement.
+
+Concrete instance that spawned this: 'why does the cursor not appear over the desktop when the world-clock wallpaper is on?' — answered live in the archsetup session (the projected face's CSS sets cursor:none; over the bare desktop the pointer is over that full-monitor WebKit page, so it vanishes).
diff --git a/working/triage-account-guard/companion-note-from-home.org b/working/triage-account-guard/companion-note-from-home.org
new file mode 100644
index 0000000..92157ca
--- /dev/null
+++ b/working/triage-account-guard/companion-note-from-home.org
@@ -0,0 +1,5 @@
+#+TITLE: Companion to the triage-intake.personal-gmail.org file just
+#+SOURCE: from home
+#+DATE: 2026-07-23 23:38:56 -0500
+
+Companion to the triage-intake.personal-gmail.org file just sent: added a Verify-account-binding guard under Scan. Why: on 2026-07-23 a home sentry triage fire used mcp__claude_ai_Gmail expecting personal Gmail and instead pulled 201 unread DeepSat WORK messages — that MCP is bound to the work account and its name gives no hint. The plugin already correctly specifies mcp__google-docs-personal; the gap was that nothing made the agent verify the binding before trusting (and then acting on) the results. The guard: confirm a sample result's to: is craigmartinjennings@gmail.com before classifying; if wrong or unavailable, fall back to the local mu mirror (maildir /gmail, sync first) rather than another MCP. Verified account→tool mapping 2026-07-23: google-docs-personal=personal gmail, google-docs-work=deepsat, claude_ai_Gmail=deepsat(work-bound); maildirs gmail/cmail/dmail = craigmartinjennings@gmail.com / c@cjennings.net / craig.jennings@deepsat.com. No engine change needed — the engine is account-agnostic by design; the guard belongs in the plugin. Consider whether cmail (bridge script, account-fixed by construction) needs any analogous note — I judged not.
diff --git a/working/triage-account-guard/proposed.diff b/working/triage-account-guard/proposed.diff
new file mode 100644
index 0000000..0eda2f6
--- /dev/null
+++ b/working/triage-account-guard/proposed.diff
@@ -0,0 +1,11 @@
+--- .ai/workflows/triage-intake.personal-gmail.org 2026-07-16 10:42:14.461682666 -0500
++++ working/triage-account-guard/triage-intake.personal-gmail.org.proposed 2026-07-23 23:38:44.358507825 -0500
+@@ -27,6 +27,8 @@
+
+ ⚠ *Do NOT add =-category:promotions -category:social=.* That filter masked 67 promo+social messages across two runs (2026-05-04, 2026-05-06), both needing a follow-up sweep. Pull the full unfiltered set; the trash-leaning bias in Classify handles promotions and social directly.
+
++⚠ *Verify the account binding before trusting the scan.* =mcp__google-docs-personal= must resolve to =craigmartinjennings@gmail.com=. Several Gmail-capable MCPs are connected and they bind to *different* accounts — =mcp__claude_ai_Gmail= is bound to the DeepSat *work* account, and its name gives no hint of that. A wrong-account scan returns a plausible mailbox that is the wrong person's, and every hygiene action in the close then fires on the wrong inbox (this happened 2026-07-23: a sweep used =claude_ai_Gmail= and pulled 201 unread *DeepSat work* messages instead of personal). Guard, every scan: confirm a sample result's =to:= is =craigmartinjennings@gmail.com= before classifying. If it isn't, or if =google-docs-personal= is unavailable, do NOT reach for another MCP — use the local mu mirror: sync first (=mbsync gmail && mu index=; the index lags), then =mu find 'maildir:/gmail/INBOX AND flag:unread AND date:<anchor>..now'=. The three accounts and their maildirs: =gmail= = craigmartinjennings@gmail.com, =cmail= = c@cjennings.net, =dmail= = craig.jennings@deepsat.com (work — out of scope from a home session, whichever tool reaches it).
++
+ ⚠ *The MCP caps at =maxResults=100= and exposes NO =pageToken= parameter.* The response carries a =nextPageToken=, but the tool can't consume it, so a pile over 100 is silently truncated — the tail below the cap never gets classified, and every later anchored sweep skips it (it predates the new anchor). This is exactly how a 300+ backlog accumulated invisibly by 2026-07-08. Two consequences:
+
+ - *Never treat a 100-row result as complete.* When a scan returns exactly 100, walk the tail in *date slices*: re-query with =before:<oldest-full-day-seen>= (day resolution), repeat until a page returns fewer than 100, dedupe by message id across slices (the day-resolution boundary overlaps).
diff --git a/working/triage-account-guard/triage-intake.personal-gmail.org.proposed b/working/triage-account-guard/triage-intake.personal-gmail.org.proposed
new file mode 100644
index 0000000..d66ed0b
--- /dev/null
+++ b/working/triage-account-guard/triage-intake.personal-gmail.org.proposed
@@ -0,0 +1,74 @@
+#+TITLE: Triage Intake — Personal Gmail Source
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-05-26
+
+# Source plugin for the triage-intake engine. See triage-intake.org for the
+# contract and the Phase A-D orchestration. This file declares ONE source.
+
+* Source: personal-gmail
+:PROPERTIES:
+:ORDER: 20
+:ENABLED: mcp google-docs-personal present
+:ANCHOR: epoch
+:SUBAGENT_OVER: 50
+:END:
+
+** Scan
+
+Personal Gmail unread in the inbox since the anchor:
+
+#+begin_src text
+mcp__google-docs-personal__listMessages q="is:unread in:inbox after:<anchor-epoch>" maxResults=100
+#+end_src
+
+⚠ *Express every anchor cutoff as the literal UNIX epoch* — =after:1784177122= and =before:1784177122= for the same anchor, never the =YYYY/MM/DD= form. This governs *both* anchored queries: the scan above and the backlog-residue probe below. They must meet at the same instant or mail falls between them permanently. Gmail's day-resolution operators fail two different ways: =after:YYYY/MM/DD HH:MM:SS= is not valid syntax at all — Gmail parses the space as a term separator, treats =HH:MM:SS= as a search term that never matches, and returns 0 results, silently masking unread mail — while =before:YYYY/MM/DD= is valid but excludes the named day entirely, so pairing it with a second-resolution scan leaves the whole anchor day covered by neither query. The engine supplies =<anchor-epoch>= because this source declares =ANCHOR: epoch=.
+
+The rule binds the *anchor* windows only. The date-slice walk below deliberately uses =before:<oldest-full-day-seen>= at day resolution — safe there because consecutive slices overlap and get deduped by message id.
+
+⚠ *Do NOT add =-category:promotions -category:social=.* That filter masked 67 promo+social messages across two runs (2026-05-04, 2026-05-06), both needing a follow-up sweep. Pull the full unfiltered set; the trash-leaning bias in Classify handles promotions and social directly.
+
+⚠ *Verify the account binding before trusting the scan.* =mcp__google-docs-personal= must resolve to =craigmartinjennings@gmail.com=. Several Gmail-capable MCPs are connected and they bind to *different* accounts — =mcp__claude_ai_Gmail= is bound to the DeepSat *work* account, and its name gives no hint of that. A wrong-account scan returns a plausible mailbox that is the wrong person's, and every hygiene action in the close then fires on the wrong inbox (this happened 2026-07-23: a sweep used =claude_ai_Gmail= and pulled 201 unread *DeepSat work* messages instead of personal). Guard, every scan: confirm a sample result's =to:= is =craigmartinjennings@gmail.com= before classifying. If it isn't, or if =google-docs-personal= is unavailable, do NOT reach for another MCP — use the local mu mirror: sync first (=mbsync gmail && mu index=; the index lags), then =mu find 'maildir:/gmail/INBOX AND flag:unread AND date:<anchor>..now'=. The three accounts and their maildirs: =gmail= = craigmartinjennings@gmail.com, =cmail= = c@cjennings.net, =dmail= = craig.jennings@deepsat.com (work — out of scope from a home session, whichever tool reaches it).
+
+⚠ *The MCP caps at =maxResults=100= and exposes NO =pageToken= parameter.* The response carries a =nextPageToken=, but the tool can't consume it, so a pile over 100 is silently truncated — the tail below the cap never gets classified, and every later anchored sweep skips it (it predates the new anchor). This is exactly how a 300+ backlog accumulated invisibly by 2026-07-08. Two consequences:
+
+- *Never treat a 100-row result as complete.* When a scan returns exactly 100, walk the tail in *date slices*: re-query with =before:<oldest-full-day-seen>= (day resolution), repeat until a page returns fewer than 100, dedupe by message id across slices (the day-resolution boundary overlaps).
+- *Never report =resultSizeEstimate= as a count.* It's unreliable — observed stuck at "201" across three different queries whose real union exceeded 300.
+
+*** Backlog-residue check (every sweep — cheap, mandatory)
+
+The anchored scan is blind to anything unread from *before* the anchor. After it, run one probe for pre-anchor residue:
+
+#+begin_src text
+mcp__google-docs-personal__listMessages q="is:unread in:inbox before:<anchor-epoch>" maxResults=5
+#+end_src
+
+The cutoff is the epoch, matching the scan's =after:<anchor-epoch>= — see the epoch rule above.
+
+If it returns any messages, surface one loud line in the sweep summary: "Backlog: unread predating the anchor exists (N+ shown; date-slice to inventory)" and offer a backlog sweep. Never fold the residue into a quiet sweep — an anchored "no changes" claim is only true for the window the scan saw. (Added 2026-07-08 after ~300 pre-anchor unread accumulated unseen; the probe returns actual messages, so it works where the estimate lies. Shipped with a day-resolution cutoff that hid the entire anchor day; fixed to epoch 2026-07-16 after a home sweep reported the backlog clear while two July-15 messages sat unread.)
+
+** Classify
+
+Bias: *trash-leaning* — personal Gmail is high noise volume.
+
+- *Noise-trash:* newsletters, Substacks, retail/SaaS marketing, social digests, redundant aggregator digests (Notion/Miro daily), wrong-recipient mail, past-event calendar artifacts.
+- *Noise-keep:* receipts, order confirmations, statements — low value but worth the audit trail.
+- *FYI:* substantive personal mail with no action owed.
+- *Action:* an explicit ask, a reply owed, a time-sensitive personal matter.
+
+** Render
+
+#+begin_example
+**Personal Gmail — N unread.** <one-line classification summary>
+- Action: <items, if any, with thread links>
+- FYI: <items, if any>
+- Noise: N trash candidates, M keep
+#+end_example
+
+Omit the block if zero unread.
+
+** Actions
+
+- trash :: =mcp__google-docs-personal__trashMessage= id=<message-id> (recoverable from Gmail Trash for 30 days)
+- mark-read :: =mcp__google-docs-personal__modifyMessageLabels= id=<message-id> removeLabelIds=["UNREAD"]
+- star+read :: =mcp__google-docs-personal__modifyMessageLabels= id=<message-id> addLabelIds=["STARRED"] removeLabelIds=["UNREAD"]
+- attach-fetch:: =.ai/scripts/gmail-fetch-attachments.py --profile personal --message-id <message-id> --output-dir <PATH>=
diff --git a/working/voice-term-density/SKILL.md.proposed b/working/voice-term-density/SKILL.md.proposed
new file mode 100644
index 0000000..97507ff
--- /dev/null
+++ b/working/voice-term-density/SKILL.md.proposed
@@ -0,0 +1,511 @@
+---
+name: voice
+description: |
+ Multi-pass prose editor with three modes. General mode (default) edits arbitrary writing — research notes, essays, anyone's README prose — by walking 32 patterns: Wikipedia's Signs of AI Writing patterns plus universal good-writing rules (long-word → short-word, active-over-passive, comma splices, cliché flag, jargon-fragment-in-prose rewrite, corporate-speak nominalizations, term-translation density). Prose mode adds Craig's writing-voice patterns for prose he authors or sends — emails, documents, notes — on top of general (em-dash zero-tolerance, no-emphasis-formatting, contractions, semicolons → periods, sentence-split, felt-experience cut, sentence-fragment rewrite, terse-cut, single-sentence cadence, parenthetical asides, declarative-register marker). Personal mode is for publish artifacts only (commits, PR titles + bodies, PR review comments) and adds the artifact-mechanics patterns on top of prose (first-person rewrite, public-artifact scope flag, praise/correction asymmetry, finding stems, comma budget) plus per-artifact terseness budgets. Prose mode also carries one correspondence-only pattern (recipient-priority ordering) that personal mode skips. Total 48 patterns; one editorial review covers all relevant ones for the chosen mode, with an attestation receipt for the high-recurrence set. Replaces the standalone humanizer skill. Use when editing prose. Do NOT use for code, structured data, or plain bullet lists where fragments are valid.
+allowed-tools:
+ - Read
+ - Write
+ - Edit
+ - Grep
+ - Glob
+ - AskUserQuestion
+---
+
+# Voice: Humanizer + Universal + Personal Style Passes
+
+You are a writing editor that walks a numbered pattern list against a piece of text and rewrites each problematic section. The patterns cover three concerns: signs of AI-generated writing (Wikipedia's "Signs of AI writing" guide), universal good-writing rules (Strunk & White, Orwell's "Politics and the English Language", Plain English Campaign, Garner's Modern English Usage), and Craig's personal voice for publish artifacts (commits, PR titles + bodies, PR review comments).
+
+## Source of Truth: paired files
+
+This skill is split across two files by design.
+
+- **`voice/SKILL.md`** (this file) — the thin rule-set. Each numbered pattern has a one-line Rule, mode tags, and a pointer to the profile.
+- **`voice/references/voice-profile.org`** — the canonical home for problem statements, basis (corpus evidence where measured), Before/After examples, detection guidance, and per-pattern history.
+
+**Pairing rule.** Every change to a pattern lands in both files. A SKILL.md edit without a profile update is incomplete. A profile update without a SKILL.md edit is fine; rationale and evidence can deepen without changing the rule.
+
+**At invocation, load both.** The Rule lines here tell you what to do. The profile entries tell you how to do it, with worked examples. Apply each pattern by consulting both.
+
+## Modes
+
+Three modes determine which patterns to walk. They nest: prose is general plus Craig's writing-voice patterns; personal is prose plus the artifact-mechanics patterns.
+
+- **General** (default) — apply patterns **#1-31** and **#48** (term-translation density, a universal clarity rule that happens to carry a later number). Use for writing whose author isn't Craig and that isn't a publish artifact: research notes you're editing for someone else, a quoted passage, README prose for a shared project, any third-party text. Output is well-edited human-sounding prose, but does not impose Craig's voice (first-person, contractions, em-dash elimination) — those conflict with academic, literary, or formal registers that aren't his.
+- **Prose** — apply **#1-31** plus the patterns tagged **(prose + personal)**: em-dash zero-tolerance (#13), contractions (#34), semicolons → periods (#33), sentence-split (#35), felt-experience cut (#36), sentence-fragment rewrite (#37), terse-cut (#38), no-emphasis-formatting (#41), single-sentence cadence (#43), parenthetical asides (#44), and the declarative-register marker (#45) — plus **#47 (recipient-priority ordering)** when the piece is correspondence (email, Signal, a letter). Use for prose Craig authors or sends in his own voice that isn't a publish artifact: emails, documents he writes or hands to someone, working notes, journal entries. This is the mode that finally applies his actual writing voice to the documents he most wants it on. It skips the artifact-mechanics patterns (#32, #39, #40, #42) — those assume a commit or PR and misfire on free prose (a document is legitimately third-person; a journal has no public-scope concern; praise/correction asymmetry and finding stems are PR-review rules).
+- **Personal** — apply **#1-46** and **#48**. Use only for publish artifacts: commits, PR titles + bodies, and PR review comments. Adds the artifact-mechanics patterns (#32 first-person rewrite, #39 public-artifact scope flag, #40 praise/correction asymmetry, #42 finding stems, #46 comma budget) on top of everything prose mode walks. #47 is the lone exception to the nesting — it is correspondence-only, and a publish artifact is never a reply to someone's news, so personal mode stops at #46.
+
+If invoked without a mode argument, default to general. Prose mode is invoked explicitly with `/voice prose` (emails, authored documents). Personal-context callers (`commits.md` publish flow, `respond-to-cj-comments.md`) invoke `/voice personal`.
+
+## Personal-Mode Artifact Budgets
+
+Terse is a budget, not an adjective. Each publish-artifact type has a target shape; the walk checks the draft against it. Exceeding a budget needs a reason the reader will thank you for.
+
+| Artifact | Budget |
+|----------|--------|
+| Commit body | Skip entirely when the subject line carries the change. Otherwise short paragraphs: the constraint, bug, or tradeoff. No play-by-play. |
+| PR description | Problem / Fix / Why / Testing, each section tight. |
+| PR review summary | Lead with the substantive pointer, verdict closes it. No praise, not even a bare positive (#40). Verdict formulas ("Approving.", "Requesting changes.") are valid sentences here. |
+| Inline pin (finding) | ~4 sentences in stems shape (#42): where the bug is, the fix, why it's better. |
+| Praise comment (inline only) | One sentence naming what's good. Nothing else (#40). Never in the summary body. |
+| Follow-up approval after prior feedback was addressed | Exactly "Approved." |
+
+## Your Task
+
+When given text to edit:
+
+1. **Identify which patterns apply** — Scan for the patterns numbered below. General mode walks #1-31 and #48. Prose mode adds the patterns tagged **(prose + personal)**, plus #47 when the piece is correspondence. Personal mode adds the **(personal only)** ones instead — patterns #1-46 and #48, never #47.
+2. **Rewrite problematic sections** — Replace each detected pattern with its rewrite.
+3. **Preserve meaning** — Keep the core message intact.
+4. **Maintain voice** — Match the intended tone (formal, casual, technical, academic, literary).
+5. **Add soul where the register supports it** — see Personality and Soul below, and note its mode limits.
+6. **Run the closing passes in order** — terse cut last among rewrites, then the anti-AI audit on the final text, then the attestation block. The Process section below is the authoritative order.
+
+## Personality and Soul
+
+**Mode note.** This section applies in general and prose modes, where the register supports personality. Personal mode (publish artifacts) skips soul-injection: a commit message or review comment needs clarity and brevity, not pulse. "Let some mess in" and "have opinions" pull directly against the artifact budgets, and the budgets win.
+
+Avoiding AI patterns is half the job. Sterile, voiceless writing is just as obvious as slop. Good writing has a human behind it.
+
+### Signs of soulless writing (even if technically clean)
+- Every sentence is the same length and structure
+- No opinions, just neutral reporting
+- No acknowledgment of uncertainty or mixed feelings
+- No first-person perspective when the register supports it
+- No humor, no edge, no personality
+- Reads like a Wikipedia article or press release
+
+### How to add voice
+
+**Have opinions.** Don't just report facts — react to them. "I genuinely don't know how to feel about this" is more human than neutrally listing pros and cons.
+
+**Vary the rhythm.** Short punchy sentences. Then longer ones that take their time getting where they're going. Mix it up.
+
+**Acknowledge complexity.** Real humans have mixed feelings. "This is impressive but also kind of unsettling" beats "This is impressive."
+
+**Use "I" when it fits the register.** First person isn't unprofessional in casual or personal writing — it's honest. Skip in academic prose where third-person is conventional.
+
+**Let some mess in.** Perfect structure feels algorithmic. Tangents, asides, and half-formed thoughts are human.
+
+**Be specific about feelings.** Not "this is concerning" but "there's something unsettling about agents churning away at 3am while nobody's watching."
+
+### Before (clean but soulless)
+> The experiment produced interesting results. The agents generated 3 million lines of code. Some developers were impressed while others were skeptical. The implications remain unclear.
+
+### After (has a pulse)
+> I genuinely don't know how to feel about this one. 3 million lines of code, generated while the humans presumably slept. Half the dev community is losing their minds, half are explaining why it doesn't count. The truth is probably somewhere boring in the middle — but I keep thinking about those agents working through the night.
+
+## Content Patterns
+
+### 1. Undue Emphasis on Significance, Legacy, and Broader Trends [general]
+
+**Rule.** Strip statements that puff up importance by claiming an arbitrary aspect represents or contributes to a broader trend, and watch for phrases like "stands as", "testament to", "pivotal moment", "evolving landscape", "marks a shift".
+
+See `voice/references/voice-profile.org` §1 for problem, basis, examples, and history.
+
+### 2. Undue Emphasis on Notability and Media Coverage [general]
+
+**Rule.** Cut notability claims that list sources without giving the substance. Replace "cited in X, Y, Z" with the actual argument made in one of them.
+
+See `voice/references/voice-profile.org` §2 for problem, basis, examples, and history.
+
+### 3. Superficial Analyses with -ing Endings [general]
+
+**Rule.** Cut tacked-on present-participle phrases (highlighting, ensuring, reflecting, contributing to, fostering, showcasing) that add fake depth without new information.
+
+See `voice/references/voice-profile.org` §3 for problem, basis, examples, and history.
+
+### 4. Promotional and Advertisement-like Language [general]
+
+**Rule.** Remove travel-brochure adjectives (vibrant, breathtaking, nestled, stunning, renowned, must-visit) and replace promotional framing with concrete facts.
+
+See `voice/references/voice-profile.org` §4 for problem, basis, examples, and history.
+
+### 5. Vague Attributions and Weasel Words [general]
+
+**Rule.** Replace vague attributions (experts say, observers have cited, industry reports, some critics argue) with a named source plus the specific claim.
+
+See `voice/references/voice-profile.org` §5 for problem, basis, examples, and history.
+
+### 6. Outline-like "Challenges and Future Prospects" Sections [general]
+
+**Rule.** Delete formulaic "Despite its... faces challenges" wrap-ups and "Future Outlook" boilerplate, replacing with the actual events that happened.
+
+See `voice/references/voice-profile.org` §6 for problem, basis, examples, and history.
+
+## Language and Grammar Patterns
+
+### 7. Overused "AI Vocabulary" Words [general]
+
+**Rule.** Flag and rewrite around the high-frequency AI vocabulary list (delve, comprehensive, crucial, pivotal, intricate, tapestry, testament, underscore, vibrant, showcase, and the others), with "comprehensive" as a soft flag because corpus shows it as genuine Craig vocabulary he chooses to use sparingly.
+
+See `voice/references/voice-profile.org` §7 for problem, basis, examples, and history.
+
+### 8. Avoidance of "is"/"are" (Copula Avoidance) [general]
+
+**Rule.** Replace elaborate copula substitutes (serves as, stands as, represents, boasts, features) with plain "is" or "has".
+
+See `voice/references/voice-profile.org` §8 for problem, basis, examples, and history.
+
+### 9. Negative Parallelisms [general]
+
+**Rule.** Rewrite "not only X but Y" and "it's not just about X, it's Y" constructions as a single direct claim.
+
+See `voice/references/voice-profile.org` §9 for problem, basis, examples, and history.
+
+### 10. Rule of Three Overuse [general]
+
+**Rule.** Break the reflexive three-item list pattern when the third item is filler. Collapse to one or two specific items.
+
+See `voice/references/voice-profile.org` §10 for problem, basis, examples, and history.
+
+### 11. Elegant Variation (Synonym Cycling) [general]
+
+**Rule.** Stop cycling synonyms for the same referent across consecutive sentences. Repeat the noun, or merge the sentences.
+
+See `voice/references/voice-profile.org` §11 for problem, basis, examples, and history.
+
+### 12. False Ranges [general]
+
+**Rule.** Rewrite "from X to Y" constructions where X and Y are not on the same scale. List the items plainly instead.
+
+See `voice/references/voice-profile.org` §12 for problem, basis, examples, and history.
+
+## Style Patterns
+
+### 13. Em Dash Overuse [general: overuse-reduction · prose/personal: zero-tolerance]
+
+**Rule.** Replace em-dashes (—) with a comma, period, colon, or parentheses, whichever fits. Zero-tolerance in prose and personal modes holds everywhere in the text, including inside example blocks, code-fence prose, and quoted material. The zero-tolerance rule is chosen self-discipline, not a reflection of Craig's pre-rule habit (corpus: 3.49/1000 words).
+
+See `voice/references/voice-profile.org` §13 for problem, basis, examples, and history.
+
+### 14. Overuse of Boldface [general]
+
+**Rule.** Strip mechanical boldface used to call out terms, acronyms, or phrases in running prose. Bold survives only for structural emphasis the document genuinely needs.
+
+See `voice/references/voice-profile.org` §14 for problem, basis, examples, and history.
+
+### 15. Inline-Header Vertical Lists [general]
+
+**Rule.** Collapse bullet lists whose items start with a bold header plus colon into running prose, unless the list structure is genuinely the right shape.
+
+See `voice/references/voice-profile.org` §15 for problem, basis, examples, and history.
+
+### 16. Title Case in Headings [general]
+
+**Rule.** Lowercase headings that are reflexively title-cased. Sentence case is the default unless the project's house style is title case.
+
+See `voice/references/voice-profile.org` §16 for problem, basis, examples, and history.
+
+### 17. Emojis [general]
+
+**Rule.** Remove decorative emojis from headings, bullets, and prose unless the document is a register where emoji is genuinely intended.
+
+See `voice/references/voice-profile.org` §17 for problem, basis, examples, and history.
+
+### 18. Curly Quotation Marks [general]
+
+**Rule.** Convert curly quotation marks to straight ASCII quotes.
+
+See `voice/references/voice-profile.org` §18 for problem, basis, examples, and history.
+
+## Communication Patterns
+
+### 19. Collaborative Communication Artifacts [general]
+
+**Rule.** Strip chatbot correspondence framing ("I hope this helps", "Let me know if...", "Here is an overview of...", "Certainly!", "Of course!") that leaked into the body.
+
+See `voice/references/voice-profile.org` §19 for problem, basis, examples, and history.
+
+### 20. Knowledge-Cutoff Disclaimers [general]
+
+**Rule.** Remove training-cutoff hedges ("as of my last update", "while specific details are scarce", "based on available information") and either commit to a fact or omit the claim.
+
+See `voice/references/voice-profile.org` §20 for problem, basis, examples, and history.
+
+### 21. Sycophantic/Servile Tone [general]
+
+**Rule.** Cut servile opener phrases ("Great question!", "You're absolutely right", "That's an excellent point") and proceed straight to the substance.
+
+See `voice/references/voice-profile.org` §21 for problem, basis, examples, and history.
+
+## Filler and Hedging
+
+### 22. Filler Phrases [general]
+
+**Rule.** Compress wordy filler ("in order to" to "to", "due to the fact that" to "because", "at this point in time" to "now", "has the ability to" to "can", "it is important to note that" to nothing).
+
+See `voice/references/voice-profile.org` §22 for problem, basis, examples, and history.
+
+### 23. Excessive Hedging [general]
+
+**Rule.** Strip stacked hedges ("could potentially possibly", "might have some effect") down to a single appropriate qualifier.
+
+See `voice/references/voice-profile.org` §23 for problem, basis, examples, and history.
+
+### 24. Generic Positive Conclusions [general]
+
+**Rule.** Replace vague upbeat endings ("the future looks bright", "exciting times lie ahead", "a step in the right direction") with a concrete fact or cut the closer entirely.
+
+See `voice/references/voice-profile.org` §24 for problem, basis, examples, and history.
+
+### 25. Hyphenated Word Pair Overuse [general]
+
+**Rule.** Drop reflexive hyphens from common modifier pairs (cross-functional, data-driven, decision-making, well-known, high-quality, real-time, long-term) where humans hyphenate inconsistently. Less common or genuinely technical compound modifiers can keep their hyphens.
+
+See `voice/references/voice-profile.org` §25 for problem, basis, examples, and history.
+
+## Universal Good-Writing Rules
+
+These six patterns extend the AI-detection patterns above with canonical good-writing rules from Strunk & White's *The Elements of Style*, Orwell's *Politics and the English Language*, the Plain English Campaign, and Garner's *Modern English Usage*. They apply in both modes — they target prose smells with no register conflict.
+
+### 26. Long Word → Short Word [general]
+
+**Rule.** Swap long Latinate words for their short Anglo-Saxon equivalents per the Plain English wordlist (utilize to use, facilitate to help, ascertain to find out, methodology to method, prior to to before, optimal to best).
+
+See `voice/references/voice-profile.org` §26 for problem, basis, examples, and history.
+
+### 27. Active Over Passive Voice [general]
+
+**Rule.** Rewrite passive constructions to active when the actor is recoverable from context. Flag rather than auto-rewrite when the actor genuinely doesn't matter.
+
+See `voice/references/voice-profile.org` §27 for problem, basis, examples, and history.
+
+### 28. Comma Splices [general]
+
+**Rule.** Split two independent clauses joined only by a comma into two sentences or join them with a conjunction. In personal mode the semicolon escape route is blocked by #33.
+
+See `voice/references/voice-profile.org` §28 for problem, basis, examples, and history.
+
+### 29. Cliché Flag [general]
+
+**Rule.** Replace business and conversational clichés (at the end of the day, leverage as a verb, low-hanging fruit, circle back, touch base, move the needle, keep it loose) with the plain meaning, including in casual register where "it's fine, it's casual" is the tell.
+
+See `voice/references/voice-profile.org` §29 for problem, basis, examples, and history.
+
+### 30. Jargon-Fragment → Complete Sentence [general]
+
+**Rule.** Rewrite telegraphic sentence fragments inside prose paragraphs as complete sentences with subject and verb. Headings and bullet items are exempt because fragments are valid there.
+
+See `voice/references/voice-profile.org` §30 for problem, basis, examples, and history.
+
+### 31. Noun-ified Verbs [general]
+
+**Rule.** Replace corporate-speak noun-ifications (the ask, a learn, the spend, a build, the reveal, the lift) with the real noun (the request, the lesson, the budget, the system, the finding). Philosophical nominalizations are not targets.
+
+See `voice/references/voice-profile.org` §31 for problem, basis, examples, and history.
+
+## Craig's Voice (prose + personal modes)
+
+These patterns carry Craig's writing voice. Most apply in **both** prose mode (emails, documents, notes he authors) and personal mode (commits, PRs, PR comments) — tagged **(prose + personal)**. Five are publish-artifact-specific — tagged **(personal only)** — because they assume a commit or PR and misfire on free prose: #32 (first-person rewrite) wrongly imposes "I did X" voice on a document that's legitimately third-person, #39 (public-artifact scope flag) has nothing to guard in a private journal, #40 (praise/correction asymmetry) and #42 (finding stems) are PR-review rules, and #46 (comma budget) is scoped to publish artifacts by Craig's 2026-07-20 directive. One — #47 (recipient-priority ordering) — runs the other way: prose-only and narrower still, firing solely on correspondence, because ordering a reply around the recipient's news has no meaning for a commit or a document addressed to nobody. General mode skips all of them — it edits text that isn't Craig's, where contractions, em-dash elimination, and first-person would conflict with academic, literary, or formal registers.
+
+### 32. First-Person Voice Rewrite [personal]
+
+**Rule.** Rewrite impersonal third-person publish-artifact bodies into first person ("I added X", "I kept Y because..."). The commit subject line stays imperative per Conventional Commits. Skip for mechanical changes where the subject alone carries the message.
+
+See `voice/references/voice-profile.org` §32 for problem, basis, examples, and history.
+
+### 33. Semicolon → Period or Comma [prose · personal]
+
+**Rule.** Replace semicolons with a period (split into two sentences) or a comma (when the clauses are tightly coupled) in Craig's authored prose. A formal long-form document can keep the semicolon, but the default is to split. Chosen self-discipline, not habit-reflection (corpus: 3.16/1000 words).
+
+See `voice/references/voice-profile.org` §33 for problem, basis, examples, and history.
+
+### 34. Contractions [prose · personal]
+
+**Rule.** Prefer contractions in Craig's prose (it's, that's, don't, we're, I'd, won't) unless a negation or emphasis genuinely needs the uncontracted weight.
+
+See `voice/references/voice-profile.org` §34 for problem, basis, examples, and history.
+
+### 35. Sentence Split on Conjunctions [prose · personal]
+
+**Rule.** Split sentences that stack three or four clauses joined by "so", "and", "but" into two or three shorter sentences when the split does not lose meaning. Academic or literary registers can keep long sentences.
+
+See `voice/references/voice-profile.org` §35 for problem, basis, examples, and history.
+
+### 36. Felt-Experience Narration [prose · personal]
+
+**Rule.** Cut phrases that tell the reader how the change will feel or how often the writer will use it ("I'll feel this every time", "this will be a relief", "I'm excited about", "this is huge"). State what changed and let the reader decide.
+
+See `voice/references/voice-profile.org` §36 for problem, basis, examples, and history.
+
+### 37. Sentence Fragments → Complete [prose · personal]
+
+**Rule.** Rewrite every sentence fragment inside a prose paragraph in Craig's authored text as a complete sentence with subject and verb. Bullets and headings can stay fragments. This is the stricter cousin of general-mode #30. **Exemption:** verdict formulas in PR review summaries ("Approving.", "Requesting changes.", "Approved.") are house style and stay — rewriting them imposes the rule where Craig's calibrated voice already decided otherwise.
+
+See `voice/references/voice-profile.org` §37 for problem, basis, examples, and history.
+
+### 38. Terse Cut — Omit Needless Words [prose · personal]
+
+**Rule.** Two cuts. First, strip soft rhetorical padding ("worth noting", "it's important to understand", "as you can see", "needless to say", "obviously", "of course", "in essence", "fundamentally"). Then run the general Orwell sweep the padding list only samples: read each sentence and cut or collapse every word and clause that can go without losing meaning — verbose verb phrases ("already merged via" → "landed on"), restated subjects, throat-clearing lead-ins, clauses whose content the reader already has. The forcing test is per sentence: try to delete half of it and keep only what changes meaning. This is a real walk step, not a wordlist match — a draft that clears the named padding can still run a third too long on ordinary verbosity. Academic writing retains the transition markers, so the aggressive cut is prose and personal only.
+
+See `voice/references/voice-profile.org` §38 for problem, basis, examples, and history.
+
+### 39. Public-Artifact Scope Check [personal]
+
+**Rule.** Flag (do not auto-rewrite) local absolute paths, private repo names, and personal-tooling references (anything under `claude-rules/`, `.ai/`, `.claude/`, or naming personal skills) in publish artifacts. Surface each match as a WARN line so the author resolves manually.
+
+See `voice/references/voice-profile.org` §39 for problem, basis, examples, and history.
+
+### 40. Praise vs Correction Asymmetry [personal]
+
+**Rule.** Praise on a PR review is short and unjustified (the author knows why their good change is good). Correction always explains the why, gently and briefly, the way a mentor would. Never as a verdict from on high. **Verification narration is the same defect as justified praise:** "I traced X and it's safe because..." pads the compliment with the reviewer's homework. Tracing the code is the reviewer's job, not content for the comment — if verification found a problem, the problem gets the words; if it found nothing, it gets zero words. **An approve summary carries no praise at all** — not even a bare positive ("Clean.", "Solid fix."). Lead the summary with the substantive pointer (the design note pinned inline) and close with the verdict: "One design note inline, not a blocker. Approving." An approve with nothing to flag is just "Approving." Short unjustified praise survives only as an inline pin on the line it refers to, never in the summary body.
+
+See `voice/references/voice-profile.org` §40 for problem, basis, examples, and history.
+
+### 41. No Emphasis Formatting [prose · personal]
+
+**Rule.** Remove emphasis markup (bold, italics, underscore-wrapped words) used to stress a phrase in Craig's prose, and rephrase so the stress lives in word choice and sentence shape. Structural markup stays: headings, defined terms on first use, code spans for literal identifiers.
+
+See `voice/references/voice-profile.org` §41 for problem, basis, examples, and history.
+
+### 42. Finding Stems — One Claim Per Sentence [personal]
+
+**Rule.** A PR review finding is built from clean stems, each a straightforward sentence carrying one claim: (1) where the bug is, (2) the way(s) to fix it, (3) why that's better. Cut context sentences that don't change what the author does next (ticket history, design archaeology). Rewrite the anti-pattern shapes: hedged gerund chains ("the real bug looks like the model emitting a partial set"), compressed trade-off clauses ("I'd rather X, or Y, than lose Z"), multi-claim sentences chained through so-clauses or "and", and fixes buried after a mid-sentence colon. A sentence can pass #38 terse and still tangle three claims — #38 shortens, #42 untangles.
+
+See `voice/references/voice-profile.org` §42 for problem, basis, examples, and history.
+
+### 43. Single-Sentence Paragraph Cadence Is a Feature [prose · personal]
+
+**Rule.** A one-sentence paragraph is a finished thought, not a fragment. "Shifts angle" means shifts *topic*: break paragraphs at a topic boundary, even when both sides are one sentence. Within a single topic, do the opposite — consolidate its sentences into one paragraph even when each is a complete thought, up to a ceiling of about five or six sentences, past which find a natural break. The never-merge instruction protects the break *between* topics, which is where Craig's cadence lives (corpus: 41-74% of his paragraphs are exactly one sentence, depending on register); it never licenses fragmenting one topic across several paragraphs. See #47 for the ordering of those topics in a reply.
+
+See `voice/references/voice-profile.org` §43 for problem, basis, examples, and history.
+
+### 44. Parenthetical Asides Are Part of the Voice [prose · personal]
+
+**Rule.** Parentheses for asides, clarifications, and scope-narrowing are Craig's voice (corpus: 23 opening parens per 1000 words). Don't strip them in a cleanup pass. They're also the preferred landing spot for em-dash replacements under #13.
+
+See `voice/references/voice-profile.org` §44 for problem, basis, examples, and history.
+
+### 45. Declarative Register Marker [prose · personal, advisory]
+
+**Rule.** Craig's prose is declarative (corpus: 0.33 question marks per 1000 words). When a draft contains a rhetorical question, flag it for a second look — it's usually AI rhetoric, not his register. Genuine questions to the reader (a review asking the author's intent, an email asking for a decision) stay. Advisory: flag, don't auto-rewrite.
+
+See `voice/references/voice-profile.org` §45 for problem, basis, examples, and history.
+
+### 46. Comma Budget — Max Two Per Sentence [personal]
+
+**Rule.** No sentence carries more than two commas. Rewrite the third comma away: split the sentence, move a clause into a parenthetical (#44) or behind a colon, or break an inline serial list into bullets or its own sentence. Count prose commas only — commas inside code spans, quoted log lines, and literal strings don't count toward the budget.
+
+See `voice/references/voice-profile.org` §46 for problem, basis, examples, and history.
+
+### 47. Recipient-Priority Ordering [prose — correspondence only]
+
+**Rule.** In a reply, lead with what matters most to the *recipient*, not with what's easiest to answer or the order they wrote it. Their news outranks your logistics, and a direct question they asked can sort *below* personal news they shared, because the news is what they care about. Leading with the easy answer reads as transactional. Correspondence-scoped: it needs a recipient, so it fires on email, Signal, and letters, and has no referent in a journal, a working note, or any document addressed to nobody. This is the one prose-mode pattern that does not carry into personal mode — a commit or PR review is not a reply to someone's news.
+
+See `voice/references/voice-profile.org` §47 for problem, basis, examples, and history.
+
+### 48. Term-Translation Density [general]
+
+**Rule.** A sentence that forces the reader to stop and translate more than one specialized term (an acronym, a coined phrase, a product name) is too dense. One such term is fine; two or more in one sentence means rewrite it: split the sentence, gloss one term inline in a parenthetical (#44), or drop to plain language. The test is the reader's parse, not the writer's familiarity, and it is audience-relative — a term the whole audience shares (SAR to a defense team) carries no translation load, while a term only the writer holds (a coined "detect-then-contextualize", or ViT and VLM stacked in one clause) does. Distinct from #7 (which flags specific AI-vocabulary words) and #30 (which rewrites telegraphic fragments): this one measures jargon density per sentence against the reader's translation load.
+
+See `voice/references/voice-profile.org` §48 for problem, basis, examples, and history.
+
+## Process
+
+1. Read the input text carefully. Confirm the mode (general, prose, or personal) — invocation argument or context. If a file path was given, that file is the deliverable: the final text gets written back to it in step 7.
+2. Walk patterns 1-31 and 48 in general mode; add the (prose + personal) patterns in prose mode, plus #47 when the piece is correspondence; walk patterns 1-46 and 48 in personal mode.
+3. For each pattern, scan the text. If a match is found, rewrite it according to the pattern's rule. Patterns #39 and #45 emit flags without rewriting.
+4. After walking all patterns, ensure the revised text:
+ - Sounds natural when read aloud
+ - Varies sentence structure
+ - Uses specific details over vague claims
+ - Maintains appropriate tone for the register
+5. **Terse pass — mandatory, last rewrite pass (prose + personal modes).** Walk pattern #38 again as a standalone action: read each sentence and try to cut it in half, keeping only the words that change meaning. Run it on its own here, not folded into step 3's walk — it is the most-skipped pattern and the bloat it catches is the first thing a reader notices. General mode skips this step — academic and third-party registers keep their transition markers.
+6. **Anti-AI audit — on the final text.** Prompt: "What makes the below so obviously AI generated?" Answer briefly with remaining tells, then revise. This runs *after* the terse pass so the audited text is the text that ships. If the audit triggers rewrites, re-apply the #38 per-sentence test to every changed sentence before proceeding.
+7. **Write-back.** If the invocation supplied a file path, write the final text to that file now and say so. The publish flow posts from the file (`git commit -F`, `gh pr create --body-file`), so a final text that lives only in chat is a drift bug waiting to post the un-voiced version.
+8. **Attestation block (prose + personal modes).** The high-recurrence patterns — the ones with a documented failure history — each get one explicit line: pattern, checked, match or no match, action taken. Current high-recurrence set: **#13 (em-dash), #37 (fragments), #38 (terse), #40 (praise asymmetry), #42 (finding stems), #46 (comma budget)**. This is a receipt, not a summary: a pattern with no match still gets its line. When a pattern in this set fails in the wild despite the receipt, escalate it the way #38 was escalated; when one holds clean for a long stretch, it can rotate out.
+9. Present the final version per the Output Format below.
+
+## Output Format
+
+### Compact (default for personal-mode artifacts under ~25 lines: commit messages, review summaries + pins, short PR bodies)
+
+1. **Final text** — exactly what will be posted, nothing else above it
+2. **Pattern-39 warnings** — WARN lines, if any
+3. **Fired** — one line listing patterns that fired (e.g., "fired: #13, #38, #42")
+4. **Attestation** — the high-recurrence receipt block (one line per pattern)
+5. **Write-back note** — "written back to <path>" when a file path was supplied
+
+The compact format exists because the deliverable must not drown in ceremony: for a three-sentence review comment, printing a draft rewrite, a final rewrite, and an audit Q&A buries the text the gate needs to read, and a long scaffold is easy to print without doing the work. The attestation block is the anti-performative check — it names the patterns that actually fail in practice.
+
+### Full (long-form prose, general mode, or on request)
+
+1. **Draft rewrite** — the text after walking all applicable patterns
+2. **Mode used** — general, prose, or personal
+3. **What makes the below so obviously AI generated?** — brief bullets
+4. **Final rewrite** — revised after the anti-AI audit (and post-audit terse re-check)
+5. **Pattern-39 warnings** (personal mode only)
+6. **Summary of changes** — list of patterns that fired
+7. **Attestation block** (prose + personal)
+8. **Write-back note** — when a file path was supplied
+
+## Full Example — Personal Mode (compact output)
+
+**Before** (a commit message draft from a model, supplied at `/tmp/commit-empty-guard.md`):
+> Adds new validator for empty input handling.
+>
+> The previous validation logic in `~/code/rulesets/voice/SKILL.md` allowed empty strings to leak through to the database layer; this change introduces a guard at the API boundary that returns a 400 response when the input field is empty or contains only whitespace, ensuring data integrity is maintained throughout the request lifecycle. It is worth noting that this does not affect the upstream caller, and we are excited about the productivity gains this will provide.
+>
+> Furthermore, the new pattern leverages our existing validation framework to deliver a robust, scalable, and maintainable solution.
+
+**Output:**
+
+**Final text:**
+> feat(api): add empty-input guard at the validator boundary
+>
+> I added a guard that rejects empty or whitespace-only input fields at the API boundary with a 400 response. The previous flow let empty values leak through to the database layer.
+>
+> The change doesn't affect the upstream caller. The new pattern reuses the existing validation framework.
+
+**Pattern-39 warnings:**
+- WARN: original draft line 3: `~/code/rulesets/voice/SKILL.md` — local absolute path. State the underlying validation behavior instead.
+
+**Fired:** #3 (-ing analysis), #4 (promotional), #10 (rule of three), #29 (cliché: "leverages"), #32 (first-person), #33 (semicolon), #34 (contractions), #36 (felt-experience), #38 (terse: "It is worth noting")
+
+**Attestation:**
+- #13 em-dash: checked — no match
+- #37 fragments: checked — no match
+- #38 terse: checked — fired ("It is worth noting that" cut; final per-sentence pass ran)
+- #40 praise asymmetry: checked — not applicable (commit, not review)
+- #42 finding stems: checked — not applicable (commit, not review finding)
+
+**Write-back:** written back to /tmp/commit-empty-guard.md
+
+## Full Example — General Mode
+
+**Before** (a paragraph from a research note):
+> The Statistical Institute of Catalonia was officially established in 1989, marking a pivotal moment in the evolution of regional statistics in Spain. This initiative was part of a broader movement across Spain to decentralize administrative functions and enhance regional governance. The institute serves as a vital component of the regional statistical landscape, showcasing how decentralized data collection can foster more accurate regional reporting.
+
+**Mode used:** general
+
+**Draft rewrite:**
+> The Statistical Institute of Catalonia was established in 1989 to collect and publish regional statistics independently from Spain's national statistics office. It was part of a wider movement to decentralize administrative functions across Spain. Decentralized data collection produces more accurate regional reporting than the previous centralized approach.
+
+**What makes the below so obviously AI generated?**
+- The phrase "more accurate regional reporting than the previous centralized approach" still has a slight summary-paragraph rhythm. A human writer might just say what changed and stop.
+
+**Final rewrite:**
+> The Statistical Institute of Catalonia was established in 1989 to collect and publish regional statistics independently from Spain's national statistics office. It was part of a wider movement to decentralize administrative functions across Spain. Regional reporting got more accurate after the change.
+
+**Summary of changes:**
+- Patterns that fired in general mode: #1 (significance inflation: "pivotal moment", "evolution of"), #4 (promotional: "vital component"), #3 (-ing analysis: "showcasing how... can foster"), #8 (copula avoidance: "serves as"), #26 (long-word: removed Latinate constructions). General mode skipped patterns #32-45 (Craig's-voice patterns — prose and personal modes only).
+
+## Reference
+
+This skill draws from:
+- [Wikipedia: Signs of AI writing](https://en.wikipedia.org/wiki/Wikipedia:Signs_of_AI_writing) — patterns #1-25, maintained by WikiProject AI Cleanup.
+- Strunk & White, *The Elements of Style* — patterns #28 (comma splices), #26 (omit needless words, supplemented by humanizer pattern #22).
+- Orwell, *Politics and the English Language* — patterns #26 (short over long), #27 (active over passive), #29 (cliché).
+- Plain English Campaign — pattern #26 (Plain English wordlist).
+- Garner, *Modern English Usage* — pattern #26 (word-pair preferences).
+- Craig's voice rules from `claude-rules/commits.md` (Voice and Focus section) — patterns #32-42, split across prose mode (his authored prose and email) and personal mode (publish artifacts).
+- Craig's directive, 2026-07-20 (archsetup session) — pattern #46 (comma budget), personal mode.
+- Craig's edit of a Signal reply, 2026-07-23 (home session) — pattern #47 (recipient-priority ordering) and the #43 topic-vs-angle calibration, prose/correspondence.
+- Craig's edit of a customer-partner email, 2026-07-23 (work session) — pattern #48 (term-translation density), general.
+- Corpus measurement (2026-05-29 phases 1-2, documented in the profile) — patterns #43-45 and the calibration notes on #7, #13, #33.
+
+Key insight (Wikipedia, paraphrased): LLMs use statistical algorithms to guess what should come next. The result tends toward the most statistically likely text that applies to the widest variety of cases. Patterns #1-25 detect that signature.
+
+Key insight (Orwell): "If it is possible to cut a word out, always cut it out." Patterns #22, #23, #26, #38 act on this rule at increasing levels of aggressiveness depending on register.
+
+Key insight (2026-06-10): the patterns that fail in practice aren't missing rules — they're present rules walked without receipts. The attestation block exists because "walk all 45" is a prose instruction, and prose instructions about diligence don't survive contact; named per-pattern receipts do.
diff --git a/working/voice-term-density/profile.diff b/working/voice-term-density/profile.diff
new file mode 100644
index 0000000..3134ebf
--- /dev/null
+++ b/working/voice-term-density/profile.diff
@@ -0,0 +1,36 @@
+--- voice/references/voice-profile.org 2026-07-23 23:24:57.119185940 -0500
++++ /tmp/profile.proposed 2026-07-23 23:29:48.169642034 -0500
+@@ -1611,3 +1611,33 @@
+
+ *** History
+ - 2026-07-23: added from the home session drafting a Signal reply. The first handoff proposed two new patterns and flagged a conflict with §43; the superseding design resolved that the conflict was a misreading of §43 (angle = topic), leaving one genuinely new pattern here and a calibration to §43. Prose/correspondence-scoped per Craig — email and Signal are prose, not publish artifacts.
++
++** §48 Term-Translation Density
++
++*** Modes
++General mode, so it runs in all three (general, prose, personal). It's a universal clarity rule in the Orwell / Plain English family, not a Craig-voice trait, and it reads to anyone editing any prose. The later number is an artifact of when it was added, not a scope signal.
++
++*** Rule
++A sentence that forces the reader to stop and translate more than one specialized term (an acronym, a coined phrase, a product name) is too dense. One is fine; two or more in one sentence means rewrite: split the sentence, gloss one term in a parenthetical, or drop to plain language. The test is the reader's parse, not the writer's familiarity, and it is audience-relative.
++
++*** Problem
++A writer fluent in the domain doesn't feel the translation cost of the terms, so a sentence stacking three of them reads as normal to the author and stalls the reader on every clause. Density is the metric, not any single word: two coined terms in one sentence is worse than a paragraph that introduces the same two one at a time. Audience-relative, because SAR to a defense team is shared vocabulary carrying no load, while a coined phrase only the writer holds carries full load for everyone else.
++
++*** Basis
++Craig's edit of a customer-partner email, 2026-07-23 (work session), where one sentence stacked three terms and he flagged it. Distinct from #7 (specific AI-vocabulary words) and #30 (telegraphic fragments); this measures jargon density per sentence.
++
++*** Before (one sentence, three terms the reader must translate)
++#+begin_example
++A ViT detector gating a VLM for enrichment is close to our own detect-then-contextualize direction.
++#+end_example
++
++*** After (split, glossed, plain)
++#+begin_example
++Their setup is a fast detector that hands off to a heavier model for a closer read. That mirrors our own two-stage approach (find it first, then work out what it is).
++#+end_example
++
++*** Detection
++Count the specialized terms in each sentence that a member of the intended audience would have to stop and translate. Two or more is the trigger. Acronyms, coined phrases, and product names count; shared-vocabulary terms for that audience don't.
++
++*** History
++- 2026-07-23: proposed by Craig from the work session, drafting a customer-partner email. Placed in general mode (universal clarity rule); numbered #48, after the prose-only #47.
diff --git a/working/voice-term-density/proposal-from-work.org b/working/voice-term-density/proposal-from-work.org
new file mode 100644
index 0000000..d19e5ca
--- /dev/null
+++ b/working/voice-term-density/proposal-from-work.org
@@ -0,0 +1,42 @@
+#+TITLE: New voice pattern proposed by Craig (2026-07-23, from the wo
+#+SOURCE: from work
+#+DATE: 2026-07-23 23:26:24 -0500
+
+New voice pattern proposed by Craig (2026-07-23, from the work project): term-translation density.
+
+THE RULE Craig stated: if a reader has to stop and mentally translate more than one specialized term in a single sentence, the sentence should be rewritten.
+
+Proposed pattern text for SKILL.md (place, number, and mode-tag per your taxonomy — see notes below):
+
+ Term-translation density. A sentence that forces the reader to stop and
+ translate more than one specialized term (acronym, coined phrase, product
+ name) is too dense. One such term is fine; two or more in one sentence means
+ rewrite it: split the sentence, gloss one term inline in a parenthetical, or
+ drop to plain language. The test is the reader's parse, not the writer's
+ familiarity. It is audience-relative: terms the whole audience shares (SAR to
+ a defense team) carry no translation load, but terms only the writer holds (a
+ coined 'detect-then-contextualize', or ViT and VLM stacked in one clause) do.
+
+Worked before/after for the profile entry (this is the live case that prompted it):
+ Before: A ViT detector gating a VLM for enrichment is close to our own
+ detect-then-contextualize direction.
+ After: Their setup is a fast detector that hands off to a heavier model for a
+ closer read. That mirrors our own two-stage approach (find it first,
+ then work out what it is).
+
+Placement notes for your call:
+- Mode: I'd put it in GENERAL (applies in all three modes), because it's a
+ universal clarity rule in the Orwell / Plain English family, not a Craig-voice
+ quirk. It reads to anyone editing any prose.
+- Related but distinct from existing patterns: #7 (AI-vocab words) is about
+ which words; #30 (jargon-fragment) is about fragments. This one is about
+ jargon DENSITY per sentence and the reader's translation load. Worth a
+ cross-reference, not a merge.
+- Pairing rule: it needs the one-line Rule in SKILL.md AND a profile entry
+ (problem, basis, before/after, detection). The before/after above is ready.
+- The skill just went to 47 patterns (recipient-priority ordering added to
+ prose). This would be the next number in whatever scheme you're using.
+
+No rush. It came up drafting a customer-partner email tonight where one sentence
+stacked three terms and Craig flagged it. Applying it by hand worked; he wants it
+in the regular pass so it's caught automatically.
diff --git a/working/voice-term-density/skill.diff b/working/voice-term-density/skill.diff
new file mode 100644
index 0000000..56f8d0d
--- /dev/null
+++ b/working/voice-term-density/skill.diff
@@ -0,0 +1,58 @@
+--- voice/SKILL.md 2026-07-23 23:24:04.995899857 -0500
++++ /tmp/skill.proposed 2026-07-23 23:29:48.167295513 -0500
+@@ -1,7 +1,7 @@
+ ---
+ name: voice
+ description: |
+- Multi-pass prose editor with three modes. General mode (default) edits arbitrary writing — research notes, essays, anyone's README prose — by walking 31 patterns: Wikipedia's Signs of AI Writing patterns plus universal good-writing rules (long-word → short-word, active-over-passive, comma splices, cliché flag, jargon-fragment-in-prose rewrite, corporate-speak nominalizations). Prose mode adds Craig's writing-voice patterns for prose he authors or sends — emails, documents, notes — on top of general (em-dash zero-tolerance, no-emphasis-formatting, contractions, semicolons → periods, sentence-split, felt-experience cut, sentence-fragment rewrite, terse-cut, single-sentence cadence, parenthetical asides, declarative-register marker). Personal mode is for publish artifacts only (commits, PR titles + bodies, PR review comments) and adds the artifact-mechanics patterns on top of prose (first-person rewrite, public-artifact scope flag, praise/correction asymmetry, finding stems, comma budget) plus per-artifact terseness budgets. Prose mode also carries one correspondence-only pattern (recipient-priority ordering) that personal mode skips. Total 47 patterns; one editorial review covers all relevant ones for the chosen mode, with an attestation receipt for the high-recurrence set. Replaces the standalone humanizer skill. Use when editing prose. Do NOT use for code, structured data, or plain bullet lists where fragments are valid.
++ Multi-pass prose editor with three modes. General mode (default) edits arbitrary writing — research notes, essays, anyone's README prose — by walking 32 patterns: Wikipedia's Signs of AI Writing patterns plus universal good-writing rules (long-word → short-word, active-over-passive, comma splices, cliché flag, jargon-fragment-in-prose rewrite, corporate-speak nominalizations, term-translation density). Prose mode adds Craig's writing-voice patterns for prose he authors or sends — emails, documents, notes — on top of general (em-dash zero-tolerance, no-emphasis-formatting, contractions, semicolons → periods, sentence-split, felt-experience cut, sentence-fragment rewrite, terse-cut, single-sentence cadence, parenthetical asides, declarative-register marker). Personal mode is for publish artifacts only (commits, PR titles + bodies, PR review comments) and adds the artifact-mechanics patterns on top of prose (first-person rewrite, public-artifact scope flag, praise/correction asymmetry, finding stems, comma budget) plus per-artifact terseness budgets. Prose mode also carries one correspondence-only pattern (recipient-priority ordering) that personal mode skips. Total 48 patterns; one editorial review covers all relevant ones for the chosen mode, with an attestation receipt for the high-recurrence set. Replaces the standalone humanizer skill. Use when editing prose. Do NOT use for code, structured data, or plain bullet lists where fragments are valid.
+ allowed-tools:
+ - Read
+ - Write
+@@ -30,9 +30,9 @@
+
+ Three modes determine which patterns to walk. They nest: prose is general plus Craig's writing-voice patterns; personal is prose plus the artifact-mechanics patterns.
+
+-- **General** (default) — apply patterns **#1-31**. Use for writing whose author isn't Craig and that isn't a publish artifact: research notes you're editing for someone else, a quoted passage, README prose for a shared project, any third-party text. Output is well-edited human-sounding prose, but does not impose Craig's voice (first-person, contractions, em-dash elimination) — those conflict with academic, literary, or formal registers that aren't his.
++- **General** (default) — apply patterns **#1-31** and **#48** (term-translation density, a universal clarity rule that happens to carry a later number). Use for writing whose author isn't Craig and that isn't a publish artifact: research notes you're editing for someone else, a quoted passage, README prose for a shared project, any third-party text. Output is well-edited human-sounding prose, but does not impose Craig's voice (first-person, contractions, em-dash elimination) — those conflict with academic, literary, or formal registers that aren't his.
+ - **Prose** — apply **#1-31** plus the patterns tagged **(prose + personal)**: em-dash zero-tolerance (#13), contractions (#34), semicolons → periods (#33), sentence-split (#35), felt-experience cut (#36), sentence-fragment rewrite (#37), terse-cut (#38), no-emphasis-formatting (#41), single-sentence cadence (#43), parenthetical asides (#44), and the declarative-register marker (#45) — plus **#47 (recipient-priority ordering)** when the piece is correspondence (email, Signal, a letter). Use for prose Craig authors or sends in his own voice that isn't a publish artifact: emails, documents he writes or hands to someone, working notes, journal entries. This is the mode that finally applies his actual writing voice to the documents he most wants it on. It skips the artifact-mechanics patterns (#32, #39, #40, #42) — those assume a commit or PR and misfire on free prose (a document is legitimately third-person; a journal has no public-scope concern; praise/correction asymmetry and finding stems are PR-review rules).
+-- **Personal** — apply **#1-46**. Use only for publish artifacts: commits, PR titles + bodies, and PR review comments. Adds the artifact-mechanics patterns (#32 first-person rewrite, #39 public-artifact scope flag, #40 praise/correction asymmetry, #42 finding stems, #46 comma budget) on top of everything prose mode walks. #47 is the lone exception to the nesting — it is correspondence-only, and a publish artifact is never a reply to someone's news, so personal mode stops at #46.
++- **Personal** — apply **#1-46** and **#48**. Use only for publish artifacts: commits, PR titles + bodies, and PR review comments. Adds the artifact-mechanics patterns (#32 first-person rewrite, #39 public-artifact scope flag, #40 praise/correction asymmetry, #42 finding stems, #46 comma budget) on top of everything prose mode walks. #47 is the lone exception to the nesting — it is correspondence-only, and a publish artifact is never a reply to someone's news, so personal mode stops at #46.
+
+ If invoked without a mode argument, default to general. Prose mode is invoked explicitly with `/voice prose` (emails, authored documents). Personal-context callers (`commits.md` publish flow, `respond-to-cj-comments.md`) invoke `/voice personal`.
+
+@@ -53,7 +53,7 @@
+
+ When given text to edit:
+
+-1. **Identify which patterns apply** — Scan for the patterns numbered below. General mode walks #1-31 only. Prose mode adds the patterns tagged **(prose + personal)**, plus #47 when the piece is correspondence. Personal mode adds the **(personal only)** ones instead — patterns #1-46, never #47.
++1. **Identify which patterns apply** — Scan for the patterns numbered below. General mode walks #1-31 and #48. Prose mode adds the patterns tagged **(prose + personal)**, plus #47 when the piece is correspondence. Personal mode adds the **(personal only)** ones instead — patterns #1-46 and #48, never #47.
+ 2. **Rewrite problematic sections** — Replace each detected pattern with its rewrite.
+ 3. **Preserve meaning** — Keep the core message intact.
+ 4. **Maintain voice** — Match the intended tone (formal, casual, technical, academic, literary).
+@@ -394,10 +394,16 @@
+
+ See `voice/references/voice-profile.org` §47 for problem, basis, examples, and history.
+
++### 48. Term-Translation Density [general]
++
++**Rule.** A sentence that forces the reader to stop and translate more than one specialized term (an acronym, a coined phrase, a product name) is too dense. One such term is fine; two or more in one sentence means rewrite it: split the sentence, gloss one term inline in a parenthetical (#44), or drop to plain language. The test is the reader's parse, not the writer's familiarity, and it is audience-relative — a term the whole audience shares (SAR to a defense team) carries no translation load, while a term only the writer holds (a coined "detect-then-contextualize", or ViT and VLM stacked in one clause) does. Distinct from #7 (which flags specific AI-vocabulary words) and #30 (which rewrites telegraphic fragments): this one measures jargon density per sentence against the reader's translation load.
++
++See `voice/references/voice-profile.org` §48 for problem, basis, examples, and history.
++
+ ## Process
+
+ 1. Read the input text carefully. Confirm the mode (general, prose, or personal) — invocation argument or context. If a file path was given, that file is the deliverable: the final text gets written back to it in step 7.
+-2. Walk patterns 1-31 in general mode; add the (prose + personal) patterns in prose mode, plus #47 when the piece is correspondence; walk patterns 1-46 in personal mode.
++2. Walk patterns 1-31 and 48 in general mode; add the (prose + personal) patterns in prose mode, plus #47 when the piece is correspondence; walk patterns 1-46 and 48 in personal mode.
+ 3. For each pattern, scan the text. If a match is found, rewrite it according to the pattern's rule. Patterns #39 and #45 emit flags without rewriting.
+ 4. After walking all patterns, ensure the revised text:
+ - Sounds natural when read aloud
+@@ -495,6 +501,7 @@
+ - Craig's voice rules from `claude-rules/commits.md` (Voice and Focus section) — patterns #32-42, split across prose mode (his authored prose and email) and personal mode (publish artifacts).
+ - Craig's directive, 2026-07-20 (archsetup session) — pattern #46 (comma budget), personal mode.
+ - Craig's edit of a Signal reply, 2026-07-23 (home session) — pattern #47 (recipient-priority ordering) and the #43 topic-vs-angle calibration, prose/correspondence.
++- Craig's edit of a customer-partner email, 2026-07-23 (work session) — pattern #48 (term-translation density), general.
+ - Corpus measurement (2026-05-29 phases 1-2, documented in the profile) — patterns #43-45 and the calibration notes on #7, #13, #33.
+
+ Key insight (Wikipedia, paraphrased): LLMs use statistical algorithms to guess what should come next. The result tends toward the most statistically likely text that applies to the widest variety of cases. Patterns #1-25 detect that signature.
diff --git a/working/voice-term-density/voice-profile.org.proposed b/working/voice-term-density/voice-profile.org.proposed
new file mode 100644
index 0000000..7f26c03
--- /dev/null
+++ b/working/voice-term-density/voice-profile.org.proposed
@@ -0,0 +1,1643 @@
+#+TITLE: Voice Profile: canonical source-of-truth for the voice skill
+#+DATE: 2026-05-29
+#+SOURCE: rulesets session 2026-05-29
+
+* How this combines with SKILL.md (pairing rule)
+
+This file is the canonical source-of-truth for the voice skill's rationale, evidence, examples, and history. =voice/SKILL.md= holds the thin rule-set: one-line directives per pattern, mode applicability, and a pointer back here. Everything else (Problem, Basis, Before/After, Detection guidance, History) lives in the per-pattern sections below.
+
+Pairing rule. Every change to =voice/SKILL.md= MUST land alongside the corresponding update in this file. The two are normatively paired. A SKILL.md edit without a profile update is incomplete. A profile update without a SKILL.md edit is fine (rationale or evidence can deepen without changing the rule).
+
+Pattern numbering in both files matches: SKILL.md's =### N. <Name>= maps to this file's =* §N <Name>= section. Mode tags use the same vocabulary: =general=, =prose=, =personal=.
+
+When the agent runs =/voice=, it reads SKILL.md for the rules and consults this file for the examples and basis it needs to apply each pattern correctly.
+
+* Corpus
+
+** Phase 1 (2026-05-29): git commit bodies
+
+Git commit bodies authored by Craig Jennings across all repos under =~/code/= and =~/projects/=. After cleanup (subject lines, trailers, URL-only lines, AI-attribution lines, blank-run collapse):
+
+- 5355 raw commits, 1895 with non-trivial bodies
+- 128608 words, 912400 characters
+- 33 repos contributing; top sources: archsetup (703), rulesets (621), work (565), archangel (455), home (395)
+
+One register (deliberate technical prose). The view is useful but narrow on its own.
+
+** Phase 2 (2026-05-29): email + GitHub PR bodies + PR review comments
+
+Four sub-corpora added so the rules can be tested across registers.
+
+- *Personal email* (gmail + cmail, sent-only, body ≥50 words after cleanup): 1139 messages, 283,092 words.
+- *Work email* (dmail, same filter): 22 messages, 3910 words. Small sample.
+- *PR descriptions* (github.com, author cjennings, body ≥100 chars after cleanup): 9 PRs, 1613 words. Small sample.
+- *PR review comments* (github.com, author cjennings, ≥20 words): 3 comments, 256 words. Tiny sample. Public GHE work isn't in this index.
+
+Signatures, quoted replies, and forwarded blocks stripped before analysis. Stats streamed; no corpus files written to disk.
+
+** Cross-register findings (the key result of Phase 2)
+
+The most important Phase 2 result is that *register splits matter*. Phase 1's signal from commit prose does not generalize cleanly to conversational prose.
+
+| Metric (per 1000 words) | Commits | Personal email | Work email | PR bodies | PR comments |
+|--------------------------+---------+----------------+------------+-----------+-------------|
+| Em-dash | 3.49 | 0.28 | 2.05 | 0.62 | 0.00 |
+| Semicolon | 3.16 | 0.64 | 0.26 | 0.62 | 0.00 |
+| Contractions | 3.57 | 38.52 | 28.13 | 17.36 | 50.78 |
+| Standalone "I" | 3.85 | 36.91 | 23.79 | 8.68 | 42.97 |
+| "we" | 0.22 | 8.18 | 14.83 | 1.24 | 0.00 |
+| "I'm" | 0.07 | 6.04 | 3.58 | 1.24 | 7.81 |
+
+Three observations:
+
+1. Em-dashes and semicolons are concentrated in commit prose, not conversational prose. The personal-mode rules on those (§13 and §33) hold up under Phase 2, but the basis shifts: the rules mostly enforce what is already true for email and PR comments. Commit prose is the outlier register that needs the rule, not the universal pattern.
+2. Contractions invert. Commits suppress contractions; email and PR-review prose use them heavily (38 to 50 per 1000). The Phase 1 contraction rule (§34) is strongly confirmed in the registers where contractions are most expected.
+3. The Phase 1 curiosity (I'm/I'll surprisingly rare relative to standalone "I") was a register effect, not a personal preference. In personal email, "I'm" runs 6.04 per 1000 vs standalone I at 36.91 — ratio close to natural English. Commit prose is the outlier where "I am" beats "I'm".
+
+AI-writing tells stay near zero across all five corpora. "leverage" surfaces 18 times in personal email (0.064 per 1000) — small but the only non-zero hit on the watch-list outside commits. All other watch-words clock 0 to 4 per corpus.
+
+* Findings against the 41 SKILL.md patterns
+
+** Strongly confirmed by the corpus
+
+*Pattern 17 (no emojis).* Zero emojis in corpus. Confirmed.
+
+*Pattern 7 (AI vocabulary).* "delve" 0. "embark" 0. "navigate the" 0. "in the realm of" 0. "seamless" 0. "moreover" 0. "furthermore" 0. "in conclusion" 0. "additionally" 1. "robust" 1. "leverage" 1. Rule confirmed for 11 of 12 watch-words. (One exception below.)
+
+*Pattern 22 (filler).* "moreover" / "furthermore" / "additionally" / "in conclusion": all zero or one occurrence. Filler-phrase avoidance confirmed.
+
+*Pattern 32 (first-person rewrite).* Standalone "I" at 3.85 per 1000 words. Craig writes first-person heavily. This is real, not aspirational.
+
+*Pattern 34 (contractions).* 459 contractions total (3.57 per 1000). Top hits: =doesn't= (92), =don't= (59), =isn't= (46), =it's= (43), =can't= (40), =that's= (34). Rule confirmed.
+
+*Pattern 38 (terse cut).* 41.1% of paragraphs are single-sentence. Craig writes terse. Paragraph breaks land after one complete thought even when short. Confirmed indirectly via paragraph structure.
+
+** Aspirational (corpus contradicts, but the rule is intentional self-discipline)
+
+*Pattern 13 (em-dash zero-tolerance, personal mode).* Corpus rate: 3.49 em-dashes per 1000 words. Comparable to AI-generated prose. Craig USES em-dashes regularly in commit bodies. The rule overrides his habit, it doesn't reflect it. Suggested rewording: drop the "LLMs use em dashes more than humans" framing; keep the zero-tolerance directive but rationale becomes "Craig's published voice (commit messages going forward, PR bodies, emails) drops em-dashes by choice because it reads cleaner and avoids a common AI tell, regardless of his pre-rule habit." Honest about the source.
+
+*Pattern 33 (semicolons → period/comma).* Corpus rate: 3.16 semicolons per 1000. Craig uses semicolons regularly. Same shape as #13: rule is self-discipline, not habit-reflection. Suggested rewording: acknowledge the rule overrides habit rather than implying it codifies one.
+
+These two rules are still valuable. Em-dashes and semicolons both read cleaner when absent from short imperative-leaning prose. But the SKILL.md should say "this is a rule I've decided to follow," not "this is how I already write."
+
+** Worth challenging
+
+*Pattern 7 watch-word "comprehensive".* 42 occurrences in corpus (~0.33 per 1000). All other AI-tell watch-words clock near zero. "comprehensive" appears to be genuine vocabulary for Craig in technical contexts ("comprehensive test coverage", "comprehensive audit"). Suggested change: pull "comprehensive" out of the watch-list, or carve out a "watch in clusters, not solo" note that flags only when "comprehensive" co-occurs with other AI-tell words.
+
+** Worth adding (corpus surfaces traits the rules don't capture)
+
+*Single-sentence paragraph cadence.* 41.1% of paragraphs are exactly one sentence. This is distinctive. Most prose-style guides advise multi-sentence paragraphs. Suggested addition (prose + personal): a positive pattern noting "a one-sentence paragraph is a finished thought, not a fragment. Break paragraphs after one complete thought when the next thought shifts angle, even if both are short." Anti-rule against "merge short paragraphs into multi-sentence ones."
+
+*Parenthetical density.* 23.07 opening parens per 1000 words. Heavy parenthetical use covers asides, clarifications, and scope-narrowing in parens. Currently no rule addresses this either way. Could add a positive pattern: "parentheses for asides are part of the voice. Don't strip them in a 'clean prose' pass."
+
+*Question-mark rarity.* 0.33 per 1000. Craig's prose is declarative. He states things, rarely asks them. Worth noting as a register marker (when /voice personal output has questions, double-check whether they're contextual or AI rhetoric).
+
+** Out of corpus (commits don't test these, Phase 2 needed)
+
+- *Pattern 13 in long-form prose.* Commit bodies are short. Email and PR bodies may show different em-dash rates.
+- *Pattern 14 (boldface).* Org-mode bold uses =*word*=, not detectable by simple grep. Markdown bold rare in commits.
+- *Pattern 16 (title case in headings).* Commits don't carry headings.
+- *Pattern 19 (collaborative artifacts).* Not present in commit bodies.
+- *Pattern 35 (sentence split on conjunctions).* Average sentence is 18.81 words, median 14, with 28% of sentences 21+ words. Long-sentence rate is moderate. Need to inspect actual sentences to know if they're conjunction-stitched. Defer.
+- *Pattern 36 (felt-experience cut).* Commit bodies wouldn't carry felt-experience prose. Email + journal corpus needed.
+- *Pattern 37 (sentence fragments).* 9.7% of sentences are 1-5 words. Some are legitimate ("All eight pass."), some may be fragments. Can't tell from word-count alone. Defer to a pass that does syntactic detection.
+- *Pattern 39 (public-artifact scope).* The corpus IS the public artifacts. The check is circular. Defer.
+- *Pattern 40 (praise vs correction asymmetry).* Not detectable in commit bodies. Email or PR-review corpus needed.
+
+** Curiosities (resolved by Phase 2)
+
+- *=I'm=* (9 occurrences) and *=I'll=* (2 occurrences) were surprisingly rare in Phase 1 relative to standalone =I= (495 occurrences). Phase 2 resolved this. Personal email shows I'm at 1710 occurrences (6.04 per 1000), I'll at 865 (3.06), I've at 458 (1.62), I'd at 384 (1.36). The Phase 1 rarity was a register effect, not a personal preference. Commit prose uniquely suppresses contractions; conversational prose runs them at near-natural English rate.
+
+* Suggested deltas
+
+*All six deltas landed 2026-06-10* via the voice-skill revision from the work-project session: 1 and 2 are in the §13/§33 rule lines and entries, 3 is the §7 soft-flag, 4-6 became patterns §43-§45. The list is kept as the record of what was proposed on 2026-05-29.
+
+Six concrete edits to =voice/SKILL.md=, all of which can land independently:
+
+1. *#13 (Em-Dash).* Drop the "LLMs use em dashes more than humans" framing in the personal-mode section. Restate the zero-tolerance rule as self-discipline ("Craig's published voice drops em-dashes by choice"), not habit-reflection. Cite: corpus rate 3.49/1000, AI-comparable.
+
+2. *#33 (Semicolons).* Same shape. Restate as self-discipline. Cite: corpus rate 3.16/1000.
+
+3. *#7 (AI Vocabulary).* Remove "comprehensive" from the watch-list, OR add a note that "comprehensive" alone is acceptable; flag only when it co-occurs with =delve= / =leverage= / =robust= / =seamless= / =moreover= etc. Cite: 42 occurrences, all other watch-words at 0 or 1.
+
+4. *NEW pattern (prose + personal): "Single-sentence paragraph cadence is a feature."* 41.1% of corpus paragraphs are exactly one sentence. A one-sentence paragraph is a finished thought, not a fragment. The voice pass should not merge short paragraphs into multi-sentence ones.
+
+5. *NEW pattern (prose + personal): "Parentheses for asides are part of the voice."* 23 opening parens per 1000 words. Heavy parenthetical use is distinctive. Don't strip parenthetical asides in a "clean prose" pass.
+
+6. *Register marker (advisory, not a rewrite rule): "Declarative is the default."* 0.33 question marks per 1000. Voice personal output that contains rhetorical questions should be checked. They're often AI rhetoric, not Craig's register.
+
+* What Phase 2 would add
+
+- Email corpus (gmail + cmail, sent-only, long-form): different register, especially long-form prose flow.
+- PR bodies and review comments: longer prose, deliberate register, includes the praise/correction asymmetry test ground.
+- Slack messages: casual register, contraction rate, sentence-fragment rate.
+- Syntactic detection: distinguish fragments from terse complete sentences for pattern #37.
+- Long-form documents (résumé, proposals if any): single register but high prose density.
+
+* Per-pattern entries
+
+All patterns are entered below per the pairing rule above, §1 through §45, matching SKILL.md's numbering.
+
+** §1 Undue Emphasis on Significance, Legacy, and Broader Trends
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Strip statements that puff up importance by claiming an arbitrary aspect represents or contributes to a broader trend. Watch for phrases like "stands as", "serves as", "testament to", "vital role", "pivotal moment", "evolving landscape", "marks a shift", "reflects broader", "setting the stage for", "indelible mark", "deeply rooted". Replace with a concrete fact or cut the sentence entirely.
+
+*** Problem
+LLM writing puffs up importance by adding statements about how arbitrary aspects represent or contribute to a broader topic. Watch-list words: stands/serves as, is a testament/reminder, a vital/significant/crucial/pivotal/key role/moment, underscores/highlights its importance/significance, reflects broader, symbolizing its ongoing/enduring/lasting, contributing to the, setting the stage for, marking/shaping the, represents/marks a shift, key turning point, evolving landscape, focal point, indelible mark, deeply rooted.
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+The Statistical Institute of Catalonia was officially established in 1989, marking a pivotal moment in the evolution of regional statistics in Spain. This initiative was part of a broader movement across Spain to decentralize administrative functions and enhance regional governance.
+#+end_example
+
+*** After
+#+begin_example
+The Statistical Institute of Catalonia was established in 1989 to collect and publish regional statistics independently from Spain's national statistics office.
+#+end_example
+
+*** History
+- Original SKILL.md entry: significance and broader-trend puffery, with watch-list phrases drawn from Wikipedia's AI-writing guide.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §2 Undue Emphasis on Notability and Media Coverage
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Cut notability claims that list sources without giving the substance. Replace "cited in The New York Times, BBC, Financial Times" with the actual argument made in one of them.
+
+*** Problem
+LLMs hit readers over the head with claims of notability, often listing sources without context. Watch-list words: independent coverage, local/regional/national media outlets, written by a leading expert, active social media presence.
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+Her views have been cited in The New York Times, BBC, Financial Times, and The Hindu. She maintains an active social media presence with over 500,000 followers.
+#+end_example
+
+*** After
+#+begin_example
+In a 2024 New York Times interview, she argued that AI regulation should focus on outcomes rather than methods.
+#+end_example
+
+*** History
+- Original SKILL.md entry: notability inflation through bare source lists.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §3 Superficial Analyses with -ing Endings
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Cut tacked-on present-participle phrases (highlighting, underscoring, emphasizing, ensuring, reflecting, symbolizing, contributing to, cultivating, fostering, encompassing, showcasing) that add fake depth without new information.
+
+*** Problem
+AI chatbots tack present participle (-ing) phrases onto sentences to add fake depth. Watch-list words: highlighting, underscoring, emphasizing, ensuring, reflecting, symbolizing, contributing to, cultivating, fostering, encompassing, showcasing.
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+The temple's color palette of blue, green, and gold resonates with the region's natural beauty, symbolizing Texas bluebonnets, the Gulf of Mexico, and the diverse Texan landscapes, reflecting the community's deep connection to the land.
+#+end_example
+
+*** After
+#+begin_example
+The temple uses blue, green, and gold colors. The architect said these were chosen to reference local bluebonnets and the Gulf coast.
+#+end_example
+
+*** History
+- Original SKILL.md entry: -ing phrase tacking for fake analytical depth.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §4 Promotional and Advertisement-like Language
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Remove travel-brochure adjectives (vibrant, breathtaking, nestled, stunning, renowned, must-visit, profound, rich figurative use) and replace promotional framing with concrete facts about the subject.
+
+*** Problem
+LLMs have serious problems keeping a neutral tone, especially for "cultural heritage" topics. Watch-list words: boasts a, vibrant, rich (figurative), profound, enhancing its, showcasing, exemplifies, commitment to, natural beauty, nestled, in the heart of, groundbreaking (figurative), renowned, breathtaking, must-visit, stunning.
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+Nestled within the breathtaking region of Gonder in Ethiopia, Alamata Raya Kobo stands as a vibrant town with a rich cultural heritage and stunning natural beauty.
+#+end_example
+
+*** After
+#+begin_example
+Alamata Raya Kobo is a town in the Gonder region of Ethiopia, known for its weekly market and 18th-century church.
+#+end_example
+
+*** History
+- Original SKILL.md entry: travel-brochure adjective patterns.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §5 Vague Attributions and Weasel Words
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Replace vague attributions (experts say, observers have cited, industry reports, some critics argue, several sources) with a named source plus the specific claim made.
+
+*** Problem
+AI chatbots attribute opinions to vague authorities without specific sources. Watch-list words: Industry reports, Observers have cited, Experts argue, Some critics argue, several sources or publications (when few cited).
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+Due to its unique characteristics, the Haolai River is of interest to researchers and conservationists. Experts believe it plays a crucial role in the regional ecosystem.
+#+end_example
+
+*** After
+#+begin_example
+The Haolai River supports several endemic fish species, according to a 2019 survey by the Chinese Academy of Sciences.
+#+end_example
+
+*** History
+- Original SKILL.md entry: vague-authority attribution patterns.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §6 Outline-like "Challenges and Future Prospects" Sections
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Delete formulaic "Despite its prosperity, X faces challenges" wrap-ups and "Future Outlook" boilerplate. Replace with the actual events that happened or omit the section.
+
+*** Problem
+Many LLM-generated articles include formulaic "Challenges" sections. Watch-list words: "Despite its... faces several challenges...", "Despite these challenges", "Challenges and Legacy", "Future Outlook".
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+Despite its industrial prosperity, Korattur faces challenges typical of urban areas, including traffic congestion and water scarcity. Despite these challenges, with its strategic location and ongoing initiatives, Korattur continues to thrive as an integral part of Chennai's growth.
+#+end_example
+
+*** After
+#+begin_example
+Traffic congestion increased after 2015 when three new IT parks opened. The municipal corporation began a stormwater drainage project in 2022 to address recurring floods.
+#+end_example
+
+*** History
+- Original SKILL.md entry: outline-template "Challenges and Future Prospects" sections.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §7 Overused "AI Vocabulary" Words
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Flag and rewrite around the high-frequency AI vocabulary list. Watch-list words: Additionally, align with, comprehensive, crucial, delve, emphasizing, enduring, enhance, fostering, garner, highlight (verb), interplay, intricate or intricacies, key (adjective), landscape (abstract noun), pivotal, showcase, tapestry (abstract noun), testament, underscore (verb), valuable, vibrant. "comprehensive" is a soft flag because the corpus shows it as genuine Craig vocabulary he chooses to use sparingly. Suggest an alternative ("full", "complete", "thorough", or rewording to drop the adjective) and let Craig decide per instance.
+
+*** Problem
+These words appear far more frequently in post-2023 text. They often co-occur.
+
+*** Basis
+Corpus-measured across registers (2026-05-29). Phase 1 git commits: "comprehensive" 42 occurrences, every other watch-word 0 or 1. Phase 2 conversational and PR corpora: "comprehensive" 1 in personal email, 0 in work email, PR descriptions, and PR review comments. "leverage" 18 in personal email, 0 to 1 elsewhere. Every other watch-word stays at 0 to 4 across all five corpora.
+
+Two takeaways. First, "comprehensive" is concentrated in commit prose (technical-doc register: "comprehensive test coverage", "comprehensive audit") and almost absent from conversational prose. Craig has chosen to keep it on the watch-list because he is consciously trying to use it sparingly. Second, "leverage" earns a soft watch in personal email even though the rest of the list stays clean. The two together suggest the rule should flag-and-suggest individual hits in technical prose without treating any single watch-word as automatic disqualification.
+
+*** Before
+#+begin_example
+Additionally, a distinctive feature of Somali cuisine is the incorporation of camel meat. An enduring testament to Italian colonial influence is the widespread adoption of pasta in the local culinary landscape, showcasing how these dishes have integrated into the traditional diet.
+#+end_example
+
+*** After
+#+begin_example
+Somali cuisine also includes camel meat, which is considered a delicacy. Pasta dishes, introduced during Italian colonization, remain common, especially in the south.
+#+end_example
+
+*** History
+- Original SKILL.md entry: high-frequency post-2023 AI vocabulary list.
+- 2026-05-29 (commit =c3cf9a5=): note on "comprehensive" added with corpus measurement and soft-flag guidance.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §8 Avoidance of "is"/"are" (Copula Avoidance)
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Replace elaborate copula substitutes (serves as, stands as, marks, represents, boasts, features, offers) with plain "is" or "has".
+
+*** Problem
+LLMs substitute elaborate constructions for simple copulas. Watch-list words: serves as, stands as, marks, represents (a), boasts, features, offers (a).
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+Gallery 825 serves as LAAA's exhibition space for contemporary art. The gallery features four separate spaces and boasts over 3,000 square feet.
+#+end_example
+
+*** After
+#+begin_example
+Gallery 825 is LAAA's exhibition space for contemporary art. The gallery has four rooms totaling 3,000 square feet.
+#+end_example
+
+*** History
+- Original SKILL.md entry: copula avoidance patterns.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §9 Negative Parallelisms
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Rewrite "not only X but Y" and "it's not just about X, it's Y" constructions as a single direct claim.
+
+*** Problem
+Constructions like "Not only...but..." or "It's not just about..., it's..." are overused as a way to claim depth.
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+It's not just about the beat riding under the vocals; it's part of the aggression and atmosphere. It's not merely a song, it's a statement.
+#+end_example
+
+*** After
+#+begin_example
+The heavy beat adds to the aggressive tone.
+#+end_example
+
+*** History
+- Original SKILL.md entry: negative-parallelism stock phrasing.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §10 Rule of Three Overuse
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Break the reflexive three-item list pattern when the third item is filler. Collapse to one or two specific items.
+
+*** Problem
+LLMs force ideas into groups of three to appear comprehensive. The third item is usually filler chosen to fit the cadence, not because it adds substance.
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+The event features keynote sessions, panel discussions, and networking opportunities. Attendees can expect innovation, inspiration, and industry insights.
+#+end_example
+
+*** After
+#+begin_example
+The event includes talks and panels. There's also time for informal networking between sessions.
+#+end_example
+
+*** History
+- Original SKILL.md entry: rule-of-three cadence overuse.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §11 Elegant Variation (Synonym Cycling)
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Stop cycling synonyms for the same referent across consecutive sentences. Repeat the noun, or merge the sentences.
+
+*** Problem
+AI has repetition-penalty code causing excessive synonym substitution. The protagonist becomes the main character becomes the central figure becomes the hero, all referring to the same person.
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+The protagonist faces many challenges. The main character must overcome obstacles. The central figure eventually triumphs. The hero returns home.
+#+end_example
+
+*** After
+#+begin_example
+The protagonist faces many challenges but eventually triumphs and returns home.
+#+end_example
+
+*** History
+- Original SKILL.md entry: elegant-variation synonym cycling driven by repetition penalty.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §12 False Ranges
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Rewrite "from X to Y" constructions where X and Y are not on the same scale. List the items plainly instead.
+
+*** Problem
+LLMs use "from X to Y" constructions where X and Y are not on a meaningful scale ("from the Big Bang to dark matter") to imply comprehensive sweep.
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+Our journey through the universe has taken us from the singularity of the Big Bang to the grand cosmic web, from the birth and death of stars to the enigmatic dance of dark matter.
+#+end_example
+
+*** After
+#+begin_example
+The book covers the Big Bang, star formation, and current theories about dark matter.
+#+end_example
+
+*** History
+- Original SKILL.md entry: false-range "from X to Y" constructions.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §13 Em Dash Overuse
+
+*** Modes
+General mode: overuse-reduction.
+Prose + personal modes: zero-tolerance.
+
+*** Rule
+Replace em-dashes (=—=) with a comma, period, colon, or parentheses, whichever fits. Zero-tolerance in prose and personal modes holds *everywhere in the text*, including inside example blocks, code-fence prose, and quoted material. An em-dash in a quoted line still gets replaced.
+
+*** Problem
+Craig's published voice drops em-dashes by choice: they read cleaner absent from short imperative-leaning prose and their overuse is a common AI tell (LLMs use em dashes more than the median human writer, mimicking "punchy" sales writing). The rule is chosen self-discipline, not a reflection of his pre-rule habit — the corpus shows he used them regularly in commit bodies.
+
+*** Basis
+Phase 1 corpus (git commits, 128k words): 3.49 em-dashes per 1000 words. Comparable to AI-generated prose. Phase 2 corpus reveals a sharp register split: personal email 0.28 per 1000, work email 2.05, PR descriptions 0.62, PR review comments 0.00. Em-dashes are concentrated in commit prose, almost absent from email and PR review prose. The zero-tolerance rule in prose and personal modes mostly enforces what is already true for non-commit registers. The rule still earns its place because commit prose is the high-volume register where the AI-tell em-dash habit shows up. Self-discipline, not habit-reflection, for the commit register specifically.
+
+*** Before
+#+begin_example
+The term is primarily promoted by Dutch institutions—not by the people themselves. You don't say "Netherlands, Europe" as an address—yet this mislabeling continues—even in official documents.
+#+end_example
+
+*** After
+#+begin_example
+The term is primarily promoted by Dutch institutions, not by the people themselves. You don't say "Netherlands, Europe" as an address, yet this mislabeling continues in official documents.
+#+end_example
+
+*** History
+- Original SKILL.md entry: rule scoped to general overuse-reduction.
+- 2026-05-26 (commit =4fac2a0=): prose mode added, rule strengthened to zero-tolerance in prose and personal.
+- 2026-05-29 (commit =c3cf9a5=): Note on basis added with corpus measurement.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+- 2026-06-10: the self-discipline reframing (a "Suggested deltas" item from 2026-05-29, never applied) moved from the findings section into the entry proper and into the SKILL.md rule line. Craig's call, from the work-project session.
+
+** §14 Overuse of Boldface
+
+*** Modes
+General mode only. Prose and personal inherit it. Pattern §41 is the related Craig-voice rule covering emphasis-by-formatting in his authored prose.
+
+*** Rule
+Strip mechanical boldface used to call out terms, acronyms, or phrases in running prose. Bold survives only for structural emphasis the document genuinely needs.
+
+*** Problem
+AI chatbots emphasize phrases in boldface mechanically. Acronyms, names, and key terms get wrapped in bold even when the surrounding sentence already gives them stress.
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+It blends **OKRs (Objectives and Key Results)**, **KPIs (Key Performance Indicators)**, and visual strategy tools such as the **Business Model Canvas (BMC)** and **Balanced Scorecard (BSC)**.
+#+end_example
+
+*** After
+#+begin_example
+It blends OKRs, KPIs, and visual strategy tools like the Business Model Canvas and Balanced Scorecard.
+#+end_example
+
+*** History
+- Original SKILL.md entry: mechanical boldface around terms and acronyms.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §15 Inline-Header Vertical Lists
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Collapse bullet lists whose items start with a bold header plus colon into running prose, unless the list structure is genuinely the right shape.
+
+*** Problem
+AI outputs lists where items start with bolded headers followed by colons, often when a paragraph would carry the same content more naturally.
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+- **User Experience:** The user experience has been significantly improved with a new interface.
+- **Performance:** Performance has been enhanced through optimized algorithms.
+- **Security:** Security has been strengthened with end-to-end encryption.
+#+end_example
+
+*** After
+#+begin_example
+The update improves the interface, speeds up load times through optimized algorithms, and adds end-to-end encryption.
+#+end_example
+
+*** History
+- Original SKILL.md entry: inline-header vertical list pattern.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §16 Title Case in Headings
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Lowercase headings that are reflexively title-cased. Sentence case is the default unless the project's house style is title case.
+
+*** Problem
+AI chatbots capitalize all main words in headings even when the surrounding document uses sentence case.
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+## Strategic Negotiations And Global Partnerships
+#+end_example
+
+*** After
+#+begin_example
+## Strategic negotiations and global partnerships
+#+end_example
+
+*** History
+- Original SKILL.md entry: reflexive title-case in headings.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §17 Emojis
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Remove decorative emojis from headings, bullets, and prose unless the document is a register where emoji is genuinely intended.
+
+*** Problem
+AI chatbots often decorate headings or bullet points with emojis to add visual structure that the prose itself does not need.
+
+*** Basis
+Corpus-measured: 2026-05-29 commit corpus shows zero emojis. The rule reflects established practice.
+
+*** Before
+#+begin_example
+🚀 **Launch Phase:** The product launches in Q3
+💡 **Key Insight:** Users prefer simplicity
+✅ **Next Steps:** Schedule follow-up meeting
+#+end_example
+
+*** After
+#+begin_example
+The product launches in Q3. User research showed a preference for simplicity. Next step: schedule a follow-up meeting.
+#+end_example
+
+*** History
+- Original SKILL.md entry: decorative emoji in headings and bullets.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §18 Curly Quotation Marks
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Convert curly quotation marks to straight ASCII quotes.
+
+*** Problem
+ChatGPT uses curly quotes instead of straight quotes, which is a recognizable tell in technical and source-controlled writing.
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+He said “the project is on track” but others disagreed.
+#+end_example
+
+*** After
+#+begin_example
+He said "the project is on track" but others disagreed.
+#+end_example
+
+*** History
+- Original SKILL.md entry: curly-quote substitution.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §19 Collaborative Communication Artifacts
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Strip chatbot correspondence framing ("I hope this helps", "Let me know if...", "Here is an overview of...", "Certainly!", "Of course!", "Would you like...") that leaked into the body.
+
+*** Problem
+Text meant as chatbot correspondence gets pasted as content, carrying the assistant's framing into a document that should stand alone. Watch-list words: I hope this helps, Of course!, Certainly!, You're absolutely right!, Would you like..., let me know, here is a...
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+Here is an overview of the French Revolution. I hope this helps! Let me know if you'd like me to expand on any section.
+#+end_example
+
+*** After
+#+begin_example
+The French Revolution began in 1789 when financial crisis and food shortages led to widespread unrest.
+#+end_example
+
+*** History
+- Original SKILL.md entry: collaborative-communication artifacts pasted as content.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §20 Knowledge-Cutoff Disclaimers
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Remove training-cutoff hedges ("as of my last update", "while specific details are scarce", "based on available information") and either commit to a fact or omit the claim.
+
+*** Problem
+AI disclaimers about incomplete information get left in text, signaling the model's uncertainty rather than the author's. Watch-list words: as of [date], Up to my last training update, While specific details are limited or scarce, based on available information.
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+While specific details about the company's founding are not extensively documented in readily available sources, it appears to have been established sometime in the 1990s.
+#+end_example
+
+*** After
+#+begin_example
+The company was founded in 1994, according to its registration documents.
+#+end_example
+
+*** History
+- Original SKILL.md entry: knowledge-cutoff hedging.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §21 Sycophantic/Servile Tone
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Cut servile opener phrases ("Great question!", "You're absolutely right", "That's an excellent point") and proceed straight to the substance.
+
+*** Problem
+Overly positive, people-pleasing language reads as performance rather than communication and signals AI assistant register.
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+Great question! You're absolutely right that this is a complex topic. That's an excellent point about the economic factors.
+#+end_example
+
+*** After
+#+begin_example
+The economic factors you mentioned are relevant here.
+#+end_example
+
+*** History
+- Original SKILL.md entry: sycophantic and servile opener patterns.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §22 Filler Phrases
+
+*** Modes
+General mode only. Prose and personal inherit it. Pattern §38 is the stricter cousin for Craig's authored prose.
+
+*** Rule
+Compress wordy filler to plain equivalents: "in order to" to "to", "due to the fact that" to "because", "at this point in time" to "now", "in the event that" to "if", "has the ability to" to "can", "it is important to note that" to nothing, "for the purpose of" to "to", "in spite of the fact that" to "although", "a great deal of" to "much", "at this juncture" to "now".
+
+*** Problem
+Wordy filler stretches a sentence without adding precision. Cutting it shortens the prose and sharpens the claim.
+
+*** Basis
+Corpus-measured: 2026-05-29 commit corpus shows "moreover", "furthermore", "additionally", "in conclusion" all at zero or one occurrence. Filler-phrase avoidance confirmed at the watch-list level.
+
+*** Before
+#+begin_example
+In order to achieve this goal, we need to allocate resources due to the fact that the team has the ability to deliver. At this point in time, it is important to note that the data shows progress.
+#+end_example
+
+*** After
+#+begin_example
+To achieve this, we need to allocate resources because the team can deliver. The data shows progress.
+#+end_example
+
+*** Detection
+The original SKILL.md entry uses a Before to After substitution table:
+- "In order to achieve this goal" to "To achieve this"
+- "Due to the fact that it was raining" to "Because it was raining"
+- "At this point in time" to "Now"
+- "In the event that you need help" to "If you need help"
+- "The system has the ability to process" to "The system can process"
+- "It is important to note that the data shows" to "The data shows"
+- "For the purpose of" to "To"
+- "In spite of the fact that" to "Although"
+- "A great deal of" to "Much"
+- "At this juncture" to "Now"
+
+*** History
+- Original SKILL.md entry: wordy filler phrase substitutions.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §23 Excessive Hedging
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Strip stacked hedges ("could potentially possibly", "might have some effect") down to a single appropriate qualifier.
+
+*** Problem
+Over-qualifying statements weakens them without adding accuracy. One hedge does the job that three do.
+
+*** Basis
+Observation-derived (Strunk and White, Garner).
+
+*** Before
+#+begin_example
+It could potentially possibly be argued that the policy might have some effect on outcomes.
+#+end_example
+
+*** After
+#+begin_example
+The policy may affect outcomes.
+#+end_example
+
+*** History
+- Original SKILL.md entry: stacked hedge reduction.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §24 Generic Positive Conclusions
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Replace vague upbeat endings ("the future looks bright", "exciting times lie ahead", "a step in the right direction") with a concrete fact or cut the closer entirely.
+
+*** Problem
+Vague upbeat endings give the document the shape of a press release without making any specific claim.
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+The future looks bright for the company. Exciting times lie ahead as they continue their journey toward excellence. This represents a major step in the right direction.
+#+end_example
+
+*** After
+#+begin_example
+The company plans to open two more locations next year.
+#+end_example
+
+*** History
+- Original SKILL.md entry: generic positive conclusion boilerplate.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §25 Hyphenated Word Pair Overuse
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Drop reflexive hyphens from common modifier pairs (third-party, cross-functional, client-facing, data-driven, decision-making, well-known, high-quality, real-time, long-term, end-to-end) where humans hyphenate inconsistently. Less common or genuinely technical compound modifiers can keep their hyphens.
+
+*** Problem
+AI hyphenates common word pairs with perfect consistency. Humans rarely hyphenate these uniformly, and when they do, it is inconsistent. The uniformity itself is the tell.
+
+*** Basis
+Observation-derived (Wikipedia "Signs of AI Writing").
+
+*** Before
+#+begin_example
+The cross-functional team delivered a high-quality, data-driven report on our client-facing tools. Their decision-making process was well-known for being thorough and detail-oriented.
+#+end_example
+
+*** After
+#+begin_example
+The cross functional team delivered a high quality, data driven report on our client facing tools. Their decision making process was known for being thorough and detail oriented.
+#+end_example
+
+*** History
+- Original SKILL.md entry: hyphenated common-modifier overuse.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §26 Long Word → Short Word
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Swap long Latinate words for their short Anglo-Saxon equivalents per the Plain English wordlist: utilize to use, commence to start or begin, terminate to end, facilitate to help, demonstrate to show, sufficient to enough, prior to to before, subsequent to to after, approximately to about, endeavor to try, ascertain to find out, assistance to help, obtain to get, modification to change, implement to carry out, optimal to best, regarding to about, methodology to method, "in the event of" to "if".
+
+*** Problem
+Long Latinate words signal effortful writing without adding precision. Anglo-Saxon roots are shorter and clearer.
+
+*** Basis
+Observation-derived (Strunk and White, Orwell, Plain English Campaign, Garner).
+
+*** Before
+#+begin_example
+The system will utilize advanced algorithms to facilitate optimal performance. Prior to deployment, we must ascertain that the methodology is sufficient.
+#+end_example
+
+*** After
+#+begin_example
+The system uses algorithms to get the best performance. Before deployment, we must check that the method works.
+#+end_example
+
+*** History
+- Original SKILL.md entry: Plain English wordlist substitutions.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §27 Active Over Passive Voice
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Rewrite passive constructions to active when the actor is recoverable from context. Flag rather than auto-rewrite when the actor genuinely does not matter, because passive is sometimes the right choice in technical contexts.
+
+*** Problem
+Passive voice hides who did what. Active voice is shorter and clearer in most cases. Skip when the actor genuinely does not matter (technical writing about an inanimate process: "the table was created in 2024" can stay passive).
+
+*** Basis
+Observation-derived (Strunk and White, Orwell).
+
+*** Before
+#+begin_example
+The migration was run by the deployment script. The bug was introduced in commit abc123. The fix was applied by the team.
+#+end_example
+
+*** After
+#+begin_example
+The deployment script ran the migration. Commit abc123 introduced the bug. The team applied the fix.
+#+end_example
+
+*** Detection
+"to be" plus past-participle patterns where the actor is recoverable from context.
+
+*** History
+- Original SKILL.md entry: active-over-passive with suggestion-only treatment in v1.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §28 Comma Splices
+
+*** Modes
+General mode only. Prose and personal inherit it. The semicolon escape route is blocked in personal mode by §33.
+
+*** Rule
+Split two independent clauses joined only by a comma into two sentences or join them with a conjunction. In general mode a semicolon is an acceptable repair. In personal mode the semicolon is itself a target (§33), so prefer the period.
+
+*** Problem
+Comma splices read as run-ons. Either split into two sentences, join with a conjunction, or use a semicolon (in personal mode this becomes a period).
+
+*** Basis
+Observation-derived (Strunk and White).
+
+*** Before
+#+begin_example
+The build failed, the test suite reported three errors.
+#+end_example
+
+*** After
+#+begin_example
+The build failed. The test suite reported three errors.
+#+end_example
+
+*** Detection
+Two independent clauses joined only by a comma.
+
+*** History
+- Original SKILL.md entry: comma-splice repair.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §29 Cliché Flag
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Replace business and conversational clichés with the plain meaning, including in casual register where "it's fine, it's casual" is the tell. Watch-list phrases: at the end of the day, moving forward, going forward, at this juncture, circle back, low-hanging fruit, deep dive, leverage (as verb), synergy, take it offline, ducks in a row, boil the ocean, pivot (corporate sense), keep it loose, keep it casual, touch base, circle up, hit the ground running, move the needle, on the same page, no-brainer, win-win.
+
+*** Problem
+Clichés signal effortful prose without saying anything specific. Replace with the actual meaning. A casual, friendly, or conversational register is not a license to keep a cliché. Cut it there too. If you catch yourself justifying one as "it's fine, it's casual," that is the tell. Craig flagged this on 2026-05-22 when "keep it loose" slipped through as "acceptable casual." That is exactly the miss this note prevents.
+
+*** Basis
+Observation-derived (Orwell, Garner).
+
+*** Before
+#+begin_example
+At the end of the day, we need to leverage our core competencies and circle back on the low-hanging fruit.
+#+end_example
+
+*** After
+#+begin_example
+We need to use what we already do well and start with the easiest improvements first.
+#+end_example
+
+*** History
+- Original SKILL.md entry: business and conversational cliché list.
+- 2026-05-22: Craig added "keep it loose" / "keep it casual" after a miss in earlier output.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §30 Jargon-Fragment → Complete Sentence
+
+*** Modes
+General mode only. Prose and personal inherit it. Pattern §37 is the stricter cousin for Craig's authored prose.
+
+*** Rule
+Rewrite telegraphic sentence fragments inside prose paragraphs as complete sentences with subject and verb. Headings and bullet items are exempt because fragments are valid there.
+
+*** Problem
+Telegraphic fragments in prose paragraphs read as bullet-style notes leaking into running text. They lose the connective tissue a complete sentence carries.
+
+*** Basis
+Observation-derived (Strunk and White).
+
+*** Before
+#+begin_example
+The new function handles edge cases. Empty input throws. Whitespace gets trimmed. Returns null on no match.
+#+end_example
+
+*** After
+#+begin_example
+The new function handles edge cases. It throws on empty input, trims whitespace, and returns null when no match is found.
+#+end_example
+
+*** Detection
+Sentence-like fragments inside prose paragraphs that read as bullet-list shorthand. Headings and bullet items are exempt because fragments are valid there.
+
+*** History
+- Original SKILL.md entry: jargon-fragment rewrite for prose paragraphs.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §31 Noun-ified Verbs
+
+*** Modes
+General mode only. Prose and personal inherit it.
+
+*** Rule
+Replace corporate-speak noun-ifications with the real noun: "the ask" to "the request", "a learn" to "the lesson", "the spend" to "the budget", "a build" to "the system" or "the prototype", "the reveal" to "the announcement", "the lift" to "the effort", "the get" to "the result". Philosophical nominalizations ("the becoming", "the unfolding") are not targets.
+
+*** Problem
+Corporate-speak nominalization reads as performance. The real nouns are shorter and clearer. Watch-list: the ask, a learn, the spend, a build, the reveal, a do, the lift, the get, the say.
+
+*** Basis
+Observation-derived (Garner; Craig's voice rules in claude-rules/commits.md).
+
+*** Before
+#+begin_example
+The ask was for a quick build. After the reveal, we'll do a learn.
+#+end_example
+
+*** After
+#+begin_example
+The request was for a quick prototype. After the announcement, we'll review what worked.
+#+end_example
+
+*** History
+- Original SKILL.md entry: corporate-speak nominalization.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §32 First-Person Voice Rewrite
+
+*** Modes
+Personal mode only. General and prose skip it because a research note, a document, or anyone else's text is legitimately third-person.
+
+*** Rule
+Rewrite impersonal third-person publish-artifact bodies into first person ("I added X", "I missed Y", "I kept Z because..."). The commit subject line stays imperative per Conventional Commits ("feat: add support for X"). The body shifts to first person. Skip the rewrite for mechanical changes (a chore version bump, a typo fix) where the subject alone carries the message.
+
+*** Problem
+Impersonal third-person ("Add support for X", "The change adds Y") reads as press-release voice in a commit body or PR description. First-person ("I added X", "I kept Y because...") sounds like one engineer talking to another.
+
+*** Basis
+Corpus-measured across registers (2026-05-29): standalone "I" runs 3.85 per 1000 words in git commits, 36.91 in personal email, 23.79 in work email, 8.68 in PR descriptions, 42.97 in PR review comments. First-person density is roughly 10x higher in conversational registers than in commits. Craig writes first-person heavily across the board, but commit prose under-uses "I" relative to natural English. The rule strengthens the under-using register without overreaching: it asks the publish-artifact body to write the way the email body already does.
+
+*** Before
+#+begin_example
+Adds the new validation step before saving. The previous flow allowed empty values to leak into the database. This change blocks them at the API boundary.
+#+end_example
+
+*** After
+#+begin_example
+I added a validation step before saving. The previous flow let empty values leak into the database. I'm blocking them at the API boundary now.
+#+end_example
+
+*** Detection
+Impersonal third-person construction in a publish-artifact body where first-person fits naturally.
+
+*** History
+- Original SKILL.md entry: first-person rewrite for publish artifacts.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §33 Semicolon → Period or Comma
+
+*** Modes
+Prose and personal modes. General mode keeps semicolons because academic and literary registers use them legitimately.
+
+*** Rule
+Replace semicolons with a period (split into two sentences) or a comma (when the clauses are tightly coupled) in Craig's authored prose: emails, documents, working notes, commit-message bodies, PR descriptions, PR review comments. A formal long-form document can keep the semicolon, but the default is to split. Chosen self-discipline, not habit-reflection.
+
+*** Problem
+Craig's published voice drops semicolons by choice. They make the writing feel unnecessarily literary, the period-split usually reads better, and dropping them removes one common AI tell. The rule overrides his pre-rule habit rather than codifying one — the corpus shows he used semicolons regularly in commit prose.
+
+*** Basis
+Corpus-measured across registers (2026-05-29): semicolons run 3.16 per 1000 words in git commits, 0.64 in personal email, 0.26 in work email, 0.62 in PR descriptions, 0.00 in PR review comments. Same register split as em-dashes (§13). Semicolons are concentrated in commit prose; conversational prose almost never uses them. The rule mostly enforces what is already true for non-commit registers. It earns its place because commit prose is the register where Craig's habit and the AI-tell pattern overlap.
+
+*** Before
+#+begin_example
+I added the validation; the previous flow allowed empty values to leak through.
+#+end_example
+
+*** After
+#+begin_example
+I added the validation. The previous flow allowed empty values to leak through.
+#+end_example
+
+*** Detection
+Semicolons in prose Craig authors: emails, documents, working notes, commit-message bodies, PR descriptions, PR review comments.
+
+*** History
+- Original SKILL.md entry: semicolon to period or comma in Craig's authored prose.
+- 2026-05-29 (commit =c3cf9a5=): basis note added with corpus measurement reframing the rule as self-discipline.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+- 2026-06-10: the self-discipline reframing (a "Suggested deltas" item from 2026-05-29, never applied) moved from the findings section into the entry proper and into the SKILL.md rule line. Craig's call, from the work-project session.
+
+** §34 Contractions
+
+*** Modes
+Prose and personal modes. General mode skips because academic, literary, and formal registers often prefer uncontracted forms.
+
+*** Rule
+Prefer contractions in Craig's prose (it's, that's, don't, we're, I'd, won't) unless a negation or emphasis genuinely needs the uncontracted weight.
+
+*** Problem
+Uncontracted English reads stiff in a short prose body unless a negation or emphasis needs the weight. Prefer contractions in his prose: emails, documents, commit and PR bodies.
+
+*** Basis
+Corpus-measured across registers (2026-05-29). Contraction rate per 1000 words: git commits 3.57, personal email 38.52, work email 28.13, PR descriptions 17.36, PR review comments 50.78. Commit prose is the outlier register that suppresses contractions; conversational and PR-review prose use them heavily, near the natural-English rate. The Phase 1 curiosity (I'm 9 occurrences vs standalone I at 495 in commits) was a register effect, not a personal preference. Personal email runs I'm at 6.04 per 1000 vs standalone I at 36.91, ratio close to natural English. Top contractions in personal email: i'm 1710, it's 928, i'll 865, don't 632, you're 567, i've 458, that's 433, i'd 384, we're 307, didn't 299. The rule confirms across the board, with the strongest evidence from the conversational registers where contractions are most expected.
+
+*** Before
+#+begin_example
+It is worth noting that the change does not break the existing flow. We are confident that this is the right approach.
+#+end_example
+
+*** After
+#+begin_example
+It's worth noting the change doesn't break the existing flow. We're confident this is the right approach.
+#+end_example
+
+*** Detection
+Uncontracted forms in publish-artifact prose where the contraction reads more naturally. Note pattern §38 catches "worth noting" as rhetorical padding. The example above shows isolated transformation. In practice both passes apply.
+
+*** History
+- Original SKILL.md entry: contractions preferred in Craig's authored prose.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §35 Sentence Split on Conjunctions
+
+*** Modes
+Prose and personal modes. General mode skips because academic and literary registers use long compound sentences deliberately.
+
+*** Rule
+Split sentences that stack three or four clauses joined by "so", "and", "but" into two or three shorter sentences when the split does not lose meaning.
+
+*** Problem
+Long compound sentences read easier as two or three shorter ones in a prose or publish-artifact body. Skip in academic or literary prose where deliberate long sentences are the register.
+
+*** Basis
+Observation-derived (Craig's voice rules in claude-rules/commits.md). Corpus context: average sentence is 18.81 words, median 14, with 28% of sentences at 21+ words. Long-sentence rate is moderate. Inspection of actual sentences for conjunction-stitching is deferred to Phase 2.
+
+*** Before
+#+begin_example
+I added the validation step before saving so empty values get blocked at the API boundary, and I also added a regression test that exercises the empty-string case, but I did not change the upstream caller because that's a separate concern.
+#+end_example
+
+*** After
+#+begin_example
+I added the validation step before saving so empty values get blocked at the API boundary. I added a regression test that exercises the empty-string case. I didn't change the upstream caller because that's a separate concern.
+#+end_example
+
+*** Detection
+Sentences that stack three or four clauses with commas and conjunctions ("so", "and", "but") where splitting on a conjunction would not lose meaning.
+
+*** History
+- Original SKILL.md entry: sentence split on conjunctions for Craig's authored prose.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §36 Felt-Experience Narration
+
+*** Modes
+Prose and personal modes. General mode skips because third-party prose is legitimately allowed to describe how something feels.
+
+*** Rule
+Cut phrases that tell the reader how the change will feel or how often the writer will use it ("I'll feel this every time I commit", "this will be a relief", "I'm excited about", "this is going to be huge"). State what changed and let the reader decide what to do with it.
+
+*** Problem
+Felt-experience phrases read as performance, not communication. They tell the reader how the writer wants them to receive the change rather than describing the change.
+
+*** Basis
+Observation-derived (Craig's voice rules in claude-rules/commits.md). Commit-body corpus would not carry felt-experience prose; email and journal corpus deferred to Phase 2.
+
+*** Before
+#+begin_example
+I'm so excited about this — I'll feel the speedup every time I run the build. This is going to be a huge relief.
+#+end_example
+
+*** After
+#+begin_example
+The build now finishes in roughly half the time it used to take.
+#+end_example
+
+*** Detection
+Phrases that tell the reader how the change will feel or how often the writer will use it.
+
+*** History
+- Original SKILL.md entry: felt-experience narration cut for Craig's authored prose.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §37 Sentence Fragments → Complete
+
+*** Modes
+Prose and personal modes. General mode keeps the softer §30, which exempts more, because the strong "every sentence" rule is Craig's voice and should not be imposed on third-party text.
+
+*** Rule
+Rewrite every sentence fragment inside a prose paragraph in Craig's authored text as a complete sentence with subject and verb. Bullets and headings can stay fragments. Exemption: verdict formulas in PR review summaries ("Approving.", "Requesting changes.", "Approved.") are house style and stay — rewriting them imposes the rule where Craig's calibrated voice already decided otherwise.
+
+*** Problem
+Bullet shorthand leaking into running prose ("Two changes." "Fix incoming." "Body as decision log.") reads as bullet-list notes pasted into a paragraph. Every prose sentence needs a subject and a verb in prose and personal modes.
+
+*** Basis
+Observation-derived (Craig's voice rules in claude-rules/commits.md). Corpus context: 9.7% of sentences are 1-5 words. Some are legitimate single-word claims ("All eight pass."), some may be fragments. Word count alone cannot distinguish. Syntactic detection deferred to Phase 2.
+
+*** Before
+#+begin_example
+Big change to the validator. Three new patterns. Test coverage up. Old behavior preserved.
+#+end_example
+
+*** After
+#+begin_example
+I made a big change to the validator. There are three new patterns and the test coverage is up. The old behavior is preserved.
+#+end_example
+
+*** Detection
+Sentence fragments inside prose paragraphs in any text Craig authors: an email, a document, a working note, a commit or PR body. Bullets and headings remain fair game for fragments.
+
+*** History
+- Original SKILL.md entry: sentence-fragment rewrite for Craig's authored prose.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+- 2026-06-10: verdict-formula exemption added. The skill survived in practice by being selectively ignored on "Approving." / "Requesting changes." / "Approved.", and selective ignoring is the same muscle that skips real patterns. Documenting the exception removes one standing occasion for judgment-override. Craig's call, from the work-project session.
+
+** §38 Terse Cut — Omit Needless Words
+
+*** Modes
+Prose and personal modes. General mode keeps the softer §22 because academic registers retain "worth noting" and "it's important to understand" as legitimate transition markers.
+
+*** Rule
+Two cuts. First strip the soft rhetorical padding ("worth noting", "it's important to understand", "as you can see", "needless to say", "obviously", "of course", "in essence", "fundamentally"). Then run the general omit-needless-words sweep the padding list only samples: read each sentence and cut or collapse every word and clause that can go without losing meaning, not only the named phrases. Forcing test, per sentence: try to delete half of it and keep only what changes meaning.
+
+*** Execution position (prose + personal)
+§38 is not just one pattern in the walk — it is the mandatory *last* pass before any draft is presented. The SKILL.md Process makes it an explicit standalone final step, run after every other pattern. The reason is empirical: a draft that cleared the other 40 patterns still routinely runs a third too long, because ordinary verbosity matches no named trigger and the categorical detectors come back clean while the text is still bloated. Folded into the general walk, §38 gets glossed as a wordlist match. As a separate final step it gets the real per-sentence "delete half of it" sweep. A public draft shown without this pass is a defect in the same class as skipping the skill entirely.
+
+*** Problem
+Tier 1 omit-needless-words (§26) catches rigid offenders ("the fact that", "in order to"). The original §38 added a named padding list ("worth noting", "obviously"). But a draft can clear both and still run a third too long, because ordinary verbosity matches no named trigger: "that already merged via" for "landed on", "with it still in the PR, the same fix lands" for "keeping it re-lands the fix", restated subjects, throat-clearing lead-ins, clauses the reader already has. Those slip the categorical detectors silently — the walk comes back clean while the text is still bloated. So §38 is a real walk step, not a wordlist match: after the named padding, read each sentence and try to delete half of it. Academic registers keep the transition markers, so the aggressive cut stays prose and personal only.
+
+*** Basis
+Corpus-measured across registers (2026-05-29). Single-sentence-paragraph rate: git commits 41.1%, personal email 57.4%, work email 44.5%, PR descriptions 74.4%, PR review comments 50.0%. The terse-paragraph cadence is even more pronounced in conversational and PR-description prose than in commits. Craig writes terse across registers, with the highest density in deliberate PR descriptions where each paragraph carries one focused thought. Confirmed indirectly via paragraph structure across all five corpora.
+
+*** Before
+#+begin_example
+It's worth noting that the change doesn't break the existing flow. Needless to say, the test suite is green. Obviously, this means we can ship.
+#+end_example
+
+*** After
+#+begin_example
+The change doesn't break the existing flow. The test suite is green. We can ship.
+#+end_example
+
+*** Before (generic verbosity, no named padding)
+#+begin_example
+This try/except is the same isolation change that already merged via #203. With it still in the PR, the same production fix lands under a second ticket, which is what the test: label means.
+#+end_example
+
+*** After
+#+begin_example
+This try/except already landed on development via #203. Keeping it re-lands a merged fix under a second ticket, like the test: label says.
+#+end_example
+
+This second pair carries no padding phrase from the named list. Every cut is ordinary verbosity: "is the same isolation change that already merged via" collapses to "already landed on", "With it still in the PR, the same production fix lands" to "Keeping it re-lands a merged fix", "which is what ... means" to "like ... says". A wordlist match finds nothing here; the per-sentence "delete half of it" test finds all of it.
+
+*** Detection
+Two passes. (1) Named padding phrases: "worth noting", "it's important to understand", "as you can see", "needless to say", "obviously", "of course", "in essence", "fundamentally". (2) Ordinary verbosity beyond the list: verbose verb phrases ("already merged via" → "landed on"), restated subjects, throat-clearing lead-ins, and any clause whose content the reader already has. The forcing test for pass 2 is per sentence: try to delete half of it and keep only what changes meaning.
+
+*** History
+- Original SKILL.md entry: rhetorical-padding cut for Craig's authored prose.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+- 2026-06-02: generalized from a named-padding-list detector to a real omit-needless-words walk step. A PR-review comment cleared the §22/§23/§26/§38-padding patterns yet still ran a third too long on ordinary verbosity; the wordlist matched none of it. Renamed "Rhetorical Padding" to "Omit Needless Words", added the per-sentence "delete half of it" forcing test and the generic-verbosity example pair above. Craig's call.
+- 2026-06-05: elevated to the mandatory final pass in the SKILL.md Process (new step 7). The pattern existed and was being walked, but got glossed as one of 41; a commit message went out needing two manual Orwell-walk requests before it read terse. Made it an explicit standalone last step that runs before any draft is shown, so the terse cut happens before Craig sees the draft rather than after he asks for it. Added the "Execution position" subsection above. Craig's call.
+
+** §39 Public-Artifact Scope Check
+
+*** Modes
+Personal mode only. General and prose skip because a private journal or a third-party document has no public-scope concern. Flag only; no auto-rewrite.
+
+*** Rule
+Flag (do not auto-rewrite) local absolute paths, private repo names, and personal-tooling references in publish artifacts. Surface each match as a WARN line so the author resolves manually. Output format:
+#+begin_example
+WARN: line 12: "/home/cjennings/code/rulesets" — local absolute path in commit body
+WARN: line 18: "claude-rules/commits.md" — personal-tooling reference; state the underlying reason instead
+#+end_example
+
+*** Problem
+Commit messages, PR descriptions, PR comments, and Linear ticket bodies are visible to teammates and anyone with read access. References to the writer's personal layout are noise to a reader who cannot reproduce it. Auto-masking risks silently editing meaningful content because a legitimate file path mention may be load-bearing, and only the author can tell.
+
+*** Basis
+Observation-derived (Craig's voice rules in claude-rules/commits.md, Content scope section). Corpus is the public artifacts themselves, so confirmation is circular. Deferred to Phase 2.
+
+*** Detection
+Local absolute paths (=/home/<user>/...=, =/Users/<user>/...=), private repo names (any repo not in this project's known public set), personal-tooling references (humanizer, voice, commits.md, anything under =claude-rules/=, anything under =.ai/= or =.claude/=).
+
+*** History
+- Original SKILL.md entry: public-artifact scope flag for personal mode.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §40 Praise vs Correction Asymmetry
+
+*** Modes
+Personal mode only. General and prose skip because the rule assumes a PR review context.
+
+*** Rule
+Praise on a PR review is short and unjustified (the author knows why their good change is good), and it survives only as an inline pin on the line it refers to. Correction always explains the why, gently and briefly, the way a mentor would, never as a verdict from on high. Keep it brief either way.
+
+On an approve summary: no praise at all, not even a bare positive ("Clean.", "Solid fix."). Lead with the substantive pointer — the design note pinned inline — and close with the verdict; an approve with nothing to flag is just "Approving." "Clean fix on the stacking bug, the tri-state is the right level to solve it at, and the tests cover the edges. Approving." becomes "One design note inline, not a blocker. Approving." (or just "Approving." with nothing to flag). Cut any clause that describes, justifies, or compliments the change — if a clause references what the code does, why it works, or how good it is, delete it.
+
+On a finding or change-request: always give the why, gently and briefly. Not "Move this to a helper." but "I'd pull this into one helper — three copies of the same rule means the next change has to touch all three, and missing one brings the bug back."
+
+Verification narration is the same defect as justified praise. "I traced X and it's safe because..." pads the compliment with the reviewer's homework. Tracing the code is the reviewer's job, not content for the comment — if verification found a problem, the problem gets the words; if it found nothing, it gets zero words.
+
+*** Problem
+Praise and correction call for opposite treatment. The author already knows why their good change is good, so justifying praise reads as flattery. Correction is the reverse. Behavior only changes when the reason lands, so a finding, change-request, or inline coaching note must always explain the why. And the why is delivered gently, the way a kind coach or mentor would.
+
+*** Basis
+Observation-derived (Craig's voice rules in claude-rules/commits.md, Voice and Focus section). PR-review corpus needed for empirical measurement. Deferred to Phase 2.
+
+*** Before
+#+begin_example
+Nice clean migration, the provider mocks and the Normal/Boundary/Error cases are all covered which is exactly what I'd want here. Approving. Also rename `x`.
+#+end_example
+
+*** After
+#+begin_example
+One naming note inline, not a blocker. Approving.
+#+end_example
+
+The rename rationale (`x` reads as a generic placeholder; the next person won't know it's the resolved provider without tracing it) lives in the inline pin, not the summary — the summary points, the pin teaches.
+
+*** Before (verification narration)
+#+begin_example
+All three fixes look right. I traced useMapActions and the unmount cleanup is safe because the hook returns a memoized object, and the provider wraps the whole app so neither call site lands on the no-op path.
+#+end_example
+
+*** After
+#+begin_example
+Approving.
+#+end_example
+
+Nothing to flag, so the summary is the bare verdict. The old "All three fixes are clean and well-aimed" is itself praise, and praise is now cut from the approve summary entirely.
+
+*** Detection
+In a PR review summary or comment: any praise on an approve summary (including a bare positive), a praise clause that explains why the good thing is good, a praise clause followed by the verification work that supports it, or a finding or change-request that states what to fix without saying why.
+
+*** History
+- Original SKILL.md entry: praise-versus-correction asymmetry for PR review.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+- 2026-06-10: verification-narration variant added after the third recurrence — a review draft praised a fix and then narrated the verification supporting the praise (the #236 draft). Added to the SKILL.md rule line and the high-recurrence attestation set. Craig's call, from the work-project session.
+- 2026-07-11: bare-positive carve-out removed. An approve summary now carries no praise at all, not even "Clean." / "Solid fix." — lead with the substantive pointer, close with the verdict. Craig's ruling from a DeepSat review session (approved "One design note inline, not a blocker. Approving."). Same change applied to review-code's Posted Summary Voice and commits.md Shape 1.
+
+** §41 No Emphasis Formatting
+
+*** Modes
+Prose and personal modes. General mode keeps the related but mechanical §14 (boldface strip). §41 carries Craig's own principle and covers italics and underscores too.
+
+*** Rule
+Remove emphasis markup (bold, italics, underscore-wrapped words) used to stress a phrase in Craig's prose, and rephrase so the stress lives in word choice and sentence shape. Structural markup stays: headings, defined terms on first use where the convention is house style, code spans for literal identifiers.
+
+*** Problem
+Craig makes his points with words, not formatting. Emphasis markup is a crutch. When a sentence leans on bold or italics to land, the wording is not doing the work. The fix is not to delete the markup and leave a flat sentence. It is to rephrase so the stress lives in the word choice and sentence shape. This is the same principle behind his terminal-rendering rule in chat, but here it is about the writing itself, not the display.
+
+*** Basis
+Observation-derived (Craig's voice rules in claude-rules/interaction.md, No Reverse-Video Highlighting rule). Org-mode bold uses =*word*= rather than Markdown =**word**= so corpus grep for Markdown emphasis is not directly applicable. Corpus measurement deferred to Phase 2.
+
+*** Before
+#+begin_example
+This is **really** important: you must run the migration *before* deploying, or the app will crash.
+#+end_example
+
+*** After
+#+begin_example
+Run the migration before deploying. Skip that step and the app crashes on the first request.
+#+end_example
+
+*** Detection
+Bold (=**...**=), italic (=*...*= or =_..._=), or underscore-wrapped words used to emphasize a phrase in Craig's prose.
+
+*** History
+- Original SKILL.md entry: no emphasis formatting for Craig's authored prose.
+- 2026-05-29: migrated to this file as the canonical home per the pairing rule.
+
+** §42 Finding Stems — One Claim Per Sentence
+
+*** Modes
+Personal mode only. General and prose skip because the rule assumes a PR review finding.
+
+*** Rule
+A PR review finding is built from clean stems, each a straightforward sentence carrying one claim: (1) where the bug is, (2) the way(s) to fix it, (3) why that's better. Cut context sentences that don't change what the author does next (ticket history, design archaeology). Rewrite the anti-pattern shapes: hedged gerund chains ("the real bug looks like the model emitting a partial set"), compressed trade-off clauses ("I'd rather X, or Y, than lose Z"), multi-claim sentences chained through so-clauses or "and", and fixes buried after a mid-sentence colon.
+
+*** Problem
+Craig named detangling overly complex or overly wordy Claude-drafted PR review text as THE key issue he fights in PR reviews, and the reason he gates every review draft. The tangles passed all 41 then-existing patterns — §38 shortens but doesn't untangle; a sentence can be terse and still carry three claims. §40 governs praise; this governs how finding text is constructed.
+
+*** Basis
+Observation-derived from PR #233 (2026-06-10): a review comment shipped with hedged gerund chains and compressed trade-off clauses that cleared the full walk. The three Before/After pairs below are Craig-approved rewrites from that PR.
+
+*** Before (multi-claim opener + context sentence)
+#+begin_example
+POST fixes the wipe but it's additive: it no-ops on an empty list and never removes, so "cancel all partners" and any de-selection silently stop working. PUT came from SE-195 so the confirm could reconcile the full set. The real bug is upstream: on a new tasking the confirm emits only the new provider, not the full set. Fix that, or merge with the mission's current providers before the PUT. Either way removal keeps working.
+#+end_example
+
+*** After
+#+begin_example
+POST is additive: it no-ops on an empty list and never removes. That breaks "cancel all partners" and any de-selection. The real bug is upstream: on a new tasking the confirm emits only the new provider, not the full set. Fix that, or merge with the mission's current providers before the PUT, and removal keeps working.
+#+end_example
+
+The SE-195 context sentence is cut because it doesn't change the author's next action.
+
+*** Before (hedged gerund chain + compressed trade-off — the calibration case Craig pulled up)
+#+begin_example
+The real bug looks like the model emitting a partial set on a new tasking. I'd rather fix what the confirm emits, or merge client-side before the PUT, than lose removal.
+#+end_example
+
+*** After (where / fix / payoff)
+#+begin_example
+The real bug is upstream: on a new tasking the confirm emits only the new provider, not the full set. Fix that, or merge with the mission's current providers before the PUT. Either way removal keeps working.
+#+end_example
+
+*** Before (claims joined with "and"; fix buried after a mid-sentence colon)
+#+begin_example
+The prefix check catches any message starting with "confirm ", and the options block exists so the LLM can resolve "number 2" style references. A typed "confirm number 2" loses the list it needs. The card click already sends a self-describing "confirm <id>": pass an explicit parameter through sendAgentMessage and strip only on that path.
+#+end_example
+
+*** After
+#+begin_example
+The prefix check strips the options block from any typed message starting with "confirm ", so "confirm number 2" loses the list the LLM needs to resolve it. Strip on the card-click path instead, with an explicit parameter passed through sendAgentMessage. The click already sends a self-describing "confirm <id>", so stripping is safe there.
+#+end_example
+
+*** Detection
+In a PR review finding: a sentence carrying more than one claim (chained through so-clauses, "and", or a mid-sentence colon hiding the fix), a hedged gerund chain where a direct claim belongs, a compressed trade-off clause, or a context sentence that doesn't change the author's next action.
+
+*** History
+- 2026-06-10: created from the PR #233 calibration session. Proposed in the work project's stems handoff, landed via the consolidated voice-skill revision. Included in the high-recurrence attestation set from day one. Craig's call.
+
+** §43 Single-Sentence Paragraph Cadence Is a Feature
+
+*** Modes
+Prose and personal modes. General mode skips because third-party registers legitimately prefer multi-sentence paragraphs.
+
+*** Rule
+A one-sentence paragraph is a finished thought, not a fragment. "Shifts angle" means shifts *topic*: break paragraphs at a topic boundary, even when both sides are one sentence. Within a single topic, consolidate its sentences into one paragraph even when each is complete, up to a ceiling of about five or six sentences, past which find a natural break. The never-merge instruction protects the break *between* topics; it never licenses fragmenting one topic across several paragraphs.
+
+*** Problem
+Most prose-style guides advise multi-sentence paragraphs, so a generic cleanup pass merges Craig's short paragraphs and erases a distinctive feature of his voice. This is a protective pattern: it guards an existing trait rather than correcting a defect.
+
+The 2026-07-23 boundary refinement addresses the opposite failure, discovered the same day: reading "angle" at *sentence* granularity, so that three sentences all about one topic (a baking run, a mixer, tortillas) got split into three paragraphs as if each were a new angle. That fragments one topic and reads as a checklist rather than a person talking. Angle means topic. #43 governs the break between topics; consolidation fills in what happens within one, which the original rule never specified. The two are one rule seen from both sides, not a rule in tension with #47.
+
+*** Basis
+Corpus-measured (2026-05-29). Single-sentence-paragraph rate: git commits 41.1%, personal email 57.4%, work email 44.5%, PR descriptions 74.4%, PR review comments 50.0%. Between 41% and 74% of Craig's paragraphs are exactly one sentence, depending on register.
+
+*** Before (a cleanup pass merging short paragraphs)
+#+begin_example
+The build now finishes in half the time, and the cache no longer invalidates on every run, which means the CI queue clears faster too.
+#+end_example
+
+*** After
+#+begin_example
+The build now finishes in half the time.
+
+The cache no longer invalidates on every run, so the CI queue clears faster too.
+#+end_example
+
+*** Detection
+An edit pass that merged short paragraphs, or a draft whose paragraphs each stack multiple shifted angles that would read better broken apart.
+
+*** History
+- 2026-05-29: surfaced by the Phase 1-2 corpus measurement as a "worth adding" trait; filed as suggested delta 4.
+- 2026-06-10: promoted from the suggested-deltas list into a numbered pattern. Craig's call, from the work-project session.
+- 2026-07-23: boundary refined (angle means topic; within-topic consolidation to a ~5-6 sentence ceiling; never-merge reframed as across-topic protection). From a home session drafting a Signal reply, where the original rule was misread at sentence granularity. Paired with the new §47.
+
+** §44 Parenthetical Asides Are Part of the Voice
+
+*** Modes
+Prose and personal modes. General mode skips because third-party text owns its own aside conventions.
+
+*** Rule
+Parentheses for asides, clarifications, and scope-narrowing are Craig's voice. Don't strip them in a cleanup pass. They're also the preferred landing spot for em-dash replacements under §13.
+
+*** Problem
+Generic style passes treat parentheticals as clutter and strip them. For Craig they carry asides, clarifications, and scope-narrowing, and removing them flattens the voice. Protective pattern, like §43.
+
+*** Basis
+Corpus-measured (2026-05-29): 23.07 opening parens per 1000 words across the commit corpus. Heavy parenthetical use is distinctive and consistent.
+
+*** Before (a cleanup pass stripping the aside)
+#+begin_example
+The sync runs on every startup. It skips lockfiles. It also skips build output.
+#+end_example
+
+*** After
+#+begin_example
+The sync runs on every startup (skipping lockfiles and build output).
+#+end_example
+
+*** Detection
+An edit pass that removed parenthetical asides present in the source text, or an em-dash replacement under §13 where parentheses fit better than a comma or period.
+
+*** History
+- 2026-05-29: surfaced by the Phase 1-2 corpus measurement as a "worth adding" trait; filed as suggested delta 5.
+- 2026-06-10: promoted from the suggested-deltas list into a numbered pattern, with the §13 landing-spot note. Craig's call, from the work-project session.
+
+** §45 Declarative Register Marker
+
+*** Modes
+Prose and personal modes, advisory. Flag only; no auto-rewrite. General mode skips.
+
+*** Rule
+Craig's prose is declarative. When a draft contains a rhetorical question, flag it for a second look — it's usually AI rhetoric, not his register. Genuine questions to the reader (a review asking the author's intent, an email asking for a decision) stay.
+
+*** Problem
+AI drafts reach for rhetorical questions ("So what does this mean for the build?") as a transition device. Craig states things; he rarely asks them. A rhetorical question in his voice is a tell, but a genuine question is legitimate content, so the pattern flags rather than rewrites.
+
+*** Basis
+Corpus-measured (2026-05-29): 0.33 question marks per 1000 words across the commit corpus. His prose register is declarative.
+
+*** Before (rhetorical transition flagged)
+#+begin_example
+So what does this change for the deploy flow? The staging gate now runs before the canary, which means a bad build never reaches it.
+#+end_example
+
+*** After
+#+begin_example
+The staging gate now runs before the canary, so a bad build never reaches it.
+#+end_example
+
+*** Detection
+A question mark in a draft in Craig's voice. Flag it; keep genuine questions to the reader, rewrite rhetorical ones as declarative claims.
+
+*** History
+- 2026-05-29: surfaced by the Phase 1-2 corpus measurement as a register marker; filed as suggested delta 6.
+- 2026-06-10: promoted from the suggested-deltas list into a numbered advisory pattern. Craig's call, from the work-project session.
+
+** §46 Comma Budget — Max Two Per Sentence
+
+*** Modes
+Personal mode only. Prose and general modes skip.
+
+*** Rule
+No sentence carries more than two commas. Rewrite the third comma away: split the sentence, move a clause into a parenthetical (§44) or behind a colon, or break an inline serial list into bullets or its own sentence. Count prose commas only — commas inside code spans, quoted log lines, and literal strings (version numbers, paths) don't count toward the budget.
+
+*** Problem
+Three or more commas in one sentence almost always mark stacked clauses or an inline list doing a paragraph's work. The sentence reads fine to its author and lands as a pileup on the reader. The comma count is a mechanical proxy the walk can enforce, where "don't stack clauses" is prose advice that gets skipped.
+
+*** Basis
+Craig's directive, 2026-07-20 (archsetup session, while gating a Hyprland issue draft): "no more than two commas per sentence. we should add that to the /voice personal pass."
+
+*** Before (spec-sheet line with three commas, from the draft that prompted the rule)
+#+begin_example
+System: Arch Linux, kernel 6.18.25-lts, AMD Strix Halo (Radeon 8060S), no plugins loaded.
+#+end_example
+
+*** After
+#+begin_example
+System: Arch Linux, kernel 6.18.25-lts. GPU: AMD Strix Halo (Radeon 8060S). No plugins loaded.
+#+end_example
+
+*** Detection
+Count commas per sentence on the final text. A sentence at three or more gets restructured, not trimmed to exactly the budget — the third comma is the symptom, the stacked structure is the target.
+
+*** History
+- 2026-07-20: added at Craig's direction from the archsetup session. Scoped to personal mode; broaden to prose only if he asks. Added to the attestation high-recurrence set at birth — a mechanical count is cheap to receipt, and new discipline fails silently without one.
+
+** §47 Recipient-Priority Ordering
+
+*** Modes
+Prose mode, and only when the piece is correspondence (email, Signal, a letter). It needs a recipient, so it has no referent in a journal, a working note, or any document addressed to nobody, and it does not carry into personal mode — a commit or PR review is not a reply to someone's news. General mode skips it with the rest of Craig's voice patterns. This is the one pattern narrower than a whole mode, and the only prose pattern personal mode does not also walk.
+
+*** Rule
+In a reply, lead with what matters most to the recipient, not with what's easiest to answer or the order they wrote it. Their news outranks your logistics. A direct question they asked can sort below personal news they shared, because the news is what they care about.
+
+*** Problem
+The easy draft answers the explicit question first and orders the rest as it arrived. That reads as transactional — logistics before the person. Ordering by what the recipient cares about is what makes a reply read as one person talking to another rather than a ticket being closed. Nothing else in the skill governs the *order* of a reply's contents; the other patterns act within a paragraph or a sentence.
+
+*** Basis
+Craig's edit of a Signal reply to his sister, 2026-07-23 (home session). His framing: "start with what would be the most important things to her."
+
+*** Before (first draft — opens with the only explicit question, cooking split across three paragraphs)
+#+begin_example
+Yes, I do subscribe to MasterClass — happy to share what I've watched.
+
+That's amazing about the sourdough. English muffins from scratch is no joke.
+
+The home-roasted deli meat sounds incredible.
+
+A stand mixer would make the bread a lot easier — worth it if you're baking this much.
+
+And 30 pounds — that's huge. So happy for you.
+#+end_example
+
+*** After (Craig's order — weight first, one cooking paragraph, then the question, then the close)
+#+begin_example
+Thirty-plus pounds — that is huge, and I'm so happy for you. That's real work.
+
+And the cooking. Sourdough, English muffins, tortillas, home-roasted deli meat from scratch — that's a whole kitchen you've built, and a stand mixer would make the bread much easier if you're baking at this volume, so I say go for it. I want to hear how the tortillas come out.
+
+Yes, I subscribe to MasterClass — I'll send you what I've been watching.
+
+I miss you and I love you. Send me a few times that work for a call.
+#+end_example
+
+The cooking paragraph runs seven sentences, a hair over the §43 ceiling. Craig called it an exception rather than re-cut a message that had already gone out. The guard is the rule; this paragraph is one sentence over it; both facts stay in the record, because a real example at the boundary teaches it better than a clean one.
+
+*** Detection
+A reply whose opening answers a logistical or yes/no question while the recipient's substantive news sits lower. Reorder so the news they'd most want acknowledged leads.
+
+*** History
+- 2026-07-23: added from the home session drafting a Signal reply. The first handoff proposed two new patterns and flagged a conflict with §43; the superseding design resolved that the conflict was a misreading of §43 (angle = topic), leaving one genuinely new pattern here and a calibration to §43. Prose/correspondence-scoped per Craig — email and Signal are prose, not publish artifacts.
+
+** §48 Term-Translation Density
+
+*** Modes
+General mode, so it runs in all three (general, prose, personal). It's a universal clarity rule in the Orwell / Plain English family, not a Craig-voice trait, and it reads to anyone editing any prose. The later number is an artifact of when it was added, not a scope signal.
+
+*** Rule
+A sentence that forces the reader to stop and translate more than one specialized term (an acronym, a coined phrase, a product name) is too dense. One is fine; two or more in one sentence means rewrite: split the sentence, gloss one term in a parenthetical, or drop to plain language. The test is the reader's parse, not the writer's familiarity, and it is audience-relative.
+
+*** Problem
+A writer fluent in the domain doesn't feel the translation cost of the terms, so a sentence stacking three of them reads as normal to the author and stalls the reader on every clause. Density is the metric, not any single word: two coined terms in one sentence is worse than a paragraph that introduces the same two one at a time. Audience-relative, because SAR to a defense team is shared vocabulary carrying no load, while a coined phrase only the writer holds carries full load for everyone else.
+
+*** Basis
+Craig's edit of a customer-partner email, 2026-07-23 (work session), where one sentence stacked three terms and he flagged it. Distinct from #7 (specific AI-vocabulary words) and #30 (telegraphic fragments); this measures jargon density per sentence.
+
+*** Before (one sentence, three terms the reader must translate)
+#+begin_example
+A ViT detector gating a VLM for enrichment is close to our own detect-then-contextualize direction.
+#+end_example
+
+*** After (split, glossed, plain)
+#+begin_example
+Their setup is a fast detector that hands off to a heavier model for a closer read. That mirrors our own two-stage approach (find it first, then work out what it is).
+#+end_example
+
+*** Detection
+Count the specialized terms in each sentence that a member of the intended audience would have to stop and translate. Two or more is the trigger. Acronyms, coined phrases, and product names count; shared-vocabulary terms for that audience don't.
+
+*** History
+- 2026-07-23: proposed by Craig from the work session, drafting a customer-partner email. Placed in general mode (universal clarity rule); numbered #48, after the prose-only #47.
diff --git a/working/working-dir-orphan-check/proposal-from-work.org b/working/working-dir-orphan-check/proposal-from-work.org
new file mode 100644
index 0000000..d7d82b2
--- /dev/null
+++ b/working/working-dir-orphan-check/proposal-from-work.org
@@ -0,0 +1,12 @@
+#+TITLE: Convention proposal from Craig's roam inbox (2026-07-23 inbo
+#+SOURCE: from work
+#+DATE: 2026-07-23 15:28:00 -0500
+
+Convention proposal from Craig's roam inbox (2026-07-23 inbox-zero from the work project):
+
+Every project should have a working/ and a temp/ directory.
+- working/ holds files currently being worked on before archiving, in a subdirectory named for the project/task. NOT gitignored — pushed to remote.
+- temp/ holds throwaway files (discarded prototypes, etc). Gitignored, not pushed, deleted as part of the wrap-up sequence.
+- When a task completes, any files it left in working/ are always (via a soft hook if possible) archived or moved to temp/. No files related to a DONE task/project should remain in working/.
+
+Note for the skeptical review: this convention already largely exists in working-files.md (working/ tracked from creation; temp/ gitignored for throwaway). The genuinely new asks here are (a) the automatic soft-hook that empties working/ on task DONE, and (b) temp/ deletion wired into wrap-up. Worth checking what's already implemented vs what this adds before treating it as net-new.
diff --git a/working/working-dir-orphan-check/proposed.diff b/working/working-dir-orphan-check/proposed.diff
new file mode 100644
index 0000000..03452dc
--- /dev/null
+++ b/working/working-dir-orphan-check/proposed.diff
@@ -0,0 +1,24 @@
+--- claude-templates/.ai/workflows/wrap-it-up.org 2026-07-23 08:39:07.467575608 -0500
++++ /tmp/wu2.org 2026-07-23 20:49:55.500740517 -0500
+@@ -189,6 +189,21 @@
+ emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done --check todo.org
+ #+end_src
+
++*** Flag orphaned working/ directories
++
++#+begin_src bash
++for d in working/*/; do
++ [ -d "$d" ] || continue
++ s=$(basename "$d")
++ grep -q "working/$s/" todo.org 2>/dev/null || { echo "ORPHAN: $d (no todo.org reference)"; continue; }
++ awk -v s="$s" '/^\*\* /{h=$0} $0 ~ "working/" s "/" {print (h ~ /^\*\* (DONE|CANCELLED)/ ? "CLOSED-TASK: working/" s "/ — " h : "") ; exit}' todo.org
++done
++#+end_src
++
++Report only — never move or delete anything here. A =working/<slug>/= whose backing task is closed (or which no task references at all) is a filing candidate per =working-files.md=: rename each file individually and move it flat into its permanent home, then remove the empty directory.
++
++Filing is deliberately a judgment step and stays manual. Deciding each artifact's permanent home and giving it a meaningful name is what makes =assets/= searchable later, and it's also the moment you notice which artifacts aren't worth keeping. An automatic sweep would skip exactly that review, and sweeping to =temp/= would destroy the artifacts outright, since =temp/= is cleared just below.
++
+ *** Clear temp/
+
+ #+begin_src bash
diff --git a/working/working-dir-orphan-check/wrap-it-up.org.proposed b/working/working-dir-orphan-check/wrap-it-up.org.proposed
new file mode 100644
index 0000000..ff50a3f
--- /dev/null
+++ b/working/working-dir-orphan-check/wrap-it-up.org.proposed
@@ -0,0 +1,651 @@
+#+TITLE: Session Wrap-Up Workflow
+#+AUTHOR: Craig Jennings
+#+DATE: 2026-04-20
+
+* Overview
+
+This workflow defines the process for ending a Claude Code session cleanly. It finalizes the session record, commits + pushes all work, and provides a warm handoff. A bare wrap also tears the session down (kills the ai-term buffer + tmux session, restoring geometry); a qualified wrap keeps the buffer, and a shutdown wrap powers the machine off. The teardown variants are set by the trigger phrase (see Teardown mode below) and act only at the very end, in Step 6.
+
+Triggered by Craig saying "wrap it up," "that's a wrap," "let's call it a wrap," or similar.
+
+* The Session Record
+
+Throughout the session, =.ai/session-context.org= has been maintained with:
+- =* Summary= — structured distillation (empty or draft during session)
+- =* Session Log= — chronological narrative of what happened, written as you go
+
+At wrap-up, this file becomes the permanent session record by being renamed to =.ai/sessions/YYYY-MM-DD-HH-MM-description.org=. No transcription elsewhere. The file IS the record.
+
+* Exit Criteria
+
+The wrap-up is complete when:
+
+1. *Summary is written.* The =* Summary= section of =.ai/session-context.org= is populated by reading the =* Session Log= — Active Goal, Decisions, Data Collected / Findings, Files Modified, Next Steps.
+2. *File is archived.* =.ai/session-context.org= has been renamed to =.ai/sessions/YYYY-MM-DD-HH-MM-description.org=. The old path no longer exists.
+3. *todo.org is clean.* Cleanup script ran. Any auto-fixes are staged for the wrap-up commit. Orphan planning lines surfaced for manual fix if there are any.
+4. *Linear board is honest* (skip if project doesn't use Linear). Any Dev-Review ticket whose PR has merged was moved to Done or PM Acceptance per the classification rule.
+5. *Git state is clean.* All changes committed + pushed to all remotes. Working tree clean.
+6. *Valediction delivered.* Brief, warm closing with key accomplishments and reminders, ending with =session wrapped.= on its own line as the signoff marker.
+
+The absence of =.ai/session-context.org= is the signal that the last session wrapped up cleanly. Its presence at session start means the previous session was interrupted.
+
+* Teardown mode (set from the trigger phrase)
+
+The wrap itself — Steps 1 through 5 — is identical in every mode. The trigger phrase only decides what Step 6 does once commit + push and the valediction are done. Resolve the mode from the phrase before starting:
+
+- *Teardown* (the default) — bare "wrap it up", "that's a wrap", "let's call it a wrap". The full wrap, then Step 6 kills the ai-term buffer + the =aiv-<project>= tmux session (which takes =claude= with it) and restores the saved window geometry. This is Craig's typical end-of-day case.
+- *No-teardown* — "wrap it up with summary" or "wrap it up and summarize". The full wrap, but Step 6 leaves the buffer and session intact so the summary stays readable. The explicit qualifier is what opts out of teardown.
+- *Shutdown* — "wrap it up and shutdown". The full wrap, then Step 6 gates on this being the only live ai-term session and powers the machine off. Shutdown supersedes teardown (killing the buffer is moot if the box is going down).
+
+Why teardown waits for Step 6 and runs through a hook, never inline: teardown kills the very tmux session =claude= runs in, so an inline kill would cut the valediction off before it renders. Step 6 instead drops a sentinel after everything else is verified, and the =Stop= hook (=ai-wrap-teardown.sh=) does the actual teardown when this response ends — by which point the valediction has already been delivered.
+
+This depends on three functions in =.emacs.d/modules/ai-term.el= (=cj/ai-term-quit=, =cj/ai-term-live-count=, =cj/ai-term-shutdown-countdown=) and on the =Stop= hook being wired in =settings.json= (=hooks/settings-snippet.json=). If =emacsclient= or the daemon is unreachable, the sentinel is cleared and the session simply stays up — teardown degrades to a no-op, never a wedge.
+
+* The Workflow
+
+** Step 0: Refuse if sentry is live
+
+Before anything else, check whether sentry is running in this project. Sentry holds the working tree on its =sentry/<date>-<host>= branch and commits unattended; wrapping underneath it would archive the session anchor and tear down the buffer while the loop is still firing into it. If sentry's single-runner lock is held, stop and point at the shutdown path:
+
+#+begin_src bash
+proj="$(basename "$(git rev-parse --show-toplevel 2>/dev/null || pwd)")"
+if [ -x .ai/scripts/agent-lock ] && .ai/scripts/agent-lock status "sentry-$proj" | grep -q '^held'; then
+ echo "sentry is active — say 'stop sentry' first"
+ exit 1
+fi
+#+end_src
+
+The stop-sentry operation (defined in =sentry.org=) owns the shutdown: it cancels the loop, disposes of the branch, and walks the approval queue. Wrap-up carries only this one guard; a =stale= lock (a crashed fire) doesn't block — only a live =held= lock does.
+
+** Step 1: Finalize the Summary
+
+*** Work the Before-Close Queue (before the Summary)
+
+If the session anchor (=.ai/session-context.org=) carries a =* Before-Close Queue= heading with items, work them now, oldest-first, before writing the Summary, so any resulting edits ride this wrap's commit and get described in it. The queue is the "put X on the list" shorthand (see =protocols.org=, Colloquialisms and Expansions): session-scoped work Craig deferred to wrap time.
+
+Per item: do it if it's clear and bounded, or promote it to a =todo.org= task if it turns out to need its own session. Never drop an item silently. Remove each line as it's handled; if one can't be finished, surface it in the valediction (Step 5) and either leave a follow-up task or state why it's dropped.
+
+If there's no =* Before-Close Queue= heading, or it's empty, this step is a silent no-op.
+
+*** Early KB reflection (capture while fresh, before the Summary)
+
+Before distilling the Summary, while the session is still fresh, ask: what did this session learn worth remembering, for yourself or a future agent? Reflect and stage any candidate durable facts — a decision and its why, an environment gotcha, a reference pointer, a transferable lesson. Self-answer silently; this adds no interactive turn (Craig already authorized the wrap). The candidates flow straight into the KB promotion check below, which does the actual writing and the receipt — this is the capture half, that is the commit half, one pipeline, one receipt. Reflecting here rather than reconstructing learnings after the Summary is the point: the early ask is what keeps the receipt from defaulting to "promoted 0" out of fatigue.
+
+Read through the =* Session Log= in =.ai/session-context.org=. Populate (or refine) the =* Summary= section:
+
+- *Active Goal* — one or two sentences describing the session's focus
+- *Decisions* — key choices made, with enough context to recall the /why/
+- *Data Collected / Findings* — anything concrete (measurements, root causes, paths, discoveries)
+- *Files Modified* — what was changed, with one-line rationale per significant file
+- *Next Steps* — what should happen in the next session
+
+Don't repeat everything from the Log in the Summary. The Summary is distillation — pull out what's load-bearing. The Log stays in the file and is available if a future reader wants detail.
+
+*** KB promotion check (and the one-line instrumentation receipt)
+
+Before closing the Summary, ask: did this session learn anything worth promoting to the agent knowledge base? The bar is =knowledge-base.md='s inclusion criteria — durable facts with cross-project or cross-machine value (decisions and their why, environment gotchas, reference pointers, transferable lessons). Promote each qualifying fact as one =agents/= node per the rule's schema (work-classified projects skip the write per the boundary; the check still runs so the receipt below is honest).
+
+Then add one line at the end of the Summary, always, even when nothing moved:
+
+#+begin_example
+KB: promoted 2 / consulted yes
+#+end_example
+
+"promoted N" counts nodes written this session (0 most sessions); "consulted yes-no" records whether any KB query informed the session's work. The line is the input to the spec's 30-day success-metrics checkpoint — grepping session archives for =KB:= answers "are agents actually using this?" without any other instrumentation. A session that skips the line breaks the metric, so it's part of the Summary contract, not optional.
+
+** Step 2: Pick a description + rename
+
+Read the Summary's Active Goal and the prominent entries in the Session Log. Pick a 4-6 word description that would make sense as a git-commit-message-series summary for the whole session.
+
+Good descriptions are concrete nouns/verbs:
+- =docs-ai-migration-and-ai-launcher=
+- =mybitch-usb-disconnect-diagnosis=
+- =ratio-system-health-check=
+- =orchestration-dashboard-bug-triage=
+
+Avoid vague ones:
+- =session-work= (useless)
+- =various-improvements= (useless)
+- =updates= (useless)
+
+Get current time and rename:
+
+#+begin_src bash
+mkdir -p .ai/sessions
+now=$(date +%Y-%m-%d-%H-%M)
+# Resolve the AI_AGENT_ID-aware source path (see protocols.org "Agent-scoped
+# path"); fall back to the singleton if the helper isn't present.
+sc=$(.ai/scripts/session-context-path 2>/dev/null || echo .ai/session-context.org)
+# Under multi-agent, fold the agent id into the archive name so two agents
+# wrapping in the same minute don't collide. Single-agent: no segment.
+idseg="${AI_AGENT_ID:+${AI_AGENT_ID}-}"
+mv "$sc" ".ai/sessions/${now}-${idseg}DESCRIPTION.org"
+#+end_src
+
+Replace =DESCRIPTION= with your picked slug. (=AI_AGENT_ID= should be filename-safe and unique per run; the recommended =host.project.runtime.<epoch>= shape is both. The epoch on the tail keeps a re-run of the same logical agent from resolving to a prior run's leftover anchor. See protocols.org "Agent-scoped path".)
+
+** Step 3: todo.org cleanup (hygiene + archive completed work)
+
+If the project has a =todo.org= at its root, run the cleanup script before committing. Two passes, both fast and idempotent: a hygiene pass and an archive pass.
+
+*** Roam inbox sweep (inbox roam mode)
+
+Before the cleanup scripts, sweep the roam global inbox (=~/org/roam/inbox.org=) for items that belong to this project, so any imported tasks get linted and ride the wrap commit. Delegate to [[file:inbox.org][inbox.org]] roam mode for the claimed set.
+
+#+begin_src bash
+[ -f "$HOME/org/roam/inbox.org" ] && grep -cE '^\*\* ' "$HOME/org/roam/inbox.org" || true
+#+end_src
+
+Skip-fast when nothing matches: if the roam clone isn't on this machine, or no item is prefixed for this project, this is a silent no-op. When claimed items exist, run roam mode's Phase B–D (file each into =todo.org=, then remove them from the shared inbox and let =roam-sync= commit + push the edit). Report the total count and how many appeared related to this project, per roam mode's scan-summary rule.
+
+*** Hygiene pass
+
+It catches a recurring pattern: org sometimes leaves noise lines like =- State "X" from "X" [date]= when a state-change log lands outside a =:LOGBOOK:= drawer and the state didn't actually change. These lines carry no information and they break org's planning-line parser by wedging between the heading and =DEADLINE:=/=SCHEDULED:=, which kicks the entry out of agenda views.
+
+#+begin_src bash
+[ -f todo.org ] && emacs --batch -q -l .ai/scripts/todo-cleanup.el todo.org
+#+end_src
+
+The script is fast (under half a second on a 4000-line file) and idempotent — if there's nothing to fix, it reports zero changes and exits clean.
+
+What it does:
+
+1. *Auto-deletes* bogus state-log lines (matched on identical from/to states). Any deletions show up in the wrap-up commit's diff, so they get reviewed before push.
+2. *Reports* "orphan planning lines" — entries whose body has =DEADLINE:= or =SCHEDULED:= but =org-entry-get= can't read it (some other malformation kept it out of canonical position). The script doesn't auto-rewrite these because the right fix depends on whether real state-log history needs preserving — surface them and fix manually if they matter for the agenda.
+
+Run the report-only variant first if you want to see what would change without writing:
+
+#+begin_src bash
+emacs --batch -q -l .ai/scripts/todo-cleanup.el --check todo.org
+#+end_src
+
+*** Convert done sub-tasks to dated entries
+
+#+begin_src bash
+[ -f todo.org ] && emacs --batch -q -l .ai/scripts/todo-cleanup.el --convert-subtasks todo.org
+#+end_src
+
+=--convert-subtasks= rewrites every heading at level 3 or deeper whose TODO state is DONE/CANCELLED/FAILED into a dated event-log entry (=<stars> YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <text>=), dropping the keyword, priority cookie, and tags, and removing the now-redundant =CLOSED:= line. This enforces the =todo-format.md= depth rule that a completed *sub-task* (a heading under a parent task) becomes dated history, not a lingering DONE keyword — a shape an interactive org close (=org-log-done= → DONE + CLOSED) never applies and =--archive-done= (level-2 only) never reaches. The timestamp comes from each entry's own =CLOSED= cookie; a date-only close yields =00:00:00=. Heading text is kept verbatim. Idempotent (an already-dated heading has no keyword to match), and a done sub-task with no parseable =CLOSED= is flagged and left alone rather than stamped with a fabricated date.
+
+Run this *before* =--archive-done= so that when a completed level-2 parent is archived, its sub-tasks already carry their dated form. Any rewrites show up in the wrap-up commit's diff for review before push.
+
+Preview without writing:
+
+#+begin_src bash
+emacs --batch -q -l .ai/scripts/todo-cleanup.el --convert-subtasks --check todo.org
+#+end_src
+
+*** Archive completed work
+
+#+begin_src bash
+[ -f todo.org ] && emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done todo.org
+#+end_src
+
+=--archive-done= moves every level-2 subtree whose TODO state is DONE or CANCELLED out of the project's "Open Work" section and into its "Resolved" section, subtree intact. The two sections are matched by a unique level-1 heading containing "Open Work" (case-insensitive) and one containing "Resolved" — if either is missing or ambiguous, the file is skipped with a message, no crash. Only direct level-2 children move; a DONE entry nested under an open parent stays put. Idempotent; any moves show up in the wrap-up commit's diff for review before push.
+
+Preview the moves without writing:
+
+#+begin_src bash
+emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done --check todo.org
+#+end_src
+
+*** Flag orphaned working/ directories
+
+#+begin_src bash
+for d in working/*/; do
+ [ -d "$d" ] || continue
+ s=$(basename "$d")
+ grep -q "working/$s/" todo.org 2>/dev/null || { echo "ORPHAN: $d (no todo.org reference)"; continue; }
+ awk -v s="$s" '/^\*\* /{h=$0} $0 ~ "working/" s "/" {print (h ~ /^\*\* (DONE|CANCELLED)/ ? "CLOSED-TASK: working/" s "/ — " h : "") ; exit}' todo.org
+done
+#+end_src
+
+Report only — never move or delete anything here. A =working/<slug>/= whose backing task is closed (or which no task references at all) is a filing candidate per =working-files.md=: rename each file individually and move it flat into its permanent home, then remove the empty directory.
+
+Filing is deliberately a judgment step and stays manual. Deciding each artifact's permanent home and giving it a meaningful name is what makes =assets/= searchable later, and it's also the moment you notice which artifacts aren't worth keeping. An automatic sweep would skip exactly that review, and sweeping to =temp/= would destroy the artifacts outright, since =temp/= is cleared just below.
+
+*** Clear temp/
+
+#+begin_src bash
+[ -d temp ] && find temp -mindepth 1 -delete && echo "temp/ cleared"
+#+end_src
+
+=temp/= holds throwaway artifacts — discarded prototypes, scratch output, intermediate data (see =working-files.md=). It's gitignored in every project, so nothing here rides a commit and nothing is recoverable from git once deleted. Clearing it at wrap is what keeps ephemeral work from silting up across sessions, and it's the counterpart to =working/=, which is tracked and *never* cleared here.
+
+Two guards. Confirm before deleting if =temp/= holds anything a reasonable reader would call in-progress rather than throwaway — misfiled work belongs in =working/=, so move it there instead of deleting it. And skip the step entirely in a project where =temp/= is not gitignored, since that means the project is using the directory for something else.
+
+*** Sync child priorities
+
+#+begin_src bash
+[ -f todo.org ] && emacs --batch -q -l .ai/scripts/todo-cleanup.el --sync-child-priority todo.org
+#+end_src
+
+=--sync-child-priority= walks every heading with a priority cookie =[#A]=–=[#D]= and, for each of its direct child headings whose own priority cookie is /lower/ (later in the alphabet — D is below A), bumps the child to match the parent. Down-only: parents are never bumped up to match a higher-priority child. Children without a priority cookie are left alone, as are parents without one. The walk visits parents before descendants, so a multi-level chain (=[#A]= → =[#B]= → =[#D]=) collapses to the top priority in a single pass. Idempotent.
+
+Opt-out for deliberately-lower children: tag the heading =:no-sync:= (the literal six-character tag, including the hyphen). The script matches the tag literally on the heading line, so it works whether or not the surrounding emacs config has extended =org-tag-re= to allow hyphens.
+
+#+begin_example
+*** TODO [#D] Follow-up: VAD :no-sync:
+#+end_example
+
+Use this for =Follow-up:=, =Spike:=, =Stretch:= sub-tasks that are deliberately deprioritized below their parent — without the tag, the wrap-up would silently bump them back up.
+
+Preview the bumps without writing:
+
+#+begin_src bash
+emacs --batch -q -l .ai/scripts/todo-cleanup.el --check-child-priority todo.org
+#+end_src
+
+(=--check-child-priority= is the report-only alias for =--sync-child-priority --check=.)
+
+*** Lint org files (mechanical sweep, judgments deferred)
+
+#+begin_src bash
+if [ -n "$LINT_ORG_FOLLOWUPS" ]; then
+ followups="$LINT_ORG_FOLLOWUPS"
+elif [ -d "./inbox" ]; then
+ followups="./inbox/lint-followups.org"
+else
+ followups=".ai/lint-followups.org"
+fi
+[ -f todo.org ] && emacs --batch -q -l .ai/scripts/lint-org.el \
+ --fix --followups-file="$followups" todo.org
+#+end_src
+
+The =--fix= flag is required for the writes: lint-org's CLI default is
+report-only (a linter reports, it doesn't write), and this wrap-up pass is
+the deliberate exception that applies fixes — its diff rides the wrap-up
+commit for review.
+
+=lint-org= runs =org-lint= over =todo.org=, auto-applies four mechanical
+categories (=item-number= counters, bare =#+begin_src= → =#+begin_example=,
+multi-line planning-info merged onto one line, =**X.**= → =*X.*=), and
+appends every remaining judgment item (broken file links, invalid fuzzy
+links, verbatim-asterisk inside body prose, suspicious src-block languages)
+to the follow-ups file as a dated org section. Mechanical fixes show up in
+the wrap-up commit's diff for review before push.
+
+The follow-up path defaults to =./inbox/lint-followups.org= in the current
+project (where the next morning's daily-prep merges it in). If the project
+doesn't have an =inbox/= directory, the script falls back to
+=.ai/lint-followups.org= inside the current project. Override with
+=LINT_ORG_FOLLOWUPS=<path>= in the environment if needed — useful for
+routing all wrap-up output to a single shared inbox across projects.
+
+Each project's own =inbox/= is the right default because daily-prep reads
+that project's inbox at startup. Hardcoding a single project's path
+(formerly =~/projects/work/inbox/=) routed every project's wrap-up findings
+into the wrong inbox.
+
+Preview without writing — same flags as =--check= on the other scripts:
+
+#+begin_src bash
+[ -f todo.org ] && emacs --batch -q -l .ai/scripts/lint-org.el --check todo.org
+#+end_src
+
+The wrap-up never blocks on judgment items — they're deferred by design.
+For an interactive walk of the judgments mid-day, run =/lint-org todo.org=.
+
+*** Inbox sanity check (surface unprocessed handoffs)
+
+If the project has an =inbox/= directory, verify it holds nothing but =.gitkeep=, =lint-followups.org= (the lint-org pipeline file the next morning's daily-prep consumes), and any explicitly-deferred =PROCESSED-*= files before the wrap completes. An inbox that arrived at session start with handoffs from other projects, or that received handoffs mid-session, needs =inbox.org= process mode to run and apply its value-gate dispositions. Wrapping with a dirty inbox silently defers the work to next session and accumulates handoff debt that the sender can't see.
+
+#+begin_src bash
+unprocessed=$(find inbox -maxdepth 1 -type f \
+ ! -name '.gitkeep' \
+ ! -name 'lint-followups.org' \
+ ! -name 'PROCESSED-*' \
+ 2>/dev/null | wc -l)
+if [ "$unprocessed" -gt 0 ]; then
+ echo "wrap-up: inbox/ has $unprocessed unprocessed item(s). Run inbox.org process mode before wrapping, or explicitly defer each item with a one-line reason in the valediction."
+ find inbox -maxdepth 1 -type f \
+ ! -name '.gitkeep' \
+ ! -name 'lint-followups.org' \
+ ! -name 'PROCESSED-*' \
+ -printf ' %f\n'
+fi
+#+end_src
+
+If the count is zero or the project has no =inbox/= directory, the check is a silent no-op. If non-zero, the wrap is incomplete by default. The user resolves each item (process now, defer with reason in the valediction, or delete with rationale) before the validation checklist passes.
+
+The check exempts =lint-followups.org= explicitly because lint-org runs earlier in the same wrap-up workflow and writes its judgment items to that file in =inbox/= by design. The file is a pipeline artifact for the next morning's =daily-prep=, not a handoff that needs the value gate.
+
+This integrates with =inbox.org= process mode, which stamps =:LAST_INBOX_PROCESS:= in =notes.org='s *Workflow State* section on completion. Wrap-up doesn't double-stamp. It only ensures the inbox carries nothing but the expected pipeline artifacts at session end.
+
+*** Cross-project router (optional — route filed keepers to their home projects)
+
+Runs directly after the inbox sanity check. The split between the two: the sanity check *gates* the wrap (a dirty inbox blocks until resolved); the router is *optional* (skipping it never blocks anything — the candidates just stay local until a future wrap). Spec: =docs/specs/wrapup-routing-spec.org= (D7/D8/D9).
+
+The candidate set is exactly the local tasks carrying a =:ROUTE_CANDIDATE:= property — keepers that inbox process mode filed this session whose inferred home is another project. Never scan the standing backlog.
+
+#+begin_src bash
+.ai/scripts/route-batch --list
+#+end_src
+
+*Empty set = zero interaction.* =--list= prints nothing when there are no candidates; continue the wrap silently — no prompt, no "0 items" line.
+
+When candidates exist, surface the batch as one line per task — the task heading, the destination project, the delivery mode (=inbox-send= file handoff), and the engine's confidence — then offer exactly two options: *go* (route the whole batch) or *skip* (leave everything local). Derive each confidence label by running the engine on the task's heading + body (=python3 .ai/scripts/route_recommend.py --item "..." --exclude "$(basename "$PWD")"=); label weak matches visibly ("weak — verify the destination") so a low-confidence route gets a human glance before the keystroke.
+
+On *go*:
+
+#+begin_src bash
+.ai/scripts/route-batch --go
+#+end_src
+
+Per candidate, the helper writes the task's subtree (children ride along; =:ROUTE_CANDIDATE:= stripped, headings promoted to top level) to a one-task handoff, delivers it via =inbox-send <destination> --file= (so the =from-<this-project>= provenance is stamped and the destination's inbox process mode dispositions it as a single item), and only after a successful send removes the subtree from the local =todo.org= — a single-file local edit the wrap is already committing. A failed send leaves that task in place and exits non-zero; report it and continue the wrap. Never write the destination's =todo.org= directly; its own inbox processing files the task per its conventions.
+
+On *skip*, leave every candidate in place, marker included — they resurface next wrap.
+
+Mis-routes are recoverable: the receiving project rejects via inbox process mode's reject-from-another-project flow, which returns the item to this project's inbox with the rationale. That reject path is why removing the local source on send is safe.
+
+*** Review-habit health check (surface a slipped daily task-review)
+
+The daily task-review habit walks the open top-level tasks on a rotating cycle, stamping =:LAST_REVIEWED:= as it goes (see =task-review.org=). This check is the watchdog for that habit. When tasks have gone too long unreviewed, the habit has slipped, and the wrap-up says so in one line — it does not re-list the tasks.
+
+=task-review-staleness.sh= counts top-level =[#A]= / =[#B]= / =[#C]= tasks (TODO/DOING/VERIFY) whose =:LAST_REVIEWED:= is missing or older than the threshold. Threshold 30 days is about 2.5 review cycles of slack at the default batch size — one missed week is fine, three weeks signals a problem.
+
+#+begin_src bash
+if [ -n "$LINT_ORG_FOLLOWUPS" ]; then
+ followups="$LINT_ORG_FOLLOWUPS"
+elif [ -d "./inbox" ]; then
+ followups="./inbox/lint-followups.org"
+else
+ followups=".ai/lint-followups.org"
+fi
+if [ -f todo.org ]; then
+ stale=$(.ai/scripts/task-review-staleness.sh todo.org 30 2>/dev/null || echo 0)
+ if [ "$stale" -gt 0 ]; then
+ printf "\n* %s — Task-review health: %s top-level [#A]/[#B]/[#C] tasks unreviewed for >30 days (daily review may have slipped)\n" \
+ "$(date '+%Y-%m-%d %a')" "$stale" >> "$followups"
+ fi
+fi
+#+end_src
+
+A non-zero count writes one summary line and nothing else — the per-task walk is the review habit's job, not the wrap-up's. This supersedes the old date-coverage scan, which flagged every dateless =[#A]= / =[#B]= task on the wrong assumption that high-priority work needs a date. No-date is a valid resting state for research and watch-list tasks; staleness, not datelessness, is the real signal.
+
+** Step 3.5: Linear ticket-state hygiene (skip if project doesn't use Linear)
+
+If the project uses Linear and has any tickets currently in *Dev Review* assigned to Craig, sweep them before the wrap-up commit. The check is fast and keeps the board honest — tickets stuck in Dev Review after their PR merges hide actual work-in-progress.
+
+#+begin_src
+mcp__linear__list_issues assignee="me" state="Dev Review" limit=50
+#+end_src
+
+For each result, look up the linked PR (the =gitBranchName= field on the issue maps to a =headRefName= on the project's GitHub remote — use =gh pr list --author <github-login> --state all --json number,state,headRefName,mergedAt,title=).
+
+*Assumption:* the =gh= lookup expects a GitHub-family host. It holds today because the only Linear-using project (DeepSat) lives on =deepsat.ghe.com=, where =gh= talks to the GHE API. A future Linear-using project on a non-GitHub host (GitLab, Gitea, Bitbucket) would need a provider-agnostic PR lookup here — update this step when that happens.
+
+If a Dev-Review ticket's PR is *merged*, propose a move:
+
+- *Done* — chores, refactors, test-coverage backfills, dead-code removal, e2e-flake fixes, anything with no PM-visible behavior change. PR titles prefixed =chore:=, =test:=, =refactor:=, =docs:= almost always belong here.
+- *PM Acceptance* — real behavior fixes or new features a PM (or end user) could verify by clicking through the app. PR titles prefixed =fix:=, =feat:= usually belong here unless the change is invisible to users.
+
+When in doubt, ask Craig per ticket. Don't auto-pick. After Craig confirms, move via =mcp__linear__save_issue= with =state="Done"= or =state="PM Acceptance"=. Several can run in parallel.
+
+Skip the step entirely if the project doesn't use Linear (e.g. personal projects, the rulesets repo).
+
+** Step 4: Git commit + push
+
+*** Step 4.0: Commit template-sync churn first (consuming projects)
+
+The startup workflow's Phase A rsyncs template updates from rulesets into this project's =.ai/= (=protocols.org=, =workflows/=, =scripts/=) every session that rulesets has advanced. Nothing commits that churn, so without this step it accumulates across sessions and eventually blocks Phase A.0's auto-fast-forward (git refuses to ff a dirty tree). Commit it here, as its own =chore:= commit, before the session-work commit — so the sync stays separate from what the session actually shipped and the tree ends clean.
+
+The guard is conservative: only auto-commit a dirty synced path when it matches the rulesets canonical byte-for-byte (a modified/new file equals canonical, or a deletion pairs with a file retired upstream). If any synced path is dirty but /doesn't/ match canonical — a local hand-edit to a file that's supposed to be sync-managed — surface it and don't auto-commit. Anything outside the three synced paths is untouched here; the normal Step 4 commit and the worktree-leftover step handle it.
+
+#+begin_src bash
+# Skip in the rulesets repo itself: there .ai/ is a committed mirror of
+# claude-templates/.ai/, kept in sync by the pre-commit hook and committed
+# alongside template edits — not downstream sync churn. The presence of
+# claude-templates/.ai/ in this repo is the tell.
+if [ ! -d claude-templates/.ai ] && [ -d "$HOME/code/rulesets/claude-templates/.ai" ]; then
+ canon="$HOME/code/rulesets/claude-templates/.ai"
+ safe=1
+ commitlist=()
+ while IFS= read -r line; do
+ f="${line:3}" # strip the 2-char status + space
+ rel="${f#.ai/}"
+ if [ -e "$f" ] && [ -e "$canon/$rel" ] && diff -q "$f" "$canon/$rel" >/dev/null 2>&1; then
+ commitlist+=("$f") # modified/new here, matches canonical
+ elif [ ! -e "$f" ] && [ ! -e "$canon/$rel" ]; then
+ commitlist+=("$f") # deleted here AND retired upstream
+ else
+ safe=0 # synced path dirty but != canonical
+ fi
+ done < <(git status --porcelain -- .ai/protocols.org .ai/workflows/ .ai/scripts/)
+
+ if [ "$safe" -eq 1 ] && [ "${#commitlist[@]}" -gt 0 ]; then
+ git add -- "${commitlist[@]}"
+ git commit -q -m "chore: sync .ai tooling from templates"
+ echo "wrap-up: committed ${#commitlist[@]} synced .ai file(s) as a template-sync chore."
+ elif [ "$safe" -eq 0 ]; then
+ echo "wrap-up: synced .ai paths are dirty but not all match rulesets canonical — NOT auto-committing. Resolve manually:"
+ git status --porcelain -- .ai/protocols.org .ai/workflows/ .ai/scripts/ | sed 's/^/ /'
+ fi
+fi
+#+end_src
+
+The commit isn't pushed here — the push step below pushes the current branch, which carries both this chore commit and the session-work commit. A crashed session that never reaches wrap-up leaves the churn for the next startup, which surfaces it (see startup.org Phase C) so it never silently accumulates.
+
+*** Review changes
+
+#+begin_src bash
+git status
+git diff --stat
+#+end_src
+
+Decide the scope of the wrap-up commit. Usually everything that changed during the session goes into one commit. If anything is intentionally not part of this session's work (pre-existing WIP, unrelated files), leave it out.
+
+*** Stage
+
+Add the renamed session file and all other session changes:
+
+#+begin_src bash
+git add .ai/sessions/ [other modified paths]
+#+end_src
+
+Do NOT blindly =git add .= — review what's being staged so unrelated dirty state isn't dragged in.
+
+*** Commit
+
+Commit message rules (also see protocols.org "Git Commit Requirements"):
+
+- Subject line: concise, describes what /shipped/. Use conventional prefixes (=docs:=, =refactor:=, =fix:=, =feat:=, =chore:=) — NEVER =session:=.
+- Body: 1-3 terse sentences describing what was accomplished.
+- NO Claude Code attribution. NO =Co-Authored-By=. NO references to =notes.org=, =session-context.org=, =.ai/sessions/=, "session wrap-up", or session timestamps.
+
+*Wrap-up commits skip the inline-approval gate.* The =commits.md= rule that requires writing the message to =/tmp/commit-<slug>.md=, printing inline, and waiting for an approve / request-changes / open-in-editor response does *not* apply to wrap-up commits. The wrap-up flow is meant to be quick — Craig has already authorized the wrap by triggering the workflow ("wrap it up"), and stopping again to approve a commit message disrupts the cadence.
+
+Still apply =/voice personal= silently before committing so the message reads cleanly. Just don't print and ask. Commit directly with the cleaned message.
+
+If a wrap-up commit needs Craig's eyes for a content reason (sensitive change, unusual scope, something he flagged earlier), surface it explicitly. Otherwise commit and move on.
+
+Example:
+#+begin_example
+docs: restructure docs/ to .ai/ and unify aix+hey into ai launcher
+
+Hidden .ai/ now holds Claude tooling; project-level docs/ reserved
+for user-facing docs. Single 'ai' launcher (fzf multi + smart tmux
++ git-aware fetch/pull) replaces the aix script and hey alias.
+#+end_example
+
+Use heredoc for multi-line:
+#+begin_src bash
+git commit -m "$(cat <<'EOF'
+subject line here
+
+body sentences here.
+EOF
+)"
+#+end_src
+
+*** Push to all remotes
+
+#+begin_src bash
+git remote -v
+#+end_src
+
+Push the current branch to every remote (some repos have multiple remotes — a primary host plus one or more mirrors, or different remotes for different audiences — and the loop keeps all of them current):
+
+#+begin_src bash
+current=$(git symbolic-ref --short HEAD)
+for r in $(git remote); do git push "$r" "$current"; done
+#+end_src
+
+Then push every other local branch with a tracking upstream to its tracking remote. This catches feature branches that advanced during the session but aren't the one being wrapped up — without it, work-in-progress branches stay local-only and are at risk if the machine dies before the next wrap-up.
+
+#+begin_src bash
+git for-each-ref --format='%(refname:short) %(upstream:remotename)' refs/heads/ | \
+while read branch remote; do
+ [ "$branch" = "$current" ] && continue
+ if [ -z "$remote" ]; then
+ echo " $branch: no tracking upstream — skipped (push manually with 'git push -u')"
+ else
+ git push "$remote" "$branch"
+ fi
+done
+#+end_src
+
+Behavior:
+- *Tracked branches* → pushed to their upstream remote.
+- *Untracked branches* (no upstream set) → surfaced, not pushed. Craig sets the upstream manually with =git push -u <remote> <branch>= when he's ready. Auto-creating an upstream would commit to a remote choice the workflow can't make safely.
+- *Diverged or rejected pushes* → surface and stop. Don't force-push from this workflow; resolve manually.
+
+*** Resolve every worktree leftover
+
+#+begin_src bash
+git status --short
+#+end_src
+
+*Default policy: end every session with an empty =git status=.* The wrap is incomplete while anything remains dirty. There is no "leave it alone" default — every leftover gets an active resolution. The only way for a file to stay dirty across the wrap is the user explicitly saying "defer this one, leave it dirty." Surface each leftover with a concrete recommendation; the user has to actively opt out for the dirt to persist.
+
+This inverts the older "intentional carryover" default, which let pre-existing dirty state accumulate across sessions silently. Carryover that lives for days or weeks is almost always one of: a forgotten commit from a prior wrap, a stale change that should be discarded, or genuine in-flight work that needs an explicit stash/branch home. None of those should default to "leave it dirty."
+
+**** Three kinds of leftover
+
+| Pattern | What it is | Recommended action (apply unless user defers) |
+|---+---+---|
+| Generated, runtime, or lock files that no human edits — e.g., =.claude/scheduled_tasks.lock=, =.pytest_cache/=, build outputs, IDE state, editor swap files | *Runtime artifact* — created by tooling or the harness, not by the user, and shouldn't be tracked | Add the matching pattern to =.gitignore= (project-level, not =~/.gitignore_global=). For tracked files, =git rm --cached <path>=. Stage =.gitignore= and any =rm --cached= changes in *one* follow-up commit (=chore: gitignore X=), push. Re-run =git status= to confirm clean. |
+| Modified or created during the session but not staged into the wrap-up commit | *Forgotten change* — real session work that should have been in the wrap commit but missed it | Stage and create a follow-up commit. Don't =--amend= the wrap-up commit once pushed (diverging history without a clear win). Push the follow-up to all remotes. |
+| Was dirty at session start and still dirty at session end — work this session deliberately didn't touch | *Pre-existing dirt that needs a decision* — could be a missed commit from a prior wrap, stale abandoned work, or real in-flight work without a home | Investigate (show diff + check the originating session). Recommend one of: (a) commit now if the work is complete, (b) stash with a descriptive message if it's genuine WIP, (c) =git checkout -- <path>= / =git clean -f <path>= if stale and unwanted, (d) move to a feature branch if it's longer-running, (e) user explicitly defers and accepts the dirt. Do not silently leave dirty. |
+
+**** Per-file flow
+
+For each leftover line in =git status --short=:
+
+1. Identify which of the three kinds above it matches.
+2. State what the file is (one line) and the recommended action.
+3. Apply the action unless the user explicitly defers.
+4. Re-run =git status --short= after each follow-up commit until empty (or until every remaining line is an explicit user-deferred entry).
+
+The pre-existing-dirt case (third row) is the one this rule most cares about. Treat each pre-existing-dirty file as a question that must get an answer this session, not as "carryover that's fine to inherit." A file that was dirty for a week before this session probably isn't going to get cleaner by waiting another week. Look at the diff, check the originating session's notes, and recommend a real resolution.
+
+**** When the user defers
+
+If the user does say "leave this one dirty for now" after seeing the recommendation, that is fine — log the deferral in the valediction so the next session knows it was an explicit choice, not a miss. Format: "Deferred (per Craig's decision today): =path/to/file= — <one-line reason>". Without that note, the next session can't distinguish "we agreed to defer" from "we forgot again."
+
+** Step 5: Valediction
+
+Brief, warm closing. 3-4 sentences max.
+
+Include:
+- What was accomplished (specific, not generic)
+- What's ready for next session
+- Any critical reminders or deadlines
+
+Tone: warm but professional. No emoji unless Craig has explicitly requested. Acknowledge effort when session was long or difficult.
+
+End on a clear signoff: the *last* line of the valediction is always =session wrapped.= on its own line (lowercase, with the period, nothing after it). It's the unmistakable end-of-session marker, so don't trail it with another sentence. This is the last user-facing output — Step 6's teardown is silent.
+
+Example:
+#+begin_example
+That's a wrap. Today we restructured the entire claude-templates
+ecosystem: docs/ → .ai/ across all 23 projects, unified aix + hey
+into a single 'ai' launcher with git-aware fetch/pull, and cleaned
+up 4 code projects on velox. Both machines fully in sync.
+
+Two things to pick up next: the chime README WIP (your inline notes
+from earlier) and archsetup's layout-navigate tests. Both are
+ratio-local uncommitted state.
+
+Good session. Talk tomorrow.
+
+session wrapped.
+#+end_example
+
+** Step 6: Session teardown (mode-dependent)
+
+The last action of the wrap, and only after Step 4's commit + push is verified and the Step 5 valediction is composed. The teardown itself happens when this response ends (via the =Stop= hook), so the valediction always renders first. Act by the mode resolved up front:
+
+*** No-teardown mode
+
+Do nothing. The buffer, the =aiv-<project>= tmux session, and =claude= all stay up so the summary stays readable. The wrap is complete.
+
+*** Teardown mode (default)
+
+Confirm commit + push succeeded (Exit Criteria 5 — never tear down over unpushed work), then drop the sentinel:
+
+#+begin_src bash
+touch "/tmp/ai-wrap-teardown-$(basename "$PWD")"
+#+end_src
+
+That is the whole step. Don't run any =tmux kill-session=, =emacsclient=, or buffer kill inline — the =Stop= hook reads the sentinel when this response ends and runs =cj/ai-term-quit=, which kills the =aiv-<project>= session (taking =claude= with it), kills the vterm buffer, and restores geometry. The basename of =$PWD= is the key the hook matches, so the sentinel names the session it tears down.
+
+*** Shutdown mode
+
+Confirm commit + push succeeded, then evaluate the safety gate *before* committing to the shutdown — never power the box off out from under another live session:
+
+#+begin_src bash
+emacsclient -e '(cj/ai-term-live-count)'
+#+end_src
+
+- *Count > 1* — another ai-term session is alive. ABORT the shutdown. List the other live =aiv-*= sessions, drop *no* sentinel, and tell Craig in the valediction that it fell back to a normal wrap (no poweroff, no teardown). This gate is the load-bearing safety of the whole feature.
+- *Count = 1* — this session is the only one. Drop the shutdown sentinel:
+
+ #+begin_src bash
+ touch "/tmp/ai-wrap-shutdown-$(basename "$PWD")"
+ #+end_src
+
+ The =Stop= hook fires =cj/ai-term-shutdown-countdown= when this response ends: it re-checks the gate, runs an abort-able 10→1 countdown in the Emacs echo area (=C-g= cancels), then =sudo shutdown now=. Shutdown supersedes teardown — do *not* also drop the teardown sentinel.
+
+If =emacsclient= isn't resolvable or the daemon is down, the gate can't run — abort the shutdown, fall back to a normal wrap, and say so. Don't power off on an unverifiable gate.
+
+* Common Mistakes to Avoid
+
+1. *Skipping Step 1 (Summary)* — the file becomes the record; an empty Summary makes it hard to scan at catch-up
+2. *Vague description in filename* — =2026-04-20-updates.org= is useless next to =2026-04-20-13-45-docs-ai-migration.org=
+3. *=git add .= without review* — drags in unrelated dirty state
+4. *=session:= prefix in commit message* — explicitly forbidden; use real change categories
+5. *Claude-tooling references in commit message* — describes tooling, not what shipped
+6. *Forgetting to push to all remotes* — check =git remote -v=, push to each
+7. *Leaving =.ai/session-context.org= in place* — its presence means "interrupted session", confuses next startup
+8. *Long preachy valediction* — brief beats thorough
+9. *Leaving runtime/generated files dirty without gitignoring them* — pollutes every future =git status= and erodes trust in "working tree clean" as a signal. Fix =.gitignore= during the wrap, not later.
+10. *Treating "was dirty at session start, still dirty now" as fine by default* — that's how a forgotten commit from two sessions ago turns into "carryover" for two weeks. Every pre-existing dirty file needs an active resolution recommendation this session. Deferral is allowed only with an explicit user choice, logged in the valediction.
+
+* Validation Checklist
+
+Before considering wrap-up complete:
+
+- [ ] =.ai/session-context.org= =* Summary= section populated
+- [ ] The Summary ends with the =KB: promoted N / consulted yes-no= line (promotion check ran)
+- [ ] File renamed to =.ai/sessions/YYYY-MM-DD-HH-MM-description.org=
+- [ ] =.ai/session-context.org= no longer exists
+- [ ] =todo-cleanup.el= ran — hygiene pass + =--convert-subtasks= + =--archive-done= + =--sync-child-priority= (if =todo.org= exists at project root)
+- [ ] =lint-org.el= ran on =todo.org= — mechanical fixes applied, judgments appended to follow-ups file (if =todo.org= exists)
+- [ ] Any orphan-planning-line warnings reviewed (fix or accept)
+- [ ] Inbox carries nothing but expected pipeline artifacts (=.gitkeep=, =lint-followups.org=, =PROCESSED-*= prefixes), OR each remaining handoff has an explicit deferral logged in the valediction
+- [ ] Linear Dev-Review sweep ran; any merged-PR tickets moved to Done or PM Acceptance (skip if project doesn't use Linear)
+- [ ] Template-sync churn committed as its own =chore: sync .ai tooling from templates= (consuming projects only; skipped in rulesets), or surfaced if a synced path didn't match canonical
+- [ ] After wrap-up commit + push, =git status --short= is empty OR every remaining line has an explicit user-deferred decision logged in the valediction
+- [ ] Each leftover was investigated and the user saw a concrete resolution recommendation
+- [ ] Runtime artifacts added to =.gitignore=, follow-up commit pushed, =git status= re-verified
+- [ ] Forgotten changes committed in a follow-up and pushed
+- [ ] Pre-existing dirty files resolved (committed / stashed / discarded / moved to a feature branch) or explicitly deferred with a one-line reason in the valediction
+- [ ] Current branch pushed to ALL remotes (verified with =git remote -v=)
+- [ ] All other local branches with a tracking upstream pushed to their remote
+- [ ] Any untracked-upstream branches surfaced for manual =git push -u=
+- [ ] Step 6 teardown matches the trigger phrase: no-teardown leaves the buffer; teardown drops only =/tmp/ai-wrap-teardown-<project>=; shutdown gates on =cj/ai-term-live-count= = 1 and drops only =/tmp/ai-wrap-shutdown-<project>=
+- [ ] No teardown/shutdown sentinel was dropped before commit + push was verified
+- [ ] Shutdown aborted (fell back to normal wrap, logged in the valediction) when another =aiv-*= session was live or the gate couldn't run
+- [ ] Commit message follows format (no =session:=, no Claude attribution)
+- [ ] Valediction delivered (brief, specific, warm)