diff options
216 files changed, 19660 insertions, 2542 deletions
diff --git a/.ai/metrics/work-the-backlog.jsonl b/.ai/metrics/work-the-backlog.jsonl index 1067b3a..88c6764 100644 --- a/.ai/metrics/work-the-backlog.jsonl +++ b/.ai/metrics/work-the-backlog.jsonl @@ -3,3 +3,13 @@ {"ts":"2026-07-02T05:22:11-04:00","run_id":"c726f526-2e35-4513-b25c-18ef61061333","project":"rulesets","caller":"speedrun","task":"template-sync-gitignored-only-changes","outcome":"implemented-committed","defer_reason":"","upfront_decision":true,"wall_clock_s":188,"commit_sha":"ed75d3c","review_findings":0} {"ts":"2026-07-02T05:58:16-04:00","run_id":"a48f2977-4493-48a3-9238-9b2f5ff5383b","project":"rulesets","caller":"loop","task":"inbox-send-filename-collision-fix","outcome":"implemented-committed","defer_reason":"","upfront_decision":false,"wall_clock_s":300,"commit_sha":"8099377","review_findings":0} {"ts":"2026-07-02T05:58:16-04:00","run_id":"a48f2977-4493-48a3-9238-9b2f5ff5383b","project":"rulesets","caller":"loop","task":"page-me-notify-info-level","outcome":"implemented-committed","defer_reason":"","upfront_decision":false,"wall_clock_s":120,"commit_sha":"a6b534f","review_findings":0} +{"ts":"2026-07-23T23:55:16-05:00","run_id":"711400a5-c6bc-4684-b928-6b818d0d33f2","project":"rulesets","caller":"speedrun","task":"inbox-send phantom empty handoff","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"0f91a8e","review_findings":1} +{"ts":"2026-07-23T23:57:59-05:00","run_id":"711400a5-c6bc-4684-b928-6b818d0d33f2","project":"rulesets","caller":"speedrun","task":"inbox-send two smaller defects","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"a053e9d","review_findings":0} +{"ts":"2026-07-24T00:02:14-05:00","run_id":"711400a5-c6bc-4684-b928-6b818d0d33f2","project":"rulesets","caller":"speedrun","task":"lint-org todo-format checkers fire on specs","outcome":"implemented-committed","upfront_decision":true,"commit_sha":"c38bab9","review_findings":0} +{"ts":"2026-07-24T00:02:49-05:00","run_id":"711400a5-c6bc-4684-b928-6b818d0d33f2","project":"rulesets","caller":"speedrun","task":"notes.org template four lint flags","outcome":"already-satisfied","defer_reason":"already-satisfied","upfront_decision":false,"commit_sha":"","review_findings":0} +{"ts":"2026-07-24T01:41:14-05:00","run_id":"sentry-fire2-1784875274","project":"rulesets","caller":"loop","task":"cj-remove-block over-deletion + unsafe write","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"17f5d48","review_findings":0} +{"ts":"2026-07-24T01:43:52-05:00","run_id":"sentry-fire2","project":"rulesets","caller":"loop","task":"route_recommend duplicate-name tier downgrade","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"1b0f284","review_findings":0} +{"ts":"2026-07-24T02:38:48-05:00","run_id":"sentry-fire3","project":"rulesets","caller":"loop","task":"audit.bats flaky teardown","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"7f45d4b","review_findings":0} +{"ts":"2026-07-24T03:39:16-05:00","run_id":"sentry-fire4","project":"rulesets","caller":"loop","task":"todo-cleanup missing backup","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"0686784","review_findings":0} +{"ts":"2026-07-24T04:36:09-05:00","run_id":"sentry-fire5","project":"rulesets","caller":"loop","task":"attachment filename sanitization","outcome":"deferred-verify","defer_reason":"needs-deliberation","upfront_decision":false,"commit_sha":"","review_findings":0} +{"ts":"2026-07-24T07:38:47-05:00","run_id":"sentry-fire8","project":"rulesets","caller":"loop","task":"bin/ lint coverage gap","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"f91feef","review_findings":1} diff --git a/.ai/notes.org b/.ai/notes.org index 71d23dc..828fde3 100644 --- a/.ai/notes.org +++ b/.ai/notes.org @@ -61,6 +61,8 @@ This section tracks decisions that need Craig's input before work can proceed. ** Current Reminders +- =[2026-07-27]= Finish the context-engineering rightsizing — Craig's explicit ask at wrap. Surface is 57,800 → 28,949 tokens; the remaining work needs *his decisions*, not execution: =verification.md= (C1 — its honesty core vs the Opus 5 over-verification warning), =interaction.md= (3,828 tok, largest remaining), the TDD rationalization table (cut or keep), and D3 the gate separation (which approval gates are preference vs guardrail). Task: "Finish context-engineering rightsizing" in todo.org. Docs in =working/context-engineering-rightsizing/= are one commit behind — reconcile them first. + - =[2026-07-14]= Review the sentry spec (docs/specs/2026-07-14-sentry-workflow-spec.org) — Craig's explicit ask at wrap: strongly suggest he reviews it before ending the next session. All 12 review findings and 10 decisions are resolved and folded in; the spec is open in his Emacs; the READY flip and the [#B] build task both wait on his deep read. ** Instructions for This Section @@ -78,10 +80,11 @@ Format: :COMMIT_AUTONOMY: yes :LOOP_MAY_COMMIT: yes +:SENTRY_MAY_IMPLEMENT: yes :LAST_SPEC_SORT: 2026-07-02 Markers maintained by workflows to record when they last ran. Read by other workflows that gate their behavior on freshness. -:LAST_AUDIT: 2026-07-04 -:LAST_INBOX_PROCESS: 2026-07-18 (10 handoffs: triage-intake redesign + birthdays feature applied; todo-cleanup seal, planning-line strip, colloquialisms convention filed; knowledge-base roam URL fixed; 2 website FYIs + 1 archsetup FYI acknowledged) +:LAST_AUDIT: 2026-07-20 (open set current — this session's shipped work (working/temp, triage-source-activation, silent-until-signal, suspend detach) closed as it went; sentry cluster consolidated (merged the /schedule tasks, added cross-host-coordination); nothing shipped-but-open per git reconcile. Live finding: the Polyglot + Subprojects scouting tasks are SCHEDULED 2026-07-20 and due.) +:LAST_INBOX_PROCESS: 2026-07-25 (consolidated home + work Claude-to-Codex MCP registry proposals into one [#B] parked spec decision; memory auditor split from the registry work) Format: one =:MARKER: YYYY-MM-DD= line per workflow. Workflows overwrite their own marker on completion. diff --git a/.ai/protocols.org b/.ai/protocols.org index 5cd69d4..b291d9e 100644 --- a/.ai/protocols.org +++ b/.ai/protocols.org @@ -84,7 +84,7 @@ Do NOT estimate, guess, or rely on memory. Just run the command. It takes one se Every session pulls rulesets first, then the local project repo. Rulesets carries the canonical behavioral rules and =.ai/= templates (the old =claude-templates= repo is folded in as a subtree at =rulesets/claude-templates/=); the project pull lands commits pushed from other machines or teammates since the last session. -Resolve any dirty-tree or merge issue at each step before moving on. Both pulls run as =git pull --ff-only= (or =git merge --ff-only= against the fetched upstream), so anything non-trivial — non-fast-forward history, dirty working tree, diverged branches — aborts. Surface the state and stop. Never auto-stash, auto-merge, or auto-rebase; the user resolves the conflict before further work. +Resolve any sync-blocking tree or merge issue at each step before moving on. The shared =git-worktree-gate sync-safe= policy permits untracked deliveries beneath =inbox/= so receiving a handoff never prevents another project from refreshing rulesets; every staged or tracked change, dirty submodule, Git operation in progress, or untracked path outside =inbox/= blocks. Both pulls run as =git pull --ff-only= (or =git merge --ff-only= against the fetched upstream), so non-fast-forward history and diverged branches also abort. Surface the state and stop. Never auto-stash, auto-merge, or auto-rebase; the user resolves the conflict before further work. Mechanics live in =startup.org= Phase A.0. The rule lives here because it governs the very first action of every session: load the freshest behavioral rules and templates before anything else runs. @@ -106,7 +106,7 @@ The epoch is baked into the id by the spawner, never minted inside =session-cont Resolve the path with =.ai/scripts/session-context-path= rather than hardcoding =.ai/session-context.org=; it prints the right path for the current =AI_AGENT_ID=. Fall back to =.ai/session-context.org= if the script isn't present (older checkouts mid-sync). Everything below — the record/recovery purpose, the update triggers, the startup existence check, the wrap-up rename — operates on that resolved path. The prose says "session-context.org" as the default name; read it as "the resolved active path" when =AI_AGENT_ID= is set. -A helper instance (a second agent running in this project while a primary session is live) follows a different contract: it skips the pulls and rsync, makes only scoped single-heading edits to shared files, leaves all git mutation to the primary, and wraps up by archiving its own context file without committing. The full rules — read/write tiers, data-integrity, light startup, helper wrap-up — live in [[file:workflows/helper-mode.org][workflows/helper-mode.org]]. A session is a helper only when something routes it there (the =ai --helper= launcher, startup's roster check, or an explicit "you are a helper" instruction); the routing itself ships behind the helper-instance feature gate and isn't live yet. +A helper instance (a second agent running in this project while a primary session is live) follows a different contract: it skips the pulls and rsync, makes only scoped single-heading edits to shared files, leaves all git mutation to the primary, and wraps up by archiving its own context file without committing. The full rules — read/write tiers, data-integrity, light startup, helper wrap-up — live in [[file:workflows/helper-mode.org][workflows/helper-mode.org]]. A session is a helper only when something routes it there: the =ai --helper= launcher (live — it checks the roster, assigns the id, and opens the helper in its own tmux window) or an explicit "you are a helper" instruction. Startup's roster check is *not* built, so a bare =claude= launched into a project that already has a live session will run full primary startup regardless. Launch helpers with =ai --helper=. This file serves two purposes with one mechanism: 1. *Crash recovery* — if the session dies mid-work, the live file is all that's left. On 2026-01-22 a session crashed during a 20-minute design discussion and all context was lost because this file wasn't being updated. @@ -187,6 +187,8 @@ Canonical rule: =~/code/rulesets/claude-rules/cross-project.md=. Every in-progress task that produces files (drafts, source documents, diagrams, scripts, sub-deliverables) gets a dedicated subdirectory under =<project-root>/working/=, named after the task. All artifacts for that task live in that subdirectory until the task is marked done. +=working/= is version-controlled from creation — it's the tracked home of in-progress work, never gitignored. Filing on completion *reorganizes* durable artifacts into permanent homes; it doesn't mark when they became durable (they were durable, and tracked, from the start). Genuinely disposable artifacts go in a gitignored =temp/= (or =/tmp=), never =working/=; the install tooling ignores =temp/= in both track and gitignore modes. + When the task ships, files are **renamed individually** (standard form: =YYYY-MM-DD-<task-slug>-<descriptor>.<ext>=) and **moved flat** into the appropriate permanent home (typically =assets/= or an area-specific =<area>/assets/=). The working subdirectory is then empty and gets deleted. ***Never rename the directory itself as a substitute for filing.*** The point is to keep =assets/= flat-searchable — a nested =assets/old-tech-deck-2026/slide.png= is harder to find than =assets/2026-05-18-tech-deck-vol2-slide-04-diagram.png=. @@ -205,6 +207,8 @@ Check =inbox/= at every task boundary (after finishing a unit of work, before re Exit 1 means handoffs are pending — process them per =inbox.org= process mode. For each accepted handoff, the act-vs-file rule: *act now* when it's clear, bounded, low-risk, in-scope, and cheaper than deferring — just do it, no asking; *file* otherwise — ask first, with filing as option 1 and "do it now" as option 2; *ask* if unsure. Exception: a proposal to change a shared asset (template workflow, rule, skill, synced script) or a substantive convention never silently acts now — it goes through the inbox engine's skeptical review and its approval (or park) step. Always reply to a handoff's sender (confirm on accept, the why on reject). Full process, the reply discipline, and the opt-in background-monitor =/loop= recipe live in =inbox.org= monitor mode. +A machine-global =Stop= hook (=inbox-boundary-check.sh=) backs this rule so it isn't prose-only. When handoffs are pending it blocks the turn once and injects the count, so a task boundary can't pass with items unseen. It soft-nudges rather than hard-blocks: on the harness re-entry it steps aside, so a mid-task pause to ask "what's next" is never hijacked into inbox processing. The rule above still governs what to do with the items; the hook only makes sure you look. + ** Recursive Reads — Honor =.aiignore= Before a naive recursive read or glob of a project tree (file inventories, "what's in this repo", broad greps), skip the noise: dependency trees (=node_modules/=, =.venv/=), build output (=dist/=, =build/=, =coverage/=), language caches (=__pycache__/=, =.pytest_cache/=, =*.pyc=), editor/OS cruft, and generated token/OAuth artifacts. These waste tokens and skew project summaries even when gitignored — a recursive read sees the disk, not git. @@ -246,13 +250,29 @@ Execute the wrap-up workflow (details in Session Protocols section below): Execute the suspend workflow ([[file:workflows/suspend.org][suspend.org]]): a capture-only mid-session pause for an abrupt departure. It appends a resume-weighted =SUSPENDED= entry to the Session Log, notes uncommitted work, and LEAVES =.ai/session-context.org= in place so the next startup resumes from it — no archive, no teardown, no valediction. The capture-only counterpart to "wrap it up" (which ends + archives + tears down) and to =/flush= (which prompts =/clear= and resumes the same session). "I need to go" is broad — if it reads as a conversational aside, confirm before suspending. +* Colloquialisms and Expansions + +Shorthand phrases Craig uses that expand to a defined action the agent applies without asking. The set is extensible: a project may add its own entries, and new shared shorthands land here. + +** "the list": the Before-Close Queue + +"Put X on the list" or "add X to the list" appends X to the Before-Close Queue, a FIFO queue of tasks and actions to finish before the session closes. Work it oldest-first at wrap-up, before teardown (=wrap-it-up.org= Step 1 works it before finalizing the Summary), and surface anything unfinished in the valediction rather than dropping it. + +The queue lives in the session anchor (=.ai/session-context.org=) under a =* Before-Close Queue= heading. Create the heading on the first "put it on the list" if it's absent, then append one line per item. It's session-scoped: it resets when the anchor is archived at wrap. Anything that must outlive the session is a =todo.org= task instead, not a list item. + +** "tell <project> <message>": cross-project handoff + +"Tell <project> <message>" drops the message in that project's =inbox/= via =inbox-send= (=python3 .ai/scripts/inbox-send.py <project> --text "<message>"=), the sanctioned cross-project handoff. Never write another project's =todo.org= or =inbox/= directly. Resolve =<project>= the way =inbox-send= does (basename match, dots stripped); if it's ambiguous, ask which project rather than guessing. + * User Information ** Calendar Management Three ways to access Craig's calendars: Google Calendar MCP (preferred, both personal + work accounts), gcalcli (fallback, personal only), Emacs org files (read-only viewer). -For tool recipes, authentication details, and credentials, see [[file:references/calendar-reference.org][calendar-reference.org]]. +For tool recipes and account details, read the calendar workflows in =.ai/workflows/=: =add-calendar-event.org=, =edit-calendar-event.org=, =delete-calendar-event.org=, =read-calendar-events.org=. They carry the MCP tool names, both account ids, the gcalcli fallback, and the conflict-check discipline. + +Credentials are needed only for a re-auth Craig performs himself. The MCP bundle's =mcp/README.org= in the rulesets repo is the authority: =gcp-oauth.keys.json= is gitignored and regenerated at install from a base64 var in the bundle, never committed. Named in prose rather than linked, because that path isn't synced into consuming projects. ** GPG Keys @@ -356,9 +376,19 @@ Craig runs a pure Wayland setup (Hyprland) and avoids XWayland/Xorg apps. - Clipboard: Use =wl-copy= and =wl-paste= (NOT =xclip= or =xsel=) - Window management: Use Hyprland commands (NOT =xkill=, =xdotool=, etc.) - Prefer Wayland-native tools over X11 equivalents -- Open URLs in browser: Use =google-chrome-stable "URL" &>/dev/null &= - - The =&>/dev/null &= is required to detach the process and suppress output - - Without it, the command may appear to hang or produce no result +- Open URLs in browser: invoke Chrome directly — never =xdg-open=, which returned success in a home session on 2026-07-26 while no tab appeared. + + Chrome is normally already running, and in that case it hands the URL to the live session and exits immediately (rc 0), printing =Opening in existing browser session.= on *stdout*. So run it in the foreground and read that line as the confirmation the tab actually opened: + + #+begin_src bash + google-chrome-stable --new-tab "URL" + #+end_src + + Don't redirect stdout away while checking for that line — verified 2026-07-27 on ratio: with =2>/dev/null= the message still appears (it isn't stderr), and with =>/dev/null= it vanishes. + + Several URLs in one invocation open as separate tabs (=google-chrome-stable --new-tab "URL1" "URL2"=). Pass them as separate words or an array — the Bash tool runs zsh, which does not word-split an unquoted =$urls= variable, so a space-joined string arrives as one malformed argument (see the zsh note below). + + *Cold start.* If Chrome is *not* already running, the command becomes the browser process and blocks. Detach that case with =&>/dev/null &=, accepting that the confirmation line is discarded — there is no session to confirm into. Don't apply the detach form unconditionally: it suppresses the very output the warm path is verified by. *** Shell aliases (=ls= → =exa=) Craig's shell aliases =ls= to =exa=, which prints nothing to non-TTY pipes (e.g. when capturing =ls= output in a Bash tool call). The result looks like the directory is empty when it isn't. @@ -412,27 +442,29 @@ Full usage: =notify --help= or see =~/.local/bin/notify= - =atq= - list all scheduled alarms - =atrm [number]= - remove an alarm by its queue number -** Paging Craig — the agent pager +** Reaching Craig — the notification vocabulary -"Page me" has two channels; pick by where Craig is. Both work from any agent runtime — nothing here is Claude-specific. +Two channels, two trigger words. "page me" is the desktop, "text me" is the phone, "text and page me" is both. Pick by where Craig is, and default to both when a run can't tell. Both work from any agent runtime (nothing here is Claude-specific). The words are what Craig says; a run deciding on its own maps the same way (away run texts, at-desk run pages, unsure does both). -- *At his laptop/desktop* — desktop =notify ... --persist= (above). It reaches him on the machine and stays up until dismissed. +- *"page me" — at his laptop/desktop.* A desktop =notify ... --persist= that reaches him on the machine and stays up until dismissed. #+begin_src bash notify info "Title" "Message" --persist #+end_src -- *Away from his laptop/desktop* — page his phone over Signal with the *agent pager*: +- *"text me" — away from his machine.* A Signal push to his phone via =agent-text=: #+begin_src bash - agent-page "Message for Craig's phone" + agent-text "Message for Craig's phone" #+end_src - =agent-page= (in =~/.local/bin= via the rulesets install) sends from the dedicated pager identity (+15045173983, registered in velox's signal-cli) to Craig's Signal account UUID, firing a normal mobile push. On velox it sends directly; on any other tailnet machine it ssh-relays the send to velox. Verified end to end 2026-07-13. Never page Craig's phone *number* — it reads as unregistered in Signal's directory; the script already targets the UUID. + =agent-text= (in =~/.local/bin= via the rulesets install) sends from the dedicated Signal identity (+15045173983) to Craig's Signal account UUID, firing a normal mobile push. The account is registered on velox (primary) and ratio (linked device), so either sends directly; a machine without it ssh-relays to velox. Verified end to end 2026-07-13 (velox) and 2026-07-20 (ratio). Never target Craig's phone *number* (it reads as unregistered in Signal's directory); the script targets the UUID. + + Caveats: a relay from a non-linked machine needs velox up on the tailnet, and each device holding the account wants a periodic =receive= (the signal-receive timer handles that). The full runbook lives in rulesets =docs/design/=. - Caveats: velox must be up and on the tailnet (the script says so and names the desktop fallback when the relay fails), and the signal-cli account wants a periodic =receive= — both tracked on the rulesets Signal-pager task, which owns the full runbook. +- *"text and page me" — both.* Fire =agent-text= and =notify= together. The phone reaches him now, the desktop note waits for his return. This is the default when a run can't tell whether he's away. -On velox, Claude sessions may also have the *signal-mcp* tool (=send_message_to_user=, same pager identity) — fine to use there, but it exists only in velox's local MCP config, so =agent-page= is the portable habit. Do *not* use the old =page-signal= shell script (removed 2026-06-12). +On velox, Claude sessions may also have the *signal-mcp* tool (=send_message_to_user=, same identity), fine to use there, but it exists only in velox's local MCP config, so =agent-text= is the portable habit. The tool was named =agent-page= before 2026-07-20; a deprecated =agent-page= shim still delegates to =agent-text=. Do *not* use the old =page-signal= shell script (removed 2026-06-12). * Session Protocols @@ -547,12 +579,13 @@ When monitoring a long-running process (rsync, large downloads, builds, VM tests ** "Wrap it up" / "That's a wrap" / "Let's call it a wrap" -When Craig says any of these phrases (or variations), execute the wrap-up workflow: [[file:workflows/wrap-it-up.org][wrap-it-up.org]]. Four steps: +When Craig says any of these phrases (or variations), execute the wrap-up workflow: [[file:workflows/wrap-it-up.org][wrap-it-up.org]]. Five load-bearing steps: 1. *Finalize the Summary* in =.ai/session-context.org= (populate the 5 subsections from the Session Log) 2. *Rename* =.ai/session-context.org= → =.ai/sessions/YYYY-MM-DD-HH-MM-description.org= 3. *Git commit + push* to all remotes (see Git Commit Requirements) -4. *Valediction* — brief, warm, specific closing +4. *Certify the clean tree* with =git-worktree-gate certify=. Any remaining staged, unstaged, untracked, submodule, or in-progress-operation state blocks wrap entirely; report each path and the exact decision needed. There is no dirty-file deferral. +5. *Valediction* — brief, warm, specific closing, reachable only after certification The absence of =.ai/session-context.org= after wrap-up is the signal that the session ended cleanly. If the file is still there at the next session start, the previous session was interrupted. diff --git a/.ai/references/calendar-reference.org b/.ai/references/calendar-reference.org deleted file mode 100644 index 5791b08..0000000 --- a/.ai/references/calendar-reference.org +++ /dev/null @@ -1,66 +0,0 @@ -#+TITLE: Calendar Reference -#+AUTHOR: Craig Jennings - -Tool recipes, authentication, and credentials for Craig's calendar -setup. Three access methods, in order of preference. - -* Google Calendar MCP Server (preferred for all calendar operations) - -Craig has the =@cocal/google-calendar-mcp= MCP server configured at user scope (=~/.claude.json=). It provides full read/write access to Google Calendar via MCP tools. - -Two accounts are authenticated: -- *personal* — craigmartinjennings@gmail.com (primary: "Craig Google") -- *work* — craig.jennings@deepsat.com (primary: "Craig Deepsat") - -MCP tools available: -- =list-events=, =search-events=, =get-event= — read events -- =create-event=, =create-events= — add events -- =update-event= — modify events -- =delete-event= — remove events -- =list-calendars=, =list-colors= — calendar metadata -- =get-freebusy= — check availability -- =manage-accounts= — add/remove/list authenticated accounts -- =respond-to-event= — accept/decline invitations -- =get-current-time= — current time in any timezone - -Use =account_id: "personal"= or =account_id: "work"= to specify which account. - -Default calendar for adding events: "Craig Google" (personal account). - -Calendar workflows are available alongside this reference: add-calendar-event, edit-calendar-event, delete-calendar-event, read-calendar-events. - -If re-authentication is needed: -- Use the =manage-accounts= MCP tool with =action: "add"= and the account nickname -- OAuth credentials: =~/projects/homelab/assets/gcp-oauth.keys.json= -- Google Cloud app is in production mode (tokens don't expire after 7 days) -- See =~/projects/homelab/.ai/gcalcli-setup.org= for Google Cloud project details - -* gcalcli (fallback for personal account only) - -Craig has =gcalcli= installed via pipx, authenticated to his personal Google account only. - -#+begin_src bash -gcalcli agenda # upcoming events -gcalcli calw # weekly view -gcalcli add --title "..." --when "..." --duration "60" # add event -gcalcli search "..." # search events -gcalcli delete "..." # delete event -#+end_src - -Use =--calendar "Craig Google"= when adding events. - -gcalcli does NOT have access to the work (DeepSat) calendar. Use the MCP server for work calendar operations. - -If gcalcli needs re-authentication, credentials are stored in the homelab project: =~/projects/homelab/assets/gcalcli-client-secret.json.gpg= (GPG encrypted). - -* Emacs org files (read-only, for viewing schedules) - -Craig's calendars are at: =~/.emacs.d/data/*cal.org= (gcal.org, dcal.org, pcal.org) - -These files are **READ-ONLY** — NEVER add anything to them. - -Use this to: -- Check meeting times and schedules -- Verify when events occurred -- See what's upcoming -- Note: only updated periodically when Emacs is running — may be stale diff --git a/.ai/scripts/agent-lock b/.ai/scripts/agent-lock new file mode 100755 index 0000000..634412c --- /dev/null +++ b/.ai/scripts/agent-lock @@ -0,0 +1,248 @@ +#!/usr/bin/env bash +# agent-lock — a mkdir-atomic advisory lock for agent workflows. +# +# Why not flock: every Bash call an agent makes is its own short-lived shell, +# so an flock taken in one /loop turn is gone by the next. This helper persists +# the lock on disk between calls (an atomic mkdir is the acquire), and a crashed +# holder's lock self-clears via age-based staleness reclaim instead of wedging +# every later acquire. +# +# Serves both of sentry's locks (the single-runner lock and the roam-write +# lock); callers pass a name, never a path — the helper owns the path scheme. +# +# Usage: +# agent-lock acquire <name> [--ttl=SECONDS] [--wait[=SECONDS]] +# Atomic acquire. exit 0 on win (fresh, or reclaimed from a stale holder); +# exit 1 when a live lock already holds <name> (deferred — a note names the +# holder on stderr). --wait polls up to SECONDS (default 30) before +# deferring; without it, acquire is single-shot win-or-lose. --ttl records +# the staleness horizon in the lock's metadata (default below). +# agent-lock refresh <name> +# Heartbeat: re-touch a held lock's mtime so it stays young. A runner +# refreshes its own lock between passes, so a live run's lock is never older +# than one pass and the TTL sizes to the longest single pass. exit 1 if the +# lock is absent (nothing to refresh). +# agent-lock release <name> +# Remove the lock. Idempotent: exit 0 even if already free. +# agent-lock status <name> +# Print "free" | "held ..." | "stale ..." plus metadata. exit 0 (a query +# never fails on lock state). +# agent-lock path <name> +# Print the resolved lock-directory path without creating it. +# +# Lock home (the helper owns this; callers pass names only): +# $AGENT_LOCK_DIR/<name>/ when AGENT_LOCK_DIR is set (tests / advanced) +# $XDG_RUNTIME_DIR/agent-locks/<name>/ the tmpfs runtime dir /run/user/<uid> +# (host-local, out of every repo, +# cleared on reboot). XDG_RUNTIME_DIR is +# the standard handle for it and is set +# in sentry's interactive launch. +# ${XDG_CACHE_HOME:-~/.cache}/agent-locks/<name>/ fallback where no runtime +# dir exists (XDG_RUNTIME_DIR unset or +# unwritable — a headless/container box) +# +# tmpfs residence is deliberate: a lock under ~/org/roam would ride roam-sync's +# `git add -A` to the other machine as a phantom hold. Host-locality is by +# construction, and reboot clears any lock a crash left behind for free. +# +# Staleness is age-based on the metadata file's mtime versus the lock's own +# recorded TTL. Heartbeat re-touches the mtime; a reclaim is always surfaced, +# never silent. + +set -euo pipefail + +DEFAULT_TTL=600 # 10 min: sized to the longest single sentry pass, since a + # live runner heartbeats between passes and stays young. +DEFAULT_WAIT=30 # bounded-wait budget for --wait (capture-guard's shape). +WAIT_INTERVAL=3 # poll cadence while waiting on a busy lock. + +usage() { + echo "usage: agent-lock {acquire|refresh|release|status|path} <name> [--ttl=N] [--wait[=N]]" >&2 + exit 2 +} + +# Resolve the base directory that holds all lock dirs, per the home scheme above. +lock_base() { + if [ -n "${AGENT_LOCK_DIR:-}" ]; then + printf '%s\n' "$AGENT_LOCK_DIR" + elif [ -n "${XDG_RUNTIME_DIR:-}" ] && [ -d "$XDG_RUNTIME_DIR" ] && [ -w "$XDG_RUNTIME_DIR" ]; then + printf '%s/agent-locks\n' "$XDG_RUNTIME_DIR" + else + printf '%s/agent-locks\n' "${XDG_CACHE_HOME:-$HOME/.cache}" + fi +} + +# Validate a lock name: non-empty, no path separators (so a name can never +# escape the base dir). +valid_name() { + case "$1" in + ''|*/*|.|..) return 1 ;; + *) return 0 ;; + esac +} + +lock_dir() { printf '%s/%s\n' "$(lock_base)" "$1"; } +meta_path() { printf '%s/meta\n' "$(lock_dir "$1")"; } + +# Read a key from a lock's metadata file; empty if absent. +meta_get() { + local key="$1" file="$2" + [ -f "$file" ] || return 0 + sed -n "s/^${key}=//p" "$file" | head -n1 +} + +# Age of a lock in whole seconds, from the metadata mtime. +lock_age() { + local file="$1" mtime now + mtime=$(stat -c %Y "$file" 2>/dev/null) || return 1 + now=$(date +%s) + printf '%s\n' "$((now - mtime))" +} + +# True when a lock dir exists but its age exceeds its recorded TTL. +is_stale() { + local name="$1" file age ttl + file="$(meta_path "$name")" + [ -f "$file" ] || return 1 + age="$(lock_age "$file")" || return 1 + ttl="$(meta_get ttl "$file")" + [ -n "$ttl" ] || ttl="$DEFAULT_TTL" + [ "$age" -gt "$ttl" ] +} + +# Write the metadata file for a freshly-taken lock. +write_meta() { + local name="$1" ttl="$2" file + file="$(meta_path "$name")" + { + printf 'pid=%s\n' "$$" + printf 'host=%s\n' "$(uname -n)" + printf 'acquired=%s\n' "$(date +%Y-%m-%dT%H:%M:%S%z)" + printf 'ttl=%s\n' "$ttl" + } > "$file" +} + +# One-line holder description for surfaced notes. +holder_desc() { + local file="$1" + printf "pid=%s host=%s age=%ss ttl=%ss" \ + "$(meta_get pid "$file")" "$(meta_get host "$file")" \ + "$(lock_age "$file" 2>/dev/null || echo '?')" "$(meta_get ttl "$file")" +} + +# Attempt a single atomic acquire. exit 0 win, 1 busy (live holder). +try_acquire() { + local name="$1" ttl="$2" dir file + dir="$(lock_dir "$name")" + file="$(meta_path "$name")" + mkdir -p "$(lock_base)" + + if mkdir "$dir" 2>/dev/null; then + write_meta "$name" "$ttl" + return 0 + fi + + # Directory exists. Reclaim it if the holder is stale; otherwise it's busy. + if is_stale "$name"; then + # Claim the stale dir atomically before removing it. `mv` of a directory is + # atomic, so when two acquirers both see the lock stale, only one's rename + # of $dir succeeds — the other's fails because $dir is already gone, and it + # falls through to busy. Never `rm -rf $dir` directly: a plain remove lets + # the loser delete the winner's freshly-created lock and double-acquire. + local claimed="$dir.stale.$$" + if mv "$dir" "$claimed" 2>/dev/null; then + echo "agent-lock: reclaimed stale lock '$name' ($(holder_desc "$claimed/meta"))" >&2 + rm -rf "$claimed" + # mkdir stays the sole grant: a concurrent fresh acquirer may win here, + # in which case our mkdir fails and we correctly defer to it. + if mkdir "$dir" 2>/dev/null; then + write_meta "$name" "$ttl" + return 0 + fi + fi + fi + return 1 +} + +cmd_acquire() { + local name="$1"; shift + local ttl="$DEFAULT_TTL" wait_total=0 + while [ $# -gt 0 ]; do + case "$1" in + --ttl=*) ttl="${1#--ttl=}" ;; + --ttl) shift; ttl="${1:-}" ;; + --wait) wait_total="$DEFAULT_WAIT" ;; + --wait=*) wait_total="${1#--wait=}" ;; + *) usage ;; + esac + shift + done + case "$ttl" in ''|*[!0-9]*) usage ;; esac + case "$wait_total" in *[!0-9]*) usage ;; esac + + local elapsed=0 + while :; do + if try_acquire "$name" "$ttl"; then + exit 0 + fi + if [ "$elapsed" -ge "$wait_total" ]; then + echo "agent-lock: '$name' busy ($(holder_desc "$(meta_path "$name")")); deferring" >&2 + exit 1 + fi + local remaining=$((wait_total - elapsed)) step + step=$(( remaining < WAIT_INTERVAL ? remaining : WAIT_INTERVAL )) + sleep "$step" + elapsed=$((elapsed + step)) + done +} + +cmd_refresh() { + local name="$1" file + file="$(meta_path "$name")" + [ -f "$file" ] || exit 1 + # Re-stamp acquired and bump mtime so the age clock restarts. + local ttl; ttl="$(meta_get ttl "$file")"; [ -n "$ttl" ] || ttl="$DEFAULT_TTL" + write_meta "$name" "$ttl" + exit 0 +} + +cmd_release() { + local name="$1" dir + dir="$(lock_dir "$name")" + rm -rf "$dir" + exit 0 +} + +cmd_status() { + local name="$1" dir file + dir="$(lock_dir "$name")" + file="$(meta_path "$name")" + if [ ! -d "$dir" ]; then + echo "free $name" + exit 0 + fi + local state="held" + is_stale "$name" && state="stale" + echo "$state $name pid=$(meta_get pid "$file") host=$(meta_get host "$file") acquired=$(meta_get acquired "$file") ttl=$(meta_get ttl "$file") age=$(lock_age "$file" 2>/dev/null || echo '?')s" + exit 0 +} + +cmd_path() { + lock_dir "$1" + exit 0 +} + +[ $# -ge 1 ] || usage +subcmd="$1"; shift +[ $# -ge 1 ] || usage +name="$1"; shift +valid_name "$name" || usage + +case "$subcmd" in + acquire) cmd_acquire "$name" "$@" ;; + refresh) cmd_refresh "$name" ;; + release) cmd_release "$name" ;; + status) cmd_status "$name" ;; + path) cmd_path "$name" ;; + *) usage ;; +esac diff --git a/.ai/scripts/apkg-to-orgdrill.py b/.ai/scripts/apkg-to-orgdrill.py new file mode 100755 index 0000000..79e24a4 --- /dev/null +++ b/.ai/scripts/apkg-to-orgdrill.py @@ -0,0 +1,251 @@ +#!/usr/bin/env -S uv run --script +# /// script +# requires-python = ">=3.11" +# dependencies = [] +# /// +"""Convert an Anki .apkg deck into an org-drill file (inverse of flashcard-to-anki.py). + +The flashcard pipeline is otherwise one-directional (org-drill -> apkg). +Decks curated on the phone, and orphaned apkgs whose .org source was never +saved, can't get back into the org source of truth. This recovers them. + +Reading needs no third-party library: an apkg is a zip holding +collection.anki2 / .anki21 (an Anki sqlite db) plus a media blob, so stdlib +zipfile + sqlite3 suffice. genanki is only needed to write apkgs, not read +them. + +Mapping (mirrors flashcard-to-anki.py's parse/build, inverted): + - Deck name (from the apkg) -> #+TITLE: + - Note Front -> ** <Front> :drill: + - Note Back (HTML) -> entry body (<br> -> newlines, + &/</> unescaped, + <hr id="answer"> stripped) + - Note tag -> * <tag> section grouping + (best-effort: the tag is a slug, + so it won't round-trip to the exact + original section title — a human + retitles) + - A fresh :ID: UUID per card -> so the output is org-drill-valid + +GUIDs in flashcard-to-anki.py are derived from the Front text, not the +:ID:, so a deck regenerated from recovered org still matches existing phone +cards by Front. Only Front/Back (Basic) note types convert; other models +(cloze, etc.) are skipped with a warning rather than silently dropped. + +Usage: + apkg-to-orgdrill.py <input.apkg> # one <deck-slug>.org per deck in cwd + apkg-to-orgdrill.py <input.apkg> --output-dir DIR + apkg-to-orgdrill.py <input.apkg> --deck "Name" --output deck.org +""" +from __future__ import annotations + +import argparse +import json +import re +import sqlite3 +import sys +import tempfile +import uuid +import zipfile +from collections import OrderedDict +from dataclasses import dataclass +from pathlib import Path + +# Collection member names Anki uses, newest schema first. +COLLECTION_NAMES = ("collection.anki21", "collection.anki2") + +_BR_RE = re.compile(r"<br\s*/?>", re.IGNORECASE) +_ANSWER_HR_RE = re.compile(r'<hr id="answer">', re.IGNORECASE) +_MEDIA_RE = re.compile(r"<img\b|\[sound:|<audio\b|<video\b", re.IGNORECASE) + + +@dataclass +class Note: + deck: str + front: str + back_html: str + tag: str + + +def html_to_org_body(back_html: str) -> list[str]: + """Invert flashcard-to-anki.py's back-of-card HTML into org body lines. + + <br> (all spellings) and a stray answer <hr> become line breaks; the + entity unescape undoes escape_html, which escaped ``&`` first — so ``&`` + is unescaped last here, or a literally-escaped ``<`` in the source + would wrongly collapse to ``<``. + """ + if not back_html: + return [] + s = _ANSWER_HR_RE.sub("\n", back_html) + s = _BR_RE.sub("\n", s) + s = s.replace("<", "<").replace(">", ">").replace("&", "&") + return s.split("\n") + + +def _slug(title: str) -> str: + return re.sub(r"[^a-z0-9]+", "-", title.lower()).strip("-") + + +def _read_collection(db_path: Path) -> list[Note]: + con = sqlite3.connect(db_path) + try: + row = con.execute("SELECT decks, models FROM col LIMIT 1").fetchone() + if row is None: + raise ValueError("collection has no col row") + decks_json, models_json = row + decks = {int(k): v["name"] for k, v in json.loads(decks_json).items()} + models = { + int(k): [f["name"] for f in v["flds"]] + for k, v in json.loads(models_json).items() + } + + # A note's deck comes from its card; the Default deck (id 1) carries + # no cards from this pipeline, so it never shows up here. + nid_to_did: dict[int, int] = {} + for nid, did in con.execute("SELECT nid, did FROM cards"): + nid_to_did.setdefault(nid, did) + + notes: list[Note] = [] + for nid, mid, flds, tags in con.execute( + "SELECT id, mid, flds, tags FROM notes" + ): + field_names = models.get(mid) + if not field_names or "Front" not in field_names or "Back" not in field_names: + print( + f"apkg-to-orgdrill: skip note {nid} — model is not a Front/Back " + f"type (fields={field_names})", + file=sys.stderr, + ) + continue + fields = flds.split("\x1f") + fi, bi = field_names.index("Front"), field_names.index("Back") + front = fields[fi] if fi < len(fields) else "" + back_html = fields[bi] if bi < len(fields) else "" + + did = nid_to_did.get(nid) + if did is None: + continue # note with no card — orphan + deck = decks.get(did) + if deck is None: + continue + + tag_list = tags.split() + tag = tag_list[0] if tag_list else "drill" + + if _MEDIA_RE.search(back_html): + print( + f"apkg-to-orgdrill: note {nid} references media; org has no " + f"media path (left inline for a human to resolve)", + file=sys.stderr, + ) + notes.append(Note(deck=deck, front=front, back_html=back_html, tag=tag)) + return notes + finally: + con.close() + + +def read_apkg(path: Path) -> list[Note]: + """Read an .apkg and return its Front/Back notes. Raises on a malformed file.""" + with zipfile.ZipFile(path) as z: # BadZipFile if it isn't a zip + names = set(z.namelist()) + col_name = next((n for n in COLLECTION_NAMES if n in names), None) + if col_name is None: + raise ValueError(f"{path}: no collection.anki2/.anki21 inside the apkg") + with tempfile.TemporaryDirectory() as td: + db_path = Path(td) / col_name + db_path.write_bytes(z.read(col_name)) + return _read_collection(db_path) + + +def notes_to_org(notes: list[Note], deck_name: str, *, new_id=None) -> str: + """Render one deck's notes as an org-drill file in the house shape.""" + if new_id is None: + new_id = lambda: str(uuid.uuid4()) # noqa: E731 + groups: "OrderedDict[str, list[Note]]" = OrderedDict() + for n in notes: + groups.setdefault(n.tag, []).append(n) + + lines: list[str] = [f"#+TITLE: {deck_name}", ""] + for tag, group in groups.items(): + lines.append(f"* {tag}") + for n in group: + lines.append(f"** {n.front} :drill:") + lines.append(":PROPERTIES:") + lines.append(f":ID: {new_id()}") + lines.append(":END:") + lines.extend(html_to_org_body(n.back_html)) + lines.append("") + return "\n".join(lines).rstrip("\n") + "\n" + + +def convert(apkg_path: Path, *, new_id=None) -> "OrderedDict[str, str]": + """apkg -> {deck_name: org_text}, one entry per deck that has Front/Back cards.""" + by_deck: "OrderedDict[str, list[Note]]" = OrderedDict() + for n in read_apkg(apkg_path): + by_deck.setdefault(n.deck, []).append(n) + out: "OrderedDict[str, str]" = OrderedDict() + for deck, deck_notes in by_deck.items(): + out[deck] = notes_to_org(deck_notes, deck, new_id=new_id) + return out + + +def main() -> int: + parser = argparse.ArgumentParser( + description="Convert an Anki .apkg deck into an org-drill file.", + ) + parser.add_argument("input", type=Path, help="Path to the .apkg file.") + parser.add_argument("--deck", help="Only convert the deck with this exact name.") + parser.add_argument( + "--output", + type=Path, + help="Output .org path. Requires a single deck (use --deck to pick one).", + ) + parser.add_argument( + "--output-dir", + type=Path, + help="Directory for per-deck .org files (default: current directory).", + ) + args = parser.parse_args() + + input_path = args.input.expanduser().resolve() + if not input_path.is_file(): + print(f"error: {input_path} not found", file=sys.stderr) + return 1 + + by_deck = convert(input_path) + if args.deck: + by_deck = OrderedDict((k, v) for k, v in by_deck.items() if k == args.deck) + if not by_deck: + print(f"error: no deck named {args.deck!r} in {input_path}", file=sys.stderr) + return 1 + if not by_deck: + print(f"error: no Front/Back cards found in {input_path}", file=sys.stderr) + return 1 + + if args.output: + if len(by_deck) != 1: + print( + f"error: --output needs a single deck; {input_path} has " + f"{len(by_deck)} ({', '.join(by_deck)}). Use --deck or --output-dir.", + file=sys.stderr, + ) + return 1 + out = args.output.expanduser().resolve() + out.parent.mkdir(parents=True, exist_ok=True) + deck, org = next(iter(by_deck.items())) + out.write_text(org, encoding="utf-8") + print(f"wrote {out} ({org.count(':drill:')} cards, deck {deck!r})") + return 0 + + out_dir = (args.output_dir or Path.cwd()).expanduser().resolve() + out_dir.mkdir(parents=True, exist_ok=True) + for deck, org in by_deck.items(): + out = out_dir / f"{_slug(deck) or 'deck'}.org" + out.write_text(org, encoding="utf-8") + print(f"wrote {out} ({org.count(':drill:')} cards, deck {deck!r})") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/.ai/scripts/cj-remove-block.py b/.ai/scripts/cj-remove-block.py index 71c7b3d..d5137a3 100755 --- a/.ai/scripts/cj-remove-block.py +++ b/.ai/scripts/cj-remove-block.py @@ -16,8 +16,12 @@ Companion to the /respond-to-cj-comments skill and to cj-scan.py. from __future__ import annotations import argparse +import os import re +import shutil import sys +import tempfile +from datetime import datetime from pathlib import Path SRC_OPEN_RE = re.compile(r"^\s*#\+begin_src\s+cj:", re.IGNORECASE) @@ -57,12 +61,83 @@ def looks_like_cj_range(lines: list[str], start: int, end: int) -> tuple[bool, s f"Line {end} does not look like a #+end_src closing fence " f"(got: {last[:60]!r})" ) + + # The range must hold exactly ONE block. Checking only the first and last + # lines let a drifted range run from one block's opener to a *later* block's + # closer: validation passed and the removal silently deleted everything + # between, prose and headings included. Drift is the case this check exists + # for, so it has to look inside the range, not just at its ends. + for offset, line in enumerate(lines[start:end - 1], start=start + 1): + if SRC_CLOSE_RE.match(line): + return False, ( + f"Range {start}..{end} covers more than one cj block — " + f"a #+end_src appears at line {offset}, before the range ends. " + f"Re-scan for current line numbers; removing this range would " + f"delete everything between the two blocks." + ) + if SRC_OPEN_RE.match(line): + return False, ( + f"Range {start}..{end} covers more than one cj block — " + f"a second #+begin_src cj: appears at line {offset}. " + f"Re-scan for current line numbers." + ) return True, "" +def _backup(path: Path) -> Path: + """Copy path to /tmp before mutating it, mirroring lint-org.el's convention. + + These are Craig's org files. lint-org.el, the other tool that rewrites them, + leaves a /tmp copy before touching anything; this matches it so a bad edit is + always recoverable without reaching for git (which only reaches the last + commit, losing intra-session work). + """ + stamp = datetime.now().strftime("%Y%m%d-%H%M%S") + base = Path(tempfile.gettempdir()) / f"{path.name}.before-cj-remove.{stamp}" + # Never overwrite an earlier backup. The skill removes several annotations + # in quick succession, so a second-resolution stamp collides and the later + # copy would replace the earlier one with already-mutated content — losing + # the pre-session original the backup exists to preserve. + dest = base + n = 2 + while dest.exists(): + dest = base.with_name(f"{base.name}-{n}") + n += 1 + shutil.copy2(path, dest) + return dest + + +def _atomic_write(path: Path, text: str) -> None: + """Write text to path via a temp sibling and os.replace. + + A bare write_text truncates the target on open, so a mid-write failure left + the org file truncated with no complete copy on disk. Writing a temp sibling + and renaming means the file is either its old content or its new content, + never a partial. + """ + # Follow a symlink to the file it names. os.replace would otherwise swap the + # symlink itself for a regular file, leaving the real target holding the old + # content — the edit silently goes nowhere. Resolving also puts the temp + # sibling on the same filesystem as the real file, which os.replace needs. + path = path.resolve() + fd, tmp = tempfile.mkstemp(dir=path.parent, prefix=f".{path.name}.", suffix=".tmp") + os.close(fd) + tmp_path = Path(tmp) + # Carry the original's permissions across. mkstemp creates 0600, and + # defaulting to the umask instead widened a deliberately-restricted file + # (a 0600 org file came back 0644). + shutil.copymode(path, tmp_path) + try: + tmp_path.write_text(text, encoding="utf-8") + os.replace(tmp_path, path) + except BaseException: + tmp_path.unlink(missing_ok=True) + raise + + def remove_range(path: Path, start: int, end: int) -> None: """Read path, validate range looks like cj content, remove the range, write back.""" - text = path.read_text() + text = path.read_text(encoding="utf-8") had_trailing_newline = text.endswith("\n") lines = text.splitlines(keepends=False) @@ -77,7 +152,9 @@ def remove_range(path: Path, start: int, end: int) -> None: new_text += "\n" elif not new_lines and had_trailing_newline: new_text = "" - path.write_text(new_text) + + _backup(path) + _atomic_write(path, new_text) def main() -> int: diff --git a/.ai/scripts/flashcard-stats.py b/.ai/scripts/flashcard-stats.py index 1fa5afb..cb580ac 100755 --- a/.ai/scripts/flashcard-stats.py +++ b/.ai/scripts/flashcard-stats.py @@ -35,7 +35,12 @@ import re import sys from pathlib import Path -CARD_RE = re.compile(r"^\*\*\s+(.+?)\s+:drill:\s*$") +# A card is a level-2 heading whose trailing org tag block includes `drill`. +# Group 1 is the front, group 2 the tag block — so a curated card multi-tagged +# :fundamental:drill: still counts (it would silently drop under a :drill:$ +# anchor, undercounting the deck). HEADING_RE bounds a card's body. +CARD_RE = re.compile(r"^\*\*\s+(.+?)\s+(:[A-Za-z0-9_@#%:]+:)\s*$") +HEADING_RE = re.compile(r"^\*{1,2}\s") ANSWER_RE = re.compile(r"^\*\*\*\s+Answer\b") PROP_START_RE = re.compile(r"^\s*:PROPERTIES:\s*$") PROP_END_RE = re.compile(r"^\s*:END:\s*$") @@ -177,7 +182,8 @@ def parse_cards(lines: list[str]) -> tuple[list[dict], int]: n = len(lines) while i < n: m = CARD_RE.match(lines[i]) - if not m: + tags = [t for t in m.group(2).split(":") if t] if m else [] + if not (m and "drill" in tags): i += 1 continue heading = m.group(1).strip() @@ -188,7 +194,7 @@ def parse_cards(lines: list[str]) -> tuple[list[dict], int]: body_lines: list[str] = [] while i < n: line = lines[i] - if line.startswith("* ") or CARD_RE.match(line): + if HEADING_RE.match(line): break if PROP_START_RE.match(line): prop_count += 1 diff --git a/.ai/scripts/flashcard-to-anki.py b/.ai/scripts/flashcard-to-anki.py index ca4c70b..e369fd8 100755 --- a/.ai/scripts/flashcard-to-anki.py +++ b/.ai/scripts/flashcard-to-anki.py @@ -10,8 +10,15 @@ Parses org-drill structure: - Top-level "* Section" headings become tags on every card under them. - Each "** Card name :drill:" entry becomes a card. Front = heading - text (sans :drill: tag). Back = entry body with newlines converted + text (sans the tag block). Back = entry body with newlines converted to <br>. + - A card may carry a second org tag ("** Card :fundamental:drill:"). + Any heading whose tag block includes `drill` is a card; the other + tags ride along as Anki tags next to the section tag, so a curated + subset stays grep-able in the source. --tag-filter <tag> emits only + the cards carrying that tag, and a subset deck built that way should + pass --guid-salt so its notes get their own GUID space (Anki dedupes + on GUID, so without it the subset imports empty against the full deck). Deck name defaults to the org #+TITLE: (so the phone deck reads as the curated title), falling back to the input basename when the source has @@ -27,6 +34,8 @@ Usage: flashcard-to-anki.py <input.org> flashcard-to-anki.py <input.org> --deck "My Deck Name" flashcard-to-anki.py <input.org> --output /path/to/deck.apkg + flashcard-to-anki.py <input.org> --tag-filter fundamental \ + --deck "DeepSat Fundamentals" --guid-salt fundamentals Requires genanki, which uv resolves automatically via the PEP 723 script metadata above. No venv or system install needed. @@ -47,6 +56,15 @@ import genanki ID_BASE = 1_500_000_000 ID_RANGE = 500_000_000 +# A card is any level-2 heading whose trailing org tag block includes `drill`. +# Group 1 is the front text, group 2 the colon-delimited tag block (e.g. +# ":fundamental:drill:") — so a curated subset can carry a second org tag +# (:fundamental:) and stay grep-able in the source without dropping from the +# full deck. HEADING_RE bounds a card's body at the next L1/L2 heading. +CARD_RE = re.compile(r"^\*\*\s+(.+?)\s+(:[A-Za-z0-9_@#%:]+:)\s*$") +HEADING_RE = re.compile(r"^\*{1,2}\s") +SECTION_RE = re.compile(r"^\*\s+(.+?)\s*$") + def stable_id(name: str, salt: str) -> int: """Derive a deterministic 32-bit id from `name` and a `salt`. @@ -120,33 +138,40 @@ def strip_org_metadata(body_lines: list[str]) -> list[str]: return cleaned -def parse(org_text: str) -> list[tuple[str, str, str]]: - """Return [(front, back_html, tag), ...] for every :drill: card.""" - cards: list[tuple[str, str, str]] = [] - current_section: str | None = None +def parse( + org_text: str, tag_filter: str | None = None +) -> list[tuple[str, str, list[str]]]: + """Return [(front, back_html, anki_tags), ...] for every :drill: card. - section_re = re.compile(r"^\*\s+(.+?)\s*$") - card_re = re.compile(r"^\*\*\s+(.+?)\s+:drill:\s*$") + A card is any level-2 heading whose trailing org tag block includes + `drill`. Non-drill org tags on the heading (e.g. :fundamental:) ride + along as Anki tags next to the section tag, so a curated subset stays + grep-able in the source. When `tag_filter` is set, only cards carrying + that org tag are returned (the subset-deck path). + """ + cards: list[tuple[str, str, list[str]]] = [] + current_section: str | None = None lines = org_text.splitlines() i = 0 while i < len(lines): line = lines[i] - sec = section_re.match(line) + sec = SECTION_RE.match(line) if sec: current_section = sec.group(1).strip() i += 1 continue - card = card_re.match(line) - if card: - front = card.group(1).strip() + m = CARD_RE.match(line) + tags = [t for t in m.group(2).split(":") if t] if m else [] + if m and "drill" in tags: + front = m.group(1).strip() body_lines: list[str] = [] i += 1 while i < len(lines): nxt = lines[i] - if nxt.startswith("* ") or card_re.match(nxt): + if HEADING_RE.match(nxt): break body_lines.append(nxt) i += 1 @@ -156,8 +181,17 @@ def parse(org_text: str) -> list[tuple[str, str, str]]: while body_lines and not body_lines[-1].strip(): body_lines.pop() back_html = "<br>".join(escape_html(ln) for ln in body_lines) - tag = section_to_tag(current_section) if current_section else "drill" - cards.append((front, back_html, tag)) + + org_tags = [t for t in tags if t != "drill"] + if tag_filter and tag_filter not in org_tags: + continue + anki_tags: list[str] = [] + if current_section: + anki_tags.append(section_to_tag(current_section)) + anki_tags.extend(org_tags) + if not anki_tags: + anki_tags = ["drill"] + cards.append((front, back_html, anki_tags)) continue i += 1 @@ -165,15 +199,28 @@ def parse(org_text: str) -> list[tuple[str, str, str]]: return cards -def build(cards: list[tuple[str, str, str]], deck_name: str) -> genanki.Deck: +def card_guid(front: str, guid_salt: str | None) -> str: + """GUID for a card's front. A salt gives a derived subset deck its own + GUID space so its notes don't collide with the full deck's (Anki dedupes + on GUID, which would otherwise import the subset empty). No salt is the + original behavior, so an unsalted deck's GUIDs and SRS state are untouched. + """ + return genanki.guid_for(guid_salt, front) if guid_salt else genanki.guid_for(front) + + +def build( + cards: list[tuple[str, str, list[str]]], + deck_name: str, + guid_salt: str | None = None, +) -> genanki.Deck: deck = genanki.Deck(stable_id(deck_name, "deck"), deck_name) model = make_model(deck_name) - for front, back, tag in cards: + for front, back, tags in cards: note = genanki.Note( model=model, fields=[front, back], - tags=[tag], - guid=genanki.guid_for(front), + tags=tags, + guid=card_guid(front, guid_salt), ) deck.add_note(note) return deck @@ -219,6 +266,16 @@ def main() -> int: help="Output .apkg path. Defaults to " "~/sync/phone/anki/<input-basename>.apkg.", ) + parser.add_argument( + "--tag-filter", + help="Emit only cards carrying this org tag (e.g. --tag-filter " + "fundamental for a curated subset deck).", + ) + parser.add_argument( + "--guid-salt", + help="Salt note GUIDs so a subset deck gets its own GUID space and " + "imports non-empty without disturbing the full deck's SRS state.", + ) args = parser.parse_args() input_path: Path = args.input.expanduser().resolve() @@ -231,12 +288,18 @@ def main() -> int: output_path: Path = (args.output or default_output_path(input_path)).expanduser().resolve() output_path.parent.mkdir(parents=True, exist_ok=True) - cards = parse(org_text) + cards = parse(org_text, tag_filter=args.tag_filter) if not cards: - print(f"error: no :drill: cards found in {input_path}", file=sys.stderr) + if args.tag_filter: + print( + f"error: no :drill: cards tagged :{args.tag_filter}: in {input_path}", + file=sys.stderr, + ) + else: + print(f"error: no :drill: cards found in {input_path}", file=sys.stderr) return 1 - deck = build(cards, deck_name) + deck = build(cards, deck_name, guid_salt=args.guid_salt) genanki.Package(deck).write_to_file(str(output_path)) print(f"wrote {output_path} ({len(cards)} cards, deck '{deck_name}')") return 0 diff --git a/.ai/scripts/inbox-send.py b/.ai/scripts/inbox-send.py index 1ebb636..663efcb 100755 --- a/.ai/scripts/inbox-send.py +++ b/.ai/scripts/inbox-send.py @@ -31,6 +31,7 @@ import os import re import shutil import sys +import tempfile from datetime import datetime from pathlib import Path @@ -48,7 +49,7 @@ def resolve_roots() -> list[Path]: config = Path.home() / ".claude" / "inbox-roots.txt" if config.is_file(): paths: list[Path] = [] - for line in config.read_text().splitlines(): + for line in config.read_text(encoding="utf-8").splitlines(): line = line.strip() if line and not line.startswith("#"): paths.append(Path(line).expanduser()) @@ -69,17 +70,28 @@ def discover_projects(roots: list[Path]) -> list[Path]: a specific project root (included directly if it qualifies). """ projects: list[Path] = [] + seen: set[Path] = set() + + def _add(p: Path) -> None: + # Dedupe on the resolved path: a roots config naming both a parent and + # one of its children would otherwise list the child project twice, at + # two different indices. + key = p.resolve() + if key not in seen: + seen.add(key) + projects.append(p) + for root in roots: if not root.is_dir(): continue if _is_project(root): - projects.append(root) + _add(root) continue for child in sorted(root.iterdir()): if not child.is_dir(): continue if _is_project(child): - projects.append(child) + _add(child) return projects @@ -194,6 +206,39 @@ def uniquify(dest: Path) -> Path: n += 1 +def _atomic_write(dest: Path, writer) -> None: + """Write to a temp file in dest's directory, then rename it into place. + + dest is another project's inbox/, and a direct write truncates the target + on open, so any mid-write failure (a full disk, an encoding error, an + interrupted process) leaves a zero-byte .org there. inbox-status counts + that phantom as a pending handoff and blocks a turn in the receiving + project over a file with no content and no sender (2026-07-23). Writing to + a temp sibling and os.replace-ing means the inbox only ever sees a complete + file. os.replace is atomic within one filesystem, and the temp sits in the + same directory as dest, so it is. + + `writer` receives the open temp path and fills it. On any failure the temp + is removed and the error re-raised, so a caught error never leaves debris. + """ + fd, tmp = tempfile.mkstemp( + dir=dest.parent, prefix=".inbox-send-", suffix=dest.suffix + ) + os.close(fd) + tmp_path = Path(tmp) + # mkstemp creates the temp 0600; give the delivered file the umask-default + # mode the old direct write produced, so inbox files stay readable as before. + umask = os.umask(0) + os.umask(umask) + os.chmod(tmp_path, 0o666 & ~umask) + try: + writer(tmp_path) + os.replace(tmp_path, dest) + except BaseException: + tmp_path.unlink(missing_ok=True) + raise + + def send_text( target_inbox: Path, message: str, @@ -209,7 +254,8 @@ def send_text( raise ValueError(f"could not derive a slug from text: {message!r}") filename = f"{now.strftime(TS_FILENAME_FMT)}-from-{source_name}-{slug}.org" dest = uniquify(target_inbox / filename) - dest.write_text(build_text_org(message, source_name, now.strftime(TS_DOC_FMT))) + body = build_text_org(message, source_name, now.strftime(TS_DOC_FMT)) + _atomic_write(dest, lambda p: p.write_text(body, encoding="utf-8")) return dest @@ -229,7 +275,7 @@ def send_file( ext = src_path.suffix filename = f"{now.strftime(TS_FILENAME_FMT)}-from-{source_name}-{slug}{ext}" dest = uniquify(target_inbox / filename) - shutil.copy2(src_path, dest) + _atomic_write(dest, lambda p: shutil.copyfile(src_path, p)) return dest @@ -310,7 +356,10 @@ def main() -> int: else: assert args.file is not None dest = send_file(target_inbox, args.file, source_name, args.name, now) - except (ValueError, FileNotFoundError) as exc: + except (ValueError, OSError) as exc: + # OSError covers FileNotFoundError (missing source), PermissionError + # (unreadable source), and any atomic-write failure — all should + # surface as the clean "inbox-send: <message>" error, never a traceback. print(f"inbox-send: {exc}", file=sys.stderr) return 1 diff --git a/.ai/scripts/inbox-status b/.ai/scripts/inbox-status index b917144..17031af 100755 --- a/.ai/scripts/inbox-status +++ b/.ai/scripts/inbox-status @@ -35,6 +35,7 @@ mapfile -t pending < <(find inbox -maxdepth 1 -type f \ ! -name '.gitkeep' \ ! -name 'lint-followups.org' \ ! -name 'PROCESSED-*' \ + ! -name '.inbox-send-*' \ -printf '%f\n' 2>/dev/null | sort) n=${#pending[@]} diff --git a/.ai/scripts/lint-org.el b/.ai/scripts/lint-org.el index 55727ef..33dc52f 100644 --- a/.ai/scripts/lint-org.el +++ b/.ai/scripts/lint-org.el @@ -38,7 +38,9 @@ ;; empty-heading bare stars with no title ;; malformed-priority-cookie [#x]-shaped token org rejected ;; level2-done-without-closed completed level-2 task with no CLOSED +;; task-missing-last-reviewed open level-2 task with no :LAST_REVIEWED: ;; subtask-done-not-dated level-3+ done sub-task still a DONE keyword +;; dated-log-heading-active-timestamp dated-log heading with a live SCHEDULED/DEADLINE ;; (anything else) surfaced as judgment with checker name ;; ;; Output format on stdout: @@ -74,6 +76,18 @@ The CLI defaults this to t (a linter reports, it doesn't write); `--fix' is what enables writes on a command-line run.") (defvar lo-current-file nil "Path of the file currently being processed.") + +(defun lo--spec-file-p () + "Non-nil when the current file lives under a docs/specs/ directory. +The four todo-format-family checkers encode todo.org completion conventions +and misfire on a spec: a spec's Decisions section legitimately carries a +level-2 DONE with no CLOSED cookie, and its review-history section carries +level-2 dated headings. docs/specs/ is the canonical spec home per the +docs-lifecycle rule, so a path segment match is the scope test. Link, +table, and structural checks still run on specs — only the todo-format +family is scoped out." + (and lo-current-file + (string-match-p "/docs/specs/" (expand-file-name lo-current-file)))) (defvar lo-followups-file nil "When non-nil, after a non-check run any judgment items are appended to this path as an org section dated today. The file is created if missing.") @@ -292,6 +306,52 @@ Craig-specific annotation marker rather than Babel src-block syntax." (lo--goto-line line) (looking-at-p "^[ \t]*#\\+begin_src[ \t]+cj:"))) +(defvar-local lo--matched-blocks-cache nil + "Cons of (TICK . REGIONS) memoizing `lo--matched-block-regions'. +TICK is the `buffer-chars-modified-tick' the regions were computed at, so a +fix applied mid-pass invalidates them.") + +(defun lo--matched-block-regions () + "Return ((BEGIN-LINE . END-LINE) ...) for every correctly paired block. +Scans lines directly rather than asking org, because org's own parser is what +mis-reads these blocks: a heading-shaped line inside a verbatim body reads as a +structural break and loses the open block. The scan applies org's real rule — +once a block is open, only its own `#+end_TYPE' closes it, so a nested +`#+begin_' or a foreign `#+end_' in the body is just text." + (let ((tick (buffer-chars-modified-tick))) + (if (eql (car lo--matched-blocks-cache) tick) + (cdr lo--matched-blocks-cache) + (let ((case-fold-search t) + (regions nil) (open-type nil) (open-line nil) (line 0)) + (save-excursion + (goto-char (point-min)) + (while (not (eobp)) + (setq line (1+ line)) + (let ((text (buffer-substring-no-properties + (line-beginning-position) (line-end-position)))) + (cond + (open-type + (when (string-match + (format "\\`[ \t]*#\\+end_%s[ \t]*\\'" + (regexp-quote open-type)) + text) + (push (cons open-line line) regions) + (setq open-type nil open-line nil))) + ((string-match "\\`[ \t]*#\\+begin_\\([^ \t\n]+\\)" text) + (setq open-type (match-string 1 text) + open-line line)))) + (forward-line 1))) + (setq lo--matched-blocks-cache (cons tick (nreverse regions))) + (cdr lo--matched-blocks-cache))))) + +(defun lo--in-matched-block-p (line) + "Non-nil when LINE sits within a correctly paired block, delimiters included. +org-lint reports `invalid-block' at the delimiter lines themselves, so the +range has to be inclusive for the suppression to reach them." + (cl-some (lambda (region) + (and (>= line (car region)) (<= line (cdr region)))) + (lo--matched-block-regions))) + (defun lo--handle-item (item) (let ((name (lo--checker-name item)) (line (lo--line item)) @@ -304,6 +364,13 @@ Craig-specific annotation marker rather than Babel src-block syntax." wrong-header-argument)) (lo--cj-comment-block-opener-p line)) nil) + ;; `invalid-block' on a block that is in fact correctly paired — the + ;; checker is org-lint's own, so this filters its output rather than + ;; fixing a local checker. A genuinely unterminated block isn't in any + ;; matched region, so it still reports. + ((and (eq name 'invalid-block) + (lo--in-matched-block-p line)) + nil) ((eq name 'item-number) (lo--apply-or-preview name line msg #'lo-fix-item-number)) ((eq name 'missing-language-in-src-block) @@ -525,6 +592,42 @@ the live file on the next `task-sorted'." "level-2 DONE/CANCELLED has no CLOSED date — add CLOSED: [YYYY-MM-DD Day]; task-sorted's aging step archives an undated completed task immediately")))))))) ;;; --------------------------------------------------------------------------- +;;; task-missing-last-reviewed check (claude-rules/todo-format.md) +;; +;; A task is stamped `:LAST_REVIEWED:' when it is *created*, not a review cycle +;; later. An agent filing a task has just written its body and graded its +;; priority, which is a review by any honest reading — so a fresh task that +;; carries no stamp reads as "never reviewed" and lands at the top of the next +;; staleness batch, where re-reviewing it is pure ceremony. Every task filed +;; during the 2026-07-23 sweep hit exactly that, which is what prompted the rule. +;; +;; Judgment-only, deliberately. The stamp's whole value is that its date is +;; true, and nothing here can know when an unstamped task was actually last +;; looked at. Auto-stamping today's date would convert a "nobody has reviewed +;; this" signal into a false "reviewed today" one — worse than the gap it +;; closes. Flag it; a human or the filing workflow supplies the honest date. +;; +;; Scope matches `task-review-staleness.sh' exactly (level-2, open keyword, +;; priority cookie), so the checker and the staleness count never disagree +;; about which headings are in the review pool. + +(defun lo--check-task-missing-last-reviewed () + "Flag an open level-2 task with a priority cookie and no `:LAST_REVIEWED:'." + (save-excursion + (goto-char (point-min)) + (let ((case-fold-search nil)) + (while (re-search-forward "^\\*\\* \\(TODO\\|DOING\\|VERIFY\\) \\[#[A-D]\\]" nil t) + (let ((hline (line-number-at-pos)) + (entry-end (save-excursion (outline-next-heading) (point)))) + (save-excursion + (forward-line 1) + (unless (re-search-forward "^[ \t]*:LAST_REVIEWED:[ \t]*[[0-9]" + entry-end t) + (lo--emit-judgment + 'task-missing-last-reviewed hline + "task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed")))))))) + +;;; --------------------------------------------------------------------------- ;;; level-3+ dated-header check (claude-rules/todo-format.md) ;; ;; The inverse of the level-2 check above. A completed sub-task — a heading at @@ -551,6 +654,41 @@ Emits one judgment item per offending heading (checker "level-3+ done sub-task should be a dated event-log entry (todo-format.md): run todo-cleanup.el --convert-subtasks to rewrite it"))))) ;;; --------------------------------------------------------------------------- +;;; dated-log heading with a stale active planning timestamp (todo-format.md) +;; +;; The mechanical backstop for the planning-line-strip rule. A dated event-log +;; heading (`<stars> YYYY-MM-DD Day @ ...', no TODO keyword) records completed +;; work — its date lives in the heading. An active `<...>' SCHEDULED or DEADLINE +;; left on it pins the entry to the agenda forever: org renders any headline with +;; an active planning timestamp, keyword or not, so a stale SCHEDULED shows as +;; weeks-overdue long after the work is done. Invisible to a keyword scan (no +;; TODO) and it survives --archive-done, so nothing else catches it. The +;; completion rewrite and todo-cleanup --convert-subtasks now strip the planning +;; line; this flags any that slipped through before that landed, the same way +;; subtask-done-not-dated backstops the depth rule. Judgment-only. + +(defun lo--check-dated-log-active-timestamp () + "Flag a dated event-log heading that still carries an active SCHEDULED/DEADLINE. +The heading matches `<stars> YYYY-MM-DD Day @ ...' with no TODO keyword; an +active `<...>' planning timestamp in its entry is the defect. An inactive +`[...]' timestamp is ignored (org doesn't render it on the agenda). Emits one +judgment item per offending heading (checker `dated-log-heading-active-timestamp')." + (save-excursion + (goto-char (point-min)) + (let ((case-fold-search nil)) + (while (re-search-forward + "^\\*+ [0-9]\\{4\\}-[0-9]\\{2\\}-[0-9]\\{2\\} [A-Za-z]+ @ " nil t) + (let ((hline (line-number-at-pos)) + (entry-end (save-excursion (outline-next-heading) (point)))) + (save-excursion + (forward-line 1) + (when (re-search-forward + "^[ \t]*\\(?:SCHEDULED\\|DEADLINE\\):[ \t]*<" entry-end t) + (lo--emit-judgment + 'dated-log-heading-active-timestamp hline + "dated-log heading carries an active SCHEDULED/DEADLINE — org renders any active planning timestamp (keyword or not), so it stays on the agenda as weeks-overdue; delete the planning line (todo-format.md)")))))))) + +;;; --------------------------------------------------------------------------- ;;; File processing (defun lo--backup (file) @@ -584,14 +722,22 @@ left unmodified and mechanical entries are recorded with :preview t." ;; After org-lint items: the custom table-standard scan. Runs on the ;; post-fix buffer; judgment-only, so order doesn't perturb fixes. (lo--check-tables) - ;; Same shape: flag level-2 dated headers (completion defects). - (lo--check-level2-dated-headers) - ;; Structural heading defects org-lint doesn't cover. + ;; Structural heading defects org-lint doesn't cover. These run on + ;; every org file, specs included. (lo--check-indented-headings) (lo--check-empty-headings) (lo--check-malformed-priority-cookies) - (lo--check-level2-done-without-closed) - (lo--check-subtask-done-not-dated) + ;; The todo-format family encodes todo.org completion conventions and + ;; misfires on a spec (a Decisions section's undated DONE, a + ;; review-history dated heading, a phases task with no LAST_REVIEWED). + ;; Scope them out of docs/specs/; link, table, and structural checks + ;; above still run there. + (unless (lo--spec-file-p) + (lo--check-level2-dated-headers) + (lo--check-level2-done-without-closed) + (lo--check-task-missing-last-reviewed) + (lo--check-subtask-done-not-dated) + (lo--check-dated-log-active-timestamp)) (when (and (not lo-check-only) (buffer-modified-p)) (save-buffer))) (with-current-buffer buf (set-buffer-modified-p nil)) diff --git a/.ai/scripts/route_recommend.py b/.ai/scripts/route_recommend.py index 7b36405..12ab132 100644 --- a/.ai/scripts/route_recommend.py +++ b/.ai/scripts/route_recommend.py @@ -71,6 +71,15 @@ def recommend(item: str, projects: list[str]) -> tuple[str | None, str]: if not projects: return (None, "none") + # Collapse identical names first. Projects are addressed by bare basename, so + # two projects sharing one across roots (~/code/notes, ~/projects/notes) arrive + # twice; both literal-match, and the tie test below then read that as ambiguity + # and downgraded a correct strong match to weak. Deduping here rather than in + # discover_destination_names protects every caller of the pure core, not just + # the CLI path. Order-preserving, and it collapses only identical names — two + # *different* projects matching is real ambiguity and still downgrades. + projects = list(dict.fromkeys(projects)) + item_lower = item.lower() item_tokens = _tokens(item) diff --git a/.ai/scripts/tests/agent-lock.bats b/.ai/scripts/tests/agent-lock.bats new file mode 100644 index 0000000..dbcffe1 --- /dev/null +++ b/.ai/scripts/tests/agent-lock.bats @@ -0,0 +1,214 @@ +#!/usr/bin/env bats +# +# Tests for claude-templates/.ai/scripts/agent-lock — a mkdir-atomic advisory +# lock helper for agent workflows (sentry's single-runner and roam-write +# locks). flock can't span an agent's tool calls: every Bash call is its own +# short-lived shell, so a flock dies with the call that took it. This helper +# persists the lock on disk between calls and self-clears after a crash via +# age-based staleness reclaim. +# +# Contract under test: +# agent-lock acquire <name> [--ttl=SECONDS] [--wait[=SECONDS]] +# exit 0 → acquired (fresh, or reclaimed from a stale prior holder). +# exit 1 → busy: a live lock holds <name>; deferred (note on stderr). +# exit 2 → usage error (bad/absent name, unknown subcommand). +# agent-lock refresh <name> → re-touch a held lock (heartbeat); exit 1 if absent. +# agent-lock release <name> → remove the lock; idempotent (exit 0 if already free). +# agent-lock status <name> → print free|held|stale + metadata; exit 0 (query). +# agent-lock path <name> → print the resolved lock dir path; does not create it. +# +# Staleness is age-based on the metadata file's mtime versus the lock's own +# recorded TTL, so a crashed holder's lock expires instead of wedging every +# later acquire. Heartbeat (refresh) re-touches the mtime, keeping a live +# holder's lock young. Every reclaim surfaces a note (never silent). +# +# Lock home: /run/user/<uid>/agent-locks/<name>/ (tmpfs: host-local, out of +# every repo, cleared on reboot), with ~/.cache/agent-locks/ as the fallback +# where no runtime dir exists. AGENT_LOCK_DIR overrides the base for tests and +# advanced callers; the helper otherwise owns the path scheme and callers pass +# only names. +# +# Strategy: AGENT_LOCK_DIR points every lock at a temp base, so tests never +# touch a real runtime dir. Staleness is exercised by aging the metadata +# file's mtime with `touch` rather than sleeping. + +SCRIPT="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/agent-lock" +BASH_BIN="$(command -v bash)" + +setup() { + TEST_DIR="$(mktemp -d -t agent-lock-bats.XXXXXX)" + LOCK_BASE="$TEST_DIR/locks" +} + +teardown() { + rm -rf "$TEST_DIR" +} + +lock() { + run env AGENT_LOCK_DIR="$LOCK_BASE" "$BASH_BIN" "$SCRIPT" "$@" +} + +# meta-file path for a lock name, for direct inspection / aging. +meta_of() { + printf '%s/%s/meta\n' "$LOCK_BASE" "$1" +} + +# ---- acquire: fresh win + metadata -------------------------------------- + +@test "acquire: fresh name wins (exit 0) and writes pid/host/timestamp/ttl" { + lock acquire job + [ "$status" -eq 0 ] + local meta; meta="$(meta_of job)" + [ -f "$meta" ] + grep -q "^pid=$$\|^pid=[0-9][0-9]*$" "$meta" + grep -q "^host=$(uname -n)$" "$meta" + grep -qE "^acquired=[0-9]{4}-[0-9]{2}-[0-9]{2}T" "$meta" + grep -qE "^ttl=[0-9]+$" "$meta" +} + +@test "acquire: honors an explicit --ttl in the metadata" { + lock acquire job --ttl=45 + [ "$status" -eq 0 ] + grep -q "^ttl=45$" "$(meta_of job)" +} + +# ---- acquire: contention (one winner) ----------------------------------- + +@test "acquire: a second acquire of a live lock defers (exit 1, note)" { + lock acquire job + [ "$status" -eq 0 ] + lock acquire job + [ "$status" -eq 1 ] + [[ "$output" == *job* ]] +} + +@test "acquire: two racing acquires yield exactly one winner" { + # Fire both without releasing; exactly one mkdir wins. + env AGENT_LOCK_DIR="$LOCK_BASE" "$BASH_BIN" "$SCRIPT" acquire race & p1=$! + env AGENT_LOCK_DIR="$LOCK_BASE" "$BASH_BIN" "$SCRIPT" acquire race & p2=$! + local r1=0 r2=0 + wait $p1 || r1=$? + wait $p2 || r2=$? + # One exits 0 (won), one exits 1 (deferred). + [ "$((r1 + r2))" -eq 1 ] +} + +# ---- release: frees the lock -------------------------------------------- + +@test "release: frees a held lock so the next acquire wins" { + lock acquire job + [ "$status" -eq 0 ] + lock release job + [ "$status" -eq 0 ] + [ ! -d "$LOCK_BASE/job" ] + lock acquire job + [ "$status" -eq 0 ] +} + +@test "release: is idempotent on an already-free lock (exit 0)" { + lock release never-held + [ "$status" -eq 0 ] +} + +# ---- staleness reclaim (surfaced, never silent) ------------------------- + +@test "acquire: reclaims a stale lock and surfaces the reclaim note" { + lock acquire job --ttl=1 + [ "$status" -eq 0 ] + # Age the metadata mtime well past the 1s TTL. + touch -d '1 hour ago' "$(meta_of job)" + lock acquire job --ttl=1 + [ "$status" -eq 0 ] + [[ "$output" == *reclaim* ]] + [[ "$output" == *job* ]] + # The reclaim installed fresh metadata (young again), not the aged holder's. + lock status job + [[ "$output" == *held* ]] + [[ "$output" != *stale* ]] +} + +@test "acquire: a lock inside its TTL is not stale (stays deferred)" { + lock acquire job --ttl=3600 + [ "$status" -eq 0 ] + lock acquire job --ttl=3600 + [ "$status" -eq 1 ] +} + +# ---- heartbeat (refresh keeps a live lock young) ------------------------ + +@test "refresh: re-touches a held lock so it is no longer stale" { + lock acquire job --ttl=1 + [ "$status" -eq 0 ] + touch -d '1 hour ago' "$(meta_of job)" + lock status job + [[ "$output" == *stale* ]] + lock refresh job + [ "$status" -eq 0 ] + lock status job + [[ "$output" == *held* ]] + [[ "$output" != *stale* ]] +} + +@test "refresh: an absent lock cannot be refreshed (exit 1)" { + lock refresh nothing + [ "$status" -eq 1 ] +} + +# ---- status query ------------------------------------------------------- + +@test "status: reports free for an unheld lock (exit 0)" { + lock status job + [ "$status" -eq 0 ] + [[ "$output" == *free* ]] +} + +@test "status: reports held with metadata for a live lock" { + lock acquire job --ttl=3600 + lock status job + [ "$status" -eq 0 ] + [[ "$output" == *held* ]] + [[ "$output" == *"host=$(uname -n)"* ]] +} + +# ---- path resolution: runtime dir home with cache fallback -------------- + +@test "path: resolves under AGENT_LOCK_DIR when set" { + lock path job + [ "$status" -eq 0 ] + [ "$output" = "$LOCK_BASE/job" ] + [ ! -d "$LOCK_BASE/job" ] # path does not create the lock +} + +@test "path: prefers the runtime dir home when no override is set" { + local rt="$TEST_DIR/run" + mkdir -p "$rt" + run env -u AGENT_LOCK_DIR XDG_RUNTIME_DIR="$rt" "$BASH_BIN" "$SCRIPT" path job + [ "$status" -eq 0 ] + [ "$output" = "$rt/agent-locks/job" ] +} + +@test "path: falls back to the cache home when no runtime dir exists" { + local home="$TEST_DIR/home" + mkdir -p "$home" + run env -u AGENT_LOCK_DIR -u XDG_RUNTIME_DIR -u XDG_CACHE_HOME \ + HOME="$home" "$BASH_BIN" "$SCRIPT" path job + [ "$status" -eq 0 ] + [ "$output" = "$home/.cache/agent-locks/job" ] +} + +# ---- usage errors ------------------------------------------------------- + +@test "usage: a missing name is a usage error (exit 2)" { + lock acquire + [ "$status" -eq 2 ] +} + +@test "usage: a name with a slash is rejected (exit 2)" { + lock acquire bad/name + [ "$status" -eq 2 ] +} + +@test "usage: an unknown subcommand is a usage error (exit 2)" { + lock frobnicate job + [ "$status" -eq 2 ] +} diff --git a/.ai/scripts/tests/flashcard-sync.bats b/.ai/scripts/tests/flashcard-sync.bats index 608a280..e6ffc21 100644 --- a/.ai/scripts/tests/flashcard-sync.bats +++ b/.ai/scripts/tests/flashcard-sync.bats @@ -6,6 +6,7 @@ setup() { SCRIPT_DIR="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)" SYNC="$SCRIPT_DIR/flashcard-sync" + STATS="$SCRIPT_DIR/flashcard-stats.py" TMP="$(mktemp -d)" } @@ -36,3 +37,27 @@ EOF [ "$status" -eq 1 ] [ ! -f "$HOME/sync/phone/anki/dirty.apkg" ] } + +@test "flashcard-stats: a multi-tagged :fundamental:drill: card still counts" { + # Regression guard: a curated card carrying a second org tag must not drop + # from the count. A :drill:$ anchor would have counted only one card here. + cat > "$TMP/multitag.org" <<'EOF' +#+TITLE: Multitag Test + +* Orbital Regimes +** What is LEO? :fundamental:drill: +:PROPERTIES: +:ID: c1 +:END: +Low Earth Orbit is the region below about 2000 kilometers. +** What is GEO? :drill: +:PROPERTIES: +:ID: c2 +:END: +Geostationary orbit sits at roughly 35786 kilometers of altitude. +EOF + run python3 "$STATS" "$TMP/multitag.org" + [ "$status" -eq 0 ] + [[ "$output" == *"Cards: 2"* ]] + [[ "$output" == *clean* ]] +} diff --git a/.ai/scripts/tests/inbox-status.bats b/.ai/scripts/tests/inbox-status.bats index bc8a734..27a497e 100644 --- a/.ai/scripts/tests/inbox-status.bats +++ b/.ai/scripts/tests/inbox-status.bats @@ -45,6 +45,18 @@ teardown() { [[ "$output" == *"0 pending"* ]] } +@test "inbox-status: ignores an in-flight .inbox-send-* temp file" { + mkdir "$TMP/inbox" + # inbox-send writes to a .inbox-send-* temp then renames it into place; + # during that window the temp must not read as a pending handoff, or a + # concurrent boundary check blocks on a file that's about to become real. + touch "$TMP/inbox/.inbox-send-abc123.org" + cd "$TMP" + run "$SCRIPT" + [ "$status" -eq 0 ] + [[ "$output" == *"0 pending"* ]] +} + @test "inbox-status: -q suppresses the per-item lines" { mkdir "$TMP/inbox" echo body > "$TMP/inbox/handoff.org" diff --git a/.ai/scripts/tests/test-lint-org.el b/.ai/scripts/tests/test-lint-org.el index 8e3e190..ceee209 100644 --- a/.ai/scripts/tests/test-lint-org.el +++ b/.ai/scripts/tests/test-lint-org.el @@ -193,6 +193,65 @@ real suspicious-language warning here #+end_src ") +;; invalid-block, false-positive case — a correctly paired example block whose +;; body holds a heading-shaped line. org's parser reads the `** ' inside the +;; verbatim body as a structural break, loses the open block, and flags BOTH +;; delimiters as "Possible incomplete block". +(defconst lo-test--verbatim-heading-block "\ +* Heading + +#+begin_example +** Feature Name or Topic +Body line. +#+end_example + +Trailing prose. +") + +;; invalid-block, literal-delimiter case — a paired src block whose body holds +;; a literal `#+end_example' plus a heading-shaped line. Only `#+end_src' +;; closes a src block, so all three findings here are false. +(defconst lo-test--literal-end-in-src "\ +* Heading + +#+begin_src text +#+end_example +** heading shaped +#+end_src +") + +;; invalid-block, uppercase-delimiter case — org accepts #+BEGIN_/#+END_ in +;; either case, and the pre-fix script flagged both delimiters here too. +(defconst lo-test--uppercase-verbatim-block "\ +* Heading + +#+BEGIN_EXAMPLE +** heading shaped +#+END_EXAMPLE +") + +;; invalid-block, genuine case — a block that really is never closed. The +;; suppression must not reach this one. +(defconst lo-test--unterminated-block "\ +* Heading + +#+begin_example +truly unterminated block body +") + +;; A genuinely unterminated block *after* a correctly paired one — verifies the +;; suppression is scoped per block rather than per file. +(defconst lo-test--paired-then-unterminated "\ +* Heading + +#+begin_example +** heading shaped +#+end_example + +#+begin_example +never closed +") + ;; Mixed fixture — each category once. (defconst lo-test--mixed "\ * Mixed @@ -392,6 +451,55 @@ suspicious-language judgment." (should (= 1 suspicious)))) ;;; --------------------------------------------------------------------------- +;;; invalid-block — false positives on correctly paired verbatim blocks + +(ert-deftest lo-verbatim-heading-block-emits-no-invalid-block () + "Normal: a paired example block containing a heading-shaped body line emits +no invalid-block judgment. Both delimiters are flagged by org-lint because the +parser treats the `** ' inside the verbatim body as a structural break." + (let* ((out (lo-test--run lo-test--verbatim-heading-block)) + (res (plist-get out :result)) + (judgments (lo-test--judgments (plist-get out :issues)))) + ;; File untouched, no fixes applied — suppression only, never a rewrite. + (should (equal lo-test--verbatim-heading-block res)) + (should (= 0 (plist-get out :fixes))) + (should-not (member 'invalid-block (lo-test--checkers judgments))))) + +(ert-deftest lo-literal-end-delimiter-in-src-emits-no-invalid-block () + "Boundary: a paired src block whose body holds a literal `#+end_example' and +a heading-shaped line emits no invalid-block judgment. Only `#+end_src' closes +a src block, so the interior delimiter is body text." + (let* ((out (lo-test--run lo-test--literal-end-in-src)) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should-not (member 'invalid-block (lo-test--checkers judgments))))) + +(ert-deftest lo-uppercase-verbatim-block-emits-no-invalid-block () + "Boundary: block delimiters are case-insensitive in org, so an uppercase +`#+BEGIN_EXAMPLE' pair is suppressed the same as a lowercase one." + (let* ((out (lo-test--run lo-test--uppercase-verbatim-block)) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should-not (member 'invalid-block (lo-test--checkers judgments))))) + +(ert-deftest lo-unterminated-block-still-emits-invalid-block () + "Error: a block that is never closed still emits its invalid-block judgment. +This is the finding the checker exists for — the suppression must not mask it." + (let* ((out (lo-test--run lo-test--unterminated-block)) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should (member 'invalid-block (lo-test--checkers judgments))))) + +(ert-deftest lo-invalid-block-suppression-is-scoped-per-block () + "Boundary: a paired block and an unterminated block in the same file — the +paired one is suppressed and the unterminated one still reports. Exactly one +invalid-block judgment, and it points at the unterminated opener (line 7)." + (let* ((out (lo-test--run lo-test--paired-then-unterminated)) + (judgments (lo-test--judgments (plist-get out :issues))) + (invalid (cl-remove-if-not + (lambda (i) (eq (plist-get i :checker) 'invalid-block)) + judgments))) + (should (= 1 (length invalid))) + (should (= 7 (plist-get (car invalid) :line))))) + +;;; --------------------------------------------------------------------------- ;;; --check mode (ert-deftest lo-check-mode-does-not-modify-file () @@ -739,6 +847,48 @@ missing-rules violation." (judgments (lo-test--judgments (plist-get out :issues)))) (should-not (member 'subtask-done-not-dated (lo-test--checkers judgments))))) +;;; dated-log-heading-active-timestamp check (stale SCHEDULED/DEADLINE on a +;;; completed dated-log entry — the home 2026-07-17 agenda-pollution bug) + +(ert-deftest lo-dated-log-active-scheduled-is-flagged () + "A dated-log entry still carrying an active SCHEDULED is flagged: org renders +it on the agenda forever despite the missing keyword." + (let* ((out (lo-test--run + "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 trip booked\nSCHEDULED: <2026-06-18 Thu>\nBody.\n")) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should (= 0 (plist-get out :fixes))) ; judgment-only, never auto-fixed + (should (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments))))) + +(ert-deftest lo-dated-log-active-deadline-is-flagged () + "An active DEADLINE on a dated-log entry is flagged too." + (let* ((out (lo-test--run + "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 shipped\nDEADLINE: <2026-06-25 Thu>\n")) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments))))) + +(ert-deftest lo-dated-log-clean-entry-not-flagged () + "A dated-log entry with no active planning timestamp is correct — not flagged." + (let* ((out (lo-test--run + "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 done cleanly\nBody only.\n")) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should-not (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments))))) + +(ert-deftest lo-dated-log-inactive-timestamp-not-flagged () + "An inactive [..] timestamp doesn't render on the agenda, so it isn't flagged — +only active <..> planning timestamps are the defect." + (let* ((out (lo-test--run + "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 recorded\nSCHEDULED: [2026-06-18 Thu]\n")) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should-not (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments))))) + +(ert-deftest lo-dated-log-active-scheduled-on-live-todo-not-flagged () + "A live TODO (keyword present) that legitimately carries an active SCHEDULED is +not a dated-log heading, so this checker leaves it alone." + (let* ((out (lo-test--run + "* Open Work\n\n** TODO [#B] Parent\n*** TODO [#C] real upcoming task\nSCHEDULED: <2026-06-18 Thu>\n")) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should-not (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments))))) + ;;; --------------------------------------------------------------------------- ;;; structural heading checks (org-lint gaps) @@ -817,3 +967,134 @@ heading, so it is not flagged — only two-or-more indented stars are." (provide 'test-lint-org) ;;; test-lint-org.el ends here + +;;; --------------------------------------------------------------------------- +;;; task-missing-last-reviewed (claude-rules/todo-format.md) + +(ert-deftest lo-task-without-last-reviewed-is-judgment () + "An open level-2 task with no :LAST_REVIEWED: is flagged." + (let* ((out (lo-test--run "* Open Work\n** TODO [#B] A task :feature:\nBody.\n")) + (js (lo-test--judgments (plist-get out :issues)))) + (should (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +(ert-deftest lo-task-with-last-reviewed-is-clean () + "A task carrying the property is not flagged." + (let* ((out (lo-test--run (concat "* Open Work\n** TODO [#B] A task :feature:\n" + ":PROPERTIES:\n:LAST_REVIEWED: 2026-07-23\n:END:\n" + "Body.\n"))) + (js (lo-test--judgments (plist-get out :issues)))) + (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +(ert-deftest lo-task-last-reviewed-accepts-org-timestamp () + "The org-native [YYYY-MM-DD Day] form counts, matching the staleness script." + (let* ((out (lo-test--run (concat "* Open Work\n** TODO [#B] A task :feature:\n" + ":PROPERTIES:\n:LAST_REVIEWED: [2026-07-23 Thu]\n:END:\n"))) + (js (lo-test--judgments (plist-get out :issues)))) + (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +(ert-deftest lo-done-task-without-last-reviewed-is-clean () + "Completed tasks leave the review pool, so they are never flagged." + (let* ((out (lo-test--run (concat "* Open Work\n** DONE [#B] A task :feature:\n" + "CLOSED: [2026-07-23 Thu]\n"))) + (js (lo-test--judgments (plist-get out :issues)))) + (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +(ert-deftest lo-subtask-without-last-reviewed-is-clean () + "Only level-2 tasks are in the review pool; deeper headings are not." + (let* ((out (lo-test--run (concat "* Open Work\n** TODO [#B] Parent :feature:\n" + ":PROPERTIES:\n:LAST_REVIEWED: 2026-07-23\n:END:\n" + "*** TODO A sub-task\n"))) + (js (lo-test--judgments (plist-get out :issues)))) + (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +(ert-deftest lo-cookieless-task-without-last-reviewed-is-clean () + "The staleness script selects on a priority cookie, so match that scope." + (let* ((out (lo-test--run "* Open Work\n** TODO Manual testing and validation\n")) + (js (lo-test--judgments (plist-get out :issues)))) + (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +(ert-deftest lo-verify-task-without-last-reviewed-is-judgment () + "VERIFY is in the review pool too." + (let* ((out (lo-test--run "* Open Work\n** VERIFY [#B] Waiting on Craig\n")) + (js (lo-test--judgments (plist-get out :issues)))) + (should (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +;;; --------------------------------------------------------------------------- +;;; todo-format checkers skip docs/specs/ files (claude-rules/todo-format.md) +;; +;; The four todo-format-family checkers encode todo.org completion conventions. +;; A spec legitimately uses ** DONE <decision> with no CLOSED cookie and +;; ** <dated> — <who> review-history headings, so those checkers misfire on +;; every spec. They must skip any file under a docs/specs/ path segment. + +(defun lo-test--run-at (relpath content) + "Write CONTENT to <tmpdir>/RELPATH, run lint on it, return :issues. +RELPATH is a relative path (may contain slashes) so a docs/specs/ segment +can be exercised — the checkers key on the file's path, not just its name." + (let* ((root (make-temp-file "lo-test-root-" t)) + (file (expand-file-name relpath root))) + (make-directory (file-name-directory file) t) + (unwind-protect + (progn + (with-temp-file file (insert content)) + (lo-test--reset) + (lo-process-file file) + (prog1 (list :issues lo-issues) + (lo-test--drop-buffer file))) + (delete-directory root t)))) + +(defconst lo-test--spec-decisions + "* Decisions [1/1]\n** DONE Some decision\n- Context: x\n" + "A spec Decisions section: a level-2 DONE with no CLOSED cookie.") + +(defconst lo-test--spec-history + "* Review history\n** 2026-07-14 Tue @ 02:03:28 -0500 — Claude — responder\n- What: x\n" + "A spec review-history section: a level-2 dated header.") + +(ert-deftest lo-todo-checkers-fire-on-a-normal-org-file () + "Baseline: the checkers DO fire on a non-spec path (the bug is scope, not silence)." + (let* ((out (lo-test--run-at "todo.org" lo-test--spec-decisions)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should (memq 'level2-done-without-closed cs)))) + +(ert-deftest lo-level2-done-without-closed-skips-specs () + (let* ((out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" lo-test--spec-decisions)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should-not (memq 'level2-done-without-closed cs)))) + +(ert-deftest lo-level2-dated-header-skips-specs () + (let* ((out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" lo-test--spec-history)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should-not (memq 'level-2-dated-header cs)))) + +(ert-deftest lo-dated-log-active-timestamp-skips-specs () + (let* ((c "* History\n** 2026-07-14 Tue @ 02:03:28 -0500 — did a thing\nSCHEDULED: <2026-07-20 Mon>\n") + (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should-not (memq 'dated-log-heading-active-timestamp cs)))) + +(ert-deftest lo-subtask-done-not-dated-skips-specs () + (let* ((c "* Work\n** TODO Parent\n*** DONE A sub-decision\n") + (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should-not (memq 'subtask-done-not-dated cs)))) + +(ert-deftest lo-link-checks-still-fire-on-specs () + "Only the todo-format family is scoped out; a broken link in a spec still flags." + (let* ((c "* X\n[[file:does-not-exist-xyz.org][link]]\n") + (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should (memq 'link-to-local-file cs)))) + +(ert-deftest lo-task-missing-last-reviewed-skips-specs () + "The fifth todo-format checker (added 2026-07-23) skips specs too — a spec's +phases section may carry ** TODO [#x] items that aren't backlog tasks." + (let* ((c "* Implementation phases\n** TODO [#B] Phase one\nBody.\n") + (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should-not (memq 'task-missing-last-reviewed cs))) + ;; And still fires on a normal file. + (let* ((c "* Work\n** TODO [#B] Real backlog task\nBody.\n") + (out (lo-test--run-at "todo.org" c)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should (memq 'task-missing-last-reviewed cs)))) diff --git a/.ai/scripts/tests/test-todo-cleanup.el b/.ai/scripts/tests/test-todo-cleanup.el index ffbf2fb..1e964b3 100644 --- a/.ai/scripts/tests/test-todo-cleanup.el +++ b/.ai/scripts/tests/test-todo-cleanup.el @@ -31,6 +31,7 @@ (defun tc-test--reset (&optional check) (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-archived-to-file 0 tc-issues nil + tc-sealed 0 tc-seal nil tc-convert-subtasks nil tc-check-only (and check t) tc-archive-done t tc-sync-child-priority nil tc-current-file nil @@ -40,6 +41,7 @@ (defun tc-test--reset-sync (&optional check) (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-archived-to-file 0 tc-issues nil + tc-sealed 0 tc-seal nil tc-check-only (and check t) tc-archive-done nil tc-sync-child-priority t tc-current-file nil @@ -514,6 +516,12 @@ gitignore todo.org, then run `--archive-done' aging with the DEFAULT archive pat .gitignore contents or nil), :archive-ignored (whether git ignores the archive), :archive-exists." (let* ((root (make-temp-file "tc-git-" t)) + ;; Private backup dir: this helper writes a file literally named + ;; todo.org and runs a real (non-check) pass, so without this its + ;; backup lands in the shared temp dir under the exact production + ;; name and is indistinguishable from a real one. + (temporary-file-directory + (file-name-as-directory (make-temp-file "tc-git-bk-" t))) (todo (expand-file-name "todo.org" root)) (archive (expand-file-name "archive/task-archive.org" root)) (gi (expand-file-name ".gitignore" root))) @@ -534,7 +542,8 @@ gitignore todo.org, then run `--archive-done' aging with the DEFAULT archive pat :archive-ignored (eq 0 (call-process "git" nil nil nil "check-ignore" "-q" archive)) :archive-exists (file-readable-p archive))) - (delete-directory root t)))) + (delete-directory root t) + (delete-directory temporary-file-directory t)))) (ert-deftest tc-age-self-protect-gitignores-archive-when-todo-ignored () "When the todo file is gitignored, the aged-out archive is added to .gitignore @@ -578,6 +587,95 @@ entry is added for it." (should (> (plist-get out :archived) 0))))) ;;; --------------------------------------------------------------------------- +;;; --archive-done retention default + +(ert-deftest tc-archive-retain-default-is-one-month () + "The shipped retention default is one month (31 days), not the legacy 7. +The defvar initializes from this defconst; the live var itself is mutated by +other tests, so the immutable defconst is the stable contract to pin." + (should (= 31 tc-archive-retain-days-default))) + +;;; --------------------------------------------------------------------------- +;;; --seal: rename the working archive to resolved-YYYY-MM-DD.org + +(defun tc-test--seal (&optional opts) + "Run `--seal' against a temp todo file with a temp archive dir. +OPTS is a plist: :archive-content (seed task-archive.org with this; nil = no +working archive), :ref (YEAR MONTH DAY seal date; default (2026 7 18)), +:check, :presealed (also create resolved-<ref>.org first, to test collision). +Returns a plist: :sealed count, :issues, :working-exists, :sealed-exists, +:sealed-name, :report." + (let* ((ref (or (plist-get opts :ref) '(2026 7 18))) + (check (plist-get opts :check)) + (archive-content (plist-get opts :archive-content)) + (todo (make-temp-file "tc-seal-todo-" nil ".org")) + (adir (make-temp-file "tc-seal-arch-" t)) + (afile (expand-file-name "task-archive.org" adir)) + (sealed-name (format "resolved-%04d-%02d-%02d.org" + (nth 0 ref) (nth 1 ref) (nth 2 ref))) + (sealed (expand-file-name sealed-name adir))) + (unwind-protect + (progn + (with-temp-file todo (insert "* Open Work\n** TODO [#A] live\n")) + (when archive-content (with-temp-file afile (insert archive-content))) + (when (plist-get opts :presealed) + (with-temp-file sealed (insert "pre-existing seal\n"))) + (tc-test--reset check) + ;; Set every mode flag explicitly: tc-test--reset leaves + ;; tc-convert-subtasks untouched, so a convert test running earlier in + ;; the suite would otherwise still own the dispatch and run convert. + (setq tc-archive-done nil tc-sync-child-priority nil + tc-convert-subtasks nil tc-seal t tc-sealed 0 + tc-archive-reference-date ref + tc-archive-file afile) + (let ((report (with-output-to-string (tc-process-file todo) (tc-emit-report)))) + (tc-test--drop-buffer todo) + (list :sealed tc-sealed + :issues tc-issues + :working-exists (file-readable-p afile) + :sealed-exists (file-readable-p sealed) + :sealed-name sealed-name + :report report))) + (tc-test--drop-buffer todo) + (delete-file todo) + (delete-directory adir t)))) + +(ert-deftest tc-seal-renames-working-archive-to-dated-file () + "Normal: --seal renames task-archive.org to resolved-<seal-date>.org." + (let ((out (tc-test--seal '(:archive-content "* Resolved (archived)\n** DONE old\n" + :ref (2026 7 18))))) + (should (= 1 (plist-get out :sealed))) + (should-not (plist-get out :working-exists)) + (should (plist-get out :sealed-exists)) + (should (equal "resolved-2026-07-18.org" (plist-get out :sealed-name))) + (should (tc-test--has (plist-get out :report) "sealed task-archive.org → resolved-2026-07-18.org")))) + +(ert-deftest tc-seal-nothing-to-seal-is-a-reported-noop () + "Boundary: no working archive present — reported no-op, nothing created." + (let ((out (tc-test--seal '(:ref (2026 7 18))))) + (should (= 0 (plist-get out :sealed))) + (should-not (plist-get out :sealed-exists)) + (should (tc-test--has (plist-get out :report) "no working archive to seal")))) + +(ert-deftest tc-seal-check-mode-previews-without-renaming () + "Boundary: --check reports the seal but leaves the working archive in place." + (let ((out (tc-test--seal '(:archive-content "* Resolved (archived)\n" + :ref (2026 7 18) :check t)))) + (should (= 1 (plist-get out :sealed))) + (should (plist-get out :working-exists)) + (should-not (plist-get out :sealed-exists)) + (should (tc-test--has (plist-get out :report) "would seal")))) + +(ert-deftest tc-seal-refuses-to-clobber-existing-sealed-file () + "Error: resolved-<today>.org already exists — refuse, leave both files intact." + (let ((out (tc-test--seal '(:archive-content "* Resolved (archived)\n" + :ref (2026 7 18) :presealed t)))) + (should (= 0 (plist-get out :sealed))) + (should (plist-get out :working-exists)) + (should (plist-get out :sealed-exists)) + (should (tc-test--has (plist-get out :report) "already exists")))) + +;;; --------------------------------------------------------------------------- ;;; Sync-child-priority harness + fixtures (defun tc-test--sync (content &optional runs check) @@ -773,7 +871,7 @@ in ISSUES, in document order." (defun tc-test--reset-convert (&optional check) (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-converted 0 tc-archived-to-file 0 - tc-issues nil + tc-issues nil tc-sealed 0 tc-seal nil tc-check-only (and check t) tc-archive-done nil tc-sync-child-priority nil tc-convert-subtasks t tc-current-file nil @@ -927,8 +1025,9 @@ CLOSED: [2026-06-27 Sat 12:50] DEADLINE: <2026-06-30 Tue> Body line. ") -(ert-deftest tc-convert-preserves-deadline-on-shared-planning-line-boundary () - "Boundary: removing the CLOSED cookie keeps a DEADLINE sharing its planning line." +(ert-deftest tc-convert-strips-deadline-sharing-the-planning-line-boundary () + "Boundary: a DEADLINE sharing the CLOSED planning line goes too — a dated-log +entry carries no active planning timestamp (todo-format.md). Body survives." (let* ((out (tc-test--convert tc-test--convert-closed-with-deadline)) (res (plist-get out :result))) (should (= 1 (plist-get out :converted))) @@ -936,8 +1035,142 @@ Body line. "^\\*\\*\\* 2026-06-27 Sat @ 12:50:00 [-+][0-9]\\{4\\} Ship the panel$" res)) (should-not (string-match-p "CLOSED:" res)) - (should (string-match-p "^DEADLINE: <2026-06-30 Tue>$" res)) + (should-not (string-match-p "DEADLINE:" res)) + (should (string-match-p "^Body line\\.$" res)))) + +(defconst tc-test--convert-closed-and-scheduled-separate-lines + "* Project Open Work +** TODO [#B] Parent task +*** DONE [#C] Book the venue :feature: +CLOSED: [2026-06-27 Sat 12:50] +SCHEDULED: <2026-06-20 Sat> +Body line. +") + +(ert-deftest tc-convert-strips-scheduled-on-its-own-line () + "Normal (the home bug): a SCHEDULED planning line on its own — the completion +rewrite dropped keyword/priority/tags but left the SCHEDULED, pinning the dated +entry to the agenda as weeks-overdue. Both planning lines go; body survives." + (let* ((out (tc-test--convert tc-test--convert-closed-and-scheduled-separate-lines)) + (res (plist-get out :result))) + (should (= 1 (plist-get out :converted))) + (should (string-match-p + "^\\*\\*\\* 2026-06-27 Sat @ 12:50:00 [-+][0-9]\\{4\\} Book the venue$" + res)) + (should-not (string-match-p "CLOSED:" res)) + (should-not (string-match-p "SCHEDULED:" res)) (should (string-match-p "^Body line\\.$" res)))) +(defconst tc-test--convert-scheduled-in-body-prose + "* Project Open Work +** TODO [#B] Parent task +*** DONE [#C] Note the mechanism :feature: +CLOSED: [2026-06-27 Sat 12:50] +An active SCHEDULED: <2026-06-20 Sat> in prose must survive. +") + +(ert-deftest tc-convert-leaves-planning-shaped-body-prose-alone () + "Boundary: a planning-shaped token inside body prose (not a canonical planning +line) is left untouched — the strip stops at the first non-planning line." + (let* ((out (tc-test--convert tc-test--convert-scheduled-in-body-prose)) + (res (plist-get out :result))) + (should (= 1 (plist-get out :converted))) + (should-not (string-match-p "CLOSED:" res)) + (should (string-match-p "An active SCHEDULED: <2026-06-20 Sat> in prose must survive\\." res)))) + (provide 'test-todo-cleanup) ;;; test-todo-cleanup.el ends here + +;;; --------------------------------------------------------------------------- +;;; Backup before mutating (parity with lint-org.el / wrap-org-table.el) +;; +;; todo-cleanup rewrites todo.org in place and left no copy behind, while both +;; sibling org-mutators back up to /tmp first. It is also the one that runs most +;; often (every wrap, every sentry cycle). Emacs's own backup does not fire under +;; --batch -q, so there was genuinely no undo short of git. + +(ert-deftest tc-backup-written-before-a-real-mutation () + "A real (non-check) run leaves a copy holding the pre-edit content. + +`temporary-file-directory' is rebound to a private dir for the duration: the +backup name derives from the *file's* basename, and the real todo.org shares +that basename, so a live sentry run writing /tmp/todo.org.before-todo-cleanup.* +would otherwise be indistinguishable from this test's own artifact. The first +version of this test globbed the shared /tmp and passed only until a real run +created one (2026-07-24)." + (let* ((dir (make-temp-file "tc-backup-" t)) + (bdir (file-name-as-directory (make-temp-file "tc-bk-" t))) + (file (expand-file-name "todo.org" dir)) + (before "* P Open Work\n** TODO [#B] parent\n*** DONE a subtask\nCLOSED: [2026-07-01 Tue]\n")) + (unwind-protect + (progn + (with-temp-file file (insert before)) + (let ((tc-check-only nil) + (tc-convert-subtasks t) + (temporary-file-directory bdir)) + (tc-process-file file)) + (let ((backups (file-expand-wildcards + (concat bdir "todo.org.before-todo-cleanup.*")))) + (should backups) + (should (string-match-p + "a subtask" + (with-temp-buffer (insert-file-contents (car backups)) + (buffer-string)))))) + (delete-directory dir t) + (delete-directory bdir t)))) + +(ert-deftest tc-no-backup-in-check-mode () + "--check writes nothing, so it must not leave a backup either. +Uses a private `temporary-file-directory' for the same isolation reason." + (let* ((dir (make-temp-file "tc-backup-" t)) + (bdir (file-name-as-directory (make-temp-file "tc-bk-" t))) + (file (expand-file-name "todo.org" dir))) + (unwind-protect + (progn + (with-temp-file file + (insert "* P Open Work\n** TODO [#B] parent\n*** DONE sub\nCLOSED: [2026-07-01 Tue]\n")) + (let ((tc-check-only t) + (tc-convert-subtasks t) + (temporary-file-directory bdir)) + (tc-process-file file)) + (should-not (file-expand-wildcards + (concat bdir "todo.org.before-todo-cleanup.*")))) + (delete-directory dir t) + (delete-directory bdir t)))) + +(ert-deftest tc-backup-never-overwrites-an-earlier-one () + "Two invocations in the same second must not collapse to one backup. + +open-tasks.org runs --convert-subtasks then --archive-done back to back, each +a sub-second batch run. With a second-resolution stamp and copy-file's +OK-IF-ALREADY-EXISTS, the second invocation overwrote the first's backup with +already-mutated content, so the true pre-session original was unrecoverable — +the exact state the backup exists to preserve (found 2026-07-24 in review)." + (let* ((dir (make-temp-file "tc-collide-" t)) + (bdir (file-name-as-directory (make-temp-file "tc-cbk-" t))) + (file (expand-file-name "todo.org" dir)) + (original (concat "* P Open Work\n** TODO [#B] parent\n*** DONE sub\n" + "CLOSED: [2026-07-01 Tue]\n" + "* P Resolved\n** DONE [#C] old\nCLOSED: [2025-01-01 Wed]\n"))) + (unwind-protect + (progn + (with-temp-file file (insert original)) + ;; Two back-to-back invocations, as the shipped workflow does. + (let ((temporary-file-directory bdir)) + (let ((tc-check-only nil) (tc-convert-subtasks t)) + (tc-process-file file)) + (let ((tc-check-only nil) (tc-convert-subtasks nil) (tc-archive-done t) + (tc-archive-retain-days nil)) + (tc-process-file file))) + (let ((backups (file-expand-wildcards + (concat bdir "todo.org.before-todo-cleanup.*")))) + ;; Both invocations kept their own backup. + (should (= (length backups) 2)) + ;; And one of them still holds the true original. + (should (cl-some (lambda (b) + (string= original + (with-temp-buffer (insert-file-contents b) + (buffer-string)))) + backups)))) + (delete-directory dir t) + (delete-directory bdir t)))) diff --git a/.ai/scripts/tests/test_apkg_to_orgdrill.py b/.ai/scripts/tests/test_apkg_to_orgdrill.py new file mode 100644 index 0000000..6a95ea4 --- /dev/null +++ b/.ai/scripts/tests/test_apkg_to_orgdrill.py @@ -0,0 +1,301 @@ +"""Tests for apkg-to-orgdrill.py — the inverse of flashcard-to-anki.py. + +The converter reads an Anki .apkg (a zip holding collection.anki2 / .anki21 +sqlite) and emits an org-drill .org in the house canonical shape. It is +stdlib-only (zipfile + sqlite3), so it imports directly — no genanki stub. + +The apkg schema these tests build by hand mirrors what genanki actually +writes, confirmed against a real apkg generated from flashcard-to-anki.py: + - col.decks : JSON {did: {"name": ...}}, always including id-1 "Default" + - col.models : JSON {mid: {"name": ..., "flds": [{"name": "Front"}, ...]}} + - notes.flds : fields joined by \x1f; tags space-padded (" tag ") + - cards : nid -> did (the Default deck carries no cards) + +The round-trip test closes the loop through flashcard-to-anki.py's own +parse(): original org -> forward parse tuples -> apkg fixture -> converter +-> recovered org -> forward parse -> assert the (front, back, tag) tuples +match. Only the apkg materialization is hand-built (the genanki boundary); +everything else is the real code on both sides. +""" +from __future__ import annotations + +import importlib.util +import json +import sqlite3 +import sys +import types +import zipfile +from pathlib import Path + +import pytest + +SCRIPTS = Path(__file__).resolve().parents[1] +CONVERTER = SCRIPTS / "apkg-to-orgdrill.py" +FORWARD = SCRIPTS / "flashcard-to-anki.py" + + +def _load(path: Path, name: str, stub_genanki: bool = False): + if stub_genanki: + sys.modules.setdefault("genanki", types.ModuleType("genanki")) + spec = importlib.util.spec_from_file_location(name, path) + assert spec and spec.loader + module = importlib.util.module_from_spec(spec) + # Register before exec: @dataclass resolves cls.__module__ via sys.modules + # (Python 3.14), which is None for an unregistered importlib module. + sys.modules[name] = module + spec.loader.exec_module(module) + return module + + +@pytest.fixture(scope="module") +def conv(): + return _load(CONVERTER, "apkg_to_orgdrill") + + +@pytest.fixture(scope="module") +def forward(): + return _load(FORWARD, "flashcard_to_anki", stub_genanki=True) + + +# --- fixture builder: write a genanki-shaped apkg by hand ------------------ + +def _make_apkg( + path: Path, + decks: dict[int, str], + models: dict[int, list[str]], + notes: list[tuple[int, int, list[str], str]], # (nid, mid, fields, tag) + cards: list[tuple[int, int]], # (nid, did) + *, + media: str = "{}", +) -> None: + """Materialize a minimal apkg matching genanki's collection.anki2 shape.""" + col_dir = path.parent / f"{path.stem}-build" + col_dir.mkdir(parents=True, exist_ok=True) + db = col_dir / "collection.anki2" + if db.exists(): + db.unlink() + con = sqlite3.connect(db) + con.execute("CREATE TABLE col (id INTEGER, decks TEXT, models TEXT)") + decks_json = {"1": {"name": "Default"}} + decks_json.update({str(did): {"name": name} for did, name in decks.items()}) + models_json = { + str(mid): {"name": f"{decks.get(list(decks)[0], 'M')} model", + "flds": [{"name": n, "ord": i} for i, n in enumerate(flds)]} + for mid, flds in models.items() + } + con.execute("INSERT INTO col (id, decks, models) VALUES (1, ?, ?)", + (json.dumps(decks_json), json.dumps(models_json))) + con.execute("CREATE TABLE notes (id INTEGER, mid INTEGER, flds TEXT, tags TEXT)") + for nid, mid, fields, tag in notes: + con.execute("INSERT INTO notes (id, mid, flds, tags) VALUES (?, ?, ?, ?)", + (nid, mid, "\x1f".join(fields), f" {tag} " if tag else " ")) + con.execute("CREATE TABLE cards (id INTEGER, nid INTEGER, did INTEGER)") + for i, (nid, did) in enumerate(cards): + con.execute("INSERT INTO cards (id, nid, did) VALUES (?, ?, ?)", (1000 + i, nid, did)) + con.commit() + con.close() + with zipfile.ZipFile(path, "w") as z: + z.write(db, "collection.anki2") + z.writestr("media", media) + + +# --- html_to_org_body ------------------------------------------------------ + +def test_html_to_org_splits_br_into_lines(conv): + assert conv.html_to_org_body("one<br>two<br>three") == ["one", "two", "three"] + + +def test_html_to_org_handles_br_variants(conv): + assert conv.html_to_org_body("a<br/>b<br />c<BR>d") == ["a", "b", "c", "d"] + + +def test_html_to_org_unescapes_entities_amp_last(conv): + # Inverts escape_html (which escapes & first): < > & -> < > &. + assert conv.html_to_org_body("x <tag> & y") == ["x <tag> & y"] + + +def test_html_to_org_preserves_a_literal_escaped_entity(conv): + # Forward-escaping the literal "<" yields "&lt;"; the inverse must + # recover "<", not "<". + assert conv.html_to_org_body("&lt;") == ["<"] + + +def test_html_to_org_strips_answer_hr(conv): + assert conv.html_to_org_body('front<hr id="answer">back') == ["front", "back"] + + +def test_html_to_org_empty_back_is_empty(conv): + assert conv.html_to_org_body("") == [] + + +# --- read_apkg ------------------------------------------------------------- + +def test_read_apkg_single_deck_recovers_front_back_tag_deck(conv, tmp_path): + apkg = tmp_path / "d.apkg" + _make_apkg( + apkg, + decks={20: "My Deck"}, + models={9: ["Front", "Back"]}, + notes=[(100, 9, ["Q1?", "A1.<br>line2"], "sec-one")], + cards=[(100, 20)], + ) + recovered = conv.read_apkg(apkg) + assert len(recovered) == 1 + note = recovered[0] + assert note.deck == "My Deck" + assert note.front == "Q1?" + assert note.back_html == "A1.<br>line2" + assert note.tag == "sec-one" + + +def test_read_apkg_multiple_decks_grouped(conv, tmp_path): + apkg = tmp_path / "multi.apkg" + _make_apkg( + apkg, + decks={20: "Deck A", 21: "Deck B"}, + models={9: ["Front", "Back"]}, + notes=[(100, 9, ["QA?", "AA"], "ta"), (101, 9, ["QB?", "AB"], "tb")], + cards=[(100, 20), (101, 21)], + ) + decks = {n.deck for n in conv.read_apkg(apkg)} + assert decks == {"Deck A", "Deck B"} + + +def test_read_apkg_skips_default_deck_without_cards(conv, tmp_path): + apkg = tmp_path / "def.apkg" + _make_apkg( + apkg, + decks={20: "Real Deck"}, + models={9: ["Front", "Back"]}, + notes=[(100, 9, ["Q?", "A"], "t")], + cards=[(100, 20)], + ) + assert {n.deck for n in conv.read_apkg(apkg)} == {"Real Deck"} + + +def test_read_apkg_warns_and_skips_non_basic_model(conv, tmp_path, capsys): + apkg = tmp_path / "cloze.apkg" + _make_apkg( + apkg, + decks={20: "Cloze Deck"}, + models={9: ["Text", "Extra"]}, # not Front/Back + notes=[(100, 9, ["some {{c1::text}}", "extra"], "t")], + cards=[(100, 20)], + ) + recovered = conv.read_apkg(apkg) + assert recovered == [] + assert "skip" in capsys.readouterr().err.lower() + + +def test_read_apkg_reads_anki21_collection_name(conv, tmp_path): + # A .anki21 collection filename must be read the same as .anki2. + apkg = tmp_path / "new.apkg" + _make_apkg( + apkg, + decks={20: "Deck"}, + models={9: ["Front", "Back"]}, + notes=[(100, 9, ["Q?", "A"], "t")], + cards=[(100, 20)], + ) + # Rewrite the zip renaming the collection member to .anki21. + with zipfile.ZipFile(apkg) as z: + data = z.read("collection.anki2") + media = z.read("media") + with zipfile.ZipFile(apkg, "w") as z: + z.writestr("collection.anki21", data) + z.writestr("media", media) + assert conv.read_apkg(apkg)[0].front == "Q?" + + +def test_read_apkg_flags_media_reference(conv, tmp_path, capsys): + apkg = tmp_path / "media.apkg" + _make_apkg( + apkg, + decks={20: "Deck"}, + models={9: ["Front", "Back"]}, + notes=[(100, 9, ["Q?", 'see <img src="x.png">'], "t")], + cards=[(100, 20)], + ) + conv.read_apkg(apkg) + assert "media" in capsys.readouterr().err.lower() + + +# --- notes_to_org ---------------------------------------------------------- + +def test_notes_to_org_emits_canonical_shape(conv): + Note = conv.Note + notes = [ + Note(deck="My Deck", front="Q1?", back_html="A1.", tag="alpha"), + Note(deck="My Deck", front="Q2?", back_html="A2.", tag="alpha"), + ] + ids = iter(["id-1", "id-2"]) + org = conv.notes_to_org(notes, "My Deck", new_id=lambda: next(ids)) + assert "#+TITLE: My Deck" in org + assert "* alpha" in org + assert "** Q1? :drill:" in org + assert ":ID: id-1" in org + assert ":ID: id-2" in org + assert org.count("* alpha") == 1 # both cards share one section + + +def test_notes_to_org_distinct_tags_get_distinct_sections(conv): + Note = conv.Note + notes = [ + Note(deck="D", front="Qa?", back_html="a", tag="alpha"), + Note(deck="D", front="Qb?", back_html="b", tag="beta"), + ] + org = conv.notes_to_org(notes, "D", new_id=lambda: "x") + assert "* alpha" in org and "* beta" in org + + +# --- round-trip through the real forward parse() --------------------------- + +def test_round_trip_matches_forward_parse_tuples(conv, forward, tmp_path): + original = ( + "#+TITLE: RT Deck\n" + "\n" + "* First Section\n" + "** What is 2+2? :drill:\n" + ":PROPERTIES:\n:ID: aaaa\n:END:\n" + "Four.\n" + "Second line with <angle> & amp.\n" + "\n" + "* Second Section\n" + "** Capital of France? :drill:\n" + "Paris.\n" + ) + tuples = forward.parse(original) # [(front, back_html, anki_tags), ...] + assert len(tuples) == 2 + + apkg = tmp_path / "rt.apkg" + _make_apkg( + apkg, + decks={20: "RT Deck"}, + models={9: ["Front", "Back"]}, + # anki_tags is a list; the apkg tags field is space-joined. + notes=[(100 + i, 9, [f, b], " ".join(tags)) + for i, (f, b, tags) in enumerate(tuples)], + cards=[(100 + i, 20) for i in range(len(tuples))], + ) + + by_deck = conv.convert(apkg) + assert set(by_deck) == {"RT Deck"} + recovered_tuples = forward.parse(by_deck["RT Deck"]) + assert recovered_tuples == tuples + + +# --- errors ---------------------------------------------------------------- + +def test_read_apkg_missing_collection_errors(conv, tmp_path): + bad = tmp_path / "bad.apkg" + with zipfile.ZipFile(bad, "w") as z: + z.writestr("media", "{}") + with pytest.raises(Exception): + conv.read_apkg(bad) + + +def test_read_apkg_not_a_zip_errors(conv, tmp_path): + notzip = tmp_path / "plain.apkg" + notzip.write_text("not a zip") + with pytest.raises(Exception): + conv.read_apkg(notzip) diff --git a/.ai/scripts/tests/test_cj_remove_block.py b/.ai/scripts/tests/test_cj_remove_block.py index 2c8dade..3cdee46 100644 --- a/.ai/scripts/tests/test_cj_remove_block.py +++ b/.ai/scripts/tests/test_cj_remove_block.py @@ -14,6 +14,34 @@ import pytest SCRIPT = Path(__file__).parent.parent / "cj-remove-block.py" +@pytest.fixture(autouse=True) +def isolated_tmpdir(tmp_path, monkeypatch): + """Give every test in this module a private TMPDIR. + + The script backs up to the system temp dir under a name derived from the + edited file's BASENAME. The real todo.org shares that basename, so any test + operating on a fixture named todo.org writes something indistinguishable + from a production backup — and an earlier version of this file globbed the + shared /tmp and unlinked every match, so a routine `make test` destroyed + Craig's real backups (found in review, 2026-07-24). + + Isolating at module scope rather than per-test is deliberate: the same bug + was fixed once in the elisp sibling and left here, so relying on each new + test to remember is exactly how it recurred. Autouse makes it structural. + """ + d = tmp_path / "_tmpdir" + d.mkdir() + # TMPDIR covers subprocess invocations of the script. + monkeypatch.setenv("TMPDIR", str(d)) + # tempfile.gettempdir() caches its answer on first call, so a test that + # loads the module in-process would keep writing to the real /tmp no matter + # what TMPDIR says. Override the cache too — this is the gap that made the + # env-var-only version still leak one backup per suite run. + import tempfile as _tempfile + monkeypatch.setattr(_tempfile, "tempdir", str(d)) + return d + + @pytest.fixture def run_remove(tmp_path): """Write content to a temp org file, run cj-remove-block, return new contents.""" @@ -155,3 +183,142 @@ class TestCjRemoveBlockSafety: err, post_content = run_remove_expecting_failure(original, start=4, end=2) assert err.returncode != 0 assert post_content == original + + +class TestMultiBlockRangeRefused: + """The validation exists to catch a drifted range, but it only checked the + first and last lines of that range. A span from one block's opening fence to + a LATER block's closing fence passed, and the removal silently deleted every + line between — real prose, headings, whole tasks — with a zero exit. Drift is + the skill's normal operating mode (respond-to-cj-comments edits the file as it + processes, and a file under cj review usually holds several blocks), so this + is the exact scenario the check was written for. Reproduced 2026-07-24.""" + + TWO_BLOCKS = ( + "* Alpha\n" + "#+begin_src cj:\n" + "note A\n" + "#+end_src\n" + "KEEP THIS LINE\n" + "* Beta\n" + "#+begin_src cj:\n" + "note B\n" + "#+end_src\n" + ) + + def test_range_spanning_two_blocks_is_refused(self, run_remove_expecting_failure): + # Lines 2..9: block one's opener through block two's closer. + err, content = run_remove_expecting_failure(self.TWO_BLOCKS, 2, 9) + assert err.returncode == 1 + assert "KEEP THIS LINE" in content, "content between the blocks was destroyed" + assert "* Beta" in content, "a heading between the blocks was destroyed" + + def test_refusal_names_the_reason(self, run_remove_expecting_failure): + err, _ = run_remove_expecting_failure(self.TWO_BLOCKS, 2, 9) + assert "more than one" in err.stderr.decode().lower() + + def test_a_correct_single_block_range_still_removes(self, run_remove): + # The fix must not over-tighten: the legitimate range still works. + out = run_remove(self.TWO_BLOCKS, 2, 4) + assert "note A" not in out + assert "KEEP THIS LINE" in out + assert "note B" in out, "the second block must be untouched" + + def test_a_nested_end_src_inside_the_range_is_refused(self, run_remove_expecting_failure): + # Any #+end_src before the final line means the range covers >1 block. + content = ( + "#+begin_src cj:\n" + "a\n" + "#+end_src\n" + "middle\n" + "#+begin_src cj:\n" + "b\n" + "#+end_src\n" + ) + err, after = run_remove_expecting_failure(content, 1, 7) + assert err.returncode == 1 + assert "middle" in after + + +class TestSafeMutation: + """The script rewrites Craig's org files (todo.org, notes.org). It wrote with + a bare write_text, which truncates the target on open, and took no backup — + so a mid-write failure left the file truncated with no copy to recover from. + lint-org.el, the other tool that mutates these files, backs up to a temp dir + first. Match that, and make the write atomic. + + Every test here redirects TMPDIR to a private directory. The backup name + derives from the file's basename, and the real todo.org shares it, so a test + globbing the shared temp dir cannot tell its own artifact from a genuine + backup — and an earlier version of this class globbed /tmp and unlinked every + match, so a routine `make test` destroyed real backups (found in review, + 2026-07-24). Never glob or delete across the shared temp dir.""" + + ONE_BLOCK = "* T\n#+begin_src cj:\nnote\n#+end_src\nkeep\n" + + def test_a_backup_is_written_before_mutating(self, tmp_path): + import subprocess, glob, os + bdir = tmp_path / "bk" + bdir.mkdir() + f = tmp_path / "todo.org" + f.write_text(self.ONE_BLOCK) + subprocess.run( + ["python3", str(SCRIPT), "--file", str(f), "--start", "2", "--end", "4"], + check=True, capture_output=True, + env={**os.environ, "TMPDIR": str(bdir)}, + ) + backups = glob.glob(str(bdir / "todo.org.before-cj-remove.*")) + assert backups, "no backup was written before mutating the org file" + assert "note" in Path(max(backups)).read_text() + + def test_no_partial_file_when_the_write_fails(self, tmp_path, monkeypatch): + import importlib.util + spec = importlib.util.spec_from_file_location("crb", SCRIPT) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + bdir = tmp_path / "bk" + bdir.mkdir() + monkeypatch.setenv("TMPDIR", str(bdir)) + f = tmp_path / "todo.org" + f.write_text(self.ONE_BLOCK) + def boom(*a, **k): + raise OSError("disk full") + monkeypatch.setattr(mod.os, "replace", boom) + with pytest.raises(OSError): + mod.remove_range(f, 2, 4) + # The original survives intact — no truncation, no partial. + assert f.read_text() == self.ONE_BLOCK + + +class TestBackupNeverOverwrites: + """Same defect class as todo-cleanup's, and more reachable here: the + respond-to-cj-comments skill removes several annotations in quick + succession, so a second-resolution stamp collides and the later backup + overwrote the earlier one with already-mutated content.""" + + TWO_BLOCKS = ( + "* A\n#+begin_src cj:\nfirst\n#+end_src\n" + "* B\n#+begin_src cj:\nsecond\n#+end_src\n" + ) + + def test_consecutive_removals_each_keep_a_backup(self, tmp_path, monkeypatch): + import subprocess, glob + bdir = tmp_path / "bk" + bdir.mkdir() + monkeypatch.setenv("TMPDIR", str(bdir)) + f = tmp_path / "todo.org" + f.write_text(self.TWO_BLOCKS) + original = f.read_text() + # Remove the second block, then the first — back to back, same second. + subprocess.run(["python3", str(SCRIPT), "--file", str(f), + "--start", "6", "--end", "8"], + check=True, capture_output=True, + env={**__import__("os").environ, "TMPDIR": str(bdir)}) + subprocess.run(["python3", str(SCRIPT), "--file", str(f), + "--start", "2", "--end", "4"], + check=True, capture_output=True, + env={**__import__("os").environ, "TMPDIR": str(bdir)}) + backups = glob.glob(str(bdir / "todo.org.before-cj-remove.*")) + assert len(backups) == 2, f"expected 2 backups, got {len(backups)}" + contents = [Path(b).read_text() for b in backups] + assert original in contents, "no backup holds the true original" diff --git a/.ai/scripts/tests/test_flashcard_stats.py b/.ai/scripts/tests/test_flashcard_stats.py index 606f7c1..46deccc 100644 --- a/.ai/scripts/tests/test_flashcard_stats.py +++ b/.ai/scripts/tests/test_flashcard_stats.py @@ -217,6 +217,31 @@ def test_parse_cards_captures_body_without_drawer_planning_or_answer_header(stat assert c["body"] == "the real answer" +def test_parse_cards_counts_a_multitag_heading_as_a_card(stats): + """A card multi-tagged :fundamental:drill: still counts; the front is clean.""" + text = "* Sec\n** Q multi? :fundamental:drill:\nthe answer\n" + cards, _ = stats.parse_cards(text.splitlines()) + assert len(cards) == 1 + assert cards[0]["heading"] == "Q multi?" + assert cards[0]["body"] == "the answer" + + +def test_parse_cards_ignores_a_tagged_heading_without_drill(stats): + """A tagged heading missing :drill: is not a drill card.""" + text = "* Sec\n** Just a note :note:\nbody\n" + cards, _ = stats.parse_cards(text.splitlines()) + assert cards == [] + + +def test_parse_cards_body_stops_at_next_multitag_card(stats): + """The body scan ends at the next L2 card even when it is multi-tagged.""" + text = "** Q1? :a:drill:\nbody1\n** Q2? :drill:\nbody2\n" + cards, _ = stats.parse_cards(text.splitlines()) + assert len(cards) == 2 + assert cards[0]["body"] == "body1" + assert cards[1]["body"] == "body2" + + def test_find_duplicate_fronts_matches_normalized_headings(stats): cards = [ {"heading": "What is LEO?"}, diff --git a/.ai/scripts/tests/test_flashcard_to_anki.py b/.ai/scripts/tests/test_flashcard_to_anki.py index 87008a8..fa38b64 100644 --- a/.ai/scripts/tests/test_flashcard_to_anki.py +++ b/.ai/scripts/tests/test_flashcard_to_anki.py @@ -158,17 +158,18 @@ Geostationary Earth Orbit. def test_parse_returns_front_back_tag_per_card(drill): cards = drill.parse(SECTIONED) assert len(cards) == 2 - assert cards[0] == ("What is LEO?", "Low Earth Orbit.", "orbital-regimes") + # The section becomes the sole Anki tag (as a one-element list). + assert cards[0] == ("What is LEO?", "Low Earth Orbit.", ["orbital-regimes"]) assert cards[1][0] == "What is GEO?" def test_parse_card_without_a_section_gets_the_drill_tag(drill): - assert drill.parse("** Lone card? :drill:\nbody\n") == [("Lone card?", "body", "drill")] + assert drill.parse("** Lone card? :drill:\nbody\n") == [("Lone card?", "body", ["drill"])] def test_parse_strips_properties_drawer_from_back(drill): text = "** Q? :drill:\n:PROPERTIES:\n:ID: abc\n:END:\nThe answer.\n" - assert drill.parse(text) == [("Q?", "The answer.", "drill")] + assert drill.parse(text) == [("Q?", "The answer.", ["drill"])] def test_parse_trims_leading_and_trailing_blank_body_lines(drill): @@ -178,7 +179,59 @@ def test_parse_trims_leading_and_trailing_blank_body_lines(drill): def test_parse_card_with_only_a_drawer_has_empty_back(drill): text = "** Q? :drill:\n:PROPERTIES:\n:ID: x\n:END:\n" - assert drill.parse(text) == [("Q?", "", "drill")] + assert drill.parse(text) == [("Q?", "", ["drill"])] + + +# --- multi-tag headings, --tag-filter, --guid-salt ------------------------- + +MULTITAG = """* Fundamentals +** What is LEO? :fundamental:drill: +Low Earth Orbit. +** What is GEO? :drill: +Geostationary Earth Orbit. +""" + + +def test_parse_multitag_heading_is_a_card_when_drill_is_present(drill): + """A heading with a second org tag still parses when drill is among them.""" + cards = drill.parse(MULTITAG) + assert len(cards) == 2 + assert cards[0][0] == "What is LEO?" + + +def test_parse_multitag_tags_ride_along_next_to_the_section_tag(drill): + """Non-drill org tags become Anki tags alongside the section tag.""" + cards = drill.parse(MULTITAG) + assert cards[0][2] == ["fundamentals", "fundamental"] # section slug + org tag + assert cards[1][2] == ["fundamentals"] # drill-only -> section only + + +def test_parse_heading_without_drill_tag_is_not_a_card(drill): + """A tagged heading missing :drill: is not a card (e.g. :note:).""" + assert drill.parse("* S\n** Just a note :note:\nbody\n") == [] + + +def test_parse_tag_filter_returns_only_cards_with_that_org_tag(drill): + """--tag-filter narrows to cards carrying the given org tag.""" + cards = drill.parse(MULTITAG, tag_filter="fundamental") + assert len(cards) == 1 + assert cards[0][0] == "What is LEO?" + + +def test_parse_body_bounded_by_any_l1_or_l2_heading(drill): + """A card body stops at the next L1/L2 heading, multi-tagged or not.""" + text = "** Q1? :a:drill:\nbody1\n** Q2? :drill:\nbody2\n" + cards = drill.parse(text) + assert cards[0][1] == "body1" + assert cards[1][1] == "body2" + + +def test_card_guid_salt_changes_the_guid(drill, monkeypatch): + """--guid-salt gives a subset deck its own GUID space; no salt is unchanged.""" + monkeypatch.setattr(drill.genanki, "guid_for", lambda *a: ":".join(a), raising=False) + assert drill.card_guid("front", None) == "front" + assert drill.card_guid("front", "fundamentals") == "fundamentals:front" + assert drill.card_guid("front", None) != drill.card_guid("front", "fundamentals") def test_parse_joins_multiline_body_with_br(drill): diff --git a/.ai/scripts/tests/test_inbox_send.py b/.ai/scripts/tests/test_inbox_send.py index f75d7a1..9b0a8c6 100644 --- a/.ai/scripts/tests/test_inbox_send.py +++ b/.ai/scripts/tests/test_inbox_send.py @@ -476,3 +476,117 @@ class TestFilenameCollisions: assert len(files) == 2 bodies = "".join(f.read_text() for f in files) assert "message one" in bodies and "message two" in bodies + + +class TestAtomicWrite: + """A send wrote straight to the destination path in another project's + inbox/, and write_text truncates on open, so any mid-write failure left a + zero-byte .org there. inbox-status counts that phantom as a pending + handoff, blocking a turn in the receiving project over a file with no + content (2026-07-23). The write must be atomic: the inbox sees a complete + file or nothing.""" + + def test_send_text_writes_utf8(self, tmp_path): + from datetime import datetime + mod = _load_module() + inbox = tmp_path / "inbox" + inbox.mkdir() + now = datetime(2026, 7, 23, 4, 36, 0) + # An em dash and an accented char — both non-ASCII. + dest = mod.send_text(inbox, "accent café and dash — here", "src", None, now) + # Reading as utf-8 must round-trip; a locale-encoded write would raise + # under a C locale, and reading back proves the bytes are utf-8. + assert "—" in dest.read_text(encoding="utf-8") + + def test_send_text_no_partial_on_write_failure(self, tmp_path, monkeypatch): + from datetime import datetime + mod = _load_module() + inbox = tmp_path / "inbox" + inbox.mkdir() + now = datetime(2026, 7, 23, 4, 36, 0) + # Force the atomic finalize to fail after the temp file is written. + def boom(*a, **k): + raise OSError("disk full") + monkeypatch.setattr(mod.os, "replace", boom) + with pytest.raises(OSError): + mod.send_text(inbox, "a message that should never half-land", "src", None, now) + # No phantom, no leftover temp: the inbox is empty. + assert list(inbox.iterdir()) == [] + + def test_send_text_leaves_no_temp_on_success(self, tmp_path): + from datetime import datetime + mod = _load_module() + inbox = tmp_path / "inbox" + inbox.mkdir() + now = datetime(2026, 7, 23, 4, 36, 0) + dest = mod.send_text(inbox, "clean send", "src", None, now) + assert list(inbox.iterdir()) == [dest] + + def test_send_file_no_partial_on_write_failure(self, tmp_path, monkeypatch): + from datetime import datetime + mod = _load_module() + inbox = tmp_path / "inbox" + inbox.mkdir() + src = tmp_path / "note.org" + src.write_text("body") + now = datetime(2026, 7, 23, 4, 36, 0) + def boom(*a, **k): + raise OSError("disk full") + monkeypatch.setattr(mod.os, "replace", boom) + with pytest.raises(OSError): + mod.send_file(inbox, src, "src", None, now) + assert list(inbox.iterdir()) == [] + + def test_send_file_leaves_no_temp_on_success(self, tmp_path): + from datetime import datetime + mod = _load_module() + inbox = tmp_path / "inbox" + inbox.mkdir() + src = tmp_path / "note.org" + src.write_text("payload") + now = datetime(2026, 7, 23, 4, 36, 0) + dest = mod.send_file(inbox, src, "src", None, now) + assert list(inbox.iterdir()) == [dest] + assert dest.read_text() == "payload" + + +class TestSmallerDefects: + """Two low-severity defects found reading inbox-send during the 2026-07-23 + sweep: an unreadable source raised an uncaught traceback instead of the + clean error every other failure path produces, and a roots config naming + both a parent and one of its children listed the same project twice.""" + + def test_unreadable_source_gives_clean_error_not_traceback( + self, project_root, run_script, tmp_path + ): + project_root("sender") + project_root("receiver") + roots = [tmp_path / "projects"] + src = tmp_path / "secret.bin" + src.write_text("x") + src.chmod(0o000) + try: + result = run_script( + ["receiver", "--file", str(src)], + cwd=tmp_path / "projects" / "sender", + roots=roots, + expect_failure=True, + ) + finally: + src.chmod(0o644) + assert result.returncode == 1 + # The clean "inbox-send: <message>" shape, not a Python traceback. + assert result.stderr.startswith("inbox-send:") + assert "Traceback" not in result.stderr + + def test_discover_projects_dedupes_parent_and_child_root(self, tmp_path): + mod = _load_module() + # A project directory, reachable both as a child of its parent root and + # as a root in its own right. + parent = tmp_path / "projects" + proj = parent / "app" + (proj / ".ai").mkdir(parents=True) + (proj / "inbox").mkdir() + found = mod.discover_projects([parent, proj]) + resolved = [p.resolve() for p in found] + assert resolved.count(proj.resolve()) == 1 diff --git a/.ai/scripts/tests/test_route_recommend.py b/.ai/scripts/tests/test_route_recommend.py index acc4755..2ec900a 100644 --- a/.ai/scripts/tests/test_route_recommend.py +++ b/.ai/scripts/tests/test_route_recommend.py @@ -122,3 +122,31 @@ def test_cli_exclude_drops_current_project(tmp_path): r = _run(["--exclude", "foo"], roots=[tmp_path / "projects"], item="fix the foo widget") assert r.returncode == 0 assert r.stdout.strip() == "none" + + +# ---------------------------------------------------------------------- +# Duplicate candidate names +# +# Projects are collapsed to bare basenames, so two projects sharing a basename +# across roots (~/code/notes and ~/projects/notes) appear twice in the candidate +# list. Both literal-match, recommend read len(strong) > 1 as an ambiguous tie, +# and a correct strong match was downgraded to weak. Latent when discovered +# 2026-07-24 (27 projects, 27 distinct basenames) but real. +# ---------------------------------------------------------------------- + +def test_duplicate_candidate_name_keeps_strong_confidence(): + assert rr.recommend("fix the notes thing", ["notes", "other"]) == ("notes", "strong") + # The same name twice must not read as a tie. + assert rr.recommend("fix the notes thing", ["notes", "notes", "other"]) == ("notes", "strong") + + +def test_genuine_ambiguity_still_downgrades(): + # Two DIFFERENT projects both matching is a real tie and stays weak — the + # dedupe must collapse identical names only, never real ambiguity. + dest, conf = rr.recommend("notes and other both", ["notes", "other"]) + assert conf == "weak" + + +def test_duplicates_do_not_change_the_chosen_destination(): + dest, _ = rr.recommend("fix the notes thing", ["notes", "notes"]) + assert dest == "notes" diff --git a/.ai/scripts/todo-cleanup.el b/.ai/scripts/todo-cleanup.el index bd8166d..516e9b1 100644 --- a/.ai/scripts/todo-cleanup.el +++ b/.ai/scripts/todo-cleanup.el @@ -5,6 +5,8 @@ ;; emacs --batch -q -l todo-cleanup.el --check todo.org # hygiene report only ;; emacs --batch -q -l todo-cleanup.el --archive-done todo.org # archive completed subtrees ;; emacs --batch -q -l todo-cleanup.el --archive-done --check todo.org # preview the archive +;; emacs --batch -q -l todo-cleanup.el --seal todo.org # seal the working archive to resolved-YYYY-MM-DD.org +;; emacs --batch -q -l todo-cleanup.el --seal --check todo.org # preview the seal ;; emacs --batch -q -l todo-cleanup.el --convert-subtasks todo.org # dated-rewrite done level-3+ sub-tasks ;; emacs --batch -q -l todo-cleanup.el --convert-subtasks --check todo.org # preview the conversion ;; emacs --batch -q -l todo-cleanup.el --sync-child-priority todo.org # bump children whose priority drifted below the parent's @@ -37,23 +39,37 @@ ;; a message. Only direct level-2 children move — a DONE entry nested under ;; an open parent stays put. ;; -;; 2. Ages the "Resolved" section: a level-2 DONE/CANCELLED subtree whose -;; CLOSED date is older than `tc-archive-retain-days' (default 7) is moved +;; 2. Ages the "Resolved" section: a level-2 DONE/CANCELLED subtree is moved ;; out to `tc-archive-file' (default `archive/task-archive.org' beside the -;; todo file), keeping only the last week of closed tasks in the file -;; itself. Only subtrees closed within the window stay; older ones, and -;; those with no parseable CLOSED date, are moved out. Set -;; `tc-archive-retain-days' to nil to disable this step (legacy in-file-only -;; behavior). The aging date is `tc-archive-reference-date' when set -;; (tests), otherwise the real current date. The archive inherits the todo -;; file's gitignore status: when the todo file is gitignored, the archive -;; path is added to .gitignore before the first write, so private task -;; history never lands in a tracked path (see +;; todo file) when its CLOSED date is older than `tc-archive-retain-days' +;; (default 31 — one month) OR its CLOSED date can't be parsed. The last +;; month of closed tasks stays browsable in the file itself; older ones age +;; out. The unparseable-CLOSED case archives too, deliberately: a +;; keyword-complete task with no readable close date is cruft, not live +;; work. Set `tc-archive-retain-days' to nil to disable this step (legacy +;; in-file-only behavior). The aging date is `tc-archive-reference-date' +;; when set (tests), otherwise the real current date. The archive inherits +;; the todo file's gitignore status: when the todo file is gitignored, the +;; archive path is added to .gitignore before the first write, so private +;; task history never lands in a tracked path (see ;; `tc--ensure-archive-gitignored'). ;; ;; Archiving is consequential, so it's never run by default; it does *not* ;; also run the hygiene passes. ;; +;; * --seal (opt-in). Renames the working archive file (`tc-archive-file', +;; default `archive/task-archive.org') to `resolved-YYYY-MM-DD.org' beside it, +;; dated by the seal run, and leaves the next `--archive-done' to recreate a +;; fresh working file. The dated file means "everything sealed as of that +;; date" — not a calendar quarter — so a task closed late in a quarter and +;; archived after the boundary is never mislabeled; cadence (e.g. quarterly) +;; becomes independent of correctness and any slip is harmless. The sealed +;; file inherits the todo file's gitignore status the same way the working +;; archive does. A no-op (reported) when there's no working archive to seal; +;; refuses to clobber an existing `resolved-<today>.org'. Honors `--check'. +;; The seal date is `tc-archive-reference-date' when set (tests), otherwise the +;; real current date. +;; ;; * --convert-subtasks (opt-in). Rewrites every level-3-and-deeper heading whose ;; TODO state is DONE/CANCELLED/FAILED into a dated event-log entry ;; (`<stars> YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <text>'), dropping the keyword, @@ -84,6 +100,12 @@ ;; --check-child-priority is the report-only alias for --sync-child-priority ;; --check. +;; Before any modification a backup is copied to +;; /tmp/<basename>.before-todo-cleanup.<YYYYMMDD-HHMMSS> +;; matching lint-org.el and wrap-org-table.el. Skipped under --check, which +;; writes nothing. +;; + (require 'org) (require 'cl-lib) (require 'calendar) @@ -102,6 +124,17 @@ sub-task is terminal too and belongs in the parent's dated history.") (defconst tc--priority-cookie-regexp "\\[#\\([A-Z]\\)\\]" "Regexp matching an org priority cookie. Match group 1 is the letter.") +(defconst tc--planning-cookie-regexp + "\\(?:CLOSED\\|DEADLINE\\|SCHEDULED\\):[ \t]*[[<][^]>\n]*[]>]" + "One org planning cookie: a CLOSED/DEADLINE/SCHEDULED keyword followed by a +bracketed (inactive) or angled (active) timestamp.") + +(defconst tc--planning-line-regexp + (concat "\\`[ \t]*\\(?:" tc--planning-cookie-regexp "[ \t]*\\)+\\'") + "A whole org planning line: nothing but planning cookies and whitespace. +Anchored to a single line's contents so a line mixing a cookie with real body +text is never matched.") + (defconst tc-no-sync-tag "no-sync" "Org tag that opts a heading and all its descendants out of `--sync-child-priority'. Inherits down: a tag on an ancestor counts for @@ -112,20 +145,30 @@ every heading below it.") (defvar tc-bumped 0) (defvar tc-converted 0) (defvar tc-issues nil) +(defvar tc-sealed 0) (defvar tc-check-only nil) (defvar tc-archive-done nil) (defvar tc-sync-child-priority nil) (defvar tc-convert-subtasks nil) +(defvar tc-seal nil) (defvar tc-current-file nil) (defvar tc-current-dir nil) (defvar tc-archived-to-file 0) -(defvar tc-archive-retain-days 7 +(defconst tc-archive-retain-days-default 31 + "Default retention window (days) for the `--archive-done' file-aging step — +one month. A closed Resolved subtree stays in-file for this long before it ages +out to `tc-archive-file'; the last month of resolved work stays browsable in the +todo file itself. Named so the \"one month\" contract is explicit and testable.") + +(defvar tc-archive-retain-days tc-archive-retain-days-default "Retention window for the `--archive-done' file-aging step. A closed Resolved subtree whose CLOSED date is within this many days of the reference date stays in the in-file Resolved section; an older one is moved out to `tc-archive-file'. -A subtree with no parseable CLOSED date stays. nil disables the aging step -entirely, leaving the legacy in-file-only behavior.") +A subtree with no parseable CLOSED date is aged out too (a keyword-complete task +with no readable close date is cruft, not live work). nil disables the aging +step entirely, leaving the legacy in-file-only behavior. Defaults to +`tc-archive-retain-days-default' (one month).") (defvar tc-archive-reference-date nil "(YEAR MONTH DAY) treated as \"today\" when aging Resolved subtrees out to a @@ -479,6 +522,51 @@ step. Honors `tc-check-only' (report only)." tc-issues)))))))))) ;;; --------------------------------------------------------------------------- +;;; --seal mode: rename the working archive to a dated resolved-YYYY-MM-DD.org + +(defun tc--seal-date-string () + "YYYY-MM-DD for the seal — `tc-archive-reference-date' when set (tests), +otherwise the real current date." + (if tc-archive-reference-date + (pcase-let ((`(,y ,m ,d) tc-archive-reference-date)) + (format "%04d-%02d-%02d" y m d)) + (format-time-string "%Y-%m-%d"))) + +(defun tc-seal-archive-file () + "Rename the working archive file to `resolved-YYYY-MM-DD.org' beside it. +The next `--archive-done' run recreates a fresh working file. No-op (reported) +when there is no working archive to seal; refuses to clobber an existing +`resolved-<today>.org'. Ensures the sealed file inherits the todo file's +gitignore status. Honors `tc-check-only'." + (let ((path (tc--archive-file-path))) + (cond + ((or (null path) (not (file-readable-p path))) + (push (list :kind 'seal-nothing :file tc-current-file) tc-issues)) + (t + (let* ((dir (file-name-directory path)) + (sealed (expand-file-name + (format "resolved-%s.org" (tc--seal-date-string)) dir))) + (cond + ((file-exists-p sealed) + (push (list :kind 'seal-collision :file tc-current-file + :detail (file-name-nondirectory sealed)) + tc-issues)) + (tc-check-only + (cl-incf tc-sealed) + (push (list :kind 'seal-would :file tc-current-file + :detail (file-name-nondirectory sealed)) + tc-issues)) + (t + ;; Ignore the sealed name before the rename so its history stays as + ;; private as the working archive it derives from. + (tc--ensure-archive-gitignored sealed) + (rename-file path sealed) + (cl-incf tc-sealed) + (push (list :kind 'seal-done :file tc-current-file + :detail (file-name-nondirectory sealed)) + tc-issues)))))))) + +;;; --------------------------------------------------------------------------- ;;; --sync-child-priority mode (defun tc--heading-priority-letter () @@ -617,6 +705,14 @@ before their descendants — a [#A] → [#B] → [#D] chain collapses in one pas ;; as written). Idempotent: an already-dated heading has no done keyword, so it ;; is skipped. A done sub-task with no parseable CLOSED cookie can't be dated, so ;; it is flagged and left alone rather than stamped with a fabricated date. +;; +;; The planning line goes entirely. A dated-log entry carries its date in the +;; heading, so CLOSED is redundant and an active DEADLINE/SCHEDULED is wrong: org +;; renders any headline with an active planning timestamp — keyword or not — so a +;; SCHEDULED left on a dated-log heading pins it to the agenda as weeks-overdue +;; long after the work is done. The conversion deletes the whole planning line, +;; not just the CLOSED cookie (todo-format.md; lint checker +;; `dated-log-heading-active-timestamp' backstops any that slip through). (defun tc--closed-parts-in-entry () "Return a plist (:year :month :day :dow :hour :minute) from the CLOSED cookie @@ -676,6 +772,27 @@ in-progress `org-map-entries' walk; markers track their headings across edits." nil 'file) (nreverse targets))) +(defun tc--strip-planning-lines-in-entry () + "Delete the canonical planning line(s) directly under the heading at point. +A planning line is one composed solely of CLOSED/DEADLINE/SCHEDULED cookies and +whitespace. Walks the lines immediately after the heading and stops at the first +non-planning line, so a planning-shaped line deeper in the body (e.g. in a code +block) is never touched. Returns the count of lines removed." + (save-excursion + (org-back-to-heading t) + (forward-line 1) + (let ((removed 0) (continue t)) + (while (and continue (not (eobp))) + (let ((line (buffer-substring-no-properties + (line-beginning-position) (line-end-position)))) + (if (string-match-p tc--planning-line-regexp line) + (progn + (delete-region (line-beginning-position) + (min (1+ (line-end-position)) (point-max))) + (cl-incf removed)) + (setq continue nil)))) + removed))) + (defun tc--convert-one-subtask (marker) "Convert the done sub-task heading at MARKER to a dated event-log entry. Under `tc-check-only' the conversion is reported but not performed." @@ -698,27 +815,13 @@ Under `tc-check-only' the conversion is reported but not performed." (push (list :kind 'convert-would :file tc-current-file :line line :heading title :new new) tc-issues) - ;; Replace the heading line, then drop the now-redundant CLOSED - ;; cookie from the entry (its date now lives in the header). Only - ;; the cookie goes: a planning line can also carry DEADLINE: or - ;; SCHEDULED: beside it, and those survive on their line. A line - ;; left blank by the removal is deleted whole. + ;; Replace the heading line, then drop the whole planning line. The + ;; date now lives in the header, so CLOSED is redundant and an active + ;; DEADLINE/SCHEDULED would wrongly pin this completed entry to the + ;; agenda (todo-format.md). Both go, not just the CLOSED cookie. (delete-region (line-beginning-position) (line-end-position)) (insert new) - (let ((end (save-excursion - (or (outline-next-heading) (goto-char (point-max))) - (point)))) - (save-excursion - (when (re-search-forward "CLOSED:[ \t]*\\[[^]]*\\][ \t]*" end t) - (replace-match "") - (let ((bol (line-beginning-position)) - (eol (line-end-position))) - (if (string-match-p "\\`[ \t]*\\'" - (buffer-substring bol eol)) - (delete-region bol (min (1+ eol) (point-max))) - (goto-char bol) - (when (looking-at "[ \t]+") - (replace-match ""))))))) + (tc--strip-planning-lines-in-entry) (push (list :kind 'convert-done :file tc-current-file :line line :heading title :new new) tc-issues))))))) @@ -735,9 +838,37 @@ event-log entry, pulling the timestamp from its CLOSED cookie. Honors ;;; --------------------------------------------------------------------------- ;;; Driver + reporting +(defun tc--backup (file) + "Copy FILE to /tmp before any modification. Skipped in --check mode. + +Matches `lint-org.el' and `wrap-org-table.el', the other tools that rewrite +these org files. todo-cleanup runs the most often of the three (every wrap, +every sentry cycle), and Emacs's own backup does not fire under --batch -q, so +without this a mechanical rewrite has no undo short of git — which recovers +only to the last commit and loses intra-session work." + (let* ((base (format "%s%s.before-todo-cleanup.%s" + temporary-file-directory + (file-name-nondirectory file) + (format-time-string "%Y%m%d-%H%M%S"))) + (backup base) + (n 2)) + ;; Never overwrite an earlier backup. A second-resolution stamp collides + ;; when two invocations run back to back, which the shipped workflow does + ;; (open-tasks.org runs --convert-subtasks then --archive-done, each a + ;; sub-second batch run). Overwriting there replaces the true pre-session + ;; original with already-mutated content — losing exactly what the backup + ;; exists to preserve. Suffix instead, so every invocation keeps its own. + (while (file-exists-p backup) + (setq backup (format "%s-%d" base n)) + (setq n (1+ n))) + (copy-file file backup nil) + backup)) + (defun tc-process-file (file) (setq tc-current-file (file-name-nondirectory file)) (setq tc-current-dir (file-name-directory (expand-file-name file))) + (unless tc-check-only + (tc--backup file)) (with-current-buffer (find-file-noselect file) (org-mode) (cond @@ -747,6 +878,8 @@ event-log entry, pulling the timestamp from its CLOSED cookie. Honors (tc-sync-child-priority-in-file)) (tc-convert-subtasks (tc-convert-subtasks-in-file)) + (tc-seal + (tc-seal-archive-file)) (t ;; Pass 1: auto-fix bogus state logs (or report under --check). (org-map-entries #'tc-fix-bogus-state-log-in-entry nil 'file) @@ -865,10 +998,26 @@ event-log entry, pulling the timestamp from its CLOSED cookie. Honors (plist-get i :file) (plist-get i :line) (plist-get i :heading) (plist-get i :detail))))))))) +(defun tc--emit-seal-report () + (dolist (i (reverse tc-issues)) + (pcase (plist-get i :kind) + ('seal-done + (princ (format "todo-cleanup --seal: sealed task-archive.org → %s\n" + (plist-get i :detail)))) + ('seal-would + (princ (format "todo-cleanup --seal: would seal task-archive.org → %s — CHECK MODE (no writes)\n" + (plist-get i :detail)))) + ('seal-collision + (princ (format "todo-cleanup --seal: %s already exists — not sealing (already sealed today?)\n" + (plist-get i :detail)))) + ('seal-nothing + (princ "todo-cleanup --seal: no working archive to seal\n"))))) + (defun tc-emit-report () (cond (tc-archive-done (tc--emit-archive-report)) (tc-sync-child-priority (tc--emit-sync-report)) (tc-convert-subtasks (tc--emit-convert-report)) + (tc-seal (tc--emit-seal-report)) (t (tc--emit-hygiene-report)))) (defun tc-main () @@ -886,6 +1035,9 @@ event-log entry, pulling the timestamp from its CLOSED cookie. Honors (when (member "--convert-subtasks" command-line-args-left) (setq tc-convert-subtasks t) (setq command-line-args-left (delete "--convert-subtasks" command-line-args-left))) + (when (member "--seal" command-line-args-left) + (setq tc-seal t) + (setq command-line-args-left (delete "--seal" command-line-args-left))) ;; --check-child-priority is the report-only alias for ;; `--sync-child-priority --check'. (when (member "--check-child-priority" command-line-args-left) @@ -893,7 +1045,7 @@ event-log entry, pulling the timestamp from its CLOSED cookie. Honors (setq command-line-args-left (delete "--check-child-priority" command-line-args-left))) (if (null command-line-args-left) (progn - (princ "Usage: emacs --batch -q -l todo-cleanup.el [--check] [--archive-done | --convert-subtasks | --sync-child-priority | --check-child-priority] FILE...\n") + (princ "Usage: emacs --batch -q -l todo-cleanup.el [--check] [--archive-done | --seal | --convert-subtasks | --sync-child-priority | --check-child-priority] FILE...\n") (kill-emacs 1)) (let ((files command-line-args-left)) (setq command-line-args-left nil) @@ -912,6 +1064,7 @@ ert-run-tests-batch-and-exit'." (cl-every (lambda (a) (cond ((member a '("--check" "--archive-done" + "--seal" "--convert-subtasks" "--sync-child-priority" "--check-child-priority")) diff --git a/.ai/sessions/2026-07-19-21-15-launcher-hardening-inbox-hook-colloquialisms.org b/.ai/sessions/2026-07-19-21-15-launcher-hardening-inbox-hook-colloquialisms.org new file mode 100644 index 0000000..8f5207c --- /dev/null +++ b/.ai/sessions/2026-07-19-21-15-launcher-hardening-inbox-hook-colloquialisms.org @@ -0,0 +1,238 @@ +#+TITLE: Session Context +#+AUTHOR: Craig Jennings +#+DATE: 2026-07-18 + +* Summary + +** Active Goal + +Three tasks shipped this session, all closed, pushed, velox synced (HEAD f6a2701). No active goal remaining; next is a fresh backlog pick. + +1. ai-launcher-hardening [#C] (113e8d8, 2b619f1, closed 33c6d7b): hardened =claude-templates/bin/ai= per its measurable acceptance criteria. Launcher tests 9 → 42, four pure decision cores extracted (=_git_prep_action=, =_order_windows=, =_match_window_id=, =_git_is_dirty=) each N/B/E, footgun audit + /refactor pass fully dispositioned, shellcheck clean + shfmt -i2 -ci + suite green before/after + live smoke correct. +2. inbox-boundary-check hook [#B] (94e54f6, closed f6a2701): soft-nudge Stop hook enforcing the task-boundary inbox check. 6 bats, wired in settings.json + snippet, protocols note. Takes effect next session. +3. colloquialisms / "the list" convention [#B] (8fd9e39, closed f6a2701): protocols.org Colloquialisms section + wrap-it-up Step 1 Before-Close Queue sub-step. 4 bats. Live now. + +KB: promoted 0 / consulted no + +** Decisions + +- The task's acceptance criteria are FIXED and live in the task body (todo.org, the 4 moves per todo-format.md's "Making an open-ended task measurable"). Follow them; don't re-derive scope. +- Interactive runtime picker is OUT of this :solo: scope (design call) — file separately if wanted. +- Characterization discipline per testing.md (refined this session, commit 179c495): record-not-spec, full Normal/Boundary/Error set per unit, negative/boundary cases are the bug-finders, extract-pure-core-when-IO-blocks IS the hardening. + +** Data Collected / Findings + +- Surface (22 fns). COVERED (via scripts/tests/ai-launcher-runtime.bats, 9 tests, runtime path): resolve_agent_cmd, build_runtime_choices, pick_runtime, build_instructions, print modes. UNCOVERED (17 to net): usage, check_deps, attach_session, create_window, maybe_add_candidate, build_candidates, fetch_candidates, git_status_indicator, annotate_candidates, auto_pull_if_clean, read_selections, sort_windows, find_window_id, prep_git_single, attach_mode, single_mode, multi_mode, print_launch_mode. +- Pure/near-pure (characterize directly, N/B/E): git_status_indicator, maybe_add_candidate (dedup), annotate_candidates (format), read_selections (parse), usage. tmux/git-coupled (extract pure core + thin wrapper): sort_windows (ordering), create_window, attach_session, find_window_id, prep_git_single, auto_pull_if_clean. ~3 functional tests over single_mode/multi_mode/attach_mode against a throwaway tmux session. +- Canonical bin/ai is claude-templates/bin/ai; installed via make install's bin loop (symlink to ~/.local/bin/ai). Tests live in scripts/tests/ (not .ai mirror). shellcheck + shfmt present; kcov NOT installed. + +** Files Modified + +This session (all pushed to origin/main, velox synced to d49be09): sentry build a8b6cf4/ccc9c26/c6383e9/8c0a56b, trial fix beb7f0b, gui-open b3195e9, flashcard apkg converter a143679 + multi-tag a14e43b, task closes a760d8e, knowledge-arch landing 179c495/2e19048/94df71e, flake fix 94015e6, launcher scoping d49be09. Nothing in flight — tree clean at the flush. + +** Next Steps + +Launcher hardening is done and pushed. Pick the next backlog task. Strong candidates surfaced this session: the two [#B] :feature: shared-asset proposals (todo-cleanup dated-seal already shipped ddbd47f; remaining backlog includes the inbox-boundary-check hook, the colloquialisms/the-list convention, and the build-to-prototype ui-prototyping extension). Manual: the sentry overnight live-trial on ratio (4-part task) stays Craig's to run. + +STANDING (still in force this session): run the suite as its OWN step and read it green before each commit; sync velox (git pull + make install, ssh 100.127.238.103) after each push; /review-code + /voice personal per commit; keep velox current. + +* Session Log + +** 2026-07-18 Sat 17:52 CDT — Startup + inbox inventory + +Ran startup.org. Phase A.0: rulesets pull skipped (dirty tree — =.claude/settings.json= carries a harness-written model flip opus → claude-fable-5[1m] against the committed opus pin bd76d98; needs Craig's call). =make install= nothing new. Project fetch clean. Phase A: no crash anchor (clean prior wrap), =.ai/= synced from templates, staleness 3 tasks >7 days, roam inbox 18 items (all foreign — archsetup/takuzu/home/clock-panel, none rulesets-claimed), KB 97 nodes / no relevant titles / best-practices path unresolved, spec-sort + host-identity probes silent. + +Inbox: 7 pending. Acted on the two trivial ones immediately: +- website priority-scheme FYI — deleted (pure FYI, loop already closed by their "done"). +- website roam-KB-hosting-moved — applied the factual origin-URL fix in =claude-rules/knowledge-base.md= (git@cjennings.net:roam.git → cjennings@cjennings.net:git/roam.git), verified ratio's =~/org/roam= remote already points at the new URL. Deleted the inbox file. Commit + reply to website pending. Surfaced to Craig: rulesets.git itself is still publicly browsable on cgit (his open decision, per website's note). + +** 2026-07-18 Sat 18:02 CDT — Triage-intake redesign applied (item 1 of 3) + +Craig approved option 1 (apply as sent). Copied both sent canonicals over =claude-templates/.ai/workflows/{triage-intake,daily-prep}.org=, synced the mirror via =sync-check.sh --fix=, full suite green (374 + 67 pytest, all ERT expected, all bats ok, make exit 0). Review ran (/review-code --staged, inline): verdict Approve — the two remaining "suggested-actions" mentions are dated history entries (correct to keep), ORDER still governs the on-request long form, daily-prep line 58 keeps old "Action items" wording but is behaviorally accurate (Minor, not fixed). Commit 6e48714 (voice-passed, gate skipped — .ai/ tracked). Deleted the 3 inbox files, replied to work (delivered to their inbox 18:02). + +Three substantive shared-asset proposals queued for skeptical-review surfacing: +1. work 07-17: todo-cleanup.el --archive-done dated-seal model (retain 7→31, unparseable-CLOSED archiving, --seal rename). Craig ratified the design at work. Verified: todo-cleanup.el still has retain default 7, no seal — proposal premise current. +2. home 07-17: strip SCHEDULED/DEADLINE on dated-rewrite completion (todo-format.md + todo-cleanup --convert-subtasks) + new lint-org checker dated-log-heading-active-timestamp. +3. work 07-18: triage-intake Phase C/D redesign (three-section digest TASKS/FYI/MISC, close-by-default, reroute modifier) + daily-prep 3b one-liner — Craig's 2026-07-18 ruling, edited canonicals attached. Verified: diffs coherent, author lines clean (no "& Claude" regression), no other template references the retired format, plugins untouched by design. + +** 2026-07-18 Sat 18:11 CDT — Items 2 & 3 filed; 2 new home handoffs arrived + +Craig said "proceed" → filed both items 2 and 3 as [#B] :feature:solo: in todo.org, each with the design preserved to docs/design/ and cross-linked (both touch todo-cleanup.el, build as one batch). Replied to work (filed) for item 2. Route checks: both none (local keepers). settings.json resolved itself — Craig's /model opus set it back to match the committed opus pin (bd76d98), tree now clean on that file. + +Task-boundary inbox check caught 5 NEW home handoffs (17:53 + 18:04), two distinct proposals: +- Item 4: upcoming-birthdays feature promotion (cover + upcoming_birthdays.py + test + full daily-prep copy). Reviewed the code — clean, stdlib-only, 19 pytest cases (Normal/Boundary/Error, leap-day, placeholder-year 1900→None, window boundaries, CLI). Isolated the daily-prep reconcile: home's copy PREDATES my triage change, so only two birthday hunks apply (Heads-Up item 2 + Phase A source 8) — NOT a wholesale overwrite, which would revert my 3b triage edit. Home flagged this correctly. +- Item 5: "Colloquialisms and Expansions" + "the list" before-close-queue convention. Touches protocols.org + wrap-it-up.org (both synced) — a norm-adoption design call. + +** 2026-07-18 Sat 18:22 CDT — Item 4 (birthdays) applied + +Craig approved. Copied upcoming_birthdays.py + test into claude-templates/.ai/scripts/, applied ONLY the two birthday hunks to canonical daily-prep (verified my 3b triage edit survived — both copies still carry "three-section digest"), synced mirror. Suite green pytest 374→393 (19 new, all pass). Commit 80ebb74. Deleted 4 inbox files, replied to home (delivered 18:22). Also answered Craig's side question: all 27 projects carry the task-boundary inbox-check instruction (protocols.org synced), but it's a prose behavioral rule, not an enforced hook. + +Remaining: item 5 (colloquialisms/the-list convention). Still uncommitted from earlier: knowledge-base.md roam-URL fix + todo.org filings (items 2,3) + 2 docs/design proposals — batch at close-out. Craig also asked to explore a hook design for the inbox check after item 4. + +** 2026-07-18 Sat 19:56 CDT — Hook exploration, item 5 filed, inbox closed, all committed + +Explored the inbox-check hook with Craig. Key framing that emerged: two rails split by "must fire at a wall-clock instant, or just needs to be seen soon?" — boundary rail (Stop hook / UserPromptSubmit, fires when the agent yields, never interrupts mid-task) vs cron/at (must-fire-now: meeting alarms, meds, deadlines). The Stop event maps onto the rule's own boundary definition ("before reporting back"). Recommended soft-nudge Stop hook (stop_hook_active guard, once per turn) + UserPromptSubmit visibility injector. Filed as [#B] :feature: with full design. Craig extended the idea to a general boundary-check reminder rail (session-save nudge, uncommitted-drift nudge, soft reminders coming due) — offered a design note, he declined (not restarting, so no cold-start concern). + +Craig aborted the wrap (not restarting after all) and said complete the tasks. Actions: +- Filed item 5 (colloquialisms / "the list" convention) as [#B] :feature: — a cross-project convention adoption touching protocols.org + wrap-it-up.org, so filed not applied; needs Craig's adoption decision + a short design pass. Replied to home, design preserved to docs/design/. +- New archsetup FYI arrived mid-work (package-drift audit accepted on their side) — pure loop-close, deleted, no reply owed. +- Committed everything: 6523ed5 (knowledge-base roam URL fix), bc1d81d (4 backlog task filings + 3 docs/design proposals). Plus earlier 6e48714 (triage redesign), 80ebb74 (birthdays). + +STATE: inbox at zero, tree clean, suite green. 4 commits ahead of origin/main (6e48714, 80ebb74, 6523ed5, bc1d81d), 0 behind — UNPUSHED, awaiting Craig's push call. :LAST_INBOX_PROCESS: stamped 2026-07-18. + +Backlog filed this session (all [#B]): todo-cleanup dated-seal, dated-log planning-line strip + lint checker (batches with the seal), inbox-boundary-check hook, colloquialisms/the-list convention. + +** 2026-07-18 Sat 20:06 CDT — Two [#C] :quick:solo: closeouts (power-through) + +Craig picked the quick wins first. Both DONE + CLOSED, task-shaped (top-level): +- coverage-summary.el local-only doc (2cb7c1b, pushed after this): stated local-only status in the .el commentary header + elisp-testing.md "Measuring it" section (the gitignored .claude/scripts/ install is by-design, not a CI gap — Craig's 2026-06-28 decision). Sent emacs-wttrin a handoff to revert its contradicting header claim. +- install-ai on PATH (d2b1bef): new claude-templates/bin/install-ai thin launcher, resolves its own symlink chain and execs scripts/install-ai.sh. make install's existing bin loop links it to ~/.local/bin/install-ai (same as ai/agent-page) — resolves the task's open question (no dedicated sync, no dotfiles copy; the symlink is the canonical). 3 launcher bats incl. symlink-invocation. Verified live: install-ai --help runs from PATH. Suite green pytest 393, all bats/ERT pass. + +Remaining top solo work: the two batched todo-cleanup [#B] :feature:solo: tasks (dated-seal + planning-line strip) — the strongest next power-through target. + +** 2026-07-18 Sat 21:04 CDT — Batched todo-cleanup build (both [#B] :feature:solo: DONE) + +Craig said yes to the batch. Built both TDD in one commit ddbd47f (10 files, +746/-79), both DONE+CLOSED. + +Task A (dated-seal): retain default 7→31 via new defconst tc-archive-retain-days-default; unparseable-CLOSED archiving made explicit in the contract/commentary; new --seal mode (tc-seal-archive-file) renames task-archive.org → resolved-YYYY-MM-DD.org beside it, dated by seal run (not quarter — avoids late-quarter mislabeling), next --archive-done recreates fresh working file; sealed file inherits gitignore status; --seal flag + dispatch + report + CLI recognizer. 5 ERT tests. + +Task B (planning strip + lint): --convert-subtasks now strips the WHOLE planning line (CLOSED+SCHEDULED+DEADLINE) via tc--strip-planning-lines-in-entry, reversing the old CLOSED-only behavior (the enshrining test tc-convert-preserves-deadline... was rewritten to assert stripping). Stops at first non-planning line so body prose survives. New lint-org checker dated-log-heading-active-timestamp (flags active <..> SCHEDULED/DEADLINE on a keyword-less dated heading, ignores inactive [..]). todo-format.md sub-task rule step 5 + VERIFY path. 5 convert + 5 lint tests. + +Debugging note: seal tests failed only in the FULL suite — root cause was a latent bug, tc-test--reset never cleared tc-convert-subtasks, so a convert test's mode flag leaked and won the tc-process-file cond over seal. Fixed the reset + made the seal harness set all mode flags explicitly. Full suite green (todo-cleanup 52, lint-org 57, pytest 393, all bats). Replied to work + home (delivered). + +STATE at 21:04: ddbd47f committed, NOT yet pushed (2 quick-win commits 2cb7c1b/d2b1bef already pushed earlier). Reconcile clean before commit (0/0). + +** 2026-07-19 Sun 04:35 CDT — SENTRY BUILD STARTED (no-approvals + auto-flush) + +Craig approved the sentry spec after his deep read and told me to build it in no-approvals + auto-flush mode. Big pivot from the launcher-hardening [#C] (that was read-only, no edits to bin/ai — left as-is, still TODO). + +Spec flipped DRAFT → READY → DOING (docs/specs/2026-07-14-sentry-workflow-spec.org, ID f6c51f27-d7a2-4b63-9ff9-5ba005a66dfb). Metadata Status → doing. Build task decomposed into 4 phase sub-tasks under "** DOING [#B] Sentry workflow" with :SPEC_ID: binding. + +BUILD PLAN (resume anchor — re-read the spec's Design + Decisions + Implementation phases if context was cleared): +- Phase 1: =agent-lock= helper (canonical claude-templates/.ai/scripts/) — mkdir-atomic acquire, PID/host/ISO metadata, age-staleness reclaim (surfaced), bounded-wait contention (~30s), heartbeat refresh, acquire/release/status subcommands. Home: /run/user/<uid>/agent-locks/<name>/ with ~/.cache/agent-locks/ fallback. bats-tested. TDD. NOTHING calls it yet. +- Phase 2: =sentry.org= engine (.ai/workflows/, mirror synced) — entry ticket :COMMIT_AUTONOMY:, interactive entry gates, ff-only reconcile, sentry/YYYY-MM-DD-<host> branch, 10-pass runner (probe→work→session-context→commit), digest + approval queue, skip-and-note, spine-exclusion + fire-end digest commit, stall-notify after 2 skips, stop-sentry op. INDEX.org. +- Phase 3: companion reconciliations — knowledge-base.md, inbox.org core §5, roam-sync.sh header, triage-intake.org, wrap-it-up.org active-sentry guard. +- Phase 4: make test green; file live-trial follow-ups (Craig runs the trial); flip spec → IMPLEMENTED, close parent. + +Mode: no-approvals (commit+push per logical unit, Step 0 reconcile + pre-push check still run), auto-flush armed. COMMIT_AUTONOMY: yes + LOOP_MAY_COMMIT: yes in notes.org. + +Starting Phase 1 (agent-lock, TDD). + +** 2026-07-19 Sun @ 04:38:35 -0500 — flushed +Auto-flush before the sentry build. Clean boundary: spec flipped DRAFT→READY→DOING and decomposed (495e36b, pushed), nothing half-edited. Resuming into Phase 1 (agent-lock helper, TDD) per Summary Next Steps. + +** 2026-07-19 Sun @ 04:52 CDT — PHASE 1 DONE (agent-lock), pushed + +Built the agent-lock helper TDD (canonical claude-templates/.ai/scripts/agent-lock + mirror). Subcommands acquire/refresh/release/status/path. mkdir-atomic acquire, PID/host/ISO/TTL metadata, mtime-based staleness reclaim (surfaced note), heartbeat refresh, XDG_RUNTIME_DIR/agent-locks home with ~/.cache fallback. 18 bats tests. + +Self-review (review-code --staged, inline) caught one real concurrency bug pre-commit: the stale-reclaim path was rm -rf + mkdir in two steps, letting two acquirers who both see a lock stale double-acquire (loser deletes winner's fresh dir). Fixed with atomic-rename claim (mv wins-or-fails, then mkdir stays sole grant); strengthened test 7 to assert fresh metadata after reclaim. Verdict cleared to Approve. + +Full suite green (make test exit 0, 0 not-ok; agent-lock 18/18). Commit a8b6cf4, pushed to origin/main (0 behind, pre-push reconcile clean). /voice personal ran on the message. + +NEXT: Phase 2 — sentry.org engine (.ai/workflows/, mirror synced). Entry ticket :COMMIT_AUTONOMY:, interactive entry gates (dirty-tree / red-suite), ff-only reconcile, sentry/YYYY-MM-DD-<host> branch, pass runner (probe→work→session-context→commit contract), digest + morning-approval queue, skip-and-note semantics, spine-exclusion + fire-end digest commit, stall-notify after 2 unmerged-branch skips, stop-sentry op, INDEX.org entry. Re-read spec Design paragraphs (branch mechanics, locks, roam writes, unattended safety, pass list, digest) + the 10 decisions. + +** 2026-07-19 Sun @ 05:06 CDT — SENTRY BUILD COMPLETE (all 4 phases, pushed) + +All four sentry phases shipped, committed, pushed to origin/main, suite green throughout: +- Phase 1 a8b6cf4 — agent-lock helper (18 bats). Pre-commit review caught + fixed a reclaim double-acquire race (atomic-rename claim). +- Phase 2 ccc9c26 — sentry.org engine + INDEX entry. All 10 decisions / 12 findings reflected. +- Phase 3 c6383e9 — roam writers (knowledge-base.md, inbox.org §5) acquire roam-write lock + edit-plus-trigger (roam-sync sole committer); roam-sync.sh header; triage-intake note; wrap-it-up Step 0 active-sentry guard. Lock-name derivation pinned identically in sentry.org + wrap-it-up (sentry-<repo-basename>). +- Phase 4 8c0a56b — make test green at HEAD; spec DOING→IMPLEMENTED (dated history + Metadata mirror); build task + phase sub-tasks closed (sub-tasks → dated event-log entries); live trial filed as structured "Manual testing and validation" task. + +FINAL STATE: tree clean (only untracked spine), 0/0 vs origin/main, spec IMPLEMENTED, sync-check clean. Live agent-lock smoke test on the real runtime dir passed (acquire→held on ratio→release). Each commit ran /review-code (inline for docs) + /voice personal. + +REMAINING (Craig's, not agent-buildable): the overnight live-trial night on ratio — arm sentry, exercise the entry gates, observe one fire end to end, run the morning branch review. Filed as the 4-part manual-testing task in todo.org. Its findings become follow-up tasks. The launcher-hardening [#C] (read-only bin/ai) is still TODO, untouched (was pre-empted by the sentry build). + +** 2026-07-19 Sun @ 15:38 CDT — Live trial feedback #1 processed (roam-denylist mis-park) + +Craig ran a full sentry session from the WORK project. First trial finding: the inbox-zero pass (P2), running from ~/projects/work, parked the whole 19-item roam inbox as a cross-project boundary crossing and refused to tidy it — over-reading knowledge-base.md's work-denylist as "don't touch roam from work." Craig ruled: roam is a shared resource; the denylist only ever gated durable agents/ KB-node writes (a confidentiality guard). Reading roam + tidying the roam inbox are allowed from any project, work included. + +Fix (commit beb7f0b, pushed): applied work's prepared knowledge-base.md "Scope of the denylist — durable KB-node writes only" paragraph (naming the 2026-07-19 mis-park), plus one-line companion notes in sentry.org P2 and inbox.org roam mode pointing at the rule. Suite green. Replied to work (inbox-send, delivered 15:35), deleted the two work inbox items. + +Commits so far: a8b6cf4, ccc9c26, c6383e9, 8c0a56b (build), beb7f0b (trial fix #1). All pushed, tree clean. + +PENDING: 2 untracked archsetup items in inbox/ (11:47, gui-open proposal — unrelated to sentry, separate pass). Awaiting any further trial findings from Craig. + +** 2026-07-19 Sun @ 15:44 CDT — Inbox cleared: archsetup gui-open proposal applied + +Processed the two archsetup items (gui-open proposal). gui-open is a dotfiles-shipped launcher (a34d479, on PATH via ~/.dotfiles) that shows a file from a short-lived agent shell reliably — detaches through systemd-run --user (no shell reap), resolves the Hyprland instance after a restart, verifies a visible client. Cleared the value gate (fixes documented fragility both rules already flagged). + +Fix (commit b3195e9, pushed): interaction.md "Showing Craig Visuals" + desktop-capture.md "Showing the user something" now launch via gui-open instead of google-chrome-stable ... & / hyprctl dispatch exec imv. Guidance-only (tool is dotfiles-owned, not a rulesets script); added an "if not on PATH, needs a dotfiles pull" note. Replied to archsetup (delivered 15:42), deleted both inbox files. Inbox now clear. + +DAILY-DRIVER NOTE: gui-open ships via dotfiles a34d479 — confirmed on ratio (symlinked ~/.local/bin/gui-open → ~/.dotfiles). velox needs a dotfiles pull to have it, or the new rule guidance references a missing tool there. Flag to Craig. + +Session commits total: a8b6cf4, ccc9c26, c6383e9, 8c0a56b (sentry build) + beb7f0b (trial fix #1) + b3195e9 (gui-open guidance). All pushed, tree clean, inbox zero. + +** 2026-07-19 Sun @ 15:52 CDT — velox brought current (standing: keep it synced this session) + +Craig: do the dotfiles pull on velox now, and keep velox up to date with our changes for the rest of the session. Reached velox over tailscale (100.127.238.103, cjennings@). +- dotfiles: git pull → c3ef604 (was 9914a25), make restow hyprland (clean, no conflicts). gui-open symlinked ~/.local/bin/gui-open → dotfiles common tier; verified: direct exec prints usage, ~/.local/bin on interactive-login PATH (positions 1-2), interactive shell resolves it. Earlier "MISSING" was only the non-interactive SSH PATH. +- rulesets: git fetch + merge --ff-only → b3195e9 (was 3ae71be; also caught up on upcoming_birthdays, install-ai, docs/design proposals). make install relinked install-ai + agent-page, rest skipped. + +STANDING INSTRUCTION (rest of session): after each push, also update velox — git pull + make install on ~/code/rulesets, and a dotfiles pull + restow if dotfiles changed. ssh via 100.127.238.103. zsh gotcha: don't word-split an unquoted $VAR for the ssh command; run ssh inline. Non-interactive SSH PATH lacks ~/.local/bin — test tools via resolved path or zsh -lic. + +** 2026-07-19 Sun @ 18:42 CDT — Speedrun complete (2 flashcard tasks shipped) + +No-approvals speedrun over the 2 flashcard :solo: tasks, both TDD + review + voice, each its own commit, pushed, velox synced after each. +- apkg-to-orgdrill.py (a143679): inverse converter, stdlib zipfile+sqlite3, 17 tests + real-genanki round-trip. Grounded the apkg schema by generating/inspecting a real one first. +- flashcard multi-tag reconcile (a14e43b): broadened CARD_RE in to-anki + stats for :fundamental:drill:, added --tag-filter + --guid-salt + drill-membership guard; re-derived against current canonical (kept the #+TITLE fix); parse() 3rd element now anki-tag list. End-to-end verified (2→1 with --tag-filter). 465/100 real-deck check needs the work deck (not runnable here). +- Closed both as dated event-log entries (a760d8e). + +Suite green throughout (419 pytest). Session commits now: sentry build (4) + trial fix + gui-open + flashcard (2) + closes. All pushed, velox current at a760d8e. + +NEXT (Craig asked): discuss the ai launcher hardening [#C] task (todo.org line ~208). Remind him what it is. + +** 2026-07-19 Sun @ 19:04 CDT — Landed the testing/acceptance knowledge-architecture change + +Craig's directive: build the characterization-test + measurable-acceptance discipline into the workflows, and decide the KB-vs-rule boundary. Answer settled in discussion: apply-every-time discipline → rules (single source, auto-loaded, review-gated contributions via inbox); pull-when-relevant cross-project facts → KB. Almost none of this is KB material (it has a rule home); the KB is the capture/holding-pen, promotion moves it OUT into rules. + +Two commits: +- 179c495 (characterization discipline): testing.md — defined a characterization test (record-not-spec, Feathers recipe), required the full Normal/Boundary/Error set per unit, and the key lens that negative/boundary cases are the bug-finders; + the "extracting the pure core IS the hardening" framing in the refactor-for-testability section. code-quality.org gained the precondition that behavior-preserving rests on a characterization net. +- 2e19048 (measurable acceptance): todo-format.md new subsection "Making an open-ended task measurable (so it can be :solo:)" — bound surface / characterization net / disposition findings / objective floor; qualifying answer = dispositioned report. work-the-backlog's keystone defer item now recognizes absence-phrased open-ended goals and routes them to get criteria. + +Scope note: start-work and /refactor are NOT rulesets files (not installed at ~/.claude/skills/; registered elsewhere). Their pointers are a follow-up in whatever repo defines them — flagged to Craig. add-tests + review-code (rulesets skills) already well-wired to testing.md. + +Suite green, lint 0/0, velox synced to 2e19048. + +** 2026-07-19 Sun @ 19:14 CDT — Flaky-test fix + a discipline note + +While landing the start-work pointer (94df71e), chained make test into the commit command and the commit went through on a RED run — the bundling verification.md warns against. The red turned out flaky (rename-ai-artifact.bats teardown race, unrelated to the md edit; suite green on re-run, change sound), but the gate failed open. Correction going forward: run the suite as its own step, read it, then commit. + +Root-caused + fixed the flake (94015e6): rename-ai-artifact.bats teardown hit "rm: cannot remove .git: Directory not empty" because git background auto-maintenance (maintenance run --auto after commit) wrote into .git after the body, racing rm -rf. Fix: gc.auto 0 + maintenance.auto false in the throwaway repo before any git command arms them. Diagnose-not-mask (no rm retry). Verified 30 file runs + 2 full-suite runs green. + +Also corrected an earlier wrong claim: start-work + refactor ARE rulesets files (.claude/commands/, symlinked to ~/.claude/commands), already wired to testing.md for characterization. The acceptance-pattern landing is fully complete — no out-of-repo follow-up. + +** 2026-07-19 Sun @ 19:20 CDT — Scoped the ai-launcher-hardening task (dogfooded the pattern) + +Applied the measurable-acceptance pattern to the launcher task (d49be09). Rewrote the open-ended body with the 4 moves: bounded surface (22 fns, 5 covered via ai-launcher-runtime.bats / 17 uncovered, all named), characterization-net plan (pure-core extraction called out for the tmux/git-coupled fns), dispositioned audit (per-fn footgun matrix + /refactor pass), objective floor (shellcheck/shfmt/green suite/per-fn coverage + ~3 functional tests over single/multi/attach_mode). Split the interactive runtime picker OUT of :solo: (design call). Task is now genuinely :solo:-ready. :LAST_REVIEWED: bumped to 2026-07-19. + +The actual hardening work (bring 17 fns under characterization tests, extract pure cores, run the audit + refactor) is a substantial next-session speedrun candidate — now bounded. + +** 2026-07-19 Sun @ 20:19:15 -0500 — flushed +Interactive flush at a clean boundary (launcher task scoped + committed d49be09, tree clean, nothing half-edited). Resuming into option 2: work the ai-launcher-hardening characterization sweep per the refreshed Summary Next Steps. + +** 2026-07-19 Sun @ ~21:30 CDT — LAUNCHER HARDENING COMPLETE (2 commits + close, pushed, velox synced) + +Worked the ai-launcher-hardening [#C] :refactor:solo: end to end per its measurable acceptance criteria. Craig present, chose push+sync+close. + +Sequence: +- Grounded in bin/ai (540 lines), confirmed green baseline (make test exit 0), pinned the gate targets: shellcheck-clean (the bash hook enforces shellcheck only; shfmt is deliberately NOT hook-gated), shfmt house style = -i 2 -ci (agent-page conforms, bin/ai did not). +- Commit 113e8d8 (net + source-testable): wrapped dispatch in main() behind a run-vs-sourced guard so the file is sourceable for unit tests; cleared 4 pre-existing shellcheck warnings on touched paths (@{u} quoted ×3 = SC1083, literal display tilde disabled = SC2088). New scripts/tests/ai-launcher-characterization.bats, 20 tests: pure/near-pure N/B/E (git_status_indicator, maybe_add_candidate, annotate_candidates, read_selections, build_candidates, usage) + 4 functional (create_window, find_window_id, sort_windows, attach_mode) against a PRIVATE tmux socket (TMUX_TMPDIR + unset TMUX) so nothing touches Craig's live ai session; throwaway repos disable gc.auto/maintenance.auto (the rename-ai flake). +- Commit 2b619f1 (refactor): extracted four pure cores — _git_prep_action (git-prep none/pull/report classifier, shared by prep_git_single + auto_pull_if_clean), _order_windows (sort_windows ordering), _match_window_id (find_window_id), _git_is_dirty (DRY the dirty triad ×3). Added 13 unit tests for the cores incl. the safety case (dirty repo never auto-pulled). shfmt -i2 -ci applied. 42 launcher tests green, live black-box smoke correct for claude+codex. +- Commit 33c6d7b: closed the task DONE+CLOSED (top-level ** → task-shaped) with the dispositioned resolution note. + +Footgun audit fully dispositioned (report delivered inline to Craig): unquoted-expansion/word-split — clean; errexit-off — intentional/documented, kept; subshell state loss — none (globals set in main shell); exit-code propagation — intentional; sort_windows race — none (two-pass park-then-reassign is correct); create_window sleep 0.1 — declined (intentional); git-prep error paths (dirty/detached/no-upstream) — all correct, dirty-safety now a named test. /refactor: 4 extractions + shfmt applied; build_candidates exit-status + create_window sleep declined. + +Honest coverage limit stated: attach_session and full end-to-end of single/multi/fetch stay partially covered (terminal step attaches/blocks on fzf, can't run headless); their decision logic was extracted into the netted cores. + +STATE: 33c6d7b pushed, 0/0 vs origin/main, velox synced to 33c6d7b (make install: ai linked), tree clean (only untracked session anchor), suite 367 ok / 0 not-ok. Interactive runtime picker stayed OUT of scope (design call, on the generic-agent-runtime parent). No memory promotion needed (nothing durable/cross-project beyond the repo record). + +** 2026-07-19 Sun @ ~22:00 CDT — No-approvals speedrun: inbox hook + colloquialisms (both [#B] DONE) + +Craig said "no approvals speedrun time" after the launcher task. Built two [#B] backlog tasks, each TDD + inline review + /voice personal, committed + pushed + velox-synced. + +Task 1 — inbox-boundary-check hook (94e54f6): soft-nudge Stop hook backing the "check inbox/ at every task boundary" rule (was prose-only). Blocks the yield once + injects the pending count when =inbox-status -q= exits 1, steps aside on =stop_hook_active= re-entry (soft, not hard-block, so a mid-task pause never wedges), self-skips on no-inbox/no-inbox-status/clean. Prefers project-local =.ai/scripts/inbox-status=, PATH fallback. Wired ahead of =ai-wrap-teardown= in =.claude/settings.json= (which the live =~/.claude/settings.json= symlinks to → live global on both boxes) + =settings-snippet.json= + README row; glob-installed by =make install-hooks=. protocols.org Inbox Monitoring Cadence notes the enforcement. 6 bats. Live-verified on ratio (clean no-op, pending fixture blocks). Loads next session (hooks read at session start), so it didn't interfere with this wrap. + +Task 2 — colloquialisms / "the list" (8fd9e39): new =* Colloquialisms and Expansions= section in protocols.org. "put X on the list" → session-scoped Before-Close Queue (=* Before-Close Queue= heading in the session anchor, resets on archive, todo.org for must-outlive); "tell <project> <msg>" → =inbox-send=. wrap-it-up Step 1 gained a "Work the Before-Close Queue (before the Summary)" sub-step so queued work rides the wrap commit. Design calls: reference in protocols.org not per-project notes.org (synced = shared norm); queue in the session anchor as home did; wrap step at front of Step 1 not a half-step (keeps "Steps 1-5" framing). 4 documentation-integrity bats. Canonical+mirror synced. + +Both closes in f6a2701. Full suite green throughout (0 not-ok). velox synced to f6a2701 (make install ok). Then Craig said wrap it up. diff --git a/.ai/sessions/2026-07-20-23-36-signal-pager-and-triage-phone-push.org b/.ai/sessions/2026-07-20-23-36-signal-pager-and-triage-phone-push.org new file mode 100644 index 0000000..fe43f61 --- /dev/null +++ b/.ai/sessions/2026-07-20-23-36-signal-pager-and-triage-phone-push.org @@ -0,0 +1,185 @@ +#+TITLE: Session Context — Sentry live trial (ratio, 2026-07-19) +#+AUTHOR: Craig Jennings + +* Summary + +** Active Goal + +Post-flush session. Two items shipped: (1) the archsetup voice-#46 (comma-budget) handoff, reviewed and committed rulesets-side (2ea5d9a); (2) the triage-intake auto-mode phone-push, built onto agent-text, signal-only per Craig's ruling (e27aea2). Items 2 (reply-correlation spec) and 3 (token-rotation discussion) from the prior next-steps list remain untouched, carried forward. + +** Decisions (this session, all shipped) + +- Voice pattern #46 (comma budget, max two per sentence): personal mode only, in the attestation high-recurrence set. Craig's 2026-07-20 direction via archsetup; committed rulesets-side 2ea5d9a. +- Triage auto-mode phone-push is signal-only (Craig's option 1): a quiet sweep's "nothing" heartbeat never pushes to the phone — silent-until-signal governs the phone channel too. Send half shipped (e27aea2); reply-polling deferred to the reply-correlation spec. +- Sentry live trial: ran on ratio, 8 hourly fires, stopped on "sentry off", branch fast-forwarded to main and deleted. +- working/ is tracked-from-creation + gitignored temp/ in both modes (Shape A). Committed. +- Triage source activation: general (synced) plugins gate on a per-project =:TRIAGE_SOURCES:= declaration; project-specific plugins active by presence; gate in triage-intake Phase 0 (interactive + unattended). Spec IMPLEMENTED. Migrations done (home + work declared their sources). +- Silent-until-signal: a POLICY not a mechanism — an in-session monitor fire heartbeats =<workflow> at HH:MM: nothing= on an empty check; detection stays in-session (MCP-safe). Applied to sentry, auto triage-intake, auto inbox-zero. Spec IMPLEMENTED. +- Suspend detach change applied to canonical. Sentry cluster consolidated (merged /schedule tasks, added cross-host-coordination task). Task audit stamped 2026-07-20. +- Polyglot: case-by-case, no option-2 machinery (bundles already compose; only coverage-makefile.txt collides, and it's a manual paste). Subprojects: don't promote (N=1). Both closed. + +** Data Collected / Findings — Signal pager (what the resume needs) + +RECONCILED (2026-07-13, in the task's dated log): there is ONE pager identity, =+15045173983=, registered in *velox's* signal-cli (account file 465310). =signal-mcp= is a velox-local MCP server (invisible from ratio). ratio's signal-cli holds only Craig's personal number (note-to-self, no push). =agent-page= already shipped (=claude-templates/bin/agent-page=): runs signal-cli directly on velox, ssh-relays from anywhere else, desktop-fallback on failure, 4 bats, live-verified. protocols.org "Paging Craig" was already rewritten around the two channels (notify desktop + agent-page phone). Craig's Signal UUID: =b1b5601e-6126-47f8-afaa-0a59f5188fde=. Reliability finding: both signal-cli accounts throw receive-staleness warnings (velox ~40 days, ratio ~26) — the Signal protocol wants regular receives, and a systemd receive timer on velox is the roam-sync-shaped fix. + +REMAINING deliverables (this is the work to do): (1) the runbook proper — send + read-replies + receive-timer + signal-cli account/setup notes, the Signal equivalent of the retired ntfy runbook, canonical home in rulesets docs/; (2) a systemd receive timer on velox (roam-sync-shaped) so receives don't go stale; (3) the ssh-over-tailnet-only vs register-ratio-as-a-linked-device decision (a design call for Craig); (4) confirm protocols.org "Paging Craig" is accurate (agent-page did most of it — verify, don't redo). Task body has full history. Source: home handoff 2026-07-04. + +** Files Modified (this session — all committed + pushed to main) + +Post-flush commits: 2ea5d9a (voice #46 comma budget — SKILL.md + voice-profile.org), 9721a49 (chore: mark archsetup handoff PROCESSED), e27aea2 (feat: triage phone-push via agent-text — canonical + mirror triage-intake.org, todo.org task closed). Left unstaged: .claude/settings.json model change (opus→fable, not this session's work — deferred). + +Earlier same-session (pre-flush): f625cf5 (working/temp feat), b02eade + 4d87f35 (triage source activation spec + build), 986d6ca + 0767af8 + 0e9958a (silent-until-signal spec + Phases), af565ba (suspend detach), 70fbe01 (link fixes), 93a2e6d (working-dir filing), c82b625 (sentry cluster), up to fecdf8c, plus the pager work (302b062, 6145489) and overnight sentry commits. + +** Next Steps + +Item 1 (triage phone-push) shipped this session. Two remain from the prior list: + +1. *Spec the reply-correlation follow-up.* The two-way gap: with the account linked on velox + ratio, a Signal reply reaches BOTH devices (one account, Signal fans out per-device, independent queues) and neither knows which page/session it answers. Only bites when two sessions page-and-wait at once (fire-and-forget is fine). Options laid out to Craig: (1) correlation tag stamped on the page, echoed in reply; (2) quote-reply matching (needs verifying signal-cli surfaces quotes); (3) single reply-owner (velox-only). Overlaps the helper-instance work (same shared-channel problem). This spec also owns the deferred triage-intake reply-polling (phone-recv) half. Also fix the runbook "Reading replies" section, which oversells it. Write as a spec in docs/specs/ (spec-create spine). + +2. *Discuss the token-rotation helper* (todo.org =[#C] Token-rotation helper for @a-bonus/google-docs-mcp OAuth refresh=, :feature:quick:) — a discussion first, not a build. Read the task body. + +Publish flow (review + voice + suite) for any commits. + +KB: promoted 0 / consulted no + +* Session Log + +** 2026-07-19 23:52 CDT — Sentry armed (entry) + +Startup ran clean: rulesets pull no-op, make install nothing new, project repo up to date on main (0/0). Session-context was absent (prior session 2026-07-19-21-15 wrapped cleanly). + +Sentry entry gates, all passed with Craig present: +- Autonomy ticket: =:COMMIT_AUTONOMY: yes= in notes.org Workflow State. +- Host: ratio (intended live-trial machine). +- Dirty-tree gate: tracked tree clean; only two untracked inbox files (=2026-07-19-2141-from-.emacs.d-version-working-directories.org=, =2026-07-19-2350-from-home-craig-approved-the-colloquialisms-the.org=) — do not block; sentry's inbox-zero pass (pass 2) handles them. +- Green-suite gate: =make test= exit 0, 377 bats ok, all pytest + ERT green. +- Prior sentry branch: none. +- Reconcile: main 0 behind / 0 ahead of upstream. +- =agent-lock= helper present. + +Created daily branch =sentry/2026-07-19-ratio= from HEAD (main @ f76bf40). Working tree now sits on this branch overnight. Arming the hourly loop next. + +Two inbox items pending at entry, both to be handled by pass 2 (inbox zero): +- .emacs.d working/ version-control ruling — a shared-asset change (working-files convention + install .gitignore behavior across projects). Parks for morning approval; does not fire unattended. +- home colloquialisms wiring approval — already shipped on the rulesets side in the prior session (protocols.org Colloquialisms section + wrap-it-up Before-Close Queue step, commits 8fd9e39/f6a2701). Inbox zero replies-and-files, confirming it's already wired. + +** 2026-07-20 00:06 CDT — Fire 1 digest (manual first fire, Craig present) + +Ran end to end on branch =sentry/2026-07-19-ratio=. Single-runner lock =sentry-rulesets= acquired at fire start, on the sentry branch, tree clean. Pass-by-pass: + +- P1 roam pull — SKIPPED: =~/org/roam= working tree dirty; roam-sync owns that case (pass is read-only ff-only otherwise). No write. +- P2 inbox zero — RAN. Two project handoffs processed. (a) home's colloquialisms + "the list" before-close-queue: already wired canonically before it arrived — replied to home confirming, marked PROCESSED. (b) .emacs.d's working/ tracked-from-creation ruling: shared-asset + convention change with a temp/-placement design decision → PARKED (see approval queue). Staged proposal at =working/sentry-2026-07-19-working-files-ruling/proposed.org=, filed a [#B] VERIFY, replied to .emacs.d with two findings. Committed 727a900. +- P3 triage intake — SKIPPED (probe-too-loose; queued as a finding). The template-synced general plugins (personal Gmail/calendar/cmail/Telegram/GitHub-PRs) are present in *every* project, so the "plugins present" probe self-activates triage everywhere — including rulesets, which is not a triage target. Running it would file Craig's personal action items into rulesets' todo.org and run trash/mark-read/star hygiene on his real accounts unattended. Skipped for scope + safety. No write. +- P4 todo cleanup — RAN: hygiene 0 fixes, --convert-subtasks 0, --archive-done 0. todo.org already clean. No write. +- P5 task audit — RAN (mechanical subset). No unambiguous autonomous staleness fixes: all open tasks carry a priority cookie + type tag except the intentional "Manual testing and validation" container; spec-lifecycle clean (P7); no dead file: links in todo.org. The full reconciliation (per-task fact-check, consolidation, parent-retirement, judgment flags) is interactive → queued. Did NOT bump :LAST_AUDIT: (only the mechanical subset ran). No write. +- P6 working-files hygiene — RAN: flagged =working/inbox-zero-phase-e/= (tracked, backs the now-IMPLEMENTED autonomous-batch-execution spec) as a filing candidate. Filing is a judgment move (3 inbound spec links need updating) → queued. No write. +- P7 spec status board — RAN: clean. All specs READY or IMPLEMENTED; the sentry spec is IMPLEMENTED; no DOING spec with a closed/missing bound parent. No write. +- P8 link integrity — RAN (report-only): ~4 candidate broken file: links in live docs — the autonomous-batch spec's =../../.ai/workflows/inbox-zero.org= (renamed to inbox.org), two folded-in spec-review siblings (wrapup-routing, pattern-catalog), and =subprojects-log.org=. Some may be intentional (review docs folded in and deleted). No unattended rewrites. No write. +- P9 git health — RAN: clean. main = origin/main (0/0), no unpushed commits on other branches, no stale merged branches, sentry branch is the only local extra. No write. +- P10 prep + symlink freshness — SKIPPED: no prep dir (work/home only). No write. + +Fire-end: only org/spine files touched this fire (todo.org, inbox renames, working/ proposal — all org) → conditional suite skipped per contract. Session-context spine is untracked in rulesets, so no digest commit needed; tracked tree is clean. Single-runner lock released. + +Branch state after fire: one commit ahead of main (727a900). Next scheduled fire at :17. + +** 2026-07-20 00:21 CDT — Fire 2 digest (scheduled, :17 cron) + +Lock =sentry-rulesets= acquired, on the sentry branch, tree clean. Little changed in the 14 min since fire 1 — one new inbox handoff, everything else steady. + +- P1 roam pull — RAN: =~/org/roam= clean this fire (was dirty in fire 1), ff-only pull → already up to date. No write. +- P2 inbox zero — RAN: one new handoff. .emacs.d routed three sentry-workflow design considerations from its own hand-run trial (codebase-gated bug/enhancement logging pass; system-health pass; sibling-machine freshness pass). Filed as =** TODO [#C] Sentry vNext passes — from live-trial design input= rather than applied (design input for the Living Document; two hinge on an unresolved cross-driver coordination question). Replied to .emacs.d, marked PROCESSED. Committed 067ed55. +- P3 triage intake — SKIPPED: same probe-too-loose reason as fire 1; the finding is already in the approval queue, not re-queued. +- P4 todo cleanup — RAN: 0 fixes. No write. +- P5 task audit — no change since fire 1; the full-audit-due item stays in the queue, not re-added. +- P6 working-files hygiene — RAN: =working/inbox-zero-phase-e/= still a filing candidate (already queued fire 1, not re-queued). =working/sentry-2026-07-19-working-files-ruling/= is an *active* working dir backing the open VERIFY task — correctly not a candidate. No new queue item. +- P7 spec status board — RAN: clean, no DOING specs. No write. +- P8 link integrity — not re-scanned: no doc changes since fire 1's scan except the new backlog task, whose one internal =file:todo.org::*...= link resolves (KB lesson-detection heading exists). Fire 1's report stands. No write. +- P9 git health — RAN: clean, main = origin/main (0/0), branch now 2 ahead. No write. +- P10 prep freshness — SKIPPED: no prep dir. No write. + +Fire-end: only org files touched → conditional suite skipped. Spine untracked → no digest commit; tracked tree clean. Lock released. Branch 2 commits ahead of main (727a900, 067ed55). No new approval-queue items this fire. + +** 2026-07-20 01:20 CDT — Fire 3 digest (scheduled, :17 cron) + +Quiet read-only fire — nothing changed in the hour since fire 2. All passes no-op or skip; no writes, no commit, no new approval-queue items. + +- P1 roam pull — RAN: clean, ff-only → already up to date. +- P2 inbox zero — RAN: 0 new items. +- P3 triage intake — SKIPPED: same probe-too-loose reason (finding already queued fire 1). +- P4 todo cleanup — RAN: 0 fixes. +- P5 task audit — no change; full-audit-due item stays queued. +- P6 working-files hygiene — RAN: same two dirs (inbox-zero-phase-e already queued; sentry-ruling active for the open VERIFY). No new item. +- P7 spec status board — RAN: clean, no DOING specs. +- P8 link integrity — not re-scanned: no doc changes since fire 1. +- P9 git health — RAN: clean, main = origin/main, branch 2 ahead. +- P10 prep freshness — SKIPPED: no prep dir. + +Branch unchanged at 2 commits ahead. Lock released. + +** 2026-07-20 02:20 CDT — Fire 4 digest (scheduled, :17 cron) + +Quiet read-only fire, same as fire 3 — nothing changed in the hour. All passes no-op or skip; no writes, no commit, no new queue items. P1 roam up to date; P2 inbox 0 new; P3 triage skipped (finding already queued); P4 0 fixes; P5 no change (full audit still queued); P6 same two working dirs (inbox-zero-phase-e queued, sentry-ruling active); P7 clean, no DOING specs; P8 not re-scanned (no doc changes); P9 clean, branch 2 ahead; P10 skipped (no prep dir). Lock released. + +** 2026-07-20 03:20 CDT — Fire 5 digest (scheduled, :17 cron) + +Quiet read-only fire, same as fires 3-4. No changes in the hour; all passes no-op or skip; no writes, no commit, no new queue items. P1 roam up to date; P2 inbox 0 new; P3 triage skipped (queued); P4 0 fixes; P5 no change; P6 same two working dirs; P7 clean; P8 not re-scanned; P9 clean, branch 2 ahead; P10 skipped. Lock released. + +** 2026-07-20 04:20 CDT — Fire 6 digest (scheduled, :17 cron) + +Quiet read-only fire, same as fires 3-5. No changes in the hour; all passes no-op or skip; no writes, no commit, no new queue items. P1 roam up to date; P2 inbox 0 new; P3 triage skipped (queued); P4 0 fixes; P5 no change; P6 same two working dirs; P7 clean; P8 not re-scanned; P9 clean, branch 2 ahead; P10 skipped. Lock released. + +** 2026-07-20 05:20 CDT — Fire 7 digest (scheduled, :17 cron) + +Quiet read-only fire, same as fires 3-6. No changes in the hour; all passes no-op or skip; no writes, no commit, no new queue items. P1 roam up to date; P2 inbox 0 new; P3 triage skipped (queued); P4 0 fixes; P5 no change; P6 same two working dirs; P7 clean; P8 not re-scanned; P9 clean, branch 2 ahead; P10 skipped. Lock released. + +** 2026-07-20 06:20 CDT — Fire 8 digest (scheduled, :17 cron) + +Quiet read-only fire, same as fires 3-7. No changes in the hour; all passes no-op or skip; no writes, no commit, no new queue items. P1 roam up to date; P2 inbox 0 new; P3 triage skipped (queued); P4 0 fixes; P5 no change; P6 same two working dirs; P7 clean; P8 not re-scanned; P9 clean, branch 2 ahead; P10 skipped. Lock released. + +* Sentry approval queue (2026-07-19) + +Judgment/parked items from the overnight fires. Review top to bottom; run or discard each. + +** [Fire 1 · P2] Apply the .emacs.d working/ tracked-from-creation ruling +- *What:* update the canonical working-files convention + add a gitignored temp/ pattern across projects. +- *Why:* Craig's ruling relayed from .emacs.d (2026-07-19). Parked because it's a shared-asset + convention change and carries a design decision (temp/ must be ignored in both track and gitignore modes — it's orthogonal to the personal-tooling set). +- *Prepared:* =working/sentry-2026-07-19-working-files-ruling/proposed.org= (exact edits + the two findings). Filed as the =** VERIFY [#B] working/ tracked-from-creation + gitignored temp/= task in todo.org. +- *Fires on approval:* edit =claude-rules/working-files.md= (tracked-from-creation + temp/ subsections), =scripts/install-ai.sh= + =scripts/sweep-gitignore-tooling.sh= (emit temp/ ignore in both modes — decide shape (a) vs (b) in the proposal), =.ai/protocols.org= Working-Files Convention (one-line mirror); then =scripts/sync-check.sh --fix= for the .ai mirror, run =make test=, commit. + +** [Fire 1 · P3] Tighten the sentry triage-intake pass probe — SPECCED during morning review +- *What:* the narrow "fix the probe" framing grew, on Craig's flexibility question, into a per-project source-activation model for triage-intake. +- *Resolution:* written up as a spec for review — =docs/specs/2026-07-20-triage-source-activation-spec.org= (DRAFT), linked from the =** TODO [#B] Triage source activation= task. General plugins gate on a per-project =:TRIAGE_SOURCES:= declaration; project-specific plugins stay active by presence; the activation layer lives in triage-intake Phase 0 so it fixes the interactive over-pull too; sentry's pass-3 probe reads the same signal. +- *Next:* Craig's deep read → DRAFT → READY → spec-response decomposes the 5 phases. Two decisions still open (declaration format, whether interactive adopts the same gate). No code landed — this item is now tracked by the spec + task, not the queue. + +** [Fire 1 · P5] A full interactive task-audit is due +- *What:* run task-audit.org interactively (its consolidation / parent-retirement / judgment-flag phases need Craig). +- *Why:* :LAST_AUDIT: is 2026-07-04 (~16 days); task-review-staleness flags 2 top-level tasks unreviewed >7 days. Fire 1's mechanical subset found nothing unambiguous to auto-fix, so the marker was deliberately not bumped. +- *Fires on approval:* "let's do a task audit" (or task-review for the lighter pass). + +** [Fire 1 · P6] File working/inbox-zero-phase-e/ to a permanent home +- *What:* file the three artifacts under =working/inbox-zero-phase-e/= per working-files.md and update inbound links. +- *Why:* the backing work (autonomous-batch-execution spec) is IMPLEMENTED, so the working dir is a filing candidate. Filing is a judgment move — 3 inbound =file:= links in =docs/specs/2026-06-16-autonomous-batch-execution-spec.org= point at it and need updating in the same move. +- *Fires on approval:* rename+move the 3 files flat into their permanent home, update the spec's 3 links, delete the empty working subdir. + +** [Fire 1 · P8] Resolve ~4 candidate broken file: links in live docs (report-only) +- *What:* fix or confirm-intentional the broken links P8 flagged. +- *Why:* report-only pass; no unattended rewrites. The clearest real one: =docs/specs/2026-06-16-autonomous-batch-execution-spec.org= links =../../.ai/workflows/inbox-zero.org= (renamed to inbox.org). Others (=wrapup-routing-spec-review.org=, a pattern-catalog inbox source, =subprojects-log.org=) may be folded-in/deleted review docs — Craig's call. +- *Fires on approval:* update the inbox-zero.org→inbox.org link; decide the rest. + +** 2026-07-20 Mon @ 15:28:41 -0500 — flushed (auto) +Auto-flush checkpoint mid-session. Session's shipped work is all committed + pushed (main == origin == velox). In flight: resuming into finishing the Signal pager task ([#B] in todo.org) — the runbook, velox receive timer, and the ssh-vs-linked-device decision. See Summary → Next Steps and Data Collected for the reconciled facts needed to resume blind. + +** 2026-07-20 Mon @ 16:48 CDT — Notification vocabulary split (page/text) + agent-page → agent-text rename +Craig's follow-on from the reply-ambiguity discussion: reserve trigger words by channel. "page me" = desktop notify, "text me" = Signal, "text and page me" = both. Renamed the tool agent-page → agent-text to match (deprecated agent-page shim delegates to it, removable later). Rewrote protocols.org "Paging Craig" → "Reaching Craig", page-me.org, work-the-backlog's away-run logic, INDEX, and the runbook; updated install-ai doc comments; bats renamed agent-page.bats → agent-text.bats (5 pass incl. shim delegation). /review-code + /voice ran; make test exit 0, shellcheck clean, no new lint. Commit 6145489, pushed; make install re-run on both ratio + velox, agent-text + shim resolve on both. Vocabulary decision arc (this session): first "page=Signal, notify=desktop" (rejected, inverted current default), then "page=desktop, message=Signal" (Craig's), then settled on "text me" for Signal (agent-text). NOTE excluded from staging: .claude/settings.json shows model opus→fable (not mine — left unstaged). + +** 2026-07-20 Mon @ 23:36 CDT — Post-flush resume: voice #46 handoff + triage phone-push (item 1) +Resumed from the auto-flush anchor. Two things landed. + +Inbox handoff (archsetup, voice pattern #46 comma budget): archsetup edited voice/SKILL.md + voice-profile.org through the shared ~/.claude/skills/voice symlink (points into rulesets) at Craig's 2026-07-20 direction — "no more than two commas per sentence" in the /voice personal pass. Reviewed both files (count 45→46 everywhere, personal-only set four→five, #46 in the attestation high-recurrence set, before/after from the Hyprland draft). /review-code approved, /voice personal on the message, make test green. Committed 2ea5d9a; replied to archsetup; marked PROCESSED (chore 9721a49). The .claude/settings.json model change (opus→fable) was left unstaged — not this session's work. + +Item 1 — triage-intake auto-mode phone-push, built. Folded a "Phone delivery" subsection into canonical triage-intake.org: a full-three-section sweep pushes to Craig's phone over Signal via agent-text, with a pointer from End-of-sweep output and a Living Document note. Craig's design ruling (option 1, signal-only): a quiet sweep's "nothing" heartbeat never reaches the phone — silent-until-signal governs the phone channel too — while the in-session heartbeat stays as proof-of-life. Send half only; reply-polling (old phone-recv) deferred to the reply-correlation follow-up (item 2), which owns the recv wiring. sync-check --fix synced the mirror; make test green; /review-code approved; /voice personal. Committed e27aea2. todo.org task closed DONE (send half shipped, recv deferred). + +Next: items 2 (reply-correlation spec) and 3 (token-rotation discussion) remain untouched — carried to the next session. + +** 2026-07-20 Mon @ 16:14 CDT — Signal pager DONE (all four deliverables + topology decision) +Finished the task end to end. (1) Runbook: docs/design/2026-07-20-signal-pager-runbook.org — send, read-replies, receive timer, signal-cli account/setup notes, the resolved decision. (2) Receive timer: scripts/signal-receive.sh + scripts/systemd/signal-receive.{service,timer} (roam-sync-shaped, 15-min, no-ops where the account isn't local, 4 bats); stowed via dotfiles common; enabled + verified on both ratio and velox; a manual drain cleared the 47-day staleness live. (3) Topology decision (Craig, option 2 — linked device over ssh-relay-only): ratio linked as Device 2 "ratio-pager", direct send verified; agent-page generalized from a velox-only check to "any machine holding the account sends directly, else relay" (bats updated). (4) protocols.org "Paging Craig": verified accurate, no edit. Publish flow: /review-code caught a runbook/script contradiction (runbook claimed no-op-clean; script errored) → added the account-presence guard + a bats case. make test green, shellcheck clean. Commits: rulesets 302b062 (feat(pager)), dotfiles d8b0462 (feat(systemd)); both pushed; velox synced. todo.org task closed DONE. Craig's follow-up worry addressed: the pager only fires on an explicit page-me or an away-run's end-of-set page; the receive timer is silent (receive-only, no push). diff --git a/.ai/sessions/2026-07-24-18-05-sentry-implement-pass-review-and-hook-fixes.org b/.ai/sessions/2026-07-24-18-05-sentry-implement-pass-review-and-hook-fixes.org new file mode 100644 index 0000000..6c60a8f --- /dev/null +++ b/.ai/sessions/2026-07-24-18-05-sentry-implement-pass-review-and-hook-fixes.org @@ -0,0 +1,386 @@ +#+TITLE: Session Context — 2026-07-23 +#+AUTHOR: Craig Jennings +#+DATE: 2026-07-23 + +* Summary + +** Active Goal + +A long session (2026-07-23 into 2026-07-24) spanning several arcs: applied two Craig-ordered sentry amendments, ran a no-approvals speedrun over four solo tasks, shipped the sentry implement-pass feature itself, ran eight overnight sentry fires that found and fixed real bugs, then subjected the whole night's output to an adversarial review that found defects in the fixes, repaired those, absorbed a repo-wide fail-open security fix from .emacs.d, and processed the inbox to zero. Ended with the branch merged to main (unpushed by Craig's choice) and the session wrapped. + +** Decisions + +- Sentry gains an opt-in solo-implementation pass (pass 12, gated on =:SENTRY_MAY_IMPLEMENT:=, separate from =:COMMIT_AUTONOMY:=) plus refactor-finding in pass 11. Craig's direction, after the discussion that the branch already contains blast radius and a skeptical premise-first review is the fact-checker that makes fixing-on-a-branch safe. Shipped to main. +- Reviews must fact-check the *premise* (reproduce the bug) before judging the diff, not just check the diff is clean. Craig's correction; saved as harness memory =feedback-reviews-verify-premise=. Across the night the premise check killed roughly one wrong hypothesis per real bug. +- Craig chose NOT to push main at wrap. The hook fail-open fix therefore stays undelivered to consuming projects until he pushes. Flagged and reaffirmed. + +** Data Collected / Findings + +- The dominant defect class across the session, five instances over two projects: a quality gate that enumerates its inputs instead of discovering them, so a new input is silently skipped and the green check reads as covered. Promoted to a KB node this wrap. +- The secret-scan pre-commit hook failed open on any git error, in ALL FIVE language bundles (not just the elisp one .emacs.d reported). Two of the five were hooks I wrote the day before by copying bash — I propagated the defect. Fixed across all eleven sites; graded [#A]. +- The adversarial review round found four real defects in my overnight fixes (mode-widening and symlink-clobbering in cj-remove-block's atomic write, a same-second backup collision in two tools, a test that deleted real backups from shared /tmp) plus one I'd left: the cj-block range check still can't prove it's deleting the block that was scanned. All repaired except the last, which needs a CLI-contract decision and is filed [#B]. +- I repeated my own worst mistake pattern three times: shipping a change whose correctness depended on shared /tmp state (the backup tests), and twice concluding causation from a single-sample measurement (the audit flake A/B, the /tmp-copy comparison). The re-run/isolation habit caught each. + +** Files Modified + +- Merged to main (267d1de): cj-remove-block range guard + atomic write, todo-cleanup backup, route_recommend dedupe, audit.bats flake fix, lint.sh bin/ coverage, plus the review-round repairs. +- Hook fail-open fix (f0c1bc4): all five bundles' pre-commit, the elisp validate-el cap removal, the cross-bundle test now discovering variants, two adopted .emacs.d bats suites. +- Earlier: sentry.org pass 11/12 + marker (pushed), the speedrun's four fixes, the two .dotfiles amendments, voice #47, four approved parked proposals. +- KB: =agents/20260724180443-enumerate-vs-discover-gate-failure.org=. + +** Next Steps + +- Push main (5+ commits ahead, all local). The hook security fix is the load-bearing one. +- The cj-block wrong-block design question: content assertion vs re-scan vs bottom-up removal. Filed [#B] at the top of todo.org. +- Seven parked VERIFYs await Craig, [#A] account-binding guard from home first, then the telegram down-is-launch fix and its engine sibling. +- Optional: ~1600 backup files accumulated in /tmp from the night's runs (harmless, cleared on reboot). + +KB: promoted 1 / consulted no + +* Session Log + +** 02:34 — Startup + +Ran startup.org. Rulesets already current; =make install= had nothing new to link; project repo clean and up to date. =.ai/= synced from templates. No prior =session-context.org= — last session (2026-07-20 23:36, signal-pager + triage phone-push) wrapped cleanly. + +Findings surfaced: 13 top-level tasks unreviewed >7 days; roam inbox holds 9 items; 8 unprocessed project inbox handoffs plus =inbox/lint-followups.org=; KB has 101 =:agent:= nodes but no best-practices node resolved and nothing matching "rulesets". Active Reminder from 2026-07-14 (Craig's deep read of the sentry spec) reads stale — sentry has since shipped and dogfooded live in takuzu and archangel. + +Craig's instruction on arrival: finish startup, then begin sentry. + +** 02:35 — Craig-ordered amendments applied before launching sentry + +Two of the eight inbox handoffs are Craig's own orders relayed from the dotfiles session, and both gate a correct sentry run, so they land before the loop starts. The rest of the inbox stays for sentry's own inbox pass. + +Amendment 1 — sentry pass list (=.dotfiles=, 2026-07-21). Sentry never checks email or messengers, and gains a bug-finding pass. The handoff shipped dotfiles' whole edited =sentry.org=, but that copy forked from a pre-silent-until-signal canonical: rsyncing it would have reverted the heartbeat/digest split committed 2026-07-20. Applied their three intended changes onto current canonical by hand instead, and kept the =:TRIAGE_SOURCES:= activation-probe language their Pass 3 rewrite had dropped. + +Reviewed the staged diff before committing and found three defects in my own edit. Pass 11 as Craig worded it ran "the project's linters and suite," which would have violated sentry's own anti-pattern 5 (no per-pass suite run) eleven times a night; it now runs linters and static analysis only and reads the entry baseline's suite result. The Living Document line still said "ten mechanical passes." And the KB-deferral parenthetical called KB promotion "the proposal's eleventh pass," which collided with the real pass 11 — reworded. Suite green (exit 0) before and after. Committed 33949c5, unpushed pending Craig's call. + +** 02:43 — Sentry armed, fire 1 (working) + +Entry gates all passed: =:COMMIT_AUTONOMY: yes=, tracked tree clean, suite green (exit 0, run twice on this content), no prior =sentry/*= branch, zero behind upstream. Branch =sentry/2026-07-23-ratio= created from HEAD. Loop armed hourly at :07 (job ef8eb9d4, session-only, auto-expires in 7 days). Craig's repo working tree belongs to sentry until the morning merge — reclaiming it early means saying "stop sentry". If he has rulesets files open in Emacs, buffers want reverting after the merge. + +Fire 1 digest: + +- P1 roam pull — ran, already up to date. +- P2 inbox zero — ran. Project inbox: 1 of 5 executed (org-drill's memory-sweep completion note, informational, marked PROCESSED + reply sent). 4 park (below). Roam inbox: 9 items, 8 belong to other projects and route at wrap-up, not here; the 1 rulesets item ("every project should have a working and a temp directory") is shipped except for one clause, queued below. +- P3 triage intake — skipped: no project-specific plugin and no =:TRIAGE_SOURCES:= declaration, so no active source. Correct behavior under the activation gate. +- P4 todo cleanup — ran, no-op. +- P5 task audit — deferred to a later fire; takuzu's dogfood note says the full audit is too heavy hourly, and fire 1 already carries the bug-finding load. +- P6 working-files hygiene — skipped: no =working/= directory. +- P7 spec status board — ran. No =DOING= spec with a closed build parent. +- P8 link integrity — ran, and the checker itself turned out to be broken (see P11). +- P9 git health — ran. =main= is 1 ahead of =origin/main= (the amendments commit, unpushed pending Craig's call). Sentry branch correctly has no upstream. No stale branches. +- P10 prep freshness — skipped: no =daily-prep/=. +- P11 bug finding — ran. Rotating area this fire: =lint-org.el= and the synced =.ai/= templates. Three defects verified and filed as graded tasks (commit dc7791d): link resolution against cwd rather than the linted file's directory [#B], todo-format checkers firing on spec files [#C], and the notes.org template tripping four flags in every project [#C]. Two of the three were independently reported by takuzu and smoke tonight, which is what prompted checking them. + +The three bug tasks record their severity x frequency arithmetic in the body, including where the two inputs disagreed, so the grades can be argued rather than just overridden. + +** 02:47 — Finished the park path the fire skipped + +The inbox-boundary Stop hook caught a real gap. Fire 1 wrote the four proposals into the approval queue and stopped there, but =inbox.org='s park path is three things, not one: move the proposal into =working/<slug>/= beside a prepared diff, file a =[#B]= VERIFY carrying the decision package, and reply to the sender. I'd done the queue entry and none of the rest, so the senders were waiting on nothing and the decisions lived only in a session anchor that gets archived. + +Completed all four. Each =working/= dir now holds the original proposal, a =proposed.diff=, and the full proposed file. Both org-file diffs were verified by linting the proposed version: the notes.org template goes from four flags to zero, and the fix turned out to need one thing smoke didn't identify — the =invalid-block= pair isn't the example content confusing the parser generally, it's the literal =** Feature Name or Topic= line inside the block, which org reads as a heading. A comma-escape on that one line clears both. Also found two more column-0 bold lines beyond the two reported. + +Replies sent to takuzu, archangel, and smoke. dotfiles' ack needed nothing back and is marked PROCESSED. Inbox is at zero. + +** 03:35 — Fire 2 (working) + +Lock acquired, branch state verified. Digest: + +- P1 roam pull — ran, already up to date. +- P2 inbox zero — ran, no-op. Project inbox at zero. Roam inbox unchanged at 9; 8 belong to other projects and route at wrap-up, the 1 rulesets item is already queued. +- P3 triage intake — skipped: no active source surviving the mail/messenger exclusion. +- P4 todo cleanup — ran in check mode across all three passes, zero fixes, zero conversions, zero archives. +- P5 task audit — mechanical subset only, per the parked guidance and takuzu's note that the full audit is too heavy hourly. Staleness is 20, up from 13 at startup, which is just tonight's 7 new tasks arriving never-reviewed. Not a defect. +- P6 working-files hygiene — ran, and this fire is the first where its probe fires, since fire 1's park work created =working/=. All four dirs have open backing VERIFY tasks, so nothing to flag. The pass works. +- P7 spec status board — ran. Seven IMPLEMENTED, two READY (inbox-workflow-consolidation, encourage-kb-contribution). No DOING spec with a closed build parent. +- P8 link integrity — ran across todo.org, notes.org, the anchor, and all nine specs. Two findings, both in the docs-lifecycle spec, both prose containing a bare =file:= that org parses as a bracketless link (=file:→id:= in a sentence about converting link types, and =keep-file:-links-through-pilot= in a hyphenated phrase). Org behaving as documented rather than a defect, so a digest line and no task. +- P9 git health — main still 1 ahead of origin, unpushed, awaiting the approval-queue item. Sentry branch correctly has no upstream. No stale branches. +- P10 prep freshness — skipped: no =daily-prep/=. +- P11 bug finding — rotating area this fire: the =.ai/scripts/= shell helpers. Shellcheck clean on all seven of the sentry-critical ones (agent-lock, agent-roster, capture-guard, flashcard-sync, inbox-status, self-inject.sh, session-context-path). One SC2034 pair in task-review-staleness.sh, verified as a false positive — the two names are positional fields in a =read -r= that exist to put =value= in the right slot. Digest line, not a task. + +*** Retraction: the [#B] I filed in fire 1 was wrong + +The main result of this fire is negative. Fire 1 filed a =[#B]= claiming lint-org resolves =file:= links against the process cwd rather than the linted file's directory. It doesn't, and the task is CANCELLED. + +The claimed evidence was that linting the notes.org template from the repo root reports two siblings missing while linting from its own directory doesn't. The first half was never run. Those findings came from the =/tmp= copy I made while preparing smoke's diff, and in =/tmp= the siblings genuinely are missing, so org-lint was right. I compared two different files and read the difference as a bug, then wrote a fix direction for a mechanism I hadn't checked. + +Caught it here only because P8 surfaced =link-to-local-file= again and I went looking for the checker in lint-org.el to fix it. It isn't there — it's org-lint's own, which contradicted my stated fix and forced the retest. Three runs plus a direct =default-directory= probe settled it. + +Correction sent to takuzu, who had been told about it. Their own finding is unaffected and still filed. The useful residue: both =link-to-local-file= and the todo-format checkers are upstream org-lint checkers, so scoping them means filtering org-lint's output, not editing a local checker — which changes the fix direction on the [#C] task too. + +** 04:35 — Fire 3 (working) + +Lock acquired, branch state clean. Digest: + +- P1 roam pull — ran, already up to date. +- P2 inbox zero — ran, no-op. Both inboxes unchanged. +- P3 triage intake — skipped: no active source surviving the mail/messenger exclusion. +- P4 todo cleanup — ran. Hygiene and convert-subtasks no-op; =--archive-done= moved one subtree, the =CANCELLED= lint-org task fire 2 retracted. Committed 962f3c0. +- P5 task audit — mechanical subset. Staleness 19, down one from the archive. +- P6 working-files hygiene — ran. All four =working/= dirs still have open backing VERIFY tasks. Nothing to flag. +- P7 spec status board — ran. Two READY specs, no DOING with a closed parent. +- P8 link integrity — ran across todo.org, notes.org, and all nine specs. Same two known prose false positives in the docs-lifecycle spec, unchanged. No new findings. +- P9 git health — main 1 ahead of origin, unpushed, still queued. No stale branches. +- P10 prep freshness — skipped: no =daily-prep/=, no broken symlinks. +- P11 bug finding — rotating area this fire: the Python scripts under =.ai/scripts/=. All 15 compile; no Python linter on this box, so the pass became a targeted read of =inbox-send.py=, the script this session has exercised hardest. Three defects verified, filed as two tasks (8995016). + +*** The inbox-send finding + +The one that matters: =inbox-send= writes straight to the destination path, and =write_text= truncates on open, so any mid-write failure leaves a zero-byte =.org= in *another project's* inbox. That phantom isn't inert — =inbox-status= counts it, so it trips the receiving project's boundary hook and blocks a turn there over a file with no content and no sender context. Meanwhile the sender saw an error and retries, so the target collects a second one. + +Reproduced end to end with a non-ASCII message under a C locale with UTF-8 mode disabled: send fails, zero-byte file lands in the destination, =inbox-status= there reports it pending. The encoding case is just the trigger I could force; the defect is the non-atomic write, which a full disk or an interrupted process reaches the same way. Graded [#B], fix direction is temp-file-plus-=os.replace= with =encoding="utf-8"= pinned. + +Two smaller ones grouped as [#C]: =send_file= raises an uncaught =PermissionError= traceback because =main= catches only =ValueError= and =FileNotFoundError=, and =discover_projects= doesn't dedupe, so a roots config naming both a parent and its child lists the same project at two indices. + +Notable that this is the first fire where the rotating area produced findings in code the night's own work depended on. Fire 1 read the linter, fire 2 the shell helpers, fire 3 the script that carried every reply I sent. + +** 05:35 — Fire 4 (working) + +Passes 1 through 10 all quiet: roam already current, both inboxes unchanged, todo cleanup no-op across all three passes, all four =working/= dirs still backed by open VERIFY tasks, two READY specs and nothing stuck, the same two known prose false positives in the docs-lifecycle spec, main still 1 ahead and unpushed, no prep dir and no broken symlinks. Staleness 21, up two from fire 3's new tasks. + +P11 rotating area: =claude-templates/bin/= — the launcher and paging scripts. Shellcheck clean on all four. The findings came from reading and from a config-sanity check, and one of them is a live condition on this machine rather than a latent code defect. + +*** ratio's second Signal account has been cold for 8 days + +=signal-cli listAccounts= on ratio warns that messages were last received 8 days ago. The natural read is that the pager channel has gone stale, which would matter — it's the "text me" path. It hasn't. Established by elimination: ratio holds two accounts, =signal-receive.sh= hardcodes the pager (=+15045173983=) as its only target, and the timer's own journal shows it draining that account cleanly every 15 minutes, the last run 2 minutes before the warning printed. So the cold account is Craig's personal number, and nothing on this machine keeps it warm. + +What I can't establish is whether that matters. The session history describes that registration as note-to-self with no push, and Signal's tolerance for a quiet linked device isn't something I verified. Filed as [#C] with the uncertainty stated rather than graded up on a guess. If it does matter, the fix is a second timer instance rather than a code change, since =signal-receive.sh= already takes an account argument — which makes it a dotfiles handoff, that repo owning the unit. Not sent: the content is speculative and a 05:35 handoff asserting a problem I haven't confirmed would land as noise. + +*** agent-text's direct send is unbounded + +The relay path passes =ConnectTimeout=10= to ssh; the local-account path calls =signal-cli send= with no bound. Since agent-text is invoked by agents, a stall blocks the calling turn with no output and no way to tell a hang from a slow send. Not hypothetical contention — the receive timer holds the same account for ~16 seconds every cadence. Filed in the same [#C]. + +** 06:35 — Fire 5 (working, but the bug hunt came up empty) + +Passes 1 through 10 all no-op: roam current, both inboxes unchanged, todo cleanup clean across all three passes, four =working/= dirs still backed, two READY specs, the same two known prose false positives, main 1 ahead unpushed, no broken symlinks. Staleness 22, up one from fire 4's new task. + +P11 rotating area: =hooks/= and =scripts/= — the machine-wide hooks and the repo's own install and maintenance scripts, the last major uncovered surface. *Zero real bugs.* All five shell hooks shellcheck clean, all Python hooks compile, =hooks/tests= present, =hooks/__pycache__= correctly gitignored with nothing tracked. Six shellcheck findings across =scripts/=, every one dispositioned as a non-bug: + +- =SC2094= (read and write the same file in one pipeline) in =install-ai.sh= and =sweep-gitignore-tooling.sh= — false positive both times. The pattern is a =>>= append with a =[ -s "$gi" ]= stat inside the group. Appending doesn't truncate, and a stat isn't a content read, so the leading-blank-line logic is correct in both the file-exists and file-absent cases. +- =SC2088= (tilde doesn't expand in quotes) in =doctor.sh= and =audit.sh= — both are display strings printed to the user, not paths used for I/O. =audit.sh= says so in a comment on the line above. +- =SC2295= (unquoted expansion inside =${..}=) in =audit.sh= and =diff-lang.sh= — technically correct, no live trigger; =$HOME= carries no glob characters. +- =SC2164= (=cd= without =|| exit=) in =lint.sh= and =status.sh= — =status.sh= already guards its path upstream. =lint.sh= is genuinely unguarded and, since it sets =-u= but not =-e=, a failed =cd= would let it lint the invocation directory instead of the repo root. Confirmed =cd ""= fails rather than silently succeeding, so the hazard is real in shape but needs the running script's own parent to be unreachable. Not reachable in practice; a one-line =|| exit= would close it if anyone touches the file. + +This is the first fire whose hunt found nothing, which is the expected shape — takuzu's dogfood report predicted the rotating hunt goes quiet after the first few fires clear the standing defects. Recording the area covered so the next fire doesn't re-read it. + +*** Two measurement errors this fire, both caught before they became findings + +Worth flagging as a pattern, since fire 1's retraction was the same class. First, a probe loop passed =--convert-subtasks --check= through an unquoted variable; zsh doesn't word-split, so it arrived as one argument and =todo-cleanup= printed a bare =normal-top-level()=. That looked like a tool crash and was my loop. Protocols warns about exactly this. + +Second, chasing that, I read =exit=0= from a bad-flag run and nearly filed a silent-pass defect — a cleanup tool that exits 0 on a bad flag would let =wrap-it-up= believe a pass ran when it didn't. The =0= was =tail='s exit code, not emacs'. Measured without the pipeline, both =todo-cleanup= and =lint-org= exit 255 on an unknown flag and on a missing file, and 0 on success. No defect. + +Three times tonight a conclusion came from a bad measurement. Twice it was caught in the same fire; once (fire 1) it reached a filed task and a sent handoff before the retest caught it. + +** 07:35 — Fire 6 (working) — the night's most consequential finding + +Passes 1 through 10 all no-op again: roam current, inboxes unchanged, todo cleanup clean, working dirs backed, two READY specs, the same two prose false positives, main 1 ahead unpushed, no broken symlinks. Staleness 22, flat. + +P11 rotating area: =languages/= and the bundle install machinery — the last major uncovered surface. Not quiet. + +*** The python and typescript bundles ship no secret-scan hook + +=bash=, =elisp=, and =go= each carry =githooks/pre-commit=, =claude/hooks/validate-*.sh=, =claude/settings.json=, and a seed =CLAUDE.md=. =python= and =typescript= carry none of the four. The pre-commit hook is the credential scanner, so installing either of those bundles gives a project no secret scan on commit — while README's Bundle structure section documents all four as what every bundle follows. + +Verified live rather than inferred, by scanning every project with =.claude/rules/=: =work= (python) and =clock-panel= (python + typescript) both have no =githooks/= and no settings. All four elisp projects have both. =work= is the one that matters — a work repo is where a leaked credential costs most and is likeliest to reach a company remote. + +The history settles intent. Both bundles were added 2026-05-31; =go='s githooks landed 2026-06-02 and =bash='s 2026-06-23. The rollout swept the bundles added *after* these two and skipped these two. And =install-lang.sh= guards its copy with =[ -d "$SRC/githooks" ]=, so the install succeeds and reports nothing missing, which is how this stayed invisible for almost two months. + +Filed [#A], SCHEDULED today — the only [#A] of the night. The fix is find-not-fix per the pass contract, so nothing was changed. The task carries a second half worth as much as the port itself: make =install-lang= warn when a bundle lacks a component the README documents, so the next partial bundle announces itself rather than installing quietly. + +*** Same drift shape, smaller + +=bash='s pre-commit has =cd "$REPO_ROOT" || exit 1=; the =go= and =elisp= copies of that line dropped the guard. Impact is genuinely low — git chdirs to the working-tree root before running a hook, so the checks run against the right tree regardless, and neither script sets =-e=. Filed [#C] for the two-character fix, mostly because the shape is the same as the [#A]: a fix landed in one bundle copy and stopped there. + +Two fires' worth of evidence now says bundle-to-bundle propagation is where this repo leaks changes. + +** 00:51 — Fire 1 of the 2026-07-24 run (working) — first fire with pass 12 live + +Lock acquired, branch =sentry/2026-07-24-ratio= verified, tree clean outside the spine. Digest: + +- P1 roam pull — ran, already up to date. +- P2 inbox zero — ran, no-op. Inbox at zero (the question-capture proposal was parked during entry). +- P3 triage intake — skipped: no active source surviving the mail/messenger exclusion. +- P4 todo cleanup — ran with real work. =--archive-done= moved three completed speedrun tasks out of Open Work into Resolved; =--convert-subtasks= normalized the tree. Committed. +- P5 task audit — mechanical subset. Staleness 13, back to the pre-speedrun baseline now that the four solo tasks closed. +- P6 working-files hygiene — ran. All =working/= dirs still have open backing VERIFY tasks (the parked proposals). No orphans. +- P7 spec status board — ran. Two READY specs, nothing stuck. +- P8 link integrity — ran. The docs-lifecycle spec still shows its two known prose false positives (=file:→id:= and =keep-file:-links-through-pilot=, bare =file:= tokens org parses as bracketless links). Correctly unaffected by tonight's spec-scoping, since =link-to-local-file= is org-lint's own checker, not a todo-format one. No new findings. +- P9 git health — main level with origin, sentry branch correctly has no upstream, no stale branches. +- P10 prep freshness — skipped: no =daily-prep/=. +- P11 bug and refactor finding — ran, first fire under the widened pass. Rotating area: the cross-project routing scripts (=route_recommend.py=, =broadcast.py=), untouched by the previous six areas. One verified latent bug filed [#D] (below). Refactor note not worth a task: =recommend='s two weak-tier branches collapse to a single =_tiebreak= call, since =_tiebreak= on a one-element list returns that element. Two lines, no behavior change, so it's a digest line rather than backlog noise. +- P12 solo-task implementation — *active this run* (=:SENTRY_MAY_IMPLEMENT: yes=) but a correct no-op: zero eligible tasks. All four open =:solo:= TODOs were completed in tonight's speedrun, so the ready bucket is empty. Nothing to implement, nothing deferred. + +*** The finding + +=route_recommend='s =discover_destination_names= collapses projects to bare basenames, so two projects sharing a basename across roots would both literal-match and read as an ambiguous tie, downgrading a correct strong match to weak. Reproduced by direct probe. Latent rather than live: 27 projects, 27 distinct basenames today. The destination stays right, only the tier is wrong, so the cost is an extra routing prompt. Filed [#D] with the order-preserving dedupe as the fix. + +** 01:33 — Fire 2 (working) — pass 12's first real implementations + +Digest: + +- P1 roam pull — ran, already up to date. +- P2 inbox zero — ran. Project inbox at zero. The roam inbox dropped 9 → 5 (another session filed some); all 5 remaining belong to archsetup or work, none rulesets-claimed, so roam mode is correctly a no-op here. +- P3 triage intake — skipped: no active source. +- P4 todo cleanup — ran, all three checks clean (fire 1 did the archiving). +- P5 task audit — mechanical subset. Staleness 13, flat. +- P6 working-files hygiene — ran. All =working/= dirs still backed by open VERIFYs. +- P7 spec status board — ran. Two READY specs, nothing stuck. +- P8 link integrity — ran. Only the docs-lifecycle spec's two known prose false positives. +- P9 git health — main level with origin, sentry branch correctly upstream-less. +- P10 prep freshness — skipped: no =daily-prep/=. +- P11 bug and refactor finding — rotating area: the cj-comment tooling (=cj-scan.py=, =cj-remove-block.py=). Two findings filed, one of them serious. +- P12 solo-task implementation — *two tasks implemented and committed to the branch* (17f5d48, 1b0f284). Both premise-checked before a line was written. + +*** The serious find: cj-remove-block destroys content + +=looks_like_cj_range= validated only the first and last lines of a range. A span from one cj block's opener to a *later* block's closer passed, and the removal then deleted everything between — prose, headings, whole tasks — silently, exit 0. That is exactly the failure the check exists to prevent, and drift is its normal case, since =respond-to-cj-comments= edits the file while processing and a file under cj review usually holds several blocks. Reproduced on a two-block fixture that lost a heading and two content lines. Filed [#B], then implemented in pass 12 after an independent re-verification on a different fixture shape. + +Same file, second defect fixed alongside: =remove_range= rewrote the org file with a bare =write_text= (truncates on open) and took no backup, so a mid-write failure would leave =todo.org= truncated with nothing to recover from. =lint-org.el= already backs these files up to =/tmp= before mutating; cj-remove-block now matches that and writes atomically. + +*** The measurement lesson, third time tonight + +A full-suite run went red on =audit.bats= test 4. I stashed my changes, saw it pass clean, and had a one-sample A/B pointing straight at my own diff. Re-ran three times with the changes restored and it passed every time. The failure is an intermittent teardown flake (=rm -rf= racing something still writing into a fixture =.git/objects=), not my change, and I nearly filed the wrong cause off a single sample. Filed [#C] with the git-background-gc theory explicitly labelled a lead rather than a verified cause. + +Committed only on a genuinely green re-run, not on the red with a hand-wave. + +** 02:33 — Fire 3 (working) — the flake's cause traced, and my own lead disproved + +Digest: + +- P1 roam pull — ran, already up to date. +- P2 inbox zero — ran, no-op. Project inbox at zero; roam holds 5, none rulesets-claimed. +- P3 triage intake — skipped: no active source. +- P4 todo cleanup — ran. =--archive-done= moved fire 2's two completed tasks to Resolved. Committed. +- P5 task audit — mechanical subset. Staleness 13, flat. +- P6 working-files hygiene — ran, all dirs backed. +- P7 spec status board — ran, two READY, nothing stuck. +- P8 link integrity — ran, only the known docs-lifecycle prose false positives. +- P9 git health — main level with origin, branch upstream-less, nothing stale. +- P10 prep freshness — skipped. +- P11 bug and refactor finding — no new findings. The fire's whole investigative budget went to confirming fire 2's filed flake, which is the honest place for it; a hunt that finds nothing new is a result. +- P12 solo-task implementation — one task implemented and committed (7f45d4b). + +*** Confirming the cause before fixing, and disproving my own lead + +Fire 2 filed the =audit.bats= flaky teardown with a stated theory: =gc.auto='s loose-object threshold. Pass 12's premise check went after that theory rather than the fix, and killed it — a fixture holds five objects against a default threshold of 6700, so that mechanism cannot fire. + +The real cause came from a =GIT_TRACE= run: =git commit= spawns =git maintenance run --auto --quiet --detach= on git 2.55. The commit returns while the detached process is still writing a pack, and teardown's =rm -rf= races it. Every failed run had left a =tmp_pack_*= behind, which is the thread that led there. + +Fix: =maintenance.auto false= plus =gc.auto 0= in the fixture, killing the background writer rather than retrying the delete (a retry loop hides a live process instead of removing it). Validated over 20 consecutive clean runs against a ~1-in-8 baseline, and recorded honestly in the task that 20 clean runs alone would be ~7% likely by luck — the trace is the evidence, the runs confirm. + +This is the second night running where the premise check changed the outcome. Fire 2 it stopped a wrong causation call; here it stopped me implementing a fix for a mechanism that was never operating. Both times the cost of checking was minutes and the cost of not checking would have been a plausible, wrong, committed change. + +** 03:33 — Fire 4 (working) — two hypotheses killed, one real gap closed + +Digest: + +- P1 roam pull — ran, up to date. P2 inbox zero — no-op, both surfaces clean. P3 triage — skipped, no active source. +- P4 todo cleanup — ran. Archived fire 3's completed task to Resolved. Committed. +- P5 task audit — mechanical subset. Staleness 13, flat all night. +- P6 working-files — all dirs backed. P7 spec board — two READY, nothing stuck. P8 link integrity — only the known prose false positives. P9 git health — clean. P10 — skipped, no prep dir. +- P11 bug and refactor finding — rotating area: the org-file mutators, chosen because =cj-remove-block= yielded a serious find in that same class last fire. One gap filed. +- P12 solo-task implementation — one task implemented and committed (0686784). + +*** What the area review actually found, and what it disproved + +The lead was that =todo-cleanup.el= might lose data. It rewrites =todo.org=, creates archive files, and moves subtrees *between* files, which is the shape that bit =cj-remove-block=. Two specific hypotheses, both tested, both wrong: + +- *"A mid-move failure loses a subtree from both files."* No. The order is delete-from-buffer, write-archive, save-todo.org-last, so an archive-write failure aborts before the save. Verified by making the archive directory unwritable: exit 255, =todo.org= byte-identical, content intact. +- *"Errors are swallowed on the mutation path."* No. The only =ignore-errors= in the file wrap =call-process "git"=, never a write. + +What survived was narrower and real: todo-cleanup mutates with *no backup*, while both sibling mutators (=lint-org.el=, =wrap-org-table.el=) copy to =/tmp= first, and =cj-remove-block= joined them last fire. It is also the one that runs most often. Confirmed empirically that Emacs's own backup does not fire under =--batch -q=, so there was genuinely no undo short of git. Filed [#C], then implemented in P12. + +Also checked my own change for regression rather than assuming: a missing input file exits 255 and creates nothing, identical to the pre-change version tested from git. + +*** Running tally on the premise habit + +Three fires, three times it changed the outcome. Fire 2 it stopped a wrong causation call. Fire 3 it disproved my own filed =gc.auto= theory before I could implement against it. Here it killed two data-loss hypotheses before they became tasks, leaving only the gap that was actually there. The pattern is consistent: the cheap check keeps a plausible story from becoming a committed change. + +** 04:33 — Fire 5 (working) — the first real defer, and last fire's fix proving itself + +Digest: + +- P1 roam pull — ran, up to date. P2 inbox zero — no-op, both surfaces clean. P3 triage — skipped, no active source. +- P4 todo cleanup — ran, archived fire 4's completed task. *Confirmed last fire's backup fix working live*: the real =--archive-done= run left =/tmp/todo.org.before-todo-cleanup.20260724-043331=. Dogfooded within an hour of shipping. +- P5 task audit — staleness 13, flat all night. P6 working-files — all backed. P7 spec board — two READY. P8 link integrity — only the known prose false positives. P9 git health — clean. P10 — skipped. +- P11 bug and refactor finding — rotating area: the attachment/email handlers, chosen because they parse genuinely untrusted input, unlike every internal-tooling area covered so far. One finding filed. +- P12 solo-task implementation — *deferred*, and correctly. First defer of the run. + +*** The find: attachment filenames are partly sanitized, in two different ways + +Both writers derive on-disk names from the =filename= an email declares, and both sanitize incompletely, covering *different* gaps. =eml-view= cleans the name but interpolates the extension raw. =gmail-fetch='s =safe_filename= handles path separators and leading =..= and nothing else. Probed both with the same adversarial set: a =; rm -rf ~= extension survives in both, a literal newline survives in both, a 300-character extension produces a 314-character filename in both. + +Two things I checked so this doesn't get over-graded later. Not RCE — files are written through Python =open=, never a shell. Not path traversal — =splitext= only returns an extension when the last dot follows the last separator, so =ext= can never hold a slash, and the traversal case is neutralized in both scripts. The genuine harms are narrower: a newline in a filename breaks downstream tooling that reads the directory as a line-delimited list, and an unbounded extension blows the 255-byte limit so a crafted attachment aborts extraction. Graded [#C] on that honest read rather than the scarier one. + +I also corrected my own framing mid-investigation. I first read this as the familiar "one sibling hardened, the other not" pattern from the last two fires. It isn't — =safe_filename= is narrower than it looks, and the two scripts are *differently* incomplete. Neither handles newlines or length. + +*** Why pass 12 deferred instead of implementing + +The fix itself is clear, but where the shared sanitizer lives is a design call: a shared helper module (clean, but a new synced template file plus =importlib= gymnastics for kebab-named scripts), duplicate it in both (self-contained, but drift — the exact defect class that produced three separate findings tonight), or patch each in place (smallest diff, permanent divergence). That is deliberation, not a quick factual question, so checklist item 4 fires and the unattended loop defers. Filed the VERIFY with the three options and my lean, and left the task *un-=:solo:=-tagged* so a later run doesn't pick it up and guess. + +This is the checklist discriminating rather than rubber-stamping. Four fires implemented; this one correctly didn't. + +sentry at 05:33: nothing (bug hunt swept =scripts/*.py=; two candidate gaps both disproved — =workflow-integrity.py= *is* gated, its bats runs the real checker against the real canonical tree under =make test=, and =update-skills.py= is an on-demand maintenance command rather than a gate. Recorded so neither gets re-investigated.) + +sentry at 06:33: nothing (bug hunt swept =wrap-org-table.el=, the third org-file mutator and the one with prior history — its load-time dispatch caused the 2026-07-09 corruption. Clean on every probe: the entry-script guard correctly refuses to dispatch when lint-org merely =require='s it, it backs up before writing like its siblings, and it is block-aware — a table inside =#+begin_example= stayed verbatim while a real over-budget table wrapped onto continuation rows with rules. Second consecutive quiet hunt.) + +** 07:33 — Fire 8 (working) — a time bomb I planted four fires ago went off + +Digest: P1-P10 all no-op (roam current, both inboxes clean, todo cleanup clean, staleness 13, working dirs backed, two READY specs, only the known prose link false positives, git clean). P11 found a real coverage gap. P12 implemented it, and the suite caught a regression of my own making. + +*** The find: the only ungated shell in the repo + +=scripts/lint.sh= sweeps =scripts/*.sh=, the language hooks, and the language githooks. It never touched =claude-templates/bin/= — zero references. Those four scripts (=ai=, =agent-text=, =agent-page=, =install-ai=) are the ones =make install= symlinks onto PATH, which makes them the *most* exposed shell in the repo and left them the only shell with no gate over it. All four are clean today, so the guard is a no-op by design; it exists so a future regression can't pass silently. The =ai= launcher was hardened to 42 tests recently and nothing enforced that going forward. + +The test pins *coverage*, not cleanliness: it plants a broken file in each swept location and asserts lint complains, so a location that stops being swept fails the suite rather than passing quietly. + +Surfaced a bigger question I did *not* answer overnight: rulesets ships shellcheck enforcement to consuming projects (the bash bundle's pre-commit, =validate-bash.sh=) and runs none on itself. Filed as a VERIFY, because turning it on would surface the false positives dispositioned earlier this session and choosing between fixing them or adding disable-directives is a preference, not a fact. + +*** The regression: my own test, detonating on schedule + +The full suite went red on the two backup tests I added in fire 4. Not a flake — a time bomb. They globbed =/tmp/todo.org.before-todo-cleanup.*=, but the backup name derives from the file's *basename*, and the real =todo.org= shares it. So a live sentry run's genuine backup was indistinguishable from the test's own artifact, and the check-mode test (which asserts *no* backup exists) failed the moment fire 5's real archive pass created one. + +They passed when written only because no real backup existed yet. Three now sit in =/tmp=. Both tests rebind =temporary-file-directory= to a private dir, and I verified they pass with the real backups present rather than by clearing them. + +Worth naming plainly: I shipped a test whose correctness depended on the state of a shared directory that the code under test writes to in production. That is the "no shared mutable state" rule in =testing.md=, and I broke it while fixing a different durability bug. The suite caught it two fires later, which is the argument for running the full suite every fire rather than only the touched file. + +* Sentry approval queue (2026-07-23) + +Five items. Each names what, why, and the exact edit. Items 2 through 5 are also filed as =VERIFY [#B]= tasks in =todo.org= with prepared diffs under =working/=, so they survive this anchor being archived — say "approve the parked <topic>" for any of them. + +** 1. Push main to origin + +What: =git switch main && git push origin main= (then switch back, or leave main checked out if sentry is done). + +Why: commit 33949c5 (the two .dotfiles amendments) is on main and unpushed. The sentry spec called this out — an unpushed commit on main diverges across ratio and velox and breaks the next startup fast-forward. It's ahead-only, so the push is clean. Held because =commits.md= requires explicit confirmation before any push. + +** 2. interaction.md — remove the fenced-code-block carve-out (org-drill) + +What: in =claude-rules/interaction.md=, the "No Reverse-Video Highlighting in Chat Output" rule currently says fenced code blocks "are acceptable when the user explicitly wants a block to copy". Replace that sentence with a plain-text-always statement covering fences as well. + +Why: org-drill relayed Craig's 2026-05-30 direction ("always always list it out without markup"), saved there as a project memory. The rule as written contradicts it, and the directive belongs in the shared rule rather than one project's memory. It's a convention change to an always-on rule, so it parks rather than lands. + +Note: this session violated the tightened form of the rule several times already (fenced blocks in chat), which is evidence for the proposal rather than against it. + +** 3. sentry.org Living Document — fold in the two dogfood runs (takuzu, archangel) + +What: append to sentry.org's pass list and notes: (a) make the rotating-angle bug hunt an official pass — already done tonight as pass 11, so this reduces to noting takuzu's corroboration; (b) add randomized property sweeps as a sanctioned quiet-fire activity; (c) note that =todo-cleanup --archive-done= touches =.gitignore= on its first archive, so an "org-only" pass can produce a real commit and trigger the fire-end suite; (d) note that in a project gitignoring =.ai/=, quiet fires produce zero commits, so =git log main..sentry/*= understates the night and the anchor's heartbeat list is the only record; (e) replace the per-fire full task audit with a mechanical subset hourly plus judgment items queued once nightly. + +Why: two independent first-live-run reports, no engine defects in either. Item (e) matches what fire 1 did by instinct (P5 deferred). sentry.org is a shared synced asset, so the edit parks. + +** 4. notes.org template — clear the four lint flags (smoke) + +What: in =claude-templates/.ai/notes.org=, rephrase the two column-0 =**bold**= lines so neither starts with =**=, and comma-escape the =#+begin_example= block's own markers. Then =scripts/sync-check.sh --fix=. + +Why: filed as a [#C] bug task tonight with the reproduction. The edit itself parks because it changes a template every project inherits. Related finding worth Craig's attention: those two flags are mechanical, not judgment, so =lint-org --fix= run anywhere would rewrite the template and create drift against canonical. + +** 5. wrap-it-up.org — delete temp/ at wrap (from Craig's own roam capture) + +What: add a cleanup sub-step to =.ai/workflows/wrap-it-up.org= that removes the project's =temp/= contents during teardown. + +Why: Craig's roam inbox item asks for exactly this ("temp is ... deleted as a part of the wrap up sequence"). The rest of that capture shipped on 2026-07-20 (=working/= tracked from creation, =temp/= gitignored, the sweep backfilling it), but =wrap-it-up.org= has no mention of =temp/= at all, so this clause was missed. Parks as both a shared-asset edit and a destructive one. + +** 02:35 — Craig-ordered amendments applied before launching sentry (earlier) + +Amendment 2 — KB personal roots (=.dotfiles=, 2026-07-21). =~/.dotfiles= classified Unknown under =knowledge-base.md='s personal-roots list, blocking KB writes from that project twice (the sentry pass-list rule tonight, the xdg-desktop-portal gotcha on 2026-07-04). Grepped for other copies of the enumeration: =knowledge-base.md:25= is the only place that lists roots for *classification*. The other three hits (=triggers.md=, =broadcast.org=, =session-harvest.org=, =work-the-backlog.org=) enumerate roots for project *discovery*, a different question, and =~/.dotfiles= reaches those through the machine-local =~/.claude/inbox-roots.txt=. Edited the one line. diff --git a/.ai/sessions/2026-07-25-09-24-invalid-block-filter-push-task-review.org b/.ai/sessions/2026-07-25-09-24-invalid-block-filter-push-task-review.org new file mode 100644 index 0000000..d0814f9 --- /dev/null +++ b/.ai/sessions/2026-07-25-09-24-invalid-block-filter-push-task-review.org @@ -0,0 +1,83 @@ +#+TITLE: Session Context — 2026-07-24 +#+AUTHOR: Craig Jennings + +* Summary + +** Active Goal + +An evening of backlog clearing in the order Craig picked: push the delivery-blocking commits, fix the lint-org =invalid-block= false positive that home had just unblocked, then run a task-review cycle. All three finished. + +** Decisions + +- The =invalid-block= fix is a filter on org-lint's output, not a local checker edit. Home settled the open question and I re-verified both halves before acting: the string appears nowhere in =lint-org.el=, and =org-lint--checkers= enumerates it in batch Emacs. Same resolution as the earlier =link-to-local-file= episode. +- Left the =,**= comma-escape in =claude-templates/.ai/notes.org= in place. The task said the fix "lets that escape be reverted," but the escape is correct org for a literal =**= inside a verbatim block, so reverting trades correctness for nothing once the finding is suppressed. Reasoning recorded in the closed task rather than acting on the permission silently. +- Task review: all seven in the batch kept as-is, no new =:quick:= or =:solo:= tags, confirmed by Craig in one pass. +- Work's sentry loop was left alone. Craig said "sentry stop" here, but the running loop belongs to the work project (branch =sentry/2026-07-25-ratio=, firing hourly into =aiv-work:1.1=), and the stop procedure's later steps — lock release, branch disposition, approval queue — need that session's context. Surfaced rather than half-executed. + +** Data Collected / Findings + +- =invalid-block= is org-lint's own checker. Reproduced the false positive three ways: a paired example block with a heading-shaped body line flags both delimiters, a paired src block holding a literal =#+end_example= flags three lines, and a genuinely unterminated block flags once and must keep doing so. +- Verified the fix against home's live fixture in both directions: its =.ai/notes.org= produced exactly the two reported findings (lines 386 and 398) under the pre-change script and zero under the new one, file untouched. +- The uppercase-delimiter path (=#+BEGIN_EXAMPLE=) was handled but untested. Confirmed it was a real trigger — 2 findings before, 0 after — before adding the boundary test. +- Part of the task-review staleness count measures work that can't be delegated rather than work nobody read. This batch was the deliberation-heavy tail, where every task needs a decision from Craig mid-stream, which is why the speedruns kept stepping past them. +- The KB orphan task cites a 2026-07-01 snapshot of 53 agent nodes. Startup counted 104, so the KB has doubled and the snapshot is worth even less than the task body assumed. +- Five =[#D]= tasks carry no =:LAST_REVIEWED:= at all. The staleness script excludes =[#D]= but =lint-org= flags them, so the two tools disagree permanently. Left alone; needs a decision about which is right. + +** Files Modified + +- =claude-templates/.ai/scripts/lint-org.el= (+ mirror) — =lo--matched-block-regions= pairs blocks by line scan under org's real rule, memoized on the buffer modification tick; =lo--handle-item= drops an =invalid-block= finding inside a paired region. +- =claude-templates/.ai/scripts/tests/test-lint-org.el= (+ mirror) — five tests: heading-in-example, literal =#+end_example= in src, uppercase delimiters, unterminated block still reports, and one file with both proving per-block scoping. +- =todo.org= — closed the =invalid-block= task with its verification record, stamped seven review dates, inserted a missing properties drawer. +- KB: =agents/20260725093500-parser-cannot-verify-its-own-misreading.org=. + +** Next Steps + +- Work's sentry is still running on ratio and untouched. Stopping it properly means saying "sentry stop" in the work session (pane =aiv-work:1.1=), which handles the branch disposition and the overnight approval queue. +- Home is waiting on this push to re-run its =invalid-block= fixture. +- Next review batch starts with the agent-source improvements and flashcard tooling tasks. +- Seven parked VERIFYs still await Craig, the =[#A]= account-binding guard from home first. +- The sentry spec still wants Craig's deep read before the READY flip. + +KB: promoted 1 / consulted no + +* Session Log + +** Startup + +Startup ran clean: rulesets already current, =make install= had nothing new to link, project repo up to date with origin (5 commits ahead, unpushed — carried over from the prior session by Craig's choice). =.ai/= synced from templates with no churn. No crash anchor. Task staleness reported 13 top-level tasks unreviewed for >7 days. + +** Inbox — home's answer on invalid-block + +One pending handoff: home answering the open question left in the =[#C] lint-org invalid-block false-positives= task — is =invalid-block= lint-org.el's own checker or org-lint's? Home says org-lint's, and I re-verified both halves rather than taking it: =grep invalid-block= over =claude-templates/.ai/scripts/lint-org.el= returns nothing, and =org-lint--checkers= enumerates =invalid-block= in batch Emacs alongside =link-to-local-file=. So the fix is a filter on org-lint's output, matching the =link-to-local-file= episode. + +Folded into the existing task as a dated sub-entry rather than filing anything new — the task was already filed and only needed its open question closed. Also recorded home's regression fixture: its own =.ai/notes.org= PENDING DECISIONS block (lines 386-398) is left unescaped on purpose and trips both delimiters, so the filter should take those two findings to zero without touching the file. Replied to home confirming, and asked them to keep the fixture unescaped pending a ping when the filter lands. Inbox back to zero. + +** Pushed main + +Craig picked the evening's order: push, then the invalid-block fix, then a task review. Pre-push reconcile showed ahead-only by 5, no divergence, so =git push origin main= went out and verified at 0/0. The secret-scan fail-open fix is now delivered to consuming projects. + +** invalid-block filter + +Built test-first. Reproduced the failure three ways before writing anything: a paired example block with a heading-shaped body line flags both delimiters; a paired src block holding a literal =#+end_example= flags three lines; a genuinely unterminated block flags once and must keep doing so. Wrote four ERT tests covering those plus a mixed file, watched three fail for the right reason, then implemented. + +The fix is =lo--matched-block-regions=: a line scan that pairs blocks under org's real rule (once open, only the block's own =#+end_TYPE= closes it), memoized on the buffer's modification tick. =lo--handle-item= then drops an =invalid-block= finding falling inside a matched region, delimiters included, since org-lint reports at the delimiters. Line-scanning instead of asking org is the whole point — org's parser is the thing that's confused. + +One test failed after the implementation on an off-by-one in my own expectation (the unterminated opener is line 7, not 8); the code was right and I corrected the test. Verified against home's live fixture both ways: two findings under the pre-change mirror copy, zero under the new canonical, home's file untouched. Synced canonical → mirror, full =make test= green at exit 0. + +Review caught two things I fixed rather than filed: the uppercase-delimiter path was handled but untested (confirmed it was a real trigger — 2 findings before, 0 after — then added the boundary test), and the cache-tick comparison used =eq= where =eql= is strictly correct. Verdict Approve, committed 8822b0d after the voice pass. Pinged home that the filter landed and the fixture can be re-run. + +Left the =,**= comma-escape in the notes template alone. The task said the fix "lets that escape be reverted," but the escape is correct org for a literal =**= in a verbatim block, so reverting buys nothing now that the finding is suppressed. Recorded the reasoning in the closed task rather than acting on the permission silently. + +** Task review + +Batch of 7 from the staleness script, oldest first. Every one came back Keep with no new =:quick:= or =:solo:= tag, and Craig confirmed the batch in one pass. Backed todo.org up to /tmp first (matching the mutator convention) and stamped by exact line number rather than a global replace, since ten other tasks carried the same 2026-07-13 date and only six were in the batch. The Sentry vNext task had no properties drawer at all, so it got one. + +Two observations worth keeping. The uniform Keep-with-no-tags result isn't the review going soft: this batch is the deliberation-heavy tail, where every task needs a decision from Craig somewhere in the middle, which is exactly why the speedruns kept stepping past them and why they aged. So part of the staleness count is measuring work that can't be handed off rather than work nobody read. And the KB orphan task cites a 2026-07-01 snapshot of 53 agent nodes; tonight's startup counted 104, so the KB has doubled and the snapshot is worth even less than the task body already assumed. + +** Inbox — home's acknowledgment + +Home replied confirming both messages landed, the fixture stays unescaped, and they'll pick the filter up after the rulesets push and their next clean startup sync. A pure FYI asking nothing, so it skipped the skeptical review and got no reply (acking an ack loops). Deleted; inbox back to zero. It does corroborate that the unpushed commit is the only thing standing between home and the fix. + +** Task review (cont.) + +Staleness went 13 → 6. Lint flags five more tasks missing =:LAST_REVIEWED:= entirely (lines 327, 336, 434, 593, 602) — all =[#D]=, which the staleness script excludes but the checker doesn't. Pre-existing, not touched. diff --git a/.ai/sessions/2026-07-25-15-34-clean-wrap-and-inbox-safe-sync.org b/.ai/sessions/2026-07-25-15-34-clean-wrap-and-inbox-safe-sync.org new file mode 100644 index 0000000..737b1c5 --- /dev/null +++ b/.ai/sessions/2026-07-25-15-34-clean-wrap-and-inbox-safe-sync.org @@ -0,0 +1,85 @@ +#+TITLE: Session Context — 2026-07-25 +#+AUTHOR: Craig Jennings + +* Summary + +** Active Goal + +Guarantee that rulesets can only report a successful wrap with a completely clean Git worktree, while allowing other projects to refresh rulesets when its only residue is untracked inbox deliveries. Implemented, reviewed, fully tested, and prepared for a strict self-hosted wrap. + +** Decisions + +- One executable, =git-worktree-gate=, owns both repository-state policies. =strict= means no staged, unstaged, untracked, dirty-submodule, or in-progress-operation state; =sync-safe= permits only untracked paths beneath =inbox/=. +- Cleanup failure is not a degraded wrap. It leaves the session open and must report each path, its Git state, why it cannot be resolved safely, and the decision Craig needs. Dirty-file deferrals, valediction, and teardown are forbidden in that state. +- Final wrap verification is a HEAD-bound certificate stored in the Git directory and freshly rechecked inside the existing teardown hook. Integrating the check avoids the concurrent-hook race documented by Codex. +- Another project's structured Edit/Write call may not resolve through an installed symlink into rulesets. The runtime hook denies it and points the sender to =inbox-send rulesets=. +- The two MCP-registry handoffs were consolidated into one parked =[#B]= specification decision. The memory auditor remains separate; no machine-owned MCP configuration was promoted into canonical rulesets. + +** Data Collected / Findings + +- The prior rulesets archive completed at 09:24; the inherited three tracked changes were written at 10:27. The repository was dirtied after wrap through a later write path, which is why wrap certification alone needed the symlink-aware boundary guard. +- The previous wrap prose contradicted itself: clean Git state was an exit criterion, but the leftover and inbox sections allowed explicit deferral. The teardown hook checked only the sentinel. +- Adversarial review found a fail-open process-substitution edge: a low-level =git status= failure could appear as an empty stream. The gate now captures status and its exit code in Git-directory temporary files and blocks on failure. +- Current Codex hooks support Stop blocking with =continue=false= and run matching commands concurrently. A user-level =hooks.json= is installed; a new Codex session must complete the normal hook review/trust step. +- Full =make test= passed twice on the implementation. The final run includes 436 core Python tests, 72 hook tests, language suites, ERT suites, and all Bats suites. Focused additions cover inbox-only pulls, staged/unstaged/untracked/submodule/operation states, status failure, certificate/HEAD drift, Claude and Codex Stop outputs, installation, and realpath-based write denial. +- The wrap roam sweep found Craig's 120-column table question. It was already enforced by =org-tables.md=, =lint-org='s =org-table-standard= judgment, and =wrap-org-table.el=; the live lint pass flagged the existing over-wide table. Removed the duplicate capture and synced roam. + +** Files Modified + +- =claude-templates/bin/git-worktree-gate= — shared strict/sync-safe classifier plus certificate/verify modes. +- Startup protocol/workflow mirrors and =claude-templates/bin/ai= — inbox-only state remains visible but no longer blocks fast-forward refresh. +- Wrap protocol/workflow mirrors and =hooks/ai-wrap-teardown.sh= — no deferral escape, actionable hard blocker, final certificate, fresh teardown verification. +- =hooks/rulesets-write-boundary.py=, Claude/Codex hook configuration, cross-project rule, Makefile, and hook documentation — prevent structured writes through installed symlinks and install the enforcement on both runtimes. +- Bats and pytest suites — repository-state, launcher, teardown, installer, and cross-project boundary regressions. +- =todo.org= and workflow state — parked the consolidated MCP registry spec decision and recorded inbox processing. +- Three inherited backlog files — retain the post-09:24 definition that speedrunnable means =:solo:=. + +** Next Steps + +- Start a new Codex session and review/trust the new user-level hooks when =/hooks= prompts; Claude already reads the linked hook configuration. +- Say "spec the MCP registry sync" when ready to design the separate host-level registry reconciler. +- The existing =inbox/lint-followups.org= pipeline retains its 14 current judgment items, including the over-wide table and older missing review stamps/links; they do not represent uncommitted work after this wrap. + +KB: promoted 0 / consulted no + +* Session Log + +** Startup and clean-worktree investigation + +Ran the required startup workflow. The canonical rulesets pull and template sync were blocked by three tracked modifications; two untracked inbox handoffs were also pending. Read the project and global behavioral rules, recent session archive, wrap workflow, relevant hook and launcher code, and the current Codex MCP and hook documentation needed to evaluate the inbox proposals. + +Investigated Craig's requirement that rulesets finish with an absolutely clean worktree while still allowing downstream projects to sync when rulesets has received inbox deliveries. The current wrap workflow states a clean exit criterion but later permits explicitly deferred dirty files, and its teardown hook checks only the wrap sentinel rather than Git state. The startup shell's tracked-change check already ignores untracked inbox files, but the general launcher dirty check does not distinguish inbox deliveries from other untracked residue. + +The three tracked files now dirty were written at 10:27, after the latest archived rulesets session wrapped at 09:24. That establishes a post-wrap write path: a strict wrap gate can guarantee the state at completion, but preventing later contamination also needs a realpath-aware cross-project write guard because globally installed rules and workflows are symlinks into this repository. + +The proposed design is one shared repository-state classifier with two policies. Strict wrap requires no staged, unstaged, or untracked entries and no dirty submodules. Inbox-safe sync permits an otherwise clean tracked/index state with untracked entries only below =inbox/=; those deliveries do not block pull or template sync. The strict check should run after the final push and again inside the ordered teardown hook, tied to the checked HEAD, while the inbox-safe policy should drive both startup and the =ai= launcher. Tests should cover every Git state, unusual path names, inbox-only sync, hook behavior, and the canonical/template mirrors. + +Two pending inbox handoffs both propose a shared Claude-to-Codex MCP registry mirror. They pass the value gate but overlap. The recommendation is to consolidate them into one =[#B]= specification with Codex-only entries preserved, atomic and redacted updates, dependency and health checks, both-machine verification, and the Claude-memory audit split into a separate task. No inbox disposition or project implementation has been applied. + +Completed the investigation plan without changing product code. The worktree proposal is ready for Craig's approval; implementation, tests, and inbox filing remain deliberately pending. + +** Clean-wrap invariant clarified + +Craig confirmed that cleanup failure must prevent wrap-up entirely. The agent must keep the session open and report exactly what remains in the Git worktree, why it could not resolve each item safely, and the specific action or decision Craig needs to supply. A warning, deferred-file exception, valediction, archived-as-complete status, or teardown is not an acceptable substitute for a clean tree. + +** Clean-wrap enforcement implemented + +Craig approved implementation and asked for a full wrap when it is done. Added =git-worktree-gate= as the single policy executable: strict mode rejects every staged, unstaged, untracked, dirty-submodule, and in-progress-operation state; sync-safe mode permits only untracked =inbox/= deliveries. Certificate and verify modes bind the final clean check to HEAD inside the Git directory. + +Wired sync-safe behavior into both startup workflow copies and the =ai= launcher. The picker labels inbox-only state distinctly and now fast-forwards a behind repository with inbox deliveries present while refusing other untracked residue. + +Made wrap cleanup fail closed: removed every dirty-file and inbox deferral escape, added the post-push clean certificate as a hard prerequisite to valediction, and required an exact path/state/needed-decision report when cleanup cannot finish. The existing teardown hook now freshly verifies the certificate and HEAD before consuming either sentinel, emits the runtime-appropriate Claude or Codex Stop blocker, and leaves the sentinel/session intact on failure. Added global Codex hook configuration and installed its symlink; Codex will require its normal hook review/trust on a new session. + +Added =rulesets-write-boundary.py= and configured Claude and Codex Edit/Write hooks. It resolves targets through symlinks and denies another project's write when the real path lands in rulesets, directing the proposal through =inbox-send rulesets=. The cross-project rule now states the installed-symlink case explicitly. + +Focused verification is green: 10 state-gate Bats cases, 13 teardown-hook cases including dirty/changed-HEAD/missing-certificate and both runtime outputs, 36 launcher cases including inbox-only pull, 5 installer cases, and 5 Python write-boundary cases. + +** Inbox — MCP registry proposals consolidated + +Processed the two pending 2026-07-25 handoffs from work and home. They were duplicate evidence for a host-level Claude-to-Codex MCP registry reconciler, not project-level implementation requests. Filed one =[#B]= parked specification decision in =todo.org= with preservation, redaction, atomicity, transport, health-check, two-machine, malformed-input, and token-rotation gates; split the memory auditor into separate future work. Deleted both inbound handoffs, stamped =:LAST_INBOX_PROCESS:=, sent acknowledgements to both source projects, and verified zero pending project handoffs. + +** Review, full verification, and wrap cleanup + +The first full =make test= run passed. Adversarial self-review then found the Git-status process-substitution fail-open and a path-resolution fail-open in the cross-project hook; fixed both, added status-error and sequencer regressions, aligned =protocols.org= with the new policies, and reran the complete suite successfully. + +Executed wrap cleanup: no sentry lock, todo hygiene/convert/archive/priority passes made no changes, lint-org applied zero mechanical changes and refreshed the 14-item judgment pipeline, project inbox remained at zero, route-batch had no candidates, and 30-day staleness was zero. The shared roam inbox held one rulesets capture about 120-column tables; verified the enforcement already exists in three layers, removed the duplicate, and pushed the roam update with the repository's sync helper. diff --git a/.ai/sessions/2026-07-27-17-02-context-engineering-rightsizing.org b/.ai/sessions/2026-07-27-17-02-context-engineering-rightsizing.org new file mode 100644 index 0000000..54020e4 --- /dev/null +++ b/.ai/sessions/2026-07-27-17-02-context-engineering-rightsizing.org @@ -0,0 +1,176 @@ +* Summary + +** Active Goal + +Review three Anthropic posts on context engineering against what this repo ships downstream, then act on the findings. Ended with the always-loaded rules surface cut from ~57,800 tokens to ~28,949 (plus 13,461 path-scoped), two mechanisms proven in a live session, and two of my own bugs found and fixed — one of which had killed Craig's work session. + +** Decisions + +- *Split rules by blast radius, not by size.* What must hold whether or not you're publishing, and where a violation is permanent and reaches other people, stays always-loaded. Everything recoverable can ride a trigger. That's what let =commits.md= and =testing.md= ship without waiting on any pilot. +- *Goal is output quality first, tokens second* (Craig's correction). Anthropic's 80% was a finding, not a target. P4/Phase 8 (effort reduction) dropped outright for trading quality for cost; P5 (positive framing over prohibition) promoted as the lever that actually targets guardrails working against output. +- *Don't apply the posts additively.* The harness system prompt already carries most of what the Opus 5 guide recommends adding, near-verbatim. Adding it to =claude-rules/= would worsen the duplicate-and-conflict problem the first post opens with. The posts' value here is subtractive. +- *Everything authored in or about the repo is first person* (Craig's instruction), with one carve-out: a comment describing what the code does stays third person, since there the code is the actor. +- *No unmerging home.* Its domains are already separated by tag (24 finances, 18 kit, 17 jrestate). The real problem is a tag namespace flattening four orthogonal axes, and cross-domain priority is a judgment no scheme can make — splitting projects hides the question rather than answering it. + +** Data Collected / Findings + +- *Path-scoping works at user level*, confirmed by =/context= in a live work session: 17 generic rules listed, the three path-scoped ones absent. Deterministic glob match, so no trial needed for that tier. +- *User-level and project-level rules both load, project wins.* work and =.emacs.d= carried 19 byte-identical duplicates, and a stale project copy silently overrode the fresh global rule. +- *My token estimates were 45% low.* Real ratio 2.28 tok/word. =commits.md= was 12,800 tokens, not the ~7,000 I claimed. =/context= reported the true per-file numbers the whole time and I used a word-count estimate because it was easier to compute from inside the repo. +- *Two loading paths, not one.* Memory files arrive via the harness; =protocols.org= and the workflows are read by startup and land in Messages. They shrink by editing the workflow, not by scoping a rule. +- *41 execution/hygiene workflows against 6 discovery/design.* The system is heavily built on the half of the problem that got easier. +- *The instructions don't practice what they demand:* =commits.md= argued terseness at 5,561 words, =interaction.md= bans bold while the rules carry 591 bold markers, =testing.md= argues TDD across eight more rows of rationalizations. +- *Two of my own mechanical guards failed the same day, both certifying success while doing damage.* =wrap-org-table.el= reflowed a table into a worse shape and =lint-org= then passed it; the wrap-teardown hook consumed a two-hour-old sentinel and killed Craig's live work session. + +** Files Modified + +Seven commits, all pushed, velox synced throughout. =6c1ea8b= peer-reasoning rule + Chrome convention + KB probe fix. =0adcb1a= =paths:= frontmatter on the three file-type rules + the lint checker that catches prose/frontmatter mismatch. =7ea1d7b= generic rules no longer ship per project, sweep + gitignored session anchor. =79ed3b0= the three rightsizing docs. =2c664cb= =hooks/session-start-disarm.sh= for the sentinel bug. =d74d98d= docs corrected against live measurements. =931f364= =commits.md= → invariant core + =publish= skill. =2f45b6e= =testing.md= → directive core + =testing-standards= skill, approval-gate signal fixed, first-person directive. + +** Next Steps + +Everything remaining needs Craig's decisions rather than execution — see the =[#B] Finish context-engineering rightsizing= task and the =[2026-07-27]= reminder. In order: reconcile the three working docs (one commit behind), then C1 (=verification.md='s honesty core vs the over-verification warning), =interaction.md=, the TDD rationalization table, and D3 (which approval gates are preference vs guardrail). + +Also open: the work sentry triage split and the recurring-loop proposal, both filed =[#B]= with their reviews. The sentry spec review is still waiting, now two weeks old. + +KB: promoted 2 / consulted no + +* Session Log + +** Startup — 2026-07-27 10:25 CDT + +Ran startup. Rulesets already current; =make install= had nothing new to link; project repo clean at f2609d9 with no upstream drift. =.ai/= synced from templates (no churn — the sync is a no-op mirror refresh in this repo). Previous session wrapped cleanly (no session-context anchor present). + +Startup signals: 6 top-level tasks unreviewed for >7 days; roam inbox empty; KB at 106 =:agent:= nodes but the best-practices node path resolved empty (=rg -l 'agent-kb-best-practices'= found nothing — worth checking whether that node exists); no spec-sort or host-identity flags; language-bundle sync silent. + +Five new inbox handoffs arrived since the last wrap. Read all five and ran the skeptical review on each before surfacing dispositions. + +Disposed of one without asking: home's 07-26 10:21 file was a pure FYI acknowledging that the parked MCP-registry spec decision and the separate memory-auditor track matched its handoff. It asked for nothing, so it needed no reply and no approval — deleted it. Four remain, all shared-asset or convention changes, all waiting on Craig's approval per the inbox engine's core §2. + +Skeptical-review conclusions on the four pending: + +- *archsetup publish-lock* — design sound, acceptance checks testable. Three gaps: TTL sizing has to survive an indefinite human approval wait (refresh-on-re-entry covers the agent, not a Craig who steps away); the blocked second session's behavior (wait / defer / stop) is unspecified; and it's a real build, not a quick edit. Recommend filing =[#B] :feature:=. +- *home Chrome convention* — protocols.org already prefers google-chrome-stable over xdg-open, so the new parts are =--new-tab=, multi-URL, and the confirmation line. The confirmation half contradicts the existing =&>/dev/null &= form, which discards exactly the message to be verified. Recommend applying with a foreground-when-running / background-on-cold-start reconciliation. +- *work sentry triage correction* — Craig's 07-27 correction supersedes his 07-21 ruling; today's work fire missed a Hayk DM and a Kostya PR-review request. The gap is that the current rule excludes by category (mail / messenger) and the new split is work-vs-personal, which category can't express. Shipping plugins: cmail, personal-gmail, personal-calendar, telegram, github-prs — no general work-mail plugin, so work's source is project-specific. A denylist of personal plugin names fails open on the next personal source added; a per-plugin eligibility declaration is the durable shape, and that's a design call. Recommend filing =[#B] :bug:= (Major × most-users-frequently = P2). +- *work peer-reasoning rule* — approved exact text, well-formed. Two notes: it's a reasoning contract in a file scoped to communication style (the framing line should widen), and "process serves the outcome" sits one reading away from licensing deviation from the mandatory gates. Its own wording says surface-before-proceeding, so no edit needed, but that's the line to watch. Recommend installing as written. + +** Inbox pass applied — commit 6c1ea8b + +Craig approved all four dispositions plus the probe fix. Two corrections from him along the way: I had inverted the render-merge guard (numerals belong to the options list, dashes to every other enumeration in the same message — I did the reverse), and processed items shouldn't be left sitting in =inbox/=. + +Shipped: the peer-reasoning section at the top of =claude-rules/interaction.md= with the file's framing line widened; the Chrome convention rewritten in canonical =protocols.org=; the KB best-practices probe switched from a content grep to a filename =find=. Filed two =[#B]= tasks (sentry triage split, repository publish-lock), both stamped =:LAST_REVIEWED: 2026-07-27=. Swept the 40-file =PROCESSED-*= backlog out of =inbox/= along with the four handoffs; =inbox-status= now reports 0. Replies sent to archsetup, home, and work. + +*The review caught my own error.* I had written that Chrome's confirmation line prints to stderr and told every project to capture it with =2>&1=. It prints to *stdout*; stderr is empty. Verified both directions on ratio before correcting. An agent following the original text would have captured stderr, seen nothing, and concluded the tab failed to open — the exact silent-failure shape as the KB probe it shipped alongside. I asserted a stream rather than checking it, inside the same change that told others to verify. Side effect: four =about:blank= tabs opened in Craig's live browser during the check. + +Deliberate departure recorded: =route_recommend= returned =work strong= for the sentry task, but the work happens in rulesets' canonical =sentry.org=, so it's a local keeper and I left it unstamped rather than letting the wrap router offer to ship it away. + +Possible KB promotion at wrap: an identifier that lives only in a filename is invisible to a content search, and the lookup fails *silently* rather than erroring. Adjacent to the existing enumerate-vs-discover node but a distinct failure surface. Two instances in one commit (the KB probe, the Chrome stream) argue it generalizes. + +Startup-workflow bug found while checking the KB nudge: Phase A resolves the best-practices node with =rg -l 'agent-kb-best-practices' "$ra"=, which greps file *content*. The node's slug lives in its filename, so the probe returns empty and the contribute nudge points at nothing — in every project, every session. The node exists at =~/org/roam/agents/20260620232112-agent-kb-best-practices.org=. Synced-workflow change, so it waits on approval too. + +** Pushed and synced velox + +Pushed 6c1ea8b to origin/main (ahead-only, reconciled immediately before). On ratio, so velox needed the pull: it fast-forwarded and =make install= linked three things it had been missing since 2026-07-25 — the Codex =hooks.json=, =rulesets-write-boundary.py=, and =git-worktree-gate=. That drift is exactly the one-time-setup case =daily-drivers.md= names: the files traveled with the pull, but nothing re-runs the installer, so the symlinks only land where someone runs it. A new Codex session on velox will now hit the hook review/trust prompt, which was already on the 2026-07-25 next-steps list. + +** Context-engineering rightsizing — analysis and rollout plan + +Craig supplied three Anthropic posts (the 2026-07-24 Claude 5 context-engineering post, the Opus 5 prompting guide, the 2026-07-06 Fable field guide) and asked for a review, proposals, and a consistency audit of what this repo ships downstream. Then he reframed twice, and both reframes were better than the question I'd been answering. + +*First reframe:* consider the files as *his prompts*, not my context. That changed the finding. My first pass measured the always-loaded surface (32,123 words — =claude-rules/= 25,386 + =protocols.org= 6,620 + CLAUDE.md 117, roughly 40k tokens before the user's first word) and proposed shrinking it. Read as a map he hands every project, the finding is different: 41 execution/hygiene workflows against 6 discovery/design, seven to one. That ratio was right when the risk was the model doing things wrong. The field guide's claim is the bottleneck moved to the human's ability to clarify unknowns, so the system is heavily built on the half that got easier. + +*Second reframe:* metrics per claim, not one go/no-go. Turns the rollout into a set of separable testable claims rather than one bet. + +Three checkable "doesn't practice what it demands" findings: =commits.md= argues terseness at 5,561 words (longest file in the set); =interaction.md= bans bold in chat while the rules carry 591 bold markers; =testing.md= mandates TDD then argues eight more rows against rationalizations. + +*The finding that changed the plan:* the harness system prompt already carries most of what the Opus 5 guide recommends adding — its task-scope block, correction-narration block, and subagent cap are present nearly verbatim, and post 1's replacement comment guidance is present as the post's own new wording. So applying the posts additively would make the duplicate-and-conflict problem worse. The posts' value here is subtractive. It also exposes a third dedup axis nobody has audited: =claude-rules/= against the harness prompt, invisible from inside the repo. + +*Pilot selection rule* (the part that matters more than the list): the six pilot files were chosen because a silent miss is *detectable*, not because they're small. Four have a mechanical checker (=lint-org= =org-table-standard=, spec-board grep, =spec-review=), two produce an error Craig sees in seconds. =daily-drivers.md= and =emacs.md= were considered and held back — low risk, but a miss surfaces too slowly to learn from inside the trial window. + +Artifacts in =working/context-engineering-rightsizing/=: =proposals.org= (P1-P6, conflicts C1-C2, the from-your-side-of-the-desk section), =rollout.org= (Phases 0-8, decisions D1-D7, target trajectory), =metrics.org= (claim-by-claim testability, pilot go/no-go with the denominator rule, turn-back vs abandon triggers). + +Two honesty notes carried into the docs: I have a stake in arguing my own instructions should be shorter, so the plan weights mechanical detectors over my self-report; and about half the posts' claims aren't testable here without an eval harness, so those are labelled judgment rather than measurement so a future session doesn't mistake an adopted opinion for a tested result. + +Not started. Awaiting D1 (confirm pilot set) and D2 (skill index in the core). + +** Path-scoping shipped (0adcb1a) and work pre-synced + +The session's biggest finding: Claude Code scopes a rule by a =paths:= field in YAML frontmatter, and none of the 20 rules had one — even though three already declared a file-type scope in their =Applies to:= prose line. So =todo-format.md= (4,494), =org-tables.md= (464), and =emacs.md= (923) loaded into every session in every project, contradicting their own first line. 5,896 words. Fixed by adding the frontmatter, plus a =lint.sh= checker that warns when prose names a concrete extension without matching frontmatter (flags exactly those three, nothing else), plus teaching the heading check to skip a frontmatter block. Always-loaded rules surface: 25,386 → 19,505. + +Also confirmed from the docs: user-level and project-level rules *both* load, and project rules take priority. So work and =.emacs.d= carry 19 byte-identical duplicate copies, and a stale project copy overrides a fresh global one — which is exactly what was happening to work's =interaction.md= between this morning's commit and its next startup. + +Pre-synced work via =scripts/sync-language-bundle.sh ~/projects/work= (rulesets' own installer, run early rather than waiting for work's startup) so Craig's next work session is a valid test rather than one running the set it loaded before the sync. Verified: all four files now match canonical, frontmatter present, and work's =.claude/= is gitignored there so nothing was dirtied. + +Open question the next session answers: does =paths:= frontmatter apply to *user-level* rules or project-level only? The docs don't draw the distinction. =/context= in a fresh session settles it — if =todo-format.md= is absent from Memory files until an org file is opened, it works. If it's listed, the frontmatter is inert (no harm) and semantic skills are the only route. + +Not done: the double-load fix. Removing the 19 duplicates means changing what =install-lang= pushes into projects, and there may be a teammate-facing reason for them. Surfaced as Craig's call, not urgent — wasteful, not harmful. + +** De-duplicated the rules layer, unblocked sync (7ea1d7b, 79ed3b0) + +Craig confirmed no teammates depend on the per-project rule copies, so I removed them. =install-lang.sh= no longer copies the generic rules; =sync-language-bundle.sh= sweeps the ones earlier installs left, guarded on the global rule existing so a machine mid-bootstrap isn't stranded with none. Swept 20 files each from work and =.emacs.d=, leaving only their language rules plus work's =publishing.md= overlay. Three existing tests encoded the old contract and were rewritten; the generic-drift test now asserts sweep-not-repair, which is the stronger fix since the drifted copy outranked the global rule while it existed. Four new tests cover the sweep, the two keep-cases, and the no-global-rule guard. + +Also gitignored =.ai/session-context.org= and =.ai/session-context.d/=. This repo tracks =.ai/=, so the live anchor read as untracked all session and =git-worktree-gate= reported rulesets sync-blocked — meaning every other project skipped its rulesets pull until wrap, every session. Craig spotted the blocked state and inferred it was why I pre-synced work; it wasn't (rules load at launch, before the startup sync runs, which was the real reason), but chasing his inference found the anchor problem, which was the better bug. + +Corrections from Craig this stretch: the goal is output quality first, token reduction second — my docs led with the wrong number and P4 (effort reduction) should be demoted or dropped since it trades quality for cost. And all authored prose goes first person; I amended the first commit rather than leaving it. Code comments stay third-person by agreement, since they describe what the code does for the next reader. + +Docs not yet updated for either the goal reordering or the last two hours of findings (path-scoping, the double-load, the harness overlap). That's the next task. + +** Inbox: archsetup ack + +archsetup acknowledged the publish-lock acceptance and the three implementation gaps, confirming the decision stays closed on its side. Pure FYI, nothing asked, no reply owed. Deleted it. Inbox back to zero. + +** Killed Craig's work session with my own hook, then fixed it (2c664cb) + +Craig's 13:20 work session was blocked repeatedly and then had its terminal closed under it. The cause was mine, from Saturday's clean-wrap work. + +=wrap-it-up= drops =/tmp/ai-wrap-teardown-<project>= so the =Stop= hook tears down once the wrap certifies clean. I deliberately made a failed certification *preserve* the sentinel, so a wrap blocked by a dirty tree could retry on a later stop. I never bounded that retry to the session. work's 11:37 wrap left an uncertified sentinel armed; the 13:20 session's stops were all blocked by it failing certification; then startup's two commits (task filing, template sync) made the tree clean, the next stop certified, and =cj/ai-term-quit= killed the tmux session mid-work. + +Two others were armed and dangerous at the same moment: archsetup's since Saturday 15:02 on a live attached terminal, and home's from 13:21 on a live session. Disarmed all three by hand (backed up to =/tmp/disarmed-sentinels=) before writing any fix, since both were minutes from the same fate. + +Fix: =hooks/session-start-disarm.sh= clears the project's sentinels at =SessionStart= — a new session means the wrap that armed one is gone. Within-session retry is untouched (the hook only runs at session start) and a test pins that so the deliberate behavior isn't lost to the fix. Four tests on the disarm including project-scoping, one on the retry. Wired into =.claude/settings.json=, installed on both machines, =wrap-it-up.org= documents the session-scoping with the worked failure. + +Diagnostic note worth keeping: I found it by reading work's own crashed session anchor, which showed startup completing normally and then stopping dead, plus its git log showing two commits at 13:21 — the exact moment the tree went clean. The anchor being left behind by the interrupted session is what made the timeline reconstructable. That's the crash-recovery purpose earning itself. + +** /context settled both open questions; docs corrected (d74d98d) + +Craig ran =/context= in work. Memory files lists 17 generic rules; =todo-format.md=, =org-tables.md=, and =emacs.md= are absent, and only =python-testing.md= and =publishing.md= come from the project's own rules dir. So *path-scoping works at user level* and *the de-duplication holds*. Both were open. + +Three corrections the live numbers forced: + +1. *My token figures were low by ~45%.* Real ratio is 2.28 tok/word, not the ~1.3 I assumed. =commits.md= is 12,800 tokens (I said ~7,000); =claude-rules/= was ~57,800/session before today, now 44,410, with 13,390 path-scoped out. Worth naming the actual error: =/context= reports per-file token counts and I used a word-count estimate instead because it was easier to compute from inside the repo. The instrument existed the whole time. +2. *Two loading paths, not one.* Memory files arrive via the harness at session start. =protocols.org= and the workflows are *read by startup*, so they land in Messages and never appear under Memory files. They shrink by editing the workflow, not by scoping a rule. My "always-loaded surface" number conflated them. +3. *The harness's own suggestion* names =commits.md=, =testing.md=, =MEMORY.md= as the top three to prune — independently the same Phase 4 list I'd proposed. + +Because a glob match is deterministic, the remaining work splits: path-scopable rules ship with no trial (=docs-lifecycle.md= on =docs/**= is next), and only semantic-condition rules need the skills route and the stop conditions. =commits.md= is the real test there — largest single item, and almost all publish machinery that only applies when a commit is in play. + +Recorded a caution the confirmation doesn't cover: path-scoping fires on a *read* of a matching file, so creating a new org file from scratch never triggers =todo-format.md=. Edits are safe (Edit requires a prior read). + +Also folded in Craig's goal correction (quality first, tokens second): P4/Phase 8 dropped outright since lowering effort trades quality for cost, P5 promoted since positive-framing-over-prohibition is what targets guardrails working against output. + +** Split commits.md: 12,800 tokens → 2,342 always-loaded (931f364) + +Craig picked the commits.md split over docs-lifecycle after I checked the latter and found I'd overstated it — =docs-lifecycle.md= scopes to "any project carrying a docs/ tree," a *project-level* condition a glob can't express, and 2 of its 6 trigger points are creation cases a read-triggered path rule misses. Only three rules ever named a concrete extension and all three are already converted, so there is no other clean path-scope candidate. + +The split line is *blast radius*, not size. Stayed always-loaded (1,027 words / ~2,342 tokens): author identity, the no-AI-attribution ban, the generated-document byline rule, the public-artifact content-scope rules, and "If You Catch Yourself." Moved to =publish/SKILL.md= (4,871 words): message format, Voice and Focus, PR description structure, Review and Publish Steps 0-2, the three review shapes, hook authorization, merge strategy, the pre-commit checklist. + +Why that line: if the skill fails to trigger I don't know the flow and have to be told — visible and recoverable. I don't silently commit with AI attribution, because that guard never moved. Only the recoverable half rides the skill-triggering bet, which is what let this ship without waiting on the pilot. + +*Verified by using it.* The skill registered mid-session and I invoked =/publish= to publish its own commit; it loaded with the full flow present. Content conserved and checked rather than assumed: 5,561 words in, 5,898 across both files (delta = frontmatter + the pointer added to the core). Repointed five cross-references in =voice=, =review-code=, =inbox.org=, and =no-approvals.org= that named moved sections. + +Always-loaded rules surface: 44,410 → ~33,950 tokens. Started the day at ~57,800. + +Noted and deliberately not done: =publish/SKILL.md= is a single 4,871-word blob, and both posts argue a long skill should use progressive disclosure internally. It loads on demand now, which is the win worth taking; splitting it further is its own change. + +Also surfaced: the Step 2 =.ai=-tracking heuristic misfires here. It reads tracked =.ai/= as "shared team repo → skip the approval gate," but rulesets tracks =.ai/= as a committed mirror while being a private single-user repo. I kept asking rather than skipping, and flagged it to Craig. + +** testing.md split, gate fixed, first-person directive added (2f45b6e) + +Three changes. *testing.md split* the same way as commits.md — by what has to be resident, not by size. Core keeps TDD-is-default and the three-category requirement (347 words), because those fire *before* any code is written, which is exactly when no skill has been summoned. Everything else → =testing-standards= skill (2,903 words): characterization recipes, per-category detail, property/mutation testing, pyramid, integration rules, naming, test-quality and mocking rules, coverage targets, spike exception, anti-patterns. + +*Approval-gate fix.* The publish flow decided whether to ask by checking whether =.ai/= is tracked, as a proxy for "team repo." Wrong in the direction that matters: rulesets, home, and work all track =.ai/= and all three are private single-user repos, so the rule skipped the gate on Craig's three most-used projects. Now checks whether any remote is on a host other than cjennings.net. Verified both directions including a synthetic GitHub remote. Every current project → gate applies, which matches how the flow has actually been run all session. + +*First-person directive* added to the always-loaded core, at Craig's instruction. One existed for commit bodies/PR prose but it moved into the publish skill, and it never covered code comments at all. Now: everything authored in or about the repo is first person, with one carve-out — a comment describing what the code *does* stays third person, since there the code is the actor. + +Also split =publish/SKILL.md= internally: PR descriptions + the three review shapes → =references/pull-requests.md=, since a plain commit never needs them. SKILL.md 5,012 → 3,888 words. + +*Surface: ~57,800 tokens this morning → ~28,949 always-loaded now* (plus 13,461 path-scoped). Largest remaining: =interaction.md= 3,828, =verification.md= 3,388, =commits.md= core 2,804, =subagents.md= 2,373, =cross-project.md= 2,305. + +Risk recorded rather than buried: testing.md's margin is thinner than commits.md's. If =testing-standards= fails to trigger mid-test-writing I lose the mocking-boundary rules — a quality regression, visible in review, but a real bet where commits.md's moved half was purely procedural. Also moved the TDD rationalization table rather than cutting it; the posts say that kind of over-argument is counterproductive now, but deleting Craig's defense against me skipping TDD is his call. diff --git a/.ai/sessions/2026-07-29-06-36-adversarial-review-flow-and-telegram-fixes.org b/.ai/sessions/2026-07-29-06-36-adversarial-review-flow-and-telegram-fixes.org new file mode 100644 index 0000000..12cbe3d --- /dev/null +++ b/.ai/sessions/2026-07-29-06-36-adversarial-review-flow-and-telegram-fixes.org @@ -0,0 +1,514 @@ +#+TITLE: Session Context — 2026-07-28 +#+AUTHOR: Craig Jennings +#+DATE: 2026-07-28 + +* Summary + +** Active Goal + +Started as inbox triage on a telegram-plugin bug report and became two things: shipping the cross-project fixes that arrived overnight, then designing and dogfooding a mandatory isolated adversarial review before every commit — which immediately found real defects in its own design and in everything filed afterward. + +** Decisions + +- *Merge colliding fixes rather than sequence them.* The parked down-is-launch diff still carried the bad =loadChats= call and cited the segfault gotcha the other fix rewrites. Applying either alone would have shipped a file arguing against itself. +- *"Adversarial", not "hostile"* (Craig). An agent told to attack manufactures findings, so the stance carries a substantiation floor: a finding not substantiated against the diff is dropped. +- *Re-review until the reviewer approves* (Craig's addition, the thing I had missed). Same reviewer continued, not a fresh one — a fresh reviewer can't tell an addressed finding from one that never existed. Bounded at three rounds or first recurrence. +- *Dispatch on every commit*, with the reviewer's own Phase 0 ruling triviality. A floor written as "small" or "mechanical" puts the judgment back with the author, whose judgment is the thing being checked. +- *Pass the requirement source, withhold the rationale.* A ticket is not the author's model; it was written first and by someone else, so it's the only input that can contradict the author's claim. +- *=cycle=, not =pass=, for one sentry loop* (Craig). home proposed =pass=; =sentry.org= already uses it as a numbered noun for the eleven hygiene passes, so =Pass 11= would have collided with =pass 12=. +- *Drop the =references/= link rather than sync the directory.* The four calendar workflows already travel and already carry the recipes. +- *Strip the wrap-org-table task to Verified / Open questions* after its review loop bounded out. The measurements were never what failed. + +** Data Collected / Findings + +- *The telegram bug.* =(telega--loadChats 'main)= sends a bare symbol on the wire; =tdat_plist_value= (=telega-dat.c=) accepts only =(=, =[=, ="=, =-=, digit, =t=, =:=, =n= and calls =assert(false)= on =m=. Verified against telega's source at four points rather than trusting the handoff. Exposure was manual triage only — sentry excludes messengers, so home's eleven overnight cycles never loaded the plugin. +- *=wrap-org-table.el= splits logical rows*, and =lint-org= doesn't merely miss it — it *causes* it. =lint-org.el:424= calls the same broken predicate, so it reports the tool's own correct output as "missing rule between rows — wrap-org-table.el reflows it" when nothing is missing, then reports the corrupted result clean. Idempotence is broken: the tool corrupts its own output on a second run. +- *The isolated reviewer earned its keep on its first four uses*, finding: that withholding the ticket made my claim self-certifying; that my =subagents.md= override reaffirmed the Prompt Contract field that would destroy the isolation; a fourth verdict (=Needs Discussion=) I'd asserted didn't exist; and four successive wrong root-cause analyses on the table bug. +- *rulesets is itself exposed* to the table bug: =todo.org='s four-row attachment-sanitization table. Don't reflow until fixed. +- *Seven =../../= link sites* across four synced workflows resolve only in rulesets. =scripts/lint.sh='s =check_md_links= was built for that class and misses them because it matches markdown syntax only. + +** Files Modified + +Seven commits, all pushed, velox synced after each. =bff0138= merged telegram fixes. =43a4cf7= post-load liveness check. =614e3b1= removed finished working dirs. =ca508a1= filed three handoff findings. =f3f5bfd= the =fire= → =cycle= rename (72 sites plus three that had leaked outside =sentry.org=). =3a933a2= the =references/= analysis. =5999f88= dropped the dead link and deleted the stale file behind it. =ecd5d7b= filed the wrap-org-table bug. + +Rules changed: =publish/SKILL.md= Step 1 rewritten (dispatch contract, four defined verdicts, the loop, bounds), =review-code/SKILL.md= (two levels of dispatch, adversarial contract, re-review mode), =claude-rules/subagents.md= (Isolation Override; Prompt Contract field 2 inverts), and the three unattended callers taught to park. + +** Next Steps + +- *Three of the four items Craig queued are untouched*: the winvm =[#C]= lint defects, the context-engineering rightsizing (needs his four decisions), and the sentry spec deep read (two weeks old). +- *Rule gap found by using the rule*: =Needs Discussion= exits to the user, but nothing says what happens after the user answers — whether the round counter resets. I treated it as a fresh review; that judgment isn't written down. +- *Item 2 is now qualified*: its fix says "run =wrap-org-table.el=", and that tool has a live corruption bug. The specific table is safe, but verify the output rather than trust it. +- Four =[#B]= bugs filed tonight and unstarted: the table splitter, the =../../= links, plus the two carried in. + +KB: promoted 1 / consulted no + +* Session Log + +** 11:55 — Startup + +Ran startup. Rulesets already current, project repo clean and current, =make +install= had nothing new to link, =.ai/= synced from templates. No crash anchor +— previous session (context-engineering rightsizing, 2026-07-27 17:02) wrapped +cleanly. + +Findings: 6 tasks unreviewed >7 days; roam inbox holds 4 items; KB at 108 +=:agent:= nodes with nothing matching this project. Spec-sort and host-identity +probes silent. Language-bundle check silent. + +** 12:05 — Inbox: the telegram segfault root cause + +Four new inbox files from =.emacs.d=, two pairs: a 06:15 intro note + plugin +file, then a 07:21 correction + superseding plugin file. The correction retracts +one secondary claim from the 06:15 write-up (that the "19 of ~50 chats" reading +was truncation caused by the bug — it wasn't; 19 is the real account size, +measured by work at the wire level). Root cause and fix unchanged. + +The proposal: =triage-intake.telegram.org= Step 1 calls =(telega--loadChats +'main)=, and that bare symbol kills =telega-server= outright. + +I verified the whole chain against telega's own source rather than taking the +handoff's word for it (=elpa/telega-20260706.2147/=): + +- =telega--loadChats= (telega-tdlib.el:2190) drops its argument straight into + the request as =:chat_list= with no conversion. Confirmed. +- The C parser =tdat_plist_value= (server/telega-dat.c:466) accepts only =(=, + =[=, ="=, =-=, a digit, =t=, =:=, or =n= to start a value; anything else + prints "Unexpected char '%c' in plist value" and calls =assert(false)=. + =main= starts with =m=. Confirmed, and the accepted-char list in the handoff + is exactly right. +- telega's own callers all pass the object: telega.el:290, telega.el:295, + telega-tdlib-events.el:516. Confirmed. +- The symbol shorthand lives in a different layer — telega-filter.el:472 and + telega-folders.el:99 convert =(eq cl-fspec 'main)= to ='(:@type + "chatListMain")=. Confirmed. + +Every claim holds. Grepped all workflows for other bare-symbol arguments to +=telega--*= wrappers: only the two sites this diff already fixes. + +*The complication I found.* There is already a parked =[#B]= VERIFY (todo.org:136, +from =.emacs.d= 2026-07-24) proposing a different fix to this same file — the +down-is-launch fix, with a prepared diff in +=working/triage-telegram-down-launch/=. Its proposed file still carries the bad +=loadChats= call at its lines 52 and 122, and its third hunk adds prose citing +"tdlib segfaults in native mode (SEGFAULT gotcha below)" — pointing at the very +section this new handoff rewrites to say the crashes were our bug, not tdlib's. +Applying either one alone leaves the file half-right. They need merging, not +sequencing. + +Also: home and work both declare =telegram= in =:TRIAGE_SOURCES:=, so both have +been killing telega-server on every sweep. Work already patched locally and +verified at the wire level. Home has not been told. + +** 11:58 — Staged the batch, closed the loops + +The Stop hook counted the four files still sitting in =inbox/=. The triage was +done; what remained was Craig's merge decision, not mine. So I staged all four +into =working/triage-telegram-segfault-rootcause/= (the 07:21 pair as +=.proposed= + correction note, the 06:15 pair kept as superseded context — its +blast-radius and follow-up sections aren't repeated in the correction). Inbox +back to clean, =inbox-status= rc=0. + +Left =working/triage-telegram-down-launch/= untouched: the parked VERIFY's body +links its =proposed.diff= by path, and moving it would break the link before +Craig has decided anything. If he takes the merge, that consolidates then. + +Replied to =.emacs.d= — confirmed the verification, and told them the one thing +they couldn't see: their own 2026-07-24 parked diff collides with this, both in +the stale =loadChats= call it still carries and in the native-mode prose it adds +citing the gotcha this rewrites. Also declined their suggested bare-symbol lint: +one consumer file, two call sites both fixed, so the stated rule is cheaper than +a checker with a single input. + +Notified home directly with the mechanism and the two-site fix, told it to patch +locally rather than wait on the canonical, and carried the correction forward +explicitly so it doesn't inherit the retracted truncation story. + +Nothing applied to the canonical. That still waits on Craig. + +** 12:10 — Craig picked the merge. Applied and committed. + +Merged both fixes into one version rather than sequencing them. Base was the +segfault-corrected file, then the three parked hunks on top: the down-is-launch +directive, the SCAN-FAILED-only-after-launch-attempted rewording, and =(setq +telega-use-docker t)= restored to the Step 1 code block. + +The reconcile that made merging worth doing. The parked hunk's new comment said +"tdlib segfaults in native mode (SEGFAULT gotcha below)", pointing at the +section the other fix rewrites to say those deaths were our own bad argument. +Left alone the file would have argued against itself. I changed the Step 1 +comment to state plainly that the two are separate concerns (the deaths happened +*in* docker mode, so docker mode is neither a defense against the loadChats bug +nor evidence for itself), and reworded the Quick Reference line from "tdlib +segfaults outside docker mode" to "crashed in native mode (2026-06-09)" with the +same disambiguation. + +That reword also removed a host-identity violation I hadn't gone looking for. +The original asserted "Craig's daemon currently has telega-use-docker nil" — a +mutable machine fact stated as fixed in a synced doc. I checked the actual +default (=telega-customize.el:514=, =defcustom telega-use-docker nil=) and wrote +the durable claim instead. + +Verified: both live call sites use the TL object, the two remaining ='main= +occurrences are inside the gotcha prose describing the bug, lint-org clean on the +changed file, mirror synced, =make test= green before (exit 0) and after (exit +0). + +todo.org: closed the parked =**= VERIFY as =DONE= + =CLOSED:= per todo-format.md +with the merge rationale in the body. Promoted its =***= engine child (SCAN +FAILED must not advance the sentinel) to top-level =**= VERIFY so it doesn't get +buried under a DONE parent. Kept its =:LAST_REVIEWED: 2026-07-24= rather than +stamping today — I moved it and judged it separate, but nobody re-derived its +content, so the older date keeps it honest. + +Review: Approve, no Critical or Important. Two Minor, both surfaced rather than +fixed. The gotcha now advises a post-load =(process-live-p ...)= check that the +Step 1 recipe doesn't actually do, and adding it would extend the recipe past the +two fixes Craig approved. + +Committed =bff0138=. Not pushed — that's a separate confirmation. + +** 12:20 — home replied, and corrected my urgency read + +home accepted, patched both call sites locally, and re-verified the diagnosis +independently rather than trusting it. Useful correction back: home declares +telegram in =:TRIAGE_SOURCES:=, but sentry excludes messengers from triage +intake, so the plugin never loaded on a sentry fire. Eleven overnight fires ran +clean against the broken file. I had assumed the sweeps were affected; the real +exposure is manual triage intake only. + +Told home its stopgap won't be reverted into a broken state — the next rsync +replaces it with canonical content carrying the same fix. On this machine that +lands as soon as its next startup runs, since the rsync reads the local rulesets +working tree. velox needs the push. + +Inbox back to zero. + +** 14:10 — Pushed, and closed the velox gap + +Pushed =1675613..bff0138= to origin after the pre-push reconcile (still 1 ahead, +0 behind). + +=.emacs.d= replied while the push was in flight. It had already verified bff0138 +here, re-run its own workflows rsync, and confirmed the corrected form landed. +It withdrew the bare-symbol lint suggestion, conceded the home omission, and +flagged one real gap: bff0138 was committed but not pushed, so velox stayed +exposed. That was true when written and stale by the time it arrived. + +Checked velox over tailscale (this host is ratio, per =uname -n=). It was 1 +behind / 0 ahead and sync-safe, so I fast-forwarded it through the same +=git-worktree-gate sync-safe= check startup uses. Velox is now at bff0138 and its +workflow file carries the corrected call, with the only bare ='main= occurrences +inside the gotcha prose. Both daily drivers covered. + +Corrected read carried into both replies: the exposure was manual triage intake +only, not the automated sweeps, because sentry excludes messengers. + +** Open follow-ups (surfaced to Craig, not acted on) + +1. The gotcha tells callers to check =(process-live-p (telega-server--proc))= + after a load, but the Step 1 recipe doesn't do it. Now that the corrected call + shouldn't kill the server, that check is what would catch a regression. Left + out deliberately as scope creep past the two approved fixes. +2. Both =working/triage-telegram-*= dirs are completed-task artifacts and want + filing per working-files.md. Revised read after checking: delete both + outright. Every file is tracked (=b19d420= and =bff0138=), so git holds them + permanently and a copy in =assets/= would only duplicate history. Nothing + links to them. + +** 15:00 — Second inbox round: .emacs.d self-correction + winvm lint findings + +=.emacs.d= wrote back to say it had overcorrected on home: it accepted "home was +in the blast radius" and then recorded that home "had been killing telega-server +on every sweep too", which home's own sentry data refutes. It fixed its task +record rather than leaving it. I told it the pattern wasn't one-sided — I made +the same move this morning, estimating home's blast radius instead of measuring +it, and home's data is what corrected me. It also offered the emacs-side half of +a completed-vs-truncated signal, which it has filed as =[#C]=, if the +=process-live-p= recipe change lands. + +=winvm= sent a link-integrity pass with three findings, all reproduced against +the rulesets source rather than only its local copy. I verified all three: + +1. =protocols.org:273= links =references/calendar-reference.org=, but the rsync + set is only =protocols.org=, =workflows/=, =scripts/=. Dead link in every + consuming project. home and =.emacs.d= have no =.ai/references/= at all. +2. =retrospectives/PRINCIPLES.org:38= violates the org-table standard. + =lint-org= confirms, checker =org-table-standard=. +3. =protocols.org= lints 8 mechanical + 19 judgment =misplaced-heading=, all + from Markdown =**bold**= in an org file. 48 bold spans, 14 line-initial. + +The check winvm couldn't run: adding =references/= to the sync brings =--delete= +with it, which would destroy project-owned files there. work is the only project +with a =.ai/references/= and its copy is byte-identical to the source, so nothing +would be lost. That clears the path to sync-it over drop-the-link. + +One refinement back to winvm on finding 3: the 8 mechanical hits are would-fix, +and the 14 line-initial bold spans are exactly what =lint-org --fix= would +rewrite without knowing they're emphasis rather than headings. So that half is +not safe to run mechanically even though the count suggests it is. + +Filed as =[#B] :bug:= (the dead link, Minor x every-project-every-sync = P2) and +=[#C] :chore:= (the two cosmetic defects, P3), both stamped =:LAST_REVIEWED:= +today since I graded them today. Verified todo.org's lint counts are unchanged +from the pre-edit baseline (mechanical=2, judgment=12, both pre-existing and +already in =inbox/lint-followups.org=). + +Inbox back to zero. todo.org is left uncommitted deliberately — Craig has a +pending decision that will likely touch it again, so batching beats a churn +commit. + +** 18:45 — Craig took both follow-ups. Liveness check staged. + +Added the liveness check to the Step 1 load call. It now returns ='server-died= +instead of an unconditional ='loaded=, and new prose routes that into the +existing SCAN FAILED definition rather than inventing a category: a server that +dies during load leaves a thin hash, which reads exactly like a quiet account, +which is the same false all-clear the down/not-loaded rule already prevents. + +Self-review caught a real defect in my own first draft. I called +=(telega-server-live-p)= bare, but Step 0 guards the same predicate with +=fboundp= because telega may not be loaded. A launch that failed outright would +have signalled void-function instead of returning the clean contract. Added the +guard, matching Step 0's idiom. Verified the predicate is exactly the +=process-live-p= expression the gotcha names (=telega-server.el:221=). + +Verified: parens balance at depth 0, lint-org 0/0, mirror identical, =make test= +green (exit 0) on the final state. + +Not committed — waiting on the approval gate. + +** 18:48 — Third inbox round + +=.emacs.d= sent a closing FYI marked no-action, agreeing my framing of the shared +failure (blast radius estimated rather than measured) named the trigger rather +than the failure. Deleted without reply, since replying to "nothing owed back" +is noise. + +home proposed renaming sentry.org's noun-sense "fire" to "pass", after Craig read +its "nine fires" as nine emergencies: "I assume you mean nine crises, not nine +loop cycles and I begin to get scared." The problem is real, well-evidenced, and +reaches Craig directly through digest headings. + +But the proposed term is wrong, and the reason home gave for it is the +disqualifier. =sentry.org= already uses "pass" as a precise numbered noun — the +pass list, the Pass Runner, "eleven finding/hygiene passes", "pass 12". With +exactly eleven hygiene passes, home's proposed =** Pass 11= heading collides with +an existing referent. That trades a term Craig misreads as urgent for one that is +genuinely ambiguous. + +Counter-proposed *cycle*: zero occurrences in the file, and Craig's own word in +the quote home cited. Checked and rejected "sweep" (3 uses) and "run" (used as a +noun). Filed =[#C] :chore:= with the grading and the collision analysis; replied +to home with the counter-proposal. + +** 18:50 — Three commits, and the collision confirmed from live evidence + +Craig approved both follow-ups and the =cycle= term. Three commits: + +- =43a4cf7= the liveness check. +- =614e3b1= removed both telegram working dirs. Filing by deletion, since every + file was already in git via =b19d420= and =bff0138= and an =assets/= copy would + only duplicate history. +- =ca508a1= filed the three handoff findings with their gradings. + +home wrote back confirming the "pass" collision was real, and that it had already +walked into it: its anchor now carries =** Pass 11= meaning the eleventh cycle, +three lines from =pass 12 (solo-task implementation)= meaning the twelfth item in +the pass list. Same file, two referents, introduced by its own normalization an +hour earlier. It found that in live evidence faster than reading the file would +have caught it. + +It was blocked on Craig's confirmation and had written a memory saying "pass", so +I sent the confirmation immediately. The memory was the urgent half — a stale one +teaches every future home session the ambiguity, where the anchor is one file. + +Recording Craig's approval flipped the sentry task to =:solo:=. The term was the +only judgment it carried, and the completion check is objective, so it can ride a +backlog run rather than waiting for someone to touch =sentry.org=. + +Pushed =bff0138..ca508a1=, velox fast-forwarded to match. + +** 19:45 — The cycle rename, and the leak home's scope missed + +home did its side first: renormalized its anchor by restoring the +pre-normalization backup and re-running fire→cycle from clean, rather than +reverse-mapping pass→cycle. That was the right call — reverse-mapping would have +needed a judgment on every instance to separate its own conversions from genuine +pass-list references, where re-running from clean makes it structural. It also +corrected its memory and handed the canonical back. + +Did the canonical rename with a script rather than by eye: protect the verb sites +by explicit pattern, assert zero unclassified =-ed/-ing= forms survive, then +substitute. 72 noun instances converted, 4 verb sites untouched (=/loop= fires +again, two "fires on approval", "record of what fired"). + +*The leak home's scope missed.* The term wasn't confined to =sentry.org=. +=wrap-it-up.org= said "a crashed fire", and =todo-cleanup.el= and its test both +said "every sentry fire" — all three naming a sentry cycle. Renaming only +=sentry.org= would have split the vocabulary across files. Found by grepping +every file that mentions sentry, then re-grepping without a context window after +the first pass truncated short-line matches and hid them. + +Verified: exactly 3 verb instances left in sentry.org, no placeholder leaked, the +digest commit template now reads =<date> <time> cycle=, capitalized plurals +handled, lint 0/0 on sentry.org, suite green (exit 0 — load-bearing here, since +=todo-cleanup.el= and its test are under test). The two =wrap-it-up.org= lint +findings are pre-existing and identical at HEAD. + +Committed =f3f5bfd=, pushed, velox fast-forwarded and verified. Notified home +(with the leak it hadn't seen) and =.emacs.d=. Closed the task =DONE=. + +Four commits this session, all pushed, both daily drivers current, inbox at zero. + +** 20:00 — Roam inbox zero, then the adversarial-review design + +Roam scan: 3 items, 1 claimed (=rulesets:= prefix), 2 unowned gear links left +for Craig. Filed the claimed one, removed it from roam under capture-guard + +roam-write lock, triggered =roam-sync=. A local =.emacs.d= FYI also cleared. + +The claimed item: "code reviews must occur before every commit an agent does, +and they should be hostile reviews from a subagent without the agent's context." + +Craig's decisions: *adversarial* rather than hostile (he took my push-back that +an agent told to attack manufactures findings), plus a requirement I had missed — +a re-review loop that runs until the reviewer approves — and that the rules must +not contradict afterward. Unopposed recommendations I proceeded on: dispatch on +every commit with the reviewer's own gate deciding triviality, and the flow lands +in the =publish= skill. + +Wrote it into three files: =publish/SKILL.md= Step 1 (dispatch contract, +adversarial-with-substantiation, the loop, bounds), =review-code/SKILL.md= (the +two levels of dispatch, the adversarial contract, re-review mode), and +=claude-rules/subagents.md= (a new Isolation Override section, since three +separate size rules there said don't dispatch small work). + +** 20:30 — Dogfooded it, and the reviewer found nine things + +Ran the new flow on its own diff: dispatched an isolated adversarial reviewer +with the diff plus a one-line claim, withholding everything else. It returned +REQUEST CHANGES with seven Important and two Minor, every one substantiated. I +checked each against the files rather than accepting them, and all nine were +real: + +1. =Skipped= from Phase 0 is not =Approve=, so trivial diffs dead-ended at a gate + with no defined pass. +2. =no-approvals.org= line 73 (the actual execution step) still described the old + inline unbounded flow; I had only updated the preamble at line 49. +3. =work-the-backlog.org= and =sentry.org= — the unattended callers — had no + receiver for "stop and surface to the user". The speedrun routes to + work-the-backlog, not the file I updated. +4. The Step 2 exception still said the review runs "when it applies", which my + rewrite had made false. +5. *The sharpest one.* Withholding the ticket/plan makes the author's claim + self-certifying and strands =review-code='s Intent-vs-Delivery criterion — the + one aimed at exactly the inherited-scope error this gate exists to catch. A + ticket is not the author's model; it is the independent record of what was + asked. I had not considered this. +6. The loop turned on the verdict token, so a single Minor could burn all three + rounds and escalate. +7. My override said it "doesn't relax the Prompt Contract" while field 2 of that + contract says paste your context verbatim — which would destroy the isolation + the whole change is built on. +8. todo.org carried an unrelated =references/= rewrite, and the new task body + listed as open the questions the same commit answered. +9. =subagents.md= still says subagent output is a claim to verify, unreconciled + with "approval is the reviewer's to give". + +All nine fixed. The =references/= hunk is split out as =3a933a2=. Round 2 sent +back to the same reviewer, which is the loop working as designed. + +** 21:30 — The loop closed at three rounds + +Round 2 (three findings): =review-code='s adversarial contract still said two +inputs, so the round-1 fix landed in =publish= but not in the text the reviewer +reads — the two files disagreed at the one seam the change was about. A +count/list regression my own fix introduced ("exactly two things" over a +three-item list). And a fourth verdict I had missed entirely: =review-code= +emits =Needs Discussion=, which is exactly what an adversarial reviewer reaches +for on an architectural objection, and the flow had no handling for it. + +Fixing =Needs Discussion= reproduced round 1's finding 3 one level down, which I +caught myself: all four unattended parking clauses keyed on "the loop hits its +bound", and a first-round =Needs Discussion= is not a bound hit. Rewrote all four +to trigger on "the review can't reach approval" with the three causes named, and +asked the reviewer to verify that independently rather than take it from me. + +Round 3: APPROVE, with one Minor — a recurrence of the count/list mismatch, since +my fourth-verdict fix left the lead-in saying "three outcomes" above four +bullets. Fixed the numeral. The committed diff therefore differs from the +approved one by exactly that word, which the flow's own Minor-only rule permits +rather than spending a fourth round. + +The reviewer also verified things I had not asked about and would not have +checked: that =failed= is already a legal outcome slug in work-the-backlog's +metrics table, that the three workflow mirrors carry identical blob hashes rather +than merely similar text, and that no fifth verdict token exists anywhere in +=review-code=. It filed one follow-up correctly rather than fixing it in-diff — +=start-work.md= Phase 7 still summarizes the publish flow instead of pointing at +it, and was stale before today. Filed =[#C] :chore:solo:=. + +Convergence shape across the three rounds: nine findings, three, one Minor. The +residue in later rounds was integration error from the previous round's fixes +rather than new design problems, which is the shape the bound is calibrated for. + +Committed =8062460=, pushed, velox fast-forwarded and verified. + +Six commits this session, all pushed, both daily drivers current, inbox at zero, +tree clean. + +** 2026-07-29 — Walking the remaining items, and a long lesson + +*Item 1, the =references/= dead link* (=5999f88=). Craig picked drop-the-link. +The adversarial review returned =Needs Discussion= — the fourth verdict, on its +first real use — and widened the fix twice, both correctly. My replacement prose +said credentials "live in the rulesets repo" without naming a file, which would +have sent readers to =calendar-reference.org=, whose three paths had been dead +since May. Now names =mcp/README.org=. And the file itself was orphaned by the +link removal, so both copies and the empty =references/= dirs are gone. Round 2 +approved with three Low findings, all against the task record: my count was wrong +(seven sites, not five) and =scripts/lint.sh='s =check_md_links= already exists +for that class, missing them only because it matches markdown syntax. + +*Rule gap found by using the rule.* =Needs Discussion= exits to the user, but I +never wrote what happens after the user answers. Treated it as a fresh review on +the reasoning that Craig adjudicated and the scope changed. Flagged to Craig; not +yet written into the skill. + +*The wrap-org-table bug* (=ecd5d7b=). work reported that the tool splits a logical +row and lint passes the result. Everything I *measured* held. Everything I +*inferred* on top was refuted, four times: + +1. Root cause "absence of rules in the input" — refuted by the double-run repro + (the tool corrupts its own correct, rule-delimited output). +2. "Tested and killed the empty-cell hypothesis" — the fixture was confounded. + With no hlines the code short-circuits at =:184= before the predicate is + reached, so I varied the empty cell while the path that reads it was switched + off, got a negative, and wrote it down as settled. +3. "The reporter's row-below observation discriminates between the paths" — it + doesn't; both produce that signature. Claimed twice. +4. "A static scan can't see primary-path exposure, needs simulation" — wrong, and + worse, I sent it to work, who built on it. Then my *corrected* advice (run the + predicate over rule-delimited groups) was also wrong: work implemented it and + showed it can't discriminate at any threshold, with worked examples where the + same structural signature has opposite correct verdicts. + +Three corrections sent to work, plus a fourth acknowledging their disproof. The +review loop *bounded out* at three rounds — the first bound-out under the new +rule, working as designed. Craig adjudicated by stripping the task to Verified / +Two-fixes-that-work / Open-questions, which is the right shape: the measurements +were never the problem. + +The fix that survived is work's read, not mine: check idempotence (reflow twice, +diff) rather than build a detector. It's true by construction and needs nobody to +decide what a group means. + +*The lesson, stated plainly.* Every refutation across three rounds landed on +inference, never on a measurement. I ran experiments and then over-read them, +repeatedly, at full confidence. work — after I'd sent them three wrong analyses — +labelled their own uncertain number as a floor on a population they couldn't +cleanly define, unprompted. That discipline is the thing to copy. + +I also found the 2026-07-27 note where I'd already observed this exact failure +and left it in a session summary instead of filing it. work's framing: a correct +observation recorded and then read as fine is the same failure as a green check +on a corrupted table. diff --git a/.ai/workflows/INDEX.org b/.ai/workflows/INDEX.org index b031dbe..f18d953 100644 --- a/.ai/workflows/INDEX.org +++ b/.ai/workflows/INDEX.org @@ -58,6 +58,9 @@ This index must list every =.org= file in =.ai/workflows/= except this one and e - Speedrun triggers: "speedrun", "no approvals speedrun", "speedrun these: <task set>" — any phrase containing "speedrun" routes here (the preset), never to =no-approvals.org= - Manual triggers: "work the backlog", "work the backlog with <task set>" (file-only defaults) - Synthesis trigger: "synthesize backlog metrics" — read the per-project metrics logs, compute trends + the corrections signal, write one =:agent:metrics:= KB node (personal projects only) +- =sentry.org= — the overnight hygiene supervisor: an interval loop (default hourly) that walks a fixed pass list (roam pull, inbox zero, triage, todo cleanup, task audit, working-files hygiene, spec board, link integrity, git health, prep freshness), commits each writing pass to a throwaway =sentry/<date>-<host>= branch (never pushed), and parks every judgment call and destructive action in a morning-approval queue. Gated on =:COMMIT_AUTONOMY: yes= plus interactive entry gates (clean tree, green suite) with Craig present. Locks via =agent-lock=; morning teardown (review, squash-merge, delete) is Craig's, never automated. + - Triggers: "start sentry", "run sentry", "arm sentry", "sentry mode", "start sentry every <interval>" + - Stop trigger: "stop sentry", "stand down sentry", "sentry off" ** Calendar @@ -110,7 +113,7 @@ This index must list every =.org= file in =.ai/workflows/= except this one and e - Situational triggers: "broadcast the <event> to all projects", "broadcast that <situation>", "let every project know I'll be away ..." - =flashcard-review.org= — review an org-drill flashcard file, restructure cards to question-form headings (no answer hints), audit content accuracy against project source-of-truth via subagent, rewrite source preserving SRS state, regenerate the Anki =.apkg= to =~/sync/phone/anki/=. Person cards use "Who is X? Tell me about their Y."; talking-points cards stay as-is. Script behavior: =flashcard-to-anki.py= strips =:PROPERTIES:= drawers + =SCHEDULED:= / =DEADLINE:= planning lines from Anki output. - Triggers: "review the flashcards", "update the flashcards", "review the drill deck", "update the drill deck", "refresh the Anki cards", "let's run the flashcard-review workflow" -- =page-me.org= — set a timed notification (desktop =notify=; phone via =agent-page= when Craig is away). +- =page-me.org= — set a timed notification. "page me" desktop =notify=, "text me" phone via =agent-text=, "text and page me" both. - Triggers: anything containing the word "page" used as a verb ("page me", "page me in 10 minutes", "page me at 3pm", "page my phone") - =status-check.org= — proactive long-running-job updates. - Triggers: "keep me posted on this", "provide status checks on this job", "let me know when it's done", "monitor this for me". Auto: any job estimated 10+ min. diff --git a/.ai/workflows/code-quality.org b/.ai/workflows/code-quality.org index 3ac3e9d..3c4ed8f 100644 --- a/.ai/workflows/code-quality.org +++ b/.ai/workflows/code-quality.org @@ -9,6 +9,13 @@ One trigger that runs every behavior-preserving quality pass over a scope of orchestrator — each pass keeps its own discipline and its own confirm gate; this workflow only sequences them and collects the residue. +*Behavior-preserving rests on a test net.* The passes below claim to preserve +behavior, but a refactor on untested code is a guess, not a preservation. Where +the scope has no tests, bring it under a characterization net first +(Normal/Boundary/Error per unit, per the =testing-standards= skill's "Adding Tests to Existing +Untested Code") — that net is what turns "behavior-preserving" from an assertion +into something the green suite actually verifies across each pass. + The passes it chains: 1. =/refactor= — structural and logic cleanup on measurable metrics (complexity, diff --git a/.ai/workflows/helper-mode.org b/.ai/workflows/helper-mode.org index a6acfa7..b32d574 100644 --- a/.ai/workflows/helper-mode.org +++ b/.ai/workflows/helper-mode.org @@ -12,13 +12,14 @@ The governing fact behind every rule below: the session-context split isolates e * When to Use This Workflow -No operator trigger phrase. A helper reaches this contract one of three ways: +No operator trigger phrase. A helper reaches this contract one of two ways: - The =ai --helper= launcher routes here after the roster confirms a live agent (the deterministic path). -- Startup's roster check finds the session is not alone and routes here instead of running normal startup (the safety net for a raw =claude= launch). - An explicit "you are a helper, follow helper-mode.org" instruction (the manual fallback). -If none of those applies — the roster shows the session is alone — this is a primary session. Run normal [[file:startup.org][startup.org]], not this. +There is deliberately no third way, and the gap matters: *startup does not check the roster*. A bare =claude= launched into a project that already has a live session runs full primary startup — pulls, rsync, inbox processing — without ever reaching this file. That safety net is designed (see Status below) but unbuilt, so nothing catches a raw launch. Use =ai --helper=. + +If neither route applies, this is a primary session. Run normal [[file:startup.org][startup.org]], not this. * Identity @@ -92,10 +93,28 @@ A helper does not run normal startup. It runs a light version: When the helper's work is done: -1. Re-run the roster (=.ai/scripts/agent-roster=) to learn whether a primary is still live. +1. Re-run the roster to learn whether a primary is still live. Pass the project root explicitly — =agent-roster= defaults to =$PWD= and keeps only agents at or inside that root, so calling it from a subdirectory hides a primary sitting at the root and reports "alone": + + #+begin_src bash + root="$(git rev-parse --show-toplevel 2>/dev/null || pwd)" + if [ -x "$root/.ai/scripts/agent-roster" ]; then + "$root/.ai/scripts/agent-roster" "$root"; rc=$? + else + rc=2 + fi + echo "roster rc=$rc" + #+end_src + + Read rc as =wrap-it-up.org= Step 0 does: 1 means a primary is still live, 0 means this helper is orphaned, and 2 (or an absent script) means unavailable — which takes the same archive-only path as 1, because leaving work uncommitted is recoverable and committing under a live primary is not. 2. *Primary still live (the normal case):* finalize the Summary in the helper's own =.ai/session-context.d/<id>.org=, archive it to =.ai/sessions/YYYY-MM-DD-HH-MM-<id>-<description>.org=, and stop. Do NOT commit, push, or run hygiene — the primary's next commit picks up the archived file and any scoped edits the helper left in the tree. 3. *Orphaned helper (roster shows the helper is now alone):* the primary already exited, so the helper assumes full closing duties — the git ban lifts because the concurrency that justified it is gone. Commit and push the tree (including the helper's own edits, which would otherwise strand as a dirty tree), per the normal wrap-up flow in [[file:wrap-it-up.org][wrap-it-up.org]]. * Status -Phase 1.5 of the generic-agent-runtime spec. This contract is the canonical home; the spawn paths (=ai --helper=, startup's roster branch) and the [[file:wrap-it-up.org][wrap-it-up.org]] helper branch route here. Those wiring pieces ship behind the spec's bats-then-drills-then-pilot gate and are not yet live; until then, the manual "you are a helper" instruction is how a session adopts this contract. +Phase 1.5 of the generic-agent-runtime spec. This contract is the canonical home; the spawn paths and the [[file:wrap-it-up.org][wrap-it-up.org]] helper branch route here. + +Live now: =ai --helper <project>= (roster check, id assignment, helper opener, its own tmux window), the explicit "you are a helper" instruction, and the wrap-it-up.org Step 0 helper branch. + +Not built yet, and worth knowing because it is the gap you can fall into: *startup has no roster check*. A second session launched as a bare =claude= in a project that already has one runs full primary startup — pulls, rsync, inbox processing — with no idea another agent is live. Until that safety net exists, =ai --helper= is not merely the preferred path, it is the only one that makes a helper without being told. + +Also unbuilt: the live-helper gate that pauses a primary's file-wide hygiene passes (=todo-cleanup.el=, =lint-org.el=, =wrap-org-table.el=) while a helper is mid-edit. Data-integrity rule 1 above describes the intended behavior; nothing enforces it yet, so a primary running hygiene can still clobber a helper's just-written scoped edit. diff --git a/.ai/workflows/inbox.org b/.ai/workflows/inbox.org index 3bd9335..6faa20f 100644 --- a/.ai/workflows/inbox.org +++ b/.ai/workflows/inbox.org @@ -163,6 +163,20 @@ An org capture is usually only a few seconds of mid-finalize state, so =--wait= - *Auto inbox zero (=/loop=) cycle* → don't surface or wait further; defer the roam reconcile to the next cycle, which is itself the retry at loop cadence. The items were already filed in Phase C, so the next cycle's Phase C status-check drops the duplicates and its Phase D removes them. Note one line: "roam reconcile deferred — a capture is still open; next cycle catches it." - *Wrap-up sub-step* → don't block the wrap. Skip the roam reconcile for this run and surface one line: "Skipped roam-inbox reconcile — a live org-capture is open against it; claimed items stay and get caught next run." The items were already filed into =todo.org= in roam mode Phase C, so the next roam run's Phase C status-check drops the duplicates and its Phase D removes them — the skip self-heals. +*The roam-write lock (around the Phase D edit).* Capture-guard protects against a live *human* capture; the roam-write lock protects against a concurrent *agent* writer (a sentry inbox pass, a KB promotion) editing =~/org/roam= at the same time. Acquire it after the capture-guard clears and release it after the edit-and-trigger, so the two guards nest — capture-guard underneath, the agent lock around the write: + +#+begin_src bash +if [ -x .ai/scripts/agent-lock ]; then + .ai/scripts/agent-lock acquire roam-write --wait || { echo "roam-write busy; deferring roam reconcile" >&2; exit 1; } +fi +# capture-guard (above), then the Phase D read-modify-write of ~/org/roam/inbox.org, +# then trigger the sync — roam-sync stays the only committer: +systemctl --user start roam-sync.service +[ -x .ai/scripts/agent-lock ] && .ai/scripts/agent-lock release roam-write +#+end_src + +Degrade gracefully when =agent-lock= isn't installed (an older checkout mid-sync): the write proceeds unlocked, today's behavior. A *present* helper reporting the lock busy after its bounded wait defers the roam reconcile (the auto-loop and wrap-up paths already defer-and-retry per the fallback list above); an *absent* helper never blocks it. + * Core §6 — Priority-scheme check This gates filing whenever there are accept-and-file items. Check whether =todo.org= has a top-of-file priority scheme (an explicit legend defining =[#A]= through =[#D]= semantics and mandatory/optional tag conventions — a =* <Project> Priority Scheme= section or similar). @@ -332,7 +346,7 @@ When Craig has put the session in no-approvals mode, an accepted item may be imp 2. *Quick* — the whole implementation, including verification, is under ~15 minutes. 3. *Solo* — you can carry it end to end without a decision from Craig. Manual verification you perform yourself is fine; needing Craig to choose an option, approve a design, or resolve an ambiguity is not. -All three → implement it, verify, then commit and push at the end of that item (the Step 0 reconcile and pre-push check from =commits.md= still run). Miss any one and it doesn't self-apply: a shared-asset or convention change needs Craig's decision, so it fails *solo* and routes to the defer-and-stage park (core §2 / core §3); an oversized item fails *quick* and gets filed. +All three → implement it, verify, then commit and push at the end of that item (the Step 0 reconcile and pre-push check from the =publish= skill still run). Miss any one and it doesn't self-apply: a shared-asset or convention change needs Craig's decision, so it fails *solo* and routes to the defer-and-stage park (core §2 / core §3); an oversized item fails *quick* and gets filed. ** Replying to handoffs @@ -355,6 +369,8 @@ If either can't be satisfied — a half-done item, a failure introduced during t Reads the *global roam inbox* (=~/org/roam/inbox.org=), Craig's cross-project GTD capture: one shared file every project can see. This mode routes each roam item to the project that owns it. The current session claims only the items belonging to THIS project, files them into the project's =todo.org=, and removes them from the shared inbox. Everything it doesn't own stays. +*Allowed from any project, work included.* Tidying the shared roam inbox is housekeeping on a shared resource, not a cross-project boundary crossing and not a durable KB-node write, so the =knowledge-base.md= work-denylist doesn't gate it (a sentry inbox-zero pass mis-parked the whole inbox as a boundary crossing from the work project on 2026-07-19 — the error this note closes). Reading roam and tidying its inbox are fine from work; only promoting a durable =agents/= node stays work-denylisted. + The aspiration is inbox zero: after this mode runs, the current project's local handoff inbox has been processed (Phase A delegates to process mode) and the shared roam inbox no longer contains items explicitly owned by this project. This is distinct from the wrap-up inbox/transcript routing feature (which moves session-filed keepers between projects). This routes the shared roam capture file by ownership prefix. @@ -466,14 +482,14 @@ A recurring, *interactive* roam check. Trigger phrase: "auto inbox zero" (match ** Per cycle 1. Run roam mode's scan (Phase A local check + Phase B roam scan), read-only — no =git pull=. The capture-guard still gates any write: use =capture-guard --wait= (core §5) so a transient capture clears itself; if it's still open after the wait, *defer this cycle's roam reconcile to the next cycle* rather than surfacing — the loop cadence is the retry, and the filed items get swept next time. The rare write hands its git to =roam-sync= (roam Phase D). -2. *Nothing found* → no inbox summary. One acknowledgement line: =ran at HH:MM, nothing found=. Nothing else. The acknowledge-only-on-empty rule keeps a quiet inbox quiet. +2. *Nothing found* → no inbox summary. One heartbeat line: =inbox zero at HH:MM: nothing= (HH:MM local, from =date=) — the silent-until-signal policy, see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=. Nothing else. Keeping a quiet inbox quiet is the whole point. 3. *Items found* → summarize the found items, file them as tasks (roam Phase C), and *append them to a displayed queue* — the harness task list, via =TaskCreate= — so the queue accumulates across cycles. Then ask: "run this batch next?" - *Yes* → chain into =work-the-backlog.org= as an explicit second step after routing completes: pass it the eligibility query over the queued items (status =TODO= + =:solo:= per the scheme header, priority-ordered), =file-only= mode, paging off, cap 1. The highest-priority eligible candidate runs; the rest wait for the next tick or a later yes. - *No* → they stay queued for a later go. This mode never implements anything itself — routing ends here, and the execution loop lives in =work-the-backlog.org=, its one home. 4. *Cross-cycle dedup.* Subsequent cycles add only *newly-found* items to the same displayed queue, never re-surfacing what's already there. Dedup against the queue (the =TaskCreate= list), not against what's already been implemented — a find that was queued-but-not-yet-run must not reappear, and one already filed into =todo.org= is dropped by roam Phase C's status check. -A find is always surfaced and filed; execution happens only through the =work-the-backlog.org= chain and waits for Craig's yes. A quiet inbox produces only the timestamped acknowledgement. =auto inbox zero= is inherently in-session because its chain step waits for that yes. +A find is always surfaced and filed; execution happens only through the =work-the-backlog.org= chain and waits for Craig's yes. A quiet inbox produces only the =inbox zero at HH:MM: nothing= heartbeat. =auto inbox zero= is inherently in-session because its chain step waits for that yes. ** Fully-unattended pass (=/schedule=) — vNext, not v1 diff --git a/.ai/workflows/no-approvals.org b/.ai/workflows/no-approvals.org index 5f54b96..6b5c7fa 100644 --- a/.ai/workflows/no-approvals.org +++ b/.ai/workflows/no-approvals.org @@ -35,7 +35,7 @@ Mode resets when: The interaction gates that step the workflow back to Craig for an "OK to proceed?" check: -- The commit-message gate in =commits.md= Step 2. Print the final commit message inline, then commit immediately. No "approve / request changes / open in editor" prompt. +- The commit-message gate in the =publish= skill, Step 2. Print the final commit message inline, then commit immediately. No "approve / request changes / open in editor" prompt. - The PR-description gate. Print the final body, then create the PR. - The PR-review-reply gate. Print the final reply, then post. - "Ready to start?" / "Plan looks like X, proceed?" gates before implementation work begins. @@ -46,10 +46,10 @@ The interaction gates that step the workflow back to Craig for an "OK to proceed The engineering-discipline gates protect quality, not Craig's interaction time. They remain in force: -- =/review-code= against the staged diff before every commit. Critical and Important findings still block. Minor findings still surface. No "proceed anyway" override unless Craig has given it explicitly for this batch. +- =/review-code= against the staged diff before every commit, dispatched as an isolated adversarial reviewer per the =publish= skill's Step 1 — no-approvals removes *interaction* gates, never the isolation. Critical and Important findings still block, and the re-review loop still runs to approval. Minor findings still surface. No "proceed anyway" override unless Craig has given it explicitly for this batch. If the review can't reach approval — three rounds, a recurring finding, or a =Needs Discussion= verdict — that is a genuine question: park the item per step 4 and move to the next one rather than committing past a standing finding. - =/voice personal= on every publish artifact (commit messages, PR titles + bodies, PR review comments). The full pattern walk happens. The printed result just doesn't wait for approval. - The full test suite + lint + compile before commit (per =verification.md=). -- Fetch-and-reconcile in =commits.md= Step 0. +- Fetch-and-reconcile in the =publish= skill, Step 0. - Session Log updates per =protocols.org=. Every state-mutating turn writes to =.ai/session-context.org= before the closing message. The log is the crash-recovery anchor while Craig is away. Missing entries lose work. - Subagent review-gate cadence (=subagents.md=). Review each subagent's output before the next dispatch. - Destructive or irreversible operations per =CLAUDE.md='s "Executing actions with care": force-push, =rm -rf=, dropping a column, dropping a branch, package removal. These need explicit consent regardless of mode. No-approvals is for *interaction* gates, not destructive-action consent. @@ -70,7 +70,7 @@ For each item: - Do the work. - Update the Session Log per the rules in =protocols.org=. -- Before any commit: run =/review-code= against the staged diff. Surface Critical and Important findings inline; fix them and re-review until clean. Minor findings show but don't block. +- Before any commit: dispatch the isolated adversarial reviewer per the =publish= skill's Step 1 — never review your own staged diff inline. Surface Critical and Important findings; fix them and send the updated diff back to the *same* reviewer until it approves. Minor findings show but don't block and never earn another round. If the review can't reach approval — three rounds, a finding that recurs after being reported fixed, or a =Needs Discussion= verdict — park the item per step 4 with the standing findings and move on; don't commit past a blocking finding. - Draft the commit message. Run =/voice personal= (the skill, or walk the patterns inline if unavailable). Print the final message inline before committing so the log shows it. - Commit and push. - One-line status between items ("Task X done, on to Y.") so Craig knows what's happening when he checks back in. diff --git a/.ai/workflows/page-me.org b/.ai/workflows/page-me.org index bfa92c6..7a3b792 100644 --- a/.ai/workflows/page-me.org +++ b/.ai/workflows/page-me.org @@ -13,9 +13,17 @@ Uses the =notify= command (info type) for consistent notifications across all AI Craig says *"page me"* (or variations like "page me in 10 minutes", "page me at 3pm"). -The word "page" is the trigger for this workflow. It means: set a timed notification. +The word "page" is the trigger for this workflow. It means: set a timed notification on the *desktop* channel (=notify=). -Previously called "set-alarm" -- renamed to "page-me" for a distinctive, short trigger phrase that won't collide with common words like "remind" or "alert." +Two sibling triggers pick a different channel; the timed =at= machinery below is identical for all three, only the fired command changes: + +- *"page me"* — desktop =notify= (this workflow's default). +- *"text me"* — a Signal push to Craig's phone via =agent-text= (the away channel). +- *"text and page me"* — both, for when he might be either place. + +Scope the triggers to the reflexive "me": "page me" and "text me", not a bare "page" or "text" in prose. The full channel vocabulary lives in protocols.org "Reaching Craig". + +"page" was chosen (renamed from the old "set-alarm") for a distinctive, short trigger that won't collide with common words like "remind" or "alert". * Problem We're Solving @@ -113,18 +121,18 @@ notify info "Page" "Your message here" --persist The =--persist= flag keeps the notification on screen until manually dismissed. All page-me notifications should use =--persist= by default. -** Paging Craig's phone (away from the machine) +** Texting Craig's phone (the "text me" channel) -The timed =notify= alarm above is the desktop channel. When Craig is away from the machine (or asks to be paged "on my phone"), use the agent pager instead — a Signal push to his phone from any machine or agent runtime: +The timed =notify= alarm above is the desktop channel. When Craig says "text me" (or a run expects him away from the machine), use =agent-text= instead, a Signal push to his phone from any machine or agent runtime: #+begin_src bash -agent-page "Build finished — ready for your eyes" +agent-text "Build finished, ready for your eyes" -# Timed phone page: same at-daemon pattern, different channel -echo "agent-page 'Meeting starts in 5'" | at 3:25pm +# Timed phone message: same at-daemon pattern, different channel +echo "agent-text 'Meeting starts in 5'" | at 3:25pm #+end_src -Channel selection and the pager's mechanics live in protocols.org "Paging Craig — the agent pager". When in doubt, fire both: the desktop notification persists for whenever he returns, the phone push reaches him now. +Channel selection and the mechanics live in protocols.org "Reaching Craig". On "text and page me", fire both: the desktop notification persists for whenever he returns, the phone push reaches him now. ** Managing Alarms diff --git a/.ai/workflows/sentry.org b/.ai/workflows/sentry.org new file mode 100644 index 0000000..b25fc14 --- /dev/null +++ b/.ai/workflows/sentry.org @@ -0,0 +1,227 @@ +#+TITLE: Sentry — Overnight Hygiene Supervisor +#+AUTHOR: Craig Jennings +#+DATE: 2026-07-19 + +* Overview + +Sentry is an interval loop that keeps a project's hygiene current while Craig is away. Each cycle walks a fixed list of passes — roam pull, inbox zero, triage (no mail or messengers), todo cleanup, task audit, working-files hygiene, spec board, link integrity, git health, prep freshness, bug and refactor finding, and (opt-in) solo-task implementation — and commits each pass's writing to a throwaway daily branch. Nothing pushes. In the morning Craig reviews the branch, squash-merges what he wants, and deletes it. + +The design goal is a project that greets the morning already tidy, with every judgment call and every destructive action parked in an approval queue rather than executed unattended. Sentry does the mechanical sweeping; Craig does the deciding. + +This file is the engine. It owns the entry gates, the branch mechanics, the lock model, the per-cycle pass runner, the digest and approval queue, the skip semantics, and the stop-sentry shutdown. The =agent-lock= helper (=.ai/scripts/agent-lock=) provides the locks. The passes reuse existing workflows (=inbox.org=, =triage-intake.org=, =clean-todo.org=, =task-audit.org=) under sentry's unattended contract. + +* When to Use This Workflow + +Craig arms sentry at the end of a session, with the machine left running, to have overnight hygiene done by morning. + +Triggers: + +- "start sentry", "run sentry", "arm sentry", "sentry mode" +- "let sentry watch this overnight", "keep this tidy overnight" +- "start sentry hourly", "start sentry every <interval>" (sets the loop interval) + +Stop trigger (see Stop Sentry below): + +- "stop sentry", "stand down sentry", "sentry off" + +Sentry is deliberately *not* auto-armed. Running it in a project is a per-project grant (the =:COMMIT_AUTONOMY:= marker) plus a deliberate launch with Craig at the terminal for the entry gates. + +* Prerequisite — the autonomy ticket + +Sentry commits unattended. =commits.md= gates commits on Craig's approval, so sentry needs standing, per-project authorization to run at all. Before anything else, read the project's =.ai/notes.org= Workflow State block for: + +: :COMMIT_AUTONOMY: yes + +If the marker is absent or not =yes=, decline to start and name the marker: + +: Sentry needs ":COMMIT_AUTONOMY: yes" in .ai/notes.org Workflow State to run — it commits unattended. Add it to grant, or run the hygiene passes by hand. + +No half-running mode: a project without the grant doesn't run sentry's read-only passes either. The grant is one line away, so this is a deliberate opt-in, not a barrier. + +A second, *independent* marker gates the solo-task implementation pass (pass 12): + +: :SENTRY_MAY_IMPLEMENT: yes + +=:COMMIT_AUTONOMY:= lets sentry commit its hygiene sweeps to the branch; =:SENTRY_MAY_IMPLEMENT:= additionally lets it implement solo, decision-free backlog tasks on the branch. The split exists because the two carry different morning costs: hygiene is a two-minute merge, implemented code is a review session. A project can run hygiene-only sentry without the implement pass, and most should until sentry has quiet weeks behind it. Absent =:SENTRY_MAY_IMPLEMENT:=, pass 12 skips; sentry still runs every other pass. Requires =:COMMIT_AUTONOMY:= alongside it — implementing implies committing. + +* Entry — interactive, with Craig present + +Craig types the sentry trigger, so the first moves run with him at the terminal. Do them in order; each gate that fails stops entry until Craig answers. + +1. *Autonomy ticket* — the prerequisite above. Absent → decline and stop. + +2. *Dirty-tree gate.* =git diff --quiet HEAD= (tracked modifications only; untracked and gitignored files never block — an inbox drop or scratch file is not in-progress work). If the tracked tree is dirty, describe what's dirty and offer, inline-numbered per =interaction.md=: + + 1. Finish the job — commit the in-progress work first (recommended if it's a coherent unit) + 2. Stash it — =git stash= and start sentry on a clean tree + 3. Roll back named changes — discard specific files (names them) + + Wait for an answer. Sentry can't start unattended from a dirty state; that's the point. + +3. *Green-suite gate.* Run the project's full suite (=make test=, or the project's equivalent — detect it). Read the output. If anything is red, describe the failures and offer to investigate before arming. The loop starts only on a green baseline, because every unattended cycle measures itself against "did I break this?" and a pre-existing red poisons that check. + +4. *Prior sentry branch.* =git branch --list 'sentry/*'=. An unmerged =sentry/*= branch from a previous night means the morning review didn't happen. Surface it and offer to squash-merge or delete it now (Craig is present); don't stack a second sentry branch on the first. + +5. *Reconcile the project branch.* Fetch and fast-forward-only against upstream — the same reconcile =startup= runs: + + : git fetch --all --prune + : git rev-list --left-right --count @{u}...HEAD + + Zero-behind → continue. Behind-only and clean → =git merge --ff-only @{u}=. Diverged → surface to Craig (he's present); don't auto-resolve. + +6. *Create the daily branch.* From HEAD: + + : git switch -c "sentry/$(date +%F)-$(uname -n)" + + The host suffix (=uname -n=) stops a same-date collision between the two daily drivers. The working tree now sits on this branch overnight — the launch hands the repo to sentry until the morning merge. Reclaiming it mid-night means stopping sentry first (see Stop Sentry). Note the Emacs buffer-revert caveat to Craig if he has the repo open: files change on disk under him overnight, so buffers want reverting after the morning merge (see =emacs.md=). + +7. *Arm the loop.* Start =/loop= at the interval (default hourly; Craig's "every <interval>" phrase overrides) with the per-cycle body being one sentry cycle (the Pass Runner below). Confirm the arming in one line: interval, branch name, project. + +* The lock model + +Two locks, both served by =.ai/scripts/agent-lock= (names only; the helper owns the paths, which live on tmpfs under =$XDG_RUNTIME_DIR/agent-locks/=, host-local and cleared on reboot). + +*Single-runner lock* (=sentry-<project>=, where =<project>= is the repo-root basename: =basename "$(git rev-parse --show-toplevel)"= — the same derivation =wrap-it-up.org='s guard uses, so the two agree on the lock name). Each cycle acquires it at cycle start and releases it at cycle end, and refreshes it between passes (the heartbeat, so a live cycle's lock never ages past one pass). If =/loop= fires again while a previous cycle still holds it, the new cycle's acquire fails and the cycle skips with one digest line — no two cycles run at once. The bounded wait is short (a few seconds); a live cycle means defer, not queue. + +*Roam-write lock* (=roam-write=). A pass that edits a file under =~/org/roam= acquires it, runs =capture-guard --wait= (the human-capture layer stays underneath), edits the working tree, triggers =systemctl --user start roam-sync.service=, and releases. The lock spans only edit-plus-trigger. Sentry never runs =git= against =~/org/roam= — roam-sync stays the repo's only committer (the 2026-06-24 one-git-owner rule). Pass 1's =pull --ff-only= is the sole, read-only exception. + +Every reclaim of a stale lock surfaces in the digest — the helper prints the reclaim note, and the cycle records it. A reclaim during a genuinely slow pass is possible, so it's never silent. + +* The Pass Runner — one contract per pass + +Each cycle, after acquiring the single-runner lock and verifying branch state (below), walks the pass list in order. Every pass follows the same four-step contract: + +1. *Probe* — a cheap existence check for the pass's target (named per pass below). Absent → the pass is one skip line in the digest and nothing more. This is what makes the pass list portable: passes self-activate where their target exists and stay silent elsewhere, with zero per-project configuration. + +2. *Work* — run the pass under the unattended contract. Quick, solo, already-agreed mechanical actions execute. Anything destructive or requiring judgment does *not* execute — it appends to the morning-approval queue (what, why, the exact command or edit that fires on approval). A pass runs fully or not at all; there is no reduced-form pass. + +3. *Session-context entry* — a pass that does or queues work appends its digest line to the =session-context.org= Session Log (path resolved via =.ai/scripts/session-context-path=) before its commit, so a crash between them still leaves the trail. Per-pass lines for an all-quiet cycle (every pass probe-skipped or no-op) are not written one by one — the cycle collapses to a single heartbeat at cycle-end (below), so an idle cycle doesn't spray one skip line per pass. + +4. *Commit* — if the pass wrote to disk, commit it: =chore(sentry): <pass> — <what changed>=. One commit per writing pass. A probe-skip or a no-op pass writes nothing and commits nothing. + +Between passes, refresh the single-runner lock (=agent-lock refresh sentry-<project>=) — the heartbeat. + +** Branch-state verification (cycle start, before the passes) + +After acquiring the lock, confirm the cycle is safe to run: + +- *On the right branch* — HEAD is =sentry/<today>-<host>=. If the loop was armed on a prior day and crossed midnight, the branch keeps the arming date; that's fine, morning teardown handles it. If HEAD is somehow *not* a sentry branch (an interrupted stop, a manual checkout), skip the whole cycle with a digest line rather than committing onto main. +- *Clean of foreign changes* — =git diff --quiet HEAD= excluding the spine set (=session-context.org= / =session-context.d/=, resolved via =session-context-path=). Sentry's own spine writes must not trip this; a genuinely unexpected dirty tree (something outside the spine changed and wasn't committed by a prior pass) poisons the cycle — skip it with a digest line, the next cycle retries. + +* Unattended safety — skip, never degrade + +With no one at the terminal, any unsafe state makes the affected scope skip with one digest line, and the next cycle retries. Unsafe states and their scope: + +- *Unexpected dirty tree* (non-spine) → skip the whole cycle. +- *Lost or un-acquirable single-runner lock* → skip the cycle (another cycle holds it, or the helper is missing). +- *A pass's own precondition unmet* (its probe fails, or a dependency is dirty) → skip that pass only. +- *Red suite at cycle-end* (see below) → the commits stay on the branch, flagged in the digest for morning review; the cycle doesn't roll back. + +Skips are never silent and never partial. Inside a *working* cycle, a pass line means the pass fully ran and a skip line names why it didn't. An *all-quiet* cycle is not a silent skip either: its single =sentry at HH:MM: nothing= heartbeat is the explicit record that every pass found nothing, standing in for a wall of identical skip lines. The anti-silence rule targets a pass that hides work it should have surfaced; a quiet cycle has surfaced that there was none. + +** Multi-day stall notification + +An unmerged prior =sentry/*= branch at cycle start (the morning review never happened) skips the cycle. After the *second consecutive* cycle skipped for this reason, send one persistent desktop notification naming the project and branch: + +: sentry stalled: <branch> unmerged — merge or delete to resume + +Then repeat at most daily. Persistent notify matches the paging convention — it stays on screen until dismissed. A multi-day stall never stays silent. + +* The pass list (v1) + +In order. Each names its detection probe. A pass whose probe fails is one skip line. + +1. *Roam pull* — =git -C ~/org/roam pull --ff-only=. Probe: =~/org/roam= is a git clone. Skipped when the roam tree is dirty (roam-sync owns that case) or the clone is absent. Read-only and ff-only — the one narrow exception to "don't touch roam git," so later passes read a fresh tree. + +2. *Inbox zero* — run =inbox.org= roam mode under the no-approvals contract: quick+solo+agreed items execute, shared-asset and convention proposals park (prepared diff, =VERIFY= task, sender reply) in the approval queue. Edits to =~/org/roam/inbox.org= take the roam-write lock + =capture-guard=. Probe: the roam clone or a project =inbox/= exists. Tidying the shared roam inbox is allowed from *any* project session, work included — it's housekeeping on a shared resource, not a durable KB-node write, so the work-denylist doesn't gate it (=knowledge-base.md=). Never park it as a cross-project boundary crossing. + +3. *Triage intake — mail and messenger sources excluded.* Run =triage-intake.org=, loading only its non-mail, non-messenger source plugins (calendar, PR/ticketing). The mail and messenger plugins — cmail, any Gmail variant, Telegram, Signal, chat DMs — are never loaded by a sentry cycle: Craig ruled 2026-07-21 that sentry doesn't check email or messengers. A manual "triage intake" still scans everything. Probe: the project has at least one *active* triage source that survives that exclusion — a project-specific plugin (=.ai/project-workflows/triage-intake.*.org=), or a non-empty =:TRIAGE_SOURCES:= declaration naming general plugins that exist. Mere presence of the template-synced general plugins does *not* activate the pass; a project that declares no sources, or whose only declared sources are mail or messengers, probe-skips (see =docs/specs/2026-07-20-triage-source-activation-spec.org=). Destructive actions (deleting, archiving, sending) queue; they never cycle unattended. + +4. *Todo cleanup* — the =clean-todo.org= mechanics (hygiene pass + =--archive-done= + =--convert-subtasks=). Probe: a root =todo.org=. Note that =--archive-done= is not purely an org-file pass on its first run in a project: it creates =archive/task-archive.org= and appends a =.gitignore= entry, so it produces a real tracked-file commit and correctly trips the cycle-end conditional suite. (archangel, first live run 2026-07-21.) + +5. *Task audit* — the *mechanical subset* of =task-audit.org= hourly (staleness counts, structural checks, cookie recomputation); the judgment half (priority regrades, consolidations, merge candidates) runs *once per night* and queues its findings rather than repeating them every cycle. Probe: a root =todo.org=. A full audit every hour is too heavy and re-surfaces the same judgment calls all night. (takuzu, first live run 2026-07-21.) Factual staleness fixes that are unambiguous still execute. + +6. *Working-files hygiene* — flag =working/<slug>/= directories whose backing task is closed (a filing candidate per =working-files.md=). Probe: a =working/= directory exists. The filing itself queues (it's a judgment move). + +7. *Spec status board* — the =docs-lifecycle= grep for spec keywords, surfacing any =DOING= spec whose bound build parent is closed. Probe: =docs/specs/= exists. + +8. *Link integrity* — broken =file:= links in the project's org files, via =lint-org.el=. Probe: =lint-org.el= present. Report-only into the digest; no unattended rewrites. + +9. *Git health* — uncommitted drift, unpushed commits on other branches, stale branches, main-behind-origin. Probe: =.git=. Report into the digest. + +10. *Prep + symlink freshness* — stale daily-prep docs, broken symlinks. Probe: the prep dir / symlinks exist (work and home only, in practice). + +11. *Bug and refactor finding* — hunt for real bugs and worthwhile refactoring opportunities in the project's codebase: static analysis (=shellcheck= for shell, the project's own linters for its languages), config sanity checks, plus one targeted code-reading area per cycle. Rotate the area across cycles and name it in the digest, so coverage accumulates over a night instead of re-reading the same corner. Randomized property sweeps (generate-and-verify against an engine's own invariants) are good quiet-cycle work here, reaching past a frozen test corpus. Expect the pass to go honestly quiet after the first few cycles find the standing defects; a quiet hunt is a result, not a failure. (takuzu, first live run 2026-07-21: three real fixes in the first four cycles, then quiet.) This pass does *not* run the test suite — the entry baseline already ran it, and re-running it hourly is anti-pattern 5; read the entry result instead. Probe: the project carries a codebase — source under version control beyond its org and tooling files. File each verified bug as a graded task in =todo.org= per the severity × frequency matrix (=todo-format.md=), and each refactoring opportunity as a =:refactor:= task, deduped against existing tasks; an unverifiable suspicion is a digest line, not a task. *Find, never fix in this pass* — the finding files a task and stops. A fix happens only in the opt-in implementation pass below, and only after the finding is a filed task that pass then re-verifies from scratch (see the premise rule there). A freshly-found "bug" can be a misread — one was filed and retracted two cycles apart on 2026-07-23 — so the file-then-verify-then-fix pipeline is deliberate: the task is the checkpoint, not a same-breath fix. (Added at Craig's order 2026-07-21, first dogfooded in dotfiles; refactor-finding added 2026-07-24.) + +12. *Solo-task implementation (opt-in — =:SENTRY_MAY_IMPLEMENT:=)* — work the backlog's solo, decision-free tasks on the branch. Probe: =.ai/notes.org= Workflow State carries =:SENTRY_MAY_IMPLEMENT: yes= *and* the project holds =:COMMIT_AUTONOMY:= (the implement pass commits). Absent the marker, skip — this pass is off by default, because it turns the morning from a two-minute merge into a code review, and that's the project owner's call. When on: invoke =work-the-backlog.org= under its unattended-loop contract (no pre-flight Q&A — there's no Craig overnight), eligibility =TODO= + =:solo:=, with the defer checklist deciding act-vs-file. The overnight-only tightening: only the *ready* bucket implements (clears every checklist item with zero open decisions); a task needing even one quick decision defers to a =VERIFY= rather than guessing, exactly as the loop caller already does. Commit each logical change to the sentry branch; *never push* — the morning review and merge is the gate, same as every other pass. The full quality bar holds (TDD, suite green before each commit, the isolated adversarial review per =publish= Step 1 with its re-review loop, =/voice=), and the review here runs the *premise check first*: reproduce the bug or confirm the problem is real before judging the diff. The review is the fact-checker that a filed claim never got, and it is what makes fixing-on-a-branch safe (Craig, 2026-07-24). A task that fails its premise check is not implemented — the finding was wrong, and that outcome is a digest line, not a commit. A task whose review never reaches approval — three rounds, a recurring finding, or a =Needs Discussion= verdict — is the same shape: no commit, and a digest line naming the standing findings, so the morning review sees what the reviewer would not pass rather than finding the task silently absent. (Added at Craig's direction 2026-07-24: overnight implement-on-branch, gated and never-pushed.) + +(KB lesson promotion — the pass the original proposal listed eleventh — is deferred to vNext. An unattended judgment pass writing to the shared knowledge base waits until sentry has quiet weeks behind it and a designed detection heuristic. See the filed lesson-detection-heuristic task.) + +* Cycle-end — conditional suite, then the digest commit + +After the passes: + +1. *Conditional suite run.* If any pass this cycle modified files *outside* the org/spine set (a code-touching pass, rare but possible via fixtures), run the full suite once. A green run confirms the cycle's commits are safe; a red run flags the digest for morning review — the commits stay on the branch (nothing is pushed, so the morning gate catches it). No per-pass suite runs: the entry run is the green baseline, and hourly per-commit runs would turn a seconds-long cycle into minutes all night. Cycles that only touched org/spine files skip this. + +2. *Heartbeat or digest, then commit.* Decide quiet vs working. A *quiet* cycle — every pass probe-skipped or no-op, nothing added to the approval queue — writes a single heartbeat line to the Session Log, =sentry at HH:MM: nothing= (HH:MM local, from =date=), and no per-pass digest block. A *working* cycle — any pass ran, wrote, or queued — writes its full per-pass digest block. Then commit any accumulated spine writes in one sweep: =chore(sentry): digest — <date> <time> cycle= for a working cycle, =chore(sentry): heartbeat — <date> <time>= for a quiet one, so even a quiet cycle leaves a clean tree for the next branch-state check (where the spine is untracked, the mirror-only case, there is nothing to commit and the heartbeat line stays in the working-tree anchor). This is the silent-until-signal policy (see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=): an all-quiet night collapses from a wall of no-op digests to a list of one-line heartbeats, while a cycle that actually did or queued something still writes the full record. + +3. *Release the single-runner lock.* + +* The digest and the approval queue + +*Digest.* A *working* cycle appends its block to the =session-context.org= Session Log (the spine the cycle already writes), so it survives a crash, rides the session archive, and is on screen in the running session. One block per working cycle: the timestamp, then one line per pass (ran + what, or skipped + why), plus any lock reclaim notes. A *quiet* cycle (nothing done or queued) writes no block — just the one heartbeat line =sentry at HH:MM: nothing= (the silent-until-signal policy). The per-pass block is a working-cycle artifact; it still carries one line per pass so a real skip inside a working cycle is never hidden. + +*Approval queue.* Destructive and judgment actions accumulate under one heading in the same file — =* Sentry approval queue (<date>)= — newest last. Each item carries three things: *what* (the action), *why* (what triggered it), and the *exact command or edit* that fires on approval. The morning review is Craig reading this heading top to bottom and running or discarding each item. + +* Morning teardown — Craig's, documented not automated + +Sentry never merges its own branch. In the morning Craig: + +1. Reviews the digest and the approval queue in =session-context.org=. +2. Runs or discards each approval-queue item. +3. Reviews the branch: =git log main..sentry/<date>-<host>= and the diff. +4. Squash-merges what he wants (=git switch main && git merge --squash sentry/<date>-<host>=, then one clean commit) or cherry-picks selectively. +5. Deletes the branch: =git branch -D sentry/<date>-<host>=. +6. Reverts any Emacs buffers still showing the pre-merge on-disk state (=emacs.md= buffer-revert caveat). + +A bad night is discarded by deleting one branch — nothing reached main, nothing was pushed. + +In a project that gitignores =.ai/=, the whole spine is untracked, so quiet cycles produce no commits at all and =git log main..sentry/<date>-<host>= understates the night's activity. There the anchor's heartbeat list is the only record of what fired. Read the anchor, not just the log. (archangel, first live run 2026-07-21.) + +* Stop Sentry + +Trigger: "stop sentry" (and synonyms above). Sentry owns its own shutdown: + +1. *Cancel the loop* — stop the =/loop= (=ScheduleWakeup= stop / the loop's stop path). No further cycles. +2. *Release the single-runner lock* if this context holds it. +3. *Branch disposition* — offer, inline-numbered: + 1. Squash-merge the day's branch into main now (walk the morning teardown steps 3-5 interactively) + 2. Leave it named for later review (=sentry/<date>-<host>= stays; review at leisure) +4. *Approval queue* — offer to walk the queued items now, or carry them (they stay under the heading for whenever Craig reviews). + +Stopping sentry is the only way to reclaim the working tree mid-night. The entry gate fronts the handoff; stop-sentry ends it. + +* Wrap-up interaction + +=wrap-it-up.org= refuses while sentry is live: it detects the single-runner lock (=agent-lock status sentry-<project>= → held) and stops with "sentry is active — say 'stop sentry' first." The shutdown logic lives here, not in wrap-up; wrap-up carries only the one guard. + +* Common Mistakes + +1. *Running without the =:COMMIT_AUTONOMY:= grant* — sentry commits unattended; the marker is the entry ticket, and its absence is a hard stop, not a degrade. +2. *Starting from a dirty or red tree* — the entry gates exist because an unattended cycle can't tell Craig's in-progress work from a regression. Answer the gate; don't bypass it. +3. *Committing onto main* — every writing pass commits to the daily =sentry/*= branch. A cycle that finds HEAD off the sentry branch skips rather than commits. +4. *Running a =git= write against =~/org/roam=* — roam-sync is the only committer. Sentry edits the tree under the roam-write lock and triggers the sync; it never commits or pushes roam. +5. *A per-pass suite run* — the suite runs at entry (baseline) and conditionally at cycle-end (only when a pass touched non-org files). Hourly per-commit runs all night is the anti-pattern the suite policy exists to prevent. +6. *Executing a judgment or destructive action unattended* — those queue for the morning with their exact command. The pass did its detection; Craig makes the call. The one sanctioned exception is pass 12's solo-task implementation, and only because it inherits work-the-backlog's full defer checklist (data-loss and irreversible actions defer, never execute) plus a premise-verifying review, and it commits to the branch rather than acting on anything live. +7. *A silent skip* — inside a working cycle, every skip writes a digest line naming why; a missing pass with no line reads as "ran clean" when it didn't. The one exception is not a violation: an all-quiet cycle collapses to a single =sentry at HH:MM: nothing= heartbeat instead of one skip line per pass — the heartbeat is the explicit "nothing to do" record, per the silent-until-signal policy. +8. *Degrading a pass to a reduced form* — a pass runs fully or skips. No half-passes. +9. *Letting an unmerged branch stall silently* — after two consecutive unmerged-branch skips, the persistent desktop notify cycles. Don't suppress it. +10. *Merging sentry's branch automatically* — the morning teardown is Craig's. Sentry creates and commits; it never merges or deletes its own branch. + +* Living Document + +Sentry ships with eleven finding/hygiene passes, one opt-in implementation pass, and a deferred KB pass. The pass list, the interval default, the =:SENTRY_MAY_IMPLEMENT:= default, and the queue-vs-execute line for each pass are the knobs most likely to move with dogfooding. The implement pass especially is new (2026-07-24) and unproven at scale — watch the corrections signal (work-the-backlog's metric for autonomous commits later reverted or hand-fixed) before widening it past the projects that opt in. Fold in what the live trial surfaces — a pass that queues too eagerly, a probe that misfires, a digest line that wants more detail. Refine as the signal arrives. + +* History + +Built 2026-07-19 from the sentry spec (=docs/specs/2026-07-14-sentry-workflow-spec.org=, ID f6c51f27-d7a2-4b63-9ff9-5ba005a66dfb) — 10 decisions and 12 review findings resolved before the build. Phase 1 shipped the =agent-lock= helper (commit =a8b6cf4=); this file is Phase 2, the engine. Phase 3 reconciles the roam writers (=inbox.org=, =knowledge-base.md=) to acquire the roam-write lock and adds the =wrap-it-up.org= guard. diff --git a/.ai/workflows/startup.org b/.ai/workflows/startup.org index 929d482..2262eea 100644 --- a/.ai/workflows/startup.org +++ b/.ai/workflows/startup.org @@ -29,10 +29,16 @@ Inside a rulesets session, the project-repo refresh below covers this — the ru #+begin_src bash rs="$HOME/code/rulesets" if [ -d "$rs/.git" ]; then - if (cd "$rs" && git diff --quiet --ignore-submodules HEAD -- 2>/dev/null); then + gate="$rs/claude-templates/bin/git-worktree-gate" + if [ -x "$gate" ] && "$gate" sync-safe "$rs"; then + (cd "$rs" && git pull --ff-only origin main 2>&1) | tail -3 + elif [ ! -x "$gate" ] \ + && (cd "$rs" && git diff --quiet --ignore-submodules HEAD -- 2>/dev/null); then + # Bootstrap fallback for a checkout old enough not to have the shared + # gate yet. The pull that follows installs it for subsequent starts. (cd "$rs" && git pull --ff-only origin main 2>&1) | tail -3 else - echo "rulesets: dirty working tree — using as-is, skipping pull" + echo "rulesets: changes beyond untracked inbox deliveries — using as-is, skipping pull" fi else echo "rulesets: not a git checkout — skipping" @@ -40,11 +46,11 @@ fi #+end_src Behavior: -- *Clean working tree* → fast-forward pull. =git pull --ff-only= refuses any merge or rebase, so the operation is either a no-op (already current) or a clean advance. -- *Dirty working tree* → skip the pull. Don't auto-stash and don't auto-merge — those would either lose work or invite conflicts at the worst possible moment (session start). +- *Clean working tree, or untracked deliveries only beneath =inbox/=* → fast-forward pull. =git pull --ff-only= refuses any merge or rebase, so the operation is either a no-op (already current) or a clean advance. Inbox files are queue input, not source-tree work, and do not block other projects from receiving rulesets updates. +- *Any staged or tracked change, dirty submodule, Git operation in progress, or untracked file outside =inbox/=* → skip the pull. Don't auto-stash and don't auto-merge — those would either lose work or invite conflicts at the worst possible moment (session start). - *Non-fast-forward history* → =--ff-only= aborts with an error. Surface that to the user; the rsync still proceeds against the working tree as-is. -*Template-freshness policy (applies to every dirty-check in the synced workflows).* "Dirty" means *tracked modifications only*. Untracked and gitignored files — an inbox drop, a file left in the tree to read, scratch output — never block a template pull, a fast-forward, or a monitoring gate. Projects were falling behind on templates because somebody sent them a task; that's the failure this policy closes. The checks here already comply (=git diff --quiet HEAD= sees only tracked changes; the ff gate uses =--untracked-files=no=), and any dirty-check added to a synced workflow follows the same rule. One deliberate exception: the rsync WIP-guard below counts untracked files *within rulesets' own synced source paths*, because an untracked half-written template is exactly the WIP it exists to hold back — that guard is about rulesets' outbound content, not the consuming project's local state. +*Template-freshness policy (applies to every dirty-check in the synced workflows).* The shared =git-worktree-gate sync-safe= policy is the source of truth: untracked files beneath =inbox/= and gitignored files do not block a pull, fast-forward, or monitoring gate; every other staged, tracked, untracked, submodule, or in-progress-operation state does. Projects must not fall behind merely because somebody sent them a task, but an arbitrary scratch file is not silently treated as safe. One deliberate exception remains: the rsync WIP-guard below is narrower than the repository gate and counts untracked files within rulesets' own synced source paths, because an untracked half-written template is exactly the WIP it exists to hold back. *** Install rulesets symlinks into ~/.claude (idempotent) @@ -74,8 +80,11 @@ if [ -d .git ]; then current=$(git symbolic-ref --short HEAD 2>/dev/null) dirty=0 - if ! git diff --quiet --ignore-submodules HEAD -- 2>/dev/null \ - || [ -n "$(git status --porcelain --untracked-files=no)" ]; then + gate="$HOME/code/rulesets/claude-templates/bin/git-worktree-gate" + if [ -x "$gate" ]; then + "$gate" sync-safe "$PWD" >/dev/null 2>&1 || dirty=1 + elif ! git diff --quiet --ignore-submodules HEAD -- 2>/dev/null \ + || [ -n "$(git status --porcelain --untracked-files=no)" ]; then dirty=1 fi @@ -107,8 +116,8 @@ fi #+end_src Behavior, per branch: -- *Behind only, current branch, clean tree* → =git merge --ff-only= advances HEAD. -- *Behind only, current branch, dirty tree* → fetched but not advanced. Surface so Craig can ff manually after dealing with the dirty state. +- *Behind only, current branch, sync-safe tree* → =git merge --ff-only= advances HEAD. An untracked =inbox/= delivery is sync-safe. +- *Behind only, current branch, sync-blocking tree* → fetched but not advanced. Surface so Craig can ff manually after dealing with the reported state. - *Behind only, non-checkout branch* → =git fetch . upstream:branch= advances the ref without touching the working tree. - *Diverged* (ahead and behind) → leave alone. Surface for Craig to resolve. Don't auto-rebase or auto-merge. - *Ahead only* or *up to date* → silent no-op. @@ -170,12 +179,14 @@ These calls have no dependencies on each other. Issue them all together in one m 10. =[ -f "$HOME/org/roam/inbox.org" ] && grep -cE '^\*\* ' "$HOME/org/roam/inbox.org" || true= — count items in the roam global inbox (=~/org/roam/inbox.org=), the roam-mode startup nudge. Silent if the roam clone isn't on this machine. Phase C reads the file when the count is non-zero, splits total vs items related to this project, and surfaces the offer (see =inbox.org= roam mode). Read-only; never files at startup. 11. KB surface prep (the read + contribute startup nudges; see =docs/specs/2026-06-16-encourage-kb-contribution-spec.org=). Gated on the agent KB clone. Counts =:agent:= nodes, lists up to 5 whose content matches the current project basename (titles only; a few most-recent nodes as a fallback when nothing matches), and resolves the best-practices node path. Read-only; silent when the clone is absent. Phase C surfaces the relevant titles (consult) and the best-practices link (contribute). + The best-practices lookup matches the node's *filename*, not its content. A roam node's slug lives only in its filename, so the earlier content-grep (=rg -l 'agent-kb-best-practices'=) matched nothing and the contribute nudge silently pointed at an empty path in every project, every session, for as long as it shipped. =find= rather than a glob keeps the probe identical under bash and zsh (zsh aborts on an unmatched glob) — the same reason the spec-sort probe below uses =find=. + #+begin_src bash ra="$HOME/org/roam/agents" if [ -d "$ra" ]; then proj=$(basename "$PWD") echo "kb-total: $(rg -l '#\+filetags:.*:agent:' "$ra" 2>/dev/null | wc -l)" - echo "kb-bestpractices: $(rg -l 'agent-kb-best-practices' "$ra" 2>/dev/null | head -1)" + echo "kb-bestpractices: $(find "$ra" -maxdepth 1 -name '*agent-kb-best-practices*.org' -print -quit 2>/dev/null)" matches=$(rg -il "$proj" "$ra" 2>/dev/null | head -5) [ -z "$matches" ] && matches=$(\ls -t "$ra"/*.org 2>/dev/null | head -3) echo "kb-relevant-titles:" diff --git a/.ai/workflows/suspend.org b/.ai/workflows/suspend.org index 3691f60..166f9c9 100644 --- a/.ai/workflows/suspend.org +++ b/.ai/workflows/suspend.org @@ -23,8 +23,10 @@ straight: Refreshes the anchor in place, prompts Craig to type =/clear=, and a hook resumes the *same* logical session in a fresh context. Craig is still here. - *suspend* (this workflow) — *leave.* Captures richly into the anchor, leaves - the file in place, and Craig walks away. The next session is a cold startup - that detects the present anchor and resumes from it. + the file in place, detaches the tmux client so the session parks in the + re-attachable set, and Craig walks away. The next session is a cold startup + that detects the present anchor and resumes from it — or Craig re-attaches the + still-live session directly. - =wrap-it-up= ([[file:wrap-it-up.org][wrap-it-up.org]]) — *end.* Writes the Summary, archives the anchor into =.ai/sessions/=, commits + pushes, and runs the phrase-dependent teardown. @@ -91,7 +93,32 @@ when the Summary body is from an earlier thread. that set — but the default shared behavior is to leave the tree alone.) 4. *Leave =.ai/session-context.org= in place.* Do not archive it. 5. *Brief handoff* — one or two lines: what was captured, where the resume - pointer is, the most-active thread. End and let Craig go. + pointer is, the most-active thread. This is the last thing Craig sees before + the view detaches (Step 6), so deliver it complete. +6. *Detach the tmux client.* As the final action, detach the client viewing the + =aiv-<project>= session so it drops out of Craig's active view while staying + alive in the background. This is a DETACH, not a teardown: the session and the + agent process keep running, nothing is killed, no context is lost. + + #+begin_src bash + sess=$(tmux display-message -p '#S' 2>/dev/null) + [ -n "$sess" ] && tmux detach-client -s "$sess" + #+end_src + + Run it as the very last tool call, after the handoff text has rendered — tmux + preserves the pane, so Craig sees the full handoff when he re-attaches. Unlike + wrap-up's teardown (which must defer to a =Stop= hook because it kills the + session the agent runs in, which would cut off the valediction), detach runs + inline: it disconnects the view but leaves the agent's session alive, so + nothing is cut off. Degrade gracefully — if not inside tmux (=$TMUX= unset, no + session), skip silently and the session simply stays attached. + + Why detach on every suspend: Craig cycles his live agent sessions in Emacs + with alt-space, and rotates through everything — including re-attaching + detached ai-term sessions — with shift+alt+space. A suspended session left + attached clutters the active rotation; detaching parks it in the + re-attachable set, which is what makes suspend-and-walk-away work. Re-attach + is one keystroke (shift+alt+space) or =tmux attach -t aiv-<project>=. * What suspend does NOT do @@ -103,8 +130,12 @@ does beyond capture: - No KB / memory promotion sweep. - No Linear / board reconciliation. - No session-record archive (the file stays live). -- No teardown (the ai-term buffer + tmux session stay up). It drops no - =Stop=-hook teardown sentinel, so the wrap-teardown hook stays dormant. +- No teardown. Suspend DETACHES the tmux client (Step 6) but never kills the + session: the =aiv-<project>= session and the agent process stay alive in the + background, only the view disconnects. It drops no =Stop=-hook teardown + sentinel, so the wrap-teardown hook stays dormant. Teardown — killing the + session — is wrap-it-up's job, not suspend's; detach is the lighter move that + parks a still-live session. - No blind commit of working files (step 3). - No valediction. A suspend is a pause, not a goodbye. diff --git a/.ai/workflows/triage-intake.org b/.ai/workflows/triage-intake.org index 9e08142..55cc939 100644 --- a/.ai/workflows/triage-intake.org +++ b/.ai/workflows/triage-intake.org @@ -11,6 +11,8 @@ Think of it as the ER intake queue: every new message, invite, and PR notificati *This file is the engine.* It carries no sources of its own. Every source it scans comes from a *source plugin* — a =triage-intake.<source>.org= file the engine loads at Phase 0. The engine is source-agnostic and project-agnostic; the project- and account-specific knowledge lives entirely in the plugins. To add a source, drop a plugin file. To change one, edit its plugin. Never wire a source into this file. +*Which sources a project pulls is a per-project choice.* A *project-specific* plugin (=.ai/project-workflows/triage-intake.*.org=, never synced) is active by presence — dropping it is the declaration. A *general* plugin (=.ai/workflows/triage-intake.*.org=, template-synced into every project — personal Gmail, cmail, calendar, Telegram, GitHub PRs) is active only when the project names its basename in a =:TRIAGE_SOURCES:= line in =.ai/notes.org= Workflow State (space-separated basenames, e.g. =:TRIAGE_SOURCES: personal-gmail cmail=). A project that declares nothing and owns no project plugin pulls nothing. This is the Phase 0 activation gate — presence is capability, the declaration is activation (see =docs/specs/2026-07-20-triage-source-activation-spec.org=). + Distinct from =daily-prep.org=: - *daily-prep* — heavier, once daily, builds the day's plan + standup brief + meeting prep + time blocks. - *triage-intake* — fast, repeatable, just answers "what's new since last check?" @@ -37,6 +39,8 @@ Typical timing: Do *not* use when running daily-prep — daily-prep already does this as Phase 3. +Also runs unattended as sentry's triage pass (=sentry.org=, pass 3): sentry invokes this engine under its no-approvals contract, where destructive actions (deleting, archiving, sending) queue for the morning-approval review instead of firing. The trigger phrases above are unchanged — a manual "triage intake" always routes here directly. + * Execution @@ -56,16 +60,18 @@ ls .ai/workflows/triage-intake.*.org .ai/project-workflows/triage-intake.*.org 2 The glob exclude is automatic: =triage-intake.*.org= matches the plugins but not this engine file (=triage-intake.org= has no second dot-segment), so the engine never loads itself. After globbing, for each plugin file: -1. Read it. -2. Evaluate its =ENABLED= precondition. If false, *announce the skip with its reason* ("skipping linear — mcp__linear not present") and move on. -3. The surviving set is the source list for Phases A-D. +1. *Activation gate.* A *general* plugin (from =.ai/workflows/=, template-synced into every project) is active only if its basename appears in the project's =:TRIAGE_SOURCES:= declaration (=.ai/notes.org= Workflow State — a space-separated list of source basenames). If it isn't declared, it is *inactive*: announce it ("inactive: personal-gmail — not in :TRIAGE_SOURCES:") and skip it. A *project-specific* plugin (from =.ai/project-workflows/=, never synced) is always active — dropping it there is itself the per-project declaration. This is what stops the synced general plugins from self-activating in projects that aren't triage targets: presence is capability, the declaration is activation (see =docs/specs/2026-07-20-triage-source-activation-spec.org=). An absent or empty =:TRIAGE_SOURCES:= means no general sources are active; a project with no declaration and no project plugin has no active sources, so triage no-ops there. +2. Read it. +3. Evaluate its =ENABLED= precondition. If false, *announce the skip with its reason* ("skipping linear — mcp__linear not present") and move on. +4. The surviving set — active and enabled — is the source list for Phases A-D. -*Announce the loaded set before scanning* so the omission can't hide: +*Announce the loaded set before scanning* so the omission can't hide — inactive (undeclared) plugins are named too, so a general plugin left out of =:TRIAGE_SOURCES:= is a visible choice, not a silent drop: #+begin_example -Loaded 5 source plugins: - general: personal-gmail, personal-calendar, cmail, github-prs +Loaded 2 source plugins (:TRIAGE_SOURCES: personal-gmail cmail): + general: personal-gmail, cmail project: deepsat-gmail + inactive (undeclared): personal-calendar, github-prs, telegram skipped: linear (mcp__linear not present) #+end_example @@ -203,6 +209,22 @@ Auto mode runs as a =/loop= in the *live session*, not a detached cron job: Running in the live session means MCP auth (Slack, Gmail, Linear) is inherited from the session — the headless-auth wall that blocks a detached cron run does not apply. A durable cross-session schedule is out of scope here; that belongs to the morning-ops orchestrator, which can later invoke auto mode's accumulate behavior as its triage limb. The close/stop commands below require a live session by design. +*** Phone delivery — push each signal sweep via =agent-text= + +Auto mode exists for when Craig is away from the desk, so a sweep that surfaces something worth seeing is delivered to his phone, not just printed into a session he isn't watching. After a sweep that renders the full three sections — one with real deltas or an unacked-list change (see "End-of-sweep output" below) — send that same output to his phone over Signal with =agent-text=: + +#+begin_src bash +agent-text "$SWEEP_SUMMARY" +#+end_src + +The pushed text is the *fuller* three-section shape, not a terse one-liner: the per-source deltas, the responses-awaiting-acknowledgment list, and the timestamp, led by a ⚠ SCAN FAILED banner if any source failed. + +*Signal-only — never on a quiet sweep.* An empty sweep (the =triage intake at HH:MM: nothing= heartbeat) does *not* push to the phone. Silent-until-signal (see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=) governs the phone channel too, so the phone stays silent until a sweep has real signal. The in-session heartbeat still prints as proof the loop ran; the phone is reserved for something that actually needs Craig. (Craig's ruling, 2026-07-20: the higher-cost channel doesn't buzz with "nothing.") + +If =agent-text= isn't on =PATH=, fall back to inline delivery and say so once. + +*Reply polling is deferred.* The send half ships here; polling the phone for Craig's replies (the =phone-recv= half of the retired ntfy design) waits on the reply-correlation follow-up. With the Signal account linked on more than one device, a reply fans out to every device and neither knows which page it answers — that has to be resolved before auto mode reads replies back. Until then auto mode pushes but does not poll, and Craig acts on a pushed summary from wherever he picks it up. + ** Preconditions and Close-out Auto mode borrows the inbox monitor-mode gates (=inbox.org= monitor mode): do not start on a dirty worktree or a red test suite — a close's batch commit would otherwise sweep up unrelated changes — and leave the tree clean and green when the loop stops. Surface a blocker with inline numbered options per =interaction.md= and wait. @@ -218,11 +240,13 @@ Each sweep runs Phase 0 (load *both* plugin dirs — the loud requirement still - DOES update an active daily-prep in Update mode and re-open it on change (per =daily-prep.org=). - DOES report, deltas-only, with loud scan-failure banners (Phase C rules unchanged). -** End-of-sweep output — three sections +** End-of-sweep output — three sections, or one heartbeat + +*Silent-until-signal (see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=).* An *empty sweep* — no deltas since the previous sweep and no change to the awaiting-acknowledgment list — collapses to a single heartbeat line and nothing else: =triage intake at HH:MM: nothing= (HH:MM local, from =date=). Detection still runs in full (Phase 0 plus the A-D scan, against the session's inherited MCP auth); only the output collapses, so a long unattended run stops filling the session with identical "no changes" blocks. A sweep with real deltas or an unacked-list change prints the full three sections below, and — when away — pushes them to Craig's phone via =agent-text= (see "Phone delivery" above). The empty-sweep heartbeat is never pushed. -1. *Deltas* — what changed since the *previous sweep* (the standard Phase C summary scoped to the inter-sweep delta; one line if nothing: "HH:MM sweep: no changes"). +1. *Deltas* — what changed since the *previous sweep* (the standard Phase C summary scoped to the inter-sweep delta). 2. *Responses awaiting your acknowledgment* — every Slack reply, email, or message directed at Craig that he hasn't acknowledged or had the agent answer. A *running list carried forward across sweeps* until Craig acks each item or closes the triage. An away user's first need is "who's waiting to hear back from me," which a delta-only sweep loses the moment it scrolls past. -3. *Timestamp* — the current date, time, and timezone on the sweep's own final line, so an away reader sees how fresh the summary is without computing it. Print it on *every* sweep, including a quiet "no changes" one — on a quiet sweep the stamp is the proof the loop ran. Generate it with: +3. *Timestamp* — the current date, time, and timezone on the sweep's own final line, so an away reader sees how fresh the summary is without computing it. Print it on every sweep that prints these three sections. On an *empty* sweep there is no separate timestamp line — the heartbeat (=triage intake at HH:MM: nothing=) is itself the freshness stamp and the proof the loop ran. Generate it with: #+begin_src bash date "+%A %Y-%m-%d %H:%M:%S %Z (%z)" @@ -416,6 +440,9 @@ Update the engine as the orchestration pattern evolves; update a plugin as its s *** Updates and Learnings +**** 2026-07-20: Phone delivery for signal sweeps (=agent-text=, send half) +Auto mode now pushes a full-three-section sweep to Craig's phone over Signal via =agent-text=, the away-from-desk delivery the retired ntfy design carried before ntfy was torn down (2026-07-04). Transport is =agent-text= (the renamed Signal pager), not ntfy. Signal-only by Craig's 2026-07-20 ruling: a quiet sweep's =nothing= heartbeat never reaches the phone — silent-until-signal governs the phone channel too, so the higher-cost channel only fires when a sweep has real signal, while the in-session heartbeat stays as proof the loop ran. Falls back to inline when =agent-text= is absent. Only the send half ships; reply polling (the old =phone-recv=) waits on the reply-correlation follow-up, because a Signal reply fans out to every linked device and neither knows which page it answers. + **** 2026-07-18: Three-section digest + close-by-default (Phase C/D rewrite) Craig's ruling after a 42h-gap sweep where the long-form report (top signals + per-source breakdown + 7-option action menu) was followed by "summarize the notable items" — and the digest that answered it was the report he wanted first. Phase C now renders ==TASKS== (work items needing Craig, solo-executable first, priority order) / ==FYI== (work context, no action owed) / ==MISC== (everything outside the project, actions stated inline), then exactly two options (close-and-file / close-and-execute-solo, the latter only when solo items exist), timestamp last. Per-source blocks and the itemized action menu are gone from the default surface (long form on request). Phase D became the close: it runs as the next action no matter what Craig replies (unless he explicitly holds), includes the mail hygiene on every scanned account without itemized confirmation, files the TASKS, clears resolved unacked items, advances the sentinel, and tears down started services. Solo = mechanical + standing-approved + no prose under Craig's name; prose sends and destructive non-mail actions stay gated. The stay-open-until-confirmed exit loop is retired. Same-day addendum: the "and reroute" modifier ("1 and reroute") — MISC items are surfaced-only by default (never filed to this project's todo.org); appending the modifier delivers each outside-project item to its owner's inbox via inbox-send per the cross-project rule. diff --git a/.ai/workflows/triage-intake.telegram.org b/.ai/workflows/triage-intake.telegram.org index 5039a8b..1319da5 100644 --- a/.ai/workflows/triage-intake.telegram.org +++ b/.ai/workflows/triage-intake.telegram.org @@ -30,12 +30,27 @@ Telega does not autostart with the Emacs daemon. "Down" is its normal state unless Craig has Telegram open in Emacs. The scan therefore runs the full lifecycle every time, never skips because the server is down: +⚠ *DOWN / not-loaded is the TRIGGER to launch, never a reason to skip or fail.* +This is the exact mistake two projects (work + home, 2026-07-24) made: they +probed telega, saw =(telega-server-live-p)= nil or telega not =featurep=, and +reported =SCAN FAILED: telegram — not loaded= or a silent SKIP — a *blind* +sweep — instead of running Step 1 to start it. A down or unloaded telega is the +normal entry state; =(telega t)= both LOADS the package and STARTS the docker +server (work confirmed: down → =(telega t)= → Ready, 18 chats). So the plugin +MUST run Step 1's launch whenever telega is down/unloaded, wait for Ready, then +scan. =SCAN FAILED= is reserved for a launch that was actually ATTEMPTED and did +not reach Ready (image missing, server crash on start, daemon unreachable) — +never for the pre-launch down state itself. The =:ENABLED:= guard above tests +whether telega is INSTALLED (=fboundp=), not whether the server is up; a down +server never disables the source. + 1. Record prior state: TELEGA_WAS_RUNNING via (telega-server-live-p). 2. Launch (only if not running): emacsclient -e "(progn (setq telega-use-docker t) (telega t) 'started)" - The setq is mandatory defense: tdlib segfaults outside docker mode - (2026-06-09), and Craig's daemon currently has telega-use-docker nil. - Wait ~2s for Ready, then (telega--loadChats 'main) until telega--chats + The setq is mandatory defense: tdlib crashed in native mode when this was + set up (2026-06-09) — a separate matter from the SEGFAULT gotcha, which is + about the loadChats argument — and Craig's daemon defaults to nil. + Wait ~2s for Ready, then (telega--loadChats '(:@type "chatListMain")) until telega--chats is populated. 3. Check messages: the maphash unread scan in ** Scan Step 2 (filters the messageContactRegistered join-notice noise). @@ -48,10 +63,13 @@ lifecycle every time, never skips because the server is down: Verify: telega-server-live-p → nil, no zevlg/telega-server container in docker ps. If Craig had it running, leave it untouched. -If any lifecycle step fails (docker image missing, server crash, daemon -unreachable), the sweep reports it as SCAN FAILED at the top of the summary -per the engine's failure rule — never as a silent skip. Craig gets real -traffic here. +If any lifecycle step fails *after the launch was attempted* (docker image +missing, server crash on start, daemon unreachable, Ready never reached), the +sweep reports it as SCAN FAILED at the top of the summary per the engine's +failure rule — never as a silent skip. This does NOT cover the ordinary +pre-launch down state: a down server means "run Step 1," not "SCAN FAILED." +Craig gets real traffic here, so a blind sweep that skipped the launch is worse +than a clean failure — it hides real unread messages behind a false all-clear. ** Scan @@ -85,22 +103,58 @@ TELEGA_WAS_RUNNING=$(emacsclient -e "(and (fboundp 'telega-server-live-p) (teleg *** Step 1 — start (docker mode) if not already running, wait for Ready #+begin_src bash -# `(telega t)` starts without popping the root buffer. Docker mode (the stable -# path — see the SEGFAULT gotcha) reconnects the persisted ~/.telega session in -# ~2s. Then load the main chat list so telega--chats populates. +# `(telega t)` starts without popping the root buffer. Docker mode reconnects the +# persisted ~/.telega session in ~2s. Then load the main chat list so +# telega--chats populates. +# +# The `(setq telega-use-docker t)` is mandatory and must come BEFORE `(telega t)`: +# tdlib crashed in native mode when this was first set up (2026-06-09), and the +# daemon's default is nil unless something (e.g. an Emacs-config :custom) has +# already forced it. It was missing here while the Quick Reference required it — +# a session that started telega without it on a native-mode daemon would take the +# untested path. Match the Quick Reference exactly. +# +# Note this is a SEPARATE concern from the SEGFAULT gotcha below: that gotcha is +# about the `loadChats` argument, and the deaths it explains happened in docker +# mode. Docker mode is not a defense against it, and it is not evidence for +# docker mode. Keep both. emacsclient -e "(progn + (setq telega-use-docker t) (unless (and (fboundp 'telega-server-live-p) (telega-server-live-p)) (telega t)) 'started)" # Poll until Ready with chats synced, or a crash/timeout. Background this with an # until-loop so the wait doesn't block; exit on Ready-with-chats OR an abnormal # server exit. Then force a chat-list load if the hash is thin: -emacsclient -e "(progn (ignore-errors (telega--loadChats 'main)) (ignore-errors (telega--loadChats 'main)) 'loaded)" +# NOTE: the chat-list argument must be a TL object, not the symbol 'main. +# `telega--loadChats' puts it straight into the request as :chat_list, and a +# bare symbol kills the server outright (see the SEGFAULT gotcha below). +# +# The liveness check on the tail is the load's only failure signal. `ignore-errors' +# catches nothing here, because a bad argument kills the server process rather than +# signalling in elisp, so without this the call returns 'loaded either way. +# The `fboundp' guard matches Step 0: if the launch failed outright telega is not +# loaded, and that should read as 'server-died like any other failure rather than +# signalling void-function. +emacsclient -e "(progn (ignore-errors (telega--loadChats '(:@type \"chatListMain\"))) (ignore-errors (telega--loadChats '(:@type \"chatListMain\"))) (if (and (fboundp 'telega-server-live-p) (telega-server-live-p)) 'loaded 'server-died))" #+end_src On a persisted session telega reaches status "Ready" within ~2s; the chat list loads over a few more. If =(hash-table-count telega--chats)= is 0 or thin, re-issue =telega--loadChats= and poll until it stabilizes. +⚠ *=server-died= is SCAN FAILED, never a quiet account.* A server that dies +during the load leaves a thin =telega--chats= hash, and a thin hash reads exactly +like an account with little unread. That is the same false all-clear the +down/not-loaded rule exists to prevent, arriving one step later in the lifecycle. +It also fits the SCAN FAILED definition above: the launch was attempted and did +not hold. So on =server-died=, report SCAN FAILED rather than scanning, and never +report a low unread count from that run. + +This is the independent evidence the SEGFAULT gotcha asks for when it says to +treat a short chat list as a real short list. Without the check there is no way +to tell the two apart, which is how the =loadChats= crash stayed invisible +through two investigations. + *** Step 2 — read unread, classified by last-message type The single most important filter: =messageContactRegistered=. Telegram counts a @@ -157,24 +211,61 @@ stays non-nil). =telega-server-kill= is what actually stops the server. Call left in =docker ps=. Skipping this whole branch when =TELEGA_WAS_RUNNING= is t is the point of Step 0: never tear down a session Craig is actively using. -⚠ *SEGFAULT GOTCHA — crashes are spontaneous; treat server death as routine.* -The dockerized =telega-server= (=zevlg/telega-server:latest=, image built -2026-06-04, tdlib 1.8.64) SIGSEGVs (exit 139) *on its own*, minutes-to-hours -into a session — 11 host coredumps between 2026-06-09 and 2026-06-11, several at -times when no triage verb was running. The 2026-06-11 investigation reproduced -the crash-free verbs and the spontaneous deaths side by side: coredump -backtraces show a corrupted stack (memory corruption in the musl build), and -no newer image exists upstream. Earlier theories — "native mode is the trigger", -"toggle-read is the trigger" — were timing coincidences; the verbs are sound. +⚠ *SEGFAULT GOTCHA — this was our bug, not tdlib's. Root-caused 2026-07-28.* +=telega-server= dies with =Unexpected char 'm' in plist value= followed by +=Assertion failed: false (telega-dat.c: tdat_plist_value: 500)=. The cause was +this workflow: Step 1 called =(telega--loadChats 'main)=. + +The chain. =telega--loadChats= is a raw TL wrapper — it drops its argument into +the request as =:chat_list= with no conversion. =telega-server--send= then +=prin1='s the whole plist, and =telega--tl-pack= passes atoms through untouched, +so the symbol goes out on the wire bare as =main=. The C parser +(=server/telega-dat.c=, =tdat_plist_value=) accepts only =(=, =[=, ="=, =-=, a +digit, =t=, =:=, or =n= to start a value. It hits =m=, prints that line, and +calls =assert(false)=, which aborts the process. The =m= in the error is +literally the first character of =main=. + +The symbol shorthand is real but belongs to a different layer: +=telega-filter.el= and =telega-folders.el= convert =(eq cl-fspec 'main)= into +='(:@type "chatListMain")=. The raw TL layer never does. telega's own callers +always pass the object (=telega.el:290=, =telega-tdlib-events.el:516=). + +Proved by experiment, not inference (2026-07-28): from a live Ready server, +=(telega--loadChats 'main)= killed it within seconds and added one coredump, +with that exact assertion; a restart plus =(telega--loadChats '(:@type +"chatListMain"))= survived three consecutive calls with no new coredump and no +assertion. + +*The previous entry here was wrong and cost real time.* It recorded the deaths +as spontaneous musl memory corruption and declared "the verbs are sound", which +sent later investigations at the docker image and tdlib versions instead of at +this file. The corrupted stack in the backtraces is what an =assert= abort looks +like, not independent evidence of a memory bug. If crashes are ever seen again +with *no* triage verb running, that is a genuinely separate cause and needs its +own investigation — do not reuse the old spontaneous-crash story to explain it. + +*This crash kills a scan; it does not silently shorten one.* An earlier draft of +this section claimed the reported "19 chats of ~50" was truncation caused by the +bad call. That was wrong, and work disproved it at the wire level on 2026-07-28: +with the corrected call their count is 19 before the first load and 19 after five +(four on =chatListMain=, one on =chatListArchive=). Nineteen is the real size of +that account. The same reading here — 19 stable across three corrected loads — +was already sitting in the evidence and should have retired the claim before it +was written down. Treat a short chat list as a real short list unless something +independently shows the server died mid-sync. + +=ignore-errors= around the call never helped — the failure is the server process +dying, not an elisp signal, so there is nothing for it to catch. That is why the +death is easy to miss from inside elisp, and why a caller should check +=(process-live-p (telega-server--proc))= after a load rather than trusting a +returned value. Operationally: docker mode stays mandatory (=telega-use-docker= = t; the setq before =(telega t)= is still the right defense), and *every action batch checks the server first* — =(process-live-p (telega-server--proc))= — restarting via -=(telega t)= when dead and re-checking Ready before firing verbs. A mid-sweep -death is recoverable, not an abort: restart, confirm Ready, resume. Durable-fix -candidates if the crashing gets worse: pin a pre-2026-06 image digest, build -=telega-server= natively against tdlib, or report upstream to zevlg with the -coredumps (=coredumpctl list /usr/bin/telega-server=). +=(telega t)= when dead and re-checking Ready before firing verbs. Any argument +handed to a =telega--*= TL wrapper must be a TL object or a plain +string/number/list, never a bare symbol. Defense in depth: even if the server does die, the scan still works because it reads the cached =telega--chats= hash, not a live query. A dead server is diff --git a/.ai/workflows/work-the-backlog.org b/.ai/workflows/work-the-backlog.org index 090841d..ea3f402 100644 --- a/.ai/workflows/work-the-backlog.org +++ b/.ai/workflows/work-the-backlog.org @@ -54,7 +54,7 @@ For the task set, in order, until the run cap is hit: 1. *Eligibility gate* (below). Ineligible → record =skipped-ineligible=, next task. 2. *Scope read* of the relevant code. Cheap; just enough to run the defer checklist. 3. *Defer checklist* (below). Any hit → defer: file the =VERIFY= naming the gap and record =deferred-VERIFY= (or, under the speedrun preset, route a quick-question gap to the pre-flight Q&A), next task. -4. *Implement* under the project's commit discipline: TDD red→green→refactor, then =/review-code --staged=, fix all Critical/Important findings, then close the task per =todo-format.md='s completion rules. Decompose into as many logical commits as the change needs — size is not capped. If implementation fails partway, leave the tree working, record =failed=, surface it, and continue to the next task. +4. *Implement* under the project's commit discipline: TDD red→green→refactor, then the isolated adversarial review (=publish= Step 1) with its re-review loop, fix all Critical/Important findings, then close the task per =todo-format.md='s completion rules. Decompose into as many logical commits as the change needs — size is not capped. If implementation fails partway, leave the tree working, record =failed=, surface it, and continue to the next task. 5. *Commit autonomy branch:* - =file-only= → surface the diff, do *not* commit. Record =implemented-diff-surfaced=. - =autonomous-commit= → =/voice personal= on the message, commit individually, push per the project's flow. Record =implemented-committed=. @@ -70,6 +70,8 @@ A task is autonomous-safe when *both* hold. This layer is a lookup, not a judgme 1. *Status is =TODO=* — never =VERIFY=, =DOING=, =DONE=, or =CANCELLED=. =VERIFY= marks "awaiting Craig's input"; auto-implementing one defeats the check it represents. The do-not-implement set is safe-by-omission: anything not plainly =TODO= (plus any project-declared "hold" marker) is out. 2. *Tagged =:solo:=* — the autonomy tag, resolved against the project's priority/tag scheme header in =todo.org= (never hardcoded). =:solo:= carries the hard definition in =todo-format.md=: completable and verifiable without Craig beyond at most one or two quick decisions answerable up front, no design deliberation. A project whose scheme declares a different autonomous-safe tag set overrides the default. +Terminology: *speedrunnable means tagged =:solo:=*. It does not mean =:quick:= or require =:quick:solo:=. The =TODO= status check above is the execution-state gate over that speedrunnable set. + Priority and =:next:= drive *ordering* within the eligible set, not eligibility ([#A] before [#B] before [#C], then the author's ordering). =:quick:= is an effort hint for batching and duration estimates — never a gate. Task *size* is deliberately absent from this gate. A large but well-specified, decision-free task is in scope and gets decomposed into per-logical-commit chunks during implementation. Size never sends a task away; only *deliberation* or *risk* does (the checklist below). @@ -80,7 +82,7 @@ Task *size* is deliberately absent from this gate. A large but well-specified, d After the scope read, run each eligible candidate through the checklist. Each item is a concrete, answerable question, not an adjective. *Any* hit — or any "unsure" — defers the task. Only a task that clears every item is implemented. -1. *Test-writability (the keystone).* Can I write the failing test from the task text — plus any decisions gathered up front — without inventing a requirement? *No / unsure* → underspecified. Under the speedrun preset, if the gap is one or two quick answerable questions, route it to the pre-flight Q&A; otherwise file a =VERIFY= noting what's missing. Under the unattended loop, file the =VERIFY= (no one to ask). +1. *Test-writability (the keystone).* Can I write the failing test from the task text — plus any decisions gathered up front — without inventing a requirement? *No / unsure* → underspecified. Under the speedrun preset, if the gap is one or two quick answerable questions, route it to the pre-flight Q&A; otherwise file a =VERIFY= noting what's missing. Under the unattended loop, file the =VERIFY= (no one to ask). *Open-ended goals are a specific, recognizable failure of this item:* a task phrased as an absence ("find bugs until none remain," "refactor until nothing worthwhile is left," "clean it up") has no writable acceptance test and so isn't really =:solo:=, even when tagged. Don't guess a stopping point — defer it and note that it needs measurable acceptance criteria (bound the surface, characterization net, dispositioned findings, objective floor — see =todo-format.md='s "Making an open-ended task measurable"). Once those are in the task body, it becomes runnable. 2. *Data-loss / irreversible / external operation.* Does implementing it require any of: =rm= of non-scratch data, =git reset --hard= / force-push, =DROP= / =DELETE= / =TRUNCATE=, file truncate/overwrite of persisted content, a schema or data migration, any external or shared-state mutation, any credential touch? *Yes* → do NOT implement; file a =VERIFY= naming the risk. This is the hard safety gate; an upfront answer never overrides it without an explicit checkpoint. 3. *Already-satisfied.* Does the scope read show the desired end-state already holds? *Yes* → file a =VERIFY= noting it and move on. Don't make a no-op change. 4. *Design deliberation.* Does the task carry an unresolved design question, a "weigh these approaches" with real tradeoffs, or a TBD that isn't a quick factual answer? *Yes* → under the speedrun preset, if it collapses to one or two quick questions, route to the pre-flight Q&A; otherwise file and surface as a =/start-work= candidate. Under the loop, file. The discriminator is *quick-answerable question* vs *deliberation* — never task size. @@ -108,7 +110,8 @@ Autonomy changes who approves, not what quality means. Per task, non-negotiable: - *TDD* per =testing.md=: red first, green, refactor. The keystone checklist item already proved the failing test is writable. - *Verification* per =verification.md=: fresh evidence, full suite green before any commit. -- *=/review-code --staged=* before every commit; Critical and Important findings block until fixed. +- *Isolated adversarial review* before every commit, dispatched per the =publish= skill's Step 1 — never an inline self-review, however small the diff. Critical and Important findings block until fixed, and each fix goes back to the *same* reviewer until it approves. Minor findings never earn another round. + - *When the review can't reach approval* — three rounds without it, a finding that recurs after being reported fixed, or a =Needs Discussion= verdict — the unattended run has no one to ask. Record the task =failed= with the standing findings in its result, leave the tree working, and continue to the next task. Never commit past a blocking finding because nobody is awake to adjudicate — an unreviewed commit landing overnight is the outcome this gate exists to prevent. - *=/voice personal=* on every commit message on the =autonomous-commit= path (or the patterns walked inline if the skill is unavailable), message printed inline so the log shows what landed. - *Task closure* per =todo-format.md=: depth-based completion (keyword + =CLOSED:= at level 2, dated rewrite at level 3+). - *One logical change per commit.* A large task becomes several commits, not one omnibus. @@ -152,7 +155,7 @@ With paging on, fire one page when the set is done or the cap is hit — end-of- notify info "Page" "<project>: <N> done, <M> remaining — <one-line summary>" --persist #+end_src -=--persist= keeps it on screen until dismissed, and =info= is the page-me urgency convention (persistent but never crash-scary). The page fires when the set completes *or* the cap stops the run — either way exactly once. The message carries the project name, the completed count, and the remaining count (with skipped tasks noted in the run summary) so Craig can confirm ready and name the next project in one reply. =notify= is the desktop paging surface; a run that expects Craig to be away also fires =agent-page= with the same message (the Signal phone channel — protocols.org "Paging Craig — the agent pager"). +=--persist= keeps it on screen until dismissed, and =info= is the notification urgency convention (persistent but never crash-scary). The notification fires when the set completes *or* the cap stops the run, either way exactly once. The message carries the project name, the completed count, and the remaining count (with skipped tasks noted in the run summary) so Craig can confirm ready and name the next project in one reply. =notify= is the desktop channel (the "page me" surface); a run that expects Craig to be away also fires =agent-text= with the same message (the Signal phone channel, "text me"). See protocols.org "Reaching Craig". * Metrics diff --git a/.ai/workflows/wrap-it-up.org b/.ai/workflows/wrap-it-up.org index 5ce88a5..ecd3d22 100644 --- a/.ai/workflows/wrap-it-up.org +++ b/.ai/workflows/wrap-it-up.org @@ -24,11 +24,13 @@ The wrap-up is complete when: 2. *File is archived.* =.ai/session-context.org= has been renamed to =.ai/sessions/YYYY-MM-DD-HH-MM-description.org=. The old path no longer exists. 3. *todo.org is clean.* Cleanup script ran. Any auto-fixes are staged for the wrap-up commit. Orphan planning lines surfaced for manual fix if there are any. 4. *Linear board is honest* (skip if project doesn't use Linear). Any Dev-Review ticket whose PR has merged was moved to Done or PM Acceptance per the classification rule. -5. *Git state is clean.* All changes committed + pushed to all remotes. Working tree clean. +5. *Git state is certified clean.* All changes are committed + pushed to all remotes, =git-worktree-gate certify= succeeded at the current HEAD, and the working tree has no staged, unstaged, untracked, submodule, or in-progress-operation state. 6. *Valediction delivered.* Brief, warm closing with key accomplishments and reminders, ending with =session wrapped.= on its own line as the signoff marker. The absence of =.ai/session-context.org= is the signal that the last session wrapped up cleanly. Its presence at session start means the previous session was interrupted. +*A helper session meets a shorter list.* Criteria 1 and 2 apply to its own context file (archived under =.ai/sessions/YYYY-MM-DD-HH-MM-<id>-<description>.org=), and 6 applies. Criteria 3, 4, and 5 do not: hygiene, the Linear pass, and all git mutation belong to the primary, so a helper that satisfied criterion 5 would have violated its contract to get there. Step 0 routes this. + * Teardown mode (set from the trigger phrase) The wrap itself — Steps 1 through 5 — is identical in every mode. The trigger phrase only decides what Step 6 does once commit + push and the valediction are done. Resolve the mode from the phrase before starting: @@ -43,8 +45,58 @@ This depends on three functions in =.emacs.d/modules/ai-term.el= (=cj/ai-term-qu * The Workflow +** Step 0: Helper branch — a helper wraps only itself + +Resolve first whether this session is a helper, because a helper's wrap is a different and much shorter workflow. Everything from Step 1 down — the hygiene passes, the inbox check, the commit, the push, the clean-tree certificate — is primary-only under the role contract in [[file:helper-mode.org][helper-mode.org]], and running any of it from a helper is exactly the concurrency failure that contract exists to prevent. + +A session is a helper when =AI_HELPER=1= in its environment (=ai --helper= sets it) or when it adopted helper-mode.org this session by instruction. If neither holds, this is a primary: skip to Step 0.5 and wrap normally. + +#+begin_src bash +echo "AI_HELPER=${AI_HELPER:-unset} AI_AGENT_ID=${AI_AGENT_ID:-unset}" +#+end_src + +For a helper, re-run the roster — the answer decides which wrap applies: + +#+begin_src bash +root="$(git rev-parse --show-toplevel 2>/dev/null || pwd)" +if [ -x "$root/.ai/scripts/agent-roster" ]; then + "$root/.ai/scripts/agent-roster" "$root"; rc=$? +else + rc=2 +fi +echo "roster rc=$rc" +#+end_src + +Pass the project root explicitly. =agent-roster= defaults to =$PWD= and keeps only agents whose cwd is at or inside that root, so running it from a subdirectory hides a primary sitting at the root — and the "alone" that produces is read below as *orphaned*, which is the one branch that commits and pushes. Capture =rc= inside the branch too: =[ -x … ] && …; echo $?= reports the status of the whole list, so an absent script reads as 1 (others live) rather than 2 (unavailable). + +- *Primary still live (rc 1)* — the normal case. Finalize the =* Summary= in the helper's own context file — same contract as Step 1, KB receipt line included (resolve it with =AI_AGENT_ID=<id> .ai/scripts/session-context-path=), archive it to =.ai/sessions/YYYY-MM-DD-HH-MM-<id>-<description>.org= so it can't collide with the primary's archive name, deliver the valediction, and stop. Do NOT commit, push, or run any hygiene pass. The helper's scoped edits stay in the tree and the primary's next commit carries them along with the archived file — say so in the valediction, so Craig knows the work is real but not yet pushed. +- *Alone (rc 0) — orphaned helper* — the primary exited first, so the git ban lifts: the concurrency that justified it is gone, and stopping here would strand the helper's edits as a dirty tree nobody owns. Run the full wrap below starting at Step 0.5, exactly as a primary would. +- *Roster unavailable (rc 2, or the script absent)* — take the archive-only path, the same as primary-still-live. Leaving work for the next session to commit is recoverable; guessing "orphaned" and committing underneath a live primary is not. + +** Step 0.5: Refuse if sentry is live + +Before anything else, check whether sentry is running in this project. Sentry holds the working tree on its =sentry/<date>-<host>= branch and commits unattended; wrapping underneath it would archive the session anchor and tear down the buffer while the loop is still firing into it. If sentry's single-runner lock is held, stop and point at the shutdown path: + +#+begin_src bash +proj="$(basename "$(git rev-parse --show-toplevel 2>/dev/null || pwd)")" +if [ -x .ai/scripts/agent-lock ] && .ai/scripts/agent-lock status "sentry-$proj" | grep -q '^held'; then + echo "sentry is active — say 'stop sentry' first" + exit 1 +fi +#+end_src + +The stop-sentry operation (defined in =sentry.org=) owns the shutdown: it cancels the loop, disposes of the branch, and walks the approval queue. Wrap-up carries only this one guard; a =stale= lock (a crashed cycle) doesn't block — only a live =held= lock does. + ** Step 1: Finalize the Summary +*** Work the Before-Close Queue (before the Summary) + +If the session anchor (=.ai/session-context.org=) carries a =* Before-Close Queue= heading with items, work them now, oldest-first, before writing the Summary, so any resulting edits ride this wrap's commit and get described in it. The queue is the "put X on the list" shorthand (see =protocols.org=, Colloquialisms and Expansions): session-scoped work Craig deferred to wrap time. + +Per item: do it if it's clear and bounded, or promote it to a =todo.org= task if it turns out to need its own session. Never drop an item silently. Remove each line as it's handled; if one can't be finished, surface it in the valediction (Step 5) and either leave a follow-up task or state why it's dropped. + +If there's no =* Before-Close Queue= heading, or it's empty, this step is a silent no-op. + *** Early KB reflection (capture while fresh, before the Summary) Before distilling the Summary, while the session is still fresh, ask: what did this session learn worth remembering, for yourself or a future agent? Reflect and stage any candidate durable facts — a decision and its why, an environment gotcha, a reference pointer, a transferable lesson. Self-answer silently; this adds no interactive turn (Craig already authorized the wrap). The candidates flow straight into the KB promotion check below, which does the actual writing and the receipt — this is the capture half, that is the commit half, one pipeline, one receipt. Reflecting here rather than reconstructing learnings after the Summary is the point: the early ask is what keeps the receipt from defaulting to "promoted 0" out of fatigue. @@ -167,6 +219,16 @@ Preview the moves without writing: emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done --check todo.org #+end_src +*** Clear temp/ + +#+begin_src bash +[ -d temp ] && find temp -mindepth 1 -delete && echo "temp/ cleared" +#+end_src + +=temp/= holds throwaway artifacts — discarded prototypes, scratch output, intermediate data (see =working-files.md=). It's gitignored in every project, so nothing here rides a commit and nothing is recoverable from git once deleted. Clearing it at wrap is what keeps ephemeral work from silting up across sessions, and it's the counterpart to =working/=, which is tracked and *never* cleared here. + +Two guards. Confirm before deleting if =temp/= holds anything a reasonable reader would call in-progress rather than throwaway — misfiled work belongs in =working/=, so move it there instead of deleting it. And skip the step entirely in a project where =temp/= is not gitignored, since that means the project is using the directory for something else. + *** Sync child priorities #+begin_src bash @@ -241,7 +303,7 @@ For an interactive walk of the judgments mid-day, run =/lint-org todo.org=. *** Inbox sanity check (surface unprocessed handoffs) -If the project has an =inbox/= directory, verify it holds nothing but =.gitkeep=, =lint-followups.org= (the lint-org pipeline file the next morning's daily-prep consumes), and any explicitly-deferred =PROCESSED-*= files before the wrap completes. An inbox that arrived at session start with handoffs from other projects, or that received handoffs mid-session, needs =inbox.org= process mode to run and apply its value-gate dispositions. Wrapping with a dirty inbox silently defers the work to next session and accumulates handoff debt that the sender can't see. +If the project has an =inbox/= directory, verify it holds nothing but =.gitkeep=, =lint-followups.org= (the lint-org pipeline file the next morning's daily-prep consumes), and =PROCESSED-*= files before the wrap completes. An inbox that arrived at session start with handoffs from other projects, or that received handoffs mid-session, needs =inbox.org= process mode to run and apply its value-gate dispositions. Wrapping with an unprocessed inbox silently defers the work to next session and accumulates handoff debt that the sender can't see. #+begin_src bash unprocessed=$(find inbox -maxdepth 1 -type f \ @@ -250,7 +312,7 @@ unprocessed=$(find inbox -maxdepth 1 -type f \ ! -name 'PROCESSED-*' \ 2>/dev/null | wc -l) if [ "$unprocessed" -gt 0 ]; then - echo "wrap-up: inbox/ has $unprocessed unprocessed item(s). Run inbox.org process mode before wrapping, or explicitly defer each item with a one-line reason in the valediction." + echo "wrap-up blocked: inbox/ has $unprocessed unprocessed item(s). Run inbox.org process mode before wrapping." find inbox -maxdepth 1 -type f \ ! -name '.gitkeep' \ ! -name 'lint-followups.org' \ @@ -259,7 +321,7 @@ if [ "$unprocessed" -gt 0 ]; then fi #+end_src -If the count is zero or the project has no =inbox/= directory, the check is a silent no-op. If non-zero, the wrap is incomplete by default. The user resolves each item (process now, defer with reason in the valediction, or delete with rationale) before the validation checklist passes. +If the count is zero or the project has no =inbox/= directory, the check is a silent no-op. If non-zero, the wrap is blocked. Process each item through its value-gate disposition, or delete it only when that workflow's rationale authorizes deletion, before continuing. The check exempts =lint-followups.org= explicitly because lint-org runs earlier in the same wrap-up workflow and writes its judgment items to that file in =inbox/= by design. The file is a pipeline artifact for the next morning's =daily-prep=, not a handoff that needs the value gate. @@ -469,17 +531,17 @@ Behavior: git status --short #+end_src -*Default policy: end every session with an empty =git status=.* The wrap is incomplete while anything remains dirty. There is no "leave it alone" default — every leftover gets an active resolution. The only way for a file to stay dirty across the wrap is the user explicitly saying "defer this one, leave it dirty." Surface each leftover with a concrete recommendation; the user has to actively opt out for the dirt to persist. +*Hard policy: end every session with an empty =git status=.* The wrap is incomplete while anything remains dirty. There is no deferral exception and no "wrapped with known changes" state: unresolved dirt means the session remains open and wrap-up does not occur. This inverts the older "intentional carryover" default, which let pre-existing dirty state accumulate across sessions silently. Carryover that lives for days or weeks is almost always one of: a forgotten commit from a prior wrap, a stale change that should be discarded, or genuine in-flight work that needs an explicit stash/branch home. None of those should default to "leave it dirty." **** Three kinds of leftover -| Pattern | What it is | Recommended action (apply unless user defers) | +| Pattern | What it is | Recommended action | |---+---+---| | Generated, runtime, or lock files that no human edits — e.g., =.claude/scheduled_tasks.lock=, =.pytest_cache/=, build outputs, IDE state, editor swap files | *Runtime artifact* — created by tooling or the harness, not by the user, and shouldn't be tracked | Add the matching pattern to =.gitignore= (project-level, not =~/.gitignore_global=). For tracked files, =git rm --cached <path>=. Stage =.gitignore= and any =rm --cached= changes in *one* follow-up commit (=chore: gitignore X=), push. Re-run =git status= to confirm clean. | | Modified or created during the session but not staged into the wrap-up commit | *Forgotten change* — real session work that should have been in the wrap commit but missed it | Stage and create a follow-up commit. Don't =--amend= the wrap-up commit once pushed (diverging history without a clear win). Push the follow-up to all remotes. | -| Was dirty at session start and still dirty at session end — work this session deliberately didn't touch | *Pre-existing dirt that needs a decision* — could be a missed commit from a prior wrap, stale abandoned work, or real in-flight work without a home | Investigate (show diff + check the originating session). Recommend one of: (a) commit now if the work is complete, (b) stash with a descriptive message if it's genuine WIP, (c) =git checkout -- <path>= / =git clean -f <path>= if stale and unwanted, (d) move to a feature branch if it's longer-running, (e) user explicitly defers and accepts the dirt. Do not silently leave dirty. | +| Was dirty at session start and still dirty at session end — work this session deliberately didn't touch | *Pre-existing dirt that needs a decision* — could be a missed commit from a prior wrap, stale abandoned work, or real in-flight work without a home | Investigate (show diff + check the originating session). Recommend one of: (a) commit now if the work is complete, (b) stash with a descriptive message if it's genuine WIP, (c) =git checkout -- <path>= / =git clean -f <path>= if stale and unwanted, or (d) move to a feature branch if it's longer-running. Do not silently leave dirty. | **** Per-file flow @@ -487,18 +549,40 @@ For each leftover line in =git status --short=: 1. Identify which of the three kinds above it matches. 2. State what the file is (one line) and the recommended action. -3. Apply the action unless the user explicitly defers. -4. Re-run =git status --short= after each follow-up commit until empty (or until every remaining line is an explicit user-deferred entry). +3. Apply the action when it is safe and authorized. +4. Re-run =git status --short= after each follow-up commit until empty. The pre-existing-dirt case (third row) is the one this rule most cares about. Treat each pre-existing-dirty file as a question that must get an answer this session, not as "carryover that's fine to inherit." A file that was dirty for a week before this session probably isn't going to get cleaner by waiting another week. Look at the diff, check the originating session's notes, and recommend a real resolution. -**** When the user defers +**** When cleanup cannot be completed -If the user does say "leave this one dirty for now" after seeing the recommendation, that is fine — log the deferral in the valediction so the next session knows it was an explicit choice, not a miss. Format: "Deferred (per Craig's decision today): =path/to/file= — <one-line reason>". Without that note, the next session can't distinguish "we agreed to defer" from "we forgot again." +Stop the wrap. Do not deliver the valediction, print =session wrapped.=, drop a teardown/shutdown sentinel, or describe the session as complete. Report: + +1. Every remaining path and its exact Git state. +2. What the file is and why the agent cannot safely resolve it alone. +3. The concrete action or decision Craig needs to provide to make the tree clean. + +An explicit decision to keep a file dirty changes the outcome from "wrapping" to "leaving the session interrupted." It never satisfies this workflow. + +*** Final clean-tree certificate — hard gate + +After all commits are pushed and every leftover appears resolved, run the shared gate: + +#+begin_src bash +gate="$(command -v git-worktree-gate 2>/dev/null || true)" +[ -n "$gate" ] || gate="$HOME/code/rulesets/claude-templates/bin/git-worktree-gate" +if [ ! -x "$gate" ]; then + echo "wrap blocked: git-worktree-gate is unavailable; install rulesets tooling and retry" + exit 1 +fi +"$gate" certify "$PWD" +#+end_src + +The certificate lives inside the Git directory, so it does not dirty the worktree. It records the exact verified HEAD. A non-zero result is a hard stop governed by "When cleanup cannot be completed" above. Step 5 is unreachable until certification succeeds. ** Step 5: Valediction -Brief, warm closing. 3-4 sentences max. +Only after the final clean-tree certificate succeeds, deliver a brief, warm closing. 3-4 sentences max. Include: - What was accomplished (specific, not generic) @@ -535,7 +619,7 @@ Do nothing. The buffer, the =aiv-<project>= tmux session, and =claude= all stay *** Teardown mode (default) -Confirm commit + push succeeded (Exit Criteria 5 — never tear down over unpushed work), then drop the sentinel: +Confirm commit + push and the final clean-tree certificate succeeded (Exit Criteria 5 — never tear down over unpushed or dirty work), then drop the sentinel: #+begin_src bash touch "/tmp/ai-wrap-teardown-$(basename "$PWD")" @@ -543,6 +627,8 @@ touch "/tmp/ai-wrap-teardown-$(basename "$PWD")" That is the whole step. Don't run any =tmux kill-session=, =emacsclient=, or buffer kill inline — the =Stop= hook reads the sentinel when this response ends and runs =cj/ai-term-quit=, which kills the =aiv-<project>= session (taking =claude= with it), kills the vterm buffer, and restores geometry. The basename of =$PWD= is the key the hook matches, so the sentinel names the session it tears down. +*The sentinel is session-scoped.* If certification fails, the =Stop= hook blocks and leaves the sentinel armed on purpose, so a wrap blocked by a dirty tree retries on a later stop without re-running this workflow. It does *not* survive the session: =session-start-disarm.sh= clears it at =SessionStart=, because a wrap that never certified is not a pending teardown once its session is gone. Before that hook existed, an uncertified sentinel sat armed indefinitely and fired in whatever session next reached a clean tree — work's 2026-07-27 11:37 wrap killed the 13:20 session mid-work, and archsetup's sat armed on a live terminal for two days. If teardown is still wanted in a new session, run this workflow again. + *** Shutdown mode Confirm commit + push succeeded, then evaluate the safety gate *before* committing to the shutdown — never power the box off out from under another live session: @@ -573,7 +659,8 @@ If =emacsclient= isn't resolvable or the daemon is down, the gate can't run — 7. *Leaving =.ai/session-context.org= in place* — its presence means "interrupted session", confuses next startup 8. *Long preachy valediction* — brief beats thorough 9. *Leaving runtime/generated files dirty without gitignoring them* — pollutes every future =git status= and erodes trust in "working tree clean" as a signal. Fix =.gitignore= during the wrap, not later. -10. *Treating "was dirty at session start, still dirty now" as fine by default* — that's how a forgotten commit from two sessions ago turns into "carryover" for two weeks. Every pre-existing dirty file needs an active resolution recommendation this session. Deferral is allowed only with an explicit user choice, logged in the valediction. +10. *Treating "was dirty at session start, still dirty now" as fine* — that's how a forgotten commit from two sessions ago turns into "carryover" for two weeks. Every pre-existing dirty file must be resolved or the wrap remains blocked. +11. *Calling a blocked cleanup a wrap* — if the strict gate fails, report the paths and needed decisions; do not valedict, certify completion, or tear down. * Validation Checklist @@ -586,19 +673,20 @@ Before considering wrap-up complete: - [ ] =todo-cleanup.el= ran — hygiene pass + =--convert-subtasks= + =--archive-done= + =--sync-child-priority= (if =todo.org= exists at project root) - [ ] =lint-org.el= ran on =todo.org= — mechanical fixes applied, judgments appended to follow-ups file (if =todo.org= exists) - [ ] Any orphan-planning-line warnings reviewed (fix or accept) -- [ ] Inbox carries nothing but expected pipeline artifacts (=.gitkeep=, =lint-followups.org=, =PROCESSED-*= prefixes), OR each remaining handoff has an explicit deferral logged in the valediction +- [ ] Inbox carries nothing but expected committed or ignored pipeline artifacts (=.gitkeep=, =lint-followups.org=, =PROCESSED-*= prefixes); any untracked inbox delivery was processed before wrap - [ ] Linear Dev-Review sweep ran; any merged-PR tickets moved to Done or PM Acceptance (skip if project doesn't use Linear) - [ ] Template-sync churn committed as its own =chore: sync .ai tooling from templates= (consuming projects only; skipped in rulesets), or surfaced if a synced path didn't match canonical -- [ ] After wrap-up commit + push, =git status --short= is empty OR every remaining line has an explicit user-deferred decision logged in the valediction +- [ ] After wrap-up commit + push, =git-worktree-gate certify "$PWD"= succeeded at the current HEAD - [ ] Each leftover was investigated and the user saw a concrete resolution recommendation - [ ] Runtime artifacts added to =.gitignore=, follow-up commit pushed, =git status= re-verified - [ ] Forgotten changes committed in a follow-up and pushed -- [ ] Pre-existing dirty files resolved (committed / stashed / discarded / moved to a feature branch) or explicitly deferred with a one-line reason in the valediction +- [ ] Pre-existing dirty files resolved (committed / stashed / discarded / moved to a feature branch); otherwise wrap stopped with an actionable blocker report - [ ] Current branch pushed to ALL remotes (verified with =git remote -v=) - [ ] All other local branches with a tracking upstream pushed to their remote - [ ] Any untracked-upstream branches surfaced for manual =git push -u= - [ ] Step 6 teardown matches the trigger phrase: no-teardown leaves the buffer; teardown drops only =/tmp/ai-wrap-teardown-<project>=; shutdown gates on =cj/ai-term-live-count= = 1 and drops only =/tmp/ai-wrap-shutdown-<project>= - [ ] No teardown/shutdown sentinel was dropped before commit + push was verified +- [ ] The teardown hook can re-verify the clean-tree certificate before consuming a sentinel - [ ] Shutdown aborted (fell back to normal wrap, logged in the valediction) when another =aiv-*= session was live or the gate couldn't run - [ ] Commit message follows format (no =session:=, no Claude attribution) - [ ] Valediction delivered (brief, specific, warm) diff --git a/.claude/commands/start-work.md b/.claude/commands/start-work.md index 85ed6b0..726cef4 100644 --- a/.claude/commands/start-work.md +++ b/.claude/commands/start-work.md @@ -171,7 +171,7 @@ Then produce a justification that covers all of these, concisely: 7. **Effort estimate.** S (under 1 hour), M (1 hour to 1 day), L (over 1 day). Rough is fine. 8. **Alternatives considered.** Is there a cheaper way? Can we defer? Can we address the root cause via a different path? 9. **Reasons not to do this.** A forced devil's-advocate verdict on whether the work should happen at all — distinct from Downsides (what the change costs) and Alternatives (cheaper paths). Surface the top three objections if real ones exist; when none rise to a genuine objection, say so in one line rather than manufacturing three (e.g. "Nothing material argues against this. No reason to defer or drop it."). Building the case against the work is cheapest at this gate, which is its purpose. -10. **Ticket quality check.** Is scope clear, are acceptance criteria concrete, are reproduction steps present for bugs? If **not clear**, stop and ask the user to choose one of: +10. **Ticket quality check.** Is scope clear, are acceptance criteria concrete, are reproduction steps present for bugs? An **open-ended goal** phrased as an absence ("find bugs until none remain," "refactor until nothing worthwhile is left," "clean it up") is the specific case where acceptance criteria aren't concrete: it has no definition of done, so there's no writable acceptance test and no clean commit at the end. Don't start it as-is. Give it measurable criteria first — bound the surface, a characterization net, dispositioned findings, an objective floor (see `todo-format.md`'s "Making an open-ended task measurable") — which is also what makes it `:solo:`-eligible. If **not clear**, stop and ask the user to choose one of: - (a) Bounce to `/brainstorm` to refine the ticket first. - (b) Ping the ticket author for clarification. - (c) Supply the missing info themselves right now, if it is easy for them to do so. diff --git a/.claude/settings.json b/.claude/settings.json index 1006916..5e1cfcc 100644 --- a/.claude/settings.json +++ b/.claude/settings.json @@ -11,11 +11,20 @@ "hooks": { "PreToolUse": [ { + "matcher": "Edit|Write", + "hooks": [ + { + "type": "command", + "command": "~/.claude/hooks/rulesets-write-boundary.py" + } + ] + }, + { "matcher": "AskUserQuestion", "hooks": [ { "type": "command", - "command": "echo '{\"hookSpecificOutput\":{\"hookEventName\":\"PreToolUse\",\"permissionDecision\":\"deny\",\"permissionDecisionReason\":\"Popup choice menus are disabled per interaction.md (No Popup Menus for Choices) — present options inline in chat as a numbered list and ask the user to reply with a number.\"}}'" + "command": "echo '{\"hookSpecificOutput\":{\"hookEventName\":\"PreToolUse\",\"permissionDecision\":\"deny\",\"permissionDecisionReason\":\"Popup choice menus are disabled per interaction.md (No Popup Menus for Choices) \u2014 present options inline in chat as a numbered list and ask the user to reply with a number.\"}}'" } ] } @@ -37,6 +46,10 @@ { "type": "command", "command": "~/.claude/hooks/session-title.sh" + }, + { + "type": "command", + "command": "~/.claude/hooks/session-start-disarm.sh" } ] }, @@ -55,6 +68,10 @@ "hooks": [ { "type": "command", + "command": "~/.claude/hooks/inbox-boundary-check.sh" + }, + { + "type": "command", "command": "~/.claude/hooks/ai-wrap-teardown.sh" } ] diff --git a/.codex/hooks.json b/.codex/hooks.json new file mode 100644 index 0000000..e14bed5 --- /dev/null +++ b/.codex/hooks.json @@ -0,0 +1,30 @@ +{ + "description": "Rulesets lifecycle enforcement shared with Claude sessions.", + "hooks": { + "PreToolUse": [ + { + "matcher": "Edit|Write", + "hooks": [ + { + "type": "command", + "command": "~/.claude/hooks/rulesets-write-boundary.py", + "timeout": 30, + "statusMessage": "Checking rulesets write boundary" + } + ] + } + ], + "Stop": [ + { + "hooks": [ + { + "type": "command", + "command": "~/.claude/hooks/ai-wrap-teardown.sh", + "timeout": 30, + "statusMessage": "Verifying clean wrap state" + } + ] + } + ] + } +} @@ -20,3 +20,11 @@ # (only the .gpg counterpart is safe to commit) mcp/secrets.env mcp/gcp-oauth.keys.json + +# Live session anchor — ephemeral, archived under a different name into +# .ai/sessions/ at wrap. Tracking .ai/ (this repo does; most projects gitignore +# it) meant the anchor showed as untracked for the whole session, which made +# git-worktree-gate report rulesets sync-blocked and every other project skip +# its rulesets pull until wrap. +.ai/session-context.org +.ai/session-context.d/ @@ -6,6 +6,7 @@ RULES_DIR := $(HOME)/.claude/rules HOOKS_DIR := $(HOME)/.claude/hooks CLAUDE_DIR := $(HOME)/.claude CODEX_DIR := $(HOME)/.codex +CODEX_HOOKS := $(CURDIR)/.codex/hooks.json LOCAL_BIN := $(HOME)/.local/bin AI_LAUNCHER := $(CURDIR)/claude-templates/bin/ai SKILLS := $(patsubst %/SKILL.md,%,$(wildcard */SKILL.md)) @@ -231,6 +232,19 @@ install: ## Symlink skills, rules, config, hooks, and bin scripts into place ln -s "$(CURDIR)/claude-templates/AGENTS.md" "$(CODEX_DIR)/AGENTS.md"; \ echo " link AGENTS.md → $(CODEX_DIR)/AGENTS.md"; \ fi + @if [ -L "$(CODEX_DIR)/hooks.json" ]; then \ + target=$$(readlink "$(CODEX_DIR)/hooks.json"); \ + if [ "$$target" = "$(CODEX_HOOKS)" ]; then \ + echo " skip hooks.json (already linked)"; \ + else \ + echo " WARN hooks.json links elsewhere ($$target) — skipping"; \ + fi; \ + elif [ -e "$(CODEX_DIR)/hooks.json" ]; then \ + echo " WARN hooks.json exists and is not a symlink — skipping"; \ + else \ + ln -s "$(CODEX_HOOKS)" "$(CODEX_DIR)/hooks.json"; \ + echo " link hooks.json → $(CODEX_DIR)/hooks.json"; \ + fi @echo "" @echo "Hooks (default):" @for hook in $(DEFAULT_HOOKS); do \ @@ -317,6 +331,13 @@ uninstall: ## Remove global symlinks from ~/.claude/ else \ echo " skip commands (not a symlink)"; \ fi + @if [ -L "$(CODEX_DIR)/hooks.json" ] \ + && [ "$$(readlink "$(CODEX_DIR)/hooks.json")" = "$(CODEX_HOOKS)" ]; then \ + rm "$(CODEX_DIR)/hooks.json"; \ + echo " rm codex hooks.json"; \ + else \ + echo " skip codex hooks.json (not our symlink)"; \ + fi @echo "" @echo "ai launcher:" @if [ -L "$(LOCAL_BIN)/ai" ]; then \ @@ -39,7 +39,12 @@ make list-languages # show available bundles #+end_src What gets installed: -- =.claude/rules/*.md= — project-scoped rules (language-specific + verification) +- =.claude/rules/*.md= — the language's own rules only. The generic rules in + =claude-rules/= are *not* copied per project: =make install= links them once + into =~/.claude/rules/=, where they load in every session on the machine. + Copying them here too loaded them twice, and project rules outrank user-level + ones, so a stale project copy silently overrode the fresh global rule. + =sync-language-bundle.sh= sweeps copies left by earlier installs. - =.claude/hooks/= — PostToolUse validation scripts - =.claude/settings.json= — permission allowlist + hook wiring - =githooks/= — git hooks (activated via =core.hooksPath=) diff --git a/claude-rules/commits.md b/claude-rules/commits.md index c4eb2cd..3283b0a 100644 --- a/claude-rules/commits.md +++ b/claude-rules/commits.md @@ -9,6 +9,7 @@ Claude, Claude Code, Anthropic, or any AI tool. Git uses the configured `user.name` and `user.email` — do not modify git config to attribute otherwise. + ## No AI Attribution — Anywhere Absolutely no AI/LLM/Claude/Anthropic attribution in: @@ -61,133 +62,6 @@ relabel a document another agent genuinely authored: if Codex wrote it, the byline stays Codex. The rule removes false co-authorship, not true authorship. -## Commit Message Format - -Commit messages follow the [Conventional Commits](https://www.conventionalcommits.org/) spec. - -### Structure - - <type>[optional scope]: <description> - - [optional body] - - [optional footer(s)] - -### Types - -- `feat:` — new feature (correlates with MINOR in SemVer) -- `fix:` — bug fix (correlates with PATCH in SemVer) -- `refactor:` — code restructuring, no behavior change -- `perf:` — performance improvement -- `test:` — adding or updating tests -- `docs:` — documentation only -- `style:` — formatting, whitespace, missing semicolons (no code-behavior change) -- `build:` — build system or external dependencies -- `ci:` — CI configuration and scripts -- `chore:` — anything else: tooling, meta, housekeeping - -The Conventional Commits spec doesn't mandate the type list. Add a new type only when the existing ones genuinely don't fit and the team will agree on what it means. - -### Scope - -A scope MAY follow the type, in parentheses, naming the affected area of the codebase: `feat(parser): add ability to parse arrays`. Use a single noun. - -### Breaking changes - -Either append `!` after the type or scope, or include a `BREAKING CHANGE:` footer (uppercase — required). Both at once is fine and adds detail. `!` alone is enough. - - feat!: drop support for Node 6 - - BREAKING CHANGE: uses JavaScript features not available in Node 6. - -### Subject line - -Imperative mood. ≤72 characters. No trailing period. The full subject is `<type>[scope]: <description>` — the 72-char limit covers the whole thing. - -### Body - -Optional. Begins one blank line after the subject. Free-form, multiple paragraphs allowed. Don't hard-wrap body lines — write each paragraph and each bullet as a single logical line and let the renderer (GitHub, Linear, `git log`) soft-wrap. Hard wraps shrink the visible render width in web UIs and cause awkward mid-sentence breaks. The same soft-wrap rule applies to PR bodies. - -Skip the body when the subject line covers the change. - -### Footers - -Optional. One blank line after the body. One per line. Format: `Token: value` or `Token #value` — the git trailer convention. The token uses `-` in place of whitespace (e.g. `Reviewed-by`, `Refs`, `Acked-by`). `BREAKING CHANGE:` is the one token allowed to contain a space, and `BREAKING-CHANGE:` is treated as a synonym. - -### How to write the message - -Write commit messages as if you're explaining the change to someone debugging a failure six months from now. Focus on what changed and why, not the play-by-play of how you typed it. Short imperative summaries like "Validate input before processing" age better than diary-style notes. - -The body, when you need it, is where context belongs — the constraint, bug, or tradeoff that forced the change. Over time the body becomes a lightweight decision log, which is more valuable than perfectly formatted messages. - -Commit messages describe what changed and why, not the process that produced the change. Don't reference code review, linting, test runs, or other workflow steps in the body (e.g. "from local review," "review surfaced," "flagged by reviewer"). Reviewers and future archaeologists want the what and the why. How you got there belongs in the PR discussion, not the commit. - -### Examples - -**Subject only:** - - docs: correct spelling of CHANGELOG - -**With scope:** - - feat(lang): add Polish language - -**With body and footer:** - - fix: prevent racing of requests - - Introduce a request id and a reference to the latest request. Dismiss incoming responses other than from the latest request. - - Remove timeouts which were used to mitigate the racing issue but are obsolete now. - - Refs: #123 - -**Breaking change with `!`:** - - feat(api)!: send an email to the customer when a product is shipped - -**Breaking change in footer:** - - feat: allow provided config object to extend other configs - - BREAKING CHANGE: `extends` key in config file is now used for extending other config files. - -## Voice and Focus - -Applies to commit bodies, PR descriptions, and PR comments (review replies, follow-up notes, thread responses). - -**Write as if to a colleague.** The reader is a teammate who'll see this in `git log`, a PR feed, or a Linear thread. "I" is allowed where natural. Don't sound abstract — name the file, the function, the constraint, the symptom. Press-release voice ("This change improves...") and committee voice ("It is recommended that...") both come out. The message has to read like one engineer talking to another, not like a generated artifact. - -**No felt-experience narration.** Don't tell the reader how the change will feel or how often you'll use it. Phrases like "I'll feel this every time I commit", "this will be a relief", "I'm excited about" — these read as performance, not communication. State what changed and let the reader decide what to do with it. - -**Don't noun-ify verbs.** "The ask", "a learn", "a reveal", "the spend", "a build" — use the real noun: "the request", "the lesson", "the finding", "the budget", "the system". Verb-as-noun reads as corporate-speak and makes the sentence feel performed. - -**No sentence fragments in prose.** Every prose sentence needs a subject and a verb. "Two changes." or "Fix incoming." or "Body as decision log." read as bullet-list shorthand even when they're standing alone in a paragraph. Bullets and headings can be fragments — prose sentences cannot. - -**"I" is the author, not the user.** First person is for what *I* did or decided in this commit ("I dropped the legacy fallback because..."). It's not for describing how the software or rule behaves for whoever uses it next. "The dialog only opens if I ask" is wrong when the rule is read by someone else — that "I" becomes ambiguous. Use third-person or passive for behavior: "opens on request", "opens when asked", "opens when the user invokes it". Code and systems are the actor; "I" stays for decisions. - -**First person where it fits.** When the subject is you or a decision you made, use "I" ("I added X", "I kept the parameter as `Any` because..."). When the subject is a team decision or shared rationale, "we" fits. When another author's prior work is the subject, name them ("Kostya's PR #116 did X"). Third-person constructions like "This PR introduces X" or "This change restores Y" read as press-release self-narration. The commit *is* the change, so don't announce it. Code and systems can stay third-person when they're the actor ("the guard rejects...", "the serializer returns...") — first person is for describing what you did or decided, not for narrating how the code behaves. - -**Brief. Terse is preferred.** A one-sentence body beats a paragraph saying the same thing. If the subject line covers it, skip the body entirely. Cut every clause that restates what the diff or the PR card already shows. Length is not a proxy for care. Rhetorical padding ("worth noting", "it's important to understand") always comes out; keep what a reader will actually use. - -**Follow-up approvals stay terse.** A re-review that just confirms prior CHANGES_REQUESTED feedback got addressed should be `Approved.` and nothing more. The fixes are visible in the diff and in the prior review thread, so restating them adds noise. The first round of substantive review gets a real comment. Subsequent sign-offs after fixes do not. Counts as a trivial one-liner under the Step 2 exception, so the draft-file flow can be skipped. - -**Kind.** PR comments and review replies are directed at a specific person. Acknowledge them when it fits ("thanks for the review") without pouring it on. When you disagree or push back, frame it as your read rather than a correction ("I think...", "my read was...", "did you mean X?"). Leave room for the other person to have seen something you didn't. A polite question beats a defensive explanation. Kindness is free and makes the next review cheaper. - -Focus on what was wrong and what was corrected. Not the mechanics. -Readers skimming `git log` or a PR want the before-state, the -after-state, and the reason. They don't need a TypeScript-variance -lesson, a compiler-inference walkthrough, or a trip through an API's -internals. Keep the "why" to one sentence unless a subtle invariant -genuinely needs more. - -Don't stack technical terms. A sentence that chains three or more type -signatures, API names, or compiler concepts reads as a jargon wall. -Break it into shorter sentences and translate to reader-facing -language. "The mock returns `Promise<Mission>`, so the resolver's -argument is `Mission`, not `unknown`" beats the full inference chain -that produces that signature. Keep the terms a reader will grep for, -drop the ones that name compiler internals. ## Content scope for public artifacts @@ -216,277 +90,54 @@ Edge case: when one of these files *is* the change (a commit in the rulesets rep **Tooling-path enumeration is the same leak.** Citing a rule as authority isn't the only way the tooling layer leaks into history. A commit whose *content* must name these paths — a `.gitignore` adding `.claude/`, `CLAUDE.md`, `.ai/` — has unavoidable, correct file content, but its *message prose* must not enumerate them ("chore: ignore .claude tooling, CLAUDE.md, and session files"). On a public or shared-remote repo that enumeration exposes the tooling layer's structure in the log just as a citation would. Name the category instead: "chore: extend gitignore for local tooling and build artifacts". The same holds for any incidental mention, not only `.gitignore` commits. Two exemptions: a commit whose change *is* one of these files (the edge case above), and private single-user repos with no shared remote, where the history is the project and there's no third party to leak to. -Different artifact types carry different content. Don't duplicate. - -**PR descriptions:** four sections, in order. - -1. **Problem** — what's wrong, with enough detail that a teammate can - recognize the same failure mode in their own work. -2. **Fix** — what changed. -3. **Why this fixes it** — causal link, one or two sentences. -4. **How it was tested** — skip for proposals, specs, or discussions; - required for shipped fixes. - -The PR is the technical artifact. It carries the detail. - -If the project's publishing overlay defines a ticket system, see it for -ticket-body conventions (a ticket body is typically just the Problem and -Fix, with the causal why and test verification left to the PR). - -**PR review comments** are conversational and don't follow this -structure — they follow the Voice and Focus rules above. - -Verbose preambles, motivational language, and context unrelated to the -problem belong out. Same conciseness pressure as commit-message bodies. - -## Review and Publish - -Commits and PRs are team-visible, permanent, and hard to amend once shared -(especially after push or after a reviewer has replied). Before executing -`git commit` or `gh pr create`, the change must pass a local code review -*and* the message must be reviewed by the user. The flow has three steps, in -order. - -### Step 0: pre-flight reconcile (mandatory) - -Before reviewing the diff, fetch from the remote and reconcile against the -upstream of the current branch. Reconciliation can change the working state -when a rebase brings in upstream commits that touch staged files, and that -would invalidate Step 1's review. Handling drift first means the review and -the commit message describe the post-reconcile state. - -1. Fetch all remotes: - - git fetch --all --prune - -2. If the current branch has no upstream (new branch, never pushed), skip - to Step 1 — there's nothing to reconcile against, and the first push - sets the upstream. - -3. Otherwise, check divergence against `@{u}`: - - git rev-list --left-right --count @{u}...HEAD - - Output is `<behind>\t<ahead>`. Decide based on the pair: - - - **0 behind, anything ahead** — no-op. Continue to Step 1. - - **Behind only, clean tree** — fast-forward: `git merge --ff-only @{u}`. - - **Behind only, dirty tree** — surface to the user. Don't auto-stash or - auto-merge. Offer to commit or stash first, or skip the reconcile and - proceed knowing the push may need attention later. - - **Diverged (behind AND ahead)** — surface to the user. Ask whether to - rebase the local commits onto upstream (default for feature branches), - merge the upstream branch in (rare; preserves both lines), or skip and - proceed with the divergence. Don't auto-rebase. - -4. **PR flow only.** Also fetch the base branch (usually `main`) and check - whether the feature branch's base is behind. Surface this informationally; - don't auto-rebase the feature branch without asking. The "X commits - behind base" badge on the PR is a follow-up decision, not a reason to - block publish. - -The startup workflow's `git fetch --all --prune` doesn't substitute for -Step 0. Upstream can advance during a long session, especially across -machines or with teammates pushing in parallel. Run Step 0 every time the -publish flow starts. - -### Step 1: local code review (mandatory) - -Run the `review-code` skill against the change: - -- Before a commit: `/review-code --staged` -- Before a PR: `/review-code` (branch diff against `main` merge-base) -- Before commenting on someone else's PR: `/review-code <PR#>` - -Surface **all** findings to the user: Critical, Important, and Minor. - -**Default block:** any Critical or Important finding stops the flow. Fix the -issues and re-run `/review-code` until the diff is clean. Minor findings are -shown but do not block. - -**Override:** the user can bypass the block with an explicit "proceed anyway" -(or equivalent wording). Without the explicit override, do not proceed to -Step 2. - -The `review-code` skill already has a Phase 0 eligibility gate that handles -trivial and ineligible diffs (whitespace-only, revert with obvious -justification, already-reviewed SHA). Trust that gate; there is no "trivial -enough to skip review" exemption on top of it. - -### Step 2: draft, review, publish - -**Voice patterns and the approval gate are two independent decisions.** Don't bundle them. - -*Voice patterns are always personal for publish artifacts.* Commit messages, PR titles + bodies, and PR review comments all go out under the user's name, so they always run through `/voice personal` (the full pattern walk — general + Craig's-voice + the artifact-mechanics patterns: first-person rewrite, public-artifact scope flag, praise/correction asymmetry, finding stems), regardless of whether `.ai/` is tracked. These three are personal-voice artifacts by definition — the skill's personal mode exists for exactly them. Pattern #39 (public-artifact scope flag) matters *most* on team-visible artifacts, so it must never be skipped on a PR comment or PR body. There is no "general-voice mode" for publish artifacts. - -*The approval gate is the only thing `.ai/`-tracking decides.* Before drafting, run this command: - -``` -git ls-files :/.ai/ 2>/dev/null | head -1 -``` - -The `:/` pathspec anchors the search to the repo root, so the command works from any subdirectory. Without it, running from a subdir returns no matches even when `.ai/` is tracked at the repo root, which silently misclassifies the project. - -- **No output** — `.ai/` is gitignored, missing, or empty (the user's personal repos). **Gate applies**: write to `/tmp`, run `/voice personal`, print inline, ask approve / request changes / open in editor, then publish only on explicit approval. -- **Any output** — one or more files under `.ai/` are tracked (a shared / team repo). **Gate skipped for velocity**: write to `/tmp`, run `/voice personal`, print inline, publish immediately. - -Either way the draft runs through `/voice personal` first. The subflows below describe the full gated path. For the gate-skipped path, run the same `/voice personal` pass, then collapse the "Ask: approve, request changes, or open in editor" step — the draft prints inline and the publish step runs immediately afterward. - -**For commit messages:** - -1. Write the proposed message to `/tmp/commit-<short-slug>.md`. -2. Run `/voice personal` on the file. Always. The skill walks its full pattern list covering signs of AI writing, universal good-writing rules (Strunk & White, Orwell, Plain English, Garner), and Craig's voice patterns (first-person rewrite, semicolons → periods/commas, contractions, sentence-split on conjunctions, felt-experience cut, sentence-fragment rewrite, terse cut for rhetorical padding, no-emphasis-formatting, public-artifact scope flag, praise/correction asymmetry, finding stems). The commit subject line stays imperative per Conventional Commits — `/voice personal` rewrites the body, not the subject. Skip the pass for purely mechanical commits (a chore version bump, a typo fix) where the subject alone carries the message. -3. Print the final draft inline in the terminal. Every line, exactly as it'll be committed. No truncation, no summary. State that the skill ran (e.g. "/voice personal — full pattern walk"). If pattern #39 (public-artifact scope) flagged anything, surface those warnings; the user resolves them manually. -4. Ask: approve, request changes, or open in editor. Wait for an explicit answer. Do not open the file in `emacsclient` (or any editor) by default — print first, edit only if asked. - - **Approve** → commit with `git commit -F /tmp/commit-<short-slug>.md`. - - **Request changes** → make them, re-run `/voice personal`, re-print inline, ask again. - - **Open in editor** → only if the user asks. `emacsclient -n /tmp/commit-<short-slug>.md`. After the editor closes, re-read the file, re-print the contents inline, and ask again. - -**For PR descriptions:** - -1. Write the title as line 1 and the body below it to `/tmp/pr-<slug>.md`. **Title format:** the conventional-commit subject (`refactor: remove dead if-count-is-not-None check in admin`). If the project defines a publishing overlay with a ticket system, follow it for the ticket suffix in the title and the cross-link line in the body (see the overlay). -2. Run `/voice personal` on the file. The PR title stays imperative per Conventional Commits — `/voice personal` rewrites the body, not the title. -3. Print the final draft inline in the terminal. Title on line 1, blank line, then body — exactly as it'll be posted. State that the skill ran. Surface any pattern #39 (public-artifact scope) warnings. -4. Ask: approve, request changes, or open in editor. Wait for an explicit answer. Do not open the file in `emacsclient` (or any editor) by default. - - **Approve** → continue to step 5. - - **Request changes** → make them, re-run `/voice personal`, re-print inline, ask again. - - **Open in editor** → only if the user asks. `emacsclient -n /tmp/pr-<ticket-or-slug>.md`. After the editor closes, re-read the file, re-print inline, ask again. -5. Split the file on the first blank line and pass the title and body to `gh pr create --title "..." --body "$(tail -n +3 <file>)"` (or a heredoc) so formatting is preserved. Add `--reviewer <user[,user...]>` in the same call when you already know who should review. -6. Request reviewers on the new PR if you didn't pass `--reviewer` at create time. Use `gh pr edit <N> --add-reviewer <user>`. If the repo has a `CODEOWNERS` file, GitHub auto-suggests based on touched paths. Still issue the explicit request so the reviewer gets notified. Pick reviewers per the team's convention for the area touched (often documented in the per-repo `CLAUDE.md`). For follow-up PRs, consider tagging the parent PR's author if their context would help. PRs without a human reviewer request stall — "checks passed" is not a substitute for review. -7. **Project publishing overlay (if present).** If the project defines a publishing overlay — a `publishing-<team>.md` rule loaded from its `.claude/rules/` — run its post-create steps now: ticket cross-linking, ticket-state moves, and any other tracker integration it specifies. A project with no overlay skips this; the PR is already open and reviewers are requested, which is the complete universal flow. - -**For PR review comments and replies (review verdicts, threaded discussion, follow-up notes on someone else's PR or your own):** - -Pick the shape first. Most reviews are Shape 1. - -- **Shape 1 — Single review** (verdict + summary body + 0+ inline pins). The default for any post that carries a verdict (`APPROVE`, `REQUEST_CHANGES`, `COMMENT`), even when the verdict has no line-specific findings. One `gh api` call posts the summary, every inline pin, and the verdict together. review notification fires once for `APPROVE` or `REQUEST_CHANGES`. -- **Shape 2 — Issue-thread comment** (no verdict). General PR discussion, not a review. No inline pins. No review notification. -- **Shape 3 — Reply on an existing inline thread**. Responding to a specific prior reviewer comment. Threads under that comment. No review notification. - -**Inline threshold for Shape 1.** Any finding that names a `path:line` belongs as an inline comment pinned to that line. Cross-cutting observations (verdict rationale, "third PR with the same pattern", overall test-coverage gaps that don't pin to one place) stay in the summary body. There's no "fold one inline into the summary" exception — a single line-specific finding still goes inline. - -**Shape 1: Single review (bundled summary + inline)** - -1. Identify findings, split into **inline-eligible** (each names a specific `path:line`) and **summary-only** (cross-cutting). Decide the verdict. - -2. Write one concatenated draft to `/tmp/pr-<N>-review.md` with explicit separators: - - ``` - === SUMMARY === - <verdict summary body> - - === INLINE path=frontend/src/foo.tsx line=440 === - <inline body 1> - - === INLINE path=frontend/src/bar.tsx line=137 === - <inline body 2> - ``` - - The separator format is exactly `=== SUMMARY ===` and `=== INLINE path=<path> line=<n> ===`. The summary block is mandatory even for verdict-only reviews. Inline blocks are zero-or-more. - -3. Run `/voice personal` on the file once. The skill walks its full pattern list across every block at the same time. The separators stay intact because they aren't prose. - -4. Print the final draft inline in the terminal. Every block — the summary body AND the full prose of every inline comment — exactly as it'll be posted, with its separator header. Print the inline in full; never describe it in place of printing it ("I'd pair it with one inline on…"). Craig approves the exact words that post under his name, so the exact words must be on screen. State that the skill ran (e.g. "/voice personal — full pattern walk across summary + 3 inline"). Surface any pattern #39 warnings. - -5. Ask: approve, request changes, or open in editor. Wait for an explicit answer. Do not open the file in `emacsclient` (or any editor) by default. - - **Approve** → continue to step 6. - - **Request changes** → make them, re-run `/voice personal` on the whole file, re-print inline, ask again. - - **Open in editor** → only if the user asks. `emacsclient -n /tmp/pr-<N>-review.md`. After the editor closes, re-read, re-print inline, ask again. - -6. Split the file on the separator lines and post in **a single** `gh api` call: - - ``` - gh api repos/<owner>/<repo>/pulls/<N>/reviews \ - --hostname <ghe-host-or-omit> \ - -F event=REQUEST_CHANGES \ - -F body="<summary block>" \ - -F "comments[][path]=<path1>" \ - -F "comments[][line]=<line1>" \ - -F "comments[][body]=<inline 1>" \ - -F "comments[][path]=<path2>" \ - -F "comments[][line]=<line2>" \ - -F "comments[][body]=<inline 2>" - ``` - - `event` is one of `APPROVE`, `REQUEST_CHANGES`, `COMMENT`. The `comments[]` array can be empty for verdicts with zero line-specific findings — the call still uses the same endpoint. Pass `--hostname` for non-`github.com` hosts (a project's publishing overlay names its host when it's a GitHub Enterprise instance). - -7. Verify the review landed. `gh api repos/<owner>/<repo>/pulls/<N>/reviews --hostname ...` returns the latest review with bundled inlines. Confirm `state` matches the verdict and the inline count matches what was posted. - -8. **Project review-notification overlay (if present).** If the project defines a publishing overlay with a review-notification step (e.g. a Slack ping to the PR author), run it now — but only for `APPROVE` and `REQUEST_CHANGES` verdicts. The overlay owns the channel, the message format, the author-mention lookup, and the threading. A project with no overlay skips notification entirely. `COMMENT` verdicts and Shapes 2-3 below never notify, overlay or not. - -**Shape 2: Issue-thread comment (no verdict)** - -Use when the post is informal discussion that shouldn't appear as a review verdict (e.g. "I'd like to discuss the X approach before you continue"). - -1. Write the proposed comment to `/tmp/pr-<N>-comment.md`. -2. Run `/voice personal`. -3. Print inline, ask approve/changes/edit, gate as in Shape 1 step 5. -4. Post: `gh pr comment <N> --body-file /tmp/pr-<N>-comment.md`. -5. Verify: `gh api repos/<owner>/<repo>/issues/<N>/comments`. -6. No review notification. - -**Shape 3: Reply on an existing inline thread** - -Use when responding to a specific prior reviewer comment. - -1. Find the parent comment ID: `gh api repos/<owner>/<repo>/pulls/<N>/comments`. -2. Write the reply to `/tmp/pr-<N>-reply-<comment-id>.md`. -3. Run `/voice personal`. -4. Print inline, ask approve/changes/edit, gate as in Shape 1 step 5. -5. Post: `gh api repos/<owner>/<repo>/pulls/<N>/comments -F in_reply_to=<comment-id> -F body="$(cat /tmp/pr-<N>-reply-<comment-id>.md)"`. -6. Verify in the same `comments` list. -7. No review notification. - -**Approve does not authorize a merge.** Reviewing a PR never authorizes merging it. Anything in `## Merge Strategy` below applies only to merges *you* are about to perform on your own branches — and even then, the merge needs its own explicit user confirmation per the rules there. A project's publishing overlay may add a team merge practice (e.g. approve-then-author-merges, where the review notification hands the merge decision to the PR author); that's an overlay concern, not a global one. - -**Exception:** trivial one-liners the user dictated verbatim in the -conversation (e.g. "commit this as `chore: bump version`", "reply just -'thanks for the review'") can skip the draft-file step in Step 2. -`/review-code` in Step 1 still runs when it applies; Phase 0 of that skill -handles trivial diffs, and acknowledgment-only replies don't need it at all. - -**Single-skill gate.** Each of the three subflows above runs `/voice personal` before printing the draft — the full pattern walk covering AI-writing signs, universal good-writing rules, Craig's voice patterns, and the artifact-mechanics patterns (first-person rewrite, public-artifact scope flag, praise/correction asymmetry, finding stems). Publish artifacts (commits, PR titles + bodies, PR review comments) always use personal mode; the `.ai/`-tracking check at the top of Step 2 decides only whether the approval gate fires, not which patterns run. Running the skill is mandatory; the printed draft must have been through it. When the user asks mid-flow for "the voice pass" on an in-progress draft, that means re-run the full pattern walk — not a subset. Always state that the skill ran when announcing the printed draft (e.g. "/voice personal — full pattern walk"). Skipping the pass without flagging it is a defect. The terse/omit-needless-words cut (pattern #38) is the *last* thing the skill does before the draft is printed: read each sentence and cut it in half, keeping only what changes meaning. The draft the user first sees must already be terse — if they have to ask for an Orwell pass after seeing it, the pass was skipped. - -**If `/voice` is unavailable.** The skill should be installed (it ships with rulesets), but a fresh or partial environment may not have it. Don't let that block the publish, and don't skip the discipline silently. Walk the same patterns inline — they're documented in the skill, and the publish flow already names which ones matter (first-person rewrite, semicolons → periods/commas, contractions, sentence-split, felt-experience cut, fragment rewrite, terse cut, the pattern #39 public-artifact scope flag, plus the AI-writing and good-writing passes). Then state that the skill was unavailable and the pass was applied by hand (e.g. "/voice unavailable — patterns walked inline"). The gate is the pattern walk, not the tooling; the skill is the convenient way to run it, not the only way. Flag the missing skill so it gets installed. - -### Hook-level authorization - -The Step 1 code review plus the Step 2 user approval together constitute the -authorization gate for the publish action. No separate hook-level approval -prompt is needed on `git commit`, `gh pr create`, `git push`, or their -variants once Step 2 has been approved. If a hook is configured, rely on the -flow above to be the source of truth; do not treat the hook as a second -independent gate. - -## Merge Strategy -- *Squash-merge is the default* for feature branches. It avoids carrying - WIP and fix-up commits into the target branch history and produces one - logical change per merge. -- State the planned merge approach (squash, rebase, or merge commit) and - the target branch *before* pushing or merging. Wait for explicit user - confirmation before `git push`, `gh pr merge`, or any equivalent. The - Review and Publish flow above approves the *content*; merge strategy is - a separate decision that needs its own confirmation. -- *Pre-push reconcile.* Right before `git push`, do one more - `git fetch <remote> <branch>` and verify the local branch is still - ahead-only against its upstream. If something landed between Step 0 and - push (review and draft together can take several minutes, and another - machine or teammate may push in that window), surface and resolve before - the push command runs. Catching drift here is cheaper than recovering - from a failed non-fast-forward push under publish-step pressure. -- Override the squash default only when there's a concrete reason: a - clean per-commit review history the user has explicitly asked for, a - multi-commit semantic narrative the team values, etc. Squash is the - safe default; document why when deviating. - -## Before Committing - -1. Check author identity: `git log -1 --format='%an <%ae>'` — should be the user. -2. Scan the message for AI-attribution language (including emojis and footers), and on a public or shared-remote repo for tooling-path enumeration — prose that lists `CLAUDE.md`, `.claude/`, `.ai/`, `todo.org`, `notes.org`, or `session-context`. Name the category, not the paths. Exempt: a commit whose change is one of those files, and private single-user repos. -3. Review the diff — only intended changes staged; no unrelated files. -4. Confirm staged files belong in the repo: nothing that the project's policy keeps untracked (the personal-tooling set in gitignore-mode projects), and in repos with a canonical/mirror split, the edit is on the canonical side — a mirror-only edit gets reverted by the next sync. -5. Run the full test suite and linters as their own step, read the result, and commit only on zero failures — never chain the run into the commit command (see `verification.md`). +## Write in the first person, as Craig + +Everything authored in or about this repo is first person: code comments, commit +messages, PR descriptions, PR review comments, and any note that lands in the +repo or its history. State a choice as a choice — "I swept the copies rather +than repairing them, because a drifted copy outranked the global rule" beats +"the copies are swept rather than repaired." + +**The "I" is Craig.** He is the author of record on every commit, comment, and +review in his repos, and these artifacts go out under his name. So the voice is +his, writing about his own work — not an agent narrating what it did on his +behalf. Never write the agent into the prose as a separate party: no "Craig +asked me to", no "I filed this for Craig", no "needs Craig's decision". Where a +decision is still open, it is *his* open decision, written as "I haven't decided +whether…" or "this needs a call I haven't made yet." + +The same holds for anyone else's work. Name them ("Kostya's PR #116 did X"), +because they are a third party. Craig is not. + +Third-person constructions like "This change introduces X" or "This PR restores +Y" read as press-release self-narration. The commit is the change, so it does +not need announcing. + +**The one carve-out: code is the actor when describing behavior.** A comment +saying *what the code does* stays third person, because the subject genuinely +is the code and not me — "the sweep only fires when the global rule exists", +"the guard rejects a malformed payload". First person is for the decision +behind it, third person for the behavior itself. Both often belong in the same +comment: what it does, then why I chose it. + +This is the rule the publish flow already applied to commit bodies. It lives +here because code comments get written constantly and the publish skill is not +loaded then. + +## The publish flow lives in the `publish` skill + +Everything about *how* a commit, PR, or review comment gets written, reviewed, +approved, and published — the pre-flight reconcile, the code-review gate, the +draft/voice/approval gate, conventional-commit format, Voice and Focus, PR +description structure, the three review shapes, merge strategy, and the +pre-commit checklist — is in the `publish` skill. Load it before drafting a +message, not after: the flow gates what gets written. + +What stays here is what must hold whether or not anything is being published, +and where a violation is permanent and reaches other people. If the skill fails +to load you will have to be told the flow, which is recoverable. The rules +below are not. ## If You Catch Yourself diff --git a/claude-rules/cross-project.md b/claude-rules/cross-project.md index 73c0e1b..c5de962 100644 --- a/claude-rules/cross-project.md +++ b/claude-rules/cross-project.md @@ -50,6 +50,14 @@ whose canonical home is `~/code/rulesets/`. When work in a downstream project needs one of these files to change, a local edit alone is a stopgap that the next sync reverts. The durable change happens only in the rulesets canonical. +Installed global paths such as `~/.claude/rules/`, `~/.claude/hooks/`, +`~/.claude/skills/`, and rulesets-owned entries in `~/.local/bin/` are +symlinks into the canonical repository, not downstream copies. Never edit +through those paths from another project's session: resolving the symlink +would dirty rulesets outside its own logging and wrap discipline. The +`rulesets-write-boundary.py` hook mechanically denies Edit/Write targets whose +real path lands there and directs the proposal through `inbox-send rulesets`. + The process, every time: 1. **Make the change locally** in the downstream project so it's usable diff --git a/claude-rules/desktop-capture.md b/claude-rules/desktop-capture.md index 0051c4d..c4a67f9 100644 --- a/claude-rules/desktop-capture.md +++ b/claude-rules/desktop-capture.md @@ -38,9 +38,12 @@ output isn't available; it needs the compositor installed. Open it on a *separate* real workspace and tell them which one, so it never grabs their active workspace. They switch when ready. Craig's viewer preference -is `imv`; launch it through the compositor (`hyprctl dispatch exec "imv -<files>"`) so it survives the agent's shell rather than a bare `&` job that gets -reaped. +is `imv`; launch it with `gui-open --image <file>` (dotfiles-shipped) rather +than a bare `&` job or `hyprctl dispatch exec`: it detaches through `systemd-run +--user` so the agent shell can't reap it, resolves the current Hyprland instance +after a compositor restart, and confirms the viewer is mapped and visible before +returning. An HTML render uses `gui-open <file>` (or `--browser`) the same way. +If `gui-open` isn't on PATH, the machine needs a dotfiles pull. ## Always clean up diff --git a/claude-rules/emacs.md b/claude-rules/emacs.md index 2c3b729..846888d 100644 --- a/claude-rules/emacs.md +++ b/claude-rules/emacs.md @@ -1,3 +1,8 @@ +--- +paths: + - "**/*.el" +--- + # Working With Craig's Running Emacs Applies to: `**/*.el` (and any task that edits Craig's Emacs configuration) diff --git a/claude-rules/interaction.md b/claude-rules/interaction.md index 8d65799..b5798bd 100644 --- a/claude-rules/interaction.md +++ b/claude-rules/interaction.md @@ -2,7 +2,46 @@ Applies to: `**/*` -How the agent communicates with the user during a session — choice prompts, status updates, decision points. +How the agent reasons with the user and communicates during a session — how interpretations are formed, then how choices, status, and decision points are presented. + +## Collaborative Peer Reasoning + +Treat the conversation as joint reasoning between peers. The user's words are evidence of intent, not merely a string to execute literally. Use the request, prior context, current state, and likely downstream consequences to form a working interpretation. + +### Infer first; clarify at material forks + +Infer the intended outcome and proceed when reasonable interpretations lead to the same action. When two plausible interpretations would produce materially different outcomes, strategies, or external effects, state the working interpretation and ask one focused question before crossing that fork. Do not interrupt for reversible implementation details that can be resolved with ordinary judgment. + +### Test conclusions before committing to them + +Do not promote the first plausible explanation or plan into a conclusion. Check the strongest reasonable alternative, test the assumptions that distinguish it, and weight the tradeoffs that decide between them. Separate verified facts, inferences, and recommendations when the distinction matters. Calibrate confidence instead of projecting certainty. + +For a consequential judgment, a useful response shape is: working interpretation, evidence, strongest alternative, recommendation, confidence, and the one clarification that would change the action. Do not force this scaffold onto simple tasks. + +### Corrections update the model + +Treat a user correction as new evidence that changes the working model. Reconcile its downstream implications immediately: assumptions, source selection, plans, scheduled actions, task state, and conclusions already reached. Do not reduce a substantive correction to a wording change or preserve stale premises silently. + +A correction is not automatically true merely because the user made it. If it conflicts with verified evidence, explain the conflict directly and identify what would resolve it. If the correction is supported, change course cleanly without defending the earlier answer. + +### Neither sycophantic nor adversarial + +Agreement follows evidence and reasoning, not deference. Disagreement serves the shared outcome, not the defense of a prior position. State a contrary view when it changes the decision, give the evidence behind it, and leave room for missing context. Once new evidence resolves the issue, stop arguing the old case. + +### Process serves the outcome + +Rules and workflows are constraints on the work, not substitutes for judgment. Apply them in service of the user's intended outcome. If a literal workflow interpretation produces a disproportionate, surprising, or strategically different result, surface that implication before proceeding. + +Failure signs: + +- Executing the narrow literal request while ignoring an evident intended outcome. +- Asking about a reversible detail while failing to clarify a consequential fork. +- Presenting one plausible account as settled without testing its strongest alternative. +- Treating a correction as local wording while leaving downstream assumptions unchanged. +- Agreeing to preserve rapport or resisting to preserve authority. +- Letting procedural completeness overwhelm the value of the task. + +This is a working collaboration contract, not immutable wording. Refine the section as real sessions expose better distinctions or failure modes; route the feedback to the canonical file rather than accumulating project-local exceptions. ## No Popup Menus for Choices @@ -54,7 +93,7 @@ In conversational output to the user, do not use Markdown bold (`**...**`) or in - Write command names, file paths, key chords, and code identifiers as plain text — `pearl-save-issue` becomes pearl-save-issue, `C-; L s s` becomes C-; L s s. - Use structure that doesn't invert colors: headers, numbered lists, dashes, parentheses, and double quotes for labels are all fine. -- Fenced code blocks (triple backtick) are acceptable when the user explicitly wants a block to copy — they don't invert the way inline spans do. Default to plain text otherwise. +- Fenced code blocks (triple backtick) are not an exception. Craig's direction is zero markup in chat output, always: "always always list it out without markup" (2026-05-30). Fences don't invert the way inline spans do, but they still read as markup he didn't ask for, and the carve-out kept reintroducing them. When he needs something to copy, give it as plain indented text, or write it to a file and name the path. This governs **chat output**, not the Markdown source of rule files, specs, or docs the user reads in an editor — those keep normal Markdown formatting. The constraint is the terminal rendering of the live conversation. @@ -64,7 +103,7 @@ Craig runs Claude Code inside Emacs EAT (through tmux). EAT renders SendUserFile Two display lanes, by what the visual is for: -- **Durable or interactive visuals** — HTML prototypes, full renders, anything Craig will study or click: open in the browser (`google-chrome-stable "file://<abs-path>" &>/dev/null &`) or imv, on a separate workspace per `desktop-capture.md`. +- **Durable or interactive visuals** — HTML prototypes, full renders, anything Craig will study or click: open with `gui-open <abs-path>` (dotfiles-shipped; detaches through `systemd-run --user` so the agent shell can't reap it, resolves the Hyprland instance after a compositor restart, and confirms the window is actually visible before returning). It auto-detects image vs HTML; `--image` / `--browser` force. Place it off Craig's active workspace and name it, per `desktop-capture.md`. - **Quick inline glances** — a chart, a did-it-render check: sixel in the terminal. Encode with ImageMagick (`magick <img> sixel:<out>`; img2sixel silently emits zero bytes on some builds) and display in a separate tmux window (`tmux new-window -d -n <name>`, then `tmux send-keys -t <name> "clear; cat <out>" Enter`, tell Craig the window name) rather than the Claude Code pane, whose TUI repaints over anything drawn into it. With native tmux sixel active, the image lives in tmux's grid and survives window switches, scrolling, and resizing. **Capability gate.** Native sixel needs two config pieces: EAT answering the XTWINOPS cell-size query (patched eat.el, owned by .emacs.d) and `terminal-features 'xterm*:sixel'` in tmux.conf (owned by dotfiles). Check before relying on it: `tmux display -p '#{client_cell_width}'` — nonzero means go; 0 means the chain is missing a piece, so fall back to the browser lane. (Diagnosed 2026-07-13: tmux won't transmit sixel until it knows the client's cell pixel size, and stock EAT 0.9.4 silently ignores the CSI 14 t query tmux uses to ask.) diff --git a/claude-rules/knowledge-base.md b/claude-rules/knowledge-base.md index d61ef03..146a5e4 100644 --- a/claude-rules/knowledge-base.md +++ b/claude-rules/knowledge-base.md @@ -22,13 +22,15 @@ Pull before querying (`git -C ~/org/roam pull --ff-only`); skip silently if offl Classify the project before any write. The source of truth is the work-root denylist below — never inference from remotes, names, or task content: - **Work** — project root is, or sits under, a denylisted root. No KB write, ever. Record durable facts per that project's own conventions. -- **Personal** — project root sits under `~/code/`, `~/projects/`, or `~/.emacs.d` and is not denylisted. KB writes allowed. +- **Personal** — project root sits under `~/code/`, `~/projects/`, `~/.emacs.d`, or `~/.dotfiles` and is not denylisted. KB writes allowed. - **Unknown** — anything else. No KB write. Work-root denylist (confirmed by Craig, 2026-06-10): `~/projects/work` **Refusal contract** (work and unknown alike): state the classification, name the durable fact in a one-line redacted summary, and say where it was or wasn't written — so Craig can re-route it deliberately instead of losing it silently. +**Scope of the denylist — durable KB-node writes only.** The work-denylist governs one thing: promoting a durable fact into a new `agents/` node. It is a confidentiality guard so work-confidential material doesn't land in the personal cross-machine store. It is *not* a general "don't touch roam from a work project" boundary. Roam is a *shared resource*, not another project's product scope. Reading it (any project) and *tidying the shared roam inbox* — processing, routing, and filing the capture items in `~/org/roam/inbox.org`, e.g. via inbox-zero — are allowed from any project session, work included; that is housekeeping on a shared resource, not a durable-fact write. Only the durable-node promotion stays work-denylisted. Do not park roam-inbox tidying as a cross-project boundary crossing (a sentry inbox-zero pass did exactly that on 2026-07-19 — the error this note closes). + A write is one node per fact, under `agents/`, roam-valid so Craig's org-roam indexes it: ``` @@ -43,7 +45,26 @@ A write is one node per fact, under `agents/`, roam-valid so Craig's org-roam in <the fact, with [[id:...]] links to related nodes> ``` -Pull before writing, commit and push after (`git -C ~/org/roam add -A && git commit && git push`) — same session discipline as any repo. Never edit Craig's hand-authored nodes; link to them. This write autonomy is scoped to the KB alone — it is not permission to send email, comment on tickets, or post to any public or external channel. +Pull before writing (`git -C ~/org/roam pull --ff-only`, read-only). Then acquire the roam-write lock, write the node, and trigger roam-sync to commit and push — roam-sync stays the roam repo's only committer (the 2026-06-24 one-git-owner rule). The tree is chronically dirty from live captures, so an agent's own `git add -A && commit` could sweep an in-flight capture into a stray commit; edit-plus-trigger avoids that. Never edit Craig's hand-authored nodes; link to them. This write autonomy is scoped to the KB alone — it is not permission to send email, comment on tickets, or post to any public or external channel. + +The write block, with the lock and the trigger: + +```sh +# Acquire the roam-write lock so a concurrent sentry pass or inbox writer can't +# race this write. Callers pass a name; agent-lock owns the path (tmpfs). +if [ -x .ai/scripts/agent-lock ]; then + if ! .ai/scripts/agent-lock acquire roam-write --wait; then + # Busy after the bounded wait — surface and stop, don't write unlocked. + echo "roam-write lock held by another writer; try again shortly" >&2 + exit 1 + fi +fi +# ... write ~/org/roam/agents/<ts>-<slug>.org ... +systemctl --user start roam-sync.service # roam-sync commits + pushes +[ -x .ai/scripts/agent-lock ] && .ai/scripts/agent-lock release roam-write +``` + +Degrade gracefully when `agent-lock` isn't installed (an older checkout mid-sync): the guard above is skipped and the write proceeds unlocked — today's behavior. Only a *present* helper reporting the lock busy after its bounded wait stops the write; an *absent* helper never blocks it. ## What goes in, what stays out diff --git a/claude-rules/org-tables.md b/claude-rules/org-tables.md index 1b70085..bc9d27d 100644 --- a/claude-rules/org-tables.md +++ b/claude-rules/org-tables.md @@ -1,3 +1,8 @@ +--- +paths: + - "**/*.org" +--- + # Org Table Standard Applies to: `**/*.org` diff --git a/claude-rules/subagents.md b/claude-rules/subagents.md index 8578dea..e52d906 100644 --- a/claude-rules/subagents.md +++ b/claude-rules/subagents.md @@ -34,6 +34,66 @@ This is the same boundary the "Don't Subagent At All" section and the "Subagenting trivial work" anti-pattern draw; treat it as an explicit gate at dispatch time. +Every size-based rule in this file — the cost gate here, "Don't Subagent At +All", the trivial-work anti-pattern — is subject to the isolation override +below. + +## Isolation Override — When Size Doesn't Gate + +Every size heuristic in this file rests on one assumption: that the main +thread could do the task itself just as well, so the only question is +whether delegating is worth the overhead. When that assumption fails, the +heuristics don't apply, and a five-line task can require a subagent that a +five-hundred-line one wouldn't. + +The assumption fails whenever **the main thread is structurally disqualified +from the task** — not slower at it, disqualified. The test: would the main +thread's own context make its answer *less* trustworthy? If holding the +context is what corrupts the judgment, then doing it inline doesn't save the +overhead, it destroys the result. The isolation *is* the deliverable, and +"it's only a small diff" is not an argument against it. + +**The standing instance is the pre-commit code review** (`publish` skill, +Step 1). The author cannot review their own change, because a self-review +checks the diff against the author's own model of it and cannot check the +model. Errors that survive a self-review are the ones that were never in the +diff — an inherited scope, an estimated blast radius, a fix correct for the +case in mind and wrong for the one never considered. So that review is +dispatched on *every* commit including a one-line one, and the ~10-tool-call +floor, the single-function rule, and the trivial-work anti-pattern are all +overridden there by design. + +Other cases with the same shape: verifying a claim the main thread already +committed to in conversation, and any second opinion where the first opinion +is already in context. If you find yourself reasoning "I already know the +answer, so a subagent is wasteful," check whether already knowing it is the +problem. + +This override widens *what* gets dispatched. Scope, constraints, and output +format are still required, and arguably matter more here, since an isolated +agent can't fall back on shared context to fill a gap. + +**Field 2 of the Prompt Contract inverts under this override, and the +inversion is the whole point.** Normally field 2 says to paste the relevant +output verbatim and include what you learned in earlier turns. Do that for an +isolation dispatch and you hand over the very model you spawned the agent to +escape — a reviewer given your findings reviews your findings. So for an +isolation dispatch, field 2 is *the artifact under test and the independent +record of what was asked, and nothing else*: the diff, a one-line claim of +what it does, and the ticket or plan where one exists. The conversation, the +rationale, and the dead ends are withheld on purpose. + +Keep the requirement source in. A ticket is not your model of the change; it +was written before the work, usually by someone else, and it is the only +thing that can contradict your claim about your own diff. + +**The output is a judgment, so the review gate resolves differently.** The +Review-Gate Cadence below says subagent output is a claim to be verified +before moving on, which is right when the deliverable is *work*. When the +deliverable is *a judgment about your work*, verifying it against your own +reading reinstates exactly the bias the dispatch removed. Disagreement goes +to the user to adjudicate, not back to the author's own judgment. + ## When to Spawn a Subagent ### Parallel-safe (spawn multiple in parallel) @@ -63,11 +123,15 @@ at dispatch time. ### Don't Subagent At All +Unless the Isolation Override applies — these are efficiency rules, and they +lapse when the main thread's own context is what makes its answer untrustworthy. + - **The target is already known** and the work fits in under ~10 tool calls. - **Single-function logic** — one Read + one Edit is faster than briefing an agent. - **You can see the answer from context** — don't spawn a researcher for - something already on screen. + something already on screen. (The inverse of this one is the override's + clearest case: when *having* seen it is the disqualification, dispatch.) ## Prompt Contract @@ -138,7 +202,11 @@ fix), then dispatch the fix with a specific contract. - **Retrying a failed subagent task in the orchestrator** — pollutes context. Dispatch a fix agent instead. - **Subagenting trivial work** — one Read + one Edit doesn't need an - agent; spawn overhead exceeds benefit. + agent; spawn overhead exceeds benefit. Except under the Isolation + Override, where a one-line diff still gets its own reviewer. +- **Reviewing your own change inline** — the mirror-image failure, and the + more expensive one. Skipping a dispatch to save overhead on a small diff + costs a review that could only have come from outside your context. - **Skipping review between tasks** — compounding bugs are much harder to unwind than any single bug. - **Letting the agent decide scope** — "figure out what needs changing" diff --git a/claude-rules/testing.md b/claude-rules/testing.md index b3fa5bf..dd15282 100644 --- a/claude-rules/testing.md +++ b/claude-rules/testing.md @@ -16,340 +16,30 @@ TDD is the default workflow for all code, including demos and prototypes. **Writ Do not skip TDD for demo code. Demos build muscle memory — the habit carries into production. -### Understand Before You Test -Before writing tests, invest time in understanding the code: +## Test Categories — required for all code -1. **Explore the codebase** — Read the module under test, its callers, and its dependencies. Understand the data flow end to end. -2. **Identify the root cause** — If fixing a bug, trace the problem to its origin. Don't test (or fix) surface symptoms when the real issue is deeper in the call chain. -3. **Reason through edge cases** — Consider boundary conditions, error states, concurrent access, and interactions with adjacent modules. Your tests should cover what could actually go wrong, not just the obvious happy path. +Every unit under test needs all three, not just the happy path: -### Adding Tests to Existing Untested Code +1. **Normal** — standard inputs, common workflows, typical volumes. +2. **Boundary** — zero, one, max, empty vs null, single-element collections, unicode, very long input, timezone and date edges. +3. **Error** — invalid input, type mismatches, network failure, missing parameters, permission denied, resource exhaustion, malformed data. -When working in a codebase without tests: +The negative and boundary cases are the ones that find bugs. A unit with only +Normal coverage is not tested, it is demonstrated. -1. Write a **characterization test** that captures current behavior before making changes -2. Use the characterization test as a safety net while refactoring -3. Then follow normal TDD for the new change +## The rest of the standard lives in the `testing-standards` skill -## Test Categories (Required for All Code) +Characterization tests for untested code, the per-category detail, combinatorial +and property-based and mutation testing, organization and the pyramid, +integration-test rules, naming, the test-quality rules (independence, +determinism, mocking boundaries, signs of overmocking), the +refactor-when-tests-are-hard principle, coverage targets, the spike exception, +and the anti-pattern list are all in the `testing-standards` skill. Load it when +writing tests. -Every unit under test requires coverage across three categories: - -### 1. Normal Cases (Happy Path) -- Standard inputs and expected use cases -- Common workflows and default configurations -- Typical data volumes - -### 2. Boundary Cases -- Minimum/maximum values (0, 1, -1, MAX_INT) -- Empty vs null vs undefined (language-appropriate) -- Single-element collections -- Unicode and internationalization (emoji, RTL text, combining characters) -- Very long strings, deeply nested structures -- Timezone boundaries (midnight, DST transitions) -- Date edge cases (leap years, month boundaries) - -### 3. Error Cases -- Invalid inputs and type mismatches -- Network failures and timeouts -- Missing required parameters -- Permission denied scenarios -- Resource exhaustion -- Malformed data - -## Combinatorial Coverage - -For functions with 3+ parameters that each take multiple values (feature-flag -combinations, config matrices, permission/role interactions, multi-field -form validation, API parameter spaces), the exhaustive test count explodes -(M^N) while 3-5 ad-hoc cases miss pair interactions. Use **pairwise / -combinatorial testing** — generate a minimal matrix that hits every 2-way -combination of parameter values. Empirically catches 60-90% of combinatorial -bugs with 80-99% fewer tests. - -Invoke `/pairwise-tests` on the offending function; continue using `/add-tests` -and the Normal/Boundary/Error discipline for the rest. The two approaches -complement: pairwise covers parameter *interactions*; category discipline -covers each parameter's individual edge space. - -Skip pairwise when: the function has 1-2 parameters (just write the cases), -the context requires *provably* exhaustive coverage (regulated systems — document -in an ADR), or the testing target is non-parametric (single happy path, -performance regression, a specific error). - -## Escalation Beyond Category and Pairwise - -The Normal/Boundary/Error categories and the pairwise matrix are the default -discipline. Two further techniques escalate beyond them — reach for them when -the default leaves a gap, not on every unit. - -### Property-Based Testing - -When an invariant holds across a broad input domain — round-trips -(`decode(encode(x)) == x`), idempotence (`f(f(x)) == f(x)`), ordering -invariants (output is always sorted), or any "output always satisfies X" — -generate inputs and assert the property instead of enumerating cases. The -generator explores corners you wouldn't think to write by hand, and a -failing case shrinks to a minimal reproducer. Use the standard tool for the -language (Hypothesis for Python, fast-check for JS, proptest for Rust). -State the property as the test name and let the framework supply the inputs. - -Reach for this when the behavior is a law over a domain rather than a fixed -set of examples. Keep category-discipline cases for the specific edges that -must always hold; the property test covers the space between them. - -### Mutation Testing - -When line coverage is high but you suspect the assertions are thin — tests -that execute the code without checking its output, or that pass with a -function body replaced by a stub — use mutation testing to measure whether -the suite actually kills injected faults. The tool flips conditionals, swaps -operators, and deletes statements, then reruns the suite; a surviving mutant -is a fault the tests didn't catch. Use mutmut or cosmic-ray for Python, -Stryker for JS. High line coverage with a low mutation score means weak -assertions, not a tested codebase. - -Reach for this on critical logic where coverage looks reassuring but you -want evidence the tests would fail on a regression. It's a diagnostic, not a -gate on every change — mutation runs are slow. - -## Test Organization - -Typical layout: - -``` -tests/ - unit/ # One test file per source file - integration/ # Multi-component workflows - e2e/ # Full system tests -``` - -Per-language files may adjust this (e.g. Elisp collates ERT tests into -`tests/test-<module>*.el` without subdirectories). - -### Testing Pyramid - -Rough proportions for most projects: -- Unit tests: 70-80% (fast, isolated, granular) -- Integration tests: 15-25% (component interactions, real dependencies) -- E2E tests: 5-10% (full system, slowest) - -Don't duplicate coverage: if unit tests fully exercise a function's logic, -integration tests should focus on *how* components interact — not repeat the -function's case coverage. - -## Integration Tests - -Integration tests exercise multiple components together. Two rules: - -**The docstring names every component integrated** and marks which are real vs -mocked. Integration failures are harder to pinpoint than unit failures; -enumerating the participants up front tells you where to start looking. - -Example: - -``` -def test_integration_refund_during_sync_updates_ledger_atomically(): - """Refund processed mid-sync updates order and ledger in one transaction. - - Components integrated: - - OrderService.refund (entry point) - - PaymentGateway.reverse (MOCKED — returns success) - - Ledger.credit (real) - - db.transaction (real) - - Validates: - - Refund rolls back if ledger write fails - - Both tables updated or neither - """ -``` - -**Write an integration test when** multiple components must work together, -state crosses function boundaries, or edge cases combine. **Don't** when -single-function behavior suffices, or when mocking would erase the interaction -you meant to test. - -## Naming Convention - -- Unit: `test_<module>_<function>_<scenario>_<expected>` -- Integration: `test_integration_<workflow>_<scenario>_<outcome>` - -Examples: -- `test_cart_apply_discount_expired_coupon_raises_error` -- `test_integration_order_sync_network_timeout_retries_three_times` - -Languages that prefer camelCase, kebab-case, or other conventions keep the -structure but use their idiom. Consistency within a project matters more than -the specific case choice. - -## Test Quality - -### Independence -- No shared mutable state between tests -- Each test runs successfully in isolation -- Explicit setup and teardown - -### Determinism -- Never hardcode dates or times — generate them relative to `now()` -- No reliance on test execution order -- No flaky network calls in unit tests -- Time/clock-mocking helpers must avoid two recurring failure modes: - - *Infinite recursion.* The helper must not call the primitive it's - replacing. If the mock for `now()` calls `now()`, the test stack - overflows. Compute the mock value from a fixed source (a captured - instant, an injected fake clock). - - *Scope-shadowing without reach.* A mock that only exists inside - the test function won't affect production code that reads the - symbol through its canonical path. Replace the symbol at its - definition site (monkey-patch the module attribute in Python, - redefine the global in Lisp, swap the package-level binding in - Go, replace the named export in JavaScript) — or inject a fake - via dependency-inversion. Don't lean on scope-shadowing - primitives (Lisp `let`, Python local rebind, JS shadowed `let`) - that fence the mock to the test's lexical scope; production code - won't see them and the test passes against the real clock. - -### Performance -- Unit tests: <100ms each -- Integration tests: <1s each -- E2E tests: <10s each -- Mark slow tests with appropriate decorators/tags - -### Mocking Boundaries -Mock external dependencies at the system boundary: -- Network calls (HTTP, gRPC, WebSocket) -- File I/O and cloud storage -- Time and dates -- Third-party service clients - -Never mock: -- The code under test -- Internal domain logic -- Framework behavior (ORM queries, middleware, hooks, buffer primitives) - -### Signs of Overmocking - -Ask yourself: - -- Would this test still pass if I replaced the function body with `raise NotImplementedError` (or equivalent)? If yes, the mocks are doing the work — you're testing mocks, not code. -- Is the mock more complex than the function being tested? Smell. -- Am I mocking internal string / parsing / decoding helpers? Those aren't boundaries — they're the work. -- Does the test break when I refactor without changing behavior? Good tests survive refactors; overmocked ones couple to implementation. - -When tests demand heavy internal mocking, the fix isn't better mocks — it's -restructuring the code (see *If Tests Are Hard to Write* below). - -### Testing Code That Uses Frameworks - -When a function mostly delegates to framework or library code, test *your* -integration logic: -- ✓ "I call the library with the right arguments in the right context" -- ✓ "I handle its return value correctly" -- ✗ "The library works in 50 scenarios" — trust it; it has its own tests - -For polyglot behavior (e.g., comment handling across C/Java/Go/JS), test 2-3 -representative modes thoroughly plus a minimal smoke test in the others. -Exhaustive permutations are diminishing returns. - -### Test Real Code, Not Copies - -Never inline or copy production code into test files. Always `require`/`import` -the module under test. Copied code passes even when production breaks — the -bug hides behind the duplicate. - -Mock dependencies at their boundary; exercise the real function body. - -### Error Behavior, Not Error Text - -Test that errors occur with the right type; don't assert exact wording: -- ✓ Right exception type (`pytest.raises(ValueError)`, `(should-error ... :type 'user-error)`) -- ✓ Regex on values the message *must* contain (e.g., the offending filename) -- ✗ `assert str(e) == "File 'foo' not found"` — breaks when prose changes even though behavior is unchanged - -Production code should emit clear, contextual errors. Tests verify the -behavior (raised, caught, returned nil) and values that must appear — not the -prose. - -## If Tests Are Hard to Write, Refactor the Code - -If a test needs extensive mocking of internal helpers, elaborate fixture -scaffolding, or mocks that recreate the function's own logic, the production -code needs restructuring — not the test. - -Signals: -- Deep nesting (callbacks inside callbacks) -- Long functions doing multiple things ("fetch AND parse AND decode AND save") -- Tests that mock internal string / parsing / I/O helpers -- Tests that break on refactors with no behavior change - -Fix: extract focused helpers (one responsibility each), test each in isolation -with real inputs, compose them in a thin outer function. Several small unit -tests plus one composition test beats one monster test behind a wall of mocks. - -## Coverage Targets - -- Business logic and domain services: **90%+** -- API endpoints and views: **80%+** -- UI components: **70%+** -- Utilities and helpers: **90%+** -- Overall project minimum: **80%+** - -New code must not decrease coverage. PRs that lower coverage require justification. - -## TDD Discipline - -TDD is non-negotiable. These are the rationalizations agents use to skip it — don't fall for them: - -| Excuse | Why It's Wrong | -|--------|----------------| -| "This is too simple to need a test" | Simple code breaks too. The test takes 30 seconds. Write it. | -| "I'll add tests after the implementation" | You won't, and even if you do, they'll test what you wrote rather than what was needed. Test-after validates implementation, not behavior. | -| "Let me just get it working first" | That's not TDD. If you can't write a failing test, you don't understand the requirement yet. | -| "This is just a refactor" | Refactors without tests are guesses. Write a characterization test first, then refactor while it stays green. | -| "I'm only changing one line" | One-line changes cause production outages. Write a test that covers the line you're changing. | -| "The existing code has no tests" | Start with a characterization test. Don't make the problem worse. | -| "This is demo/prototype code" | Demos build habits. Untested demo code becomes untested production code. | -| "I need to spike first" | Spikes are fine — under the protocol below. Throw the spike away, then write the first failing test before productionizing. | - -If you catch yourself thinking any of these, stop and write the test. - -### The Spike Exception (Disciplined) - -TDD stays the default. The one sanctioned way to write code before a test is -a spike — exploratory code that answers "is this approach even viable?" when -you can't yet write a meaningful failing test because the shape of the -solution is unknown. A spike is disciplined only when all three hold: - -1. **Timebox it.** Set a limit before starting (an hour, an afternoon) and - stop when it's up. An open-ended spike is just untested implementation - wearing a different name. -2. **Do not commit spike code.** The spike is a learning artifact, not a - deliverable. It never enters the branch history. Keep it in a scratch - file or a throwaway worktree. -3. **Throw the spike away, then start with a failing test.** Once the spike - has answered the viability question, delete it. Write the first failing - test against the now-understood behavior, then productionize under normal - Red/Green/Refactor. The production code is written test-first even though - the exploration wasn't — you don't promote the spike into production by - bolting tests on after. - -The spike buys understanding, not code. If you find yourself keeping the -spike because rewriting it feels wasteful, the timebox was too long or the -problem was tractable enough to TDD from the start. - -## Anti-Patterns (Do Not Do) - -- Hardcoded dates or timestamps (they rot) -- Testing implementation details instead of behavior -- Mocking the thing you're testing -- Mocking internal helpers (string ops, parsing, decoding) — those are the work -- Inlining production code into test files — always `require` / `import` the real module -- Asserting exact error-message text instead of type + key values -- Shared mutable state between tests -- Non-deterministic tests (random without seed, network in unit tests) -- Testing framework behavior instead of your code -- Ignoring or skipping failing tests without a tracking issue +What stays here is what has to be true before any code is written, which is when +no skill has been summoned yet: test first, and cover all three categories. ## Content scope diff --git a/claude-rules/todo-format.md b/claude-rules/todo-format.md index 2cdc76c..8038b98 100644 --- a/claude-rules/todo-format.md +++ b/claude-rules/todo-format.md @@ -1,3 +1,8 @@ +--- +paths: + - "**/*.org" +--- + # Todo Entry Format Applies to: `**/*.org` (org-mode todo and inbox files) @@ -5,6 +10,56 @@ Applies to: `**/*.org` (org-mode todo and inbox files) How task entries are structured in org-mode todo files (`todo.org`, `inbox.org`, any GTD-style org file). Same shape across every project. +## Stamp `:LAST_REVIEWED:` when you create the task, not a cycle later + +Every task filed at `**` with a priority cookie carries a `:LAST_REVIEWED:` +property from the moment it's written: + +``` +** TODO [#B] Terse topic phrase :tag: +:PROPERTIES: +:LAST_REVIEWED: 2026-07-23 +:END: +Body. +``` + +Use today's date, from `date +%F`. The org-native `[YYYY-MM-DD Day]` form is +equally valid; both parse. + +**Why.** Writing a task *is* reviewing it. Whoever files it has just written +the body, chosen the wording, and graded the priority against the scheme — that +is the same judgment `task-review` applies, made with better context, because +the reason for the task is still in the room. Leaving the stamp off asserts the +opposite: `task-review-staleness.sh` sorts a missing property first, as +never-reviewed, so a task filed today arrives at the top of tomorrow's review +batch and gets "reviewed" by someone re-deriving what its author knew a day +earlier. That is ceremony, and ceremony teaches people to click through the +real thing. + +The concrete case: the 2026-07-23 sweep filed eight tasks in one night. Every +one landed unstamped, and the staleness count went 13 → 22 while the list got +*more* accurate, not less. The number stopped measuring drift and started +measuring recent activity. + +**This applies to every path that files a task**, not just the inbox: inbox +filing, triage intake, spec decomposition, task audit, a bug found mid-session, +a task you write by hand. If you wrote a task body today, stamp it today. + +**What it does not do.** The stamp never means "correct forever" — it means +"a person judged this on that date." A task filed today still enters the +review rotation on the normal cycle; it just enters it on the *next* cycle +rather than immediately. And it's a claim about a real event, so don't stamp +a task you didn't actually consider: a bulk import of someone else's list +is genuinely unreviewed, and stamping it would convert "nobody has read +these" into a false "reviewed today." + +**Enforcement.** `lint-org.el`'s `task-missing-last-reviewed` checker flags any +open `**` task with a priority cookie and no stamp, scoped to exactly the +headings `task-review-staleness.sh` selects, so the checker and the staleness +count never disagree. It's judgment-only and never auto-fixes: nothing can know +when an unstamped task was actually last considered, and writing today's date +onto an old one would destroy the very signal the property carries. + ## Priority and Tag Scheme Header Every project's `todo.org` opens with a top-level section named @@ -41,6 +96,9 @@ fixed definitions everywhere, because autonomous execution eligibility gate and trusts the author's tag rather than re-deriving autonomy at run time. +“Speedrunnable” is shorthand for `:solo:`. The `:quick:` tag plays no part +in that definition. + - **`:solo:` — autonomy.** The task can be completed *and verified* without Craig's involvement beyond at most one or two quick decisions that can be stated and answered before work starts. No open design question, no @@ -64,6 +122,49 @@ step** in the task-review and task-audit workflows, so the run-time gate can trust the tag. A review or audit that skips the `:solo:`/`:quick:` assessment is incomplete. +### Making an open-ended task measurable (so it can be `:solo:`) + +A task phrased as the *absence* of something — "find bugs until none are +visible," "refactor until no worthwhile opportunities remain," "clean this +up until it's good" — cannot be `:solo:`, because it fails the +*verifiable-by-the-agent* gate. Absence isn't falsifiable: an agent can +always look once more, so "done" is a judgment call, which is exactly what +`:solo:` forbids. The fix is not to drop the task but to give it an +objective completion criterion. Four moves convert a fuzzy goal into a +measurable one: + +1. **Bound the surface.** Enumerate the concrete units the task covers (the + N functions, the M files, the named code paths). The done-set is that + list, not the platonic set of all possible defects. Every claim is made + against the list, so "covered" is checkable where "found everything" + isn't. +2. **Net the behavior.** Bring the surface under characterization tests + (Normal/Boundary/Error per unit — see `testing.md`, and the + `testing-standards` skill for the characterization recipe) before changing + anything. This is the objective floor: writing a characterization test is + mechanical (record what the code does, not what it should), so it scales + across the surface, and it doubles as the safety net that makes any + later refactor falsifiable. +3. **Disposition every finding.** Run the relevant audits (a fixed + footgun/OWASP checklist, `/refactor`, `/review-code`) and give **every** + finding a verdict: fixed, filed as its own task, or declined with a + one-line reason. "Looked and it's fine" is not a disposition. The + measurement is zero undispositioned findings, not zero findings. +4. **Gate on an objective floor.** Static analysis clean (linter, + type-checker, `shellcheck`), the test suite green before and after, and + coverage of the enumerated surface (a per-unit test checklist, or a real + coverage number where the tooling exists). + +The **qualifying answer** is then a dispositioned report — surface split +covered/uncovered, tests before → after, static-analysis result, the audit +matrix fully dispositioned, all green — not a claim of perfection. The +honest limit stays honest ("no visible bugs" means "every enumerated path +passes its characterization set and clears the audit," never "zero bugs +exist"), but the criterion is now falsifiable, which is what lets the task +carry `:solo:`. Write these criteria into the task body at creation or +review time; a task that can't be given them stays non-`:solo:` until it +can. + ### Bug priority from severity × frequency (mandatory where a codebase exists) Some projects carry a codebase — source the project maintains under version @@ -217,6 +318,7 @@ A completed sub-task disappears as a task and becomes an in-place event-log entr 2. Generate the timestamp with `date "+%Y-%m-%d %a @ %H:%M:%S %z"`. 3. Reword the original imperative title into the past-tense action that landed. Trim or restate if the original wording doesn't fit the action. 4. Drop the `TODO`/`DOING` keyword, the priority cookie, and the tags. The body stays as the record of what was done (if useful). +5. Remove any `SCHEDULED:`/`DEADLINE:` planning line. The completion time lives in the heading now, so `CLOSED:` is redundant and an active planning date on a historical log entry is always wrong. Org renders any headline carrying an active `<...>` `SCHEDULED`/`DEADLINE` on the agenda, keyword or not, so a stale one pins the finished entry there as weeks-overdue forever. An interactive close (`org-log-done`) stamps `CLOSED:` but never strips a pre-existing planning line, which is exactly how the stale dates survive. **Example:** @@ -226,7 +328,7 @@ becomes *** 2026-05-15 Fri @ 12:58:08 -0500 Wired yasnippet for universal availability -**Enforcement.** This is applied at close time by whoever closes the task, but an interactive org close (`org-log-done` flips the keyword to `DONE` and stamps `CLOSED:`) never applies the dated rewrite, so level-3+ closes accumulate as `DONE` keywords. `todo-cleanup.el --convert-subtasks` (run in the `clean-todo` and wrap-up cleanup passes) normalizes them mechanically: it rewrites any level-3+ `DONE`/`CANCELLED`/`FAILED` heading into the dated form above, pulling the timestamp from the `CLOSED` cookie and keeping the heading text verbatim (a batch tool can't reliably past-tense a title — polish wording by hand where it matters). `lint-org.el` flags any that slip through (checker `subtask-done-not-dated`). So the depth rule holds even when tasks are closed interactively rather than by an agent applying this section. +**Enforcement.** This is applied at close time by whoever closes the task, but an interactive org close (`org-log-done` flips the keyword to `DONE` and stamps `CLOSED:`) never applies the dated rewrite, so level-3+ closes accumulate as `DONE` keywords. `todo-cleanup.el --convert-subtasks` (run in the `clean-todo` and wrap-up cleanup passes) normalizes them mechanically: it rewrites any level-3+ `DONE`/`CANCELLED`/`FAILED` heading into the dated form above, pulling the timestamp from the `CLOSED` cookie, dropping the whole planning line (`CLOSED`, `SCHEDULED`, and `DEADLINE` together — step 5), and keeping the heading text verbatim (a batch tool can't reliably past-tense a title — polish wording by hand where it matters). `lint-org.el` flags any that slip through: checker `subtask-done-not-dated` for a still-keyworded sub-task, and `dated-log-heading-active-timestamp` for a dated entry that kept an active `SCHEDULED`/`DEADLINE`. So the depth rule holds even when tasks are closed interactively rather than by an agent applying this section. ### Why depth-based @@ -309,6 +411,10 @@ tasks — dated entries at `***` and deeper, terminal keyword at `**`. *** 2026-05-15 Fri @ 14:00:00 -0500 <what was answered or done> Generate the timestamp with `date "+%Y-%m-%d %a @ %H:%M:%S %z"`. + Remove any `SCHEDULED:`/`DEADLINE:` planning line too, same as the + sub-task rule above — a dated event-log entry carries its date in the + heading, and an active planning date left on it pins the finished entry + to the agenda forever. - **At `**` — terminal keyword, like any top-level task.** Change `VERIFY` to `DONE` (answered / check passed) or `CANCELLED` (abandoned), diff --git a/claude-rules/working-files.md b/claude-rules/working-files.md index 2432268..b915579 100644 --- a/claude-rules/working-files.md +++ b/claude-rules/working-files.md @@ -26,6 +26,28 @@ hits a nested path instead of a single canonical name. Always rename the files individually with a shared prefix so they sort together but live as flat siblings in `assets/`. +## `working/` Is Version-Controlled From Creation + +`working/` holds the project work currently being developed, so it is +tracked in git the moment a task subdirectory and its artifacts are created — +not staged locally and excluded until it graduates. `working/` is the tracked +home of in-progress work, and it is never added to `.gitignore` (the +install/sweep tooling deliberately leaves it out). + +Filing on completion **reorganizes** durable artifacts into their permanent +homes; it does not mark the point at which they first become durable. The +artifacts were durable — and tracked — from creation. Graduation is a move, +not a promotion from throwaway to keep. + +The corollary: genuinely disposable work does not belong in `working/`. +Ephemeral, single-use, or regenerable artifacts — scratch output, a +throwaway conversion, intermediate data you will delete — go in a project-root +`temp/` directory (gitignored) or system `/tmp`, never in `working/`. +`working/` is for work that will graduate; `temp/` is for work that will be +thrown away. The install tooling ignores `temp/` in both track and +gitignore-mode projects, since ephemerality is independent of whether a +project tracks its `.ai/` tooling. + ## Directory Layout <project-root>/ diff --git a/claude-templates/.ai/notes.org b/claude-templates/.ai/notes.org index c16863b..311a86f 100644 --- a/claude-templates/.ai/notes.org +++ b/claude-templates/.ai/notes.org @@ -6,19 +6,19 @@ This file contains project-specific information for this project. -**When to read this:** +- When to read this: - At the start of EVERY session (after reading protocols.org) - When needing project context or history - When checking reminders or pending decisions -**What's in this file:** +- What's in this file: - Project-specific context and goals - Pending decisions - Active reminders -**Session history is NOT in this file.** Each session's record lives in =.ai/sessions/YYYY-MM-DD-HH-MM-description.org= — one file per session. Catch-up reads the Summary sections of the most recent 5. +- Session history is NOT in this file. Each session's record lives in =.ai/sessions/YYYY-MM-DD-HH-MM-description.org= — one file per session. Catch-up reads the Summary sections of the most recent 5. -**For protocols and conventions, see:** [[file:protocols.org][protocols.org]] +- For protocols and conventions, see [[file:protocols.org][protocols.org]]. * Project-Specific Context @@ -48,9 +48,9 @@ This section tracks decisions that need Craig's input before work can proceed. - Include: What needs to be decided, options available, why it matters - Remove decisions once resolved (the resolution is captured in the Session Log of the session where it was resolved) -**Example format:** +- Example format: #+begin_example -** Feature Name or Topic +,** Feature Name or Topic Craig needs to decide on [specific question]. diff --git a/claude-templates/.ai/protocols.org b/claude-templates/.ai/protocols.org index 5cd69d4..b291d9e 100644 --- a/claude-templates/.ai/protocols.org +++ b/claude-templates/.ai/protocols.org @@ -84,7 +84,7 @@ Do NOT estimate, guess, or rely on memory. Just run the command. It takes one se Every session pulls rulesets first, then the local project repo. Rulesets carries the canonical behavioral rules and =.ai/= templates (the old =claude-templates= repo is folded in as a subtree at =rulesets/claude-templates/=); the project pull lands commits pushed from other machines or teammates since the last session. -Resolve any dirty-tree or merge issue at each step before moving on. Both pulls run as =git pull --ff-only= (or =git merge --ff-only= against the fetched upstream), so anything non-trivial — non-fast-forward history, dirty working tree, diverged branches — aborts. Surface the state and stop. Never auto-stash, auto-merge, or auto-rebase; the user resolves the conflict before further work. +Resolve any sync-blocking tree or merge issue at each step before moving on. The shared =git-worktree-gate sync-safe= policy permits untracked deliveries beneath =inbox/= so receiving a handoff never prevents another project from refreshing rulesets; every staged or tracked change, dirty submodule, Git operation in progress, or untracked path outside =inbox/= blocks. Both pulls run as =git pull --ff-only= (or =git merge --ff-only= against the fetched upstream), so non-fast-forward history and diverged branches also abort. Surface the state and stop. Never auto-stash, auto-merge, or auto-rebase; the user resolves the conflict before further work. Mechanics live in =startup.org= Phase A.0. The rule lives here because it governs the very first action of every session: load the freshest behavioral rules and templates before anything else runs. @@ -106,7 +106,7 @@ The epoch is baked into the id by the spawner, never minted inside =session-cont Resolve the path with =.ai/scripts/session-context-path= rather than hardcoding =.ai/session-context.org=; it prints the right path for the current =AI_AGENT_ID=. Fall back to =.ai/session-context.org= if the script isn't present (older checkouts mid-sync). Everything below — the record/recovery purpose, the update triggers, the startup existence check, the wrap-up rename — operates on that resolved path. The prose says "session-context.org" as the default name; read it as "the resolved active path" when =AI_AGENT_ID= is set. -A helper instance (a second agent running in this project while a primary session is live) follows a different contract: it skips the pulls and rsync, makes only scoped single-heading edits to shared files, leaves all git mutation to the primary, and wraps up by archiving its own context file without committing. The full rules — read/write tiers, data-integrity, light startup, helper wrap-up — live in [[file:workflows/helper-mode.org][workflows/helper-mode.org]]. A session is a helper only when something routes it there (the =ai --helper= launcher, startup's roster check, or an explicit "you are a helper" instruction); the routing itself ships behind the helper-instance feature gate and isn't live yet. +A helper instance (a second agent running in this project while a primary session is live) follows a different contract: it skips the pulls and rsync, makes only scoped single-heading edits to shared files, leaves all git mutation to the primary, and wraps up by archiving its own context file without committing. The full rules — read/write tiers, data-integrity, light startup, helper wrap-up — live in [[file:workflows/helper-mode.org][workflows/helper-mode.org]]. A session is a helper only when something routes it there: the =ai --helper= launcher (live — it checks the roster, assigns the id, and opens the helper in its own tmux window) or an explicit "you are a helper" instruction. Startup's roster check is *not* built, so a bare =claude= launched into a project that already has a live session will run full primary startup regardless. Launch helpers with =ai --helper=. This file serves two purposes with one mechanism: 1. *Crash recovery* — if the session dies mid-work, the live file is all that's left. On 2026-01-22 a session crashed during a 20-minute design discussion and all context was lost because this file wasn't being updated. @@ -187,6 +187,8 @@ Canonical rule: =~/code/rulesets/claude-rules/cross-project.md=. Every in-progress task that produces files (drafts, source documents, diagrams, scripts, sub-deliverables) gets a dedicated subdirectory under =<project-root>/working/=, named after the task. All artifacts for that task live in that subdirectory until the task is marked done. +=working/= is version-controlled from creation — it's the tracked home of in-progress work, never gitignored. Filing on completion *reorganizes* durable artifacts into permanent homes; it doesn't mark when they became durable (they were durable, and tracked, from the start). Genuinely disposable artifacts go in a gitignored =temp/= (or =/tmp=), never =working/=; the install tooling ignores =temp/= in both track and gitignore modes. + When the task ships, files are **renamed individually** (standard form: =YYYY-MM-DD-<task-slug>-<descriptor>.<ext>=) and **moved flat** into the appropriate permanent home (typically =assets/= or an area-specific =<area>/assets/=). The working subdirectory is then empty and gets deleted. ***Never rename the directory itself as a substitute for filing.*** The point is to keep =assets/= flat-searchable — a nested =assets/old-tech-deck-2026/slide.png= is harder to find than =assets/2026-05-18-tech-deck-vol2-slide-04-diagram.png=. @@ -205,6 +207,8 @@ Check =inbox/= at every task boundary (after finishing a unit of work, before re Exit 1 means handoffs are pending — process them per =inbox.org= process mode. For each accepted handoff, the act-vs-file rule: *act now* when it's clear, bounded, low-risk, in-scope, and cheaper than deferring — just do it, no asking; *file* otherwise — ask first, with filing as option 1 and "do it now" as option 2; *ask* if unsure. Exception: a proposal to change a shared asset (template workflow, rule, skill, synced script) or a substantive convention never silently acts now — it goes through the inbox engine's skeptical review and its approval (or park) step. Always reply to a handoff's sender (confirm on accept, the why on reject). Full process, the reply discipline, and the opt-in background-monitor =/loop= recipe live in =inbox.org= monitor mode. +A machine-global =Stop= hook (=inbox-boundary-check.sh=) backs this rule so it isn't prose-only. When handoffs are pending it blocks the turn once and injects the count, so a task boundary can't pass with items unseen. It soft-nudges rather than hard-blocks: on the harness re-entry it steps aside, so a mid-task pause to ask "what's next" is never hijacked into inbox processing. The rule above still governs what to do with the items; the hook only makes sure you look. + ** Recursive Reads — Honor =.aiignore= Before a naive recursive read or glob of a project tree (file inventories, "what's in this repo", broad greps), skip the noise: dependency trees (=node_modules/=, =.venv/=), build output (=dist/=, =build/=, =coverage/=), language caches (=__pycache__/=, =.pytest_cache/=, =*.pyc=), editor/OS cruft, and generated token/OAuth artifacts. These waste tokens and skew project summaries even when gitignored — a recursive read sees the disk, not git. @@ -246,13 +250,29 @@ Execute the wrap-up workflow (details in Session Protocols section below): Execute the suspend workflow ([[file:workflows/suspend.org][suspend.org]]): a capture-only mid-session pause for an abrupt departure. It appends a resume-weighted =SUSPENDED= entry to the Session Log, notes uncommitted work, and LEAVES =.ai/session-context.org= in place so the next startup resumes from it — no archive, no teardown, no valediction. The capture-only counterpart to "wrap it up" (which ends + archives + tears down) and to =/flush= (which prompts =/clear= and resumes the same session). "I need to go" is broad — if it reads as a conversational aside, confirm before suspending. +* Colloquialisms and Expansions + +Shorthand phrases Craig uses that expand to a defined action the agent applies without asking. The set is extensible: a project may add its own entries, and new shared shorthands land here. + +** "the list": the Before-Close Queue + +"Put X on the list" or "add X to the list" appends X to the Before-Close Queue, a FIFO queue of tasks and actions to finish before the session closes. Work it oldest-first at wrap-up, before teardown (=wrap-it-up.org= Step 1 works it before finalizing the Summary), and surface anything unfinished in the valediction rather than dropping it. + +The queue lives in the session anchor (=.ai/session-context.org=) under a =* Before-Close Queue= heading. Create the heading on the first "put it on the list" if it's absent, then append one line per item. It's session-scoped: it resets when the anchor is archived at wrap. Anything that must outlive the session is a =todo.org= task instead, not a list item. + +** "tell <project> <message>": cross-project handoff + +"Tell <project> <message>" drops the message in that project's =inbox/= via =inbox-send= (=python3 .ai/scripts/inbox-send.py <project> --text "<message>"=), the sanctioned cross-project handoff. Never write another project's =todo.org= or =inbox/= directly. Resolve =<project>= the way =inbox-send= does (basename match, dots stripped); if it's ambiguous, ask which project rather than guessing. + * User Information ** Calendar Management Three ways to access Craig's calendars: Google Calendar MCP (preferred, both personal + work accounts), gcalcli (fallback, personal only), Emacs org files (read-only viewer). -For tool recipes, authentication details, and credentials, see [[file:references/calendar-reference.org][calendar-reference.org]]. +For tool recipes and account details, read the calendar workflows in =.ai/workflows/=: =add-calendar-event.org=, =edit-calendar-event.org=, =delete-calendar-event.org=, =read-calendar-events.org=. They carry the MCP tool names, both account ids, the gcalcli fallback, and the conflict-check discipline. + +Credentials are needed only for a re-auth Craig performs himself. The MCP bundle's =mcp/README.org= in the rulesets repo is the authority: =gcp-oauth.keys.json= is gitignored and regenerated at install from a base64 var in the bundle, never committed. Named in prose rather than linked, because that path isn't synced into consuming projects. ** GPG Keys @@ -356,9 +376,19 @@ Craig runs a pure Wayland setup (Hyprland) and avoids XWayland/Xorg apps. - Clipboard: Use =wl-copy= and =wl-paste= (NOT =xclip= or =xsel=) - Window management: Use Hyprland commands (NOT =xkill=, =xdotool=, etc.) - Prefer Wayland-native tools over X11 equivalents -- Open URLs in browser: Use =google-chrome-stable "URL" &>/dev/null &= - - The =&>/dev/null &= is required to detach the process and suppress output - - Without it, the command may appear to hang or produce no result +- Open URLs in browser: invoke Chrome directly — never =xdg-open=, which returned success in a home session on 2026-07-26 while no tab appeared. + + Chrome is normally already running, and in that case it hands the URL to the live session and exits immediately (rc 0), printing =Opening in existing browser session.= on *stdout*. So run it in the foreground and read that line as the confirmation the tab actually opened: + + #+begin_src bash + google-chrome-stable --new-tab "URL" + #+end_src + + Don't redirect stdout away while checking for that line — verified 2026-07-27 on ratio: with =2>/dev/null= the message still appears (it isn't stderr), and with =>/dev/null= it vanishes. + + Several URLs in one invocation open as separate tabs (=google-chrome-stable --new-tab "URL1" "URL2"=). Pass them as separate words or an array — the Bash tool runs zsh, which does not word-split an unquoted =$urls= variable, so a space-joined string arrives as one malformed argument (see the zsh note below). + + *Cold start.* If Chrome is *not* already running, the command becomes the browser process and blocks. Detach that case with =&>/dev/null &=, accepting that the confirmation line is discarded — there is no session to confirm into. Don't apply the detach form unconditionally: it suppresses the very output the warm path is verified by. *** Shell aliases (=ls= → =exa=) Craig's shell aliases =ls= to =exa=, which prints nothing to non-TTY pipes (e.g. when capturing =ls= output in a Bash tool call). The result looks like the directory is empty when it isn't. @@ -412,27 +442,29 @@ Full usage: =notify --help= or see =~/.local/bin/notify= - =atq= - list all scheduled alarms - =atrm [number]= - remove an alarm by its queue number -** Paging Craig — the agent pager +** Reaching Craig — the notification vocabulary -"Page me" has two channels; pick by where Craig is. Both work from any agent runtime — nothing here is Claude-specific. +Two channels, two trigger words. "page me" is the desktop, "text me" is the phone, "text and page me" is both. Pick by where Craig is, and default to both when a run can't tell. Both work from any agent runtime (nothing here is Claude-specific). The words are what Craig says; a run deciding on its own maps the same way (away run texts, at-desk run pages, unsure does both). -- *At his laptop/desktop* — desktop =notify ... --persist= (above). It reaches him on the machine and stays up until dismissed. +- *"page me" — at his laptop/desktop.* A desktop =notify ... --persist= that reaches him on the machine and stays up until dismissed. #+begin_src bash notify info "Title" "Message" --persist #+end_src -- *Away from his laptop/desktop* — page his phone over Signal with the *agent pager*: +- *"text me" — away from his machine.* A Signal push to his phone via =agent-text=: #+begin_src bash - agent-page "Message for Craig's phone" + agent-text "Message for Craig's phone" #+end_src - =agent-page= (in =~/.local/bin= via the rulesets install) sends from the dedicated pager identity (+15045173983, registered in velox's signal-cli) to Craig's Signal account UUID, firing a normal mobile push. On velox it sends directly; on any other tailnet machine it ssh-relays the send to velox. Verified end to end 2026-07-13. Never page Craig's phone *number* — it reads as unregistered in Signal's directory; the script already targets the UUID. + =agent-text= (in =~/.local/bin= via the rulesets install) sends from the dedicated Signal identity (+15045173983) to Craig's Signal account UUID, firing a normal mobile push. The account is registered on velox (primary) and ratio (linked device), so either sends directly; a machine without it ssh-relays to velox. Verified end to end 2026-07-13 (velox) and 2026-07-20 (ratio). Never target Craig's phone *number* (it reads as unregistered in Signal's directory); the script targets the UUID. + + Caveats: a relay from a non-linked machine needs velox up on the tailnet, and each device holding the account wants a periodic =receive= (the signal-receive timer handles that). The full runbook lives in rulesets =docs/design/=. - Caveats: velox must be up and on the tailnet (the script says so and names the desktop fallback when the relay fails), and the signal-cli account wants a periodic =receive= — both tracked on the rulesets Signal-pager task, which owns the full runbook. +- *"text and page me" — both.* Fire =agent-text= and =notify= together. The phone reaches him now, the desktop note waits for his return. This is the default when a run can't tell whether he's away. -On velox, Claude sessions may also have the *signal-mcp* tool (=send_message_to_user=, same pager identity) — fine to use there, but it exists only in velox's local MCP config, so =agent-page= is the portable habit. Do *not* use the old =page-signal= shell script (removed 2026-06-12). +On velox, Claude sessions may also have the *signal-mcp* tool (=send_message_to_user=, same identity), fine to use there, but it exists only in velox's local MCP config, so =agent-text= is the portable habit. The tool was named =agent-page= before 2026-07-20; a deprecated =agent-page= shim still delegates to =agent-text=. Do *not* use the old =page-signal= shell script (removed 2026-06-12). * Session Protocols @@ -547,12 +579,13 @@ When monitoring a long-running process (rsync, large downloads, builds, VM tests ** "Wrap it up" / "That's a wrap" / "Let's call it a wrap" -When Craig says any of these phrases (or variations), execute the wrap-up workflow: [[file:workflows/wrap-it-up.org][wrap-it-up.org]]. Four steps: +When Craig says any of these phrases (or variations), execute the wrap-up workflow: [[file:workflows/wrap-it-up.org][wrap-it-up.org]]. Five load-bearing steps: 1. *Finalize the Summary* in =.ai/session-context.org= (populate the 5 subsections from the Session Log) 2. *Rename* =.ai/session-context.org= → =.ai/sessions/YYYY-MM-DD-HH-MM-description.org= 3. *Git commit + push* to all remotes (see Git Commit Requirements) -4. *Valediction* — brief, warm, specific closing +4. *Certify the clean tree* with =git-worktree-gate certify=. Any remaining staged, unstaged, untracked, submodule, or in-progress-operation state blocks wrap entirely; report each path and the exact decision needed. There is no dirty-file deferral. +5. *Valediction* — brief, warm, specific closing, reachable only after certification The absence of =.ai/session-context.org= after wrap-up is the signal that the session ended cleanly. If the file is still there at the next session start, the previous session was interrupted. diff --git a/claude-templates/.ai/references/calendar-reference.org b/claude-templates/.ai/references/calendar-reference.org deleted file mode 100644 index 5791b08..0000000 --- a/claude-templates/.ai/references/calendar-reference.org +++ /dev/null @@ -1,66 +0,0 @@ -#+TITLE: Calendar Reference -#+AUTHOR: Craig Jennings - -Tool recipes, authentication, and credentials for Craig's calendar -setup. Three access methods, in order of preference. - -* Google Calendar MCP Server (preferred for all calendar operations) - -Craig has the =@cocal/google-calendar-mcp= MCP server configured at user scope (=~/.claude.json=). It provides full read/write access to Google Calendar via MCP tools. - -Two accounts are authenticated: -- *personal* — craigmartinjennings@gmail.com (primary: "Craig Google") -- *work* — craig.jennings@deepsat.com (primary: "Craig Deepsat") - -MCP tools available: -- =list-events=, =search-events=, =get-event= — read events -- =create-event=, =create-events= — add events -- =update-event= — modify events -- =delete-event= — remove events -- =list-calendars=, =list-colors= — calendar metadata -- =get-freebusy= — check availability -- =manage-accounts= — add/remove/list authenticated accounts -- =respond-to-event= — accept/decline invitations -- =get-current-time= — current time in any timezone - -Use =account_id: "personal"= or =account_id: "work"= to specify which account. - -Default calendar for adding events: "Craig Google" (personal account). - -Calendar workflows are available alongside this reference: add-calendar-event, edit-calendar-event, delete-calendar-event, read-calendar-events. - -If re-authentication is needed: -- Use the =manage-accounts= MCP tool with =action: "add"= and the account nickname -- OAuth credentials: =~/projects/homelab/assets/gcp-oauth.keys.json= -- Google Cloud app is in production mode (tokens don't expire after 7 days) -- See =~/projects/homelab/.ai/gcalcli-setup.org= for Google Cloud project details - -* gcalcli (fallback for personal account only) - -Craig has =gcalcli= installed via pipx, authenticated to his personal Google account only. - -#+begin_src bash -gcalcli agenda # upcoming events -gcalcli calw # weekly view -gcalcli add --title "..." --when "..." --duration "60" # add event -gcalcli search "..." # search events -gcalcli delete "..." # delete event -#+end_src - -Use =--calendar "Craig Google"= when adding events. - -gcalcli does NOT have access to the work (DeepSat) calendar. Use the MCP server for work calendar operations. - -If gcalcli needs re-authentication, credentials are stored in the homelab project: =~/projects/homelab/assets/gcalcli-client-secret.json.gpg= (GPG encrypted). - -* Emacs org files (read-only, for viewing schedules) - -Craig's calendars are at: =~/.emacs.d/data/*cal.org= (gcal.org, dcal.org, pcal.org) - -These files are **READ-ONLY** — NEVER add anything to them. - -Use this to: -- Check meeting times and schedules -- Verify when events occurred -- See what's upcoming -- Note: only updated periodically when Emacs is running — may be stale diff --git a/claude-templates/.ai/scripts/agent-lock b/claude-templates/.ai/scripts/agent-lock new file mode 100755 index 0000000..634412c --- /dev/null +++ b/claude-templates/.ai/scripts/agent-lock @@ -0,0 +1,248 @@ +#!/usr/bin/env bash +# agent-lock — a mkdir-atomic advisory lock for agent workflows. +# +# Why not flock: every Bash call an agent makes is its own short-lived shell, +# so an flock taken in one /loop turn is gone by the next. This helper persists +# the lock on disk between calls (an atomic mkdir is the acquire), and a crashed +# holder's lock self-clears via age-based staleness reclaim instead of wedging +# every later acquire. +# +# Serves both of sentry's locks (the single-runner lock and the roam-write +# lock); callers pass a name, never a path — the helper owns the path scheme. +# +# Usage: +# agent-lock acquire <name> [--ttl=SECONDS] [--wait[=SECONDS]] +# Atomic acquire. exit 0 on win (fresh, or reclaimed from a stale holder); +# exit 1 when a live lock already holds <name> (deferred — a note names the +# holder on stderr). --wait polls up to SECONDS (default 30) before +# deferring; without it, acquire is single-shot win-or-lose. --ttl records +# the staleness horizon in the lock's metadata (default below). +# agent-lock refresh <name> +# Heartbeat: re-touch a held lock's mtime so it stays young. A runner +# refreshes its own lock between passes, so a live run's lock is never older +# than one pass and the TTL sizes to the longest single pass. exit 1 if the +# lock is absent (nothing to refresh). +# agent-lock release <name> +# Remove the lock. Idempotent: exit 0 even if already free. +# agent-lock status <name> +# Print "free" | "held ..." | "stale ..." plus metadata. exit 0 (a query +# never fails on lock state). +# agent-lock path <name> +# Print the resolved lock-directory path without creating it. +# +# Lock home (the helper owns this; callers pass names only): +# $AGENT_LOCK_DIR/<name>/ when AGENT_LOCK_DIR is set (tests / advanced) +# $XDG_RUNTIME_DIR/agent-locks/<name>/ the tmpfs runtime dir /run/user/<uid> +# (host-local, out of every repo, +# cleared on reboot). XDG_RUNTIME_DIR is +# the standard handle for it and is set +# in sentry's interactive launch. +# ${XDG_CACHE_HOME:-~/.cache}/agent-locks/<name>/ fallback where no runtime +# dir exists (XDG_RUNTIME_DIR unset or +# unwritable — a headless/container box) +# +# tmpfs residence is deliberate: a lock under ~/org/roam would ride roam-sync's +# `git add -A` to the other machine as a phantom hold. Host-locality is by +# construction, and reboot clears any lock a crash left behind for free. +# +# Staleness is age-based on the metadata file's mtime versus the lock's own +# recorded TTL. Heartbeat re-touches the mtime; a reclaim is always surfaced, +# never silent. + +set -euo pipefail + +DEFAULT_TTL=600 # 10 min: sized to the longest single sentry pass, since a + # live runner heartbeats between passes and stays young. +DEFAULT_WAIT=30 # bounded-wait budget for --wait (capture-guard's shape). +WAIT_INTERVAL=3 # poll cadence while waiting on a busy lock. + +usage() { + echo "usage: agent-lock {acquire|refresh|release|status|path} <name> [--ttl=N] [--wait[=N]]" >&2 + exit 2 +} + +# Resolve the base directory that holds all lock dirs, per the home scheme above. +lock_base() { + if [ -n "${AGENT_LOCK_DIR:-}" ]; then + printf '%s\n' "$AGENT_LOCK_DIR" + elif [ -n "${XDG_RUNTIME_DIR:-}" ] && [ -d "$XDG_RUNTIME_DIR" ] && [ -w "$XDG_RUNTIME_DIR" ]; then + printf '%s/agent-locks\n' "$XDG_RUNTIME_DIR" + else + printf '%s/agent-locks\n' "${XDG_CACHE_HOME:-$HOME/.cache}" + fi +} + +# Validate a lock name: non-empty, no path separators (so a name can never +# escape the base dir). +valid_name() { + case "$1" in + ''|*/*|.|..) return 1 ;; + *) return 0 ;; + esac +} + +lock_dir() { printf '%s/%s\n' "$(lock_base)" "$1"; } +meta_path() { printf '%s/meta\n' "$(lock_dir "$1")"; } + +# Read a key from a lock's metadata file; empty if absent. +meta_get() { + local key="$1" file="$2" + [ -f "$file" ] || return 0 + sed -n "s/^${key}=//p" "$file" | head -n1 +} + +# Age of a lock in whole seconds, from the metadata mtime. +lock_age() { + local file="$1" mtime now + mtime=$(stat -c %Y "$file" 2>/dev/null) || return 1 + now=$(date +%s) + printf '%s\n' "$((now - mtime))" +} + +# True when a lock dir exists but its age exceeds its recorded TTL. +is_stale() { + local name="$1" file age ttl + file="$(meta_path "$name")" + [ -f "$file" ] || return 1 + age="$(lock_age "$file")" || return 1 + ttl="$(meta_get ttl "$file")" + [ -n "$ttl" ] || ttl="$DEFAULT_TTL" + [ "$age" -gt "$ttl" ] +} + +# Write the metadata file for a freshly-taken lock. +write_meta() { + local name="$1" ttl="$2" file + file="$(meta_path "$name")" + { + printf 'pid=%s\n' "$$" + printf 'host=%s\n' "$(uname -n)" + printf 'acquired=%s\n' "$(date +%Y-%m-%dT%H:%M:%S%z)" + printf 'ttl=%s\n' "$ttl" + } > "$file" +} + +# One-line holder description for surfaced notes. +holder_desc() { + local file="$1" + printf "pid=%s host=%s age=%ss ttl=%ss" \ + "$(meta_get pid "$file")" "$(meta_get host "$file")" \ + "$(lock_age "$file" 2>/dev/null || echo '?')" "$(meta_get ttl "$file")" +} + +# Attempt a single atomic acquire. exit 0 win, 1 busy (live holder). +try_acquire() { + local name="$1" ttl="$2" dir file + dir="$(lock_dir "$name")" + file="$(meta_path "$name")" + mkdir -p "$(lock_base)" + + if mkdir "$dir" 2>/dev/null; then + write_meta "$name" "$ttl" + return 0 + fi + + # Directory exists. Reclaim it if the holder is stale; otherwise it's busy. + if is_stale "$name"; then + # Claim the stale dir atomically before removing it. `mv` of a directory is + # atomic, so when two acquirers both see the lock stale, only one's rename + # of $dir succeeds — the other's fails because $dir is already gone, and it + # falls through to busy. Never `rm -rf $dir` directly: a plain remove lets + # the loser delete the winner's freshly-created lock and double-acquire. + local claimed="$dir.stale.$$" + if mv "$dir" "$claimed" 2>/dev/null; then + echo "agent-lock: reclaimed stale lock '$name' ($(holder_desc "$claimed/meta"))" >&2 + rm -rf "$claimed" + # mkdir stays the sole grant: a concurrent fresh acquirer may win here, + # in which case our mkdir fails and we correctly defer to it. + if mkdir "$dir" 2>/dev/null; then + write_meta "$name" "$ttl" + return 0 + fi + fi + fi + return 1 +} + +cmd_acquire() { + local name="$1"; shift + local ttl="$DEFAULT_TTL" wait_total=0 + while [ $# -gt 0 ]; do + case "$1" in + --ttl=*) ttl="${1#--ttl=}" ;; + --ttl) shift; ttl="${1:-}" ;; + --wait) wait_total="$DEFAULT_WAIT" ;; + --wait=*) wait_total="${1#--wait=}" ;; + *) usage ;; + esac + shift + done + case "$ttl" in ''|*[!0-9]*) usage ;; esac + case "$wait_total" in *[!0-9]*) usage ;; esac + + local elapsed=0 + while :; do + if try_acquire "$name" "$ttl"; then + exit 0 + fi + if [ "$elapsed" -ge "$wait_total" ]; then + echo "agent-lock: '$name' busy ($(holder_desc "$(meta_path "$name")")); deferring" >&2 + exit 1 + fi + local remaining=$((wait_total - elapsed)) step + step=$(( remaining < WAIT_INTERVAL ? remaining : WAIT_INTERVAL )) + sleep "$step" + elapsed=$((elapsed + step)) + done +} + +cmd_refresh() { + local name="$1" file + file="$(meta_path "$name")" + [ -f "$file" ] || exit 1 + # Re-stamp acquired and bump mtime so the age clock restarts. + local ttl; ttl="$(meta_get ttl "$file")"; [ -n "$ttl" ] || ttl="$DEFAULT_TTL" + write_meta "$name" "$ttl" + exit 0 +} + +cmd_release() { + local name="$1" dir + dir="$(lock_dir "$name")" + rm -rf "$dir" + exit 0 +} + +cmd_status() { + local name="$1" dir file + dir="$(lock_dir "$name")" + file="$(meta_path "$name")" + if [ ! -d "$dir" ]; then + echo "free $name" + exit 0 + fi + local state="held" + is_stale "$name" && state="stale" + echo "$state $name pid=$(meta_get pid "$file") host=$(meta_get host "$file") acquired=$(meta_get acquired "$file") ttl=$(meta_get ttl "$file") age=$(lock_age "$file" 2>/dev/null || echo '?')s" + exit 0 +} + +cmd_path() { + lock_dir "$1" + exit 0 +} + +[ $# -ge 1 ] || usage +subcmd="$1"; shift +[ $# -ge 1 ] || usage +name="$1"; shift +valid_name "$name" || usage + +case "$subcmd" in + acquire) cmd_acquire "$name" "$@" ;; + refresh) cmd_refresh "$name" ;; + release) cmd_release "$name" ;; + status) cmd_status "$name" ;; + path) cmd_path "$name" ;; + *) usage ;; +esac diff --git a/claude-templates/.ai/scripts/apkg-to-orgdrill.py b/claude-templates/.ai/scripts/apkg-to-orgdrill.py new file mode 100755 index 0000000..79e24a4 --- /dev/null +++ b/claude-templates/.ai/scripts/apkg-to-orgdrill.py @@ -0,0 +1,251 @@ +#!/usr/bin/env -S uv run --script +# /// script +# requires-python = ">=3.11" +# dependencies = [] +# /// +"""Convert an Anki .apkg deck into an org-drill file (inverse of flashcard-to-anki.py). + +The flashcard pipeline is otherwise one-directional (org-drill -> apkg). +Decks curated on the phone, and orphaned apkgs whose .org source was never +saved, can't get back into the org source of truth. This recovers them. + +Reading needs no third-party library: an apkg is a zip holding +collection.anki2 / .anki21 (an Anki sqlite db) plus a media blob, so stdlib +zipfile + sqlite3 suffice. genanki is only needed to write apkgs, not read +them. + +Mapping (mirrors flashcard-to-anki.py's parse/build, inverted): + - Deck name (from the apkg) -> #+TITLE: + - Note Front -> ** <Front> :drill: + - Note Back (HTML) -> entry body (<br> -> newlines, + &/</> unescaped, + <hr id="answer"> stripped) + - Note tag -> * <tag> section grouping + (best-effort: the tag is a slug, + so it won't round-trip to the exact + original section title — a human + retitles) + - A fresh :ID: UUID per card -> so the output is org-drill-valid + +GUIDs in flashcard-to-anki.py are derived from the Front text, not the +:ID:, so a deck regenerated from recovered org still matches existing phone +cards by Front. Only Front/Back (Basic) note types convert; other models +(cloze, etc.) are skipped with a warning rather than silently dropped. + +Usage: + apkg-to-orgdrill.py <input.apkg> # one <deck-slug>.org per deck in cwd + apkg-to-orgdrill.py <input.apkg> --output-dir DIR + apkg-to-orgdrill.py <input.apkg> --deck "Name" --output deck.org +""" +from __future__ import annotations + +import argparse +import json +import re +import sqlite3 +import sys +import tempfile +import uuid +import zipfile +from collections import OrderedDict +from dataclasses import dataclass +from pathlib import Path + +# Collection member names Anki uses, newest schema first. +COLLECTION_NAMES = ("collection.anki21", "collection.anki2") + +_BR_RE = re.compile(r"<br\s*/?>", re.IGNORECASE) +_ANSWER_HR_RE = re.compile(r'<hr id="answer">', re.IGNORECASE) +_MEDIA_RE = re.compile(r"<img\b|\[sound:|<audio\b|<video\b", re.IGNORECASE) + + +@dataclass +class Note: + deck: str + front: str + back_html: str + tag: str + + +def html_to_org_body(back_html: str) -> list[str]: + """Invert flashcard-to-anki.py's back-of-card HTML into org body lines. + + <br> (all spellings) and a stray answer <hr> become line breaks; the + entity unescape undoes escape_html, which escaped ``&`` first — so ``&`` + is unescaped last here, or a literally-escaped ``<`` in the source + would wrongly collapse to ``<``. + """ + if not back_html: + return [] + s = _ANSWER_HR_RE.sub("\n", back_html) + s = _BR_RE.sub("\n", s) + s = s.replace("<", "<").replace(">", ">").replace("&", "&") + return s.split("\n") + + +def _slug(title: str) -> str: + return re.sub(r"[^a-z0-9]+", "-", title.lower()).strip("-") + + +def _read_collection(db_path: Path) -> list[Note]: + con = sqlite3.connect(db_path) + try: + row = con.execute("SELECT decks, models FROM col LIMIT 1").fetchone() + if row is None: + raise ValueError("collection has no col row") + decks_json, models_json = row + decks = {int(k): v["name"] for k, v in json.loads(decks_json).items()} + models = { + int(k): [f["name"] for f in v["flds"]] + for k, v in json.loads(models_json).items() + } + + # A note's deck comes from its card; the Default deck (id 1) carries + # no cards from this pipeline, so it never shows up here. + nid_to_did: dict[int, int] = {} + for nid, did in con.execute("SELECT nid, did FROM cards"): + nid_to_did.setdefault(nid, did) + + notes: list[Note] = [] + for nid, mid, flds, tags in con.execute( + "SELECT id, mid, flds, tags FROM notes" + ): + field_names = models.get(mid) + if not field_names or "Front" not in field_names or "Back" not in field_names: + print( + f"apkg-to-orgdrill: skip note {nid} — model is not a Front/Back " + f"type (fields={field_names})", + file=sys.stderr, + ) + continue + fields = flds.split("\x1f") + fi, bi = field_names.index("Front"), field_names.index("Back") + front = fields[fi] if fi < len(fields) else "" + back_html = fields[bi] if bi < len(fields) else "" + + did = nid_to_did.get(nid) + if did is None: + continue # note with no card — orphan + deck = decks.get(did) + if deck is None: + continue + + tag_list = tags.split() + tag = tag_list[0] if tag_list else "drill" + + if _MEDIA_RE.search(back_html): + print( + f"apkg-to-orgdrill: note {nid} references media; org has no " + f"media path (left inline for a human to resolve)", + file=sys.stderr, + ) + notes.append(Note(deck=deck, front=front, back_html=back_html, tag=tag)) + return notes + finally: + con.close() + + +def read_apkg(path: Path) -> list[Note]: + """Read an .apkg and return its Front/Back notes. Raises on a malformed file.""" + with zipfile.ZipFile(path) as z: # BadZipFile if it isn't a zip + names = set(z.namelist()) + col_name = next((n for n in COLLECTION_NAMES if n in names), None) + if col_name is None: + raise ValueError(f"{path}: no collection.anki2/.anki21 inside the apkg") + with tempfile.TemporaryDirectory() as td: + db_path = Path(td) / col_name + db_path.write_bytes(z.read(col_name)) + return _read_collection(db_path) + + +def notes_to_org(notes: list[Note], deck_name: str, *, new_id=None) -> str: + """Render one deck's notes as an org-drill file in the house shape.""" + if new_id is None: + new_id = lambda: str(uuid.uuid4()) # noqa: E731 + groups: "OrderedDict[str, list[Note]]" = OrderedDict() + for n in notes: + groups.setdefault(n.tag, []).append(n) + + lines: list[str] = [f"#+TITLE: {deck_name}", ""] + for tag, group in groups.items(): + lines.append(f"* {tag}") + for n in group: + lines.append(f"** {n.front} :drill:") + lines.append(":PROPERTIES:") + lines.append(f":ID: {new_id()}") + lines.append(":END:") + lines.extend(html_to_org_body(n.back_html)) + lines.append("") + return "\n".join(lines).rstrip("\n") + "\n" + + +def convert(apkg_path: Path, *, new_id=None) -> "OrderedDict[str, str]": + """apkg -> {deck_name: org_text}, one entry per deck that has Front/Back cards.""" + by_deck: "OrderedDict[str, list[Note]]" = OrderedDict() + for n in read_apkg(apkg_path): + by_deck.setdefault(n.deck, []).append(n) + out: "OrderedDict[str, str]" = OrderedDict() + for deck, deck_notes in by_deck.items(): + out[deck] = notes_to_org(deck_notes, deck, new_id=new_id) + return out + + +def main() -> int: + parser = argparse.ArgumentParser( + description="Convert an Anki .apkg deck into an org-drill file.", + ) + parser.add_argument("input", type=Path, help="Path to the .apkg file.") + parser.add_argument("--deck", help="Only convert the deck with this exact name.") + parser.add_argument( + "--output", + type=Path, + help="Output .org path. Requires a single deck (use --deck to pick one).", + ) + parser.add_argument( + "--output-dir", + type=Path, + help="Directory for per-deck .org files (default: current directory).", + ) + args = parser.parse_args() + + input_path = args.input.expanduser().resolve() + if not input_path.is_file(): + print(f"error: {input_path} not found", file=sys.stderr) + return 1 + + by_deck = convert(input_path) + if args.deck: + by_deck = OrderedDict((k, v) for k, v in by_deck.items() if k == args.deck) + if not by_deck: + print(f"error: no deck named {args.deck!r} in {input_path}", file=sys.stderr) + return 1 + if not by_deck: + print(f"error: no Front/Back cards found in {input_path}", file=sys.stderr) + return 1 + + if args.output: + if len(by_deck) != 1: + print( + f"error: --output needs a single deck; {input_path} has " + f"{len(by_deck)} ({', '.join(by_deck)}). Use --deck or --output-dir.", + file=sys.stderr, + ) + return 1 + out = args.output.expanduser().resolve() + out.parent.mkdir(parents=True, exist_ok=True) + deck, org = next(iter(by_deck.items())) + out.write_text(org, encoding="utf-8") + print(f"wrote {out} ({org.count(':drill:')} cards, deck {deck!r})") + return 0 + + out_dir = (args.output_dir or Path.cwd()).expanduser().resolve() + out_dir.mkdir(parents=True, exist_ok=True) + for deck, org in by_deck.items(): + out = out_dir / f"{_slug(deck) or 'deck'}.org" + out.write_text(org, encoding="utf-8") + print(f"wrote {out} ({org.count(':drill:')} cards, deck {deck!r})") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/claude-templates/.ai/scripts/cj-remove-block.py b/claude-templates/.ai/scripts/cj-remove-block.py index 71c7b3d..d5137a3 100755 --- a/claude-templates/.ai/scripts/cj-remove-block.py +++ b/claude-templates/.ai/scripts/cj-remove-block.py @@ -16,8 +16,12 @@ Companion to the /respond-to-cj-comments skill and to cj-scan.py. from __future__ import annotations import argparse +import os import re +import shutil import sys +import tempfile +from datetime import datetime from pathlib import Path SRC_OPEN_RE = re.compile(r"^\s*#\+begin_src\s+cj:", re.IGNORECASE) @@ -57,12 +61,83 @@ def looks_like_cj_range(lines: list[str], start: int, end: int) -> tuple[bool, s f"Line {end} does not look like a #+end_src closing fence " f"(got: {last[:60]!r})" ) + + # The range must hold exactly ONE block. Checking only the first and last + # lines let a drifted range run from one block's opener to a *later* block's + # closer: validation passed and the removal silently deleted everything + # between, prose and headings included. Drift is the case this check exists + # for, so it has to look inside the range, not just at its ends. + for offset, line in enumerate(lines[start:end - 1], start=start + 1): + if SRC_CLOSE_RE.match(line): + return False, ( + f"Range {start}..{end} covers more than one cj block — " + f"a #+end_src appears at line {offset}, before the range ends. " + f"Re-scan for current line numbers; removing this range would " + f"delete everything between the two blocks." + ) + if SRC_OPEN_RE.match(line): + return False, ( + f"Range {start}..{end} covers more than one cj block — " + f"a second #+begin_src cj: appears at line {offset}. " + f"Re-scan for current line numbers." + ) return True, "" +def _backup(path: Path) -> Path: + """Copy path to /tmp before mutating it, mirroring lint-org.el's convention. + + These are Craig's org files. lint-org.el, the other tool that rewrites them, + leaves a /tmp copy before touching anything; this matches it so a bad edit is + always recoverable without reaching for git (which only reaches the last + commit, losing intra-session work). + """ + stamp = datetime.now().strftime("%Y%m%d-%H%M%S") + base = Path(tempfile.gettempdir()) / f"{path.name}.before-cj-remove.{stamp}" + # Never overwrite an earlier backup. The skill removes several annotations + # in quick succession, so a second-resolution stamp collides and the later + # copy would replace the earlier one with already-mutated content — losing + # the pre-session original the backup exists to preserve. + dest = base + n = 2 + while dest.exists(): + dest = base.with_name(f"{base.name}-{n}") + n += 1 + shutil.copy2(path, dest) + return dest + + +def _atomic_write(path: Path, text: str) -> None: + """Write text to path via a temp sibling and os.replace. + + A bare write_text truncates the target on open, so a mid-write failure left + the org file truncated with no complete copy on disk. Writing a temp sibling + and renaming means the file is either its old content or its new content, + never a partial. + """ + # Follow a symlink to the file it names. os.replace would otherwise swap the + # symlink itself for a regular file, leaving the real target holding the old + # content — the edit silently goes nowhere. Resolving also puts the temp + # sibling on the same filesystem as the real file, which os.replace needs. + path = path.resolve() + fd, tmp = tempfile.mkstemp(dir=path.parent, prefix=f".{path.name}.", suffix=".tmp") + os.close(fd) + tmp_path = Path(tmp) + # Carry the original's permissions across. mkstemp creates 0600, and + # defaulting to the umask instead widened a deliberately-restricted file + # (a 0600 org file came back 0644). + shutil.copymode(path, tmp_path) + try: + tmp_path.write_text(text, encoding="utf-8") + os.replace(tmp_path, path) + except BaseException: + tmp_path.unlink(missing_ok=True) + raise + + def remove_range(path: Path, start: int, end: int) -> None: """Read path, validate range looks like cj content, remove the range, write back.""" - text = path.read_text() + text = path.read_text(encoding="utf-8") had_trailing_newline = text.endswith("\n") lines = text.splitlines(keepends=False) @@ -77,7 +152,9 @@ def remove_range(path: Path, start: int, end: int) -> None: new_text += "\n" elif not new_lines and had_trailing_newline: new_text = "" - path.write_text(new_text) + + _backup(path) + _atomic_write(path, new_text) def main() -> int: diff --git a/claude-templates/.ai/scripts/flashcard-stats.py b/claude-templates/.ai/scripts/flashcard-stats.py index 1fa5afb..cb580ac 100755 --- a/claude-templates/.ai/scripts/flashcard-stats.py +++ b/claude-templates/.ai/scripts/flashcard-stats.py @@ -35,7 +35,12 @@ import re import sys from pathlib import Path -CARD_RE = re.compile(r"^\*\*\s+(.+?)\s+:drill:\s*$") +# A card is a level-2 heading whose trailing org tag block includes `drill`. +# Group 1 is the front, group 2 the tag block — so a curated card multi-tagged +# :fundamental:drill: still counts (it would silently drop under a :drill:$ +# anchor, undercounting the deck). HEADING_RE bounds a card's body. +CARD_RE = re.compile(r"^\*\*\s+(.+?)\s+(:[A-Za-z0-9_@#%:]+:)\s*$") +HEADING_RE = re.compile(r"^\*{1,2}\s") ANSWER_RE = re.compile(r"^\*\*\*\s+Answer\b") PROP_START_RE = re.compile(r"^\s*:PROPERTIES:\s*$") PROP_END_RE = re.compile(r"^\s*:END:\s*$") @@ -177,7 +182,8 @@ def parse_cards(lines: list[str]) -> tuple[list[dict], int]: n = len(lines) while i < n: m = CARD_RE.match(lines[i]) - if not m: + tags = [t for t in m.group(2).split(":") if t] if m else [] + if not (m and "drill" in tags): i += 1 continue heading = m.group(1).strip() @@ -188,7 +194,7 @@ def parse_cards(lines: list[str]) -> tuple[list[dict], int]: body_lines: list[str] = [] while i < n: line = lines[i] - if line.startswith("* ") or CARD_RE.match(line): + if HEADING_RE.match(line): break if PROP_START_RE.match(line): prop_count += 1 diff --git a/claude-templates/.ai/scripts/flashcard-to-anki.py b/claude-templates/.ai/scripts/flashcard-to-anki.py index ca4c70b..e369fd8 100755 --- a/claude-templates/.ai/scripts/flashcard-to-anki.py +++ b/claude-templates/.ai/scripts/flashcard-to-anki.py @@ -10,8 +10,15 @@ Parses org-drill structure: - Top-level "* Section" headings become tags on every card under them. - Each "** Card name :drill:" entry becomes a card. Front = heading - text (sans :drill: tag). Back = entry body with newlines converted + text (sans the tag block). Back = entry body with newlines converted to <br>. + - A card may carry a second org tag ("** Card :fundamental:drill:"). + Any heading whose tag block includes `drill` is a card; the other + tags ride along as Anki tags next to the section tag, so a curated + subset stays grep-able in the source. --tag-filter <tag> emits only + the cards carrying that tag, and a subset deck built that way should + pass --guid-salt so its notes get their own GUID space (Anki dedupes + on GUID, so without it the subset imports empty against the full deck). Deck name defaults to the org #+TITLE: (so the phone deck reads as the curated title), falling back to the input basename when the source has @@ -27,6 +34,8 @@ Usage: flashcard-to-anki.py <input.org> flashcard-to-anki.py <input.org> --deck "My Deck Name" flashcard-to-anki.py <input.org> --output /path/to/deck.apkg + flashcard-to-anki.py <input.org> --tag-filter fundamental \ + --deck "DeepSat Fundamentals" --guid-salt fundamentals Requires genanki, which uv resolves automatically via the PEP 723 script metadata above. No venv or system install needed. @@ -47,6 +56,15 @@ import genanki ID_BASE = 1_500_000_000 ID_RANGE = 500_000_000 +# A card is any level-2 heading whose trailing org tag block includes `drill`. +# Group 1 is the front text, group 2 the colon-delimited tag block (e.g. +# ":fundamental:drill:") — so a curated subset can carry a second org tag +# (:fundamental:) and stay grep-able in the source without dropping from the +# full deck. HEADING_RE bounds a card's body at the next L1/L2 heading. +CARD_RE = re.compile(r"^\*\*\s+(.+?)\s+(:[A-Za-z0-9_@#%:]+:)\s*$") +HEADING_RE = re.compile(r"^\*{1,2}\s") +SECTION_RE = re.compile(r"^\*\s+(.+?)\s*$") + def stable_id(name: str, salt: str) -> int: """Derive a deterministic 32-bit id from `name` and a `salt`. @@ -120,33 +138,40 @@ def strip_org_metadata(body_lines: list[str]) -> list[str]: return cleaned -def parse(org_text: str) -> list[tuple[str, str, str]]: - """Return [(front, back_html, tag), ...] for every :drill: card.""" - cards: list[tuple[str, str, str]] = [] - current_section: str | None = None +def parse( + org_text: str, tag_filter: str | None = None +) -> list[tuple[str, str, list[str]]]: + """Return [(front, back_html, anki_tags), ...] for every :drill: card. - section_re = re.compile(r"^\*\s+(.+?)\s*$") - card_re = re.compile(r"^\*\*\s+(.+?)\s+:drill:\s*$") + A card is any level-2 heading whose trailing org tag block includes + `drill`. Non-drill org tags on the heading (e.g. :fundamental:) ride + along as Anki tags next to the section tag, so a curated subset stays + grep-able in the source. When `tag_filter` is set, only cards carrying + that org tag are returned (the subset-deck path). + """ + cards: list[tuple[str, str, list[str]]] = [] + current_section: str | None = None lines = org_text.splitlines() i = 0 while i < len(lines): line = lines[i] - sec = section_re.match(line) + sec = SECTION_RE.match(line) if sec: current_section = sec.group(1).strip() i += 1 continue - card = card_re.match(line) - if card: - front = card.group(1).strip() + m = CARD_RE.match(line) + tags = [t for t in m.group(2).split(":") if t] if m else [] + if m and "drill" in tags: + front = m.group(1).strip() body_lines: list[str] = [] i += 1 while i < len(lines): nxt = lines[i] - if nxt.startswith("* ") or card_re.match(nxt): + if HEADING_RE.match(nxt): break body_lines.append(nxt) i += 1 @@ -156,8 +181,17 @@ def parse(org_text: str) -> list[tuple[str, str, str]]: while body_lines and not body_lines[-1].strip(): body_lines.pop() back_html = "<br>".join(escape_html(ln) for ln in body_lines) - tag = section_to_tag(current_section) if current_section else "drill" - cards.append((front, back_html, tag)) + + org_tags = [t for t in tags if t != "drill"] + if tag_filter and tag_filter not in org_tags: + continue + anki_tags: list[str] = [] + if current_section: + anki_tags.append(section_to_tag(current_section)) + anki_tags.extend(org_tags) + if not anki_tags: + anki_tags = ["drill"] + cards.append((front, back_html, anki_tags)) continue i += 1 @@ -165,15 +199,28 @@ def parse(org_text: str) -> list[tuple[str, str, str]]: return cards -def build(cards: list[tuple[str, str, str]], deck_name: str) -> genanki.Deck: +def card_guid(front: str, guid_salt: str | None) -> str: + """GUID for a card's front. A salt gives a derived subset deck its own + GUID space so its notes don't collide with the full deck's (Anki dedupes + on GUID, which would otherwise import the subset empty). No salt is the + original behavior, so an unsalted deck's GUIDs and SRS state are untouched. + """ + return genanki.guid_for(guid_salt, front) if guid_salt else genanki.guid_for(front) + + +def build( + cards: list[tuple[str, str, list[str]]], + deck_name: str, + guid_salt: str | None = None, +) -> genanki.Deck: deck = genanki.Deck(stable_id(deck_name, "deck"), deck_name) model = make_model(deck_name) - for front, back, tag in cards: + for front, back, tags in cards: note = genanki.Note( model=model, fields=[front, back], - tags=[tag], - guid=genanki.guid_for(front), + tags=tags, + guid=card_guid(front, guid_salt), ) deck.add_note(note) return deck @@ -219,6 +266,16 @@ def main() -> int: help="Output .apkg path. Defaults to " "~/sync/phone/anki/<input-basename>.apkg.", ) + parser.add_argument( + "--tag-filter", + help="Emit only cards carrying this org tag (e.g. --tag-filter " + "fundamental for a curated subset deck).", + ) + parser.add_argument( + "--guid-salt", + help="Salt note GUIDs so a subset deck gets its own GUID space and " + "imports non-empty without disturbing the full deck's SRS state.", + ) args = parser.parse_args() input_path: Path = args.input.expanduser().resolve() @@ -231,12 +288,18 @@ def main() -> int: output_path: Path = (args.output or default_output_path(input_path)).expanduser().resolve() output_path.parent.mkdir(parents=True, exist_ok=True) - cards = parse(org_text) + cards = parse(org_text, tag_filter=args.tag_filter) if not cards: - print(f"error: no :drill: cards found in {input_path}", file=sys.stderr) + if args.tag_filter: + print( + f"error: no :drill: cards tagged :{args.tag_filter}: in {input_path}", + file=sys.stderr, + ) + else: + print(f"error: no :drill: cards found in {input_path}", file=sys.stderr) return 1 - deck = build(cards, deck_name) + deck = build(cards, deck_name, guid_salt=args.guid_salt) genanki.Package(deck).write_to_file(str(output_path)) print(f"wrote {output_path} ({len(cards)} cards, deck '{deck_name}')") return 0 diff --git a/claude-templates/.ai/scripts/inbox-send.py b/claude-templates/.ai/scripts/inbox-send.py index 1ebb636..663efcb 100755 --- a/claude-templates/.ai/scripts/inbox-send.py +++ b/claude-templates/.ai/scripts/inbox-send.py @@ -31,6 +31,7 @@ import os import re import shutil import sys +import tempfile from datetime import datetime from pathlib import Path @@ -48,7 +49,7 @@ def resolve_roots() -> list[Path]: config = Path.home() / ".claude" / "inbox-roots.txt" if config.is_file(): paths: list[Path] = [] - for line in config.read_text().splitlines(): + for line in config.read_text(encoding="utf-8").splitlines(): line = line.strip() if line and not line.startswith("#"): paths.append(Path(line).expanduser()) @@ -69,17 +70,28 @@ def discover_projects(roots: list[Path]) -> list[Path]: a specific project root (included directly if it qualifies). """ projects: list[Path] = [] + seen: set[Path] = set() + + def _add(p: Path) -> None: + # Dedupe on the resolved path: a roots config naming both a parent and + # one of its children would otherwise list the child project twice, at + # two different indices. + key = p.resolve() + if key not in seen: + seen.add(key) + projects.append(p) + for root in roots: if not root.is_dir(): continue if _is_project(root): - projects.append(root) + _add(root) continue for child in sorted(root.iterdir()): if not child.is_dir(): continue if _is_project(child): - projects.append(child) + _add(child) return projects @@ -194,6 +206,39 @@ def uniquify(dest: Path) -> Path: n += 1 +def _atomic_write(dest: Path, writer) -> None: + """Write to a temp file in dest's directory, then rename it into place. + + dest is another project's inbox/, and a direct write truncates the target + on open, so any mid-write failure (a full disk, an encoding error, an + interrupted process) leaves a zero-byte .org there. inbox-status counts + that phantom as a pending handoff and blocks a turn in the receiving + project over a file with no content and no sender (2026-07-23). Writing to + a temp sibling and os.replace-ing means the inbox only ever sees a complete + file. os.replace is atomic within one filesystem, and the temp sits in the + same directory as dest, so it is. + + `writer` receives the open temp path and fills it. On any failure the temp + is removed and the error re-raised, so a caught error never leaves debris. + """ + fd, tmp = tempfile.mkstemp( + dir=dest.parent, prefix=".inbox-send-", suffix=dest.suffix + ) + os.close(fd) + tmp_path = Path(tmp) + # mkstemp creates the temp 0600; give the delivered file the umask-default + # mode the old direct write produced, so inbox files stay readable as before. + umask = os.umask(0) + os.umask(umask) + os.chmod(tmp_path, 0o666 & ~umask) + try: + writer(tmp_path) + os.replace(tmp_path, dest) + except BaseException: + tmp_path.unlink(missing_ok=True) + raise + + def send_text( target_inbox: Path, message: str, @@ -209,7 +254,8 @@ def send_text( raise ValueError(f"could not derive a slug from text: {message!r}") filename = f"{now.strftime(TS_FILENAME_FMT)}-from-{source_name}-{slug}.org" dest = uniquify(target_inbox / filename) - dest.write_text(build_text_org(message, source_name, now.strftime(TS_DOC_FMT))) + body = build_text_org(message, source_name, now.strftime(TS_DOC_FMT)) + _atomic_write(dest, lambda p: p.write_text(body, encoding="utf-8")) return dest @@ -229,7 +275,7 @@ def send_file( ext = src_path.suffix filename = f"{now.strftime(TS_FILENAME_FMT)}-from-{source_name}-{slug}{ext}" dest = uniquify(target_inbox / filename) - shutil.copy2(src_path, dest) + _atomic_write(dest, lambda p: shutil.copyfile(src_path, p)) return dest @@ -310,7 +356,10 @@ def main() -> int: else: assert args.file is not None dest = send_file(target_inbox, args.file, source_name, args.name, now) - except (ValueError, FileNotFoundError) as exc: + except (ValueError, OSError) as exc: + # OSError covers FileNotFoundError (missing source), PermissionError + # (unreadable source), and any atomic-write failure — all should + # surface as the clean "inbox-send: <message>" error, never a traceback. print(f"inbox-send: {exc}", file=sys.stderr) return 1 diff --git a/claude-templates/.ai/scripts/inbox-status b/claude-templates/.ai/scripts/inbox-status index b917144..17031af 100755 --- a/claude-templates/.ai/scripts/inbox-status +++ b/claude-templates/.ai/scripts/inbox-status @@ -35,6 +35,7 @@ mapfile -t pending < <(find inbox -maxdepth 1 -type f \ ! -name '.gitkeep' \ ! -name 'lint-followups.org' \ ! -name 'PROCESSED-*' \ + ! -name '.inbox-send-*' \ -printf '%f\n' 2>/dev/null | sort) n=${#pending[@]} diff --git a/claude-templates/.ai/scripts/lint-org.el b/claude-templates/.ai/scripts/lint-org.el index 55727ef..33dc52f 100644 --- a/claude-templates/.ai/scripts/lint-org.el +++ b/claude-templates/.ai/scripts/lint-org.el @@ -38,7 +38,9 @@ ;; empty-heading bare stars with no title ;; malformed-priority-cookie [#x]-shaped token org rejected ;; level2-done-without-closed completed level-2 task with no CLOSED +;; task-missing-last-reviewed open level-2 task with no :LAST_REVIEWED: ;; subtask-done-not-dated level-3+ done sub-task still a DONE keyword +;; dated-log-heading-active-timestamp dated-log heading with a live SCHEDULED/DEADLINE ;; (anything else) surfaced as judgment with checker name ;; ;; Output format on stdout: @@ -74,6 +76,18 @@ The CLI defaults this to t (a linter reports, it doesn't write); `--fix' is what enables writes on a command-line run.") (defvar lo-current-file nil "Path of the file currently being processed.") + +(defun lo--spec-file-p () + "Non-nil when the current file lives under a docs/specs/ directory. +The four todo-format-family checkers encode todo.org completion conventions +and misfire on a spec: a spec's Decisions section legitimately carries a +level-2 DONE with no CLOSED cookie, and its review-history section carries +level-2 dated headings. docs/specs/ is the canonical spec home per the +docs-lifecycle rule, so a path segment match is the scope test. Link, +table, and structural checks still run on specs — only the todo-format +family is scoped out." + (and lo-current-file + (string-match-p "/docs/specs/" (expand-file-name lo-current-file)))) (defvar lo-followups-file nil "When non-nil, after a non-check run any judgment items are appended to this path as an org section dated today. The file is created if missing.") @@ -292,6 +306,52 @@ Craig-specific annotation marker rather than Babel src-block syntax." (lo--goto-line line) (looking-at-p "^[ \t]*#\\+begin_src[ \t]+cj:"))) +(defvar-local lo--matched-blocks-cache nil + "Cons of (TICK . REGIONS) memoizing `lo--matched-block-regions'. +TICK is the `buffer-chars-modified-tick' the regions were computed at, so a +fix applied mid-pass invalidates them.") + +(defun lo--matched-block-regions () + "Return ((BEGIN-LINE . END-LINE) ...) for every correctly paired block. +Scans lines directly rather than asking org, because org's own parser is what +mis-reads these blocks: a heading-shaped line inside a verbatim body reads as a +structural break and loses the open block. The scan applies org's real rule — +once a block is open, only its own `#+end_TYPE' closes it, so a nested +`#+begin_' or a foreign `#+end_' in the body is just text." + (let ((tick (buffer-chars-modified-tick))) + (if (eql (car lo--matched-blocks-cache) tick) + (cdr lo--matched-blocks-cache) + (let ((case-fold-search t) + (regions nil) (open-type nil) (open-line nil) (line 0)) + (save-excursion + (goto-char (point-min)) + (while (not (eobp)) + (setq line (1+ line)) + (let ((text (buffer-substring-no-properties + (line-beginning-position) (line-end-position)))) + (cond + (open-type + (when (string-match + (format "\\`[ \t]*#\\+end_%s[ \t]*\\'" + (regexp-quote open-type)) + text) + (push (cons open-line line) regions) + (setq open-type nil open-line nil))) + ((string-match "\\`[ \t]*#\\+begin_\\([^ \t\n]+\\)" text) + (setq open-type (match-string 1 text) + open-line line)))) + (forward-line 1))) + (setq lo--matched-blocks-cache (cons tick (nreverse regions))) + (cdr lo--matched-blocks-cache))))) + +(defun lo--in-matched-block-p (line) + "Non-nil when LINE sits within a correctly paired block, delimiters included. +org-lint reports `invalid-block' at the delimiter lines themselves, so the +range has to be inclusive for the suppression to reach them." + (cl-some (lambda (region) + (and (>= line (car region)) (<= line (cdr region)))) + (lo--matched-block-regions))) + (defun lo--handle-item (item) (let ((name (lo--checker-name item)) (line (lo--line item)) @@ -304,6 +364,13 @@ Craig-specific annotation marker rather than Babel src-block syntax." wrong-header-argument)) (lo--cj-comment-block-opener-p line)) nil) + ;; `invalid-block' on a block that is in fact correctly paired — the + ;; checker is org-lint's own, so this filters its output rather than + ;; fixing a local checker. A genuinely unterminated block isn't in any + ;; matched region, so it still reports. + ((and (eq name 'invalid-block) + (lo--in-matched-block-p line)) + nil) ((eq name 'item-number) (lo--apply-or-preview name line msg #'lo-fix-item-number)) ((eq name 'missing-language-in-src-block) @@ -525,6 +592,42 @@ the live file on the next `task-sorted'." "level-2 DONE/CANCELLED has no CLOSED date — add CLOSED: [YYYY-MM-DD Day]; task-sorted's aging step archives an undated completed task immediately")))))))) ;;; --------------------------------------------------------------------------- +;;; task-missing-last-reviewed check (claude-rules/todo-format.md) +;; +;; A task is stamped `:LAST_REVIEWED:' when it is *created*, not a review cycle +;; later. An agent filing a task has just written its body and graded its +;; priority, which is a review by any honest reading — so a fresh task that +;; carries no stamp reads as "never reviewed" and lands at the top of the next +;; staleness batch, where re-reviewing it is pure ceremony. Every task filed +;; during the 2026-07-23 sweep hit exactly that, which is what prompted the rule. +;; +;; Judgment-only, deliberately. The stamp's whole value is that its date is +;; true, and nothing here can know when an unstamped task was actually last +;; looked at. Auto-stamping today's date would convert a "nobody has reviewed +;; this" signal into a false "reviewed today" one — worse than the gap it +;; closes. Flag it; a human or the filing workflow supplies the honest date. +;; +;; Scope matches `task-review-staleness.sh' exactly (level-2, open keyword, +;; priority cookie), so the checker and the staleness count never disagree +;; about which headings are in the review pool. + +(defun lo--check-task-missing-last-reviewed () + "Flag an open level-2 task with a priority cookie and no `:LAST_REVIEWED:'." + (save-excursion + (goto-char (point-min)) + (let ((case-fold-search nil)) + (while (re-search-forward "^\\*\\* \\(TODO\\|DOING\\|VERIFY\\) \\[#[A-D]\\]" nil t) + (let ((hline (line-number-at-pos)) + (entry-end (save-excursion (outline-next-heading) (point)))) + (save-excursion + (forward-line 1) + (unless (re-search-forward "^[ \t]*:LAST_REVIEWED:[ \t]*[[0-9]" + entry-end t) + (lo--emit-judgment + 'task-missing-last-reviewed hline + "task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed")))))))) + +;;; --------------------------------------------------------------------------- ;;; level-3+ dated-header check (claude-rules/todo-format.md) ;; ;; The inverse of the level-2 check above. A completed sub-task — a heading at @@ -551,6 +654,41 @@ Emits one judgment item per offending heading (checker "level-3+ done sub-task should be a dated event-log entry (todo-format.md): run todo-cleanup.el --convert-subtasks to rewrite it"))))) ;;; --------------------------------------------------------------------------- +;;; dated-log heading with a stale active planning timestamp (todo-format.md) +;; +;; The mechanical backstop for the planning-line-strip rule. A dated event-log +;; heading (`<stars> YYYY-MM-DD Day @ ...', no TODO keyword) records completed +;; work — its date lives in the heading. An active `<...>' SCHEDULED or DEADLINE +;; left on it pins the entry to the agenda forever: org renders any headline with +;; an active planning timestamp, keyword or not, so a stale SCHEDULED shows as +;; weeks-overdue long after the work is done. Invisible to a keyword scan (no +;; TODO) and it survives --archive-done, so nothing else catches it. The +;; completion rewrite and todo-cleanup --convert-subtasks now strip the planning +;; line; this flags any that slipped through before that landed, the same way +;; subtask-done-not-dated backstops the depth rule. Judgment-only. + +(defun lo--check-dated-log-active-timestamp () + "Flag a dated event-log heading that still carries an active SCHEDULED/DEADLINE. +The heading matches `<stars> YYYY-MM-DD Day @ ...' with no TODO keyword; an +active `<...>' planning timestamp in its entry is the defect. An inactive +`[...]' timestamp is ignored (org doesn't render it on the agenda). Emits one +judgment item per offending heading (checker `dated-log-heading-active-timestamp')." + (save-excursion + (goto-char (point-min)) + (let ((case-fold-search nil)) + (while (re-search-forward + "^\\*+ [0-9]\\{4\\}-[0-9]\\{2\\}-[0-9]\\{2\\} [A-Za-z]+ @ " nil t) + (let ((hline (line-number-at-pos)) + (entry-end (save-excursion (outline-next-heading) (point)))) + (save-excursion + (forward-line 1) + (when (re-search-forward + "^[ \t]*\\(?:SCHEDULED\\|DEADLINE\\):[ \t]*<" entry-end t) + (lo--emit-judgment + 'dated-log-heading-active-timestamp hline + "dated-log heading carries an active SCHEDULED/DEADLINE — org renders any active planning timestamp (keyword or not), so it stays on the agenda as weeks-overdue; delete the planning line (todo-format.md)")))))))) + +;;; --------------------------------------------------------------------------- ;;; File processing (defun lo--backup (file) @@ -584,14 +722,22 @@ left unmodified and mechanical entries are recorded with :preview t." ;; After org-lint items: the custom table-standard scan. Runs on the ;; post-fix buffer; judgment-only, so order doesn't perturb fixes. (lo--check-tables) - ;; Same shape: flag level-2 dated headers (completion defects). - (lo--check-level2-dated-headers) - ;; Structural heading defects org-lint doesn't cover. + ;; Structural heading defects org-lint doesn't cover. These run on + ;; every org file, specs included. (lo--check-indented-headings) (lo--check-empty-headings) (lo--check-malformed-priority-cookies) - (lo--check-level2-done-without-closed) - (lo--check-subtask-done-not-dated) + ;; The todo-format family encodes todo.org completion conventions and + ;; misfires on a spec (a Decisions section's undated DONE, a + ;; review-history dated heading, a phases task with no LAST_REVIEWED). + ;; Scope them out of docs/specs/; link, table, and structural checks + ;; above still run there. + (unless (lo--spec-file-p) + (lo--check-level2-dated-headers) + (lo--check-level2-done-without-closed) + (lo--check-task-missing-last-reviewed) + (lo--check-subtask-done-not-dated) + (lo--check-dated-log-active-timestamp)) (when (and (not lo-check-only) (buffer-modified-p)) (save-buffer))) (with-current-buffer buf (set-buffer-modified-p nil)) diff --git a/claude-templates/.ai/scripts/route_recommend.py b/claude-templates/.ai/scripts/route_recommend.py index 7b36405..12ab132 100644 --- a/claude-templates/.ai/scripts/route_recommend.py +++ b/claude-templates/.ai/scripts/route_recommend.py @@ -71,6 +71,15 @@ def recommend(item: str, projects: list[str]) -> tuple[str | None, str]: if not projects: return (None, "none") + # Collapse identical names first. Projects are addressed by bare basename, so + # two projects sharing one across roots (~/code/notes, ~/projects/notes) arrive + # twice; both literal-match, and the tie test below then read that as ambiguity + # and downgraded a correct strong match to weak. Deduping here rather than in + # discover_destination_names protects every caller of the pure core, not just + # the CLI path. Order-preserving, and it collapses only identical names — two + # *different* projects matching is real ambiguity and still downgrades. + projects = list(dict.fromkeys(projects)) + item_lower = item.lower() item_tokens = _tokens(item) diff --git a/claude-templates/.ai/scripts/tests/agent-lock.bats b/claude-templates/.ai/scripts/tests/agent-lock.bats new file mode 100644 index 0000000..dbcffe1 --- /dev/null +++ b/claude-templates/.ai/scripts/tests/agent-lock.bats @@ -0,0 +1,214 @@ +#!/usr/bin/env bats +# +# Tests for claude-templates/.ai/scripts/agent-lock — a mkdir-atomic advisory +# lock helper for agent workflows (sentry's single-runner and roam-write +# locks). flock can't span an agent's tool calls: every Bash call is its own +# short-lived shell, so a flock dies with the call that took it. This helper +# persists the lock on disk between calls and self-clears after a crash via +# age-based staleness reclaim. +# +# Contract under test: +# agent-lock acquire <name> [--ttl=SECONDS] [--wait[=SECONDS]] +# exit 0 → acquired (fresh, or reclaimed from a stale prior holder). +# exit 1 → busy: a live lock holds <name>; deferred (note on stderr). +# exit 2 → usage error (bad/absent name, unknown subcommand). +# agent-lock refresh <name> → re-touch a held lock (heartbeat); exit 1 if absent. +# agent-lock release <name> → remove the lock; idempotent (exit 0 if already free). +# agent-lock status <name> → print free|held|stale + metadata; exit 0 (query). +# agent-lock path <name> → print the resolved lock dir path; does not create it. +# +# Staleness is age-based on the metadata file's mtime versus the lock's own +# recorded TTL, so a crashed holder's lock expires instead of wedging every +# later acquire. Heartbeat (refresh) re-touches the mtime, keeping a live +# holder's lock young. Every reclaim surfaces a note (never silent). +# +# Lock home: /run/user/<uid>/agent-locks/<name>/ (tmpfs: host-local, out of +# every repo, cleared on reboot), with ~/.cache/agent-locks/ as the fallback +# where no runtime dir exists. AGENT_LOCK_DIR overrides the base for tests and +# advanced callers; the helper otherwise owns the path scheme and callers pass +# only names. +# +# Strategy: AGENT_LOCK_DIR points every lock at a temp base, so tests never +# touch a real runtime dir. Staleness is exercised by aging the metadata +# file's mtime with `touch` rather than sleeping. + +SCRIPT="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/agent-lock" +BASH_BIN="$(command -v bash)" + +setup() { + TEST_DIR="$(mktemp -d -t agent-lock-bats.XXXXXX)" + LOCK_BASE="$TEST_DIR/locks" +} + +teardown() { + rm -rf "$TEST_DIR" +} + +lock() { + run env AGENT_LOCK_DIR="$LOCK_BASE" "$BASH_BIN" "$SCRIPT" "$@" +} + +# meta-file path for a lock name, for direct inspection / aging. +meta_of() { + printf '%s/%s/meta\n' "$LOCK_BASE" "$1" +} + +# ---- acquire: fresh win + metadata -------------------------------------- + +@test "acquire: fresh name wins (exit 0) and writes pid/host/timestamp/ttl" { + lock acquire job + [ "$status" -eq 0 ] + local meta; meta="$(meta_of job)" + [ -f "$meta" ] + grep -q "^pid=$$\|^pid=[0-9][0-9]*$" "$meta" + grep -q "^host=$(uname -n)$" "$meta" + grep -qE "^acquired=[0-9]{4}-[0-9]{2}-[0-9]{2}T" "$meta" + grep -qE "^ttl=[0-9]+$" "$meta" +} + +@test "acquire: honors an explicit --ttl in the metadata" { + lock acquire job --ttl=45 + [ "$status" -eq 0 ] + grep -q "^ttl=45$" "$(meta_of job)" +} + +# ---- acquire: contention (one winner) ----------------------------------- + +@test "acquire: a second acquire of a live lock defers (exit 1, note)" { + lock acquire job + [ "$status" -eq 0 ] + lock acquire job + [ "$status" -eq 1 ] + [[ "$output" == *job* ]] +} + +@test "acquire: two racing acquires yield exactly one winner" { + # Fire both without releasing; exactly one mkdir wins. + env AGENT_LOCK_DIR="$LOCK_BASE" "$BASH_BIN" "$SCRIPT" acquire race & p1=$! + env AGENT_LOCK_DIR="$LOCK_BASE" "$BASH_BIN" "$SCRIPT" acquire race & p2=$! + local r1=0 r2=0 + wait $p1 || r1=$? + wait $p2 || r2=$? + # One exits 0 (won), one exits 1 (deferred). + [ "$((r1 + r2))" -eq 1 ] +} + +# ---- release: frees the lock -------------------------------------------- + +@test "release: frees a held lock so the next acquire wins" { + lock acquire job + [ "$status" -eq 0 ] + lock release job + [ "$status" -eq 0 ] + [ ! -d "$LOCK_BASE/job" ] + lock acquire job + [ "$status" -eq 0 ] +} + +@test "release: is idempotent on an already-free lock (exit 0)" { + lock release never-held + [ "$status" -eq 0 ] +} + +# ---- staleness reclaim (surfaced, never silent) ------------------------- + +@test "acquire: reclaims a stale lock and surfaces the reclaim note" { + lock acquire job --ttl=1 + [ "$status" -eq 0 ] + # Age the metadata mtime well past the 1s TTL. + touch -d '1 hour ago' "$(meta_of job)" + lock acquire job --ttl=1 + [ "$status" -eq 0 ] + [[ "$output" == *reclaim* ]] + [[ "$output" == *job* ]] + # The reclaim installed fresh metadata (young again), not the aged holder's. + lock status job + [[ "$output" == *held* ]] + [[ "$output" != *stale* ]] +} + +@test "acquire: a lock inside its TTL is not stale (stays deferred)" { + lock acquire job --ttl=3600 + [ "$status" -eq 0 ] + lock acquire job --ttl=3600 + [ "$status" -eq 1 ] +} + +# ---- heartbeat (refresh keeps a live lock young) ------------------------ + +@test "refresh: re-touches a held lock so it is no longer stale" { + lock acquire job --ttl=1 + [ "$status" -eq 0 ] + touch -d '1 hour ago' "$(meta_of job)" + lock status job + [[ "$output" == *stale* ]] + lock refresh job + [ "$status" -eq 0 ] + lock status job + [[ "$output" == *held* ]] + [[ "$output" != *stale* ]] +} + +@test "refresh: an absent lock cannot be refreshed (exit 1)" { + lock refresh nothing + [ "$status" -eq 1 ] +} + +# ---- status query ------------------------------------------------------- + +@test "status: reports free for an unheld lock (exit 0)" { + lock status job + [ "$status" -eq 0 ] + [[ "$output" == *free* ]] +} + +@test "status: reports held with metadata for a live lock" { + lock acquire job --ttl=3600 + lock status job + [ "$status" -eq 0 ] + [[ "$output" == *held* ]] + [[ "$output" == *"host=$(uname -n)"* ]] +} + +# ---- path resolution: runtime dir home with cache fallback -------------- + +@test "path: resolves under AGENT_LOCK_DIR when set" { + lock path job + [ "$status" -eq 0 ] + [ "$output" = "$LOCK_BASE/job" ] + [ ! -d "$LOCK_BASE/job" ] # path does not create the lock +} + +@test "path: prefers the runtime dir home when no override is set" { + local rt="$TEST_DIR/run" + mkdir -p "$rt" + run env -u AGENT_LOCK_DIR XDG_RUNTIME_DIR="$rt" "$BASH_BIN" "$SCRIPT" path job + [ "$status" -eq 0 ] + [ "$output" = "$rt/agent-locks/job" ] +} + +@test "path: falls back to the cache home when no runtime dir exists" { + local home="$TEST_DIR/home" + mkdir -p "$home" + run env -u AGENT_LOCK_DIR -u XDG_RUNTIME_DIR -u XDG_CACHE_HOME \ + HOME="$home" "$BASH_BIN" "$SCRIPT" path job + [ "$status" -eq 0 ] + [ "$output" = "$home/.cache/agent-locks/job" ] +} + +# ---- usage errors ------------------------------------------------------- + +@test "usage: a missing name is a usage error (exit 2)" { + lock acquire + [ "$status" -eq 2 ] +} + +@test "usage: a name with a slash is rejected (exit 2)" { + lock acquire bad/name + [ "$status" -eq 2 ] +} + +@test "usage: an unknown subcommand is a usage error (exit 2)" { + lock frobnicate job + [ "$status" -eq 2 ] +} diff --git a/claude-templates/.ai/scripts/tests/flashcard-sync.bats b/claude-templates/.ai/scripts/tests/flashcard-sync.bats index 608a280..e6ffc21 100644 --- a/claude-templates/.ai/scripts/tests/flashcard-sync.bats +++ b/claude-templates/.ai/scripts/tests/flashcard-sync.bats @@ -6,6 +6,7 @@ setup() { SCRIPT_DIR="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)" SYNC="$SCRIPT_DIR/flashcard-sync" + STATS="$SCRIPT_DIR/flashcard-stats.py" TMP="$(mktemp -d)" } @@ -36,3 +37,27 @@ EOF [ "$status" -eq 1 ] [ ! -f "$HOME/sync/phone/anki/dirty.apkg" ] } + +@test "flashcard-stats: a multi-tagged :fundamental:drill: card still counts" { + # Regression guard: a curated card carrying a second org tag must not drop + # from the count. A :drill:$ anchor would have counted only one card here. + cat > "$TMP/multitag.org" <<'EOF' +#+TITLE: Multitag Test + +* Orbital Regimes +** What is LEO? :fundamental:drill: +:PROPERTIES: +:ID: c1 +:END: +Low Earth Orbit is the region below about 2000 kilometers. +** What is GEO? :drill: +:PROPERTIES: +:ID: c2 +:END: +Geostationary orbit sits at roughly 35786 kilometers of altitude. +EOF + run python3 "$STATS" "$TMP/multitag.org" + [ "$status" -eq 0 ] + [[ "$output" == *"Cards: 2"* ]] + [[ "$output" == *clean* ]] +} diff --git a/claude-templates/.ai/scripts/tests/inbox-status.bats b/claude-templates/.ai/scripts/tests/inbox-status.bats index bc8a734..27a497e 100644 --- a/claude-templates/.ai/scripts/tests/inbox-status.bats +++ b/claude-templates/.ai/scripts/tests/inbox-status.bats @@ -45,6 +45,18 @@ teardown() { [[ "$output" == *"0 pending"* ]] } +@test "inbox-status: ignores an in-flight .inbox-send-* temp file" { + mkdir "$TMP/inbox" + # inbox-send writes to a .inbox-send-* temp then renames it into place; + # during that window the temp must not read as a pending handoff, or a + # concurrent boundary check blocks on a file that's about to become real. + touch "$TMP/inbox/.inbox-send-abc123.org" + cd "$TMP" + run "$SCRIPT" + [ "$status" -eq 0 ] + [[ "$output" == *"0 pending"* ]] +} + @test "inbox-status: -q suppresses the per-item lines" { mkdir "$TMP/inbox" echo body > "$TMP/inbox/handoff.org" diff --git a/claude-templates/.ai/scripts/tests/test-lint-org.el b/claude-templates/.ai/scripts/tests/test-lint-org.el index 8e3e190..ceee209 100644 --- a/claude-templates/.ai/scripts/tests/test-lint-org.el +++ b/claude-templates/.ai/scripts/tests/test-lint-org.el @@ -193,6 +193,65 @@ real suspicious-language warning here #+end_src ") +;; invalid-block, false-positive case — a correctly paired example block whose +;; body holds a heading-shaped line. org's parser reads the `** ' inside the +;; verbatim body as a structural break, loses the open block, and flags BOTH +;; delimiters as "Possible incomplete block". +(defconst lo-test--verbatim-heading-block "\ +* Heading + +#+begin_example +** Feature Name or Topic +Body line. +#+end_example + +Trailing prose. +") + +;; invalid-block, literal-delimiter case — a paired src block whose body holds +;; a literal `#+end_example' plus a heading-shaped line. Only `#+end_src' +;; closes a src block, so all three findings here are false. +(defconst lo-test--literal-end-in-src "\ +* Heading + +#+begin_src text +#+end_example +** heading shaped +#+end_src +") + +;; invalid-block, uppercase-delimiter case — org accepts #+BEGIN_/#+END_ in +;; either case, and the pre-fix script flagged both delimiters here too. +(defconst lo-test--uppercase-verbatim-block "\ +* Heading + +#+BEGIN_EXAMPLE +** heading shaped +#+END_EXAMPLE +") + +;; invalid-block, genuine case — a block that really is never closed. The +;; suppression must not reach this one. +(defconst lo-test--unterminated-block "\ +* Heading + +#+begin_example +truly unterminated block body +") + +;; A genuinely unterminated block *after* a correctly paired one — verifies the +;; suppression is scoped per block rather than per file. +(defconst lo-test--paired-then-unterminated "\ +* Heading + +#+begin_example +** heading shaped +#+end_example + +#+begin_example +never closed +") + ;; Mixed fixture — each category once. (defconst lo-test--mixed "\ * Mixed @@ -392,6 +451,55 @@ suspicious-language judgment." (should (= 1 suspicious)))) ;;; --------------------------------------------------------------------------- +;;; invalid-block — false positives on correctly paired verbatim blocks + +(ert-deftest lo-verbatim-heading-block-emits-no-invalid-block () + "Normal: a paired example block containing a heading-shaped body line emits +no invalid-block judgment. Both delimiters are flagged by org-lint because the +parser treats the `** ' inside the verbatim body as a structural break." + (let* ((out (lo-test--run lo-test--verbatim-heading-block)) + (res (plist-get out :result)) + (judgments (lo-test--judgments (plist-get out :issues)))) + ;; File untouched, no fixes applied — suppression only, never a rewrite. + (should (equal lo-test--verbatim-heading-block res)) + (should (= 0 (plist-get out :fixes))) + (should-not (member 'invalid-block (lo-test--checkers judgments))))) + +(ert-deftest lo-literal-end-delimiter-in-src-emits-no-invalid-block () + "Boundary: a paired src block whose body holds a literal `#+end_example' and +a heading-shaped line emits no invalid-block judgment. Only `#+end_src' closes +a src block, so the interior delimiter is body text." + (let* ((out (lo-test--run lo-test--literal-end-in-src)) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should-not (member 'invalid-block (lo-test--checkers judgments))))) + +(ert-deftest lo-uppercase-verbatim-block-emits-no-invalid-block () + "Boundary: block delimiters are case-insensitive in org, so an uppercase +`#+BEGIN_EXAMPLE' pair is suppressed the same as a lowercase one." + (let* ((out (lo-test--run lo-test--uppercase-verbatim-block)) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should-not (member 'invalid-block (lo-test--checkers judgments))))) + +(ert-deftest lo-unterminated-block-still-emits-invalid-block () + "Error: a block that is never closed still emits its invalid-block judgment. +This is the finding the checker exists for — the suppression must not mask it." + (let* ((out (lo-test--run lo-test--unterminated-block)) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should (member 'invalid-block (lo-test--checkers judgments))))) + +(ert-deftest lo-invalid-block-suppression-is-scoped-per-block () + "Boundary: a paired block and an unterminated block in the same file — the +paired one is suppressed and the unterminated one still reports. Exactly one +invalid-block judgment, and it points at the unterminated opener (line 7)." + (let* ((out (lo-test--run lo-test--paired-then-unterminated)) + (judgments (lo-test--judgments (plist-get out :issues))) + (invalid (cl-remove-if-not + (lambda (i) (eq (plist-get i :checker) 'invalid-block)) + judgments))) + (should (= 1 (length invalid))) + (should (= 7 (plist-get (car invalid) :line))))) + +;;; --------------------------------------------------------------------------- ;;; --check mode (ert-deftest lo-check-mode-does-not-modify-file () @@ -739,6 +847,48 @@ missing-rules violation." (judgments (lo-test--judgments (plist-get out :issues)))) (should-not (member 'subtask-done-not-dated (lo-test--checkers judgments))))) +;;; dated-log-heading-active-timestamp check (stale SCHEDULED/DEADLINE on a +;;; completed dated-log entry — the home 2026-07-17 agenda-pollution bug) + +(ert-deftest lo-dated-log-active-scheduled-is-flagged () + "A dated-log entry still carrying an active SCHEDULED is flagged: org renders +it on the agenda forever despite the missing keyword." + (let* ((out (lo-test--run + "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 trip booked\nSCHEDULED: <2026-06-18 Thu>\nBody.\n")) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should (= 0 (plist-get out :fixes))) ; judgment-only, never auto-fixed + (should (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments))))) + +(ert-deftest lo-dated-log-active-deadline-is-flagged () + "An active DEADLINE on a dated-log entry is flagged too." + (let* ((out (lo-test--run + "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 shipped\nDEADLINE: <2026-06-25 Thu>\n")) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments))))) + +(ert-deftest lo-dated-log-clean-entry-not-flagged () + "A dated-log entry with no active planning timestamp is correct — not flagged." + (let* ((out (lo-test--run + "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 done cleanly\nBody only.\n")) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should-not (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments))))) + +(ert-deftest lo-dated-log-inactive-timestamp-not-flagged () + "An inactive [..] timestamp doesn't render on the agenda, so it isn't flagged — +only active <..> planning timestamps are the defect." + (let* ((out (lo-test--run + "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 recorded\nSCHEDULED: [2026-06-18 Thu]\n")) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should-not (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments))))) + +(ert-deftest lo-dated-log-active-scheduled-on-live-todo-not-flagged () + "A live TODO (keyword present) that legitimately carries an active SCHEDULED is +not a dated-log heading, so this checker leaves it alone." + (let* ((out (lo-test--run + "* Open Work\n\n** TODO [#B] Parent\n*** TODO [#C] real upcoming task\nSCHEDULED: <2026-06-18 Thu>\n")) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should-not (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments))))) + ;;; --------------------------------------------------------------------------- ;;; structural heading checks (org-lint gaps) @@ -817,3 +967,134 @@ heading, so it is not flagged — only two-or-more indented stars are." (provide 'test-lint-org) ;;; test-lint-org.el ends here + +;;; --------------------------------------------------------------------------- +;;; task-missing-last-reviewed (claude-rules/todo-format.md) + +(ert-deftest lo-task-without-last-reviewed-is-judgment () + "An open level-2 task with no :LAST_REVIEWED: is flagged." + (let* ((out (lo-test--run "* Open Work\n** TODO [#B] A task :feature:\nBody.\n")) + (js (lo-test--judgments (plist-get out :issues)))) + (should (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +(ert-deftest lo-task-with-last-reviewed-is-clean () + "A task carrying the property is not flagged." + (let* ((out (lo-test--run (concat "* Open Work\n** TODO [#B] A task :feature:\n" + ":PROPERTIES:\n:LAST_REVIEWED: 2026-07-23\n:END:\n" + "Body.\n"))) + (js (lo-test--judgments (plist-get out :issues)))) + (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +(ert-deftest lo-task-last-reviewed-accepts-org-timestamp () + "The org-native [YYYY-MM-DD Day] form counts, matching the staleness script." + (let* ((out (lo-test--run (concat "* Open Work\n** TODO [#B] A task :feature:\n" + ":PROPERTIES:\n:LAST_REVIEWED: [2026-07-23 Thu]\n:END:\n"))) + (js (lo-test--judgments (plist-get out :issues)))) + (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +(ert-deftest lo-done-task-without-last-reviewed-is-clean () + "Completed tasks leave the review pool, so they are never flagged." + (let* ((out (lo-test--run (concat "* Open Work\n** DONE [#B] A task :feature:\n" + "CLOSED: [2026-07-23 Thu]\n"))) + (js (lo-test--judgments (plist-get out :issues)))) + (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +(ert-deftest lo-subtask-without-last-reviewed-is-clean () + "Only level-2 tasks are in the review pool; deeper headings are not." + (let* ((out (lo-test--run (concat "* Open Work\n** TODO [#B] Parent :feature:\n" + ":PROPERTIES:\n:LAST_REVIEWED: 2026-07-23\n:END:\n" + "*** TODO A sub-task\n"))) + (js (lo-test--judgments (plist-get out :issues)))) + (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +(ert-deftest lo-cookieless-task-without-last-reviewed-is-clean () + "The staleness script selects on a priority cookie, so match that scope." + (let* ((out (lo-test--run "* Open Work\n** TODO Manual testing and validation\n")) + (js (lo-test--judgments (plist-get out :issues)))) + (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +(ert-deftest lo-verify-task-without-last-reviewed-is-judgment () + "VERIFY is in the review pool too." + (let* ((out (lo-test--run "* Open Work\n** VERIFY [#B] Waiting on Craig\n")) + (js (lo-test--judgments (plist-get out :issues)))) + (should (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +;;; --------------------------------------------------------------------------- +;;; todo-format checkers skip docs/specs/ files (claude-rules/todo-format.md) +;; +;; The four todo-format-family checkers encode todo.org completion conventions. +;; A spec legitimately uses ** DONE <decision> with no CLOSED cookie and +;; ** <dated> — <who> review-history headings, so those checkers misfire on +;; every spec. They must skip any file under a docs/specs/ path segment. + +(defun lo-test--run-at (relpath content) + "Write CONTENT to <tmpdir>/RELPATH, run lint on it, return :issues. +RELPATH is a relative path (may contain slashes) so a docs/specs/ segment +can be exercised — the checkers key on the file's path, not just its name." + (let* ((root (make-temp-file "lo-test-root-" t)) + (file (expand-file-name relpath root))) + (make-directory (file-name-directory file) t) + (unwind-protect + (progn + (with-temp-file file (insert content)) + (lo-test--reset) + (lo-process-file file) + (prog1 (list :issues lo-issues) + (lo-test--drop-buffer file))) + (delete-directory root t)))) + +(defconst lo-test--spec-decisions + "* Decisions [1/1]\n** DONE Some decision\n- Context: x\n" + "A spec Decisions section: a level-2 DONE with no CLOSED cookie.") + +(defconst lo-test--spec-history + "* Review history\n** 2026-07-14 Tue @ 02:03:28 -0500 — Claude — responder\n- What: x\n" + "A spec review-history section: a level-2 dated header.") + +(ert-deftest lo-todo-checkers-fire-on-a-normal-org-file () + "Baseline: the checkers DO fire on a non-spec path (the bug is scope, not silence)." + (let* ((out (lo-test--run-at "todo.org" lo-test--spec-decisions)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should (memq 'level2-done-without-closed cs)))) + +(ert-deftest lo-level2-done-without-closed-skips-specs () + (let* ((out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" lo-test--spec-decisions)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should-not (memq 'level2-done-without-closed cs)))) + +(ert-deftest lo-level2-dated-header-skips-specs () + (let* ((out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" lo-test--spec-history)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should-not (memq 'level-2-dated-header cs)))) + +(ert-deftest lo-dated-log-active-timestamp-skips-specs () + (let* ((c "* History\n** 2026-07-14 Tue @ 02:03:28 -0500 — did a thing\nSCHEDULED: <2026-07-20 Mon>\n") + (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should-not (memq 'dated-log-heading-active-timestamp cs)))) + +(ert-deftest lo-subtask-done-not-dated-skips-specs () + (let* ((c "* Work\n** TODO Parent\n*** DONE A sub-decision\n") + (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should-not (memq 'subtask-done-not-dated cs)))) + +(ert-deftest lo-link-checks-still-fire-on-specs () + "Only the todo-format family is scoped out; a broken link in a spec still flags." + (let* ((c "* X\n[[file:does-not-exist-xyz.org][link]]\n") + (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should (memq 'link-to-local-file cs)))) + +(ert-deftest lo-task-missing-last-reviewed-skips-specs () + "The fifth todo-format checker (added 2026-07-23) skips specs too — a spec's +phases section may carry ** TODO [#x] items that aren't backlog tasks." + (let* ((c "* Implementation phases\n** TODO [#B] Phase one\nBody.\n") + (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should-not (memq 'task-missing-last-reviewed cs))) + ;; And still fires on a normal file. + (let* ((c "* Work\n** TODO [#B] Real backlog task\nBody.\n") + (out (lo-test--run-at "todo.org" c)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should (memq 'task-missing-last-reviewed cs)))) diff --git a/claude-templates/.ai/scripts/tests/test-todo-cleanup.el b/claude-templates/.ai/scripts/tests/test-todo-cleanup.el index ffbf2fb..1e964b3 100644 --- a/claude-templates/.ai/scripts/tests/test-todo-cleanup.el +++ b/claude-templates/.ai/scripts/tests/test-todo-cleanup.el @@ -31,6 +31,7 @@ (defun tc-test--reset (&optional check) (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-archived-to-file 0 tc-issues nil + tc-sealed 0 tc-seal nil tc-convert-subtasks nil tc-check-only (and check t) tc-archive-done t tc-sync-child-priority nil tc-current-file nil @@ -40,6 +41,7 @@ (defun tc-test--reset-sync (&optional check) (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-archived-to-file 0 tc-issues nil + tc-sealed 0 tc-seal nil tc-check-only (and check t) tc-archive-done nil tc-sync-child-priority t tc-current-file nil @@ -514,6 +516,12 @@ gitignore todo.org, then run `--archive-done' aging with the DEFAULT archive pat .gitignore contents or nil), :archive-ignored (whether git ignores the archive), :archive-exists." (let* ((root (make-temp-file "tc-git-" t)) + ;; Private backup dir: this helper writes a file literally named + ;; todo.org and runs a real (non-check) pass, so without this its + ;; backup lands in the shared temp dir under the exact production + ;; name and is indistinguishable from a real one. + (temporary-file-directory + (file-name-as-directory (make-temp-file "tc-git-bk-" t))) (todo (expand-file-name "todo.org" root)) (archive (expand-file-name "archive/task-archive.org" root)) (gi (expand-file-name ".gitignore" root))) @@ -534,7 +542,8 @@ gitignore todo.org, then run `--archive-done' aging with the DEFAULT archive pat :archive-ignored (eq 0 (call-process "git" nil nil nil "check-ignore" "-q" archive)) :archive-exists (file-readable-p archive))) - (delete-directory root t)))) + (delete-directory root t) + (delete-directory temporary-file-directory t)))) (ert-deftest tc-age-self-protect-gitignores-archive-when-todo-ignored () "When the todo file is gitignored, the aged-out archive is added to .gitignore @@ -578,6 +587,95 @@ entry is added for it." (should (> (plist-get out :archived) 0))))) ;;; --------------------------------------------------------------------------- +;;; --archive-done retention default + +(ert-deftest tc-archive-retain-default-is-one-month () + "The shipped retention default is one month (31 days), not the legacy 7. +The defvar initializes from this defconst; the live var itself is mutated by +other tests, so the immutable defconst is the stable contract to pin." + (should (= 31 tc-archive-retain-days-default))) + +;;; --------------------------------------------------------------------------- +;;; --seal: rename the working archive to resolved-YYYY-MM-DD.org + +(defun tc-test--seal (&optional opts) + "Run `--seal' against a temp todo file with a temp archive dir. +OPTS is a plist: :archive-content (seed task-archive.org with this; nil = no +working archive), :ref (YEAR MONTH DAY seal date; default (2026 7 18)), +:check, :presealed (also create resolved-<ref>.org first, to test collision). +Returns a plist: :sealed count, :issues, :working-exists, :sealed-exists, +:sealed-name, :report." + (let* ((ref (or (plist-get opts :ref) '(2026 7 18))) + (check (plist-get opts :check)) + (archive-content (plist-get opts :archive-content)) + (todo (make-temp-file "tc-seal-todo-" nil ".org")) + (adir (make-temp-file "tc-seal-arch-" t)) + (afile (expand-file-name "task-archive.org" adir)) + (sealed-name (format "resolved-%04d-%02d-%02d.org" + (nth 0 ref) (nth 1 ref) (nth 2 ref))) + (sealed (expand-file-name sealed-name adir))) + (unwind-protect + (progn + (with-temp-file todo (insert "* Open Work\n** TODO [#A] live\n")) + (when archive-content (with-temp-file afile (insert archive-content))) + (when (plist-get opts :presealed) + (with-temp-file sealed (insert "pre-existing seal\n"))) + (tc-test--reset check) + ;; Set every mode flag explicitly: tc-test--reset leaves + ;; tc-convert-subtasks untouched, so a convert test running earlier in + ;; the suite would otherwise still own the dispatch and run convert. + (setq tc-archive-done nil tc-sync-child-priority nil + tc-convert-subtasks nil tc-seal t tc-sealed 0 + tc-archive-reference-date ref + tc-archive-file afile) + (let ((report (with-output-to-string (tc-process-file todo) (tc-emit-report)))) + (tc-test--drop-buffer todo) + (list :sealed tc-sealed + :issues tc-issues + :working-exists (file-readable-p afile) + :sealed-exists (file-readable-p sealed) + :sealed-name sealed-name + :report report))) + (tc-test--drop-buffer todo) + (delete-file todo) + (delete-directory adir t)))) + +(ert-deftest tc-seal-renames-working-archive-to-dated-file () + "Normal: --seal renames task-archive.org to resolved-<seal-date>.org." + (let ((out (tc-test--seal '(:archive-content "* Resolved (archived)\n** DONE old\n" + :ref (2026 7 18))))) + (should (= 1 (plist-get out :sealed))) + (should-not (plist-get out :working-exists)) + (should (plist-get out :sealed-exists)) + (should (equal "resolved-2026-07-18.org" (plist-get out :sealed-name))) + (should (tc-test--has (plist-get out :report) "sealed task-archive.org → resolved-2026-07-18.org")))) + +(ert-deftest tc-seal-nothing-to-seal-is-a-reported-noop () + "Boundary: no working archive present — reported no-op, nothing created." + (let ((out (tc-test--seal '(:ref (2026 7 18))))) + (should (= 0 (plist-get out :sealed))) + (should-not (plist-get out :sealed-exists)) + (should (tc-test--has (plist-get out :report) "no working archive to seal")))) + +(ert-deftest tc-seal-check-mode-previews-without-renaming () + "Boundary: --check reports the seal but leaves the working archive in place." + (let ((out (tc-test--seal '(:archive-content "* Resolved (archived)\n" + :ref (2026 7 18) :check t)))) + (should (= 1 (plist-get out :sealed))) + (should (plist-get out :working-exists)) + (should-not (plist-get out :sealed-exists)) + (should (tc-test--has (plist-get out :report) "would seal")))) + +(ert-deftest tc-seal-refuses-to-clobber-existing-sealed-file () + "Error: resolved-<today>.org already exists — refuse, leave both files intact." + (let ((out (tc-test--seal '(:archive-content "* Resolved (archived)\n" + :ref (2026 7 18) :presealed t)))) + (should (= 0 (plist-get out :sealed))) + (should (plist-get out :working-exists)) + (should (plist-get out :sealed-exists)) + (should (tc-test--has (plist-get out :report) "already exists")))) + +;;; --------------------------------------------------------------------------- ;;; Sync-child-priority harness + fixtures (defun tc-test--sync (content &optional runs check) @@ -773,7 +871,7 @@ in ISSUES, in document order." (defun tc-test--reset-convert (&optional check) (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-converted 0 tc-archived-to-file 0 - tc-issues nil + tc-issues nil tc-sealed 0 tc-seal nil tc-check-only (and check t) tc-archive-done nil tc-sync-child-priority nil tc-convert-subtasks t tc-current-file nil @@ -927,8 +1025,9 @@ CLOSED: [2026-06-27 Sat 12:50] DEADLINE: <2026-06-30 Tue> Body line. ") -(ert-deftest tc-convert-preserves-deadline-on-shared-planning-line-boundary () - "Boundary: removing the CLOSED cookie keeps a DEADLINE sharing its planning line." +(ert-deftest tc-convert-strips-deadline-sharing-the-planning-line-boundary () + "Boundary: a DEADLINE sharing the CLOSED planning line goes too — a dated-log +entry carries no active planning timestamp (todo-format.md). Body survives." (let* ((out (tc-test--convert tc-test--convert-closed-with-deadline)) (res (plist-get out :result))) (should (= 1 (plist-get out :converted))) @@ -936,8 +1035,142 @@ Body line. "^\\*\\*\\* 2026-06-27 Sat @ 12:50:00 [-+][0-9]\\{4\\} Ship the panel$" res)) (should-not (string-match-p "CLOSED:" res)) - (should (string-match-p "^DEADLINE: <2026-06-30 Tue>$" res)) + (should-not (string-match-p "DEADLINE:" res)) + (should (string-match-p "^Body line\\.$" res)))) + +(defconst tc-test--convert-closed-and-scheduled-separate-lines + "* Project Open Work +** TODO [#B] Parent task +*** DONE [#C] Book the venue :feature: +CLOSED: [2026-06-27 Sat 12:50] +SCHEDULED: <2026-06-20 Sat> +Body line. +") + +(ert-deftest tc-convert-strips-scheduled-on-its-own-line () + "Normal (the home bug): a SCHEDULED planning line on its own — the completion +rewrite dropped keyword/priority/tags but left the SCHEDULED, pinning the dated +entry to the agenda as weeks-overdue. Both planning lines go; body survives." + (let* ((out (tc-test--convert tc-test--convert-closed-and-scheduled-separate-lines)) + (res (plist-get out :result))) + (should (= 1 (plist-get out :converted))) + (should (string-match-p + "^\\*\\*\\* 2026-06-27 Sat @ 12:50:00 [-+][0-9]\\{4\\} Book the venue$" + res)) + (should-not (string-match-p "CLOSED:" res)) + (should-not (string-match-p "SCHEDULED:" res)) (should (string-match-p "^Body line\\.$" res)))) +(defconst tc-test--convert-scheduled-in-body-prose + "* Project Open Work +** TODO [#B] Parent task +*** DONE [#C] Note the mechanism :feature: +CLOSED: [2026-06-27 Sat 12:50] +An active SCHEDULED: <2026-06-20 Sat> in prose must survive. +") + +(ert-deftest tc-convert-leaves-planning-shaped-body-prose-alone () + "Boundary: a planning-shaped token inside body prose (not a canonical planning +line) is left untouched — the strip stops at the first non-planning line." + (let* ((out (tc-test--convert tc-test--convert-scheduled-in-body-prose)) + (res (plist-get out :result))) + (should (= 1 (plist-get out :converted))) + (should-not (string-match-p "CLOSED:" res)) + (should (string-match-p "An active SCHEDULED: <2026-06-20 Sat> in prose must survive\\." res)))) + (provide 'test-todo-cleanup) ;;; test-todo-cleanup.el ends here + +;;; --------------------------------------------------------------------------- +;;; Backup before mutating (parity with lint-org.el / wrap-org-table.el) +;; +;; todo-cleanup rewrites todo.org in place and left no copy behind, while both +;; sibling org-mutators back up to /tmp first. It is also the one that runs most +;; often (every wrap, every sentry cycle). Emacs's own backup does not fire under +;; --batch -q, so there was genuinely no undo short of git. + +(ert-deftest tc-backup-written-before-a-real-mutation () + "A real (non-check) run leaves a copy holding the pre-edit content. + +`temporary-file-directory' is rebound to a private dir for the duration: the +backup name derives from the *file's* basename, and the real todo.org shares +that basename, so a live sentry run writing /tmp/todo.org.before-todo-cleanup.* +would otherwise be indistinguishable from this test's own artifact. The first +version of this test globbed the shared /tmp and passed only until a real run +created one (2026-07-24)." + (let* ((dir (make-temp-file "tc-backup-" t)) + (bdir (file-name-as-directory (make-temp-file "tc-bk-" t))) + (file (expand-file-name "todo.org" dir)) + (before "* P Open Work\n** TODO [#B] parent\n*** DONE a subtask\nCLOSED: [2026-07-01 Tue]\n")) + (unwind-protect + (progn + (with-temp-file file (insert before)) + (let ((tc-check-only nil) + (tc-convert-subtasks t) + (temporary-file-directory bdir)) + (tc-process-file file)) + (let ((backups (file-expand-wildcards + (concat bdir "todo.org.before-todo-cleanup.*")))) + (should backups) + (should (string-match-p + "a subtask" + (with-temp-buffer (insert-file-contents (car backups)) + (buffer-string)))))) + (delete-directory dir t) + (delete-directory bdir t)))) + +(ert-deftest tc-no-backup-in-check-mode () + "--check writes nothing, so it must not leave a backup either. +Uses a private `temporary-file-directory' for the same isolation reason." + (let* ((dir (make-temp-file "tc-backup-" t)) + (bdir (file-name-as-directory (make-temp-file "tc-bk-" t))) + (file (expand-file-name "todo.org" dir))) + (unwind-protect + (progn + (with-temp-file file + (insert "* P Open Work\n** TODO [#B] parent\n*** DONE sub\nCLOSED: [2026-07-01 Tue]\n")) + (let ((tc-check-only t) + (tc-convert-subtasks t) + (temporary-file-directory bdir)) + (tc-process-file file)) + (should-not (file-expand-wildcards + (concat bdir "todo.org.before-todo-cleanup.*")))) + (delete-directory dir t) + (delete-directory bdir t)))) + +(ert-deftest tc-backup-never-overwrites-an-earlier-one () + "Two invocations in the same second must not collapse to one backup. + +open-tasks.org runs --convert-subtasks then --archive-done back to back, each +a sub-second batch run. With a second-resolution stamp and copy-file's +OK-IF-ALREADY-EXISTS, the second invocation overwrote the first's backup with +already-mutated content, so the true pre-session original was unrecoverable — +the exact state the backup exists to preserve (found 2026-07-24 in review)." + (let* ((dir (make-temp-file "tc-collide-" t)) + (bdir (file-name-as-directory (make-temp-file "tc-cbk-" t))) + (file (expand-file-name "todo.org" dir)) + (original (concat "* P Open Work\n** TODO [#B] parent\n*** DONE sub\n" + "CLOSED: [2026-07-01 Tue]\n" + "* P Resolved\n** DONE [#C] old\nCLOSED: [2025-01-01 Wed]\n"))) + (unwind-protect + (progn + (with-temp-file file (insert original)) + ;; Two back-to-back invocations, as the shipped workflow does. + (let ((temporary-file-directory bdir)) + (let ((tc-check-only nil) (tc-convert-subtasks t)) + (tc-process-file file)) + (let ((tc-check-only nil) (tc-convert-subtasks nil) (tc-archive-done t) + (tc-archive-retain-days nil)) + (tc-process-file file))) + (let ((backups (file-expand-wildcards + (concat bdir "todo.org.before-todo-cleanup.*")))) + ;; Both invocations kept their own backup. + (should (= (length backups) 2)) + ;; And one of them still holds the true original. + (should (cl-some (lambda (b) + (string= original + (with-temp-buffer (insert-file-contents b) + (buffer-string)))) + backups)))) + (delete-directory dir t) + (delete-directory bdir t)))) diff --git a/claude-templates/.ai/scripts/tests/test_apkg_to_orgdrill.py b/claude-templates/.ai/scripts/tests/test_apkg_to_orgdrill.py new file mode 100644 index 0000000..6a95ea4 --- /dev/null +++ b/claude-templates/.ai/scripts/tests/test_apkg_to_orgdrill.py @@ -0,0 +1,301 @@ +"""Tests for apkg-to-orgdrill.py — the inverse of flashcard-to-anki.py. + +The converter reads an Anki .apkg (a zip holding collection.anki2 / .anki21 +sqlite) and emits an org-drill .org in the house canonical shape. It is +stdlib-only (zipfile + sqlite3), so it imports directly — no genanki stub. + +The apkg schema these tests build by hand mirrors what genanki actually +writes, confirmed against a real apkg generated from flashcard-to-anki.py: + - col.decks : JSON {did: {"name": ...}}, always including id-1 "Default" + - col.models : JSON {mid: {"name": ..., "flds": [{"name": "Front"}, ...]}} + - notes.flds : fields joined by \x1f; tags space-padded (" tag ") + - cards : nid -> did (the Default deck carries no cards) + +The round-trip test closes the loop through flashcard-to-anki.py's own +parse(): original org -> forward parse tuples -> apkg fixture -> converter +-> recovered org -> forward parse -> assert the (front, back, tag) tuples +match. Only the apkg materialization is hand-built (the genanki boundary); +everything else is the real code on both sides. +""" +from __future__ import annotations + +import importlib.util +import json +import sqlite3 +import sys +import types +import zipfile +from pathlib import Path + +import pytest + +SCRIPTS = Path(__file__).resolve().parents[1] +CONVERTER = SCRIPTS / "apkg-to-orgdrill.py" +FORWARD = SCRIPTS / "flashcard-to-anki.py" + + +def _load(path: Path, name: str, stub_genanki: bool = False): + if stub_genanki: + sys.modules.setdefault("genanki", types.ModuleType("genanki")) + spec = importlib.util.spec_from_file_location(name, path) + assert spec and spec.loader + module = importlib.util.module_from_spec(spec) + # Register before exec: @dataclass resolves cls.__module__ via sys.modules + # (Python 3.14), which is None for an unregistered importlib module. + sys.modules[name] = module + spec.loader.exec_module(module) + return module + + +@pytest.fixture(scope="module") +def conv(): + return _load(CONVERTER, "apkg_to_orgdrill") + + +@pytest.fixture(scope="module") +def forward(): + return _load(FORWARD, "flashcard_to_anki", stub_genanki=True) + + +# --- fixture builder: write a genanki-shaped apkg by hand ------------------ + +def _make_apkg( + path: Path, + decks: dict[int, str], + models: dict[int, list[str]], + notes: list[tuple[int, int, list[str], str]], # (nid, mid, fields, tag) + cards: list[tuple[int, int]], # (nid, did) + *, + media: str = "{}", +) -> None: + """Materialize a minimal apkg matching genanki's collection.anki2 shape.""" + col_dir = path.parent / f"{path.stem}-build" + col_dir.mkdir(parents=True, exist_ok=True) + db = col_dir / "collection.anki2" + if db.exists(): + db.unlink() + con = sqlite3.connect(db) + con.execute("CREATE TABLE col (id INTEGER, decks TEXT, models TEXT)") + decks_json = {"1": {"name": "Default"}} + decks_json.update({str(did): {"name": name} for did, name in decks.items()}) + models_json = { + str(mid): {"name": f"{decks.get(list(decks)[0], 'M')} model", + "flds": [{"name": n, "ord": i} for i, n in enumerate(flds)]} + for mid, flds in models.items() + } + con.execute("INSERT INTO col (id, decks, models) VALUES (1, ?, ?)", + (json.dumps(decks_json), json.dumps(models_json))) + con.execute("CREATE TABLE notes (id INTEGER, mid INTEGER, flds TEXT, tags TEXT)") + for nid, mid, fields, tag in notes: + con.execute("INSERT INTO notes (id, mid, flds, tags) VALUES (?, ?, ?, ?)", + (nid, mid, "\x1f".join(fields), f" {tag} " if tag else " ")) + con.execute("CREATE TABLE cards (id INTEGER, nid INTEGER, did INTEGER)") + for i, (nid, did) in enumerate(cards): + con.execute("INSERT INTO cards (id, nid, did) VALUES (?, ?, ?)", (1000 + i, nid, did)) + con.commit() + con.close() + with zipfile.ZipFile(path, "w") as z: + z.write(db, "collection.anki2") + z.writestr("media", media) + + +# --- html_to_org_body ------------------------------------------------------ + +def test_html_to_org_splits_br_into_lines(conv): + assert conv.html_to_org_body("one<br>two<br>three") == ["one", "two", "three"] + + +def test_html_to_org_handles_br_variants(conv): + assert conv.html_to_org_body("a<br/>b<br />c<BR>d") == ["a", "b", "c", "d"] + + +def test_html_to_org_unescapes_entities_amp_last(conv): + # Inverts escape_html (which escapes & first): < > & -> < > &. + assert conv.html_to_org_body("x <tag> & y") == ["x <tag> & y"] + + +def test_html_to_org_preserves_a_literal_escaped_entity(conv): + # Forward-escaping the literal "<" yields "&lt;"; the inverse must + # recover "<", not "<". + assert conv.html_to_org_body("&lt;") == ["<"] + + +def test_html_to_org_strips_answer_hr(conv): + assert conv.html_to_org_body('front<hr id="answer">back') == ["front", "back"] + + +def test_html_to_org_empty_back_is_empty(conv): + assert conv.html_to_org_body("") == [] + + +# --- read_apkg ------------------------------------------------------------- + +def test_read_apkg_single_deck_recovers_front_back_tag_deck(conv, tmp_path): + apkg = tmp_path / "d.apkg" + _make_apkg( + apkg, + decks={20: "My Deck"}, + models={9: ["Front", "Back"]}, + notes=[(100, 9, ["Q1?", "A1.<br>line2"], "sec-one")], + cards=[(100, 20)], + ) + recovered = conv.read_apkg(apkg) + assert len(recovered) == 1 + note = recovered[0] + assert note.deck == "My Deck" + assert note.front == "Q1?" + assert note.back_html == "A1.<br>line2" + assert note.tag == "sec-one" + + +def test_read_apkg_multiple_decks_grouped(conv, tmp_path): + apkg = tmp_path / "multi.apkg" + _make_apkg( + apkg, + decks={20: "Deck A", 21: "Deck B"}, + models={9: ["Front", "Back"]}, + notes=[(100, 9, ["QA?", "AA"], "ta"), (101, 9, ["QB?", "AB"], "tb")], + cards=[(100, 20), (101, 21)], + ) + decks = {n.deck for n in conv.read_apkg(apkg)} + assert decks == {"Deck A", "Deck B"} + + +def test_read_apkg_skips_default_deck_without_cards(conv, tmp_path): + apkg = tmp_path / "def.apkg" + _make_apkg( + apkg, + decks={20: "Real Deck"}, + models={9: ["Front", "Back"]}, + notes=[(100, 9, ["Q?", "A"], "t")], + cards=[(100, 20)], + ) + assert {n.deck for n in conv.read_apkg(apkg)} == {"Real Deck"} + + +def test_read_apkg_warns_and_skips_non_basic_model(conv, tmp_path, capsys): + apkg = tmp_path / "cloze.apkg" + _make_apkg( + apkg, + decks={20: "Cloze Deck"}, + models={9: ["Text", "Extra"]}, # not Front/Back + notes=[(100, 9, ["some {{c1::text}}", "extra"], "t")], + cards=[(100, 20)], + ) + recovered = conv.read_apkg(apkg) + assert recovered == [] + assert "skip" in capsys.readouterr().err.lower() + + +def test_read_apkg_reads_anki21_collection_name(conv, tmp_path): + # A .anki21 collection filename must be read the same as .anki2. + apkg = tmp_path / "new.apkg" + _make_apkg( + apkg, + decks={20: "Deck"}, + models={9: ["Front", "Back"]}, + notes=[(100, 9, ["Q?", "A"], "t")], + cards=[(100, 20)], + ) + # Rewrite the zip renaming the collection member to .anki21. + with zipfile.ZipFile(apkg) as z: + data = z.read("collection.anki2") + media = z.read("media") + with zipfile.ZipFile(apkg, "w") as z: + z.writestr("collection.anki21", data) + z.writestr("media", media) + assert conv.read_apkg(apkg)[0].front == "Q?" + + +def test_read_apkg_flags_media_reference(conv, tmp_path, capsys): + apkg = tmp_path / "media.apkg" + _make_apkg( + apkg, + decks={20: "Deck"}, + models={9: ["Front", "Back"]}, + notes=[(100, 9, ["Q?", 'see <img src="x.png">'], "t")], + cards=[(100, 20)], + ) + conv.read_apkg(apkg) + assert "media" in capsys.readouterr().err.lower() + + +# --- notes_to_org ---------------------------------------------------------- + +def test_notes_to_org_emits_canonical_shape(conv): + Note = conv.Note + notes = [ + Note(deck="My Deck", front="Q1?", back_html="A1.", tag="alpha"), + Note(deck="My Deck", front="Q2?", back_html="A2.", tag="alpha"), + ] + ids = iter(["id-1", "id-2"]) + org = conv.notes_to_org(notes, "My Deck", new_id=lambda: next(ids)) + assert "#+TITLE: My Deck" in org + assert "* alpha" in org + assert "** Q1? :drill:" in org + assert ":ID: id-1" in org + assert ":ID: id-2" in org + assert org.count("* alpha") == 1 # both cards share one section + + +def test_notes_to_org_distinct_tags_get_distinct_sections(conv): + Note = conv.Note + notes = [ + Note(deck="D", front="Qa?", back_html="a", tag="alpha"), + Note(deck="D", front="Qb?", back_html="b", tag="beta"), + ] + org = conv.notes_to_org(notes, "D", new_id=lambda: "x") + assert "* alpha" in org and "* beta" in org + + +# --- round-trip through the real forward parse() --------------------------- + +def test_round_trip_matches_forward_parse_tuples(conv, forward, tmp_path): + original = ( + "#+TITLE: RT Deck\n" + "\n" + "* First Section\n" + "** What is 2+2? :drill:\n" + ":PROPERTIES:\n:ID: aaaa\n:END:\n" + "Four.\n" + "Second line with <angle> & amp.\n" + "\n" + "* Second Section\n" + "** Capital of France? :drill:\n" + "Paris.\n" + ) + tuples = forward.parse(original) # [(front, back_html, anki_tags), ...] + assert len(tuples) == 2 + + apkg = tmp_path / "rt.apkg" + _make_apkg( + apkg, + decks={20: "RT Deck"}, + models={9: ["Front", "Back"]}, + # anki_tags is a list; the apkg tags field is space-joined. + notes=[(100 + i, 9, [f, b], " ".join(tags)) + for i, (f, b, tags) in enumerate(tuples)], + cards=[(100 + i, 20) for i in range(len(tuples))], + ) + + by_deck = conv.convert(apkg) + assert set(by_deck) == {"RT Deck"} + recovered_tuples = forward.parse(by_deck["RT Deck"]) + assert recovered_tuples == tuples + + +# --- errors ---------------------------------------------------------------- + +def test_read_apkg_missing_collection_errors(conv, tmp_path): + bad = tmp_path / "bad.apkg" + with zipfile.ZipFile(bad, "w") as z: + z.writestr("media", "{}") + with pytest.raises(Exception): + conv.read_apkg(bad) + + +def test_read_apkg_not_a_zip_errors(conv, tmp_path): + notzip = tmp_path / "plain.apkg" + notzip.write_text("not a zip") + with pytest.raises(Exception): + conv.read_apkg(notzip) diff --git a/claude-templates/.ai/scripts/tests/test_cj_remove_block.py b/claude-templates/.ai/scripts/tests/test_cj_remove_block.py index 2c8dade..3cdee46 100644 --- a/claude-templates/.ai/scripts/tests/test_cj_remove_block.py +++ b/claude-templates/.ai/scripts/tests/test_cj_remove_block.py @@ -14,6 +14,34 @@ import pytest SCRIPT = Path(__file__).parent.parent / "cj-remove-block.py" +@pytest.fixture(autouse=True) +def isolated_tmpdir(tmp_path, monkeypatch): + """Give every test in this module a private TMPDIR. + + The script backs up to the system temp dir under a name derived from the + edited file's BASENAME. The real todo.org shares that basename, so any test + operating on a fixture named todo.org writes something indistinguishable + from a production backup — and an earlier version of this file globbed the + shared /tmp and unlinked every match, so a routine `make test` destroyed + Craig's real backups (found in review, 2026-07-24). + + Isolating at module scope rather than per-test is deliberate: the same bug + was fixed once in the elisp sibling and left here, so relying on each new + test to remember is exactly how it recurred. Autouse makes it structural. + """ + d = tmp_path / "_tmpdir" + d.mkdir() + # TMPDIR covers subprocess invocations of the script. + monkeypatch.setenv("TMPDIR", str(d)) + # tempfile.gettempdir() caches its answer on first call, so a test that + # loads the module in-process would keep writing to the real /tmp no matter + # what TMPDIR says. Override the cache too — this is the gap that made the + # env-var-only version still leak one backup per suite run. + import tempfile as _tempfile + monkeypatch.setattr(_tempfile, "tempdir", str(d)) + return d + + @pytest.fixture def run_remove(tmp_path): """Write content to a temp org file, run cj-remove-block, return new contents.""" @@ -155,3 +183,142 @@ class TestCjRemoveBlockSafety: err, post_content = run_remove_expecting_failure(original, start=4, end=2) assert err.returncode != 0 assert post_content == original + + +class TestMultiBlockRangeRefused: + """The validation exists to catch a drifted range, but it only checked the + first and last lines of that range. A span from one block's opening fence to + a LATER block's closing fence passed, and the removal silently deleted every + line between — real prose, headings, whole tasks — with a zero exit. Drift is + the skill's normal operating mode (respond-to-cj-comments edits the file as it + processes, and a file under cj review usually holds several blocks), so this + is the exact scenario the check was written for. Reproduced 2026-07-24.""" + + TWO_BLOCKS = ( + "* Alpha\n" + "#+begin_src cj:\n" + "note A\n" + "#+end_src\n" + "KEEP THIS LINE\n" + "* Beta\n" + "#+begin_src cj:\n" + "note B\n" + "#+end_src\n" + ) + + def test_range_spanning_two_blocks_is_refused(self, run_remove_expecting_failure): + # Lines 2..9: block one's opener through block two's closer. + err, content = run_remove_expecting_failure(self.TWO_BLOCKS, 2, 9) + assert err.returncode == 1 + assert "KEEP THIS LINE" in content, "content between the blocks was destroyed" + assert "* Beta" in content, "a heading between the blocks was destroyed" + + def test_refusal_names_the_reason(self, run_remove_expecting_failure): + err, _ = run_remove_expecting_failure(self.TWO_BLOCKS, 2, 9) + assert "more than one" in err.stderr.decode().lower() + + def test_a_correct_single_block_range_still_removes(self, run_remove): + # The fix must not over-tighten: the legitimate range still works. + out = run_remove(self.TWO_BLOCKS, 2, 4) + assert "note A" not in out + assert "KEEP THIS LINE" in out + assert "note B" in out, "the second block must be untouched" + + def test_a_nested_end_src_inside_the_range_is_refused(self, run_remove_expecting_failure): + # Any #+end_src before the final line means the range covers >1 block. + content = ( + "#+begin_src cj:\n" + "a\n" + "#+end_src\n" + "middle\n" + "#+begin_src cj:\n" + "b\n" + "#+end_src\n" + ) + err, after = run_remove_expecting_failure(content, 1, 7) + assert err.returncode == 1 + assert "middle" in after + + +class TestSafeMutation: + """The script rewrites Craig's org files (todo.org, notes.org). It wrote with + a bare write_text, which truncates the target on open, and took no backup — + so a mid-write failure left the file truncated with no copy to recover from. + lint-org.el, the other tool that mutates these files, backs up to a temp dir + first. Match that, and make the write atomic. + + Every test here redirects TMPDIR to a private directory. The backup name + derives from the file's basename, and the real todo.org shares it, so a test + globbing the shared temp dir cannot tell its own artifact from a genuine + backup — and an earlier version of this class globbed /tmp and unlinked every + match, so a routine `make test` destroyed real backups (found in review, + 2026-07-24). Never glob or delete across the shared temp dir.""" + + ONE_BLOCK = "* T\n#+begin_src cj:\nnote\n#+end_src\nkeep\n" + + def test_a_backup_is_written_before_mutating(self, tmp_path): + import subprocess, glob, os + bdir = tmp_path / "bk" + bdir.mkdir() + f = tmp_path / "todo.org" + f.write_text(self.ONE_BLOCK) + subprocess.run( + ["python3", str(SCRIPT), "--file", str(f), "--start", "2", "--end", "4"], + check=True, capture_output=True, + env={**os.environ, "TMPDIR": str(bdir)}, + ) + backups = glob.glob(str(bdir / "todo.org.before-cj-remove.*")) + assert backups, "no backup was written before mutating the org file" + assert "note" in Path(max(backups)).read_text() + + def test_no_partial_file_when_the_write_fails(self, tmp_path, monkeypatch): + import importlib.util + spec = importlib.util.spec_from_file_location("crb", SCRIPT) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + bdir = tmp_path / "bk" + bdir.mkdir() + monkeypatch.setenv("TMPDIR", str(bdir)) + f = tmp_path / "todo.org" + f.write_text(self.ONE_BLOCK) + def boom(*a, **k): + raise OSError("disk full") + monkeypatch.setattr(mod.os, "replace", boom) + with pytest.raises(OSError): + mod.remove_range(f, 2, 4) + # The original survives intact — no truncation, no partial. + assert f.read_text() == self.ONE_BLOCK + + +class TestBackupNeverOverwrites: + """Same defect class as todo-cleanup's, and more reachable here: the + respond-to-cj-comments skill removes several annotations in quick + succession, so a second-resolution stamp collides and the later backup + overwrote the earlier one with already-mutated content.""" + + TWO_BLOCKS = ( + "* A\n#+begin_src cj:\nfirst\n#+end_src\n" + "* B\n#+begin_src cj:\nsecond\n#+end_src\n" + ) + + def test_consecutive_removals_each_keep_a_backup(self, tmp_path, monkeypatch): + import subprocess, glob + bdir = tmp_path / "bk" + bdir.mkdir() + monkeypatch.setenv("TMPDIR", str(bdir)) + f = tmp_path / "todo.org" + f.write_text(self.TWO_BLOCKS) + original = f.read_text() + # Remove the second block, then the first — back to back, same second. + subprocess.run(["python3", str(SCRIPT), "--file", str(f), + "--start", "6", "--end", "8"], + check=True, capture_output=True, + env={**__import__("os").environ, "TMPDIR": str(bdir)}) + subprocess.run(["python3", str(SCRIPT), "--file", str(f), + "--start", "2", "--end", "4"], + check=True, capture_output=True, + env={**__import__("os").environ, "TMPDIR": str(bdir)}) + backups = glob.glob(str(bdir / "todo.org.before-cj-remove.*")) + assert len(backups) == 2, f"expected 2 backups, got {len(backups)}" + contents = [Path(b).read_text() for b in backups] + assert original in contents, "no backup holds the true original" diff --git a/claude-templates/.ai/scripts/tests/test_flashcard_stats.py b/claude-templates/.ai/scripts/tests/test_flashcard_stats.py index 606f7c1..46deccc 100644 --- a/claude-templates/.ai/scripts/tests/test_flashcard_stats.py +++ b/claude-templates/.ai/scripts/tests/test_flashcard_stats.py @@ -217,6 +217,31 @@ def test_parse_cards_captures_body_without_drawer_planning_or_answer_header(stat assert c["body"] == "the real answer" +def test_parse_cards_counts_a_multitag_heading_as_a_card(stats): + """A card multi-tagged :fundamental:drill: still counts; the front is clean.""" + text = "* Sec\n** Q multi? :fundamental:drill:\nthe answer\n" + cards, _ = stats.parse_cards(text.splitlines()) + assert len(cards) == 1 + assert cards[0]["heading"] == "Q multi?" + assert cards[0]["body"] == "the answer" + + +def test_parse_cards_ignores_a_tagged_heading_without_drill(stats): + """A tagged heading missing :drill: is not a drill card.""" + text = "* Sec\n** Just a note :note:\nbody\n" + cards, _ = stats.parse_cards(text.splitlines()) + assert cards == [] + + +def test_parse_cards_body_stops_at_next_multitag_card(stats): + """The body scan ends at the next L2 card even when it is multi-tagged.""" + text = "** Q1? :a:drill:\nbody1\n** Q2? :drill:\nbody2\n" + cards, _ = stats.parse_cards(text.splitlines()) + assert len(cards) == 2 + assert cards[0]["body"] == "body1" + assert cards[1]["body"] == "body2" + + def test_find_duplicate_fronts_matches_normalized_headings(stats): cards = [ {"heading": "What is LEO?"}, diff --git a/claude-templates/.ai/scripts/tests/test_flashcard_to_anki.py b/claude-templates/.ai/scripts/tests/test_flashcard_to_anki.py index 87008a8..fa38b64 100644 --- a/claude-templates/.ai/scripts/tests/test_flashcard_to_anki.py +++ b/claude-templates/.ai/scripts/tests/test_flashcard_to_anki.py @@ -158,17 +158,18 @@ Geostationary Earth Orbit. def test_parse_returns_front_back_tag_per_card(drill): cards = drill.parse(SECTIONED) assert len(cards) == 2 - assert cards[0] == ("What is LEO?", "Low Earth Orbit.", "orbital-regimes") + # The section becomes the sole Anki tag (as a one-element list). + assert cards[0] == ("What is LEO?", "Low Earth Orbit.", ["orbital-regimes"]) assert cards[1][0] == "What is GEO?" def test_parse_card_without_a_section_gets_the_drill_tag(drill): - assert drill.parse("** Lone card? :drill:\nbody\n") == [("Lone card?", "body", "drill")] + assert drill.parse("** Lone card? :drill:\nbody\n") == [("Lone card?", "body", ["drill"])] def test_parse_strips_properties_drawer_from_back(drill): text = "** Q? :drill:\n:PROPERTIES:\n:ID: abc\n:END:\nThe answer.\n" - assert drill.parse(text) == [("Q?", "The answer.", "drill")] + assert drill.parse(text) == [("Q?", "The answer.", ["drill"])] def test_parse_trims_leading_and_trailing_blank_body_lines(drill): @@ -178,7 +179,59 @@ def test_parse_trims_leading_and_trailing_blank_body_lines(drill): def test_parse_card_with_only_a_drawer_has_empty_back(drill): text = "** Q? :drill:\n:PROPERTIES:\n:ID: x\n:END:\n" - assert drill.parse(text) == [("Q?", "", "drill")] + assert drill.parse(text) == [("Q?", "", ["drill"])] + + +# --- multi-tag headings, --tag-filter, --guid-salt ------------------------- + +MULTITAG = """* Fundamentals +** What is LEO? :fundamental:drill: +Low Earth Orbit. +** What is GEO? :drill: +Geostationary Earth Orbit. +""" + + +def test_parse_multitag_heading_is_a_card_when_drill_is_present(drill): + """A heading with a second org tag still parses when drill is among them.""" + cards = drill.parse(MULTITAG) + assert len(cards) == 2 + assert cards[0][0] == "What is LEO?" + + +def test_parse_multitag_tags_ride_along_next_to_the_section_tag(drill): + """Non-drill org tags become Anki tags alongside the section tag.""" + cards = drill.parse(MULTITAG) + assert cards[0][2] == ["fundamentals", "fundamental"] # section slug + org tag + assert cards[1][2] == ["fundamentals"] # drill-only -> section only + + +def test_parse_heading_without_drill_tag_is_not_a_card(drill): + """A tagged heading missing :drill: is not a card (e.g. :note:).""" + assert drill.parse("* S\n** Just a note :note:\nbody\n") == [] + + +def test_parse_tag_filter_returns_only_cards_with_that_org_tag(drill): + """--tag-filter narrows to cards carrying the given org tag.""" + cards = drill.parse(MULTITAG, tag_filter="fundamental") + assert len(cards) == 1 + assert cards[0][0] == "What is LEO?" + + +def test_parse_body_bounded_by_any_l1_or_l2_heading(drill): + """A card body stops at the next L1/L2 heading, multi-tagged or not.""" + text = "** Q1? :a:drill:\nbody1\n** Q2? :drill:\nbody2\n" + cards = drill.parse(text) + assert cards[0][1] == "body1" + assert cards[1][1] == "body2" + + +def test_card_guid_salt_changes_the_guid(drill, monkeypatch): + """--guid-salt gives a subset deck its own GUID space; no salt is unchanged.""" + monkeypatch.setattr(drill.genanki, "guid_for", lambda *a: ":".join(a), raising=False) + assert drill.card_guid("front", None) == "front" + assert drill.card_guid("front", "fundamentals") == "fundamentals:front" + assert drill.card_guid("front", None) != drill.card_guid("front", "fundamentals") def test_parse_joins_multiline_body_with_br(drill): diff --git a/claude-templates/.ai/scripts/tests/test_inbox_send.py b/claude-templates/.ai/scripts/tests/test_inbox_send.py index f75d7a1..9b0a8c6 100644 --- a/claude-templates/.ai/scripts/tests/test_inbox_send.py +++ b/claude-templates/.ai/scripts/tests/test_inbox_send.py @@ -476,3 +476,117 @@ class TestFilenameCollisions: assert len(files) == 2 bodies = "".join(f.read_text() for f in files) assert "message one" in bodies and "message two" in bodies + + +class TestAtomicWrite: + """A send wrote straight to the destination path in another project's + inbox/, and write_text truncates on open, so any mid-write failure left a + zero-byte .org there. inbox-status counts that phantom as a pending + handoff, blocking a turn in the receiving project over a file with no + content (2026-07-23). The write must be atomic: the inbox sees a complete + file or nothing.""" + + def test_send_text_writes_utf8(self, tmp_path): + from datetime import datetime + mod = _load_module() + inbox = tmp_path / "inbox" + inbox.mkdir() + now = datetime(2026, 7, 23, 4, 36, 0) + # An em dash and an accented char — both non-ASCII. + dest = mod.send_text(inbox, "accent café and dash — here", "src", None, now) + # Reading as utf-8 must round-trip; a locale-encoded write would raise + # under a C locale, and reading back proves the bytes are utf-8. + assert "—" in dest.read_text(encoding="utf-8") + + def test_send_text_no_partial_on_write_failure(self, tmp_path, monkeypatch): + from datetime import datetime + mod = _load_module() + inbox = tmp_path / "inbox" + inbox.mkdir() + now = datetime(2026, 7, 23, 4, 36, 0) + # Force the atomic finalize to fail after the temp file is written. + def boom(*a, **k): + raise OSError("disk full") + monkeypatch.setattr(mod.os, "replace", boom) + with pytest.raises(OSError): + mod.send_text(inbox, "a message that should never half-land", "src", None, now) + # No phantom, no leftover temp: the inbox is empty. + assert list(inbox.iterdir()) == [] + + def test_send_text_leaves_no_temp_on_success(self, tmp_path): + from datetime import datetime + mod = _load_module() + inbox = tmp_path / "inbox" + inbox.mkdir() + now = datetime(2026, 7, 23, 4, 36, 0) + dest = mod.send_text(inbox, "clean send", "src", None, now) + assert list(inbox.iterdir()) == [dest] + + def test_send_file_no_partial_on_write_failure(self, tmp_path, monkeypatch): + from datetime import datetime + mod = _load_module() + inbox = tmp_path / "inbox" + inbox.mkdir() + src = tmp_path / "note.org" + src.write_text("body") + now = datetime(2026, 7, 23, 4, 36, 0) + def boom(*a, **k): + raise OSError("disk full") + monkeypatch.setattr(mod.os, "replace", boom) + with pytest.raises(OSError): + mod.send_file(inbox, src, "src", None, now) + assert list(inbox.iterdir()) == [] + + def test_send_file_leaves_no_temp_on_success(self, tmp_path): + from datetime import datetime + mod = _load_module() + inbox = tmp_path / "inbox" + inbox.mkdir() + src = tmp_path / "note.org" + src.write_text("payload") + now = datetime(2026, 7, 23, 4, 36, 0) + dest = mod.send_file(inbox, src, "src", None, now) + assert list(inbox.iterdir()) == [dest] + assert dest.read_text() == "payload" + + +class TestSmallerDefects: + """Two low-severity defects found reading inbox-send during the 2026-07-23 + sweep: an unreadable source raised an uncaught traceback instead of the + clean error every other failure path produces, and a roots config naming + both a parent and one of its children listed the same project twice.""" + + def test_unreadable_source_gives_clean_error_not_traceback( + self, project_root, run_script, tmp_path + ): + project_root("sender") + project_root("receiver") + roots = [tmp_path / "projects"] + src = tmp_path / "secret.bin" + src.write_text("x") + src.chmod(0o000) + try: + result = run_script( + ["receiver", "--file", str(src)], + cwd=tmp_path / "projects" / "sender", + roots=roots, + expect_failure=True, + ) + finally: + src.chmod(0o644) + assert result.returncode == 1 + # The clean "inbox-send: <message>" shape, not a Python traceback. + assert result.stderr.startswith("inbox-send:") + assert "Traceback" not in result.stderr + + def test_discover_projects_dedupes_parent_and_child_root(self, tmp_path): + mod = _load_module() + # A project directory, reachable both as a child of its parent root and + # as a root in its own right. + parent = tmp_path / "projects" + proj = parent / "app" + (proj / ".ai").mkdir(parents=True) + (proj / "inbox").mkdir() + found = mod.discover_projects([parent, proj]) + resolved = [p.resolve() for p in found] + assert resolved.count(proj.resolve()) == 1 diff --git a/claude-templates/.ai/scripts/tests/test_route_recommend.py b/claude-templates/.ai/scripts/tests/test_route_recommend.py index acc4755..2ec900a 100644 --- a/claude-templates/.ai/scripts/tests/test_route_recommend.py +++ b/claude-templates/.ai/scripts/tests/test_route_recommend.py @@ -122,3 +122,31 @@ def test_cli_exclude_drops_current_project(tmp_path): r = _run(["--exclude", "foo"], roots=[tmp_path / "projects"], item="fix the foo widget") assert r.returncode == 0 assert r.stdout.strip() == "none" + + +# ---------------------------------------------------------------------- +# Duplicate candidate names +# +# Projects are collapsed to bare basenames, so two projects sharing a basename +# across roots (~/code/notes and ~/projects/notes) appear twice in the candidate +# list. Both literal-match, recommend read len(strong) > 1 as an ambiguous tie, +# and a correct strong match was downgraded to weak. Latent when discovered +# 2026-07-24 (27 projects, 27 distinct basenames) but real. +# ---------------------------------------------------------------------- + +def test_duplicate_candidate_name_keeps_strong_confidence(): + assert rr.recommend("fix the notes thing", ["notes", "other"]) == ("notes", "strong") + # The same name twice must not read as a tie. + assert rr.recommend("fix the notes thing", ["notes", "notes", "other"]) == ("notes", "strong") + + +def test_genuine_ambiguity_still_downgrades(): + # Two DIFFERENT projects both matching is a real tie and stays weak — the + # dedupe must collapse identical names only, never real ambiguity. + dest, conf = rr.recommend("notes and other both", ["notes", "other"]) + assert conf == "weak" + + +def test_duplicates_do_not_change_the_chosen_destination(): + dest, _ = rr.recommend("fix the notes thing", ["notes", "notes"]) + assert dest == "notes" diff --git a/claude-templates/.ai/scripts/todo-cleanup.el b/claude-templates/.ai/scripts/todo-cleanup.el index bd8166d..516e9b1 100644 --- a/claude-templates/.ai/scripts/todo-cleanup.el +++ b/claude-templates/.ai/scripts/todo-cleanup.el @@ -5,6 +5,8 @@ ;; emacs --batch -q -l todo-cleanup.el --check todo.org # hygiene report only ;; emacs --batch -q -l todo-cleanup.el --archive-done todo.org # archive completed subtrees ;; emacs --batch -q -l todo-cleanup.el --archive-done --check todo.org # preview the archive +;; emacs --batch -q -l todo-cleanup.el --seal todo.org # seal the working archive to resolved-YYYY-MM-DD.org +;; emacs --batch -q -l todo-cleanup.el --seal --check todo.org # preview the seal ;; emacs --batch -q -l todo-cleanup.el --convert-subtasks todo.org # dated-rewrite done level-3+ sub-tasks ;; emacs --batch -q -l todo-cleanup.el --convert-subtasks --check todo.org # preview the conversion ;; emacs --batch -q -l todo-cleanup.el --sync-child-priority todo.org # bump children whose priority drifted below the parent's @@ -37,23 +39,37 @@ ;; a message. Only direct level-2 children move — a DONE entry nested under ;; an open parent stays put. ;; -;; 2. Ages the "Resolved" section: a level-2 DONE/CANCELLED subtree whose -;; CLOSED date is older than `tc-archive-retain-days' (default 7) is moved +;; 2. Ages the "Resolved" section: a level-2 DONE/CANCELLED subtree is moved ;; out to `tc-archive-file' (default `archive/task-archive.org' beside the -;; todo file), keeping only the last week of closed tasks in the file -;; itself. Only subtrees closed within the window stay; older ones, and -;; those with no parseable CLOSED date, are moved out. Set -;; `tc-archive-retain-days' to nil to disable this step (legacy in-file-only -;; behavior). The aging date is `tc-archive-reference-date' when set -;; (tests), otherwise the real current date. The archive inherits the todo -;; file's gitignore status: when the todo file is gitignored, the archive -;; path is added to .gitignore before the first write, so private task -;; history never lands in a tracked path (see +;; todo file) when its CLOSED date is older than `tc-archive-retain-days' +;; (default 31 — one month) OR its CLOSED date can't be parsed. The last +;; month of closed tasks stays browsable in the file itself; older ones age +;; out. The unparseable-CLOSED case archives too, deliberately: a +;; keyword-complete task with no readable close date is cruft, not live +;; work. Set `tc-archive-retain-days' to nil to disable this step (legacy +;; in-file-only behavior). The aging date is `tc-archive-reference-date' +;; when set (tests), otherwise the real current date. The archive inherits +;; the todo file's gitignore status: when the todo file is gitignored, the +;; archive path is added to .gitignore before the first write, so private +;; task history never lands in a tracked path (see ;; `tc--ensure-archive-gitignored'). ;; ;; Archiving is consequential, so it's never run by default; it does *not* ;; also run the hygiene passes. ;; +;; * --seal (opt-in). Renames the working archive file (`tc-archive-file', +;; default `archive/task-archive.org') to `resolved-YYYY-MM-DD.org' beside it, +;; dated by the seal run, and leaves the next `--archive-done' to recreate a +;; fresh working file. The dated file means "everything sealed as of that +;; date" — not a calendar quarter — so a task closed late in a quarter and +;; archived after the boundary is never mislabeled; cadence (e.g. quarterly) +;; becomes independent of correctness and any slip is harmless. The sealed +;; file inherits the todo file's gitignore status the same way the working +;; archive does. A no-op (reported) when there's no working archive to seal; +;; refuses to clobber an existing `resolved-<today>.org'. Honors `--check'. +;; The seal date is `tc-archive-reference-date' when set (tests), otherwise the +;; real current date. +;; ;; * --convert-subtasks (opt-in). Rewrites every level-3-and-deeper heading whose ;; TODO state is DONE/CANCELLED/FAILED into a dated event-log entry ;; (`<stars> YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <text>'), dropping the keyword, @@ -84,6 +100,12 @@ ;; --check-child-priority is the report-only alias for --sync-child-priority ;; --check. +;; Before any modification a backup is copied to +;; /tmp/<basename>.before-todo-cleanup.<YYYYMMDD-HHMMSS> +;; matching lint-org.el and wrap-org-table.el. Skipped under --check, which +;; writes nothing. +;; + (require 'org) (require 'cl-lib) (require 'calendar) @@ -102,6 +124,17 @@ sub-task is terminal too and belongs in the parent's dated history.") (defconst tc--priority-cookie-regexp "\\[#\\([A-Z]\\)\\]" "Regexp matching an org priority cookie. Match group 1 is the letter.") +(defconst tc--planning-cookie-regexp + "\\(?:CLOSED\\|DEADLINE\\|SCHEDULED\\):[ \t]*[[<][^]>\n]*[]>]" + "One org planning cookie: a CLOSED/DEADLINE/SCHEDULED keyword followed by a +bracketed (inactive) or angled (active) timestamp.") + +(defconst tc--planning-line-regexp + (concat "\\`[ \t]*\\(?:" tc--planning-cookie-regexp "[ \t]*\\)+\\'") + "A whole org planning line: nothing but planning cookies and whitespace. +Anchored to a single line's contents so a line mixing a cookie with real body +text is never matched.") + (defconst tc-no-sync-tag "no-sync" "Org tag that opts a heading and all its descendants out of `--sync-child-priority'. Inherits down: a tag on an ancestor counts for @@ -112,20 +145,30 @@ every heading below it.") (defvar tc-bumped 0) (defvar tc-converted 0) (defvar tc-issues nil) +(defvar tc-sealed 0) (defvar tc-check-only nil) (defvar tc-archive-done nil) (defvar tc-sync-child-priority nil) (defvar tc-convert-subtasks nil) +(defvar tc-seal nil) (defvar tc-current-file nil) (defvar tc-current-dir nil) (defvar tc-archived-to-file 0) -(defvar tc-archive-retain-days 7 +(defconst tc-archive-retain-days-default 31 + "Default retention window (days) for the `--archive-done' file-aging step — +one month. A closed Resolved subtree stays in-file for this long before it ages +out to `tc-archive-file'; the last month of resolved work stays browsable in the +todo file itself. Named so the \"one month\" contract is explicit and testable.") + +(defvar tc-archive-retain-days tc-archive-retain-days-default "Retention window for the `--archive-done' file-aging step. A closed Resolved subtree whose CLOSED date is within this many days of the reference date stays in the in-file Resolved section; an older one is moved out to `tc-archive-file'. -A subtree with no parseable CLOSED date stays. nil disables the aging step -entirely, leaving the legacy in-file-only behavior.") +A subtree with no parseable CLOSED date is aged out too (a keyword-complete task +with no readable close date is cruft, not live work). nil disables the aging +step entirely, leaving the legacy in-file-only behavior. Defaults to +`tc-archive-retain-days-default' (one month).") (defvar tc-archive-reference-date nil "(YEAR MONTH DAY) treated as \"today\" when aging Resolved subtrees out to a @@ -479,6 +522,51 @@ step. Honors `tc-check-only' (report only)." tc-issues)))))))))) ;;; --------------------------------------------------------------------------- +;;; --seal mode: rename the working archive to a dated resolved-YYYY-MM-DD.org + +(defun tc--seal-date-string () + "YYYY-MM-DD for the seal — `tc-archive-reference-date' when set (tests), +otherwise the real current date." + (if tc-archive-reference-date + (pcase-let ((`(,y ,m ,d) tc-archive-reference-date)) + (format "%04d-%02d-%02d" y m d)) + (format-time-string "%Y-%m-%d"))) + +(defun tc-seal-archive-file () + "Rename the working archive file to `resolved-YYYY-MM-DD.org' beside it. +The next `--archive-done' run recreates a fresh working file. No-op (reported) +when there is no working archive to seal; refuses to clobber an existing +`resolved-<today>.org'. Ensures the sealed file inherits the todo file's +gitignore status. Honors `tc-check-only'." + (let ((path (tc--archive-file-path))) + (cond + ((or (null path) (not (file-readable-p path))) + (push (list :kind 'seal-nothing :file tc-current-file) tc-issues)) + (t + (let* ((dir (file-name-directory path)) + (sealed (expand-file-name + (format "resolved-%s.org" (tc--seal-date-string)) dir))) + (cond + ((file-exists-p sealed) + (push (list :kind 'seal-collision :file tc-current-file + :detail (file-name-nondirectory sealed)) + tc-issues)) + (tc-check-only + (cl-incf tc-sealed) + (push (list :kind 'seal-would :file tc-current-file + :detail (file-name-nondirectory sealed)) + tc-issues)) + (t + ;; Ignore the sealed name before the rename so its history stays as + ;; private as the working archive it derives from. + (tc--ensure-archive-gitignored sealed) + (rename-file path sealed) + (cl-incf tc-sealed) + (push (list :kind 'seal-done :file tc-current-file + :detail (file-name-nondirectory sealed)) + tc-issues)))))))) + +;;; --------------------------------------------------------------------------- ;;; --sync-child-priority mode (defun tc--heading-priority-letter () @@ -617,6 +705,14 @@ before their descendants — a [#A] → [#B] → [#D] chain collapses in one pas ;; as written). Idempotent: an already-dated heading has no done keyword, so it ;; is skipped. A done sub-task with no parseable CLOSED cookie can't be dated, so ;; it is flagged and left alone rather than stamped with a fabricated date. +;; +;; The planning line goes entirely. A dated-log entry carries its date in the +;; heading, so CLOSED is redundant and an active DEADLINE/SCHEDULED is wrong: org +;; renders any headline with an active planning timestamp — keyword or not — so a +;; SCHEDULED left on a dated-log heading pins it to the agenda as weeks-overdue +;; long after the work is done. The conversion deletes the whole planning line, +;; not just the CLOSED cookie (todo-format.md; lint checker +;; `dated-log-heading-active-timestamp' backstops any that slip through). (defun tc--closed-parts-in-entry () "Return a plist (:year :month :day :dow :hour :minute) from the CLOSED cookie @@ -676,6 +772,27 @@ in-progress `org-map-entries' walk; markers track their headings across edits." nil 'file) (nreverse targets))) +(defun tc--strip-planning-lines-in-entry () + "Delete the canonical planning line(s) directly under the heading at point. +A planning line is one composed solely of CLOSED/DEADLINE/SCHEDULED cookies and +whitespace. Walks the lines immediately after the heading and stops at the first +non-planning line, so a planning-shaped line deeper in the body (e.g. in a code +block) is never touched. Returns the count of lines removed." + (save-excursion + (org-back-to-heading t) + (forward-line 1) + (let ((removed 0) (continue t)) + (while (and continue (not (eobp))) + (let ((line (buffer-substring-no-properties + (line-beginning-position) (line-end-position)))) + (if (string-match-p tc--planning-line-regexp line) + (progn + (delete-region (line-beginning-position) + (min (1+ (line-end-position)) (point-max))) + (cl-incf removed)) + (setq continue nil)))) + removed))) + (defun tc--convert-one-subtask (marker) "Convert the done sub-task heading at MARKER to a dated event-log entry. Under `tc-check-only' the conversion is reported but not performed." @@ -698,27 +815,13 @@ Under `tc-check-only' the conversion is reported but not performed." (push (list :kind 'convert-would :file tc-current-file :line line :heading title :new new) tc-issues) - ;; Replace the heading line, then drop the now-redundant CLOSED - ;; cookie from the entry (its date now lives in the header). Only - ;; the cookie goes: a planning line can also carry DEADLINE: or - ;; SCHEDULED: beside it, and those survive on their line. A line - ;; left blank by the removal is deleted whole. + ;; Replace the heading line, then drop the whole planning line. The + ;; date now lives in the header, so CLOSED is redundant and an active + ;; DEADLINE/SCHEDULED would wrongly pin this completed entry to the + ;; agenda (todo-format.md). Both go, not just the CLOSED cookie. (delete-region (line-beginning-position) (line-end-position)) (insert new) - (let ((end (save-excursion - (or (outline-next-heading) (goto-char (point-max))) - (point)))) - (save-excursion - (when (re-search-forward "CLOSED:[ \t]*\\[[^]]*\\][ \t]*" end t) - (replace-match "") - (let ((bol (line-beginning-position)) - (eol (line-end-position))) - (if (string-match-p "\\`[ \t]*\\'" - (buffer-substring bol eol)) - (delete-region bol (min (1+ eol) (point-max))) - (goto-char bol) - (when (looking-at "[ \t]+") - (replace-match ""))))))) + (tc--strip-planning-lines-in-entry) (push (list :kind 'convert-done :file tc-current-file :line line :heading title :new new) tc-issues))))))) @@ -735,9 +838,37 @@ event-log entry, pulling the timestamp from its CLOSED cookie. Honors ;;; --------------------------------------------------------------------------- ;;; Driver + reporting +(defun tc--backup (file) + "Copy FILE to /tmp before any modification. Skipped in --check mode. + +Matches `lint-org.el' and `wrap-org-table.el', the other tools that rewrite +these org files. todo-cleanup runs the most often of the three (every wrap, +every sentry cycle), and Emacs's own backup does not fire under --batch -q, so +without this a mechanical rewrite has no undo short of git — which recovers +only to the last commit and loses intra-session work." + (let* ((base (format "%s%s.before-todo-cleanup.%s" + temporary-file-directory + (file-name-nondirectory file) + (format-time-string "%Y%m%d-%H%M%S"))) + (backup base) + (n 2)) + ;; Never overwrite an earlier backup. A second-resolution stamp collides + ;; when two invocations run back to back, which the shipped workflow does + ;; (open-tasks.org runs --convert-subtasks then --archive-done, each a + ;; sub-second batch run). Overwriting there replaces the true pre-session + ;; original with already-mutated content — losing exactly what the backup + ;; exists to preserve. Suffix instead, so every invocation keeps its own. + (while (file-exists-p backup) + (setq backup (format "%s-%d" base n)) + (setq n (1+ n))) + (copy-file file backup nil) + backup)) + (defun tc-process-file (file) (setq tc-current-file (file-name-nondirectory file)) (setq tc-current-dir (file-name-directory (expand-file-name file))) + (unless tc-check-only + (tc--backup file)) (with-current-buffer (find-file-noselect file) (org-mode) (cond @@ -747,6 +878,8 @@ event-log entry, pulling the timestamp from its CLOSED cookie. Honors (tc-sync-child-priority-in-file)) (tc-convert-subtasks (tc-convert-subtasks-in-file)) + (tc-seal + (tc-seal-archive-file)) (t ;; Pass 1: auto-fix bogus state logs (or report under --check). (org-map-entries #'tc-fix-bogus-state-log-in-entry nil 'file) @@ -865,10 +998,26 @@ event-log entry, pulling the timestamp from its CLOSED cookie. Honors (plist-get i :file) (plist-get i :line) (plist-get i :heading) (plist-get i :detail))))))))) +(defun tc--emit-seal-report () + (dolist (i (reverse tc-issues)) + (pcase (plist-get i :kind) + ('seal-done + (princ (format "todo-cleanup --seal: sealed task-archive.org → %s\n" + (plist-get i :detail)))) + ('seal-would + (princ (format "todo-cleanup --seal: would seal task-archive.org → %s — CHECK MODE (no writes)\n" + (plist-get i :detail)))) + ('seal-collision + (princ (format "todo-cleanup --seal: %s already exists — not sealing (already sealed today?)\n" + (plist-get i :detail)))) + ('seal-nothing + (princ "todo-cleanup --seal: no working archive to seal\n"))))) + (defun tc-emit-report () (cond (tc-archive-done (tc--emit-archive-report)) (tc-sync-child-priority (tc--emit-sync-report)) (tc-convert-subtasks (tc--emit-convert-report)) + (tc-seal (tc--emit-seal-report)) (t (tc--emit-hygiene-report)))) (defun tc-main () @@ -886,6 +1035,9 @@ event-log entry, pulling the timestamp from its CLOSED cookie. Honors (when (member "--convert-subtasks" command-line-args-left) (setq tc-convert-subtasks t) (setq command-line-args-left (delete "--convert-subtasks" command-line-args-left))) + (when (member "--seal" command-line-args-left) + (setq tc-seal t) + (setq command-line-args-left (delete "--seal" command-line-args-left))) ;; --check-child-priority is the report-only alias for ;; `--sync-child-priority --check'. (when (member "--check-child-priority" command-line-args-left) @@ -893,7 +1045,7 @@ event-log entry, pulling the timestamp from its CLOSED cookie. Honors (setq command-line-args-left (delete "--check-child-priority" command-line-args-left))) (if (null command-line-args-left) (progn - (princ "Usage: emacs --batch -q -l todo-cleanup.el [--check] [--archive-done | --convert-subtasks | --sync-child-priority | --check-child-priority] FILE...\n") + (princ "Usage: emacs --batch -q -l todo-cleanup.el [--check] [--archive-done | --seal | --convert-subtasks | --sync-child-priority | --check-child-priority] FILE...\n") (kill-emacs 1)) (let ((files command-line-args-left)) (setq command-line-args-left nil) @@ -912,6 +1064,7 @@ ert-run-tests-batch-and-exit'." (cl-every (lambda (a) (cond ((member a '("--check" "--archive-done" + "--seal" "--convert-subtasks" "--sync-child-priority" "--check-child-priority")) diff --git a/claude-templates/.ai/workflows/INDEX.org b/claude-templates/.ai/workflows/INDEX.org index b031dbe..f18d953 100644 --- a/claude-templates/.ai/workflows/INDEX.org +++ b/claude-templates/.ai/workflows/INDEX.org @@ -58,6 +58,9 @@ This index must list every =.org= file in =.ai/workflows/= except this one and e - Speedrun triggers: "speedrun", "no approvals speedrun", "speedrun these: <task set>" — any phrase containing "speedrun" routes here (the preset), never to =no-approvals.org= - Manual triggers: "work the backlog", "work the backlog with <task set>" (file-only defaults) - Synthesis trigger: "synthesize backlog metrics" — read the per-project metrics logs, compute trends + the corrections signal, write one =:agent:metrics:= KB node (personal projects only) +- =sentry.org= — the overnight hygiene supervisor: an interval loop (default hourly) that walks a fixed pass list (roam pull, inbox zero, triage, todo cleanup, task audit, working-files hygiene, spec board, link integrity, git health, prep freshness), commits each writing pass to a throwaway =sentry/<date>-<host>= branch (never pushed), and parks every judgment call and destructive action in a morning-approval queue. Gated on =:COMMIT_AUTONOMY: yes= plus interactive entry gates (clean tree, green suite) with Craig present. Locks via =agent-lock=; morning teardown (review, squash-merge, delete) is Craig's, never automated. + - Triggers: "start sentry", "run sentry", "arm sentry", "sentry mode", "start sentry every <interval>" + - Stop trigger: "stop sentry", "stand down sentry", "sentry off" ** Calendar @@ -110,7 +113,7 @@ This index must list every =.org= file in =.ai/workflows/= except this one and e - Situational triggers: "broadcast the <event> to all projects", "broadcast that <situation>", "let every project know I'll be away ..." - =flashcard-review.org= — review an org-drill flashcard file, restructure cards to question-form headings (no answer hints), audit content accuracy against project source-of-truth via subagent, rewrite source preserving SRS state, regenerate the Anki =.apkg= to =~/sync/phone/anki/=. Person cards use "Who is X? Tell me about their Y."; talking-points cards stay as-is. Script behavior: =flashcard-to-anki.py= strips =:PROPERTIES:= drawers + =SCHEDULED:= / =DEADLINE:= planning lines from Anki output. - Triggers: "review the flashcards", "update the flashcards", "review the drill deck", "update the drill deck", "refresh the Anki cards", "let's run the flashcard-review workflow" -- =page-me.org= — set a timed notification (desktop =notify=; phone via =agent-page= when Craig is away). +- =page-me.org= — set a timed notification. "page me" desktop =notify=, "text me" phone via =agent-text=, "text and page me" both. - Triggers: anything containing the word "page" used as a verb ("page me", "page me in 10 minutes", "page me at 3pm", "page my phone") - =status-check.org= — proactive long-running-job updates. - Triggers: "keep me posted on this", "provide status checks on this job", "let me know when it's done", "monitor this for me". Auto: any job estimated 10+ min. diff --git a/claude-templates/.ai/workflows/code-quality.org b/claude-templates/.ai/workflows/code-quality.org index 3ac3e9d..3c4ed8f 100644 --- a/claude-templates/.ai/workflows/code-quality.org +++ b/claude-templates/.ai/workflows/code-quality.org @@ -9,6 +9,13 @@ One trigger that runs every behavior-preserving quality pass over a scope of orchestrator — each pass keeps its own discipline and its own confirm gate; this workflow only sequences them and collects the residue. +*Behavior-preserving rests on a test net.* The passes below claim to preserve +behavior, but a refactor on untested code is a guess, not a preservation. Where +the scope has no tests, bring it under a characterization net first +(Normal/Boundary/Error per unit, per the =testing-standards= skill's "Adding Tests to Existing +Untested Code") — that net is what turns "behavior-preserving" from an assertion +into something the green suite actually verifies across each pass. + The passes it chains: 1. =/refactor= — structural and logic cleanup on measurable metrics (complexity, diff --git a/claude-templates/.ai/workflows/helper-mode.org b/claude-templates/.ai/workflows/helper-mode.org index a6acfa7..b32d574 100644 --- a/claude-templates/.ai/workflows/helper-mode.org +++ b/claude-templates/.ai/workflows/helper-mode.org @@ -12,13 +12,14 @@ The governing fact behind every rule below: the session-context split isolates e * When to Use This Workflow -No operator trigger phrase. A helper reaches this contract one of three ways: +No operator trigger phrase. A helper reaches this contract one of two ways: - The =ai --helper= launcher routes here after the roster confirms a live agent (the deterministic path). -- Startup's roster check finds the session is not alone and routes here instead of running normal startup (the safety net for a raw =claude= launch). - An explicit "you are a helper, follow helper-mode.org" instruction (the manual fallback). -If none of those applies — the roster shows the session is alone — this is a primary session. Run normal [[file:startup.org][startup.org]], not this. +There is deliberately no third way, and the gap matters: *startup does not check the roster*. A bare =claude= launched into a project that already has a live session runs full primary startup — pulls, rsync, inbox processing — without ever reaching this file. That safety net is designed (see Status below) but unbuilt, so nothing catches a raw launch. Use =ai --helper=. + +If neither route applies, this is a primary session. Run normal [[file:startup.org][startup.org]], not this. * Identity @@ -92,10 +93,28 @@ A helper does not run normal startup. It runs a light version: When the helper's work is done: -1. Re-run the roster (=.ai/scripts/agent-roster=) to learn whether a primary is still live. +1. Re-run the roster to learn whether a primary is still live. Pass the project root explicitly — =agent-roster= defaults to =$PWD= and keeps only agents at or inside that root, so calling it from a subdirectory hides a primary sitting at the root and reports "alone": + + #+begin_src bash + root="$(git rev-parse --show-toplevel 2>/dev/null || pwd)" + if [ -x "$root/.ai/scripts/agent-roster" ]; then + "$root/.ai/scripts/agent-roster" "$root"; rc=$? + else + rc=2 + fi + echo "roster rc=$rc" + #+end_src + + Read rc as =wrap-it-up.org= Step 0 does: 1 means a primary is still live, 0 means this helper is orphaned, and 2 (or an absent script) means unavailable — which takes the same archive-only path as 1, because leaving work uncommitted is recoverable and committing under a live primary is not. 2. *Primary still live (the normal case):* finalize the Summary in the helper's own =.ai/session-context.d/<id>.org=, archive it to =.ai/sessions/YYYY-MM-DD-HH-MM-<id>-<description>.org=, and stop. Do NOT commit, push, or run hygiene — the primary's next commit picks up the archived file and any scoped edits the helper left in the tree. 3. *Orphaned helper (roster shows the helper is now alone):* the primary already exited, so the helper assumes full closing duties — the git ban lifts because the concurrency that justified it is gone. Commit and push the tree (including the helper's own edits, which would otherwise strand as a dirty tree), per the normal wrap-up flow in [[file:wrap-it-up.org][wrap-it-up.org]]. * Status -Phase 1.5 of the generic-agent-runtime spec. This contract is the canonical home; the spawn paths (=ai --helper=, startup's roster branch) and the [[file:wrap-it-up.org][wrap-it-up.org]] helper branch route here. Those wiring pieces ship behind the spec's bats-then-drills-then-pilot gate and are not yet live; until then, the manual "you are a helper" instruction is how a session adopts this contract. +Phase 1.5 of the generic-agent-runtime spec. This contract is the canonical home; the spawn paths and the [[file:wrap-it-up.org][wrap-it-up.org]] helper branch route here. + +Live now: =ai --helper <project>= (roster check, id assignment, helper opener, its own tmux window), the explicit "you are a helper" instruction, and the wrap-it-up.org Step 0 helper branch. + +Not built yet, and worth knowing because it is the gap you can fall into: *startup has no roster check*. A second session launched as a bare =claude= in a project that already has one runs full primary startup — pulls, rsync, inbox processing — with no idea another agent is live. Until that safety net exists, =ai --helper= is not merely the preferred path, it is the only one that makes a helper without being told. + +Also unbuilt: the live-helper gate that pauses a primary's file-wide hygiene passes (=todo-cleanup.el=, =lint-org.el=, =wrap-org-table.el=) while a helper is mid-edit. Data-integrity rule 1 above describes the intended behavior; nothing enforces it yet, so a primary running hygiene can still clobber a helper's just-written scoped edit. diff --git a/claude-templates/.ai/workflows/inbox.org b/claude-templates/.ai/workflows/inbox.org index 3bd9335..6faa20f 100644 --- a/claude-templates/.ai/workflows/inbox.org +++ b/claude-templates/.ai/workflows/inbox.org @@ -163,6 +163,20 @@ An org capture is usually only a few seconds of mid-finalize state, so =--wait= - *Auto inbox zero (=/loop=) cycle* → don't surface or wait further; defer the roam reconcile to the next cycle, which is itself the retry at loop cadence. The items were already filed in Phase C, so the next cycle's Phase C status-check drops the duplicates and its Phase D removes them. Note one line: "roam reconcile deferred — a capture is still open; next cycle catches it." - *Wrap-up sub-step* → don't block the wrap. Skip the roam reconcile for this run and surface one line: "Skipped roam-inbox reconcile — a live org-capture is open against it; claimed items stay and get caught next run." The items were already filed into =todo.org= in roam mode Phase C, so the next roam run's Phase C status-check drops the duplicates and its Phase D removes them — the skip self-heals. +*The roam-write lock (around the Phase D edit).* Capture-guard protects against a live *human* capture; the roam-write lock protects against a concurrent *agent* writer (a sentry inbox pass, a KB promotion) editing =~/org/roam= at the same time. Acquire it after the capture-guard clears and release it after the edit-and-trigger, so the two guards nest — capture-guard underneath, the agent lock around the write: + +#+begin_src bash +if [ -x .ai/scripts/agent-lock ]; then + .ai/scripts/agent-lock acquire roam-write --wait || { echo "roam-write busy; deferring roam reconcile" >&2; exit 1; } +fi +# capture-guard (above), then the Phase D read-modify-write of ~/org/roam/inbox.org, +# then trigger the sync — roam-sync stays the only committer: +systemctl --user start roam-sync.service +[ -x .ai/scripts/agent-lock ] && .ai/scripts/agent-lock release roam-write +#+end_src + +Degrade gracefully when =agent-lock= isn't installed (an older checkout mid-sync): the write proceeds unlocked, today's behavior. A *present* helper reporting the lock busy after its bounded wait defers the roam reconcile (the auto-loop and wrap-up paths already defer-and-retry per the fallback list above); an *absent* helper never blocks it. + * Core §6 — Priority-scheme check This gates filing whenever there are accept-and-file items. Check whether =todo.org= has a top-of-file priority scheme (an explicit legend defining =[#A]= through =[#D]= semantics and mandatory/optional tag conventions — a =* <Project> Priority Scheme= section or similar). @@ -332,7 +346,7 @@ When Craig has put the session in no-approvals mode, an accepted item may be imp 2. *Quick* — the whole implementation, including verification, is under ~15 minutes. 3. *Solo* — you can carry it end to end without a decision from Craig. Manual verification you perform yourself is fine; needing Craig to choose an option, approve a design, or resolve an ambiguity is not. -All three → implement it, verify, then commit and push at the end of that item (the Step 0 reconcile and pre-push check from =commits.md= still run). Miss any one and it doesn't self-apply: a shared-asset or convention change needs Craig's decision, so it fails *solo* and routes to the defer-and-stage park (core §2 / core §3); an oversized item fails *quick* and gets filed. +All three → implement it, verify, then commit and push at the end of that item (the Step 0 reconcile and pre-push check from the =publish= skill still run). Miss any one and it doesn't self-apply: a shared-asset or convention change needs Craig's decision, so it fails *solo* and routes to the defer-and-stage park (core §2 / core §3); an oversized item fails *quick* and gets filed. ** Replying to handoffs @@ -355,6 +369,8 @@ If either can't be satisfied — a half-done item, a failure introduced during t Reads the *global roam inbox* (=~/org/roam/inbox.org=), Craig's cross-project GTD capture: one shared file every project can see. This mode routes each roam item to the project that owns it. The current session claims only the items belonging to THIS project, files them into the project's =todo.org=, and removes them from the shared inbox. Everything it doesn't own stays. +*Allowed from any project, work included.* Tidying the shared roam inbox is housekeeping on a shared resource, not a cross-project boundary crossing and not a durable KB-node write, so the =knowledge-base.md= work-denylist doesn't gate it (a sentry inbox-zero pass mis-parked the whole inbox as a boundary crossing from the work project on 2026-07-19 — the error this note closes). Reading roam and tidying its inbox are fine from work; only promoting a durable =agents/= node stays work-denylisted. + The aspiration is inbox zero: after this mode runs, the current project's local handoff inbox has been processed (Phase A delegates to process mode) and the shared roam inbox no longer contains items explicitly owned by this project. This is distinct from the wrap-up inbox/transcript routing feature (which moves session-filed keepers between projects). This routes the shared roam capture file by ownership prefix. @@ -466,14 +482,14 @@ A recurring, *interactive* roam check. Trigger phrase: "auto inbox zero" (match ** Per cycle 1. Run roam mode's scan (Phase A local check + Phase B roam scan), read-only — no =git pull=. The capture-guard still gates any write: use =capture-guard --wait= (core §5) so a transient capture clears itself; if it's still open after the wait, *defer this cycle's roam reconcile to the next cycle* rather than surfacing — the loop cadence is the retry, and the filed items get swept next time. The rare write hands its git to =roam-sync= (roam Phase D). -2. *Nothing found* → no inbox summary. One acknowledgement line: =ran at HH:MM, nothing found=. Nothing else. The acknowledge-only-on-empty rule keeps a quiet inbox quiet. +2. *Nothing found* → no inbox summary. One heartbeat line: =inbox zero at HH:MM: nothing= (HH:MM local, from =date=) — the silent-until-signal policy, see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=. Nothing else. Keeping a quiet inbox quiet is the whole point. 3. *Items found* → summarize the found items, file them as tasks (roam Phase C), and *append them to a displayed queue* — the harness task list, via =TaskCreate= — so the queue accumulates across cycles. Then ask: "run this batch next?" - *Yes* → chain into =work-the-backlog.org= as an explicit second step after routing completes: pass it the eligibility query over the queued items (status =TODO= + =:solo:= per the scheme header, priority-ordered), =file-only= mode, paging off, cap 1. The highest-priority eligible candidate runs; the rest wait for the next tick or a later yes. - *No* → they stay queued for a later go. This mode never implements anything itself — routing ends here, and the execution loop lives in =work-the-backlog.org=, its one home. 4. *Cross-cycle dedup.* Subsequent cycles add only *newly-found* items to the same displayed queue, never re-surfacing what's already there. Dedup against the queue (the =TaskCreate= list), not against what's already been implemented — a find that was queued-but-not-yet-run must not reappear, and one already filed into =todo.org= is dropped by roam Phase C's status check. -A find is always surfaced and filed; execution happens only through the =work-the-backlog.org= chain and waits for Craig's yes. A quiet inbox produces only the timestamped acknowledgement. =auto inbox zero= is inherently in-session because its chain step waits for that yes. +A find is always surfaced and filed; execution happens only through the =work-the-backlog.org= chain and waits for Craig's yes. A quiet inbox produces only the =inbox zero at HH:MM: nothing= heartbeat. =auto inbox zero= is inherently in-session because its chain step waits for that yes. ** Fully-unattended pass (=/schedule=) — vNext, not v1 diff --git a/claude-templates/.ai/workflows/no-approvals.org b/claude-templates/.ai/workflows/no-approvals.org index 5f54b96..6b5c7fa 100644 --- a/claude-templates/.ai/workflows/no-approvals.org +++ b/claude-templates/.ai/workflows/no-approvals.org @@ -35,7 +35,7 @@ Mode resets when: The interaction gates that step the workflow back to Craig for an "OK to proceed?" check: -- The commit-message gate in =commits.md= Step 2. Print the final commit message inline, then commit immediately. No "approve / request changes / open in editor" prompt. +- The commit-message gate in the =publish= skill, Step 2. Print the final commit message inline, then commit immediately. No "approve / request changes / open in editor" prompt. - The PR-description gate. Print the final body, then create the PR. - The PR-review-reply gate. Print the final reply, then post. - "Ready to start?" / "Plan looks like X, proceed?" gates before implementation work begins. @@ -46,10 +46,10 @@ The interaction gates that step the workflow back to Craig for an "OK to proceed The engineering-discipline gates protect quality, not Craig's interaction time. They remain in force: -- =/review-code= against the staged diff before every commit. Critical and Important findings still block. Minor findings still surface. No "proceed anyway" override unless Craig has given it explicitly for this batch. +- =/review-code= against the staged diff before every commit, dispatched as an isolated adversarial reviewer per the =publish= skill's Step 1 — no-approvals removes *interaction* gates, never the isolation. Critical and Important findings still block, and the re-review loop still runs to approval. Minor findings still surface. No "proceed anyway" override unless Craig has given it explicitly for this batch. If the review can't reach approval — three rounds, a recurring finding, or a =Needs Discussion= verdict — that is a genuine question: park the item per step 4 and move to the next one rather than committing past a standing finding. - =/voice personal= on every publish artifact (commit messages, PR titles + bodies, PR review comments). The full pattern walk happens. The printed result just doesn't wait for approval. - The full test suite + lint + compile before commit (per =verification.md=). -- Fetch-and-reconcile in =commits.md= Step 0. +- Fetch-and-reconcile in the =publish= skill, Step 0. - Session Log updates per =protocols.org=. Every state-mutating turn writes to =.ai/session-context.org= before the closing message. The log is the crash-recovery anchor while Craig is away. Missing entries lose work. - Subagent review-gate cadence (=subagents.md=). Review each subagent's output before the next dispatch. - Destructive or irreversible operations per =CLAUDE.md='s "Executing actions with care": force-push, =rm -rf=, dropping a column, dropping a branch, package removal. These need explicit consent regardless of mode. No-approvals is for *interaction* gates, not destructive-action consent. @@ -70,7 +70,7 @@ For each item: - Do the work. - Update the Session Log per the rules in =protocols.org=. -- Before any commit: run =/review-code= against the staged diff. Surface Critical and Important findings inline; fix them and re-review until clean. Minor findings show but don't block. +- Before any commit: dispatch the isolated adversarial reviewer per the =publish= skill's Step 1 — never review your own staged diff inline. Surface Critical and Important findings; fix them and send the updated diff back to the *same* reviewer until it approves. Minor findings show but don't block and never earn another round. If the review can't reach approval — three rounds, a finding that recurs after being reported fixed, or a =Needs Discussion= verdict — park the item per step 4 with the standing findings and move on; don't commit past a blocking finding. - Draft the commit message. Run =/voice personal= (the skill, or walk the patterns inline if unavailable). Print the final message inline before committing so the log shows it. - Commit and push. - One-line status between items ("Task X done, on to Y.") so Craig knows what's happening when he checks back in. diff --git a/claude-templates/.ai/workflows/page-me.org b/claude-templates/.ai/workflows/page-me.org index bfa92c6..7a3b792 100644 --- a/claude-templates/.ai/workflows/page-me.org +++ b/claude-templates/.ai/workflows/page-me.org @@ -13,9 +13,17 @@ Uses the =notify= command (info type) for consistent notifications across all AI Craig says *"page me"* (or variations like "page me in 10 minutes", "page me at 3pm"). -The word "page" is the trigger for this workflow. It means: set a timed notification. +The word "page" is the trigger for this workflow. It means: set a timed notification on the *desktop* channel (=notify=). -Previously called "set-alarm" -- renamed to "page-me" for a distinctive, short trigger phrase that won't collide with common words like "remind" or "alert." +Two sibling triggers pick a different channel; the timed =at= machinery below is identical for all three, only the fired command changes: + +- *"page me"* — desktop =notify= (this workflow's default). +- *"text me"* — a Signal push to Craig's phone via =agent-text= (the away channel). +- *"text and page me"* — both, for when he might be either place. + +Scope the triggers to the reflexive "me": "page me" and "text me", not a bare "page" or "text" in prose. The full channel vocabulary lives in protocols.org "Reaching Craig". + +"page" was chosen (renamed from the old "set-alarm") for a distinctive, short trigger that won't collide with common words like "remind" or "alert". * Problem We're Solving @@ -113,18 +121,18 @@ notify info "Page" "Your message here" --persist The =--persist= flag keeps the notification on screen until manually dismissed. All page-me notifications should use =--persist= by default. -** Paging Craig's phone (away from the machine) +** Texting Craig's phone (the "text me" channel) -The timed =notify= alarm above is the desktop channel. When Craig is away from the machine (or asks to be paged "on my phone"), use the agent pager instead — a Signal push to his phone from any machine or agent runtime: +The timed =notify= alarm above is the desktop channel. When Craig says "text me" (or a run expects him away from the machine), use =agent-text= instead, a Signal push to his phone from any machine or agent runtime: #+begin_src bash -agent-page "Build finished — ready for your eyes" +agent-text "Build finished, ready for your eyes" -# Timed phone page: same at-daemon pattern, different channel -echo "agent-page 'Meeting starts in 5'" | at 3:25pm +# Timed phone message: same at-daemon pattern, different channel +echo "agent-text 'Meeting starts in 5'" | at 3:25pm #+end_src -Channel selection and the pager's mechanics live in protocols.org "Paging Craig — the agent pager". When in doubt, fire both: the desktop notification persists for whenever he returns, the phone push reaches him now. +Channel selection and the mechanics live in protocols.org "Reaching Craig". On "text and page me", fire both: the desktop notification persists for whenever he returns, the phone push reaches him now. ** Managing Alarms diff --git a/claude-templates/.ai/workflows/sentry.org b/claude-templates/.ai/workflows/sentry.org new file mode 100644 index 0000000..b25fc14 --- /dev/null +++ b/claude-templates/.ai/workflows/sentry.org @@ -0,0 +1,227 @@ +#+TITLE: Sentry — Overnight Hygiene Supervisor +#+AUTHOR: Craig Jennings +#+DATE: 2026-07-19 + +* Overview + +Sentry is an interval loop that keeps a project's hygiene current while Craig is away. Each cycle walks a fixed list of passes — roam pull, inbox zero, triage (no mail or messengers), todo cleanup, task audit, working-files hygiene, spec board, link integrity, git health, prep freshness, bug and refactor finding, and (opt-in) solo-task implementation — and commits each pass's writing to a throwaway daily branch. Nothing pushes. In the morning Craig reviews the branch, squash-merges what he wants, and deletes it. + +The design goal is a project that greets the morning already tidy, with every judgment call and every destructive action parked in an approval queue rather than executed unattended. Sentry does the mechanical sweeping; Craig does the deciding. + +This file is the engine. It owns the entry gates, the branch mechanics, the lock model, the per-cycle pass runner, the digest and approval queue, the skip semantics, and the stop-sentry shutdown. The =agent-lock= helper (=.ai/scripts/agent-lock=) provides the locks. The passes reuse existing workflows (=inbox.org=, =triage-intake.org=, =clean-todo.org=, =task-audit.org=) under sentry's unattended contract. + +* When to Use This Workflow + +Craig arms sentry at the end of a session, with the machine left running, to have overnight hygiene done by morning. + +Triggers: + +- "start sentry", "run sentry", "arm sentry", "sentry mode" +- "let sentry watch this overnight", "keep this tidy overnight" +- "start sentry hourly", "start sentry every <interval>" (sets the loop interval) + +Stop trigger (see Stop Sentry below): + +- "stop sentry", "stand down sentry", "sentry off" + +Sentry is deliberately *not* auto-armed. Running it in a project is a per-project grant (the =:COMMIT_AUTONOMY:= marker) plus a deliberate launch with Craig at the terminal for the entry gates. + +* Prerequisite — the autonomy ticket + +Sentry commits unattended. =commits.md= gates commits on Craig's approval, so sentry needs standing, per-project authorization to run at all. Before anything else, read the project's =.ai/notes.org= Workflow State block for: + +: :COMMIT_AUTONOMY: yes + +If the marker is absent or not =yes=, decline to start and name the marker: + +: Sentry needs ":COMMIT_AUTONOMY: yes" in .ai/notes.org Workflow State to run — it commits unattended. Add it to grant, or run the hygiene passes by hand. + +No half-running mode: a project without the grant doesn't run sentry's read-only passes either. The grant is one line away, so this is a deliberate opt-in, not a barrier. + +A second, *independent* marker gates the solo-task implementation pass (pass 12): + +: :SENTRY_MAY_IMPLEMENT: yes + +=:COMMIT_AUTONOMY:= lets sentry commit its hygiene sweeps to the branch; =:SENTRY_MAY_IMPLEMENT:= additionally lets it implement solo, decision-free backlog tasks on the branch. The split exists because the two carry different morning costs: hygiene is a two-minute merge, implemented code is a review session. A project can run hygiene-only sentry without the implement pass, and most should until sentry has quiet weeks behind it. Absent =:SENTRY_MAY_IMPLEMENT:=, pass 12 skips; sentry still runs every other pass. Requires =:COMMIT_AUTONOMY:= alongside it — implementing implies committing. + +* Entry — interactive, with Craig present + +Craig types the sentry trigger, so the first moves run with him at the terminal. Do them in order; each gate that fails stops entry until Craig answers. + +1. *Autonomy ticket* — the prerequisite above. Absent → decline and stop. + +2. *Dirty-tree gate.* =git diff --quiet HEAD= (tracked modifications only; untracked and gitignored files never block — an inbox drop or scratch file is not in-progress work). If the tracked tree is dirty, describe what's dirty and offer, inline-numbered per =interaction.md=: + + 1. Finish the job — commit the in-progress work first (recommended if it's a coherent unit) + 2. Stash it — =git stash= and start sentry on a clean tree + 3. Roll back named changes — discard specific files (names them) + + Wait for an answer. Sentry can't start unattended from a dirty state; that's the point. + +3. *Green-suite gate.* Run the project's full suite (=make test=, or the project's equivalent — detect it). Read the output. If anything is red, describe the failures and offer to investigate before arming. The loop starts only on a green baseline, because every unattended cycle measures itself against "did I break this?" and a pre-existing red poisons that check. + +4. *Prior sentry branch.* =git branch --list 'sentry/*'=. An unmerged =sentry/*= branch from a previous night means the morning review didn't happen. Surface it and offer to squash-merge or delete it now (Craig is present); don't stack a second sentry branch on the first. + +5. *Reconcile the project branch.* Fetch and fast-forward-only against upstream — the same reconcile =startup= runs: + + : git fetch --all --prune + : git rev-list --left-right --count @{u}...HEAD + + Zero-behind → continue. Behind-only and clean → =git merge --ff-only @{u}=. Diverged → surface to Craig (he's present); don't auto-resolve. + +6. *Create the daily branch.* From HEAD: + + : git switch -c "sentry/$(date +%F)-$(uname -n)" + + The host suffix (=uname -n=) stops a same-date collision between the two daily drivers. The working tree now sits on this branch overnight — the launch hands the repo to sentry until the morning merge. Reclaiming it mid-night means stopping sentry first (see Stop Sentry). Note the Emacs buffer-revert caveat to Craig if he has the repo open: files change on disk under him overnight, so buffers want reverting after the morning merge (see =emacs.md=). + +7. *Arm the loop.* Start =/loop= at the interval (default hourly; Craig's "every <interval>" phrase overrides) with the per-cycle body being one sentry cycle (the Pass Runner below). Confirm the arming in one line: interval, branch name, project. + +* The lock model + +Two locks, both served by =.ai/scripts/agent-lock= (names only; the helper owns the paths, which live on tmpfs under =$XDG_RUNTIME_DIR/agent-locks/=, host-local and cleared on reboot). + +*Single-runner lock* (=sentry-<project>=, where =<project>= is the repo-root basename: =basename "$(git rev-parse --show-toplevel)"= — the same derivation =wrap-it-up.org='s guard uses, so the two agree on the lock name). Each cycle acquires it at cycle start and releases it at cycle end, and refreshes it between passes (the heartbeat, so a live cycle's lock never ages past one pass). If =/loop= fires again while a previous cycle still holds it, the new cycle's acquire fails and the cycle skips with one digest line — no two cycles run at once. The bounded wait is short (a few seconds); a live cycle means defer, not queue. + +*Roam-write lock* (=roam-write=). A pass that edits a file under =~/org/roam= acquires it, runs =capture-guard --wait= (the human-capture layer stays underneath), edits the working tree, triggers =systemctl --user start roam-sync.service=, and releases. The lock spans only edit-plus-trigger. Sentry never runs =git= against =~/org/roam= — roam-sync stays the repo's only committer (the 2026-06-24 one-git-owner rule). Pass 1's =pull --ff-only= is the sole, read-only exception. + +Every reclaim of a stale lock surfaces in the digest — the helper prints the reclaim note, and the cycle records it. A reclaim during a genuinely slow pass is possible, so it's never silent. + +* The Pass Runner — one contract per pass + +Each cycle, after acquiring the single-runner lock and verifying branch state (below), walks the pass list in order. Every pass follows the same four-step contract: + +1. *Probe* — a cheap existence check for the pass's target (named per pass below). Absent → the pass is one skip line in the digest and nothing more. This is what makes the pass list portable: passes self-activate where their target exists and stay silent elsewhere, with zero per-project configuration. + +2. *Work* — run the pass under the unattended contract. Quick, solo, already-agreed mechanical actions execute. Anything destructive or requiring judgment does *not* execute — it appends to the morning-approval queue (what, why, the exact command or edit that fires on approval). A pass runs fully or not at all; there is no reduced-form pass. + +3. *Session-context entry* — a pass that does or queues work appends its digest line to the =session-context.org= Session Log (path resolved via =.ai/scripts/session-context-path=) before its commit, so a crash between them still leaves the trail. Per-pass lines for an all-quiet cycle (every pass probe-skipped or no-op) are not written one by one — the cycle collapses to a single heartbeat at cycle-end (below), so an idle cycle doesn't spray one skip line per pass. + +4. *Commit* — if the pass wrote to disk, commit it: =chore(sentry): <pass> — <what changed>=. One commit per writing pass. A probe-skip or a no-op pass writes nothing and commits nothing. + +Between passes, refresh the single-runner lock (=agent-lock refresh sentry-<project>=) — the heartbeat. + +** Branch-state verification (cycle start, before the passes) + +After acquiring the lock, confirm the cycle is safe to run: + +- *On the right branch* — HEAD is =sentry/<today>-<host>=. If the loop was armed on a prior day and crossed midnight, the branch keeps the arming date; that's fine, morning teardown handles it. If HEAD is somehow *not* a sentry branch (an interrupted stop, a manual checkout), skip the whole cycle with a digest line rather than committing onto main. +- *Clean of foreign changes* — =git diff --quiet HEAD= excluding the spine set (=session-context.org= / =session-context.d/=, resolved via =session-context-path=). Sentry's own spine writes must not trip this; a genuinely unexpected dirty tree (something outside the spine changed and wasn't committed by a prior pass) poisons the cycle — skip it with a digest line, the next cycle retries. + +* Unattended safety — skip, never degrade + +With no one at the terminal, any unsafe state makes the affected scope skip with one digest line, and the next cycle retries. Unsafe states and their scope: + +- *Unexpected dirty tree* (non-spine) → skip the whole cycle. +- *Lost or un-acquirable single-runner lock* → skip the cycle (another cycle holds it, or the helper is missing). +- *A pass's own precondition unmet* (its probe fails, or a dependency is dirty) → skip that pass only. +- *Red suite at cycle-end* (see below) → the commits stay on the branch, flagged in the digest for morning review; the cycle doesn't roll back. + +Skips are never silent and never partial. Inside a *working* cycle, a pass line means the pass fully ran and a skip line names why it didn't. An *all-quiet* cycle is not a silent skip either: its single =sentry at HH:MM: nothing= heartbeat is the explicit record that every pass found nothing, standing in for a wall of identical skip lines. The anti-silence rule targets a pass that hides work it should have surfaced; a quiet cycle has surfaced that there was none. + +** Multi-day stall notification + +An unmerged prior =sentry/*= branch at cycle start (the morning review never happened) skips the cycle. After the *second consecutive* cycle skipped for this reason, send one persistent desktop notification naming the project and branch: + +: sentry stalled: <branch> unmerged — merge or delete to resume + +Then repeat at most daily. Persistent notify matches the paging convention — it stays on screen until dismissed. A multi-day stall never stays silent. + +* The pass list (v1) + +In order. Each names its detection probe. A pass whose probe fails is one skip line. + +1. *Roam pull* — =git -C ~/org/roam pull --ff-only=. Probe: =~/org/roam= is a git clone. Skipped when the roam tree is dirty (roam-sync owns that case) or the clone is absent. Read-only and ff-only — the one narrow exception to "don't touch roam git," so later passes read a fresh tree. + +2. *Inbox zero* — run =inbox.org= roam mode under the no-approvals contract: quick+solo+agreed items execute, shared-asset and convention proposals park (prepared diff, =VERIFY= task, sender reply) in the approval queue. Edits to =~/org/roam/inbox.org= take the roam-write lock + =capture-guard=. Probe: the roam clone or a project =inbox/= exists. Tidying the shared roam inbox is allowed from *any* project session, work included — it's housekeeping on a shared resource, not a durable KB-node write, so the work-denylist doesn't gate it (=knowledge-base.md=). Never park it as a cross-project boundary crossing. + +3. *Triage intake — mail and messenger sources excluded.* Run =triage-intake.org=, loading only its non-mail, non-messenger source plugins (calendar, PR/ticketing). The mail and messenger plugins — cmail, any Gmail variant, Telegram, Signal, chat DMs — are never loaded by a sentry cycle: Craig ruled 2026-07-21 that sentry doesn't check email or messengers. A manual "triage intake" still scans everything. Probe: the project has at least one *active* triage source that survives that exclusion — a project-specific plugin (=.ai/project-workflows/triage-intake.*.org=), or a non-empty =:TRIAGE_SOURCES:= declaration naming general plugins that exist. Mere presence of the template-synced general plugins does *not* activate the pass; a project that declares no sources, or whose only declared sources are mail or messengers, probe-skips (see =docs/specs/2026-07-20-triage-source-activation-spec.org=). Destructive actions (deleting, archiving, sending) queue; they never cycle unattended. + +4. *Todo cleanup* — the =clean-todo.org= mechanics (hygiene pass + =--archive-done= + =--convert-subtasks=). Probe: a root =todo.org=. Note that =--archive-done= is not purely an org-file pass on its first run in a project: it creates =archive/task-archive.org= and appends a =.gitignore= entry, so it produces a real tracked-file commit and correctly trips the cycle-end conditional suite. (archangel, first live run 2026-07-21.) + +5. *Task audit* — the *mechanical subset* of =task-audit.org= hourly (staleness counts, structural checks, cookie recomputation); the judgment half (priority regrades, consolidations, merge candidates) runs *once per night* and queues its findings rather than repeating them every cycle. Probe: a root =todo.org=. A full audit every hour is too heavy and re-surfaces the same judgment calls all night. (takuzu, first live run 2026-07-21.) Factual staleness fixes that are unambiguous still execute. + +6. *Working-files hygiene* — flag =working/<slug>/= directories whose backing task is closed (a filing candidate per =working-files.md=). Probe: a =working/= directory exists. The filing itself queues (it's a judgment move). + +7. *Spec status board* — the =docs-lifecycle= grep for spec keywords, surfacing any =DOING= spec whose bound build parent is closed. Probe: =docs/specs/= exists. + +8. *Link integrity* — broken =file:= links in the project's org files, via =lint-org.el=. Probe: =lint-org.el= present. Report-only into the digest; no unattended rewrites. + +9. *Git health* — uncommitted drift, unpushed commits on other branches, stale branches, main-behind-origin. Probe: =.git=. Report into the digest. + +10. *Prep + symlink freshness* — stale daily-prep docs, broken symlinks. Probe: the prep dir / symlinks exist (work and home only, in practice). + +11. *Bug and refactor finding* — hunt for real bugs and worthwhile refactoring opportunities in the project's codebase: static analysis (=shellcheck= for shell, the project's own linters for its languages), config sanity checks, plus one targeted code-reading area per cycle. Rotate the area across cycles and name it in the digest, so coverage accumulates over a night instead of re-reading the same corner. Randomized property sweeps (generate-and-verify against an engine's own invariants) are good quiet-cycle work here, reaching past a frozen test corpus. Expect the pass to go honestly quiet after the first few cycles find the standing defects; a quiet hunt is a result, not a failure. (takuzu, first live run 2026-07-21: three real fixes in the first four cycles, then quiet.) This pass does *not* run the test suite — the entry baseline already ran it, and re-running it hourly is anti-pattern 5; read the entry result instead. Probe: the project carries a codebase — source under version control beyond its org and tooling files. File each verified bug as a graded task in =todo.org= per the severity × frequency matrix (=todo-format.md=), and each refactoring opportunity as a =:refactor:= task, deduped against existing tasks; an unverifiable suspicion is a digest line, not a task. *Find, never fix in this pass* — the finding files a task and stops. A fix happens only in the opt-in implementation pass below, and only after the finding is a filed task that pass then re-verifies from scratch (see the premise rule there). A freshly-found "bug" can be a misread — one was filed and retracted two cycles apart on 2026-07-23 — so the file-then-verify-then-fix pipeline is deliberate: the task is the checkpoint, not a same-breath fix. (Added at Craig's order 2026-07-21, first dogfooded in dotfiles; refactor-finding added 2026-07-24.) + +12. *Solo-task implementation (opt-in — =:SENTRY_MAY_IMPLEMENT:=)* — work the backlog's solo, decision-free tasks on the branch. Probe: =.ai/notes.org= Workflow State carries =:SENTRY_MAY_IMPLEMENT: yes= *and* the project holds =:COMMIT_AUTONOMY:= (the implement pass commits). Absent the marker, skip — this pass is off by default, because it turns the morning from a two-minute merge into a code review, and that's the project owner's call. When on: invoke =work-the-backlog.org= under its unattended-loop contract (no pre-flight Q&A — there's no Craig overnight), eligibility =TODO= + =:solo:=, with the defer checklist deciding act-vs-file. The overnight-only tightening: only the *ready* bucket implements (clears every checklist item with zero open decisions); a task needing even one quick decision defers to a =VERIFY= rather than guessing, exactly as the loop caller already does. Commit each logical change to the sentry branch; *never push* — the morning review and merge is the gate, same as every other pass. The full quality bar holds (TDD, suite green before each commit, the isolated adversarial review per =publish= Step 1 with its re-review loop, =/voice=), and the review here runs the *premise check first*: reproduce the bug or confirm the problem is real before judging the diff. The review is the fact-checker that a filed claim never got, and it is what makes fixing-on-a-branch safe (Craig, 2026-07-24). A task that fails its premise check is not implemented — the finding was wrong, and that outcome is a digest line, not a commit. A task whose review never reaches approval — three rounds, a recurring finding, or a =Needs Discussion= verdict — is the same shape: no commit, and a digest line naming the standing findings, so the morning review sees what the reviewer would not pass rather than finding the task silently absent. (Added at Craig's direction 2026-07-24: overnight implement-on-branch, gated and never-pushed.) + +(KB lesson promotion — the pass the original proposal listed eleventh — is deferred to vNext. An unattended judgment pass writing to the shared knowledge base waits until sentry has quiet weeks behind it and a designed detection heuristic. See the filed lesson-detection-heuristic task.) + +* Cycle-end — conditional suite, then the digest commit + +After the passes: + +1. *Conditional suite run.* If any pass this cycle modified files *outside* the org/spine set (a code-touching pass, rare but possible via fixtures), run the full suite once. A green run confirms the cycle's commits are safe; a red run flags the digest for morning review — the commits stay on the branch (nothing is pushed, so the morning gate catches it). No per-pass suite runs: the entry run is the green baseline, and hourly per-commit runs would turn a seconds-long cycle into minutes all night. Cycles that only touched org/spine files skip this. + +2. *Heartbeat or digest, then commit.* Decide quiet vs working. A *quiet* cycle — every pass probe-skipped or no-op, nothing added to the approval queue — writes a single heartbeat line to the Session Log, =sentry at HH:MM: nothing= (HH:MM local, from =date=), and no per-pass digest block. A *working* cycle — any pass ran, wrote, or queued — writes its full per-pass digest block. Then commit any accumulated spine writes in one sweep: =chore(sentry): digest — <date> <time> cycle= for a working cycle, =chore(sentry): heartbeat — <date> <time>= for a quiet one, so even a quiet cycle leaves a clean tree for the next branch-state check (where the spine is untracked, the mirror-only case, there is nothing to commit and the heartbeat line stays in the working-tree anchor). This is the silent-until-signal policy (see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=): an all-quiet night collapses from a wall of no-op digests to a list of one-line heartbeats, while a cycle that actually did or queued something still writes the full record. + +3. *Release the single-runner lock.* + +* The digest and the approval queue + +*Digest.* A *working* cycle appends its block to the =session-context.org= Session Log (the spine the cycle already writes), so it survives a crash, rides the session archive, and is on screen in the running session. One block per working cycle: the timestamp, then one line per pass (ran + what, or skipped + why), plus any lock reclaim notes. A *quiet* cycle (nothing done or queued) writes no block — just the one heartbeat line =sentry at HH:MM: nothing= (the silent-until-signal policy). The per-pass block is a working-cycle artifact; it still carries one line per pass so a real skip inside a working cycle is never hidden. + +*Approval queue.* Destructive and judgment actions accumulate under one heading in the same file — =* Sentry approval queue (<date>)= — newest last. Each item carries three things: *what* (the action), *why* (what triggered it), and the *exact command or edit* that fires on approval. The morning review is Craig reading this heading top to bottom and running or discarding each item. + +* Morning teardown — Craig's, documented not automated + +Sentry never merges its own branch. In the morning Craig: + +1. Reviews the digest and the approval queue in =session-context.org=. +2. Runs or discards each approval-queue item. +3. Reviews the branch: =git log main..sentry/<date>-<host>= and the diff. +4. Squash-merges what he wants (=git switch main && git merge --squash sentry/<date>-<host>=, then one clean commit) or cherry-picks selectively. +5. Deletes the branch: =git branch -D sentry/<date>-<host>=. +6. Reverts any Emacs buffers still showing the pre-merge on-disk state (=emacs.md= buffer-revert caveat). + +A bad night is discarded by deleting one branch — nothing reached main, nothing was pushed. + +In a project that gitignores =.ai/=, the whole spine is untracked, so quiet cycles produce no commits at all and =git log main..sentry/<date>-<host>= understates the night's activity. There the anchor's heartbeat list is the only record of what fired. Read the anchor, not just the log. (archangel, first live run 2026-07-21.) + +* Stop Sentry + +Trigger: "stop sentry" (and synonyms above). Sentry owns its own shutdown: + +1. *Cancel the loop* — stop the =/loop= (=ScheduleWakeup= stop / the loop's stop path). No further cycles. +2. *Release the single-runner lock* if this context holds it. +3. *Branch disposition* — offer, inline-numbered: + 1. Squash-merge the day's branch into main now (walk the morning teardown steps 3-5 interactively) + 2. Leave it named for later review (=sentry/<date>-<host>= stays; review at leisure) +4. *Approval queue* — offer to walk the queued items now, or carry them (they stay under the heading for whenever Craig reviews). + +Stopping sentry is the only way to reclaim the working tree mid-night. The entry gate fronts the handoff; stop-sentry ends it. + +* Wrap-up interaction + +=wrap-it-up.org= refuses while sentry is live: it detects the single-runner lock (=agent-lock status sentry-<project>= → held) and stops with "sentry is active — say 'stop sentry' first." The shutdown logic lives here, not in wrap-up; wrap-up carries only the one guard. + +* Common Mistakes + +1. *Running without the =:COMMIT_AUTONOMY:= grant* — sentry commits unattended; the marker is the entry ticket, and its absence is a hard stop, not a degrade. +2. *Starting from a dirty or red tree* — the entry gates exist because an unattended cycle can't tell Craig's in-progress work from a regression. Answer the gate; don't bypass it. +3. *Committing onto main* — every writing pass commits to the daily =sentry/*= branch. A cycle that finds HEAD off the sentry branch skips rather than commits. +4. *Running a =git= write against =~/org/roam=* — roam-sync is the only committer. Sentry edits the tree under the roam-write lock and triggers the sync; it never commits or pushes roam. +5. *A per-pass suite run* — the suite runs at entry (baseline) and conditionally at cycle-end (only when a pass touched non-org files). Hourly per-commit runs all night is the anti-pattern the suite policy exists to prevent. +6. *Executing a judgment or destructive action unattended* — those queue for the morning with their exact command. The pass did its detection; Craig makes the call. The one sanctioned exception is pass 12's solo-task implementation, and only because it inherits work-the-backlog's full defer checklist (data-loss and irreversible actions defer, never execute) plus a premise-verifying review, and it commits to the branch rather than acting on anything live. +7. *A silent skip* — inside a working cycle, every skip writes a digest line naming why; a missing pass with no line reads as "ran clean" when it didn't. The one exception is not a violation: an all-quiet cycle collapses to a single =sentry at HH:MM: nothing= heartbeat instead of one skip line per pass — the heartbeat is the explicit "nothing to do" record, per the silent-until-signal policy. +8. *Degrading a pass to a reduced form* — a pass runs fully or skips. No half-passes. +9. *Letting an unmerged branch stall silently* — after two consecutive unmerged-branch skips, the persistent desktop notify cycles. Don't suppress it. +10. *Merging sentry's branch automatically* — the morning teardown is Craig's. Sentry creates and commits; it never merges or deletes its own branch. + +* Living Document + +Sentry ships with eleven finding/hygiene passes, one opt-in implementation pass, and a deferred KB pass. The pass list, the interval default, the =:SENTRY_MAY_IMPLEMENT:= default, and the queue-vs-execute line for each pass are the knobs most likely to move with dogfooding. The implement pass especially is new (2026-07-24) and unproven at scale — watch the corrections signal (work-the-backlog's metric for autonomous commits later reverted or hand-fixed) before widening it past the projects that opt in. Fold in what the live trial surfaces — a pass that queues too eagerly, a probe that misfires, a digest line that wants more detail. Refine as the signal arrives. + +* History + +Built 2026-07-19 from the sentry spec (=docs/specs/2026-07-14-sentry-workflow-spec.org=, ID f6c51f27-d7a2-4b63-9ff9-5ba005a66dfb) — 10 decisions and 12 review findings resolved before the build. Phase 1 shipped the =agent-lock= helper (commit =a8b6cf4=); this file is Phase 2, the engine. Phase 3 reconciles the roam writers (=inbox.org=, =knowledge-base.md=) to acquire the roam-write lock and adds the =wrap-it-up.org= guard. diff --git a/claude-templates/.ai/workflows/startup.org b/claude-templates/.ai/workflows/startup.org index 929d482..2262eea 100644 --- a/claude-templates/.ai/workflows/startup.org +++ b/claude-templates/.ai/workflows/startup.org @@ -29,10 +29,16 @@ Inside a rulesets session, the project-repo refresh below covers this — the ru #+begin_src bash rs="$HOME/code/rulesets" if [ -d "$rs/.git" ]; then - if (cd "$rs" && git diff --quiet --ignore-submodules HEAD -- 2>/dev/null); then + gate="$rs/claude-templates/bin/git-worktree-gate" + if [ -x "$gate" ] && "$gate" sync-safe "$rs"; then + (cd "$rs" && git pull --ff-only origin main 2>&1) | tail -3 + elif [ ! -x "$gate" ] \ + && (cd "$rs" && git diff --quiet --ignore-submodules HEAD -- 2>/dev/null); then + # Bootstrap fallback for a checkout old enough not to have the shared + # gate yet. The pull that follows installs it for subsequent starts. (cd "$rs" && git pull --ff-only origin main 2>&1) | tail -3 else - echo "rulesets: dirty working tree — using as-is, skipping pull" + echo "rulesets: changes beyond untracked inbox deliveries — using as-is, skipping pull" fi else echo "rulesets: not a git checkout — skipping" @@ -40,11 +46,11 @@ fi #+end_src Behavior: -- *Clean working tree* → fast-forward pull. =git pull --ff-only= refuses any merge or rebase, so the operation is either a no-op (already current) or a clean advance. -- *Dirty working tree* → skip the pull. Don't auto-stash and don't auto-merge — those would either lose work or invite conflicts at the worst possible moment (session start). +- *Clean working tree, or untracked deliveries only beneath =inbox/=* → fast-forward pull. =git pull --ff-only= refuses any merge or rebase, so the operation is either a no-op (already current) or a clean advance. Inbox files are queue input, not source-tree work, and do not block other projects from receiving rulesets updates. +- *Any staged or tracked change, dirty submodule, Git operation in progress, or untracked file outside =inbox/=* → skip the pull. Don't auto-stash and don't auto-merge — those would either lose work or invite conflicts at the worst possible moment (session start). - *Non-fast-forward history* → =--ff-only= aborts with an error. Surface that to the user; the rsync still proceeds against the working tree as-is. -*Template-freshness policy (applies to every dirty-check in the synced workflows).* "Dirty" means *tracked modifications only*. Untracked and gitignored files — an inbox drop, a file left in the tree to read, scratch output — never block a template pull, a fast-forward, or a monitoring gate. Projects were falling behind on templates because somebody sent them a task; that's the failure this policy closes. The checks here already comply (=git diff --quiet HEAD= sees only tracked changes; the ff gate uses =--untracked-files=no=), and any dirty-check added to a synced workflow follows the same rule. One deliberate exception: the rsync WIP-guard below counts untracked files *within rulesets' own synced source paths*, because an untracked half-written template is exactly the WIP it exists to hold back — that guard is about rulesets' outbound content, not the consuming project's local state. +*Template-freshness policy (applies to every dirty-check in the synced workflows).* The shared =git-worktree-gate sync-safe= policy is the source of truth: untracked files beneath =inbox/= and gitignored files do not block a pull, fast-forward, or monitoring gate; every other staged, tracked, untracked, submodule, or in-progress-operation state does. Projects must not fall behind merely because somebody sent them a task, but an arbitrary scratch file is not silently treated as safe. One deliberate exception remains: the rsync WIP-guard below is narrower than the repository gate and counts untracked files within rulesets' own synced source paths, because an untracked half-written template is exactly the WIP it exists to hold back. *** Install rulesets symlinks into ~/.claude (idempotent) @@ -74,8 +80,11 @@ if [ -d .git ]; then current=$(git symbolic-ref --short HEAD 2>/dev/null) dirty=0 - if ! git diff --quiet --ignore-submodules HEAD -- 2>/dev/null \ - || [ -n "$(git status --porcelain --untracked-files=no)" ]; then + gate="$HOME/code/rulesets/claude-templates/bin/git-worktree-gate" + if [ -x "$gate" ]; then + "$gate" sync-safe "$PWD" >/dev/null 2>&1 || dirty=1 + elif ! git diff --quiet --ignore-submodules HEAD -- 2>/dev/null \ + || [ -n "$(git status --porcelain --untracked-files=no)" ]; then dirty=1 fi @@ -107,8 +116,8 @@ fi #+end_src Behavior, per branch: -- *Behind only, current branch, clean tree* → =git merge --ff-only= advances HEAD. -- *Behind only, current branch, dirty tree* → fetched but not advanced. Surface so Craig can ff manually after dealing with the dirty state. +- *Behind only, current branch, sync-safe tree* → =git merge --ff-only= advances HEAD. An untracked =inbox/= delivery is sync-safe. +- *Behind only, current branch, sync-blocking tree* → fetched but not advanced. Surface so Craig can ff manually after dealing with the reported state. - *Behind only, non-checkout branch* → =git fetch . upstream:branch= advances the ref without touching the working tree. - *Diverged* (ahead and behind) → leave alone. Surface for Craig to resolve. Don't auto-rebase or auto-merge. - *Ahead only* or *up to date* → silent no-op. @@ -170,12 +179,14 @@ These calls have no dependencies on each other. Issue them all together in one m 10. =[ -f "$HOME/org/roam/inbox.org" ] && grep -cE '^\*\* ' "$HOME/org/roam/inbox.org" || true= — count items in the roam global inbox (=~/org/roam/inbox.org=), the roam-mode startup nudge. Silent if the roam clone isn't on this machine. Phase C reads the file when the count is non-zero, splits total vs items related to this project, and surfaces the offer (see =inbox.org= roam mode). Read-only; never files at startup. 11. KB surface prep (the read + contribute startup nudges; see =docs/specs/2026-06-16-encourage-kb-contribution-spec.org=). Gated on the agent KB clone. Counts =:agent:= nodes, lists up to 5 whose content matches the current project basename (titles only; a few most-recent nodes as a fallback when nothing matches), and resolves the best-practices node path. Read-only; silent when the clone is absent. Phase C surfaces the relevant titles (consult) and the best-practices link (contribute). + The best-practices lookup matches the node's *filename*, not its content. A roam node's slug lives only in its filename, so the earlier content-grep (=rg -l 'agent-kb-best-practices'=) matched nothing and the contribute nudge silently pointed at an empty path in every project, every session, for as long as it shipped. =find= rather than a glob keeps the probe identical under bash and zsh (zsh aborts on an unmatched glob) — the same reason the spec-sort probe below uses =find=. + #+begin_src bash ra="$HOME/org/roam/agents" if [ -d "$ra" ]; then proj=$(basename "$PWD") echo "kb-total: $(rg -l '#\+filetags:.*:agent:' "$ra" 2>/dev/null | wc -l)" - echo "kb-bestpractices: $(rg -l 'agent-kb-best-practices' "$ra" 2>/dev/null | head -1)" + echo "kb-bestpractices: $(find "$ra" -maxdepth 1 -name '*agent-kb-best-practices*.org' -print -quit 2>/dev/null)" matches=$(rg -il "$proj" "$ra" 2>/dev/null | head -5) [ -z "$matches" ] && matches=$(\ls -t "$ra"/*.org 2>/dev/null | head -3) echo "kb-relevant-titles:" diff --git a/claude-templates/.ai/workflows/suspend.org b/claude-templates/.ai/workflows/suspend.org index 3691f60..166f9c9 100644 --- a/claude-templates/.ai/workflows/suspend.org +++ b/claude-templates/.ai/workflows/suspend.org @@ -23,8 +23,10 @@ straight: Refreshes the anchor in place, prompts Craig to type =/clear=, and a hook resumes the *same* logical session in a fresh context. Craig is still here. - *suspend* (this workflow) — *leave.* Captures richly into the anchor, leaves - the file in place, and Craig walks away. The next session is a cold startup - that detects the present anchor and resumes from it. + the file in place, detaches the tmux client so the session parks in the + re-attachable set, and Craig walks away. The next session is a cold startup + that detects the present anchor and resumes from it — or Craig re-attaches the + still-live session directly. - =wrap-it-up= ([[file:wrap-it-up.org][wrap-it-up.org]]) — *end.* Writes the Summary, archives the anchor into =.ai/sessions/=, commits + pushes, and runs the phrase-dependent teardown. @@ -91,7 +93,32 @@ when the Summary body is from an earlier thread. that set — but the default shared behavior is to leave the tree alone.) 4. *Leave =.ai/session-context.org= in place.* Do not archive it. 5. *Brief handoff* — one or two lines: what was captured, where the resume - pointer is, the most-active thread. End and let Craig go. + pointer is, the most-active thread. This is the last thing Craig sees before + the view detaches (Step 6), so deliver it complete. +6. *Detach the tmux client.* As the final action, detach the client viewing the + =aiv-<project>= session so it drops out of Craig's active view while staying + alive in the background. This is a DETACH, not a teardown: the session and the + agent process keep running, nothing is killed, no context is lost. + + #+begin_src bash + sess=$(tmux display-message -p '#S' 2>/dev/null) + [ -n "$sess" ] && tmux detach-client -s "$sess" + #+end_src + + Run it as the very last tool call, after the handoff text has rendered — tmux + preserves the pane, so Craig sees the full handoff when he re-attaches. Unlike + wrap-up's teardown (which must defer to a =Stop= hook because it kills the + session the agent runs in, which would cut off the valediction), detach runs + inline: it disconnects the view but leaves the agent's session alive, so + nothing is cut off. Degrade gracefully — if not inside tmux (=$TMUX= unset, no + session), skip silently and the session simply stays attached. + + Why detach on every suspend: Craig cycles his live agent sessions in Emacs + with alt-space, and rotates through everything — including re-attaching + detached ai-term sessions — with shift+alt+space. A suspended session left + attached clutters the active rotation; detaching parks it in the + re-attachable set, which is what makes suspend-and-walk-away work. Re-attach + is one keystroke (shift+alt+space) or =tmux attach -t aiv-<project>=. * What suspend does NOT do @@ -103,8 +130,12 @@ does beyond capture: - No KB / memory promotion sweep. - No Linear / board reconciliation. - No session-record archive (the file stays live). -- No teardown (the ai-term buffer + tmux session stay up). It drops no - =Stop=-hook teardown sentinel, so the wrap-teardown hook stays dormant. +- No teardown. Suspend DETACHES the tmux client (Step 6) but never kills the + session: the =aiv-<project>= session and the agent process stay alive in the + background, only the view disconnects. It drops no =Stop=-hook teardown + sentinel, so the wrap-teardown hook stays dormant. Teardown — killing the + session — is wrap-it-up's job, not suspend's; detach is the lighter move that + parks a still-live session. - No blind commit of working files (step 3). - No valediction. A suspend is a pause, not a goodbye. diff --git a/claude-templates/.ai/workflows/triage-intake.org b/claude-templates/.ai/workflows/triage-intake.org index 9e08142..55cc939 100644 --- a/claude-templates/.ai/workflows/triage-intake.org +++ b/claude-templates/.ai/workflows/triage-intake.org @@ -11,6 +11,8 @@ Think of it as the ER intake queue: every new message, invite, and PR notificati *This file is the engine.* It carries no sources of its own. Every source it scans comes from a *source plugin* — a =triage-intake.<source>.org= file the engine loads at Phase 0. The engine is source-agnostic and project-agnostic; the project- and account-specific knowledge lives entirely in the plugins. To add a source, drop a plugin file. To change one, edit its plugin. Never wire a source into this file. +*Which sources a project pulls is a per-project choice.* A *project-specific* plugin (=.ai/project-workflows/triage-intake.*.org=, never synced) is active by presence — dropping it is the declaration. A *general* plugin (=.ai/workflows/triage-intake.*.org=, template-synced into every project — personal Gmail, cmail, calendar, Telegram, GitHub PRs) is active only when the project names its basename in a =:TRIAGE_SOURCES:= line in =.ai/notes.org= Workflow State (space-separated basenames, e.g. =:TRIAGE_SOURCES: personal-gmail cmail=). A project that declares nothing and owns no project plugin pulls nothing. This is the Phase 0 activation gate — presence is capability, the declaration is activation (see =docs/specs/2026-07-20-triage-source-activation-spec.org=). + Distinct from =daily-prep.org=: - *daily-prep* — heavier, once daily, builds the day's plan + standup brief + meeting prep + time blocks. - *triage-intake* — fast, repeatable, just answers "what's new since last check?" @@ -37,6 +39,8 @@ Typical timing: Do *not* use when running daily-prep — daily-prep already does this as Phase 3. +Also runs unattended as sentry's triage pass (=sentry.org=, pass 3): sentry invokes this engine under its no-approvals contract, where destructive actions (deleting, archiving, sending) queue for the morning-approval review instead of firing. The trigger phrases above are unchanged — a manual "triage intake" always routes here directly. + * Execution @@ -56,16 +60,18 @@ ls .ai/workflows/triage-intake.*.org .ai/project-workflows/triage-intake.*.org 2 The glob exclude is automatic: =triage-intake.*.org= matches the plugins but not this engine file (=triage-intake.org= has no second dot-segment), so the engine never loads itself. After globbing, for each plugin file: -1. Read it. -2. Evaluate its =ENABLED= precondition. If false, *announce the skip with its reason* ("skipping linear — mcp__linear not present") and move on. -3. The surviving set is the source list for Phases A-D. +1. *Activation gate.* A *general* plugin (from =.ai/workflows/=, template-synced into every project) is active only if its basename appears in the project's =:TRIAGE_SOURCES:= declaration (=.ai/notes.org= Workflow State — a space-separated list of source basenames). If it isn't declared, it is *inactive*: announce it ("inactive: personal-gmail — not in :TRIAGE_SOURCES:") and skip it. A *project-specific* plugin (from =.ai/project-workflows/=, never synced) is always active — dropping it there is itself the per-project declaration. This is what stops the synced general plugins from self-activating in projects that aren't triage targets: presence is capability, the declaration is activation (see =docs/specs/2026-07-20-triage-source-activation-spec.org=). An absent or empty =:TRIAGE_SOURCES:= means no general sources are active; a project with no declaration and no project plugin has no active sources, so triage no-ops there. +2. Read it. +3. Evaluate its =ENABLED= precondition. If false, *announce the skip with its reason* ("skipping linear — mcp__linear not present") and move on. +4. The surviving set — active and enabled — is the source list for Phases A-D. -*Announce the loaded set before scanning* so the omission can't hide: +*Announce the loaded set before scanning* so the omission can't hide — inactive (undeclared) plugins are named too, so a general plugin left out of =:TRIAGE_SOURCES:= is a visible choice, not a silent drop: #+begin_example -Loaded 5 source plugins: - general: personal-gmail, personal-calendar, cmail, github-prs +Loaded 2 source plugins (:TRIAGE_SOURCES: personal-gmail cmail): + general: personal-gmail, cmail project: deepsat-gmail + inactive (undeclared): personal-calendar, github-prs, telegram skipped: linear (mcp__linear not present) #+end_example @@ -203,6 +209,22 @@ Auto mode runs as a =/loop= in the *live session*, not a detached cron job: Running in the live session means MCP auth (Slack, Gmail, Linear) is inherited from the session — the headless-auth wall that blocks a detached cron run does not apply. A durable cross-session schedule is out of scope here; that belongs to the morning-ops orchestrator, which can later invoke auto mode's accumulate behavior as its triage limb. The close/stop commands below require a live session by design. +*** Phone delivery — push each signal sweep via =agent-text= + +Auto mode exists for when Craig is away from the desk, so a sweep that surfaces something worth seeing is delivered to his phone, not just printed into a session he isn't watching. After a sweep that renders the full three sections — one with real deltas or an unacked-list change (see "End-of-sweep output" below) — send that same output to his phone over Signal with =agent-text=: + +#+begin_src bash +agent-text "$SWEEP_SUMMARY" +#+end_src + +The pushed text is the *fuller* three-section shape, not a terse one-liner: the per-source deltas, the responses-awaiting-acknowledgment list, and the timestamp, led by a ⚠ SCAN FAILED banner if any source failed. + +*Signal-only — never on a quiet sweep.* An empty sweep (the =triage intake at HH:MM: nothing= heartbeat) does *not* push to the phone. Silent-until-signal (see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=) governs the phone channel too, so the phone stays silent until a sweep has real signal. The in-session heartbeat still prints as proof the loop ran; the phone is reserved for something that actually needs Craig. (Craig's ruling, 2026-07-20: the higher-cost channel doesn't buzz with "nothing.") + +If =agent-text= isn't on =PATH=, fall back to inline delivery and say so once. + +*Reply polling is deferred.* The send half ships here; polling the phone for Craig's replies (the =phone-recv= half of the retired ntfy design) waits on the reply-correlation follow-up. With the Signal account linked on more than one device, a reply fans out to every device and neither knows which page it answers — that has to be resolved before auto mode reads replies back. Until then auto mode pushes but does not poll, and Craig acts on a pushed summary from wherever he picks it up. + ** Preconditions and Close-out Auto mode borrows the inbox monitor-mode gates (=inbox.org= monitor mode): do not start on a dirty worktree or a red test suite — a close's batch commit would otherwise sweep up unrelated changes — and leave the tree clean and green when the loop stops. Surface a blocker with inline numbered options per =interaction.md= and wait. @@ -218,11 +240,13 @@ Each sweep runs Phase 0 (load *both* plugin dirs — the loud requirement still - DOES update an active daily-prep in Update mode and re-open it on change (per =daily-prep.org=). - DOES report, deltas-only, with loud scan-failure banners (Phase C rules unchanged). -** End-of-sweep output — three sections +** End-of-sweep output — three sections, or one heartbeat + +*Silent-until-signal (see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=).* An *empty sweep* — no deltas since the previous sweep and no change to the awaiting-acknowledgment list — collapses to a single heartbeat line and nothing else: =triage intake at HH:MM: nothing= (HH:MM local, from =date=). Detection still runs in full (Phase 0 plus the A-D scan, against the session's inherited MCP auth); only the output collapses, so a long unattended run stops filling the session with identical "no changes" blocks. A sweep with real deltas or an unacked-list change prints the full three sections below, and — when away — pushes them to Craig's phone via =agent-text= (see "Phone delivery" above). The empty-sweep heartbeat is never pushed. -1. *Deltas* — what changed since the *previous sweep* (the standard Phase C summary scoped to the inter-sweep delta; one line if nothing: "HH:MM sweep: no changes"). +1. *Deltas* — what changed since the *previous sweep* (the standard Phase C summary scoped to the inter-sweep delta). 2. *Responses awaiting your acknowledgment* — every Slack reply, email, or message directed at Craig that he hasn't acknowledged or had the agent answer. A *running list carried forward across sweeps* until Craig acks each item or closes the triage. An away user's first need is "who's waiting to hear back from me," which a delta-only sweep loses the moment it scrolls past. -3. *Timestamp* — the current date, time, and timezone on the sweep's own final line, so an away reader sees how fresh the summary is without computing it. Print it on *every* sweep, including a quiet "no changes" one — on a quiet sweep the stamp is the proof the loop ran. Generate it with: +3. *Timestamp* — the current date, time, and timezone on the sweep's own final line, so an away reader sees how fresh the summary is without computing it. Print it on every sweep that prints these three sections. On an *empty* sweep there is no separate timestamp line — the heartbeat (=triage intake at HH:MM: nothing=) is itself the freshness stamp and the proof the loop ran. Generate it with: #+begin_src bash date "+%A %Y-%m-%d %H:%M:%S %Z (%z)" @@ -416,6 +440,9 @@ Update the engine as the orchestration pattern evolves; update a plugin as its s *** Updates and Learnings +**** 2026-07-20: Phone delivery for signal sweeps (=agent-text=, send half) +Auto mode now pushes a full-three-section sweep to Craig's phone over Signal via =agent-text=, the away-from-desk delivery the retired ntfy design carried before ntfy was torn down (2026-07-04). Transport is =agent-text= (the renamed Signal pager), not ntfy. Signal-only by Craig's 2026-07-20 ruling: a quiet sweep's =nothing= heartbeat never reaches the phone — silent-until-signal governs the phone channel too, so the higher-cost channel only fires when a sweep has real signal, while the in-session heartbeat stays as proof the loop ran. Falls back to inline when =agent-text= is absent. Only the send half ships; reply polling (the old =phone-recv=) waits on the reply-correlation follow-up, because a Signal reply fans out to every linked device and neither knows which page it answers. + **** 2026-07-18: Three-section digest + close-by-default (Phase C/D rewrite) Craig's ruling after a 42h-gap sweep where the long-form report (top signals + per-source breakdown + 7-option action menu) was followed by "summarize the notable items" — and the digest that answered it was the report he wanted first. Phase C now renders ==TASKS== (work items needing Craig, solo-executable first, priority order) / ==FYI== (work context, no action owed) / ==MISC== (everything outside the project, actions stated inline), then exactly two options (close-and-file / close-and-execute-solo, the latter only when solo items exist), timestamp last. Per-source blocks and the itemized action menu are gone from the default surface (long form on request). Phase D became the close: it runs as the next action no matter what Craig replies (unless he explicitly holds), includes the mail hygiene on every scanned account without itemized confirmation, files the TASKS, clears resolved unacked items, advances the sentinel, and tears down started services. Solo = mechanical + standing-approved + no prose under Craig's name; prose sends and destructive non-mail actions stay gated. The stay-open-until-confirmed exit loop is retired. Same-day addendum: the "and reroute" modifier ("1 and reroute") — MISC items are surfaced-only by default (never filed to this project's todo.org); appending the modifier delivers each outside-project item to its owner's inbox via inbox-send per the cross-project rule. diff --git a/claude-templates/.ai/workflows/triage-intake.telegram.org b/claude-templates/.ai/workflows/triage-intake.telegram.org index 5039a8b..1319da5 100644 --- a/claude-templates/.ai/workflows/triage-intake.telegram.org +++ b/claude-templates/.ai/workflows/triage-intake.telegram.org @@ -30,12 +30,27 @@ Telega does not autostart with the Emacs daemon. "Down" is its normal state unless Craig has Telegram open in Emacs. The scan therefore runs the full lifecycle every time, never skips because the server is down: +⚠ *DOWN / not-loaded is the TRIGGER to launch, never a reason to skip or fail.* +This is the exact mistake two projects (work + home, 2026-07-24) made: they +probed telega, saw =(telega-server-live-p)= nil or telega not =featurep=, and +reported =SCAN FAILED: telegram — not loaded= or a silent SKIP — a *blind* +sweep — instead of running Step 1 to start it. A down or unloaded telega is the +normal entry state; =(telega t)= both LOADS the package and STARTS the docker +server (work confirmed: down → =(telega t)= → Ready, 18 chats). So the plugin +MUST run Step 1's launch whenever telega is down/unloaded, wait for Ready, then +scan. =SCAN FAILED= is reserved for a launch that was actually ATTEMPTED and did +not reach Ready (image missing, server crash on start, daemon unreachable) — +never for the pre-launch down state itself. The =:ENABLED:= guard above tests +whether telega is INSTALLED (=fboundp=), not whether the server is up; a down +server never disables the source. + 1. Record prior state: TELEGA_WAS_RUNNING via (telega-server-live-p). 2. Launch (only if not running): emacsclient -e "(progn (setq telega-use-docker t) (telega t) 'started)" - The setq is mandatory defense: tdlib segfaults outside docker mode - (2026-06-09), and Craig's daemon currently has telega-use-docker nil. - Wait ~2s for Ready, then (telega--loadChats 'main) until telega--chats + The setq is mandatory defense: tdlib crashed in native mode when this was + set up (2026-06-09) — a separate matter from the SEGFAULT gotcha, which is + about the loadChats argument — and Craig's daemon defaults to nil. + Wait ~2s for Ready, then (telega--loadChats '(:@type "chatListMain")) until telega--chats is populated. 3. Check messages: the maphash unread scan in ** Scan Step 2 (filters the messageContactRegistered join-notice noise). @@ -48,10 +63,13 @@ lifecycle every time, never skips because the server is down: Verify: telega-server-live-p → nil, no zevlg/telega-server container in docker ps. If Craig had it running, leave it untouched. -If any lifecycle step fails (docker image missing, server crash, daemon -unreachable), the sweep reports it as SCAN FAILED at the top of the summary -per the engine's failure rule — never as a silent skip. Craig gets real -traffic here. +If any lifecycle step fails *after the launch was attempted* (docker image +missing, server crash on start, daemon unreachable, Ready never reached), the +sweep reports it as SCAN FAILED at the top of the summary per the engine's +failure rule — never as a silent skip. This does NOT cover the ordinary +pre-launch down state: a down server means "run Step 1," not "SCAN FAILED." +Craig gets real traffic here, so a blind sweep that skipped the launch is worse +than a clean failure — it hides real unread messages behind a false all-clear. ** Scan @@ -85,22 +103,58 @@ TELEGA_WAS_RUNNING=$(emacsclient -e "(and (fboundp 'telega-server-live-p) (teleg *** Step 1 — start (docker mode) if not already running, wait for Ready #+begin_src bash -# `(telega t)` starts without popping the root buffer. Docker mode (the stable -# path — see the SEGFAULT gotcha) reconnects the persisted ~/.telega session in -# ~2s. Then load the main chat list so telega--chats populates. +# `(telega t)` starts without popping the root buffer. Docker mode reconnects the +# persisted ~/.telega session in ~2s. Then load the main chat list so +# telega--chats populates. +# +# The `(setq telega-use-docker t)` is mandatory and must come BEFORE `(telega t)`: +# tdlib crashed in native mode when this was first set up (2026-06-09), and the +# daemon's default is nil unless something (e.g. an Emacs-config :custom) has +# already forced it. It was missing here while the Quick Reference required it — +# a session that started telega without it on a native-mode daemon would take the +# untested path. Match the Quick Reference exactly. +# +# Note this is a SEPARATE concern from the SEGFAULT gotcha below: that gotcha is +# about the `loadChats` argument, and the deaths it explains happened in docker +# mode. Docker mode is not a defense against it, and it is not evidence for +# docker mode. Keep both. emacsclient -e "(progn + (setq telega-use-docker t) (unless (and (fboundp 'telega-server-live-p) (telega-server-live-p)) (telega t)) 'started)" # Poll until Ready with chats synced, or a crash/timeout. Background this with an # until-loop so the wait doesn't block; exit on Ready-with-chats OR an abnormal # server exit. Then force a chat-list load if the hash is thin: -emacsclient -e "(progn (ignore-errors (telega--loadChats 'main)) (ignore-errors (telega--loadChats 'main)) 'loaded)" +# NOTE: the chat-list argument must be a TL object, not the symbol 'main. +# `telega--loadChats' puts it straight into the request as :chat_list, and a +# bare symbol kills the server outright (see the SEGFAULT gotcha below). +# +# The liveness check on the tail is the load's only failure signal. `ignore-errors' +# catches nothing here, because a bad argument kills the server process rather than +# signalling in elisp, so without this the call returns 'loaded either way. +# The `fboundp' guard matches Step 0: if the launch failed outright telega is not +# loaded, and that should read as 'server-died like any other failure rather than +# signalling void-function. +emacsclient -e "(progn (ignore-errors (telega--loadChats '(:@type \"chatListMain\"))) (ignore-errors (telega--loadChats '(:@type \"chatListMain\"))) (if (and (fboundp 'telega-server-live-p) (telega-server-live-p)) 'loaded 'server-died))" #+end_src On a persisted session telega reaches status "Ready" within ~2s; the chat list loads over a few more. If =(hash-table-count telega--chats)= is 0 or thin, re-issue =telega--loadChats= and poll until it stabilizes. +⚠ *=server-died= is SCAN FAILED, never a quiet account.* A server that dies +during the load leaves a thin =telega--chats= hash, and a thin hash reads exactly +like an account with little unread. That is the same false all-clear the +down/not-loaded rule exists to prevent, arriving one step later in the lifecycle. +It also fits the SCAN FAILED definition above: the launch was attempted and did +not hold. So on =server-died=, report SCAN FAILED rather than scanning, and never +report a low unread count from that run. + +This is the independent evidence the SEGFAULT gotcha asks for when it says to +treat a short chat list as a real short list. Without the check there is no way +to tell the two apart, which is how the =loadChats= crash stayed invisible +through two investigations. + *** Step 2 — read unread, classified by last-message type The single most important filter: =messageContactRegistered=. Telegram counts a @@ -157,24 +211,61 @@ stays non-nil). =telega-server-kill= is what actually stops the server. Call left in =docker ps=. Skipping this whole branch when =TELEGA_WAS_RUNNING= is t is the point of Step 0: never tear down a session Craig is actively using. -⚠ *SEGFAULT GOTCHA — crashes are spontaneous; treat server death as routine.* -The dockerized =telega-server= (=zevlg/telega-server:latest=, image built -2026-06-04, tdlib 1.8.64) SIGSEGVs (exit 139) *on its own*, minutes-to-hours -into a session — 11 host coredumps between 2026-06-09 and 2026-06-11, several at -times when no triage verb was running. The 2026-06-11 investigation reproduced -the crash-free verbs and the spontaneous deaths side by side: coredump -backtraces show a corrupted stack (memory corruption in the musl build), and -no newer image exists upstream. Earlier theories — "native mode is the trigger", -"toggle-read is the trigger" — were timing coincidences; the verbs are sound. +⚠ *SEGFAULT GOTCHA — this was our bug, not tdlib's. Root-caused 2026-07-28.* +=telega-server= dies with =Unexpected char 'm' in plist value= followed by +=Assertion failed: false (telega-dat.c: tdat_plist_value: 500)=. The cause was +this workflow: Step 1 called =(telega--loadChats 'main)=. + +The chain. =telega--loadChats= is a raw TL wrapper — it drops its argument into +the request as =:chat_list= with no conversion. =telega-server--send= then +=prin1='s the whole plist, and =telega--tl-pack= passes atoms through untouched, +so the symbol goes out on the wire bare as =main=. The C parser +(=server/telega-dat.c=, =tdat_plist_value=) accepts only =(=, =[=, ="=, =-=, a +digit, =t=, =:=, or =n= to start a value. It hits =m=, prints that line, and +calls =assert(false)=, which aborts the process. The =m= in the error is +literally the first character of =main=. + +The symbol shorthand is real but belongs to a different layer: +=telega-filter.el= and =telega-folders.el= convert =(eq cl-fspec 'main)= into +='(:@type "chatListMain")=. The raw TL layer never does. telega's own callers +always pass the object (=telega.el:290=, =telega-tdlib-events.el:516=). + +Proved by experiment, not inference (2026-07-28): from a live Ready server, +=(telega--loadChats 'main)= killed it within seconds and added one coredump, +with that exact assertion; a restart plus =(telega--loadChats '(:@type +"chatListMain"))= survived three consecutive calls with no new coredump and no +assertion. + +*The previous entry here was wrong and cost real time.* It recorded the deaths +as spontaneous musl memory corruption and declared "the verbs are sound", which +sent later investigations at the docker image and tdlib versions instead of at +this file. The corrupted stack in the backtraces is what an =assert= abort looks +like, not independent evidence of a memory bug. If crashes are ever seen again +with *no* triage verb running, that is a genuinely separate cause and needs its +own investigation — do not reuse the old spontaneous-crash story to explain it. + +*This crash kills a scan; it does not silently shorten one.* An earlier draft of +this section claimed the reported "19 chats of ~50" was truncation caused by the +bad call. That was wrong, and work disproved it at the wire level on 2026-07-28: +with the corrected call their count is 19 before the first load and 19 after five +(four on =chatListMain=, one on =chatListArchive=). Nineteen is the real size of +that account. The same reading here — 19 stable across three corrected loads — +was already sitting in the evidence and should have retired the claim before it +was written down. Treat a short chat list as a real short list unless something +independently shows the server died mid-sync. + +=ignore-errors= around the call never helped — the failure is the server process +dying, not an elisp signal, so there is nothing for it to catch. That is why the +death is easy to miss from inside elisp, and why a caller should check +=(process-live-p (telega-server--proc))= after a load rather than trusting a +returned value. Operationally: docker mode stays mandatory (=telega-use-docker= = t; the setq before =(telega t)= is still the right defense), and *every action batch checks the server first* — =(process-live-p (telega-server--proc))= — restarting via -=(telega t)= when dead and re-checking Ready before firing verbs. A mid-sweep -death is recoverable, not an abort: restart, confirm Ready, resume. Durable-fix -candidates if the crashing gets worse: pin a pre-2026-06 image digest, build -=telega-server= natively against tdlib, or report upstream to zevlg with the -coredumps (=coredumpctl list /usr/bin/telega-server=). +=(telega t)= when dead and re-checking Ready before firing verbs. Any argument +handed to a =telega--*= TL wrapper must be a TL object or a plain +string/number/list, never a bare symbol. Defense in depth: even if the server does die, the scan still works because it reads the cached =telega--chats= hash, not a live query. A dead server is diff --git a/claude-templates/.ai/workflows/work-the-backlog.org b/claude-templates/.ai/workflows/work-the-backlog.org index 090841d..ea3f402 100644 --- a/claude-templates/.ai/workflows/work-the-backlog.org +++ b/claude-templates/.ai/workflows/work-the-backlog.org @@ -54,7 +54,7 @@ For the task set, in order, until the run cap is hit: 1. *Eligibility gate* (below). Ineligible → record =skipped-ineligible=, next task. 2. *Scope read* of the relevant code. Cheap; just enough to run the defer checklist. 3. *Defer checklist* (below). Any hit → defer: file the =VERIFY= naming the gap and record =deferred-VERIFY= (or, under the speedrun preset, route a quick-question gap to the pre-flight Q&A), next task. -4. *Implement* under the project's commit discipline: TDD red→green→refactor, then =/review-code --staged=, fix all Critical/Important findings, then close the task per =todo-format.md='s completion rules. Decompose into as many logical commits as the change needs — size is not capped. If implementation fails partway, leave the tree working, record =failed=, surface it, and continue to the next task. +4. *Implement* under the project's commit discipline: TDD red→green→refactor, then the isolated adversarial review (=publish= Step 1) with its re-review loop, fix all Critical/Important findings, then close the task per =todo-format.md='s completion rules. Decompose into as many logical commits as the change needs — size is not capped. If implementation fails partway, leave the tree working, record =failed=, surface it, and continue to the next task. 5. *Commit autonomy branch:* - =file-only= → surface the diff, do *not* commit. Record =implemented-diff-surfaced=. - =autonomous-commit= → =/voice personal= on the message, commit individually, push per the project's flow. Record =implemented-committed=. @@ -70,6 +70,8 @@ A task is autonomous-safe when *both* hold. This layer is a lookup, not a judgme 1. *Status is =TODO=* — never =VERIFY=, =DOING=, =DONE=, or =CANCELLED=. =VERIFY= marks "awaiting Craig's input"; auto-implementing one defeats the check it represents. The do-not-implement set is safe-by-omission: anything not plainly =TODO= (plus any project-declared "hold" marker) is out. 2. *Tagged =:solo:=* — the autonomy tag, resolved against the project's priority/tag scheme header in =todo.org= (never hardcoded). =:solo:= carries the hard definition in =todo-format.md=: completable and verifiable without Craig beyond at most one or two quick decisions answerable up front, no design deliberation. A project whose scheme declares a different autonomous-safe tag set overrides the default. +Terminology: *speedrunnable means tagged =:solo:=*. It does not mean =:quick:= or require =:quick:solo:=. The =TODO= status check above is the execution-state gate over that speedrunnable set. + Priority and =:next:= drive *ordering* within the eligible set, not eligibility ([#A] before [#B] before [#C], then the author's ordering). =:quick:= is an effort hint for batching and duration estimates — never a gate. Task *size* is deliberately absent from this gate. A large but well-specified, decision-free task is in scope and gets decomposed into per-logical-commit chunks during implementation. Size never sends a task away; only *deliberation* or *risk* does (the checklist below). @@ -80,7 +82,7 @@ Task *size* is deliberately absent from this gate. A large but well-specified, d After the scope read, run each eligible candidate through the checklist. Each item is a concrete, answerable question, not an adjective. *Any* hit — or any "unsure" — defers the task. Only a task that clears every item is implemented. -1. *Test-writability (the keystone).* Can I write the failing test from the task text — plus any decisions gathered up front — without inventing a requirement? *No / unsure* → underspecified. Under the speedrun preset, if the gap is one or two quick answerable questions, route it to the pre-flight Q&A; otherwise file a =VERIFY= noting what's missing. Under the unattended loop, file the =VERIFY= (no one to ask). +1. *Test-writability (the keystone).* Can I write the failing test from the task text — plus any decisions gathered up front — without inventing a requirement? *No / unsure* → underspecified. Under the speedrun preset, if the gap is one or two quick answerable questions, route it to the pre-flight Q&A; otherwise file a =VERIFY= noting what's missing. Under the unattended loop, file the =VERIFY= (no one to ask). *Open-ended goals are a specific, recognizable failure of this item:* a task phrased as an absence ("find bugs until none remain," "refactor until nothing worthwhile is left," "clean it up") has no writable acceptance test and so isn't really =:solo:=, even when tagged. Don't guess a stopping point — defer it and note that it needs measurable acceptance criteria (bound the surface, characterization net, dispositioned findings, objective floor — see =todo-format.md='s "Making an open-ended task measurable"). Once those are in the task body, it becomes runnable. 2. *Data-loss / irreversible / external operation.* Does implementing it require any of: =rm= of non-scratch data, =git reset --hard= / force-push, =DROP= / =DELETE= / =TRUNCATE=, file truncate/overwrite of persisted content, a schema or data migration, any external or shared-state mutation, any credential touch? *Yes* → do NOT implement; file a =VERIFY= naming the risk. This is the hard safety gate; an upfront answer never overrides it without an explicit checkpoint. 3. *Already-satisfied.* Does the scope read show the desired end-state already holds? *Yes* → file a =VERIFY= noting it and move on. Don't make a no-op change. 4. *Design deliberation.* Does the task carry an unresolved design question, a "weigh these approaches" with real tradeoffs, or a TBD that isn't a quick factual answer? *Yes* → under the speedrun preset, if it collapses to one or two quick questions, route to the pre-flight Q&A; otherwise file and surface as a =/start-work= candidate. Under the loop, file. The discriminator is *quick-answerable question* vs *deliberation* — never task size. @@ -108,7 +110,8 @@ Autonomy changes who approves, not what quality means. Per task, non-negotiable: - *TDD* per =testing.md=: red first, green, refactor. The keystone checklist item already proved the failing test is writable. - *Verification* per =verification.md=: fresh evidence, full suite green before any commit. -- *=/review-code --staged=* before every commit; Critical and Important findings block until fixed. +- *Isolated adversarial review* before every commit, dispatched per the =publish= skill's Step 1 — never an inline self-review, however small the diff. Critical and Important findings block until fixed, and each fix goes back to the *same* reviewer until it approves. Minor findings never earn another round. + - *When the review can't reach approval* — three rounds without it, a finding that recurs after being reported fixed, or a =Needs Discussion= verdict — the unattended run has no one to ask. Record the task =failed= with the standing findings in its result, leave the tree working, and continue to the next task. Never commit past a blocking finding because nobody is awake to adjudicate — an unreviewed commit landing overnight is the outcome this gate exists to prevent. - *=/voice personal=* on every commit message on the =autonomous-commit= path (or the patterns walked inline if the skill is unavailable), message printed inline so the log shows what landed. - *Task closure* per =todo-format.md=: depth-based completion (keyword + =CLOSED:= at level 2, dated rewrite at level 3+). - *One logical change per commit.* A large task becomes several commits, not one omnibus. @@ -152,7 +155,7 @@ With paging on, fire one page when the set is done or the cap is hit — end-of- notify info "Page" "<project>: <N> done, <M> remaining — <one-line summary>" --persist #+end_src -=--persist= keeps it on screen until dismissed, and =info= is the page-me urgency convention (persistent but never crash-scary). The page fires when the set completes *or* the cap stops the run — either way exactly once. The message carries the project name, the completed count, and the remaining count (with skipped tasks noted in the run summary) so Craig can confirm ready and name the next project in one reply. =notify= is the desktop paging surface; a run that expects Craig to be away also fires =agent-page= with the same message (the Signal phone channel — protocols.org "Paging Craig — the agent pager"). +=--persist= keeps it on screen until dismissed, and =info= is the notification urgency convention (persistent but never crash-scary). The notification fires when the set completes *or* the cap stops the run, either way exactly once. The message carries the project name, the completed count, and the remaining count (with skipped tasks noted in the run summary) so Craig can confirm ready and name the next project in one reply. =notify= is the desktop channel (the "page me" surface); a run that expects Craig to be away also fires =agent-text= with the same message (the Signal phone channel, "text me"). See protocols.org "Reaching Craig". * Metrics diff --git a/claude-templates/.ai/workflows/wrap-it-up.org b/claude-templates/.ai/workflows/wrap-it-up.org index 5ce88a5..ecd3d22 100644 --- a/claude-templates/.ai/workflows/wrap-it-up.org +++ b/claude-templates/.ai/workflows/wrap-it-up.org @@ -24,11 +24,13 @@ The wrap-up is complete when: 2. *File is archived.* =.ai/session-context.org= has been renamed to =.ai/sessions/YYYY-MM-DD-HH-MM-description.org=. The old path no longer exists. 3. *todo.org is clean.* Cleanup script ran. Any auto-fixes are staged for the wrap-up commit. Orphan planning lines surfaced for manual fix if there are any. 4. *Linear board is honest* (skip if project doesn't use Linear). Any Dev-Review ticket whose PR has merged was moved to Done or PM Acceptance per the classification rule. -5. *Git state is clean.* All changes committed + pushed to all remotes. Working tree clean. +5. *Git state is certified clean.* All changes are committed + pushed to all remotes, =git-worktree-gate certify= succeeded at the current HEAD, and the working tree has no staged, unstaged, untracked, submodule, or in-progress-operation state. 6. *Valediction delivered.* Brief, warm closing with key accomplishments and reminders, ending with =session wrapped.= on its own line as the signoff marker. The absence of =.ai/session-context.org= is the signal that the last session wrapped up cleanly. Its presence at session start means the previous session was interrupted. +*A helper session meets a shorter list.* Criteria 1 and 2 apply to its own context file (archived under =.ai/sessions/YYYY-MM-DD-HH-MM-<id>-<description>.org=), and 6 applies. Criteria 3, 4, and 5 do not: hygiene, the Linear pass, and all git mutation belong to the primary, so a helper that satisfied criterion 5 would have violated its contract to get there. Step 0 routes this. + * Teardown mode (set from the trigger phrase) The wrap itself — Steps 1 through 5 — is identical in every mode. The trigger phrase only decides what Step 6 does once commit + push and the valediction are done. Resolve the mode from the phrase before starting: @@ -43,8 +45,58 @@ This depends on three functions in =.emacs.d/modules/ai-term.el= (=cj/ai-term-qu * The Workflow +** Step 0: Helper branch — a helper wraps only itself + +Resolve first whether this session is a helper, because a helper's wrap is a different and much shorter workflow. Everything from Step 1 down — the hygiene passes, the inbox check, the commit, the push, the clean-tree certificate — is primary-only under the role contract in [[file:helper-mode.org][helper-mode.org]], and running any of it from a helper is exactly the concurrency failure that contract exists to prevent. + +A session is a helper when =AI_HELPER=1= in its environment (=ai --helper= sets it) or when it adopted helper-mode.org this session by instruction. If neither holds, this is a primary: skip to Step 0.5 and wrap normally. + +#+begin_src bash +echo "AI_HELPER=${AI_HELPER:-unset} AI_AGENT_ID=${AI_AGENT_ID:-unset}" +#+end_src + +For a helper, re-run the roster — the answer decides which wrap applies: + +#+begin_src bash +root="$(git rev-parse --show-toplevel 2>/dev/null || pwd)" +if [ -x "$root/.ai/scripts/agent-roster" ]; then + "$root/.ai/scripts/agent-roster" "$root"; rc=$? +else + rc=2 +fi +echo "roster rc=$rc" +#+end_src + +Pass the project root explicitly. =agent-roster= defaults to =$PWD= and keeps only agents whose cwd is at or inside that root, so running it from a subdirectory hides a primary sitting at the root — and the "alone" that produces is read below as *orphaned*, which is the one branch that commits and pushes. Capture =rc= inside the branch too: =[ -x … ] && …; echo $?= reports the status of the whole list, so an absent script reads as 1 (others live) rather than 2 (unavailable). + +- *Primary still live (rc 1)* — the normal case. Finalize the =* Summary= in the helper's own context file — same contract as Step 1, KB receipt line included (resolve it with =AI_AGENT_ID=<id> .ai/scripts/session-context-path=), archive it to =.ai/sessions/YYYY-MM-DD-HH-MM-<id>-<description>.org= so it can't collide with the primary's archive name, deliver the valediction, and stop. Do NOT commit, push, or run any hygiene pass. The helper's scoped edits stay in the tree and the primary's next commit carries them along with the archived file — say so in the valediction, so Craig knows the work is real but not yet pushed. +- *Alone (rc 0) — orphaned helper* — the primary exited first, so the git ban lifts: the concurrency that justified it is gone, and stopping here would strand the helper's edits as a dirty tree nobody owns. Run the full wrap below starting at Step 0.5, exactly as a primary would. +- *Roster unavailable (rc 2, or the script absent)* — take the archive-only path, the same as primary-still-live. Leaving work for the next session to commit is recoverable; guessing "orphaned" and committing underneath a live primary is not. + +** Step 0.5: Refuse if sentry is live + +Before anything else, check whether sentry is running in this project. Sentry holds the working tree on its =sentry/<date>-<host>= branch and commits unattended; wrapping underneath it would archive the session anchor and tear down the buffer while the loop is still firing into it. If sentry's single-runner lock is held, stop and point at the shutdown path: + +#+begin_src bash +proj="$(basename "$(git rev-parse --show-toplevel 2>/dev/null || pwd)")" +if [ -x .ai/scripts/agent-lock ] && .ai/scripts/agent-lock status "sentry-$proj" | grep -q '^held'; then + echo "sentry is active — say 'stop sentry' first" + exit 1 +fi +#+end_src + +The stop-sentry operation (defined in =sentry.org=) owns the shutdown: it cancels the loop, disposes of the branch, and walks the approval queue. Wrap-up carries only this one guard; a =stale= lock (a crashed cycle) doesn't block — only a live =held= lock does. + ** Step 1: Finalize the Summary +*** Work the Before-Close Queue (before the Summary) + +If the session anchor (=.ai/session-context.org=) carries a =* Before-Close Queue= heading with items, work them now, oldest-first, before writing the Summary, so any resulting edits ride this wrap's commit and get described in it. The queue is the "put X on the list" shorthand (see =protocols.org=, Colloquialisms and Expansions): session-scoped work Craig deferred to wrap time. + +Per item: do it if it's clear and bounded, or promote it to a =todo.org= task if it turns out to need its own session. Never drop an item silently. Remove each line as it's handled; if one can't be finished, surface it in the valediction (Step 5) and either leave a follow-up task or state why it's dropped. + +If there's no =* Before-Close Queue= heading, or it's empty, this step is a silent no-op. + *** Early KB reflection (capture while fresh, before the Summary) Before distilling the Summary, while the session is still fresh, ask: what did this session learn worth remembering, for yourself or a future agent? Reflect and stage any candidate durable facts — a decision and its why, an environment gotcha, a reference pointer, a transferable lesson. Self-answer silently; this adds no interactive turn (Craig already authorized the wrap). The candidates flow straight into the KB promotion check below, which does the actual writing and the receipt — this is the capture half, that is the commit half, one pipeline, one receipt. Reflecting here rather than reconstructing learnings after the Summary is the point: the early ask is what keeps the receipt from defaulting to "promoted 0" out of fatigue. @@ -167,6 +219,16 @@ Preview the moves without writing: emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done --check todo.org #+end_src +*** Clear temp/ + +#+begin_src bash +[ -d temp ] && find temp -mindepth 1 -delete && echo "temp/ cleared" +#+end_src + +=temp/= holds throwaway artifacts — discarded prototypes, scratch output, intermediate data (see =working-files.md=). It's gitignored in every project, so nothing here rides a commit and nothing is recoverable from git once deleted. Clearing it at wrap is what keeps ephemeral work from silting up across sessions, and it's the counterpart to =working/=, which is tracked and *never* cleared here. + +Two guards. Confirm before deleting if =temp/= holds anything a reasonable reader would call in-progress rather than throwaway — misfiled work belongs in =working/=, so move it there instead of deleting it. And skip the step entirely in a project where =temp/= is not gitignored, since that means the project is using the directory for something else. + *** Sync child priorities #+begin_src bash @@ -241,7 +303,7 @@ For an interactive walk of the judgments mid-day, run =/lint-org todo.org=. *** Inbox sanity check (surface unprocessed handoffs) -If the project has an =inbox/= directory, verify it holds nothing but =.gitkeep=, =lint-followups.org= (the lint-org pipeline file the next morning's daily-prep consumes), and any explicitly-deferred =PROCESSED-*= files before the wrap completes. An inbox that arrived at session start with handoffs from other projects, or that received handoffs mid-session, needs =inbox.org= process mode to run and apply its value-gate dispositions. Wrapping with a dirty inbox silently defers the work to next session and accumulates handoff debt that the sender can't see. +If the project has an =inbox/= directory, verify it holds nothing but =.gitkeep=, =lint-followups.org= (the lint-org pipeline file the next morning's daily-prep consumes), and =PROCESSED-*= files before the wrap completes. An inbox that arrived at session start with handoffs from other projects, or that received handoffs mid-session, needs =inbox.org= process mode to run and apply its value-gate dispositions. Wrapping with an unprocessed inbox silently defers the work to next session and accumulates handoff debt that the sender can't see. #+begin_src bash unprocessed=$(find inbox -maxdepth 1 -type f \ @@ -250,7 +312,7 @@ unprocessed=$(find inbox -maxdepth 1 -type f \ ! -name 'PROCESSED-*' \ 2>/dev/null | wc -l) if [ "$unprocessed" -gt 0 ]; then - echo "wrap-up: inbox/ has $unprocessed unprocessed item(s). Run inbox.org process mode before wrapping, or explicitly defer each item with a one-line reason in the valediction." + echo "wrap-up blocked: inbox/ has $unprocessed unprocessed item(s). Run inbox.org process mode before wrapping." find inbox -maxdepth 1 -type f \ ! -name '.gitkeep' \ ! -name 'lint-followups.org' \ @@ -259,7 +321,7 @@ if [ "$unprocessed" -gt 0 ]; then fi #+end_src -If the count is zero or the project has no =inbox/= directory, the check is a silent no-op. If non-zero, the wrap is incomplete by default. The user resolves each item (process now, defer with reason in the valediction, or delete with rationale) before the validation checklist passes. +If the count is zero or the project has no =inbox/= directory, the check is a silent no-op. If non-zero, the wrap is blocked. Process each item through its value-gate disposition, or delete it only when that workflow's rationale authorizes deletion, before continuing. The check exempts =lint-followups.org= explicitly because lint-org runs earlier in the same wrap-up workflow and writes its judgment items to that file in =inbox/= by design. The file is a pipeline artifact for the next morning's =daily-prep=, not a handoff that needs the value gate. @@ -469,17 +531,17 @@ Behavior: git status --short #+end_src -*Default policy: end every session with an empty =git status=.* The wrap is incomplete while anything remains dirty. There is no "leave it alone" default — every leftover gets an active resolution. The only way for a file to stay dirty across the wrap is the user explicitly saying "defer this one, leave it dirty." Surface each leftover with a concrete recommendation; the user has to actively opt out for the dirt to persist. +*Hard policy: end every session with an empty =git status=.* The wrap is incomplete while anything remains dirty. There is no deferral exception and no "wrapped with known changes" state: unresolved dirt means the session remains open and wrap-up does not occur. This inverts the older "intentional carryover" default, which let pre-existing dirty state accumulate across sessions silently. Carryover that lives for days or weeks is almost always one of: a forgotten commit from a prior wrap, a stale change that should be discarded, or genuine in-flight work that needs an explicit stash/branch home. None of those should default to "leave it dirty." **** Three kinds of leftover -| Pattern | What it is | Recommended action (apply unless user defers) | +| Pattern | What it is | Recommended action | |---+---+---| | Generated, runtime, or lock files that no human edits — e.g., =.claude/scheduled_tasks.lock=, =.pytest_cache/=, build outputs, IDE state, editor swap files | *Runtime artifact* — created by tooling or the harness, not by the user, and shouldn't be tracked | Add the matching pattern to =.gitignore= (project-level, not =~/.gitignore_global=). For tracked files, =git rm --cached <path>=. Stage =.gitignore= and any =rm --cached= changes in *one* follow-up commit (=chore: gitignore X=), push. Re-run =git status= to confirm clean. | | Modified or created during the session but not staged into the wrap-up commit | *Forgotten change* — real session work that should have been in the wrap commit but missed it | Stage and create a follow-up commit. Don't =--amend= the wrap-up commit once pushed (diverging history without a clear win). Push the follow-up to all remotes. | -| Was dirty at session start and still dirty at session end — work this session deliberately didn't touch | *Pre-existing dirt that needs a decision* — could be a missed commit from a prior wrap, stale abandoned work, or real in-flight work without a home | Investigate (show diff + check the originating session). Recommend one of: (a) commit now if the work is complete, (b) stash with a descriptive message if it's genuine WIP, (c) =git checkout -- <path>= / =git clean -f <path>= if stale and unwanted, (d) move to a feature branch if it's longer-running, (e) user explicitly defers and accepts the dirt. Do not silently leave dirty. | +| Was dirty at session start and still dirty at session end — work this session deliberately didn't touch | *Pre-existing dirt that needs a decision* — could be a missed commit from a prior wrap, stale abandoned work, or real in-flight work without a home | Investigate (show diff + check the originating session). Recommend one of: (a) commit now if the work is complete, (b) stash with a descriptive message if it's genuine WIP, (c) =git checkout -- <path>= / =git clean -f <path>= if stale and unwanted, or (d) move to a feature branch if it's longer-running. Do not silently leave dirty. | **** Per-file flow @@ -487,18 +549,40 @@ For each leftover line in =git status --short=: 1. Identify which of the three kinds above it matches. 2. State what the file is (one line) and the recommended action. -3. Apply the action unless the user explicitly defers. -4. Re-run =git status --short= after each follow-up commit until empty (or until every remaining line is an explicit user-deferred entry). +3. Apply the action when it is safe and authorized. +4. Re-run =git status --short= after each follow-up commit until empty. The pre-existing-dirt case (third row) is the one this rule most cares about. Treat each pre-existing-dirty file as a question that must get an answer this session, not as "carryover that's fine to inherit." A file that was dirty for a week before this session probably isn't going to get cleaner by waiting another week. Look at the diff, check the originating session's notes, and recommend a real resolution. -**** When the user defers +**** When cleanup cannot be completed -If the user does say "leave this one dirty for now" after seeing the recommendation, that is fine — log the deferral in the valediction so the next session knows it was an explicit choice, not a miss. Format: "Deferred (per Craig's decision today): =path/to/file= — <one-line reason>". Without that note, the next session can't distinguish "we agreed to defer" from "we forgot again." +Stop the wrap. Do not deliver the valediction, print =session wrapped.=, drop a teardown/shutdown sentinel, or describe the session as complete. Report: + +1. Every remaining path and its exact Git state. +2. What the file is and why the agent cannot safely resolve it alone. +3. The concrete action or decision Craig needs to provide to make the tree clean. + +An explicit decision to keep a file dirty changes the outcome from "wrapping" to "leaving the session interrupted." It never satisfies this workflow. + +*** Final clean-tree certificate — hard gate + +After all commits are pushed and every leftover appears resolved, run the shared gate: + +#+begin_src bash +gate="$(command -v git-worktree-gate 2>/dev/null || true)" +[ -n "$gate" ] || gate="$HOME/code/rulesets/claude-templates/bin/git-worktree-gate" +if [ ! -x "$gate" ]; then + echo "wrap blocked: git-worktree-gate is unavailable; install rulesets tooling and retry" + exit 1 +fi +"$gate" certify "$PWD" +#+end_src + +The certificate lives inside the Git directory, so it does not dirty the worktree. It records the exact verified HEAD. A non-zero result is a hard stop governed by "When cleanup cannot be completed" above. Step 5 is unreachable until certification succeeds. ** Step 5: Valediction -Brief, warm closing. 3-4 sentences max. +Only after the final clean-tree certificate succeeds, deliver a brief, warm closing. 3-4 sentences max. Include: - What was accomplished (specific, not generic) @@ -535,7 +619,7 @@ Do nothing. The buffer, the =aiv-<project>= tmux session, and =claude= all stay *** Teardown mode (default) -Confirm commit + push succeeded (Exit Criteria 5 — never tear down over unpushed work), then drop the sentinel: +Confirm commit + push and the final clean-tree certificate succeeded (Exit Criteria 5 — never tear down over unpushed or dirty work), then drop the sentinel: #+begin_src bash touch "/tmp/ai-wrap-teardown-$(basename "$PWD")" @@ -543,6 +627,8 @@ touch "/tmp/ai-wrap-teardown-$(basename "$PWD")" That is the whole step. Don't run any =tmux kill-session=, =emacsclient=, or buffer kill inline — the =Stop= hook reads the sentinel when this response ends and runs =cj/ai-term-quit=, which kills the =aiv-<project>= session (taking =claude= with it), kills the vterm buffer, and restores geometry. The basename of =$PWD= is the key the hook matches, so the sentinel names the session it tears down. +*The sentinel is session-scoped.* If certification fails, the =Stop= hook blocks and leaves the sentinel armed on purpose, so a wrap blocked by a dirty tree retries on a later stop without re-running this workflow. It does *not* survive the session: =session-start-disarm.sh= clears it at =SessionStart=, because a wrap that never certified is not a pending teardown once its session is gone. Before that hook existed, an uncertified sentinel sat armed indefinitely and fired in whatever session next reached a clean tree — work's 2026-07-27 11:37 wrap killed the 13:20 session mid-work, and archsetup's sat armed on a live terminal for two days. If teardown is still wanted in a new session, run this workflow again. + *** Shutdown mode Confirm commit + push succeeded, then evaluate the safety gate *before* committing to the shutdown — never power the box off out from under another live session: @@ -573,7 +659,8 @@ If =emacsclient= isn't resolvable or the daemon is down, the gate can't run — 7. *Leaving =.ai/session-context.org= in place* — its presence means "interrupted session", confuses next startup 8. *Long preachy valediction* — brief beats thorough 9. *Leaving runtime/generated files dirty without gitignoring them* — pollutes every future =git status= and erodes trust in "working tree clean" as a signal. Fix =.gitignore= during the wrap, not later. -10. *Treating "was dirty at session start, still dirty now" as fine by default* — that's how a forgotten commit from two sessions ago turns into "carryover" for two weeks. Every pre-existing dirty file needs an active resolution recommendation this session. Deferral is allowed only with an explicit user choice, logged in the valediction. +10. *Treating "was dirty at session start, still dirty now" as fine* — that's how a forgotten commit from two sessions ago turns into "carryover" for two weeks. Every pre-existing dirty file must be resolved or the wrap remains blocked. +11. *Calling a blocked cleanup a wrap* — if the strict gate fails, report the paths and needed decisions; do not valedict, certify completion, or tear down. * Validation Checklist @@ -586,19 +673,20 @@ Before considering wrap-up complete: - [ ] =todo-cleanup.el= ran — hygiene pass + =--convert-subtasks= + =--archive-done= + =--sync-child-priority= (if =todo.org= exists at project root) - [ ] =lint-org.el= ran on =todo.org= — mechanical fixes applied, judgments appended to follow-ups file (if =todo.org= exists) - [ ] Any orphan-planning-line warnings reviewed (fix or accept) -- [ ] Inbox carries nothing but expected pipeline artifacts (=.gitkeep=, =lint-followups.org=, =PROCESSED-*= prefixes), OR each remaining handoff has an explicit deferral logged in the valediction +- [ ] Inbox carries nothing but expected committed or ignored pipeline artifacts (=.gitkeep=, =lint-followups.org=, =PROCESSED-*= prefixes); any untracked inbox delivery was processed before wrap - [ ] Linear Dev-Review sweep ran; any merged-PR tickets moved to Done or PM Acceptance (skip if project doesn't use Linear) - [ ] Template-sync churn committed as its own =chore: sync .ai tooling from templates= (consuming projects only; skipped in rulesets), or surfaced if a synced path didn't match canonical -- [ ] After wrap-up commit + push, =git status --short= is empty OR every remaining line has an explicit user-deferred decision logged in the valediction +- [ ] After wrap-up commit + push, =git-worktree-gate certify "$PWD"= succeeded at the current HEAD - [ ] Each leftover was investigated and the user saw a concrete resolution recommendation - [ ] Runtime artifacts added to =.gitignore=, follow-up commit pushed, =git status= re-verified - [ ] Forgotten changes committed in a follow-up and pushed -- [ ] Pre-existing dirty files resolved (committed / stashed / discarded / moved to a feature branch) or explicitly deferred with a one-line reason in the valediction +- [ ] Pre-existing dirty files resolved (committed / stashed / discarded / moved to a feature branch); otherwise wrap stopped with an actionable blocker report - [ ] Current branch pushed to ALL remotes (verified with =git remote -v=) - [ ] All other local branches with a tracking upstream pushed to their remote - [ ] Any untracked-upstream branches surfaced for manual =git push -u= - [ ] Step 6 teardown matches the trigger phrase: no-teardown leaves the buffer; teardown drops only =/tmp/ai-wrap-teardown-<project>=; shutdown gates on =cj/ai-term-live-count= = 1 and drops only =/tmp/ai-wrap-shutdown-<project>= - [ ] No teardown/shutdown sentinel was dropped before commit + push was verified +- [ ] The teardown hook can re-verify the clean-tree certificate before consuming a sentinel - [ ] Shutdown aborted (fell back to normal wrap, logged in the valediction) when another =aiv-*= session was live or the gate couldn't run - [ ] Commit message follows format (no =session:=, no Claude attribution) - [ ] Valediction delivered (brief, specific, warm) diff --git a/claude-templates/bin/agent-page b/claude-templates/bin/agent-page index 11a5264..728ee78 100755 --- a/claude-templates/bin/agent-page +++ b/claude-templates/bin/agent-page @@ -1,48 +1,12 @@ #!/bin/bash -# agent-page — page Craig's phone over Signal, from any machine or agent runtime. +# agent-page — deprecated alias for agent-text. # -# Usage: agent-page <message...> -# -# The pager identity (+15045173983) is registered in velox's signal-cli, so -# velox sends directly and every other machine ssh-relays the send over the -# tailnet. The recipient is Craig's Signal account UUID — his phone number -# reads as unregistered in Signal's directory, so never page the number. -# Verified end to end 2026-07-13 (phone push confirmed). -# -# This is the AWAY channel. At his desk, use the desktop channel instead: -# notify info "Title" "Message" --persist -# See protocols.org "Paging Craig" for choosing between them. -# -# Known caveats (tracked on the rulesets Signal-pager task): velox must be -# up and reachable on the tailnet, and the signal-cli account wants a -# periodic `receive` (staleness warnings appear otherwise). +# The Signal phone tool was renamed agent-text on 2026-07-20, when the +# notification vocabulary split into "text me" (Signal) and "page me" (desktop). +# This shim keeps old callers and other machines working until they re-install +# and pick up agent-text directly. Remove it in a later cleanup once nothing +# references agent-page. # # Source: ~/code/rulesets/claude-templates/bin/agent-page -# Install: make -C ~/code/rulesets install - -PAGER_ACCOUNT="+15045173983" -CRAIG_UUID="b1b5601e-6126-47f8-afaa-0a59f5188fde" -VELOX_HOST="velox.tailf3bb8c.ts.net" - -if [ $# -eq 0 ]; then - echo "usage: agent-page <message...>" >&2 - exit 2 -fi - -msg="$*" - -if [ "$(uname -n)" = "velox" ]; then - signal-cli -a "$PAGER_ACCOUNT" send -m "$msg" "$CRAIG_UUID" - rc=$? -else - # printf %q hardens the message for the remote shell. - ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new \ - "$VELOX_HOST" \ - "signal-cli -a $PAGER_ACCOUNT send -m $(printf '%q' "$msg") $CRAIG_UUID" - rc=$? -fi -if [ "$rc" -ne 0 ]; then - echo "agent-page: phone page failed (velox down or unreachable?) — fall back to the desktop channel: notify info 'Page' '<message>' --persist" >&2 -fi -exit "$rc" +exec "$(dirname "$(readlink -f "$0")")/agent-text" "$@" diff --git a/claude-templates/bin/agent-text b/claude-templates/bin/agent-text new file mode 100755 index 0000000..86aa933 --- /dev/null +++ b/claude-templates/bin/agent-text @@ -0,0 +1,57 @@ +#!/bin/bash +# agent-text — text Craig's phone over Signal, from any machine or agent runtime. +# The Signal half of the notification vocabulary: "text me" reaches the phone, +# "page me" is the desktop channel (notify). See protocols.org "Reaching Craig". +# +# Usage: agent-text <message...> +# +# The Signal identity (+15045173983) is registered in velox's signal-cli, and +# any daily driver linked as a device of that account (ratio, 2026-07-20) can +# send directly too. So the dispatch is: if the account is registered in the +# local signal-cli, send directly; otherwise ssh-relay the send to velox over +# the tailnet. A direct send from a linked device still lands when velox is +# down (the reason ratio was linked). The recipient is Craig's Signal account +# UUID; his phone number reads as unregistered in Signal's directory, so never +# target the number. Verified end to end 2026-07-13 (velox) and 2026-07-20 +# (ratio, direct). +# +# This is the AWAY channel. At his desk, use the desktop channel instead: +# notify info "Title" "Message" --persist +# See protocols.org "Reaching Craig" for choosing between them. +# +# Known caveats (full runbook in rulesets docs/design/): a relay from a +# non-linked machine needs velox up on the tailnet, and each device holding the +# account wants a periodic `receive` (staleness warnings appear otherwise); the +# signal-receive timer handles that. +# +# Source: ~/code/rulesets/claude-templates/bin/agent-text +# Install: make -C ~/code/rulesets install + +SIGNAL_ACCOUNT="+15045173983" +CRAIG_UUID="b1b5601e-6126-47f8-afaa-0a59f5188fde" +VELOX_HOST="velox.tailf3bb8c.ts.net" + +if [ $# -eq 0 ]; then + echo "usage: agent-text <message...>" >&2 + exit 2 +fi + +msg="$*" + +# The account is local if this machine's signal-cli holds it: the registered +# primary (velox) or any linked device. Those send directly. +if signal-cli listAccounts 2>/dev/null | grep -q "$SIGNAL_ACCOUNT"; then + signal-cli -a "$SIGNAL_ACCOUNT" send -m "$msg" "$CRAIG_UUID" + rc=$? +else + # printf %q hardens the message for the remote shell. + ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new \ + "$VELOX_HOST" \ + "signal-cli -a $SIGNAL_ACCOUNT send -m $(printf '%q' "$msg") $CRAIG_UUID" + rc=$? +fi + +if [ "$rc" -ne 0 ]; then + echo "agent-text: phone message failed (velox down or unreachable?); fall back to the desktop channel: notify info 'Message' '<message>' --persist" >&2 +fi +exit "$rc" diff --git a/claude-templates/bin/ai b/claude-templates/bin/ai index cf17875..3440ee2 100755 --- a/claude-templates/bin/ai +++ b/claude-templates/bin/ai @@ -18,6 +18,17 @@ # ollama; model per AI_LOCAL_MODEL, default gpt-oss:120b). # Also settable via AI_RUNTIME. # +# ai --helper <dir> Open a SECOND session in a project that already has a +# live one, under the helper-mode.org role contract: reads +# freely, makes only scoped edits, never mutates git, and +# skips git prep because the primary owns pulls. Runs +# agent-roster first — with no other agent live it warns +# and falls back to a normal primary launch (which does +# run git prep). Run it from a terminal of your own: the +# roster excludes its caller's own process ancestry, so +# invoking it from inside an agent session hides that +# session and silently downgrades to a primary launch. +# # ai --attach Attach to the existing 'ai' session without changes. # # ai -h | --help Show this help. @@ -43,9 +54,18 @@ LOCAL_MODEL="${AI_LOCAL_MODEL:-gpt-oss:120b}" # Strix Halo 2026-07-13). resolve_agent_cmd() { case "$RUNTIME" in - claude) AGENT_BIN="claude"; AGENT_CMD="claude" ;; - codex) AGENT_BIN="codex"; AGENT_CMD="codex" ;; - local) AGENT_BIN="codex"; AGENT_CMD="codex --oss --local-provider=ollama -m $LOCAL_MODEL" ;; + claude) + AGENT_BIN="claude" + AGENT_CMD="claude" + ;; + codex) + AGENT_BIN="codex" + AGENT_CMD="codex" + ;; + local) + AGENT_BIN="codex" + AGENT_CMD="codex --oss --local-provider=ollama -m $LOCAL_MODEL" + ;; *) echo "ai: unknown runtime '$RUNTIME' — valid runtimes: claude, codex, local" >&2 exit 2 @@ -58,7 +78,7 @@ resolve_agent_cmd() { # drives them) and a live ollama answer; a dead server just drops the lines. build_runtime_choices() { command -v claude >/dev/null 2>&1 && echo "claude — Claude Code" - command -v codex >/dev/null 2>&1 && echo "codex — ChatGPT (Codex CLI)" + command -v codex >/dev/null 2>&1 && echo "codex — ChatGPT (Codex CLI)" if command -v codex >/dev/null 2>&1 && command -v ollama >/dev/null 2>&1; then timeout 3 ollama list 2>/dev/null | tail -n +2 | awk 'NF {print "local:" $1 " — ollama"}' fi @@ -73,7 +93,7 @@ pick_runtime() { [ -z "$choice" ] && return 1 case "$choice" in claude*) RUNTIME="claude" ;; - codex*) RUNTIME="codex" ;; + codex*) RUNTIME="codex" ;; local:*) RUNTIME="local" LOCAL_MODEL="${choice#local:}" @@ -101,8 +121,18 @@ build_instructions() { printf 'This is %s %s project. Follow all instructions in .ai/protocols.org.' "$(uname -n)" "$name" } +# The opening line for a helper session. Deliberately does NOT name +# protocols.org: a helper must not run normal startup (pulls, rsync, inbox +# processing all belong to the primary), and helper-mode.org sends it to +# protocols.org itself once the role contract is loaded. +build_helper_instructions() { + local name="$1" + printf 'This is %s %s project. You are a helper session: another agent is already live here. Read and follow .ai/workflows/helper-mode.org — it is your role contract. Do not run the normal startup workflow.' \ + "$(uname -n)" "$name" +} + usage() { - sed -n '2,23p' "$0" | sed 's|^# \?||' + sed -n '2,34p' "$0" | sed 's|^# \?||' exit 0 } @@ -115,6 +145,129 @@ check_deps() { done } +# ---------- pure decision cores (no tmux/git I/O; unit-tested directly) ---------- + +# Decide what a git-prep pass should do from a repo's already-computed state. +# Inputs: has_upstream (1/0), dirty (1/0), ahead, behind. Echoes one of: +# none — no upstream, or in sync: nothing to do +# pull — clean and purely behind: safe to fast-forward +# report — ahead, dirty, or behind-while-dirty: show a summary, don't pull +_git_prep_action() { + local has_upstream="$1" dirty="$2" ahead="$3" behind="$4" + [ "$has_upstream" -eq 1 ] || { + echo none + return + } + if [ "$dirty" -eq 0 ] && [ "$ahead" -eq 0 ] && [ "$behind" -gt 0 ]; then + echo pull + elif [ "$ahead" -gt 0 ] || [ "$behind" -gt 0 ] || [ "$dirty" -eq 1 ]; then + echo report + else + echo none + fi +} + +# Decide what a `--helper` launch actually becomes, from the roster's verdict. +# Input is agent-roster's exit status — 0 alone, 1 others live, 2 unavailable — +# or the literal "absent" when no roster script is installed. Echoes one of: +# helper — confirmed: another agent is live here +# primary — refuted: nobody else is here, so --helper is a no-op +# helper-unverified — the roster couldn't answer +# Unverifiable resolves toward helper on purpose. `--helper` is the operator +# asserting a primary is live, and helper mode is the strictly less destructive +# guess: a helper that turns out to be alone merely does less, while a primary +# that turns out not to be alone runs pulls and rsync under a live session. +_helper_launch_mode() { + case "$1" in + 1) echo helper ;; + 0) echo primary ;; + *) echo helper-unverified ;; + esac +} + +# A helper's agent id: helper-<rand4>, per helper-mode.org's identity rule. +# Four hex digits is enough — the id only has to be unique among the agents +# live in one project at one moment, and the archived session file carries the +# date and time as well. Two draws because bash's RANDOM is 15-bit, so a single +# one would never set the top bit and the first hex digit would always be 0-7. +_helper_id() { + printf 'helper-%04x\n' $(( ((RANDOM << 1) ^ RANDOM) & 0xffff )) +} + +# Reduce an id to the characters session-context-path keeps, so the launcher and +# the path resolver agree on what a given id means. This is also a safety fix, +# not just tidiness: the id is interpolated into the command line typed into the +# pane, so an id carrying a space or a ';' would split the assignment off from +# the command and run something else instead of launching the helper. +# printf without a newline on purpose: tr -c would translate a trailing newline +# into an underscore too, silently appending one to every sanitized id. +_sanitize_agent_id() { + printf '%s' "$1" | tr -c 'A-Za-z0-9._-' '_' + printf '\n' +} + +# Resolve the id for a helper launch: an explicitly-exported one when it is +# free, otherwise a fresh one. +# +# The reuse check is the load-bearing part. A helper's own pane exports +# AI_AGENT_ID, so `ai --helper` invoked from inside a helper inherits its +# parent's id rather than being given one deliberately. Honoring that blindly +# points two live agents at one .ai/session-context.d/<id>.org, which is the +# lost-update collision the whole helper contract exists to avoid. +_resolve_helper_id() { + local dir="$1" + local want="${AI_AGENT_ID:-}" + local tries=0 + + if [ -n "$want" ]; then + want="$(_sanitize_agent_id "$want")" + if [ ! -e "$dir/.ai/session-context.d/$want.org" ]; then + printf '%s\n' "$want" + return + fi + echo "ai: agent id '$want' is already live in $(basename "$dir") — assigning a fresh one" >&2 + fi + + # A minted id gets the same free-anchor check as a supplied one. The odds of + # a chance collision are small, but a guard that only covers the path the + # caller controls leaves the collision it exists to prevent reachable. + # Bounded so a full or unreadable directory can't spin here. + while [ "$tries" -lt 8 ]; do + want="$(_helper_id)" + [ -e "$dir/.ai/session-context.d/$want.org" ] || break + tries=$((tries + 1)) + done + printf '%s\n' "$want" +} + +# Re-order "name<TAB>wid" lines (stdin) into the launcher's window order: +# non-project windows alphabetically, then project windows alphabetically. +# $1 is a newline-separated list of project window names. +# +# A helper window is named "<project>:<agent-id>", so it matches on the prefix +# before the first colon rather than on the whole name. That keeps it sorted +# next to the project it helps instead of landing among the unrelated windows. +_order_windows() { + local project_names="$1" wname wid others="" projects="" + while IFS=$'\t' read -r wname wid; do + [ -z "$wname" ] && continue + if printf '%s\n' "$project_names" | grep -qxF "$wname" || + printf '%s\n' "$project_names" | grep -qxF "${wname%%:*}"; then + projects+="${wname}"$'\t'"${wid}"$'\n' + else + others+="${wname}"$'\t'"${wid}"$'\n' + fi + done + others=$(printf '%s' "$others" | sort -t$'\t' -k1,1f) + projects=$(printf '%s' "$projects" | sort -t$'\t' -k1,1f) + printf '%s\n%s\n' "$others" "$projects" | sed '/^$/d' +} + +# Emit the window id whose name (field 1 of "name<TAB>wid" stdin) equals $1. +_match_window_id() { + awk -F'\t' -v n="$1" '$1 == n { print $2; exit }' +} + # ---------- shared helpers ---------- attach_session() { @@ -138,6 +291,9 @@ create_window() { # Add a directory to candidates only if it's a Claude-template project. maybe_add_candidate() { local dir="$1" + # The "~/" is a deliberate literal display prefix, re-expanded downstream via + # ${c/#\~/$HOME}; it must not expand here, so SC2088 doesn't apply. + # shellcheck disable=SC2088 [ -f "$dir/.ai/protocols.org" ] && candidates+=("~/${dir#"$HOME"/}") } @@ -174,12 +330,36 @@ fetch_candidates() { wait } +# Resolve the shared state gate installed beside this launcher. Keeping the +# policy in one executable prevents startup, the picker, and wrap-up from +# developing different meanings of "safe to sync." +_git_gate_path() { + local gate="${GIT_WORKTREE_GATE:-}" + [ -n "$gate" ] || gate="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/git-worktree-gate" + [ -x "$gate" ] && printf '%s\n' "$gate" +} + +# True (exit 0) when strict wrap would reject the worktree. +_git_is_dirty() { + local dir="$1" gate + gate="$(_git_gate_path)" || return 0 + ! "$gate" strict "$dir" >/dev/null 2>&1 +} + +# True (exit 0) when startup sync must stop. Untracked inbox deliveries are +# safe queue input; every tracked, staged, or other untracked change blocks. +_git_blocks_sync() { + local dir="$1" gate + gate="$(_git_gate_path)" || return 0 + ! "$gate" sync-safe "$dir" >/dev/null 2>&1 +} + # Return " (↑N ↓N dirty)" or " (✓)" if clean. git_status_indicator() { local dir="$1" upstream ahead=0 behind=0 parts=() [ -d "$dir/.git" ] || return 0 - upstream=$(git -C "$dir" rev-parse --abbrev-ref --symbolic-full-name @{u} 2>/dev/null || true) + upstream=$(git -C "$dir" rev-parse --abbrev-ref --symbolic-full-name "@{u}" 2>/dev/null || true) if [ -n "$upstream" ]; then ahead=$(git -C "$dir" rev-list --count "$upstream..HEAD" 2>/dev/null || echo 0) behind=$(git -C "$dir" rev-list --count "HEAD..$upstream" 2>/dev/null || echo 0) @@ -189,10 +369,12 @@ git_status_indicator() { parts+=("no upstream") fi - if ! git -C "$dir" diff --quiet 2>/dev/null \ - || ! git -C "$dir" diff --cached --quiet 2>/dev/null \ - || [ -n "$(git -C "$dir" ls-files --others --exclude-standard 2>/dev/null)" ]; then - parts+=("dirty") + if _git_is_dirty "$dir"; then + if _git_blocks_sync "$dir"; then + parts+=("dirty") + else + parts+=("inbox") + fi fi if [ ${#parts[@]} -gt 0 ]; then @@ -214,27 +396,21 @@ annotate_candidates() { candidates=("${annotated[@]}") } -# Pull if clean, behind, not ahead. No-op otherwise. +# Pull if sync-safe, behind, not ahead. Inbox-only queue input is sync-safe. auto_pull_if_clean() { local dir="$1" upstream ahead behind [ -d "$dir/.git" ] || return 0 + _git_blocks_sync "$dir" && return 0 - if ! git -C "$dir" diff --quiet 2>/dev/null \ - || ! git -C "$dir" diff --cached --quiet 2>/dev/null \ - || [ -n "$(git -C "$dir" ls-files --others --exclude-standard 2>/dev/null)" ]; then - return 0 - fi - - upstream=$(git -C "$dir" rev-parse --abbrev-ref --symbolic-full-name @{u} 2>/dev/null || true) + upstream=$(git -C "$dir" rev-parse --abbrev-ref --symbolic-full-name "@{u}" 2>/dev/null || true) [ -z "$upstream" ] && return 0 ahead=$(git -C "$dir" rev-list --count "$upstream..HEAD" 2>/dev/null || echo 0) - [ "${ahead:-0}" -gt 0 ] 2>/dev/null && return 0 - behind=$(git -C "$dir" rev-list --count "HEAD..$upstream" 2>/dev/null || echo 0) - [ "${behind:-0}" -eq 0 ] 2>/dev/null && return 0 - git -C "$dir" pull --ff-only --quiet 2>/dev/null || true + # dirty=0 and has_upstream=1 are guaranteed by the early returns above. + [ "$(_git_prep_action 1 0 "${ahead:-0}" "${behind:-0}")" = pull ] && + git -C "$dir" pull --ff-only --quiet 2>/dev/null || true } # Strip " (annotation)" suffix from fzf output so downstream gets raw paths. @@ -247,7 +423,7 @@ read_selections() { # Re-order windows: non-project windows at base-index, projects alphabetically after. sort_windows() { - local windows others="" projects="" base_idx project_names="" + local windows base_idx project_names="" ordered base_idx=$(tmux show-option -gv base-index 2>/dev/null || echo 0) windows=$(tmux list-windows -t "$SESSION" -F '#{window_name}'$'\t''#{window_id}') @@ -256,83 +432,95 @@ sort_windows() { project_names+="$(basename "${c/#\~/$HOME}")"$'\n' done - while IFS=$'\t' read -r wname wid; do - [ -z "$wname" ] && continue - if echo "$project_names" | grep -qxF "$wname"; then - projects+="${wname}"$'\t'"${wid}"$'\n' - else - others+="${wname}"$'\t'"${wid}"$'\n' - fi - done <<<"$windows" - others=$(echo -n "$others" | sort -t$'\t' -k1,1f) - projects=$(echo -n "$projects" | sort -t$'\t' -k1,1f) - - local all - all=$(printf '%s\n' "$others" "$projects" | sed '/^$/d') + ordered=$(printf '%s\n' "$windows" | _order_windows "$project_names") + [ -z "$ordered" ] && return 0 + # First pass parks every window above the live range so the second pass can + # reassign the target indices without colliding with a window already there. local i=900 while IFS=$'\t' read -r _n wid; do + [ -z "$wid" ] && continue tmux move-window -s "$wid" -t "$SESSION:$i" i=$((i + 1)) - done <<<"$all" + done <<<"$ordered" i=$base_idx - if [ -n "$others" ]; then - while IFS=$'\t' read -r _n wid; do - tmux move-window -s "$wid" -t "$SESSION:$i" - i=$((i + 1)) - done <<<"$others" - fi - if [ -n "$projects" ]; then - while IFS=$'\t' read -r _n wid; do - tmux move-window -s "$wid" -t "$SESSION:$i" - i=$((i + 1)) - done <<<"$projects" - fi + while IFS=$'\t' read -r _n wid; do + [ -z "$wid" ] && continue + tmux move-window -s "$wid" -t "$SESSION:$i" + i=$((i + 1)) + done <<<"$ordered" } # Find existing window id in ai session by window name; empty if none. find_window_id() { - local name="$1" - tmux list-windows -t "$SESSION" -F '#{window_name}'$'\t''#{window_id}' 2>/dev/null \ - | awk -F'\t' -v n="$name" '$1 == n {print $2; exit}' + tmux list-windows -t "$SESSION" -F '#{window_name}'$'\t''#{window_id}' 2>/dev/null | + _match_window_id "$1" } # Git prep for a single directory. Uses FETCH_HEAD cache to skip back-to-back # fetches. Pulls automatically if clean-and-behind; prints one-line summary # if diverged/dirty/ahead. prep_git_single() { - local dir="$1" gitdir upstream ahead=0 behind=0 dirty="" age fetch_stale=1 parts=() + local dir="$1" gitdir upstream ahead=0 behind=0 dirty=0 age fetch_stale=1 parts=() git -C "$dir" rev-parse --is-inside-work-tree >/dev/null 2>&1 || return 0 gitdir=$(git -C "$dir" rev-parse --git-dir 2>/dev/null) if [ -f "$gitdir/FETCH_HEAD" ]; then - age=$(( $(date +%s) - $(stat -c %Y "$gitdir/FETCH_HEAD" 2>/dev/null || echo 0) )) + age=$(($(date +%s) - $(stat -c %Y "$gitdir/FETCH_HEAD" 2>/dev/null || echo 0))) [ "$age" -lt 600 ] && fetch_stale=0 fi [ "$fetch_stale" -eq 1 ] && git -C "$dir" fetch --quiet 2>/dev/null || true - upstream=$(git -C "$dir" rev-parse --abbrev-ref --symbolic-full-name @{u} 2>/dev/null || true) + upstream=$(git -C "$dir" rev-parse --abbrev-ref --symbolic-full-name "@{u}" 2>/dev/null || true) [ -z "$upstream" ] && return 0 ahead=$(git -C "$dir" rev-list --count "$upstream..HEAD" 2>/dev/null || echo 0) behind=$(git -C "$dir" rev-list --count "HEAD..$upstream" 2>/dev/null || echo 0) + _git_blocks_sync "$dir" && dirty=1 - if ! git -C "$dir" diff --quiet 2>/dev/null \ - || ! git -C "$dir" diff --cached --quiet 2>/dev/null \ - || [ -n "$(git -C "$dir" ls-files --others --exclude-standard 2>/dev/null)" ]; then - dirty="dirty" - fi + case "$(_git_prep_action 1 "$dirty" "${ahead:-0}" "${behind:-0}")" in + pull) + echo "ai: pulling $behind commit(s) from $upstream..." >&2 + git -C "$dir" pull --ff-only --quiet + ;; + report) + [ "${ahead:-0}" -gt 0 ] && parts+=("↑$ahead") + [ "${behind:-0}" -gt 0 ] && parts+=("↓$behind") + [ "$dirty" -eq 1 ] && parts+=("dirty") + echo "ai: $(basename "$dir") — ${parts[*]}" >&2 + ;; + esac +} - if [ -z "$dirty" ] && [ "${ahead:-0}" -eq 0 ] && [ "${behind:-0}" -gt 0 ]; then - echo "ai: pulling $behind commit(s) from $upstream..." >&2 - git -C "$dir" pull --ff-only --quiet - elif [ "${ahead:-0}" -gt 0 ] || [ "${behind:-0}" -gt 0 ] || [ -n "$dirty" ]; then - [ "${ahead:-0}" -gt 0 ] && parts+=("↑$ahead") - [ "${behind:-0}" -gt 0 ] && parts+=("↓$behind") - [ -n "$dirty" ] && parts+=("$dirty") - echo "ai: $(basename "$dir") — ${parts[*]}" >&2 +# Run the project's agent-roster and turn its verdict into a launch decision. +# The decision is the only thing on stdout; warnings go to stderr so callers +# can capture one without the other. +_resolve_helper_launch() { + # Two statements on purpose: a name assigned in a `local` is not yet visible + # to a later assignment in that same `local`, so building the roster path in + # this line would read the CALLER's $dir — right only by coincidence. + local dir="$1" + local roster="$dir/.ai/scripts/agent-roster" rc decision + if [ -x "$roster" ]; then + # The roster prints the other agents it found; only its exit code matters + # here, and its stdout must not reach a --print-launch caller's output. + "$roster" "$dir" >/dev/null 2>&1 + rc=$? + else + rc=absent fi + + decision="$(_helper_launch_mode "$rc")" + case "$decision" in + primary) + echo "ai: --helper found no other agent live in $(basename "$dir") — opening a normal primary session instead" >&2 + ;; + helper-unverified) + echo "ai: could not verify another agent is live in $(basename "$dir") — roster unavailable; opening a helper anyway" >&2 + ;; + esac + echo "$decision" } # ---------- modes ---------- @@ -349,7 +537,10 @@ attach_mode() { # Open a single project (or focus existing window). single_mode() { local arg="$1" dir name wid existing - dir="$(cd "$arg" 2>/dev/null && pwd)" || { echo "ai: cannot access '$arg'" >&2; return 1; } + dir="$(cd "$arg" 2>/dev/null && pwd)" || { + echo "ai: cannot access '$arg'" >&2 + return 1 + } if [ ! -f "$dir/.ai/protocols.org" ]; then echo "ai: $dir has no .ai/protocols.org — not a Claude-template project" >&2 @@ -385,6 +576,55 @@ single_mode() { attach_session } +# Open a helper session: a second agent in a project that already has a live +# one. Two deliberate differences from single_mode. It never focuses an +# existing window — a second session is the entire point, and focusing the +# primary's window is the one outcome that can't be what was asked for. And it +# never runs git prep, because every pull belongs to the primary under the +# helper contract. +helper_mode() { + local arg="$1" dir name id wid wname decision instructions + dir="$(cd "$arg" 2>/dev/null && pwd)" || { + echo "ai: cannot access '$arg'" >&2 + return 1 + } + + if [ ! -f "$dir/.ai/protocols.org" ]; then + echo "ai: $dir has no .ai/protocols.org — not an agent-template project" >&2 + return 1 + fi + + name="$(basename "$dir")" + + # Nobody else is here, so there is nothing to be a helper to. Fall through to + # the normal launch rather than opening a crippled session. + decision="$(_resolve_helper_launch "$dir")" + if [ "$decision" = primary ]; then + single_mode "$arg" + return $? + fi + + id="$(_resolve_helper_id "$dir")" + wname="$name:$id" + instructions=$(build_helper_instructions "$name") + + if tmux has-session -t "$SESSION" 2>/dev/null; then + wid=$(tmux new-window -a -t "$SESSION:{end}" -n "$wname" -c "$dir" -P -F '#{window_id}') + sleep 0.1 + else + wid=$(tmux new-session -d -s "$SESSION" -n "$wname" -c "$dir" -P -F '#{window_id}') + fi + + # The id rides in the launched process's environment, which is what + # session-context-path reads to resolve .ai/session-context.d/<id>.org. + tmux send-keys -t "$wid" \ + "${LAUNCH_PREFIX}AI_AGENT_ID=$id AI_HELPER=1 $AGENT_CMD \"$instructions\"" Enter + + sort_windows + tmux select-window -t "$wid" + attach_session +} + # Multi-select via fzf (the original aix flow). multi_mode() { local filtered=() selections first_wid="" @@ -442,7 +682,7 @@ multi_mode() { dir="${entry/#\~/$HOME}" name="$(basename "$dir")" auto_pull_if_clean "$dir" - create_window "$dir" "$name" > /dev/null + create_window "$dir" "$name" >/dev/null done else # Add windows to existing session @@ -466,75 +706,119 @@ multi_mode() { # opening line with no tmux or fzf involved. print_launch_mode() { local arg="$1" dir name - dir="$(cd "$arg" 2>/dev/null && pwd)" || { echo "ai: cannot access '$arg'" >&2; exit 1; } + dir="$(cd "$arg" 2>/dev/null && pwd)" || { + echo "ai: cannot access '$arg'" >&2 + exit 1 + } if [ ! -f "$dir/.ai/protocols.org" ]; then echo "ai: $dir has no .ai/protocols.org — not an agent-template project" >&2 exit 1 fi name="$(basename "$dir")" + + # The roster runs here too, so the printed line reflects the decision a real + # run would make — including the downgrade to a primary launch. + if [ -n "$HELPER_MODE" ] && [ "$(_resolve_helper_launch "$dir")" != primary ]; then + printf 'AI_AGENT_ID=%s AI_HELPER=1 %s "%s"\n' \ + "$(_resolve_helper_id "$dir")" "$AGENT_CMD" "$(build_helper_instructions "$name")" + exit 0 + fi + printf '%s "%s"\n' "$AGENT_CMD" "$(build_instructions "$name")" exit 0 } # ---------- dispatch ---------- -print_launch="" -runtime_explicit="${AI_RUNTIME:+1}" -while [ $# -gt 0 ]; do - case "$1" in - -h|--help) - usage - ;; - --runtime) - [ -z "${2:-}" ] && { echo "ai: --runtime needs a value — valid runtimes: claude, codex, local" >&2; exit 2; } - RUNTIME="$2" - runtime_explicit=1 - shift 2 - ;; - --runtime=*) - RUNTIME="${1#--runtime=}" - runtime_explicit=1 - shift - ;; - --print-launch) - print_launch=1 - shift +# Argument parsing + mode dispatch. Wrapped so the file can be sourced (by the +# launcher's bats tests) to exercise individual functions without running a +# real launch. When executed as a program, BASH_SOURCE[0] equals $0 and the +# dispatch runs exactly as before; when sourced, it's skipped. +main() { + print_launch="" + HELPER_MODE="" + runtime_explicit="${AI_RUNTIME:+1}" + while [ $# -gt 0 ]; do + case "$1" in + -h | --help) + usage + ;; + --helper) + HELPER_MODE=1 + shift + ;; + --runtime) + [ -z "${2:-}" ] && { + echo "ai: --runtime needs a value — valid runtimes: claude, codex, local" >&2 + exit 2 + } + RUNTIME="$2" + runtime_explicit=1 + shift 2 + ;; + --runtime=*) + RUNTIME="${1#--runtime=}" + runtime_explicit=1 + shift + ;; + --print-launch) + print_launch=1 + shift + ;; + --print-runtimes) + build_runtime_choices + exit 0 + ;; + *) + break + ;; + esac + done + + resolve_agent_cmd + + # A helper is always scoped to one named project. There is no roster to check + # and no primary to help without one, so this can't fall back to the picker. + if [ -n "$HELPER_MODE" ] && [ -z "${1:-}" ]; then + echo "ai: --helper needs a project directory" >&2 + exit 2 + fi + + if [ -n "$print_launch" ]; then + [ $# -eq 0 ] && { + echo "ai: --print-launch needs a project directory" >&2 + exit 2 + } + print_launch_mode "$1" + fi + + case "${1:-}" in + --attach) + check_deps + attach_mode ;; - --print-runtimes) - build_runtime_choices - exit 0 + "") + # Bare `ai`: pick the agent first (skipped when --runtime or AI_RUNTIME + # already chose), then the familiar project multi-select. + if [ -z "$runtime_explicit" ]; then + pick_runtime || exit 0 + fi + check_deps + multi_mode ;; *) - break + check_deps + for arg in "$@"; do + if [ -n "$HELPER_MODE" ]; then + helper_mode "$arg" + else + single_mode "$arg" + fi + done ;; esac -done - -resolve_agent_cmd +} -if [ -n "$print_launch" ]; then - [ $# -eq 0 ] && { echo "ai: --print-launch needs a project directory" >&2; exit 2; } - print_launch_mode "$1" +if [ "${BASH_SOURCE[0]}" = "${0}" ]; then + main "$@" fi - -case "${1:-}" in - --attach) - check_deps - attach_mode - ;; - "") - # Bare `ai`: pick the agent first (skipped when --runtime or AI_RUNTIME - # already chose), then the familiar project multi-select. - if [ -z "$runtime_explicit" ]; then - pick_runtime || exit 0 - fi - check_deps - multi_mode - ;; - *) - check_deps - for arg in "$@"; do - single_mode "$arg" - done - ;; -esac diff --git a/claude-templates/bin/git-worktree-gate b/claude-templates/bin/git-worktree-gate new file mode 100755 index 0000000..e453fd1 --- /dev/null +++ b/claude-templates/bin/git-worktree-gate @@ -0,0 +1,185 @@ +#!/usr/bin/env bash +# git-worktree-gate — one definition of safe Git state for startup and wrap. +# +# Modes: +# strict [DIR] Require an entirely empty worktree. +# sync-safe [DIR] Permit untracked inbox/ deliveries, but nothing else. +# certify [DIR] Strict-check, then record the verified HEAD in the git dir. +# verify [DIR] Strict-check and require the recorded HEAD to still match. +# +# Ignored files are deliberately outside Git's clean-worktree contract. + +set -u + +mode="${1:-}" +repo="${2:-.}" + +usage() { + echo "usage: git-worktree-gate {strict|sync-safe|certify|verify} [DIR]" >&2 + exit 2 +} + +case "$mode" in + strict|sync-safe|certify|verify) ;; + *) usage ;; +esac + +root="$(git -C "$repo" rev-parse --show-toplevel 2>/dev/null)" || { + echo "git-worktree-gate: $repo is not inside a Git worktree" >&2 + exit 2 +} +gitdir="$(git -C "$root" rev-parse --absolute-git-dir 2>/dev/null)" || { + echo "git-worktree-gate: cannot resolve the Git directory for $root" >&2 + exit 2 +} +certificate="$gitdir/ai-wrap-clean" + +quote_path() { + printf '%q' "$1" +} + +operation_in_progress() { + local marker + for marker in MERGE_HEAD CHERRY_PICK_HEAD REVERT_HEAD BISECT_LOG; do + [ -e "$gitdir/$marker" ] && { + printf '%s' "$marker" + return 0 + } + done + for marker in rebase-merge rebase-apply sequencer; do + [ -d "$gitdir/$marker" ] && { + printf '%s' "$marker" + return 0 + } + done + return 1 +} + +describe_entry() { + local xy="$1" path="$2" original="${3:-}" + local index="${xy:0:1}" worktree="${xy:1:1}" label="" + + if [ "$xy" = "??" ]; then + label="untracked; add and commit it, move it outside the repository, or remove it if unwanted" + elif [ "$xy" = "!!" ]; then + label="ignored" + elif [[ "$xy" = *U* || "$xy" = "AA" || "$xy" = "DD" ]]; then + label="unmerged; resolve the conflict and commit the result" + elif [ "$index" != " " ] && [ "$worktree" != " " ]; then + label="staged and unstaged changes; review both layers, then commit or restore them" + elif [ "$index" != " " ]; then + label="staged change; commit it or unstage and restore it" + else + label="unstaged tracked change; commit it or restore it" + fi + + printf ' %s ' "$xy" + quote_path "$path" + if [ -n "$original" ]; then + printf ' (from ' + quote_path "$original" + printf ')' + fi + printf ' — %s\n' "$label" +} + +check_state() { + local policy="$1" xy path original="" blocked=0 op="" + local status_tmp="" status_err="" status_detail="" + local -a report=() + + if op="$(operation_in_progress)"; then + report+=(" Git operation in progress: $op — finish or abort it") + blocked=1 + fi + + status_tmp="$(mktemp "$gitdir/ai-worktree-status.tmp.XXXXXX")" || { + echo "wrap blocked: cannot allocate a Git-state check file" >&2 + return 1 + } + status_err="$(mktemp "$gitdir/ai-worktree-status.err.XXXXXX")" || { + rm -f "$status_tmp" + echo "wrap blocked: cannot allocate a Git-state error file" >&2 + return 1 + } + + if ! git -C "$root" status --porcelain=v1 -z \ + --untracked-files=all --ignore-submodules=none \ + >"$status_tmp" 2>"$status_err"; then + status_detail="$(head -1 "$status_err")" + [ -n "$status_detail" ] || status_detail="unknown Git error" + report+=(" git status failed — $status_detail") + blocked=1 + else + while IFS= read -r -d '' entry; do + xy="${entry:0:2}" + path="${entry:3}" + original="" + if [[ "${xy:0:1}" = "R" || "${xy:0:1}" = "C" ]]; then + IFS= read -r -d '' original || true + fi + + if [ "$policy" = "sync-safe" ] \ + && [ "$xy" = "??" ] \ + && [[ "$path" = inbox/* ]]; then + continue + fi + + report+=("$(describe_entry "$xy" "$path" "$original")") + blocked=1 + done <"$status_tmp" + fi + rm -f "$status_tmp" "$status_err" + + if [ "$blocked" -ne 0 ]; then + if [ "$policy" = "sync-safe" ]; then + echo "sync blocked: rulesets has changes other than untracked inbox deliveries" >&2 + else + echo "wrap blocked: Git worktree is not completely clean" >&2 + fi + printf '%s\n' "${report[@]}" >&2 + return 1 + fi + return 0 +} + +case "$mode" in + strict) + check_state strict + ;; + sync-safe) + check_state sync-safe + ;; + certify) + check_state strict || exit 1 + head="$(git -C "$root" rev-parse HEAD 2>/dev/null)" || { + echo "wrap blocked: cannot resolve HEAD" >&2 + exit 1 + } + tmp="$(mktemp "$gitdir/ai-wrap-clean.tmp.XXXXXX")" || exit 1 + chmod 600 "$tmp" + { + printf 'head=%s\n' "$head" + printf 'root=%s\n' "$root" + } >"$tmp" + mv "$tmp" "$certificate" + ;; + verify) + check_state strict || exit 1 + [ -f "$certificate" ] || { + echo "wrap blocked: no clean-tree certificate exists; rerun the final wrap verification" >&2 + exit 1 + } + certified_head="$(sed -n 's/^head=//p' "$certificate" | head -1)" + certified_root="$(sed -n 's/^root=//p' "$certificate" | head -1)" + current_head="$(git -C "$root" rev-parse HEAD 2>/dev/null)" || exit 1 + [ "$certified_root" = "$root" ] || { + echo "wrap blocked: clean-tree certificate belongs to a different worktree" >&2 + exit 1 + } + [ -n "$certified_head" ] && [ "$certified_head" = "$current_head" ] || { + echo "wrap blocked: HEAD changed after clean-tree certification; rerun the final wrap verification" >&2 + exit 1 + } + ;; +esac diff --git a/claude-templates/bin/install-ai b/claude-templates/bin/install-ai index 88cb8e5..3283c4a 100755 --- a/claude-templates/bin/install-ai +++ b/claude-templates/bin/install-ai @@ -2,7 +2,7 @@ # install-ai — PATH-facing launcher for the fresh-project bootstrapper. # # make install symlinks this into ~/.local/bin/install-ai (same bin loop that -# links `ai` and `agent-page`), so `install-ai [--track|--gitignore] [PROJECT]` +# links `ai` and `agent-text`), so `install-ai [--track|--gitignore] [PROJECT]` # runs from anywhere. The real logic lives in scripts/install-ai.sh; this # resolves its own location through the ~/.local/bin symlink and execs that # script by its true repo path, so the script's own repo-root computation diff --git a/docs/design/2026-05-28-pattern-catalog-prompt-labels-and-defaults.org b/docs/design/2026-05-28-pattern-catalog-prompt-labels-and-defaults.org index cc2cb77..9ee29d6 100644 --- a/docs/design/2026-05-28-pattern-catalog-prompt-labels-and-defaults.org +++ b/docs/design/2026-05-28-pattern-catalog-prompt-labels-and-defaults.org @@ -28,7 +28,7 @@ A generic =pearl--with-sentinel SENTINEL CANDIDATES= helper lets each call site ** Why for the catalog -This is the "no hidden affordances" pattern from the earlier note ([[file:2026-05-28-0003-from-pearl-rulesets-followup-no-empty-input.org]]) sharpened with a second rule: *if the affordance is visible, its label has to match what picking it does*. Visibility without accuracy is its own problem. A label that says "none" when the behavior is "any" is no better than an invisible empty-input idiom — both leave the user holding the wrong model. +This is the "no hidden affordances" pattern from the earlier note (a processed pearl-rulesets inbox item, since removed) sharpened with a second rule: *if the affordance is visible, its label has to match what picking it does*. Visibility without accuracy is its own problem. A label that says "none" when the behavior is "any" is no better than an invisible empty-input idiom — both leave the user holding the wrong model. Catalog shape suggestion: this is the same principle as Pattern 3, in a follow-up form. Either a single entry that captures both halves (visible + accurate) or two cross-linked entries. diff --git a/working/inbox-zero-phase-e/proposed.diff b/docs/design/2026-06-16-inbox-zero-phase-e-proposal.diff index e3d8ee8..e3d8ee8 100644 --- a/working/inbox-zero-phase-e/proposed.diff +++ b/docs/design/2026-06-16-inbox-zero-phase-e-proposal.diff diff --git a/working/inbox-zero-phase-e/proposed-inbox-zero.org b/docs/design/2026-06-16-inbox-zero-phase-e-proposal.org index 5fa7e12..5fa7e12 100644 --- a/working/inbox-zero-phase-e/proposed-inbox-zero.org +++ b/docs/design/2026-06-16-inbox-zero-phase-e-proposal.org diff --git a/working/inbox-zero-phase-e/sender-note.org b/docs/design/2026-06-16-inbox-zero-phase-e-sender-note.org index 08e7650..08e7650 100644 --- a/working/inbox-zero-phase-e/sender-note.org +++ b/docs/design/2026-06-16-inbox-zero-phase-e-sender-note.org diff --git a/docs/design/2026-07-15-subprojects-convention-home-instance.org b/docs/design/2026-07-15-subprojects-convention-home-instance.org index a1c4397..b031a41 100644 --- a/docs/design/2026-07-15-subprojects-convention-home-instance.org +++ b/docs/design/2026-07-15-subprojects-convention-home-instance.org @@ -195,7 +195,7 @@ Every real use emits a signal, and the signal is *captured*, not silently acted on: - Brief answers the question → *hit*. - Brief is missing something, stale, or wrong → *miss*, logged in - [[file:subprojects-log.org][subprojects-log.org]]. + subprojects-log.org (a home-side artifact; no rulesets file). - Each miss drives *two* updates: *local* (fix that brief) and, if the same kind of miss recurs across subprojects, *structural* (fix this spine/spec, not just the one file). The structural half is the actual self-improvement — diff --git a/docs/design/2026-07-20-signal-pager-runbook.org b/docs/design/2026-07-20-signal-pager-runbook.org new file mode 100644 index 0000000..f31ed18 --- /dev/null +++ b/docs/design/2026-07-20-signal-pager-runbook.org @@ -0,0 +1,198 @@ +#+TITLE: Signal Pager Runbook +#+AUTHOR: Craig Jennings +#+DATE: 2026-07-20 + +The operational reference for the agent pager — how a page reaches Craig's +phone, how his replies come back, how the account stays healthy, and the +signal-cli setup behind it. This is the Signal successor to the retired ntfy +runbook. Canonical home is rulesets because the pager is cross-machine tooling. + +* What the pager is + +One Signal identity, =+15045173983=, registered in *velox's* signal-cli +(account file 465310, velox is the primary device). It is a dedicated pager +number, not Craig's personal Signal. Pages go *from* that identity *to* Craig's +own Signal account, which fires a normal mobile push on his phone. + +As of 2026-07-20 the identity spans two devices: velox (primary) and ratio +(linked device "ratio-pager"). Any machine holding the account sends directly; +a machine that doesn't relays to velox over the tailnet. + +Two constants the tooling depends on: + +- Pager account: =+15045173983= (primary on velox, linked on ratio). +- Recipient: Craig's Signal account UUID =b1b5601e-6126-47f8-afaa-0a59f5188fde=. + His phone *number* reads as unregistered in Signal's directory — always + target the UUID, never the number. + +velox is the laptop that travels with Craig, so the pager account rides with +him; ratio holds it too, so a page still lands when velox is down. + +* Choosing a channel + +Two trigger words, two channels, and both work from any agent runtime (nothing +here is Claude-specific). protocols.org "Reaching Craig" is the short version +pointed at every project; this runbook is the full one for the Signal side. + +- *"page me"* — desktop notification, stays up until dismissed: + + #+begin_src bash + notify info "Title" "Message" --persist + #+end_src + +- *"text me"* — the phone, over Signal: + + #+begin_src bash + agent-text "Message for Craig's phone" + #+end_src + +- *"text and page me"* — both. The default when a run can't tell whether he's + away: the desktop one is free and the phone one reaches him if he is. + +* Sending a text + +=agent-text= (shipped at =claude-templates/bin/agent-text=, installed to +=~/.local/bin= by =make -C ~/code/rulesets install=) is the interface. It hides +the machine topology by checking whether the account is registered in the +local signal-cli: + +- If the account is local (velox's primary or a linked device like ratio), it + sends directly — no velox dependency. +- Otherwise it ssh-relays the send to velox over the tailnet. +- On failure (velox down or unreachable from a non-linked machine) it prints the + desktop fallback line and exits non-zero, so a caller can tell the page did not + land. + +The raw command it runs, for reference or a manual send from velox: + +#+begin_src bash +signal-cli -a +15045173983 send -m "your message" b1b5601e-6126-47f8-afaa-0a59f5188fde +#+end_src + +From another machine, the same send relayed over the tailnet: + +#+begin_src bash +ssh velox.tailf3bb8c.ts.net \ + "signal-cli -a +15045173983 send -m 'your message' b1b5601e-6126-47f8-afaa-0a59f5188fde" +#+end_src + +Prefer =agent-text= over the raw command — it hardens the message for the remote +shell and handles the fallback. Reach for the raw form only when debugging. + +* Reading replies + +Craig replies to a page straight from Signal on his phone. The reply is a normal +data message *to* the pager account, so it is waiting in the pager's inbound +queue until something receives it. + +Drain the queue and read what is there: + +#+begin_src bash +# On velox: +signal-cli -a +15045173983 receive --timeout 10 +# From another machine: +ssh velox.tailf3bb8c.ts.net "signal-cli -a +15045173983 receive --timeout 10" +#+end_src + +=receive= prints every queued envelope and exits 0 once the queue drains or the +timeout elapses. A text reply from Craig arrives as an envelope from his UUID +carrying a =Body:= line — that line is the reply text. Most envelopes are +delivery/read receipts and typing indicators (no =Body:=); the reply you want is +the data message with body text. For a script that waits on a reply, add +=--send-read-receipts= so his phone shows the page was read, and parse stdout for +the =Body:= line on an envelope from =b1b5601e-…=. + +Note: =receive= is destructive — it consumes the queue. Whatever drains the +queue (an on-demand read, or the warm-keeping timer below) is what sees the +reply, and it is seen once. An agent that pages and then waits for an answer +should do its own =receive= rather than race the timer. + +* Keeping the account warm (receive timer) + +Signal expects a registered account to receive regularly. Left alone, the pager +account drifts stale — signal-cli warns "Messages have been last received N days +ago" (observed at 47 days on 2026-07-20 before a manual drain reset it). A stale +account is a reliability risk on the one channel that reaches Craig when he is +away. + +The fix mirrors roam-sync: a systemd user timer that drains the queue on a +cadence, keeping the account warm and, as a bonus, picking up async replies. With +the account linked on both machines, each device wants its own regular receive, +so the timer runs on *both* velox and ratio (the shared =common= dotfiles +package, same home as roam-sync). + +- Script: =scripts/signal-receive.sh= (rulesets, so both machines get it on + =git pull=). It no-ops cleanly on a machine that lacks the account. +- Units: =scripts/signal-receive.service= + =.timer= (reference copies under + =scripts/systemd/=; the stowed copies live in =common/.config/systemd/user/= + of the dotfiles repo, so both machines get them). +- Cadence: every 15 minutes (=OnUnitActiveSec=15min=), matching roam-sync. + +Enable on each machine (one-time, per daily-drivers.md's one-time-setup class): + +#+begin_src bash +# After the dotfiles + rulesets pull, on each daily driver: +systemctl --user daemon-reload +systemctl --user enable --now signal-receive.timer +systemctl --user status signal-receive.service # confirm a clean receive +#+end_src + +* signal-cli setup notes + +- *Version:* signal-cli 0.14.5 on velox (2026-07-20). +- *Accounts:* velox's signal-cli holds the pager identity =+15045173983= as the + registered primary (account file 465310). ratio's signal-cli holds two + accounts: Craig's personal number =+15103169357= (its own primary, + note-to-self only — no phone push) and the pager identity as a *linked device* + (Device 2, "ratio-pager", linked 2026-07-20). Both accounts coexist; target + the pager with =-a +15045173983=. A future daily driver joins the same way. +- *signal-mcp:* on velox, Claude sessions may also expose a =signal-mcp= tool + (=send_message_to_user=, same pager identity) configured in velox's global + =~/.claude.json=. It works there but is invisible from any other machine and + from non-Claude runtimes, so =agent-text= is the portable habit. The old + =page-signal= shell script was removed 2026-06-12 — do not resurrect it. +- *Linking a device:* to add a second signal-cli as a linked device of the pager + account (see the open decision below), provision it from the new machine and + approve the link from the account holder: + + #+begin_src bash + # On the new machine — prints a tsdevice:/ URI (render as QR to approve): + signal-cli link -n "ratio-pager" + # Approve from velox (the primary device): + signal-cli -a +15045173983 addDevice --uri "tsdevice:/?uuid=…" + #+end_src + +* Decision — linked device (2026-07-20) + +The topology question — ssh-relay only vs. registering daily drivers as linked +devices — was decided in favor of linked devices. ssh-relay only was simpler +(one identity, one receive point) but had a single point of failure: a page +failed when velox was down or off the tailnet. + +Registering ratio as a linked device removes that: ratio sends directly, so a +page lands even when velox is down. The costs, both paid: linked-device +provisioning per machine (the =link= / =addDevice= handshake above), and each +device wanting its own regular =receive= — so the warm-keeping timer moved from a +velox-only home to the shared =common= package, running on both. + +Adding another daily driver later is the same handshake plus a dotfiles stow; +the timer and =agent-text= already generalize to "any machine holding the +account." + +* History + +- 2026-07-04 — home retired ntfy (self-hosted on ratio) and tore it down, + switching agent paging to Signal. Handoff to rulesets to document and own. +- 2026-07-13 — reconciled to one pager identity on velox; =agent-page= shipped + (direct on velox, ssh-relay elsewhere, desktop fallback); protocols.org "Paging + Craig" rewritten around the two channels. +- 2026-07-20 — this runbook; receive-timer script + units added; a manual drain + cleared the 47-day staleness live; ratio linked as a device of the pager + account and its direct send verified; the tool (still named =agent-page= that + morning) generalized to send directly from any machine holding the account; + receive timer moved to the shared =common= package and enabled on both machines. +- 2026-07-20 (later) — notification vocabulary split: "page me" is the desktop + channel, "text me" is Signal, "text and page me" is both. The tool was renamed + =agent-page= → =agent-text= to match, with a deprecated =agent-page= shim + delegating to it. protocols.org section renamed "Paging Craig" → "Reaching + Craig". diff --git a/docs/specs/2026-06-16-autonomous-batch-execution-spec.org b/docs/specs/2026-06-16-autonomous-batch-execution-spec.org index fe3458b..a42adc3 100644 --- a/docs/specs/2026-06-16-autonomous-batch-execution-spec.org +++ b/docs/specs/2026-06-16-autonomous-batch-execution-spec.org @@ -21,7 +21,7 @@ |----------+--------------------------------------------------------------------| | Date | 2026-06-16 | |----------+--------------------------------------------------------------------| -| Related | [[file:../../working/inbox-zero-phase-e/proposed-inbox-zero.org][Phase E proposal]]; [[file:../design/2026-06-15-fix-speedrun-workflow-proposal.org][speedrun proposal]] | +| Related | [[file:../design/2026-06-16-inbox-zero-phase-e-proposal.org][Phase E proposal]]; [[file:../design/2026-06-15-fix-speedrun-workflow-proposal.org][speedrun proposal]] | |----------+--------------------------------------------------------------------| * Summary @@ -375,16 +375,16 @@ Verification is by invocation against a project's real =todo.org=: run the loop * References / Appendix -- [[file:../../working/inbox-zero-phase-e/proposed-inbox-zero.org][Phase E proposal (inbox-zero stopgap)]] and [[file:../../working/inbox-zero-phase-e/sender-note.org][its sender note with the 5 open questions]]. +- [[file:../design/2026-06-16-inbox-zero-phase-e-proposal.org][Phase E proposal (inbox-zero stopgap)]] and [[file:../design/2026-06-16-inbox-zero-phase-e-sender-note.org][its sender note with the 5 open questions]]. - [[file:../design/2026-06-15-fix-speedrun-workflow-proposal.org][speedrun proposal]] (file retains its original on-disk name pending a rename pass). -- [[file:../../.ai/workflows/inbox-zero.org][inbox-zero.org (canonical, A-D)]] — the routing workflow this feature decouples from. +- [[file:../../.ai/workflows/inbox.org][inbox.org (canonical A-D; was inbox-zero.org)]] — the routing workflow this feature decouples from. - =~/code/rulesets/claude-rules/knowledge-base.md= — the org-roam write contract the synthesis step follows. * Review and iteration history ** 2026-06-16 Tue — author - What: initial draft reconciling the Phase E and fix-speedrun proposals into one work-the-backlog.org feature, plus the effectiveness-measurement instrumentation. - Why: two overlapping proposals arrived within a day; building them separately would duplicate the execution loop and let it drift. Craig also asked explicitly for measurement + org-roam synthesis. -- Artifacts: this spec; the two source proposals under docs/design/ and working/inbox-zero-phase-e/. +- Artifacts: this spec; the two source proposals under docs/design/ (the Phase E proposal, diff, and sender note now filed there). ** 2026-06-28 Sun — revision (Craig) - What: removed the task-size gate (size no longer defers; large tasks decompose into per-commit chunks); recast the act-vs-file rule as a crisp four-item defer checklist keyed on test-writability; added crisp =:solo:= / =:quick:= definitions destined for =todo-format.md= and made their assessment mandatory in task-review + task-audit; added the speedrun's pre-flight decision-gathering step (batch the quick questions up front, "skip this" drops a task, then run hands-off); renamed "fix speedrun" → "no-approvals speedrun" in prose. Status stays draft pending ratification of the revised decisions. - Why: the original criteria were adjectives, not checkable; the size gate forced Craig to stay at his desk for anything non-trivial, defeating the away-from-desk use case; and decision-needing tasks were over-deferred when many need only a quick upfront answer. diff --git a/docs/specs/2026-07-14-sentry-workflow-spec.org b/docs/specs/2026-07-14-sentry-workflow-spec.org index fe1c155..307d0be 100644 --- a/docs/specs/2026-07-14-sentry-workflow-spec.org +++ b/docs/specs/2026-07-14-sentry-workflow-spec.org @@ -4,15 +4,18 @@ #+TODO: TODO | DONE #+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED -* DRAFT sentry workflow +* IMPLEMENTED sentry workflow :PROPERTIES: :ID: f6c51f27-d7a2-4b63-9ff9-5ba005a66dfb :END: +- 2026-07-19 Sun @ 05:04:00 -0500 — IMPLEMENTED: all four phases built, committed, and pushed (agent-lock a8b6cf4, engine ccc9c26, companions c6383e9). Full suite green throughout. The overnight live trial is handed to Craig as a structured manual-testing task; its findings file as follow-up tasks, not a build gate. +- 2026-07-19 Sun @ 04:35:57 -0500 — DOING: build started. Decomposed into the four implementation phases as todo.org build tasks under the sentry parent (SPEC_ID-bound); running in no-approvals + auto-flush mode. +- 2026-07-19 Sun @ 04:35:57 -0500 — READY: Craig completed his deep read and approved the spec for build. Gate passed. - 2026-07-14 Tue @ 02:03:28 -0500 — all 12 review findings dispositioned live with Craig and folded into the design; decisions now 10/10; still DRAFT pending Craig's deep read. - 2026-07-14 Tue @ 00:52:03 -0500 — drafted, all nine design decisions resolved live with Craig during the authoring session. * Metadata -| Status | draft | +| Status | implemented | |----------+------------------------------------------| | Owner | Craig Jennings | |----------+------------------------------------------| diff --git a/docs/specs/2026-07-20-silent-until-signal-monitors-spec.org b/docs/specs/2026-07-20-silent-until-signal-monitors-spec.org new file mode 100644 index 0000000..c15d10f --- /dev/null +++ b/docs/specs/2026-07-20-silent-until-signal-monitors-spec.org @@ -0,0 +1,126 @@ +#+TITLE: Silent-Until-Signal Monitor Loops — Spec +#+AUTHOR: Craig Jennings +#+DATE: 2026-07-20 +#+TODO: TODO | DONE +#+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED + +* IMPLEMENTED silent-until-signal monitor loops +:PROPERTIES: +:ID: af592bd6-d3e6-47e2-8804-2a287b4d9303 +:END: +- 2026-07-20 Mon @ 13:34:53 -0500 — READY (Craig approved after his read) → building Phases 2-5 immediately → IMPLEMENTED. Phase 2 (auto triage-intake heartbeat), Phase 3 (auto inbox-zero heartbeat), Phase 4 (shared-policy home: the spec is the single definition, each workflow states its own heartbeat and references it), Phase 5 (manual-testing checklist). All phases shipped. +- 2026-07-20 Mon @ 12:30:00 -0500 — Phase 1 (sentry quiet-fire heartbeat) implemented ahead of the full READY gate, at Craig's direction — a quiet sentry fire now collapses to =sentry at HH:MM: nothing=, and the "no silent skip" discipline is reconciled (the heartbeat is the explicit "nothing" record). Phases 2-5 (triage, inbox, shared-policy home, verification) still pending Craig's deep read. Spec stays DRAFT. +- 2026-07-20 Mon @ 12:14:20 -0500 — drafted during the morning session. Design decisions resolved live with Craig off the .emacs.d proposal and the sentry live-trial evidence. DRAFT pending his deep read. + +* Metadata +| Status | draft | +|----------+--------------------------------------------------------------| +| Owner | Craig Jennings | +|----------+--------------------------------------------------------------| +| Reviewer | (spec-review, next session) | +|----------+--------------------------------------------------------------| +| Related | [[file:../../todo.org::*Silent-until-signal for in-session monitor loops][todo.org — silent-until-signal task]] | +|----------+--------------------------------------------------------------| + +* Summary + +An in-session monitor loop (sentry, auto triage-intake, auto inbox-zero) fires the model once per interval. Today each fire narrates a full turn even when nothing changed, so a long session fills with walls of "quiet fire" / "no new items" output. The fix is a policy, not a mechanism: every monitor fire does cheap detection first, and on an empty check collapses to a single labelled heartbeat line ("sentry at 03:20: nothing") and stops. Only a fire that finds a genuinely new item spends a full surface-and-judge turn. Detection stays in-session, so the policy applies uniformly to file-based and MCP-auth loops alike. + +* Problem / Context + +In-session cron/loop monitors surface a visible model turn on every fire. When nothing changed, that turn is pure noise — a per-pass sentry digest full of SKIP lines, or a triage/inbox "nothing new" report. Over a night or a long session the signal (the one fire that found something) drowns in the empty ticks. The rulesets sentry live trial (2026-07-20) demonstrated it directly: fires 1-2 did real work, fires 3-8 were near-identical walls of no-op lines. + +Origin: a Craig-approved proposal from .emacs.d (2026-07-20), captured off an auto inbox-zero session filling with empty-check noise. + +** Why not an external watcher (the rejected shape) + +The proposal's first instinct was a shell-level watcher (systemd timer or backgrounded loop) doing detection outside the model, producing zero output on an empty check and dropping a handoff into inbox/ only on a real item — so the model runs zero turns when nothing changed. That is truly-zero-idle, but it has a disqualifying cost: it moves detection out of the session, and auto triage-intake's sources (Gmail, Slack, Linear) are reachable only through the session's inherited MCP auth. triage-intake.org is explicit that it runs in the live session precisely because "the headless-auth wall that blocks a detached cron run does not apply." A detached watcher cannot scan those sources at all. The external-watcher shape would therefore split the three targets into two incompatible cases (file-detectable vs MCP-auth) and still leave triage unsolved. + +The reframing (Craig, 2026-07-20): the noise is a *policy* problem — when to spend a full model turn — not a missing piece of infrastructure. Keeping detection in-session and making the empty fire cheap solves the actual complaint without the watcher, without per-machine daemon setup, and without breaking MCP auth. It applies to all three loops uniformly. + +* Goals and Non-Goals + +** Goals +- An empty monitor fire produces exactly one labelled heartbeat line, not a full narrated turn: =<workflow> at HH:MM: nothing=. +- A fire that detects a genuinely new item does the full surface-and-judge turn unchanged. +- One uniform policy across sentry, auto triage-intake, and auto inbox-zero — no per-loop special-casing. +- Detection stays in-session so MCP-auth loops (triage) get the same treatment as file-based loops. +- Each loop reuses the seen-state it already keeps; no new watcher-owned seen-list. + +** Non-Goals +- No external watcher, systemd timer, or Monitor-tool daemon. Detection is the loop body's own cheap check. +- No truly-zero-idle (no model turn at all on empty). That needs an external watcher and is the shape rejected above; if it is ever wanted for the file-based loops only, it is a separate vNext, logged not built. +- No change to what a loop does when it *does* find something — the surface/judge/act behaviour is untouched. +- No change to the one-shot (non-loop) invocations of these workflows. + +* Design + +** The policy + +Every in-session monitor fire runs in two steps: + +1. *Detect (cheap, silent).* Run the loop's existing detection against its seen-state — sentry's pass probes and branch/digest state, triage's sentinel scan, the inbox monitor's disposition check. This is Bash/tool work, not narration. +2. *Branch on the result.* + - *Nothing new* → emit one line, =<workflow> at HH:MM: nothing=, and end the fire. No digest, no per-pass lines, no report. + - *Something new* → the full existing turn: surface, judge, act/queue, and its normal richer output. + +The heartbeat is the whole output of an empty fire. Its format is fixed: the workflow's short name, =at=, =HH:MM= (local, from =date=), then =: = and the result word (=nothing= for an empty check). Examples: =sentry at 03:20: nothing=, =triage intake at 03:30: nothing=, =inbox zero at 03:30: nothing=. + +** Per-workflow application + +- *Sentry.* A fire whose passes all probe-skip or no-op collapses to =sentry at HH:MM: nothing=. No per-pass digest block is written for a quiet fire. A fire that runs or queues anything writes its full digest as today. (This supersedes the trial's behaviour, where fires 3-8 each wrote a full no-op digest.) +- *Auto triage-intake.* An auto-mode sweep that finds nothing across its enabled sources collapses to =triage intake at HH:MM: nothing=. Detection stays in-session, so the MCP sources are scanned normally; only the output on empty changes. +- *Auto inbox-zero.* A roam-mode cycle that finds no new inbox items collapses to =inbox zero at HH:MM: nothing=. + +** Seen-state (no new artifact) + +Because detection stays in-session, each loop keeps using the state it already maintains to know what is "new": triage's =.ai/last-triage-intake= sentinel, sentry's branch + digest, the inbox monitor's per-cycle disposition. The proposal's "watcher-owned seen-list replacing the in-anchor Dispositioned list" is dropped — it was an artifact of the external-watcher shape, which is not being built. + +** The accepted consequence + +An in-session loop still fires the model once per interval; the harness invokes it each time. So "silent" means the empty fire is *cheap* (one detection pass plus one heartbeat line), not *absent*. This is the deliberate trade for uniformity and for keeping the MCP auth that makes triage possible. The heartbeat also doubles as a liveness pulse: a visible "still running, nothing to do" beats silence that is indistinguishable from a stalled loop. + +* Decisions + +** DONE Policy, not mechanism — detection stays in-session +CLOSED: [2026-07-20 Mon] +Craig reframed the proposal: silent-until-signal is a policy about when to spend a full model turn, not a new watcher. Keeping detection in-session dissolves the MCP-auth split (triage's sources need session auth) and needs no per-machine daemon. + +** DONE Empty fire → one labelled heartbeat line (not fully silent) +CLOSED: [2026-07-20 Mon] +Format =<workflow> at HH:MM: nothing=. A visible pulse is worth one line so a running loop is distinguishable from a stalled one; full silence was the rejected alternative. + +** DONE Applies to sentry, auto triage-intake, and auto inbox-zero uniformly +CLOSED: [2026-07-20 Mon] +The two MCP-auth loops are first-class targets, not just sentry, precisely because detection stays in-session. + +** DONE No external watcher; no truly-zero-idle in v1 +CLOSED: [2026-07-20 Mon] +The systemd-timer / Monitor-tool / backgrounded-shell shapes are out. Truly-zero-idle (no turn on empty) is the only thing they'd buy, it only works for file-based loops, and it breaks triage. Logged as a possible file-only vNext, not built. + +** DONE Reuse each loop's existing seen-state +CLOSED: [2026-07-20 Mon] +No new watcher-owned seen-list; the sentinel / branch-digest / disposition each loop already keeps is the detection state. + +* Implementation Phases + +** Phase 1 — sentry quiet-fire heartbeat — DONE 2026-07-20 +Edit =.ai/workflows/sentry.org= (canonical + mirror): a fire whose passes all probe-skip or no-op writes =sentry at HH:MM: nothing= instead of a full per-pass digest; a fire that runs or queues anything writes the full digest unchanged. Update the digest section and the Common Mistakes "silent skip" note to draw the quiet-fire-vs-working-fire line. Run sync-check. (Separable and the highest-value piece — do first.) + +Shipped 2026-07-20: five edits to sentry.org — Pass Runner step 3 (per-pass lines collapse on a quiet fire), Fire-end step 2 (the heartbeat-vs-digest decision + heartbeat commit variant), the digest section (working-block vs quiet-heartbeat), Unattended-safety (quiet fire is not a silent skip), Common Mistakes #7 (the carve-out). Canonical + mirror synced, lint clean. + +** Phase 2 — auto triage-intake heartbeat +Edit =.ai/workflows/triage-intake.org= Auto mode: an empty sweep collapses to =triage intake at HH:MM: nothing=. Preserve in-session detection and the accumulate-don't-mutate contract. Run sync-check. + +** Phase 3 — auto inbox-zero heartbeat +Edit =.ai/workflows/inbox.org= Auto inbox zero mode: an empty cycle collapses to =inbox zero at HH:MM: nothing=. Run sync-check. + +** Phase 4 — state the shared policy once +Factor the policy statement into one place both loops and sentry point at (a short section in =inbox.org= monitor-mode core, or a =claude-rules/= note if it reads as cross-cutting), so the three workflows reference one definition rather than restating it. Decide the home during build. + +** Phase 5 — verification +Workflow prose, no bats surface. Verify by exercising: a sentry quiet fire prints one heartbeat line and writes no digest; a working fire still writes its full digest and commits. For the two MCP loops, confirm an empty sweep prints the heartbeat and a sweep with a planted item still does the full turn. Add a manual-testing checklist entry per =verification.md= where a live check is the only proof. + +* Prototype / UI + +Not applicable — no UI surface; the deliverable is the one-line heartbeat format. diff --git a/docs/specs/2026-07-20-triage-source-activation-spec.org b/docs/specs/2026-07-20-triage-source-activation-spec.org new file mode 100644 index 0000000..dce1ec1 --- /dev/null +++ b/docs/specs/2026-07-20-triage-source-activation-spec.org @@ -0,0 +1,127 @@ +#+TITLE: Triage Source Activation — Spec +#+AUTHOR: Craig Jennings +#+DATE: 2026-07-20 +#+TODO: TODO | DONE +#+TODO: DRAFT READY DOING | IMPLEMENTED SUPERSEDED CANCELLED + +* IMPLEMENTED triage source activation +:PROPERTIES: +:ID: af73ef0b-cd1d-46f1-9e1d-62695733a4de +:END: +- 2026-07-20 Mon @ 13:42:32 -0500 — READY (Craig approved after his read, both open decisions resolved via cj comments: declaration format "good. approved", interactive gate "all of them") → built and IMPLEMENTED same session. Phase 0 activation gate in triage-intake.org, sentry pass-3 probe, engine-intro documentation (the template notes.org has no Workflow State block, so the declaration is documented in triage-intake.org rather than there — deviation from the drafted Phase 3), migration handoffs to home + work, a manual-testing entry. Canonical + mirror synced. +- 2026-07-20 Mon @ 08:43:23 -0500 — drafted during the morning sentry review. The activation model converged live with Craig off the sentry live-trial finding. DRAFT pending his deep read. + +* Metadata +| Status | draft | +|----------+--------------------------------------------------------------| +| Owner | Craig Jennings | +|----------+--------------------------------------------------------------| +| Reviewer | (spec-review, next session) | +|----------+--------------------------------------------------------------| +| Related | [[file:../../todo.org::*Triage source activation][todo.org — Triage source activation task]] | +|----------+--------------------------------------------------------------| + +* Summary + +triage-intake should pull only the sources a project has chosen to pull. Today it runs every plugin it can discover, and the general (personal-account) plugins are template-synced into every project, so every project looks like it wants to triage Craig's personal Gmail, cmail, calendar, and Telegram. The fix is one activation layer: a general plugin runs only when the project names it in a =:TRIAGE_SOURCES:= declaration; a project-specific plugin stays active by its presence, which is already a deliberate per-project act. Sentry's pass-3 probe then reads the same signal. + +* Problem / Context + +The 8-fire sentry live trial (rulesets, 2026-07-20) surfaced this. Sentry's pass 3 probe is "triage source plugins present for this project," and triage-intake discovers sources by globbing =.ai/workflows/triage-intake.*.org= (general, template-synced) plus =.ai/project-workflows/triage-intake.*.org= (project-specific, never synced). Because the general plugins sync into every project, "plugins present" is true everywhere, so the triage pass self-activates in every project — including rulesets, which is not a triage target. + +The harm is two-layered, and sentry's existing safety rule only catches one layer. Sentry's pass-3 line says "destructive actions queue; they never fire unattended," which would hold back the trash/mark-read/star hygiene. But triage-intake also reads the accounts and files each Action item to the local =todo.org= as a =:quick:reactive:= task. Reading and filing are not destructive, so nothing queues them. Under sentry, triage would still authenticate to Craig's personal inboxes overnight and file his personal action items into whatever project the fire runs in. Wrong scope, and it touched real accounts to get there. + +The same over-pull exists interactively: running triage-intake by hand in rulesets today would pull personal Gmail too. The trial only made the unattended case visible. + +Root cause: the model conflates *presence* with *activation*. A plugin has two existing gates — it must be globbed (presence) and pass its =ENABLED= precondition (capability: "is the gmail MCP reachable?"). Neither answers the question the trial exposed: *should this project pull this source?* For the general plugins, presence comes from sync and capability is true on Craig's machine everywhere, so both gates pass in every project. + +Asymmetry that shapes the fix: project-specific plugins do not leak. They live in =.ai/project-workflows/=, are never synced, and exist only where someone deliberately dropped one. A project that polls an RSS feed puts an =triage-intake.rss-*.org= plugin there, and it runs in that project and nowhere else. Only the general synced plugins leak. So the missing activation layer only needs to gate the general plugins. + +* Goals and Non-Goals + +** Goals +- Per-project source selection: a project pulls exactly the sources it declares, nothing more. +- Off-limits by default: a general source that a project has not named is never pulled there. +- Project-specific sources keep working by presence — dropping the plugin is the declaration (Craig's RSS-feed case works unchanged). +- One fix covers both paths: the activation layer lives in triage-intake, so interactive and unattended (sentry) runs both respect it. +- Sentry's pass-3 probe reads the same activation signal rather than mere plugin presence. + +** Non-Goals +- No change to the per-plugin =ENABLED= capability check — it stays the "can I reach this source" gate. +- No change to the four-bucket classification, the digest shape, or the close behavior. +- No auto-migration across projects. Projects that pull general sources today declare them via handoff; the change never edits another project's config unattended. +- No new plugin discovery mechanism — the two-directory glob stays. + +* Design + +** Two plugin classes, one new activation layer + +- *Project-specific plugins* (=.ai/project-workflows/triage-intake.*.org=): active by presence. The plugin exists only because the project author put it there, which is itself the per-project declaration. No further gate. This is where a project's own sources live — an RSS feed, a work Linear, a work Slack. +- *General plugins* (=.ai/workflows/triage-intake.*.org=, template-synced — personal Gmail, cmail, calendar, Telegram, GitHub PRs): active only when the project names the source in its =:TRIAGE_SOURCES:= declaration. Present-but-undeclared means available-not-active: the plugin is on disk, but this project does not pull it. + +** The declaration + +A line in the project's =.ai/notes.org= Workflow State block, alongside =:COMMIT_AUTONOMY:= and =:LAST_AUDIT:=: + +: :TRIAGE_SOURCES: personal-gmail cmail + +Space-separated source names matching general-plugin basenames (=personal-gmail=, =cmail=, =personal-calendar=, =telegram=, =github-prs=). Absent or empty means no general sources are active for this project. Project-specific plugins are unaffected by this line — they run regardless, because presence is their declaration. + +** triage-intake Phase 0 change + +Phase 0 keeps globbing both directories. The loaded-set computation changes: for each *general* plugin, additionally require its basename to appear in =:TRIAGE_SOURCES:=; skip it with an announced reason otherwise ("skipping personal-gmail — not in :TRIAGE_SOURCES:"). Project-specific plugins skip this check. The =ENABLED= capability check still runs on the survivors. The announce-loaded-set block already exists and gains an "inactive (undeclared)" line so the omission stays visible rather than silent — the same anti-silence discipline Phase 0 already enforces. + +** Sentry pass-3 probe change + +The probe changes from "triage source plugins present" to "the project has at least one active triage source" — any project-specific plugin present, or a non-empty =:TRIAGE_SOURCES:= intersecting the general plugins on disk. rulesets, declaring nothing and owning no project-specific plugin, probe-skips cleanly. + +** Migration + +Projects that pull general sources today (home, and possibly work) each add a =:TRIAGE_SOURCES:= line, or their triage goes quiet. This is a per-project handoff, not an automated sweep — the change can't safely guess each project's intended source set. Projects that only ever ran project-specific plugins need no migration. + +* Decisions + +** DONE Activation gates general plugins only; project-specific stay active-by-presence +CLOSED: [2026-07-20 Mon] +Converged live with Craig. Project-specific plugins are already per-project (never synced), so they need no gate; only the general synced plugins leak, so only they need a declaration. + +** DONE The activation layer lives in triage-intake Phase 0, not only the sentry probe +CLOSED: [2026-07-20 Mon] +Placing it in the engine fixes the interactive over-pull too. Sentry inherits the signal rather than reimplementing it. + +** DONE Off-limits by default +CLOSED: [2026-07-20 Mon] +An undeclared general source is never pulled. Opt-in is explicit; there is no "pull everything discovered" default. + +** DONE Migration is per-project handoffs, never an unattended edit of another project's config +CLOSED: [2026-07-20 Mon] +The change can't guess a project's intended source set, and cross-project auto-edits violate the boundary rule. + +** DONE Declaration format — =:TRIAGE_SOURCES:= space-separated basenames in notes.org Workflow State +CLOSED: [2026-07-20 Mon] +Approved by Craig (2026-07-20). =:TRIAGE_SOURCES: personal-gmail cmail= in notes.org Workflow State, mirroring =:COMMIT_AUTONOMY:=. An empty-but-present marker and an absent one are treated identically — both mean "no general sources active" — so there's no need to distinguish them. A project-specific-only project carries no marker; its absence is correct, since those plugins activate by presence. + +** DONE Interactive triage adopts the same gate as unattended +CLOSED: [2026-07-20 Mon] +Craig: "all of them" (2026-07-20). The activation gate lives in triage-intake Phase 0 and applies to every path — interactive and unattended alike — so running triage by hand in a project also respects its =:TRIAGE_SOURCES:= declaration. One activation layer, no per-path special-casing. + +* Implementation Phases + +** Phase 1 — triage-intake Phase 0 activation rule +Edit =.ai/workflows/triage-intake.org= (canonical =claude-templates/.ai/workflows/= + mirror): add the general-vs-project-specific activation rule to Phase 0, the =:TRIAGE_SOURCES:= read, and the "inactive (undeclared)" announce line. Run sync-check. + +** Phase 2 — sentry pass-3 probe +Edit =.ai/workflows/sentry.org= (canonical + mirror): change the pass-3 probe to "any active triage source present" and note the activation source. Run sync-check. + +** Phase 3 — document the declaration +Document =:TRIAGE_SOURCES:= in the notes.org Workflow State reference (the template =claude-templates/.ai/notes.org= Workflow State block) so new projects see it. Cross-reference from triage-intake. + +** Phase 4 — migration handoffs +=inbox-send= home (and work, if it pulls general sources) the =:TRIAGE_SOURCES:= line each should add, with the reason. No auto-edit. + +** Phase 5 — verification +triage-intake is a workflow (prose), not a script, so there's no bats surface for the activation rule directly. Verify by a scripted check that a project with no declaration loads zero general plugins (a small fixture around the Phase-0 glob-and-filter logic if it's extracted, or a manual-testing checklist entry otherwise). Confirm rulesets probe-skips triage under sentry, and home still pulls its declared sources. + +* Prototype / UI + +Not applicable — no UI surface. diff --git a/docs/specs/wrapup-routing-spec.org b/docs/specs/wrapup-routing-spec.org index 1bdc0f3..07d9ec6 100644 --- a/docs/specs/wrapup-routing-spec.org +++ b/docs/specs/wrapup-routing-spec.org @@ -218,7 +218,7 @@ Everything else accepted as written: H1 (inbox-route supersedes direct-move; D2/ ** 2026-06-21 Sun @ 01:58:41 -0400 — Claude Code (rulesets) — reviewer - What: spec-review pass. Rubric *Not ready*, two blocking findings. H1: the inbox-route alternative (inbox-send each routable keeper to the destination's inbox/, let its own process-inbox file it) supersedes the direct-move design — reshape D2, drop Phase 2 and D3's provenance burden. H2: pin the candidate-set marking to Option A (tag =:ROUTE_CANDIDATE:= at process-inbox file time). Four medium findings (M1 confidence tiers, M2 empty-set silence, M3 paired process-inbox edit phase, M4 cross-project.md note). Full review + drop-in implementation tasks in the review file. - Why: Craig challenged D2 directly (why edit a foreign todo.org rather than use the sanctioned inbox-send path). The review confirmed it: inbox-send already emits the exact provenance D3 reinvents, process-inbox already files per-item with the destination's own gate, cross-project.md sanctions the inbox path, and a verified precondition reverses the spec's assumption — chime and yt-sync have inbox/ but no todo.org, so direct-move silently drops keepers headed there while inbox-route degrades gracefully. -- Artifacts: [[file:wrapup-routing-spec-review.org][review file]]. Next: spec-response to disposition H1/H2 (recommend accept both), which moves the rubric to Ready. +- Artifacts: the review file (since folded into this spec). Next: spec-response to disposition H1/H2 (recommend accept both), which moves the rubric to Ready. ** 2026-06-21 Sun @ 02:06:37 -0400 — Craig Jennings + Claude Code (rulesets) — responder - What: folded the spec-review in. Accepted H1 (inbox-route) and H2 (tag at file time); superseded D2 and D3; added D7 (deliver via =inbox-send=), D8 (=:ROUTE_CANDIDATE:= marker at file time), D9 (local source removal + reject-flow recovery). Rewrote Summary, Goals, Design mechanics, Implementation phases (dropped the atomic-move helper — Phase 2 is now the =process-inbox= marker edit), and Acceptance criteria for the inbox-route. One modify (D9) refines H1's vague source-handling. Cookie [9/9]; Status → Ready. diff --git a/hooks/README.md b/hooks/README.md index 71b3613..9c4268d 100644 --- a/hooks/README.md +++ b/hooks/README.md @@ -10,6 +10,9 @@ Machine-wide Claude Code hooks that install into `~/.claude/hooks/` and apply to | `git-commit-confirm.py` | `PreToolUse(Bash)` | Silent-unless-suspicious gate on `git commit`. Only prompts when the message contains AI-attribution patterns, the message can't be parsed (editor would open), no files are staged, or the git author is unusable. Clean commits pass through without a modal. Parses both HEREDOC and `-m`/`--message` forms. | | `gh-pr-create-confirm.py` | `PreToolUse(Bash)` | Gates `gh pr create` behind a confirmation modal showing title, base←head, reviewers, labels, assignees, milestone, draft flag, and body (HEREDOC or quoted). | | `destructive-bash-confirm.py` | `PreToolUse(Bash)` | Gates destructive commands (`git push --force`, `git reset --hard`, `git clean -f`, `git branch -D`, `rm -rf`) with a modal showing the command, local context (branch, uncommitted file counts, targeted paths), and a warning banner. Elevates severity when force-pushing protected branches or targeting root/home/wildcard paths. | +| `inbox-boundary-check.sh` | `Stop` | Soft-nudges the agent to process pending `inbox/` handoffs before yielding. Blocks the stop once (injects a reason with the pending count) when `inbox-status -q` reports pending items; steps aside on the harness re-entry (`stop_hook_active`) so a mid-task pause or an unprocessable item never wedges. No-ops in any project without an `inbox/` or without `inbox-status`. | +| `ai-wrap-teardown.sh` | `Stop` | Re-verifies the strict clean-tree certificate and its HEAD before consuming a wrap sentinel. Dirty or uncertified state blocks Stop with an actionable report; verified state tears down the matching ai-term session or starts the guarded shutdown countdown. Emits the appropriate Claude or Codex response shape. | +| `rulesets-write-boundary.py` | `PreToolUse(Edit\|Write)` | Resolves write targets through symlinks and denies cross-project edits that land inside rulesets. Directs the sender through `inbox-send rulesets`; rulesets sessions themselves pass. Codex maps `apply_patch` to the same matcher. | Shared library (not a hook): `_common.py` — `read_payload()`, `respond_ask()`, `scan_attribution()`. Installed as a sibling symlink so the two Python hooks can `from _common import …` at runtime. diff --git a/hooks/ai-wrap-teardown.sh b/hooks/ai-wrap-teardown.sh index 6133075..ca73ae6 100755 --- a/hooks/ai-wrap-teardown.sh +++ b/hooks/ai-wrap-teardown.sh @@ -39,8 +39,10 @@ # "command": "~/.claude/hooks/ai-wrap-teardown.sh" } ] } ] set -u +payload="$(cat)" + # Stop-hook stdin JSON carries cwd; basename it to the project / aiv- session. -cwd="$(jq -r '.cwd // empty' 2>/dev/null)" +cwd="$(printf '%s' "$payload" | jq -r '.cwd // empty' 2>/dev/null)" [ -z "$cwd" ] && cwd="$PWD" proj="$(basename "$cwd")" @@ -53,15 +55,54 @@ fire() { emacsclient -e "$1" >/dev/null 2>&1 || true } +# A sentinel means wrap-up claimed completion. Re-prove that claim immediately +# before consuming it: the certificate binds a prior strict check to HEAD, and +# verify also performs a fresh strict check to catch late writes. +verify_wrap() { + local gate="${GIT_WORKTREE_GATE:-}" detail + if [ -z "$gate" ]; then + gate="$(command -v git-worktree-gate 2>/dev/null || true)" + fi + if [ -z "$gate" ] || [ ! -x "$gate" ]; then + gate="$HOME/code/rulesets/claude-templates/bin/git-worktree-gate" + fi + if [ ! -x "$gate" ]; then + detail="wrap blocked: git-worktree-gate is unavailable; install rulesets tooling and retry" + else + detail="$("$gate" verify "$cwd" 2>&1)" && return 0 + fi + + # Codex and Claude consume different Stop-hook response fields. Codex + # command-hook payloads always include model; Claude's do not. + if printf '%s' "$payload" | jq -e '.model? != null' >/dev/null 2>&1; then + jq -n --arg reason "$detail" \ + '{continue:false, stopReason:$reason, systemMessage:$reason}' + else + jq -n --arg reason "$detail" \ + '{decision:"block", reason:$reason}' + fi + return 1 +} + +consume_certificate() { + local dir + dir="$(git -C "$cwd" rev-parse --absolute-git-dir 2>/dev/null)" || return 0 + rm -f "$dir/ai-wrap-clean" +} + # Shutdown supersedes teardown when both are somehow present. if [ -f "$shutdown_sentinel" ]; then + verify_wrap || exit 0 rm -f "$shutdown_sentinel" "$teardown_sentinel" + consume_certificate fire '(cj/ai-term-shutdown-countdown)' exit 0 fi if [ -f "$teardown_sentinel" ]; then + verify_wrap || exit 0 rm -f "$teardown_sentinel" + consume_certificate fire "(cj/ai-term-quit \"${proj}\")" exit 0 fi diff --git a/hooks/inbox-boundary-check.sh b/hooks/inbox-boundary-check.sh new file mode 100755 index 0000000..916c000 --- /dev/null +++ b/hooks/inbox-boundary-check.sh @@ -0,0 +1,52 @@ +#!/usr/bin/env bash +# Stop hook: soft-nudge the agent to process pending inbox/ handoffs before it +# yields control back to Craig. +# +# The "check inbox/ at every task boundary" rule (protocols.org, Inbox +# Monitoring Cadence) is otherwise prose-only, so it holds only as well as the +# agent remembers it. A Stop is the agent finishing a turn and about to yield, +# which is the harness's closest event to the rule's own trigger ("after +# finishing a unit of work, before reporting back"). When handoffs are pending +# this hook blocks the yield once and injects a reason, so the agent processes +# them instead of returning with items unseen. +# +# Soft-nudge, not hard-block: on the harness re-entry (stop_hook_active: true) +# the hook steps aside and lets the turn end. An item the agent genuinely can't +# process, or a mid-task pause to ask Craig a question (also a Stop), never +# wedges the session. +# +# Self-skips everywhere it doesn't apply: no inbox/ dir, no inbox-status, or a +# clean inbox each exit 0 silently. One hook, every project, no config. +# +# Wire in ~/.claude/settings.json (see hooks/settings-snippet.json), Stop array. +set -u + +payload="$(cat)" + +# Re-entry after our own nudge: don't nudge twice, let the turn end. +active="$(printf '%s' "$payload" | jq -r '.stop_hook_active // false' 2>/dev/null)" +[ "$active" = "true" ] && exit 0 + +cwd="$(printf '%s' "$payload" | jq -r '.cwd // empty' 2>/dev/null)" +[ -z "$cwd" ] && cwd="$PWD" +cd "$cwd" 2>/dev/null || exit 0 + +# Nothing to enforce without an inbox to watch. +[ -d inbox ] || exit 0 + +# Prefer the project-local inbox-status; fall back to one on PATH. Absent both, +# degrade to a no-op rather than block on a check we can't run. +status_bin=".ai/scripts/inbox-status" +[ -x "$status_bin" ] || status_bin="$(command -v inbox-status 2>/dev/null)" || exit 0 +[ -n "$status_bin" ] || exit 0 + +# inbox-status -q: exit 1 = pending, 0 = clean, 2 = no inbox. Only 1 nudges. +out="$("$status_bin" -q 2>/dev/null)" +[ $? -eq 1 ] || exit 0 + +count="$(printf '%s' "$out" | grep -oE '[0-9]+' | head -1)" +[ -n "$count" ] || count="Some" + +reason="$count pending inbox handoff(s) in $(basename "$cwd"). Process per inbox.org before yielding." +printf '{"decision":"block","reason":%s}\n' "$(printf '%s' "$reason" | jq -Rs .)" +exit 0 diff --git a/hooks/rulesets-write-boundary.py b/hooks/rulesets-write-boundary.py new file mode 100755 index 0000000..35088ea --- /dev/null +++ b/hooks/rulesets-write-boundary.py @@ -0,0 +1,94 @@ +#!/usr/bin/env python3 +"""Block cross-project Edit/Write calls that resolve into rulesets. + +Global Claude rules, hooks, skills, commands, and bin tools are symlinks into +~/code/rulesets. Editing one from another project's session silently dirties +rulesets and can block every later startup. The sanctioned path is inbox-send; +the rulesets session owns the canonical edit and its wrap. +""" + +from __future__ import annotations + +import os +import re +from pathlib import Path +from typing import Any, Iterable + +from _common import read_payload, respond_deny + + +PATCH_PATH = re.compile( + r"^\*\*\* (?:Add|Update|Delete) File: (.+)$|^\*\*\* Move to: (.+)$", + re.MULTILINE, +) + + +def inside(path: Path, parent: Path) -> bool: + try: + path.relative_to(parent) + return True + except ValueError: + return False + + +def candidate_paths(tool_input: Any) -> Iterable[str]: + if isinstance(tool_input, dict): + for key, value in tool_input.items(): + if key in {"file_path", "path"} and isinstance(value, str): + yield value + elif key in {"patch", "input"} and isinstance(value, str): + for match in PATCH_PATH.finditer(value): + yield match.group(1) or match.group(2) + elif isinstance(tool_input, str): + for match in PATCH_PATH.finditer(tool_input): + yield match.group(1) or match.group(2) + + +def resolve(raw: str, cwd: Path) -> Path: + expanded = Path(os.path.expandvars(os.path.expanduser(raw))) + if not expanded.is_absolute(): + expanded = cwd / expanded + return expanded.resolve(strict=False) + + +def main() -> int: + payload = read_payload() + tool_name = payload.get("tool_name", "") + if tool_name not in {"Edit", "Write", "apply_patch"}: + return 0 + + rulesets = Path( + os.environ.get("RULESETS_ROOT", "~/code/rulesets") + ).expanduser().resolve(strict=False) + cwd = Path(payload.get("cwd") or os.getcwd()).resolve(strict=False) + + # A rulesets session owns its own canonical files. + if inside(cwd, rulesets): + return 0 + + blocked: list[str] = [] + for raw in candidate_paths(payload.get("tool_input", {})): + try: + target = resolve(raw, cwd) + except (OSError, RuntimeError): + blocked.append(f"{raw} (real path could not be verified)") + continue + if inside(target, rulesets): + blocked.append(str(target)) + + if not blocked: + return 0 + + shown = ", ".join(blocked) + reason = ( + "Blocked cross-project write into rulesets: " + f"{shown}. This path may have been reached through an installed " + "symlink. Send the proposed change with inbox-send rulesets instead, " + "then let a rulesets session apply and wrap it." + ) + respond_deny(reason, system_message=reason) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/hooks/session-start-disarm.sh b/hooks/session-start-disarm.sh new file mode 100755 index 0000000..520d8c0 --- /dev/null +++ b/hooks/session-start-disarm.sh @@ -0,0 +1,42 @@ +#!/usr/bin/env bash +# +# SessionStart: disarm any wrap sentinel left over from a previous session. +# +# wrap-it-up drops /tmp/ai-wrap-teardown-<project> (or -shutdown-) to ask the +# Stop hook to kill the tmux session once the wrap certifies clean. The Stop +# hook deliberately PRESERVES that sentinel when certification fails, so a wrap +# blocked by a dirty tree can retry on a later stop in the same session without +# the user re-running the workflow. +# +# Nothing bounded that retry to the session. A sentinel armed by a wrap that +# never certified survived indefinitely and fired in whatever session next +# happened to reach a clean tree: +# +# work, 2026-07-27. The 11:37 wrap requested teardown, failed certification +# on a dirty tree, and left the sentinel armed. A fresh session started at +# 13:20, committed twice during startup, went clean — and the next stop +# consumed the two-hour-old sentinel and killed the terminal mid-work. +# archsetup's had been armed for two days on a live attached session. +# +# A new session means the wrap that armed the sentinel is gone, so its pending +# teardown is meaningless: clear it. Within-session retry is untouched, because +# this only runs at session start. If the user still wants teardown, wrap-it-up +# re-arms it. +# +# Scoped to the current project's sentinels only — a concurrent session in +# another project keeps its own. +# +# Silent and exit 0 always. A SessionStart hook must never block a session from +# starting, and there is nothing here a user needs told. + +set -u + +payload="$(cat 2>/dev/null || true)" + +cwd="$(printf '%s' "$payload" | jq -r '.cwd // empty' 2>/dev/null)" +[ -z "$cwd" ] && cwd="$PWD" +proj="$(basename "$cwd")" + +rm -f "/tmp/ai-wrap-teardown-${proj}" "/tmp/ai-wrap-shutdown-${proj}" + +exit 0 diff --git a/hooks/settings-snippet.json b/hooks/settings-snippet.json index 0f0e784..50e3d31 100644 --- a/hooks/settings-snippet.json +++ b/hooks/settings-snippet.json @@ -23,6 +23,12 @@ ], "PreToolUse": [ { + "matcher": "Edit|Write", + "hooks": [ + { "type": "command", "command": "~/.claude/hooks/rulesets-write-boundary.py" } + ] + }, + { "matcher": "Bash", "hooks": [ { "type": "command", "command": "~/.claude/hooks/git-commit-confirm.py" }, @@ -34,6 +40,7 @@ "Stop": [ { "hooks": [ + { "type": "command", "command": "~/.claude/hooks/inbox-boundary-check.sh" }, { "type": "command", "command": "~/.claude/hooks/ai-wrap-teardown.sh" } ] } diff --git a/hooks/tests/test_rulesets_write_boundary.py b/hooks/tests/test_rulesets_write_boundary.py new file mode 100644 index 0000000..826a941 --- /dev/null +++ b/hooks/tests/test_rulesets_write_boundary.py @@ -0,0 +1,110 @@ +import json +import os +import subprocess +import sys +from pathlib import Path + + +SCRIPT = Path(__file__).parents[1] / "rulesets-write-boundary.py" + + +def run_hook(payload: dict, rulesets: Path) -> dict | None: + proc = subprocess.run( + [sys.executable, str(SCRIPT)], + input=json.dumps(payload), + text=True, + capture_output=True, + env={**os.environ, "RULESETS_ROOT": str(rulesets)}, + check=True, + ) + return json.loads(proc.stdout) if proc.stdout else None + + +def test_allows_write_from_rulesets_session(tmp_path): + rulesets = tmp_path / "rulesets" + rulesets.mkdir() + target = rulesets / "file" + result = run_hook( + { + "cwd": str(rulesets), + "tool_name": "Write", + "tool_input": {"file_path": str(target)}, + }, + rulesets, + ) + assert result is None + + +def test_blocks_absolute_cross_project_write(tmp_path): + rulesets = tmp_path / "rulesets" + other = tmp_path / "other" + rulesets.mkdir() + other.mkdir() + result = run_hook( + { + "cwd": str(other), + "tool_name": "Edit", + "tool_input": {"file_path": str(rulesets / "rule.md")}, + }, + rulesets, + ) + assert result["hookSpecificOutput"]["permissionDecision"] == "deny" + assert "inbox-send rulesets" in result["hookSpecificOutput"][ + "permissionDecisionReason" + ] + + +def test_blocks_write_reached_through_symlink(tmp_path): + rulesets = tmp_path / "rulesets" + other = tmp_path / "other" + installed = tmp_path / "installed" + rulesets.mkdir() + other.mkdir() + (rulesets / "rules").mkdir() + installed.symlink_to(rulesets / "rules", target_is_directory=True) + result = run_hook( + { + "cwd": str(other), + "tool_name": "Write", + "tool_input": {"file_path": str(installed / "todo-format.md")}, + }, + rulesets, + ) + assert result["hookSpecificOutput"]["permissionDecision"] == "deny" + assert str(rulesets) in result["systemMessage"] + + +def test_blocks_apply_patch_target(tmp_path): + rulesets = tmp_path / "rulesets" + other = tmp_path / "other" + rulesets.mkdir() + other.mkdir() + patch = ( + f"*** Begin Patch\n*** Update File: {rulesets / 'file'}\n" + "@@\n-old\n+new\n*** End Patch\n" + ) + result = run_hook( + { + "cwd": str(other), + "tool_name": "apply_patch", + "tool_input": {"input": patch}, + }, + rulesets, + ) + assert result["hookSpecificOutput"]["permissionDecision"] == "deny" + + +def test_allows_unrelated_write(tmp_path): + rulesets = tmp_path / "rulesets" + other = tmp_path / "other" + rulesets.mkdir() + other.mkdir() + result = run_hook( + { + "cwd": str(other), + "tool_name": "Edit", + "tool_input": {"file_path": str(other / "file")}, + }, + rulesets, + ) + assert result is None diff --git a/inbox/PROCESSED-2026-06-11-1703-from-home-consolidation-handoff-rulesets.org b/inbox/PROCESSED-2026-06-11-1703-from-home-consolidation-handoff-rulesets.org deleted file mode 100644 index f0a86b7..0000000 --- a/inbox/PROCESSED-2026-06-11-1703-from-home-consolidation-handoff-rulesets.org +++ /dev/null @@ -1,46 +0,0 @@ -#+TITLE: Home is consolidating all personal ~/projects AI projects into itself — heads-up + okay requested -#+DATE: 2026-06-11 - -* What's happening - -Craig approved and started a migration that folds every AI-managed project under ~/projects (except work) into the home project as area subdirectories, with full git history preserved via git filter-repo + merge. End state: ~/projects holds home and work only; ~/code stays the place for standalone code projects. One todo.org, one priority scheme, one .ai session for all personal project management. - -The spec rode along in this same inbox drop (from-home file with "project-consolidation-spec" in the name). It went through a full spec-review cycle (Codex, two passes: Not ready → Ready) and carries the per-fold manifest contract, a two-mode restore runbook, and a layered rollback story (snapper snapshot, untouched server bares, pre-fold tags, retired dirs, 30-day cooling-off). - -* How far we've gone (as of 2026-06-11 ~17:00 CDT) - -- Phase 0 done: git-filter-repo verified, snapper snapshot 5621, memory-dir tar in ~/backups/, pre-consolidation tag pushed. -- Phase 1 done: philosophy folded (pilot). Mode-B rollback drill passed in a disposable clone — the fold is provably removable using only its manifest. -- Phase 2 done: clipper folded (dress rehearsal). First todo.org import with :MIGRATED_FROM: markers and a fold-time triage; first live-link fix (a dirvish bookmark in ~/.emacs.d). -- Both sources retired to ~/projects/.retired/ (not deleted). Server bares untouched. -- Manifests at home:docs/consolidation-manifest-{philosophy,clipper}.org; live inventory gate at docs/consolidation-manifest-inventory.org. - -* Learnings and adjustments so far - -- A manifest can't embed its own redistribution commit's sha (amend changes it). The manifest row now reads "the commit introducing this manifest"; only the merge sha is recorded literally. -- git clone --no-local is the right clone shape for filter-repo's fresh-clone safety check; --force is banned from the runbook. -- The fold-merge branch is read from the source (fb-photo-scraper sits on master, not main — assume nothing). -- Staged freeze instead of full shutdown: the live inventory gate is regenerated per project immediately before its fold, so Craig can keep using not-yet-folded projects until their turn. -- Imported [#D] tasks fall outside the task-review staleness pool (it tracks A-C only) — by design, but worth knowing when verifying an import. -- Source todo.org "Reference" sections (non-task content) merge into the area's notes.org, not into home's todo.org. - -* What we'd like rulesets to think about - -- Edge cases we may not have seen: anything in the templates, workflows, or scripts that assumes one .ai per project under ~/projects (cross-agent-comms discovery, inbox-send target resolution, the ai launcher's project scan, broadcast). -- Future-project plans: whether new personal "projects" should now start as areas inside home rather than standalone ~/projects entries, and whether the templates should say so. -- Any rulesets docs or workflows that name the folding projects as handoff/broadcast targets (finances, jr-estate, etc.) — they'll need updating in our Phase 7 ecosystem pass; a list from your side would help us not miss any. -- The knowledge-base work-root denylist (~/projects/work) is unaffected. - -* Ask: confirmed okay to continue - -Reply to home's inbox with a confirmed okay (or concerns) for folding the remaining projects. The remaining list, in planned order: - -1. jr-estate (Phase 3 — 704M history, 18-task triage, first memory merge) -2. danneel (Phase 4 — 613M history, 10-task triage) -3. finances (Phase 5 — 251M history, 43-task triage) -4. documents, elibrary, health, kit (Phase 6 folds, together) -5. website + little-elisper — relocate to ~/code as standalone AI projects (Phase 6) -6. fb-photo-scraper — delete (unmodified upstream clone; origin recorded in spec D2) -7. Phase 7 ecosystem pass: link sweep, velox migration, rulesets-reference updates, KB node, cooling clock. - -We hold the remaining folds until your reply lands. diff --git a/inbox/PROCESSED-2026-06-11-1703-from-home-project-consolidation-spec.org b/inbox/PROCESSED-2026-06-11-1703-from-home-project-consolidation-spec.org deleted file mode 100644 index b557012..0000000 --- a/inbox/PROCESSED-2026-06-11-1703-from-home-project-consolidation-spec.org +++ /dev/null @@ -1,350 +0,0 @@ -#+TITLE: Project Consolidation — Fold ~/projects AI Projects into Home — Spec -#+AUTHOR: Craig Jennings -#+DATE: 2026-06-11 - -* Metadata -| Status | Ready — Codex spec-review confirmed 2026-06-11 | -| Owner | Craig Jennings | -| Reviewer | Codex (spec-review, 2026-06-11) | -| Related | [[file:../todo.org::*Project consolidation into home][todo.org task]] | - -* Summary - -Fold every AI-managed project under =~/projects/= (except =work=) into the =home= project as area subdirectories, with full git history preserved, so all personal tasks live in one =todo.org= and can be prioritized against each other. The end state: =~/projects/= holds =home= and =work=; =~/code/= holds standalone code projects. Every step is reversible — originals are snapshotted, server bare repos stay untouched until a cooling-off period ends, and a written restore procedure covers both single-project and full rollback. - -* Problem / Context - -Craig manages 12 AI projects under =~/projects/= beside =home= and =work=. Each has its own =.ai/= session machinery, =todo.org=, inbox, memory dir, and git repo on cjennings.net. That isolation was the design — but the life-management projects (finances, jr-estate, danneel, health, kit, clipper, documents…) are all facets of one life, and their tasks compete for the same hours. Today there is no single surface where a [#A] in jr-estate can be weighed against a [#A] in finances; each project's priorities are graded against siblings only. Sessions fragment the same way: a morning touching finances, danneel, and home means three separate session launches, three inboxes, three memory stores, and cross-project handoff files between them. - -The forces: (1) one prioritization surface requires one task file (or at least one repo); (2) these projects are critical — legal disputes, estate settlement, finances — so the migration must be provably reversible; (3) per-project git histories carry evidentiary and reference value (especially danneel and jr-estate) and must survive queryably; (4) the =.ai/= ecosystem (sessions, memories, inboxes, workflows) has per-project state that must merge without loss; (5) other machines (velox at minimum) hold clones whose remotes must not break silently. - -Survey of the candidates (2026-06-11): - -| Project | What it is | .git | Sessions | Open tasks | Memories | Inbox | -|------------------+---------------------------------------------+-------+----------+------------+----------+-------| -| clipper | 19 Clipper St SF rental property | 23M | 6 | 1 | 0 | 0 | -| danneel | 4319 Danneel construction dispute (legal) | 613M | 45 | 10 | 0 | 0 | -| documents | Disaster-prep document vault | 61M | 7 | 8 | 0 | 2 | -| elibrary | Ebook + music library management | 2.4M | 6 | 0 | 2 | 3 | -| fb-photo-scraper | Third-party FB gallery scraper (clone) | 3.4M | 0 | 0 | 0 | 0 | -| finances | Personal finance (Craig + Christine) | 251M | 24 | 43 | 1 | 4 | -| health | Personal health management | 5.9M | 27 | 8 | 1 | 3 | -| jr-estate | JR estate settlement (legal, trust, taxes) | 704M | 44 | 18 | 7 | 2 | -| kit | Keep In Touch — relationship management | 17M | 20 | 20 | 3 | 4 | -| little-elisper | The Little LISPer worked in elisp (study) | 4.7M | 2 | 0 | 0 | 0 | -| philosophy | Philosophy study + discussion notes | 30M | 5 | 0 | 0 | 0 | -| website | Personal Hugo site (deployed from homelab) | 8.6M | 10 | 11 | 0 | 3 | - -All have origin on cjennings.net except fb-photo-scraper (GitHub clone). All =.ai/= dirs are tracked (personal-project model). No project has a live =session-context.org=. Several have small dirty trees (elibrary 2, finances 1, health 1, kit 4, little-elisper 1 files) that must be resolved pre-fold. - -* Goals and Non-Goals - -** Goals -- One repo, one =todo.org=, one priority scheme, one =.ai/= session for all personal (non-work) project management. -- Full git history of every folded project preserved and queryable in place (=git log <area>/= works). -- All =.ai/= state merged without loss: session archives, memories, inboxes, someday-maybe, project workflows/scripts. -- A written, tested restore path for any single project and for the whole migration. -- =~/code/= becomes the only home for standalone code projects. - -** Non-Goals -- No restructuring of home's existing content (homelab docs, assets, music reconciliation dirs stay where they are; re-nesting them under an =infra/= area is vNext). -- No change to the =work= project in any way. -- No task content rewriting beyond re-grading priorities to the unified scheme — bodies, links, and histories move as-is. -- No server-side bare-repo deletion during the migration (archival is a separate, later, post-cooling step). -- No renaming of the =home= project. - -** Scope tiers -- v1: fold the nine life-management projects (clipper, danneel, documents, elibrary, finances, health, jr-estate, kit, philosophy); relocate the code-shaped three (website, little-elisper, fb-photo-scraper) per Decision 2; ecosystem updates (emacs agenda, rulesets references, velox). -- Out of scope: work; any =~/code/= project; home's internal restructure. -- vNext: server bare-repo archival after cooling-off; optional =infra/= re-nesting of homelab content; per-area README normalization. - -* Design - -** End-state layout - -Each folded project becomes a top-level area directory in home, preserving its internal structure: - -#+begin_example -~/projects/home/ - clipper/ danneel/ documents/ elibrary/ - finances/ health/ jr-estate/ kit/ philosophy/ - assets/ docs/ homelab-inventory/ inbox/ scripts/ (existing home content, unchanged) - todo.org (unified) - .ai/ (single session machinery) -#+end_example - -This matches the established area pattern (work's =deepsat/assets/=, the working-files convention's =<area>/assets/=). Each area keeps its own =assets/=, =docs/=, internal org files, and a per-area =.gitignore= carrying its old ignore patterns (git honors nested ignores; prefixing patterns into the root ignore is error-prone). - -** Git history — filter-repo then merge - -For each source project, on a throwaway clone (never the original). =--no-local= forces a real transport-style clone instead of hardlinked/shared objects — the conservative shape git-filter-repo's manual recommends, and what makes the clone a genuine fresh-clone safety check. =branch= is the source's actual default branch (fb-photo-scraper is on =master=; assume nothing): - -#+begin_example -branch=$(git -C ~/projects/<name> symbolic-ref --short HEAD) -git clone --no-local ~/projects/<name> /tmp/fold-<name> -cd /tmp/fold-<name> -git filter-repo --to-subdirectory-filter <name> -cd ~/projects/home -git remote add fold-<name> /tmp/fold-<name> -git fetch fold-<name> -git merge --allow-unrelated-histories -m "feat(<name>): fold <name> project into home" "fold-<name>/$branch" -git remote remove fold-<name> -#+end_example - -=git filter-repo --force= is not part of this runbook. If filter-repo refuses to run, the fresh-clone safety check failed — stop, write down why, and fix the clone rather than overriding. - -=filter-repo= rewrites every historical path under =<name>/=, so after the merge =git log <name>/somefile= shows the file's full history with original commit messages, authors, and dates. The original repo and its server bare are never touched — the rewrite happens on the temp clone only. - -The merged home =.git= grows to roughly 1.8G (dominated by danneel 613M + jr-estate 704M + finances 251M). Acceptable for a private single-user server; noted as a clone-time cost. - -** The .ai/ and task merge (per project, after the git merge) - -The git merge lands the project's files under =<name>/=, including its old =.ai/= and =todo.org=. A post-merge commit then redistributes that state: - -1. /Sessions:/ =git mv <name>/.ai/sessions/*= into home's =.ai/sessions/=, inserting the area into the name: =YYYY-MM-DD-HH-MM-<desc>.org= → =YYYY-MM-DD-HH-MM-<name>-<desc>.org=. Dated names make collisions near-impossible; the prefix preserves provenance. -2. /todo.org:/ append the project's open work as a new top-level section =* <Area> Open Work= (matching =* Home Open Work=), and its resolved section likewise. Each imported section heading carries a properties drawer marking provenance — =:MIGRATED_FROM: <name>= and =:MIGRATED_ON: YYYY-MM-DD= — so the import boundary stays visible to future edits and to the restore runbook. Walk the incoming tasks with Craig to re-grade priorities onto the unified scheme (home's A-D impact/urgency ladder, generalized beyond infra) and add an area tag (=:finances:=, =:jrestate:=, =:danneel:=, …). Then delete =<name>/todo.org=. -3. /notes.org:/ the project's Project-Specific Context moves to =<name>/notes.org= (area-local reference, linked from home's =.ai/notes.org=). Active Reminders and Pending Decisions merge into home's =.ai/notes.org= with area attribution. -4. /Inbox:/ process each project's inbox to zero before the fold (preferred), or move unprocessed items into home's =inbox/= renamed =YYYY-MM-DD-from-<name>-<orig>.ext=. -5. /Workflows + scripts:/ copy =<name>/.ai/project-workflows/*= and =project-scripts/*= into home's, after a filename-collision check (13 workflow files exist across sources; any collision is resolved by area-prefixing the incoming file). Then delete the area's old =.ai/= machinery (=protocols.org=, =workflows/=, =scripts/= — all template-synced duplicates). -6. /someday-maybe.org:/ append under an =* <Area>= header in home's. -7. /Memory:/ copy =~/.claude/projects/-home-cjennings-projects-<name>/memory/*.md= into home's memory dir (rename on slug collision), append index lines to =MEMORY.md=, dedupe against existing entries, then archive the source memory dir into the backup tar (it lives outside git). -8. /Links:/ =grep -rn "projects/<name>" ~/projects/home ~/.emacs.d ~/sync/org ~/org/roam= and fix every absolute reference to the new path. Record the before-count, fix, re-run, and classify any remaining hits as historical (session archives, this spec) or live — live hits block the fold's close. Relative links inside the area survive the move untouched because internal structure is preserved. - -** Per-fold manifest - -Every fold produces a tracked manifest at =docs/consolidation-manifest-<name>.org=, written as the fold proceeds and committed with the redistribution commit. The manifest is the restore contract and the verification record — without it, "every step is reversible" is a slogan. It records: - -- /Source state:/ source path, HEAD sha, branch, =git status --porcelain=v1= output at gate time (must be empty), origin URL. -- /Tracked universe:/ the =git ls-files= listing from the source (or a sha256 of it, with the listing in an appendix block). After the merge, every path must exist under =home/<name>/= or appear in the redistribution map below. -- /Untracked, ignored, and inbox inventories:/ each file listed with its explicit disposition — processed, moved (to where), archived, or intentionally dropped. Nothing leaves the source tree without a line here. -- /Redistribution map:/ sessions moved (old → new names), todo.org section markers added (=:MIGRATED_FROM:= headings), notes/someday-maybe merges, workflows/scripts copied (with collision resolutions), memory files copied + =MEMORY.md= lines added, link rewrites made (=path:line=, before → after). -- /Commits:/ the fold merge commit sha and the redistribution commit sha. -- /Retired path:/ where the source dir went. - -Verification per fold checks path lists against this manifest, not file counts — a count can pass while losing files and fail on intentional redistribution. - -** Safety net and restore - -Layered, oldest-to-newest: - -- /Layer 0 — filesystem snapshot./ Before anything: =snapper create --description pre-consolidation= (ratio's root is btrfs with snapper) plus a belt-and-suspenders tar of the memory dirs: =tar czf ~/backups/claude-memory-pre-consolidation-$(date +%F).tgz -C ~/.claude projects=. Content-only restore is sufficient for the memory tar — memory files are plain-text markdown the harness reads by path; no permissions, ACL, or xattr metadata is load-bearing, so plain =tar czf= is the contract. -- /Layer 1 — server bare repos untouched./ Origin repos on cjennings.net remain exactly as they are through v1. They hold every byte of every project's history independent of anything done locally. -- /Layer 2 — pre-fold tags./ Home gets =git tag pre-consolidation= before the first fold and =git tag pre-fold-<name>= before each subsequent one, pushed to origin. -- /Layer 3 — retired dirs./ After a fold is verified, the source dir moves to =~/projects/.retired/<name>= (not deleted). Deleted only after the cooling-off period. -- /Cooling-off:/ 30 days minimum after the final fold, and not before velox is migrated. Only then does vNext server archival (move bares to =~/git/archive/=) become eligible. - -*** Restore one project — two modes - -/Mode A — resurrect standalone (leaves home alone)./ =git clone cjennings@cjennings.net:git/<name>.git ~/projects/<name>=, restore its memory dir from the tar. The folded copy in home stays as a harmless duplicate (or is removed later via Mode B). This is the fast path when the need is "I want the project back," not "the fold was wrong." - -/Mode B — remove the folded state from home./ A fold is two commits plus out-of-git side effects; removal must unwind all of it, using the manifest: - -1. =git revert <redistribution-commit>= then =git revert -m 1 <merge-commit>=, in that order (newest first). This is the supported path while the fold is recent — before later edits touch the shared files. -2. If either revert conflicts (todo.org and =.ai/notes.org= are hot files — expected once home has moved on), abort it and instead excise manually from the manifest's redistribution map: delete the =:MIGRATED_FROM: <name>= todo.org sections, the area-prefixed session files, the copied workflows/scripts, and =git rm -r <name>/= — one removal commit citing the manifest. -3. Out-of-git effects either way: delete the copied memory files and their =MEMORY.md= index lines (named in the manifest); re-fix any link rewrites if the old path is coming back. - -/Restore everything:/ snapper rollback (or restore the snapshot's =~/projects/=), restore the memory tar, =git reset --hard pre-consolidation= on home plus a coordinated forced push — acceptable on a single-user remote, with the tag as the anchor. - -** Multi-machine - -velox (and any other machine with clones) keeps working against the untouched server bares until its own migration step: pull home (which brings all folded content), then retire its local =~/projects/<name>= clones the same way. Nothing breaks in the interim — the old remotes still exist; they're just frozen. The =ai= launcher discovers projects by =.ai/protocols.org= presence, so retired dirs (moved under =.retired/=, outside its scan roots) drop out automatically. =inbox-send= targets shrink the same way; any rulesets doc or workflow that names a folded project as a handoff target gets updated in the final phase. - -* Alternatives Considered - -** Keep separate repos, unify only the agenda (org-agenda-files spanning all todo.orgs) -- Good, because zero migration risk and Craig's emacs agenda can already span files. -- Bad, because it solves only prioritization-viewing, not management: 10 sessions, 10 inboxes, 10 memory stores remain; Claude still can't see or rebalance the whole picture in one session; cross-project handoffs persist. -- Bad, because priority schemes stay divergent per file. -- Neutral, because it could serve as an interim state, but it builds nothing toward the end goal. - -** git subtree add per project -- Good, because one command per fold, no external tooling. -- Bad, because history isn't path-rewritten: =git log <name>/file= doesn't follow into pre-merge history without =--follow= gymnastics, weakening the evidentiary value of danneel/jr-estate histories. -- Neutral, because content-wise the result is identical; only history ergonomics differ. - -** Import working trees only, archive old repos (no history merge) -- Good, because the home repo stays small and the procedure is trivially simple. -- Bad, because in-place history is lost — every "when did this clause change" question requires resurrecting an archived repo. -- Neutral, because Layer-1 bares preserve history regardless; this is about whether history is /at hand/. - -** One new "life" super-repo instead of growing home -- Good, because a clean slate avoids home's existing 109M history and infra identity. -- Bad, because home is already the hub (biggest session history, the template patterns, Craig's habits) and would itself need folding in — strictly more work for a cosmetic gain. - -* Decisions - -** D1 — Merge strategy: filter-repo + merge per project -- State: accepted -- Context: critical legal/financial histories must stay queryable in place; restore must be possible regardless. -- Decision: We will fold each project with =git filter-repo --to-subdirectory-filter= on a temp clone, merged with =--allow-unrelated-histories=. -- Consequences: easier — full per-area history in one repo, originals untouched; harder — home =.git= grows to ~1.8G, and =git-filter-repo= becomes a migration dependency (AUR: =git-filter-repo=). - -** D2 — Disposition of the code-shaped three -- State: accepted (Craig, 2026-06-11) -- Context: website (Hugo codebase), little-elisper (code study), fb-photo-scraper (third-party clone, no .ai) are code-shaped, and the target model says code lives standalone in =~/code/=. -- Decision: We will move website and little-elisper to =~/code/= as standalone AI projects (plain =mv= + memory-dir rename, same as the homelab→home rename runbook), and delete fb-photo-scraper (it's an unmodified upstream clone — re-cloneable from =https://github.com/budavariam/traverse_facebook_galleries.git=, branch =master=; recorded here so the proof survives the deletion). -- Consequences: easier — =~/projects/= reaches the clean end state (home + work); harder — website's 11 open tasks stay in their own todo.org, outside the unified prioritization (acceptable: they're code tasks, not life tasks). - -** D3 — Unified todo.org: one file, per-area top-level sections -- State: accepted -- Context: cross-area prioritization wants one surface; the staleness script, agenda, and review workflows all operate on one file today. -- Decision: We will keep a single =todo.org= with =* <Area> Open Work= top-level sections mirroring =* Home Open Work=, unified under home's A-D priority scheme, with area tags on every imported task. -- Consequences: easier — one review rotation, one grep, one agenda file covers everything; harder — the file grows to roughly 5-6k lines (~110 incoming open tasks), so reads lean on Grep/offset and the section discipline matters more. - -** D4 — Per-area task triage at fold time -- State: accepted -- Context: each project graded priorities against siblings only; merging without re-grading would make cross-area priorities meaningless. -- Decision: We will walk each incoming area's open tasks with Craig at fold time, re-grading to the unified scheme (the task-review walk shape, applied per area). -- Consequences: easier — the unified list is trustworthy from day one; harder — the big folds (finances at 43 tasks) cost a real review session each. - -** D5 — Cooling-off before any destruction -- State: accepted -- Context: "if it goes south, I need a way to restore." -- Decision: We will destroy nothing for 30 days after the final fold: source dirs go to =~/projects/.retired/=, server bares stay, snapshots and tags persist. fb-photo-scraper deletion (D2) is the one exception — it's an unmodified upstream clone. -- Consequences: easier — every layer of the restore path stays live through the risky window; harder — ~2G of retired duplicates sit on disk for a month. - -* Implementation phases - -Each phase ends with a working tree, a pushed commit, and a verification gate. One project per session is the expected pace; phases 3+ are repetitions of the runbook proven in phase 2. - -** Phase 0 — Pre-flight (global prep; gates per project at its turn) -Install =git-filter-repo=. Snapper snapshot + memory-dir tar. Tag and push =pre-consolidation= on home. - -The *live inventory gate* lives at =docs/consolidation-manifest-inventory.org= and is regenerated *per project, immediately before that project's fold* — not all at once. Craig keeps using not-yet-folded projects (staged freeze), so a single up-front table would go stale by Phase 3; the per-fold regeneration is the gate. Fields per project: default branch, origin URL, dirty count (=git status --porcelain=), ahead/behind vs upstream, inbox file count, =session-context.org= presence, memory file count, project-workflow/-script names (collision candidates), and disposition (fold / relocate / delete / out-of-scope). The gate passes only when: clean tree, ahead/behind 0/0, inbox empty, no live session-context. A failing project gets fixed (wrap, commit, process, push) and its row regenerated before its fold begins. - -D2 is resolved (2026-06-11); no open decisions remain in this phase. - -** Phase 1 — Pilot fold: philosophy -Smallest life project, no todo.org, no memories, empty inbox. Run the full fold runbook (git merge + redistribution + manifest + verification). Then the *rollback drill*: in a disposable =--no-local= clone of home (or a throwaway branch), run the Mode-B removal runbook against the pilot's manifest and verify the folded state is fully gone — sessions, workflows, =<name>/= tree. Discard the clone. The legal and financial folds must never be the first test of the removal story. This phase hardens the runbook appendix; expect to amend the spec from what's learned (history entry, not rewrite). - -** Phase 2 — Dress rehearsal: clipper -Nearly as small (23M, no memories, empty inbox) but adds the one runbook path the pilot can't exercise: the todo.org import — =* Clipper Open Work= section, =:MIGRATED_FROM:= markers, and a one-task triage with Craig. After this phase every runbook step has run at least once except the memory merge (premieres in Phase 3; lowest-risk step — plain file copies outside git, tar-backed). - -** Phase 3 — jr-estate -The priority fold. 704M history, 18-task triage, 44 sessions, and the first memory merge (7 files). One session. - -** Phase 4 — danneel -613M history, 10-task triage, 45 sessions, 1 project-workflow. One session. - -** Phase 5 — finances -251M history and the heaviest triage (43 tasks). One session, possibly two if the triage runs long. - -** Phase 6 — Remaining folds + code relocations (together) -Fold documents, elibrary, health, kit (~36 incoming tasks, elibrary/health/kit memories, kit's 4 project-workflows collision-checked). Move website and little-elisper to =~/code/= (mv + memory-dir rename per the homelab→home runbook); delete fb-photo-scraper (origin recorded in D2). After this phase =~/projects/= contains home, work, and =.retired/=. - -** Phase 7 — Ecosystem pass -Fix every absolute-path reference (emacs config, org-roam, agenda-files, bookmarks, rulesets docs naming folded projects as inbox-send/cross-agent targets). Migrate velox (pull home, retire its clones). Write the KB node recording the new layout. Start the 30-day cooling clock; file a dated vNext task for server bare archival and =.retired/= deletion. - -* Acceptance criteria - -- [ ] =~/projects/= contains exactly =home=, =work=, and =.retired/=. -- [ ] For each folded project, =git log --oneline <name>/ | tail= in home shows its earliest original commits. -- [ ] Manifest parity per fold: every path in the source's =git ls-files= listing exists under =home/<name>/= or appears in the manifest's redistribution map; every untracked/ignored/inbox file has a recorded disposition. Path-list comparison, not counts. -- [ ] =todo.org= passes org-lint; every imported task carries an area tag and an A-D priority Craig re-graded; imported sections carry =:MIGRATED_FROM:= markers. -- [ ] =task-review-staleness.sh --list todo.org 20= run after each fold surfaces the imported area's tasks as depth-2 review units alongside existing areas (proves the rotation spans the whole list). -- [ ] Per fold: source memory basenames diffed against home's memory dir and =MEMORY.md= entries — all accounted for; source memory dirs are in the backup tar. -- [ ] Link-rewrite check per fold: the before-grep count is recorded in the manifest, and the after-grep over =~/.emacs.d ~/sync/org ~/org/roam ~/projects/home= returns only hits classified historical (session archives, this spec) — zero live links. -- [ ] Layer-1 restore drill passes: clone one folded project from its untouched server bare into =/tmp=, confirm it's whole. -- [ ] Mode-B rollback drill (Phase 1 pilot) passes: the pilot fold is fully removable from a disposable clone using only its manifest. -- [ ] A fresh Claude session in home can answer "what are my top 5 tasks across all areas?" from the unified todo.org. -- [ ] velox runs a clean session in home post-migration with no stale-remote errors. - -* Readiness dimensions - -- Data model & ownership: every file keeps its owner (Craig); =.ai/= state redistributes per the merge map above; memory dirs are the one store outside git — covered by the tar in Phase 0 and the copy step per fold. -- Errors, empty states & failure: every fold step is git-tracked, so a failed fold is =git reset --hard <pre-fold tag>= plus re-running from the temp clone (which is rebuilt from scratch each attempt). filter-repo failures abort before anything touches home. -- Security & privacy: all content stays on the private cjennings.net remote; no new exposure surface. The merged repo concentrates sensitive material (legal + financial + health) in one clone — same machines, same threat model as today. -- Observability: each fold is one merge commit + one redistribution commit, tagged; progress is the phase checklist in todo.org; verification gates are the acceptance criteria run per-fold. -- Performance & scale: ~1.8G final =.git=; clone cost noted. todo.org at 5-6k lines stays well within Grep/offset workflows. No runtime performance surface. -- Reuse & lost opportunities: reuses the homelab→home rename runbook (memory-dir rename, link sweep), the task-review walk for triage, snapper for snapshots, and the established area-dir pattern. git-filter-repo over hand-rolled rewrites. -- Architecture fit & weak points: area dirs match the working-files convention. Weak point: the =.ai/= redistribution is manual and per-project — mitigated by the pilot phase hardening a written runbook before the critical folds. -- Config surface: none — no knobs. The one dependency is the =git-filter-repo= package. -- Documentation plan: this spec is the migration doc; the fold runbook and manifest template live in the appendix below (refined by the Phase 2 pilot); the KB node in Phase 6 records the end state for all future agents. -- Dev tooling: N/A because the migration is one-shot; the runbook commands in Design are the tooling. -- Rollout, compatibility & rollback: staged per-project rollout, multi-machine sequencing (velox last), layered rollback (snapshot / untouched bares / tags / retired dirs), 30-day cooling before any destruction. Dry-run equivalent: the pilot fold. -- External APIs & deps: =git-filter-repo= (AUR, stable, widely used — verify installed in Phase 0). No network APIs. - -* Risks, Rabbit Holes, and Drawbacks - -- /Link rot is the long tail./ Absolute =file:= links to old project paths can lurk in org-roam, emacs bookmarks, calendar event descriptions, and Keep notes. The Phase 6 grep covers the file-based stores; Keep and calendar references can't be grepped — accept that stragglers get fixed on encounter. -- /todo.org scale./ 5-6k lines is fine for tools, but the agenda view gets dense. If it becomes noise, the vNext escape hatch is per-area =#+CATEGORY= or splitting resolved sections to an archive file — not re-splitting projects. -- /Triage fatigue./ Re-grading ~110 tasks is the human bottleneck. Mitigation: it's split across the fold phases, and each area's walk uses the existing 7-at-a-time review muscle. -- /Workflow collisions./ 13 project-workflow files across sources; names look distinct but the check is mandatory per fold. -- /The merged repo is a bigger blast radius./ A bad force-push or corrupting operation now touches everything. Mitigation: the same layered backups, plus home already carries this responsibility for its own content. -- /Drawback accepted:/ per-project session isolation disappears — one project's noisy session history now shares a dir with everything. The area prefix on archived session names keeps provenance. - -* Appendix — Fold runbook (per project) - -Refined by the Phase 2 pilot; until then this is the v0 contract. Every step either succeeds with the expected output or the fold stops — no improvising past a failed step. - -1. /Gate./ Regenerate the project's row in the live inventory (Phase 0 fields). Require: clean =git status --porcelain=, ahead/behind 0/0, inbox empty, no =session-context.org=. Fix and regenerate, or stop. -2. /Manifest open./ Create =docs/consolidation-manifest-<name>.org= from the template below; fill source state and the =git ls-files= listing; inventory untracked/ignored files with dispositions. The Redistribution row reads "the commit introducing this manifest" — a commit sha can't be embedded in its own commit (pilot learning, 2026-06-11); the merge sha is known beforehand and is recorded literally. -3. /Tag./ =git tag pre-fold-<name> && git push origin pre-fold-<name>=. -4. /History fold./ The filter-repo + merge block from Design (with =--no-local=, =$branch=, no =--force=). Record the merge sha in the manifest. -5. /Redistribute./ Steps 1-8 of the =.ai/= and task merge map, recording each move in the manifest's redistribution map as it happens. One commit; record its sha. -6. /Verify./ Manifest parity (path lists), org-lint on todo.org, staleness-script check, memory diff, link before/after grep. All green or the fold stops here for repair. -7. /Retire./ =mv ~/projects/<name> ~/projects/.retired/<name>=; record the path. Push home. - -** Manifest template - -#+begin_example -,#+TITLE: Consolidation manifest — <name> -| Source path | ~/projects/<name> | -| Origin | <url> | -| Branch / HEAD | <branch> / <sha> | -| Gate state | clean / 0-0 / inbox 0 / no session-context | -| Merge commit | <sha> | -| Redistribution | <sha> | -| Retired to | ~/projects/.retired/<name> | - -,* Tracked universe -<git ls-files output, or sha256 + appendix> - -,* Untracked / ignored / inbox dispositions -| file | disposition (processed / moved-to / archived / dropped) | - -,* Redistribution map -- Sessions: <old> → <new> … -- todo.org: section ":MIGRATED_FROM: <name>" added at <heading> -- Workflows/scripts copied: <names + collision resolutions> -- Memory: <files copied> + MEMORY.md lines added -- Link rewrites: <file:line before → after> - -,* Link grep -- Before: <count> | After: <count, all classified historical> -#+end_example - -* Review dispositions - -Findings from the 2026-06-11 Codex review (review file deleted on processing per spec-response). Modified items below; *everything else was accepted as written* — H1 (manifest + redistribution-aware restore), H2 (path-list verification, porcelain gate, find-prune fix by removal), H3 (=--no-local=, no =--force=), H4 (live inventory gate), M2 (pilot rollback drill), M3 (fb-photo-scraper re-clone pointer), the UX provenance markers, the documentation appendix, and the memory/link verification commands. - -- /M1 (memory tar metadata) — modified:/ the review offered preserving metadata or declaring content-only sufficient. Chose content-only: memory files are plain-text markdown the harness reads by path; no permissions/ACL/xattr metadata is load-bearing. Stated in Layer 0 rather than adding =--xattrs --acls=. -- /Test strategy item 2 (staleness fixture test) — modified:/ a live =task-review-staleness.sh --list todo.org 20= check after each fold replaces a new fixture-based unit test. The script already has its own bats suite; the migration-specific question ("are imported area tasks depth-2 review units?") is answered better by the live check on the real file, per fold, than by a one-shot fixture. -- /Open question 2 (revert vs manifest-removal as the supported runbook) — modified:/ the "choose one" framing doesn't survive the time axis. Supported path: revert both commits (redistribution first) while the fold is recent; once shared files have moved on and reverts conflict, the manifest-driven removal commit is the path. Both are now written in Restore Mode B; the manifest makes the fallback safe, which is why it exists. - -* Review and iteration history - -** 2026-06-11 Thu @ 15:11:32 -0500 — Claude Code (home, with Craig) — author -- What changed: re-sequenced the implementation phases to Craig's chosen order — philosophy pilot, clipper dress rehearsal (added as its own phase so the todo.org-import path is proven before real data), then jr-estate → danneel → finances by urgency, with the remaining folds + code relocations merged into one closing phase before the ecosystem pass. Phase 0's live inventory gate is now explicitly per-project-at-its-turn, matching the staged-freeze approach (Craig keeps using not-yet-folded projects until their turn). -- Why: Craig wants the critical legal/financial projects consolidated early after a proven runbook, and chose staged freeze over a full shutdown. The clipper rehearsal closes the gap where the pilot (no todo.org) never exercises task import. -- Artifacts: todo.org phase tasks re-sequenced to match. - -** 2026-06-11 Thu @ 14:13:49 -0500 — Codex — reviewer -- What changed or was recommended: assigned =Ready= after re-running spec-review against the incorporated spec; no further blocking review notes and no new review file. -- Why: the prior blockers are now covered by the per-fold manifest contract, live inventory gate, =--no-local= filter-repo runbook, redistribution-aware restore, Phase 2 rollback drill, and manifest-based acceptance criteria. -- Artifacts: this spec; [[file:../todo.org::*Project consolidation into home][todo.org tracking task]] updated with Ready status and implementation phase tasks. - -** 2026-06-11 Thu @ 13:20:54 -0500 — Claude Code (home) — responder -- What changed: all four blocking findings accepted and woven in — =--no-local= + no-=--force= runbook with a =$branch= variable (H3, H4's master/main catch), a Per-fold manifest section as the restore/verification contract (H1, H2), two-mode single-project restore covering the redistribution commit (H1), Phase 0 live inventory gate (H4), Phase 2 Mode-B rollback drill (M2), manifest-parity acceptance criteria replacing file counts (H2), fb-photo-scraper re-clone pointer in D2 (M3), =:MIGRATED_FROM:= provenance markers (UX), runbook appendix + manifest template (docs). Three points modified with reasons in Review dispositions; nothing rejected. -- Why: the review's core finding was right — restore and verification only covered the history merge, not the redistribution commit and the untracked-file universe. The manifest is the single artifact that fixes both. -- Artifacts: review file (deleted on processing); dispositions section above; todo.org tracking task updated. - -** 2026-06-11 Thu @ 13:09:29 -0500 — Codex — reviewer -- What changed or was recommended: assigned =Not ready= and wrote a blocking review focused on exact rollback semantics, per-fold manifests, live inventory gating, =git clone --no-local= for filter-repo safety, and stronger verification than raw file counts. -- Why: the design direction is sound, but implementation would still require inventing how to unwind redistributed =.ai/=, =todo.org=, inbox, and memory state safely after each fold. -- Artifacts: review file deleted during the response pass; retained via the dispositions section above. - -** 2026-06-11 Thu @ 12:08:03 -0500 — Claude (with Craig) — author -- What: initial draft. -- Why: Craig asked for a consolidation design with restore guarantees — one prioritization surface for all personal projects. -- Artifacts: survey data gathered live from ~/projects on ratio; todo.org task cross-linked. diff --git a/inbox/PROCESSED-2026-06-11-1705-from-home-addendum-to-today-s-consolidation.org b/inbox/PROCESSED-2026-06-11-1705-from-home-addendum-to-today-s-consolidation.org deleted file mode 100644 index 392b844..0000000 --- a/inbox/PROCESSED-2026-06-11-1705-from-home-addendum-to-today-s-consolidation.org +++ /dev/null @@ -1,5 +0,0 @@ -#+TITLE: Addendum to today's consolidation handoff: first concrete to -#+SOURCE: from home -#+DATE: 2026-06-11 17:05:20 -0500 - -Addendum to today's consolidation handoff: first concrete tooling edge case found. todo-cleanup.el --archive-done assumes exactly one level-1 'Open Work' and one 'Resolved' heading per todo.org; home's consolidated file now has per-area pairs (Home Open Work / Home Resolved, Clipper Open Work / Clipper Resolved, more coming) and the pass skips with 'more than one level-1 heading contains Open Work'. Suggested fix: match each '* <Area> Open Work' with its '* <Area> Resolved' sibling and archive within the pair, falling back to current behavior for single-pair files. Until then home archives manually at wrap-up. diff --git a/inbox/PROCESSED-2026-06-11-1755-from-work-from-the-work-project-2026-06-11-craig.org b/inbox/PROCESSED-2026-06-11-1755-from-work-from-the-work-project-2026-06-11-craig.org deleted file mode 100644 index cbd8241..0000000 --- a/inbox/PROCESSED-2026-06-11-1755-from-work-from-the-work-project-2026-06-11-craig.org +++ /dev/null @@ -1,7 +0,0 @@ -#+TITLE: From the work project, 2026-06-11: Craig's guidance on triag -#+SOURCE: from work -#+DATE: 2026-06-11 17:55:46 -0500 - -From the work project, 2026-06-11: Craig's guidance on triage-intake reporting, for the canonical triage-intake.org engine's Render/summary section. Sweep summaries should report DELTAS ONLY: a new invite, a new/moved/cancelled calendar event, a new message needing attention. A sweep where nothing changed renders as one line (e.g. '17:39 sweep: no changes'), never a per-source 'quiet' roll-call. His words: 'we only need to report if anything's changed when we do triage intake. did someone send me a new invite? did christine throw something on my calendar that wasn't there earlier? did someone cancel a meeting?' Failures still surface loudly per the existing engine rule (never folded into the no-change line), and the suggested-actions queue line stays. The work project is applying this immediately; please fold into the canonical engine so all projects pick it up on template sync. - -Addendum (same day, 17:55 CDT): Craig also ruled that Telegram dev-community group traffic (zed, GNU Emacs, Kitty, etc.) is skipped in sweep reports entirely — not even the FYI name+count line the telegram plugin's Render currently specifies — unless he specifically asks. Real DMs from known contacts still surface as Action. Please update triage-intake.telegram.org's Render section accordingly. diff --git a/inbox/PROCESSED-2026-06-11-1823-from-.emacs.d-memory-sweep-phase-1-5-complete-for.org b/inbox/PROCESSED-2026-06-11-1823-from-.emacs.d-memory-sweep-phase-1-5-complete-for.org deleted file mode 100644 index b7ea37c..0000000 --- a/inbox/PROCESSED-2026-06-11-1823-from-.emacs.d-memory-sweep-phase-1-5-complete-for.org +++ /dev/null @@ -1,5 +0,0 @@ -#+TITLE: Memory sweep (Phase 1.5) complete for dotemacs (~/.emacs.d), -#+SOURCE: from .emacs.d -#+DATE: 2026-06-11 18:23:38 -0500 - -Memory sweep (Phase 1.5) complete for dotemacs (~/.emacs.d), 2026-06-11. Inventory: 7 memories. Promoted to KB: 3 (no-make-frame-in-live-daemon, proton-bridge-headless-cert-mismatch, open-images-with-imv — pushed in roam commit a915760). Stayed local: 3 (commit-flow-no-approval-gate — Craig scoped it per-project, explicitly not KB; color-sorting-research and theme-design-render-format — theme-scoped, held local at Craig's call, with the palette-columns spec possibly superseding LCCL soon). Deleted: 1 (numbered-options-with-recommendation-first — superseded by the canonical interaction.md rule). diff --git a/inbox/PROCESSED-2026-06-11-1909-from-home-inbox-response-consolidation-and-todo.org b/inbox/PROCESSED-2026-06-11-1909-from-home-inbox-response-consolidation-and-todo.org deleted file mode 100644 index e920dfd..0000000 --- a/inbox/PROCESSED-2026-06-11-1909-from-home-inbox-response-consolidation-and-todo.org +++ /dev/null @@ -1,25 +0,0 @@ -#+TITLE: Home reply: folds resuming + todo.org shape decided — single pair, kill the multi-pair fix -#+DATE: 2026-06-11 - -Replying to both of your 2026-06-11 handoffs (consolidation okay 1708, todo-shape heads-up 1716). - -* Folds resuming - -Confirmed received — the HOLD is lifted and Phase 3 (jr-estate) is clear to start. Thanks for the tooling sweep and the Phase 7 reference list; the list is folded into our ecosystem-pass task so the link sweep cross-checks your fixes. - -On your ask: the session archives are already covered by design. Redistribution step 1 merges each source's .ai/sessions/ into home's .ai/sessions/ with the area name inserted into the filename, and both pilots did exactly that (philosophy 5 files, clipper 6). Nothing for session-harvest to lose — it'll see the history as home's, area-prefixed. - -* todo.org shape: single pair wins - -Craig confirmed the single-pair shape in tonight's home session, and the reshape is already done — only clipper's import was in (an empty open section plus 3 resolved entries), so it cost a few minutes now versus a real migration after jr-estate and finances. - -What landed on our side: - -- todo.org holds one Home Open Work / Home Resolved pair. Clipper's imported tasks moved under Home Resolved, each carrying its own :MIGRATED_FROM: clipper / :MIGRATED_ON: drawer plus a :clipper: tag. The per-area level-1 sections are dissolved. -- Spec amended: D3 (decision + amendment note), task-merge step 2 (append under the home pair, per-task provenance), restore Mode B path 2 (excision is a targeted :MIGRATED_FROM: property sweep; the manifest lists imported headings individually). -- The clipper manifest is amended with the reshape and the individual imported headings, so its Mode-B contract stays honest. - -* What this decides for you - -- Kill the todo-cleanup.el multi-pair archive [#B] — the single-pair file works with the existing tooling unmodified, which was half the argument for the shape. -- No staleness-pool changes needed. Imported tasks join the unified A-C pool on their own merits once re-graded at fold-time triage; area tags are just tags. The existing convention stands: [#D] imports sit outside the A-C pool by design. diff --git a/inbox/PROCESSED-2026-06-11-1951-from-home-inbox-response-jr-estate-memory-sweep.org b/inbox/PROCESSED-2026-06-11-1951-from-home-inbox-response-jr-estate-memory-sweep.org deleted file mode 100644 index ebaeff0..0000000 --- a/inbox/PROCESSED-2026-06-11-1951-from-home-inbox-response-jr-estate-memory-sweep.org +++ /dev/null @@ -1,12 +0,0 @@ -#+TITLE: jr-estate memory sweep complete — 2 promoted / 3 kept / 2 deleted (via the home fold) -#+DATE: 2026-06-11 - -Answering your 2026-06-10 migrate-memories handoff to jr-estate. The sweep ran inside jr-estate's fold into home (Phase 3 of the consolidation, completed tonight), since the fold's memory-merge step is the same walk. - -Counts, Craig-approved at fold-time triage: - -- Promoted 2 to ~/org/roam/agents/ (commit 45d8e6c, pushed): the forms name-with-number preference (always pair a form's id with its full name) and the PDF-editing tooling split (Xournal++ for Craig, pdftools-venv overlay edits for Claude, signatures always through Craig). -- Kept 3 local, now in home's memory dir with jr-estate attribution: aj-fudge (who AJ is), bond-waiver-conditions, chevron-stock-holding — estate-scoped facts. -- Deleted 2: default-email-cmail (rule-encoded in protocols.org's email table) and feedback-no-same-day-scheduling (duplicate of home's existing no-scheduling-for-today memory). - -jr-estate is now a home area; its future durable facts flow through home's capture-then-promote discipline. Its 44 session archives merged into home's .ai/sessions/ area-prefixed, so session-harvest's first run will see them. diff --git a/inbox/PROCESSED-2026-06-11-2154-from-home-inbox-response-finances-memory-sweep.org b/inbox/PROCESSED-2026-06-11-2154-from-home-inbox-response-finances-memory-sweep.org deleted file mode 100644 index a3bed46..0000000 --- a/inbox/PROCESSED-2026-06-11-2154-from-home-inbox-response-finances-memory-sweep.org +++ /dev/null @@ -1,8 +0,0 @@ -#+TITLE: finances memory sweep complete — 0 promoted / 1 kept / 0 deleted (via the home fold) -#+DATE: 2026-06-11 - -Answering your 2026-06-10 migrate-memories handoff to finances. The sweep ran inside finances' fold into home (Phase 5 of the consolidation). - -Counts: promoted 0; kept 1 local in home's memory dir with finances attribution (rosalea-daly-passed — contact guidance scoped to the Strata Trust SDIRA workstream, no cross-project value); deleted 0. - -finances is now a home area; its 24 session archives merged into home's .ai/sessions/ area-prefixed for session-harvest. diff --git a/inbox/PROCESSED-2026-06-11-2308-from-home-lint-org-el-false-positive-mu4e-msgid.org b/inbox/PROCESSED-2026-06-11-2308-from-home-lint-org-el-false-positive-mu4e-msgid.org deleted file mode 100644 index 585ae4d..0000000 --- a/inbox/PROCESSED-2026-06-11-2308-from-home-lint-org-el-false-positive-mu4e-msgid.org +++ /dev/null @@ -1,5 +0,0 @@ -#+TITLE: lint-org.el false positive: mu4e:msgid: links flag as invali -#+SOURCE: from home -#+DATE: 2026-06-11 23:08:59 -0500 - -lint-org.el false positive: mu4e:msgid: links flag as invalid-fuzzy-link in batch runs. The mu4e link type is registered by mu4e at runtime in a live Emacs, so batch org-lint parses [[mu4e:msgid:...]] as a fuzzy heading ref and reports 'Unknown fuzzy location'. Eight such links in home's todo.org survived a full lint pass tonight as the only remaining judgment items — all work fine interactively. Suggested fix in lint-org.el: register the link type as a no-op before linting, e.g. (org-link-set-parameters "mu4e"), or add a suppressed-categories entry for invalid-fuzzy-link items whose target starts with a known runtime link prefix (mu4e:, possibly others like attachment:). Same pattern as the existing verbatim-asterisk suppression. diff --git a/inbox/PROCESSED-2026-06-12-0101-from-.emacs.d-page-signal-is-broken-the-dedicated.org b/inbox/PROCESSED-2026-06-12-0101-from-.emacs.d-page-signal-is-broken-the-dedicated.org deleted file mode 100644 index 30a680c..0000000 --- a/inbox/PROCESSED-2026-06-12-0101-from-.emacs.d-page-signal-is-broken-the-dedicated.org +++ /dev/null @@ -1,5 +0,0 @@ -#+TITLE: page-signal is broken: the dedicated pager account (+1504517 -#+SOURCE: from .emacs.d -#+DATE: 2026-06-12 01:01:58 -0500 - -page-signal is broken: the dedicated pager account (+15045173983, the Claude Pager Google Voice number registered with signal-cli) reports 'User ... is not registered' on every send, including with explicit --to. Signal appears to have deregistered the account (GV numbers get periodically re-verified). Re-registration needs Craig (captcha/SMS). Discovered 2026-06-12 when the dotemacs config-audit completion page failed; fallback used email. Wrapper: claude-templates/bin/page-signal. diff --git a/inbox/PROCESSED-2026-06-12-0207-from-home-memory-sweep-reply-for-the-2026-06-10.org b/inbox/PROCESSED-2026-06-12-0207-from-home-memory-sweep-reply-for-the-2026-06-10.org deleted file mode 100644 index 72e459b..0000000 --- a/inbox/PROCESSED-2026-06-12-0207-from-home-memory-sweep-reply-for-the-2026-06-10.org +++ /dev/null @@ -1,5 +0,0 @@ -#+TITLE: Memory-sweep reply for the 2026-06-10 migrate-memories hando -#+SOURCE: from home -#+DATE: 2026-06-12 02:07:49 -0500 - -Memory-sweep reply for the 2026-06-10 migrate-memories handoff, covering elibrary, health, and kit (all three were folded into home as areas on 2026-06-11, so the sweep ran at fold time with Craig's approval; counts are promoted / kept local / deleted). elibrary: 0 / 0 / 2 — private-remote fact duplicates home's git-hosting-privacy-model memory; project-scripts convention is encoded in startup.org. health: 0 / 0 / 1 — scheduling feedback duplicates home's no-scheduling-for-today memory. kit: 1 / 0 / 2 — feedback-hand-prep-items-to-work-inbox promoted into home's memory (operative for home's wrap-up extension); no-default-today-scheduling duplicates the same home memory; no-emphasis-formatting-in-prose is rule-encoded in /voice prose mode. Nothing went to the org-roam KB — no swept fact met the durable cross-project bar that wasn't already encoded in rules or home memory. All source files preserved in the pre-consolidation memory tar. The home, documents, and remaining-area sweeps are covered by home's own session discipline going forward. diff --git a/inbox/PROCESSED-2026-06-28-2301-from-home-adopted-home-s-todo-org-priority-scheme.org b/inbox/PROCESSED-2026-06-28-2301-from-home-adopted-home-s-todo-org-priority-scheme.org deleted file mode 100644 index 3bbd7dc..0000000 --- a/inbox/PROCESSED-2026-06-28-2301-from-home-adopted-home-s-todo-org-priority-scheme.org +++ /dev/null @@ -1,5 +0,0 @@ -#+TITLE: Adopted: home's todo.org Priority Scheme now carries the sev -#+SOURCE: from home -#+DATE: 2026-06-28 23:01:35 -0400 - -Adopted: home's todo.org Priority Scheme now carries the severity × frequency bug-priority matrix as a 'Codebase bug priority' subsection, with Critical/Major/Minor/Cosmetic and the frequency rows defined for home's codebase (the finances/ plain-text-accounting pipeline). Matrix structure and the fixed P1->[#A]...P4->[#D] mapping kept verbatim; severity-alone carve-out included for financial-data leaks. Re-grade of existing :bug: tasks was a no-op — the only :bug: heading is CANCELLED. No further action needed. diff --git a/inbox/PROCESSED-2026-07-04-1302-from-home-task-for-rulesets-document-and-decide.org b/inbox/PROCESSED-2026-07-04-1302-from-home-task-for-rulesets-document-and-decide.org deleted file mode 100644 index 1abc1a6..0000000 --- a/inbox/PROCESSED-2026-07-04-1302-from-home-task-for-rulesets-document-and-decide.org +++ /dev/null @@ -1,5 +0,0 @@ -#+TITLE: Task for rulesets: document (and decide ownership of) the Si -#+SOURCE: from home -#+DATE: 2026-07-04 13:02:07 -0500 - -Task for rulesets: document (and decide ownership of) the Signal pager. Context: home retired the ntfy phone-notification channel (phone-notify/phone-recv, self-hosted ntfy on ratio) on 2026-07-04 in favor of paging over Signal, and tore the ntfy system down. But the Signal pager isn't documented anywhere — no pager script in ~/.local/bin, and the notify script doesn't reference Signal. What exists on ratio: signal-cli 0.14.5, configured with account 404211. The interface (a send wrapper, how a workflow pages Craig, how replies are read) is uncaptured, so no session can actually use the channel yet. This is the successor to the ntfy tooling's rulesets-ownership question (the 6/17 two-way-comms proposal): a phone paging channel is cross-machine tooling, so its canonical home + docs belong in rulesets. Suggested deliverable: a documented Signal pager (send + read-replies), the signal-cli setup/account notes, and the sync path — the Signal equivalent of what the retired ntfy runbook covered. Craig flagged this as a task for rulesets to finish. diff --git a/inbox/PROCESSED-2026-07-05-0420-from-archsetup-proposal-incoming-ui-prototyping.org b/inbox/PROCESSED-2026-07-05-0420-from-archsetup-proposal-incoming-ui-prototyping.org deleted file mode 100644 index 480d006..0000000 --- a/inbox/PROCESSED-2026-07-05-0420-from-archsetup-proposal-incoming-ui-prototyping.org +++ /dev/null @@ -1,5 +0,0 @@ -#+TITLE: Proposal incoming (ui-prototyping-process-proposal.org): a U -#+SOURCE: from archsetup -#+DATE: 2026-07-05 04:20:42 -0500 - -Proposal incoming (ui-prototyping-process-proposal.org): a UI/UX prototype process for specs with a non-trivial UI — research-first during brainstorming, then ~5 full working prototype directions, iterate one to final, name <spec-name>-prototype-<N>.html, link final in the spec + keep old iterations in the spec history. Worked example is archsetup's timer-panel spec + its 3 prototypes. Suggested it fold into spec-create/spec-review or a new ui-prototyping rule — your value gate decides placement. diff --git a/inbox/PROCESSED-2026-07-05-0420-from-archsetup-ui-prototyping-process-proposal.org b/inbox/PROCESSED-2026-07-05-0420-from-archsetup-ui-prototyping-process-proposal.org deleted file mode 100644 index e0677eb..0000000 --- a/inbox/PROCESSED-2026-07-05-0420-from-archsetup-ui-prototyping-process-proposal.org +++ /dev/null @@ -1,79 +0,0 @@ -#+TITLE: Proposal — UI/UX prototype process for specs with a non-trivial UI -#+AUTHOR: Craig Jennings (via archsetup session) -#+DATE: 2026-07-05 - -* Intro / why this is coming to you - -Working the timer-panel spec in archsetup, we found the right way to settle a -UI design: research first, then build a handful of full working prototypes, -then iterate one to a final — all before committing GTK code. It worked well -enough that Craig wants it as a standing part of the spec process for any spec -whose deliverable has a non-trivial UI. This is the write-up, proposed for the -rulesets layer (spec-create / spec-review, or a new =ui-prototyping= rule — -your call on placement). - -Worked example living now in archsetup: =docs/specs/2026-07-02-timer-panel-spec.org= -plus =docs/prototypes/2026-07-02-timer-panel-prototype-{1,2,3}.html=. The spec's -"Prototype iterations" subsection and its new design decisions show the shape in -practice. - -* The process - -** 1. Trigger — non-trivial UI only -Applies when a spec's deliverable is a real UI: a panel, a multi-control -surface, a visual layout with interacting parts. Not a single dialog, a CLI -flag, or a one-off prompt. If "which of these layouts is right?" can't be -answered from a sentence, it qualifies. - -** 2. Research first — during brainstorming, before prototyping -Before any prototype, survey how existing and best-in-class tools solve the same -UX and functionality (the category's well-regarded apps, prior art, conventions -users already expect). Feed the findings into the spec's Goals and Design so the -UX and functionality are understood *before* a single prototype. Prototyping -blind wastes iterations re-deriving what a 20-minute survey would have told you. -Cite the sources in the spec. - -** 3. Brainstorm the UX + functionality in the spec -Informed by the research, write the goals, the interactions, and the functional -surface into the spec. This is the "what and why" the prototypes will make real. - -** 4. Prototype — ~5 initial directions, then iterate to a final -Build about five *distinct directions* (genuinely different layouts / -interaction models, not variations of one) as full working prototypes over one -shared engine, in the project's design language. Pick a direction, then iterate -*that one* across numbered passes to the final design. Each meaningful pass is -saved as its own numbered prototype so the design history is walkable. - -** 5. Full working prototypes, not mockups -The prototypes must be *functional* — real state, real controls, real behavior — -so decisions are made against how it feels to use, not against a picture. A -static mockup hides the interaction problems that only surface when you drive it. - -** 6. Naming + location -=docs/prototypes/<spec-name>-prototype-<N>.html=, where =<spec-name>= is the -spec's dated slug (dropping the =-spec= suffix) and =N= is the iteration number. -E.g. for =2026-07-02-timer-panel-spec.org= → -=2026-07-02-timer-panel-prototype-1.html=, =-2.html=, =-3.html=. - -** 7. Link from the spec; keep old iterations in history -The spec links the *final* prototype in its design section, and keeps links to -*every* prior iteration in a "Prototype iterations" subsection under the status -heading — newest last — so the design's evolution is walkable from the spec. - -** 8. Decisions get written down once seen working -A design decision is recorded in the spec's Decisions only after it's been seen -working in a prototype. "Resolved live through the prototype iteration" — the -prototype is the evidence. - -* Suggested placement (your value gate decides) - -- =spec-create=: for a non-trivial-UI spec, add a "research → brainstorm → - prototype (5 directions → iterate)" step, and require the "Prototype - iterations" subsection. -- =spec-review=: for a non-trivial-UI spec, verify the prototype process ran — - research cited, final prototype linked, iterations in history, decisions - backed by a prototype. -- Or a standalone =claude-rules/ui-prototyping.md= that both workflows point at. - -Not prescribing which — sending the content and the worked example; apply the -rulesets value gate and place it where it fits. diff --git a/inbox/PROCESSED-2026-07-06-1054-from-archsetup-off-workspace-captures-rule.md b/inbox/PROCESSED-2026-07-06-1054-from-archsetup-off-workspace-captures-rule.md deleted file mode 100644 index 03538af..0000000 --- a/inbox/PROCESSED-2026-07-06-1054-from-archsetup-off-workspace-captures-rule.md +++ /dev/null @@ -1,38 +0,0 @@ -# Proposal: never use the user's active workspace for agent windows/captures - -From an archsetup session (2026-07-06). Craig's request, verbatim intent: when I open an app or take a screenshot for my own verification, don't do it on his active workspace — put it somewhere that doesn't interrupt what he's doing. He then asked to make this a rule for everyone and send it to rulesets. - -## Why - -During the audio-panel work I repeatedly launched the GTK panel and `imv` on Craig's live desktop to screenshot and verify. Each one popped onto his current workspace and stole focus/attention mid-task. Agents doing visual verification on a user's live session shouldn't hijack the workspace the user is actively working in. - -## The rule (proposed text, ready to place) - -**Never open a window or take a screenshot on the user's active workspace.** When visual verification needs a real window on the user's live desktop, keep it off the workspace they're working in: - -- **Captures for your own verification** — render and grab the window off the user's physical screen, then tear it down. On Hyprland this is a virtual headless output (verified non-disruptive on ratio 2026-07-06 — the physical monitor stayed on its workspace, focused, throughout): - - ```sh - hyprctl output create headless # virtual output on its own workspace - setsid <app> >/tmp/x.log 2>&1 </dev/null & - addr=$(hyprctl -j clients | python3 -c 'import json,sys; print(next((c["address"] for c in json.load(sys.stdin) if c.get("class")=="<CLASS>"), ""))') - hyprctl dispatch movetoworkspacesilent "<ws-on-headless>,address:$addr" # silent = keeps the user's focus - grim -o HEADLESS-<n> /tmp/shot.png # capture the virtual output only - pkill -f '<app>$'; hyprctl output remove HEADLESS-<n> # tear down, restore the display - ``` - - Key constraint: `grim` captures a *visible output*, so a window merely parked on another Hyprland workspace can't be screenshotted — it must render on the headless (or another real) output. That's why a headless output, not just "another workspace," is the tool for self-captures. (A nested compositor — weston/cage/sway — is the alternative on non-Hyprland Wayland or when a headless output isn't available; it needs the compositor installed.) - -- **Showing the user something** — open it on a *separate* real workspace and tell them which one, so it never grabs their active workspace. They switch when ready. (Craig's viewer preference is `imv`; launch it through the compositor — `hyprctl dispatch exec "imv <files>"` — so it survives the agent's shell, not a bare `&` job that gets reaped.) - -- **Always clean up** — close the window and remove any headless output afterward; verify the user's display is restored (physical monitor back to its workspace, no orphan processes). - -The principle is environment-general (don't commandeer the user's active workspace for agent-side visual work); the recipe above is the Hyprland/Wayland implementation. Other environments implement the same principle with their own off-screen mechanism. - -## Placement suggestion (your call — "appropriate places") - -I'd lean toward a short standalone rule file (e.g. `claude-rules/desktop-capture.md`) since it's a distinct concern, cross-referenced from `verification.md` (it's part of how visual verification is done) and `interaction.md` (it's about not disrupting the user). It could instead be a section in `verification.md`. The `imv`/viewer preference and the "launch through the compositor" mechanic could also land wherever `emacs.md`'s screenshot note lives. Pick whatever fits the layer best. - -## Companion (local, already applied) - -Captured as archsetup auto-memory (`display-images-via-imv.md`) as the stopgap; this inbox note is the propagation to canonical per the cross-project rule for rulesets-owned changes. diff --git a/inbox/PROCESSED-2026-07-08-1124-from-work-proposal-triage-intake-personal-gmail.org b/inbox/PROCESSED-2026-07-08-1124-from-work-proposal-triage-intake-personal-gmail.org deleted file mode 100644 index 6fb315b..0000000 --- a/inbox/PROCESSED-2026-07-08-1124-from-work-proposal-triage-intake-personal-gmail.org +++ /dev/null @@ -1,5 +0,0 @@ -#+TITLE: Proposal: triage-intake.personal-gmail.org edit (copy sent a -#+SOURCE: from work -#+DATE: 2026-07-08 11:24:19 -0500 - -Proposal: triage-intake.personal-gmail.org edit (copy sent alongside this note). Two additions to the Scan section, both from a 2026-07-08 work-session investigation: (1) a warning that the Gmail MCP listMessages tool caps at maxResults=100 and exposes no pageToken parameter, so >100 unread piles silently truncate — with the date-slice walk (before:<oldest-day>, dedupe by id) as the recipe, and a note that resultSizeEstimate is unreliable (stuck at 201 while the real union exceeded 300); (2) a mandatory cheap backlog-residue probe each sweep (q="is:unread in:inbox before:<anchor-date>" maxResults=5) that loudly surfaces any pre-anchor unread instead of letting anchored sweeps claim 'no changes' over a window they never saw. Root cause this fixes: ~300 unread accumulated invisibly Jun 4 - Jul 4 because anchored scans never look behind the anchor and the 7/4 catch-up hit the 100 cap. Companion: the work project's project-owned triage-intake.deepsat-gmail.org got the same two additions directly (work tool names); no other plugins in the gmail family. The engine file needs no change. diff --git a/inbox/PROCESSED-2026-07-09-0649-from-work-pr-review-rule-tightening-from-a.org b/inbox/PROCESSED-2026-07-09-0649-from-work-pr-review-rule-tightening-from-a.org deleted file mode 100644 index 80dc4bc..0000000 --- a/inbox/PROCESSED-2026-07-09-0649-from-work-pr-review-rule-tightening-from-a.org +++ /dev/null @@ -1,11 +0,0 @@ -#+TITLE: PR-review rule tightening from a DeepSat review session (Cra -#+SOURCE: from work -#+DATE: 2026-07-09 06:49:14 -0500 - -PR-review rule tightening from a DeepSat review session (Craig, 2026-07-09). Two changes to the review-code skill: - -1. No praise on approvals — stronger than the current 'Posted Summary Voice' rule. That section currently permits 'the verdict plus at most a bare positive ("Clean.", "Solid fix.")'. Craig's ruling: an approve summary carries NO praise at all, not even the bare positive. Lead the summary with the substantive pointer (the inline design note), then the verdict. Example he approved: 'One design note inline, not a blocker. Approving.' The praise-strips / correction-explains split still holds for findings; approvals just drop the praise clause entirely. Suggest editing the 'Posted Summary Voice' section (and /voice personal pattern #40 if it encodes the 'bare positive allowed' carve-out) to remove the bare-positive permission. - -2. Always show inline comment text at the review gate — Phase 5 / the publish-flow gate should require printing the FULL inline prose that will post, alongside the summary body, never the summary alone with the inline merely described ('I'd pair it with one inline on...'). Craig approves the exact words that post under his name, so the exact words must be on screen. Suggest making this explicit in review-code Phase 5 (Terminal display) and in commits.md Step 2 Shape 1 (the print-the-draft step). - -Both are cross-project (any PR review), so they belong in the rulesets layer, not just the DeepSat project. Also captured in the work project's harness memory as feedback_no_praise_on_approvals_show_inline for immediate use. diff --git a/inbox/PROCESSED-2026-07-09-1341-from-work-bug-data-loss-wrap-org-table-el-and.org b/inbox/PROCESSED-2026-07-09-1341-from-work-bug-data-loss-wrap-org-table-el-and.org deleted file mode 100644 index a128d0a..0000000 --- a/inbox/PROCESSED-2026-07-09-1341-from-work-bug-data-loss-wrap-org-table-el-and.org +++ /dev/null @@ -1,28 +0,0 @@ -#+TITLE: BUG (data loss): wrap-org-table.el and lint-org.el corrupt o -#+SOURCE: from work -#+DATE: 2026-07-09 13:41:56 -0500 - -BUG (data loss): wrap-org-table.el and lint-org.el corrupt org example blocks. - -Both scripts scan for lines beginning with "|" and rewrite them as org tables. They do not skip #+begin_example / #+begin_src / #+begin_quote regions, so ASCII art using pipe characters gets mangled into tables. Both write to disk with no confirmation. - -Reproduced 2026-07-09 in the work project against an architecture doc containing an ASCII pipeline diagram that uses | and v as flow arrows: - - Before: After: - | | | - v |---| - v - -and a plain indented block became a bordered org table with |---| rules inserted between every line. - -Two separable defects: - -1. Table detection is line-based. Both helpers should use org-element-at-point (or org-in-block-p) to skip example/src/quote/verse blocks rather than matching /^\s*|/. - -2. lint-org.el mutates its input. Passing five files to it reformatted all five on disk -- one of them by 1949 lines. Its documented job is to report judgment items. A linter must not write. If the reformat is wanted, it belongs behind an explicit --fix flag. - -Impact: silent data loss on any org file that mixes tables and example blocks, which is most architecture docs. In this case the good content was already staged in git and was recoverable. It would not have been otherwise. - -No local fix attempted: .ai/scripts/ is rulesets-owned and the startup rsync would revert it. Filed as a task on the work side (Org-table helpers corrupt example blocks) so it is tracked there until the canonical fix lands. - -Suggested test: run wrap-org-table.el against a file containing a #+begin_example block whose lines start with "|" and assert the block is byte-identical afterward. diff --git a/inbox/PROCESSED-2026-07-09-1400-from-work-handoff-ai-attribution-cleanup-done.org b/inbox/PROCESSED-2026-07-09-1400-from-work-handoff-ai-attribution-cleanup-done.org deleted file mode 100644 index a37a54a..0000000 --- a/inbox/PROCESSED-2026-07-09-1400-from-work-handoff-ai-attribution-cleanup-done.org +++ /dev/null @@ -1,22 +0,0 @@ -#+TITLE: HANDOFF: AI-attribution cleanup done from the work session ( -#+SOURCE: from work -#+DATE: 2026-07-09 14:00:05 -0500 - -HANDOFF: AI-attribution cleanup done from the work session (Craig approved doing it from here). - -What changed in rulesets, uncommitted in your working tree: - -1. 103 files: rewrote the org header "#+AUTHOR: Craig Jennings & Claude" to "#+AUTHOR: Craig Jennings". Breakdown: 43 .ai/workflows, 43 claude-templates/.ai/workflows, 5 docs/specs, 5 docs/design, 2 .ai (notes, protocols), 2 claude-templates/.ai, 2 references, 1 working/. Exactly one line changed per file (103 insertions, 103 deletions, zero non-AUTHOR lines). Prose mentions of Claude were not touched. - -2. claude-rules/commits.md: added "Document author metadata" to the No-AI-Attribution list, plus a new subsection "Generated documents carry the human author only". It explains that the propagation mechanism is imitation (no template stamps the line; agents copy it from neighbouring files), names the employer-policy stakes, and carves out two exceptions. - -Held back deliberately, please confirm you agree: -- .ai/sessions/ (14 files) left as-is. They are historical records of what happened, not live artifacts. -- docs/design/2026-05-28-generic-agent-runtime-spec.org and its -review sibling keep "#+AUTHOR: Codex". Codex actually wrote them, so renaming would be a false attribution rather than removing one. -- .ai/scripts/tests/fixtures/todo-sample.org keeps "#+AUTHOR: synthetic fixture" (test data). - -Not committed. The tracked tree was clean at 0/0 against origin/main before these edits, so the diff is exactly this change and nothing else. Review and commit on your side. - -Why it came up: the work project noticed its generated daily-prep docs carry the co-author line. Craig pointed out that his own repo tolerates it, but employers whose policy is that work product carries employee names alone would not. Nothing has leaked yet: the arch docs pushed to the company GHE lost the #+AUTHOR line in the pandoc conversion. The exposure was prospective, via a planned Markdown-to-Notion publisher. - -Companion item already in your inbox: the wrap-org-table.el / lint-org.el data-loss bug (2026-07-09-1341). diff --git a/inbox/PROCESSED-2026-07-09-1636-from-chime-staleness-proposal.txt b/inbox/PROCESSED-2026-07-09-1636-from-chime-staleness-proposal.txt deleted file mode 100644 index cc741b2..0000000 --- a/inbox/PROCESSED-2026-07-09-1636-from-chime-staleness-proposal.txt +++ /dev/null @@ -1,17 +0,0 @@ -Proposal: task-review-staleness.sh should accept org-style :LAST_REVIEWED: values, or fail loudly. - -Hit this in chime today. I stamped :LAST_REVIEWED: [2026-07-09 Thu] — an org inactive timestamp, matching the CREATED: and CLOSED: cookies sitting right next to it in the same drawer. The script expects a bare 2026-07-09. - -The failure is silent and inverted. In count mode, `date -d "[2026-07-09 Thu]"` fails, and the unparseable branch counts the task as STALE. So a freshly-reviewed task reports as never-reviewed, and a full review pass leaves the startup nudge saying exactly what it said before. In list mode the sort-key regex also rejects it, so the task sorts as 0000-00-00 (oldest) and gets re-walked first. Both modes punish the stamp for being in the wrong format, and neither says so. - -I fixed my side (bare dates, matching the precedent in home/todo.org). The trap is worth closing, because the bracketed form is the plausible guess: every other date in an org PROPERTIES drawer is bracketed, and nothing in task-review.org or todo-format.md says LAST_REVIEWED is different. - -Two options, either fine: - -1. Accept both. Strip a leading bracket and a trailing weekday + bracket before parsing, inside extract_tasks. Org-native stamps then work and existing bare stamps keep working. - -2. Fail loudly. When a LAST_REVIEWED value is present but unparseable, print a warning naming the file, line, and value rather than folding it into the stale count. A malformed stamp is a data error, and treating it as "never reviewed" hides it forever. - -I lean toward both: accept the org form, warn on anything still unparseable. Today a project can run task reviews for months while the staleness count never drops, and nothing ever explains why. - -Also worth a line in todo-format.md or task-review.org stating the expected format. The script is currently the only place it's written down, and you have to read its awk to find it. diff --git a/inbox/PROCESSED-2026-07-09-1745-from-chime-matrix-proposal.txt b/inbox/PROCESSED-2026-07-09-1745-from-chime-matrix-proposal.txt deleted file mode 100644 index 53c20c8..0000000 --- a/inbox/PROCESSED-2026-07-09-1745-from-chime-matrix-proposal.txt +++ /dev/null @@ -1,35 +0,0 @@ -Proposal: todo-format.md's bug matrix should warn against double-counting rarity. - -Adding the severity x frequency matrix to chime today, I mis-graded a bug by exactly one mistake, and I think the rule invites it. - -The task: chime's async watchdog interrupts a child that outlives its timeout, but never escalates to kill. I graded it Minor severity ("a zombie child and a leaked process buffer accumulate slowly") x rare edge case ("the watchdog must fire AND the child must survive SIGINT") = P4 = [#D]. - -Then I read async.el. Its cleanup guards on (eq 'exit (process-status proc)) and only kills the process buffer in the zero-exit branch, so a signal-killed child (status 'signal) skips it entirely. Every watchdog interrupt leaks a buffer; the surviving-SIGINT zombie is the rare sub-case, not the leak. And the watchdog nils the process handle, so the same tick spawns a replacement — if the hang cause persists, another child is abandoned every timeout period. Roughly 30 leaked buffers an hour, indefinitely, invisibly. A real incident had a child stuck 15+ hours. - -Severity was the wrong input, not frequency. "Accumulates slowly" describes a bounded trickle. This accumulates at a fixed rate forever once entered, with no workaround short of restarting Emacs. That's Major. Major x rare edge = P3 = [#C], which is where it landed. - -The generalizable error: I let the rarity of *entering* the failure state discount the *severity* of being in it. But frequency already carries that rarity. Grading it twice buries exactly the bugs that compound — the ones where a rare trigger produces unbounded harm. - -Suggested addition to the matrix section: - - Don't double-count rarity. Grade severity by the rate of harm once the - failure state is entered, not by how rare it is to enter. Frequency - already carries the rarity; letting it discount severity too grades the - same fact twice, and that buries compounding bugs. A leak that repeats - every timeout period until the process restarts is Major even when - reaching that state is a rare edge case. - -Two smaller additions I made to chime's scheme, both of which I'd put in the global rule: - - Record the grading in the task body — the severity band, the frequency - row, and the arithmetic. A bare priority cookie can't be argued with; a - stated read can be re-checked against the source and corrected. That's - how this task moved [#D] -> [#C] an hour after I graded it. - - Disagreeing with a grade means fixing an input. If a letter looks wrong, - re-read the severity band and the frequency row against the source and - correct whichever is wrong. Don't override the letter directly — that - turns the matrix into a formality and puts you back to grading by - instinct. - -The last one is the one I care about. Craig's first instinct on seeing the [#D] was to restore the [#B] it had before. The matrix earns its keep only if a disagreement forces a re-read of the inputs rather than a manual override, and the rule text currently doesn't say so. diff --git a/inbox/PROCESSED-2026-07-11-0222-from-.emacs.d-ui-prototype-rule-proposal.org b/inbox/PROCESSED-2026-07-11-0222-from-.emacs.d-ui-prototype-rule-proposal.org deleted file mode 100644 index eeb60e7..0000000 --- a/inbox/PROCESSED-2026-07-11-0222-from-.emacs.d-ui-prototype-rule-proposal.org +++ /dev/null @@ -1,55 +0,0 @@ -#+TITLE: Proposal: UI features require a prototype phase before build-to-spec -#+AUTHOR: Craig Jennings -#+DATE: 2026-07-11 - -* The rule (Craig approved promotion 2026-07-11) - -When a spec'd feature has a UI, don't build straight from the spec. After the -initial spec, run several UI/UX prototype feedback loops — iteratively driving -the remaining functionality alongside the look and feel — then fold what settled -back into the spec. Only then "build to the prototype": the built feature should -look and behave as close to the prototype as possible, and any deviation is -documented in an addendum section of the prototype/spec itself. - -* Where it should live (rulesets session decides exact home) - -Two natural homes; probably both: - -1. The =brainstorm= skill's Phase 3. Today Phase 3 presents the design in chunks - and stops. Add: if the feature has a UI, the design isn't "accepted" until it - has been through a prototype UI/UX phase (several feedback loops), and the - spec records what the prototype settled. The spec's Next Steps then say - "build to the prototype," not "build to the spec." - -2. A spec-lifecycle rule (=docs-lifecycle.md=, or a small new workflow rule). - The lifecycle for a UI feature gains a prototype stage between DRAFT/READY and - build, and the spec carries a "Prototype & deviations" addendum section that - the build keeps current. - -Companion touch points to reconcile: =spec-create= (emit the prototype stage + -the deviations-addendum section for UI specs), =spec-response= (a UI spec -decomposes into a prototype loop first, then build-to-prototype tasks), -=start-work= (its verify phase already drives the UI end-to-end; here the bar is -"matches the prototype," with deviations logged). - -* Why — worked example (takuzu, 2026-07-11) - -Building the takuzu (Binairo) Emacs game. The spec chose "colored tiles, glyph -overlay optional." Built straight to that and the first launch in Craig's actual -terminal frame was all black — the dark background-color faces read as black and -the cursor was a GUI-only =:box=, so the whole grid was invisible. The -colored-tiles-are-readable assumption was false in the real environment. A -prototype loop caught it on the first screenshot; had we "built to spec" and -called it done, we'd have shipped an unusable board. The fix (glyphs as the -primary signal, inverse-video cursor) is exactly the kind of adjustment that only -surfaces by looking at the running UI, and it now needs to flow back into the -spec so the build target is the prototype, not the original spec text. - -Jotto (the other game spec'd the same day) also has a UI and will follow the same -path. - -* Requested action - -Fold the rule into the brainstorm skill and the spec-lifecycle rules so it -governs every project's UI features, then re-sync. This proposal is the durable -channel; the two game projects apply it locally in the meantime. diff --git a/inbox/lint-followups.org b/inbox/lint-followups.org new file mode 100644 index 0000000..9d2bd8f --- /dev/null +++ b/inbox/lint-followups.org @@ -0,0 +1,18 @@ +* 2026-07-20 Mon — Task-review health: 1 top-level [#A]/[#B]/[#C] tasks unreviewed for >30 days (daily review may have slipped) + +* lint-org follow-ups — todo.org (2026-07-29) +** TODO misplaced-heading — Possibly misplaced heading line (line 2263) +** TODO link-to-local-file — Link to non-existent local file "working/hook-fail-open/validate-el.diff" (line 2239) +** TODO misplaced-planning-info — Misplaced planning info line (line 2228) +** TODO link-to-local-file — Link to non-existent local file "working/hook-fail-open/pre-commit.diff" (line 2223) +** TODO misplaced-planning-info — Misplaced planning info line (line 2208) +** TODO org-table-standard — table violates the org-table standard: no closing rule; missing rule between rows — wrap-org-table.el reflows it (line 314) +** TODO task-missing-last-reviewed — task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed (line 472) +** TODO task-missing-last-reviewed — task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed (line 475) +** TODO task-missing-last-reviewed — task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed (line 482) +** TODO task-missing-last-reviewed — task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed (line 490) +** TODO task-missing-last-reviewed — task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed (line 493) +** TODO task-missing-last-reviewed — task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed (line 502) +** TODO task-missing-last-reviewed — task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed (line 600) +** TODO task-missing-last-reviewed — task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed (line 759) +** TODO task-missing-last-reviewed — task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed (line 768) diff --git a/languages/bash/githooks/pre-commit b/languages/bash/githooks/pre-commit index 880d5cf..1520690 100755 --- a/languages/bash/githooks/pre-commit +++ b/languages/bash/githooks/pre-commit @@ -18,8 +18,18 @@ cd "$REPO_ROOT" || exit 1 SECRET_PATTERNS_CS='(AKIA[0-9A-Z]{16}|sk-[a-zA-Z0-9_-]{20,}|-----BEGIN (RSA|DSA|EC|OPENSSH|PGP)( PRIVATE)?( KEY| KEY BLOCK)?-----)' SECRET_PATTERNS_CI='(api[_-]?key|api[_-]?secret|auth[_-]?token|secret[_-]?key|bearer[_-]?token|access[_-]?token|password)[[:space:]]*[:=][[:space:]]*["'"'"'][^"'"'"']{16,}["'"'"']' -added_lines="$(git diff --cached -U0 --diff-filter=AM \ - | grep '^+' | grep -v '^+++' || true)" +# Read the diff on its own so a git failure is distinguishable from "grep +# matched nothing". Both end in a non-zero status, but only one of them means +# there is nothing to scan; piping them together and swallowing the result with +# `|| true` made a broken git look like a clean commit — the scan searched an +# empty string, found nothing, and the secret went in. +if ! staged_diff="$(git diff --cached -U0 --diff-filter=AM)"; then + echo "pre-commit: cannot read the staged diff — refusing to skip the secret scan" >&2 + exit 1 +fi + +# The greps keep their `|| true`: exiting 1 on no match is their normal result. +added_lines="$(printf '%s\n' "$staged_diff" | grep '^+' | grep -v '^+++' || true)" cs_hits="$(printf '%s\n' "$added_lines" | grep -nE "$SECRET_PATTERNS_CS" || true)" ci_hits="$(printf '%s\n' "$added_lines" | grep -niE "$SECRET_PATTERNS_CI" || true)" @@ -37,8 +47,14 @@ if [ -n "$secret_hits" ]; then fi # --- 2. shellcheck on staged .sh / .bash files --- -staged_sh="$(git diff --cached --name-only --diff-filter=AM \ - | grep -E '\.(sh|bash)$' || true)" +# Same split as the secret scan above: a git failure must not read as "no files +# staged", which would skip the language check silently. +if ! staged_names="$(git diff --cached --name-only --diff-filter=AM)"; then + echo "pre-commit: cannot read the staged file list — refusing to skip the check" >&2 + exit 1 +fi + +staged_sh="$(printf '%s\n' "$staged_names" | grep -E '\.(sh|bash)$' || true)" if [ -n "$staged_sh" ] && command -v shellcheck >/dev/null 2>&1; then failed="" diff --git a/languages/elisp/claude/hooks/validate-el.sh b/languages/elisp/claude/hooks/validate-el.sh index d028789..870eefe 100755 --- a/languages/elisp/claude/hooks/validate-el.sh +++ b/languages/elisp/claude/hooks/validate-el.sh @@ -39,8 +39,6 @@ f="$(jq -r '.tool_input.file_path // .tool_response.filePath // empty')" [ -z "$f" ] && exit 0 [ "${f##*.}" = "el" ] || exit 0 -MAX_AUTO_TEST_FILES=20 # skip if more matches than this (large test suites) - # --- Phase 1: syntax + byte-compile --- case "$f" in */init.el|*/early-init.el) @@ -96,7 +94,7 @@ case "$f" in esac count="${#tests[@]}" -if [ "$count" -ge 1 ] && [ "$count" -le "$MAX_AUTO_TEST_FILES" ]; then +if [ "$count" -ge 1 ]; then load_args=() for t in "${tests[@]}"; do load_args+=("-l" "$t"); done if ! output="$(emacs --batch --no-site-file --no-site-lisp \ diff --git a/languages/elisp/githooks/pre-commit b/languages/elisp/githooks/pre-commit index 27f280c..a87bedf 100755 --- a/languages/elisp/githooks/pre-commit +++ b/languages/elisp/githooks/pre-commit @@ -5,7 +5,7 @@ set -u REPO_ROOT="$(git rev-parse --show-toplevel)" -cd "$REPO_ROOT" +cd "$REPO_ROOT" || exit 1 # --- 1. Secret scan --- # Patterns for common credentials. Scans only added lines in the staged diff. @@ -18,8 +18,18 @@ cd "$REPO_ROOT" SECRET_PATTERNS_CS='(AKIA[0-9A-Z]{16}|sk-[a-zA-Z0-9_-]{20,}|-----BEGIN (RSA|DSA|EC|OPENSSH|PGP)( PRIVATE)?( KEY| KEY BLOCK)?-----)' SECRET_PATTERNS_CI='(api[_-]?key|api[_-]?secret|auth[_-]?token|secret[_-]?key|bearer[_-]?token|access[_-]?token|password)[[:space:]]*[:=][[:space:]]*["'"'"'][^"'"'"']{16,}["'"'"']' -added_lines="$(git diff --cached -U0 --diff-filter=AM \ - | grep '^+' | grep -v '^+++' || true)" +# Read the diff on its own so a git failure is distinguishable from "grep +# matched nothing". Both end in a non-zero status, but only one of them means +# there is nothing to scan; piping them together and swallowing the result with +# `|| true` made a broken git look like a clean commit — the scan searched an +# empty string, found nothing, and the secret went in. +if ! staged_diff="$(git diff --cached -U0 --diff-filter=AM)"; then + echo "pre-commit: cannot read the staged diff — refusing to skip the secret scan" >&2 + exit 1 +fi + +# The greps keep their `|| true`: exiting 1 on no match is their normal result. +added_lines="$(printf '%s\n' "$staged_diff" | grep '^+' | grep -v '^+++' || true)" cs_hits="$(printf '%s\n' "$added_lines" | grep -nE "$SECRET_PATTERNS_CS" || true)" ci_hits="$(printf '%s\n' "$added_lines" | grep -niE "$SECRET_PATTERNS_CI" || true)" @@ -37,7 +47,14 @@ if [ -n "$secret_hits" ]; then fi # --- 2. Paren check on staged .el files --- -staged_el="$(git diff --cached --name-only --diff-filter=AM | grep '\.el$' || true)" +# Same split as the secret scan above: a git failure must not read as "no files +# staged", which would skip the language check silently. +if ! staged_names="$(git diff --cached --name-only --diff-filter=AM)"; then + echo "pre-commit: cannot read the staged file list — refusing to skip the check" >&2 + exit 1 +fi + +staged_el="$(printf '%s\n' "$staged_names" | grep '\.el$' || true)" if [ -n "$staged_el" ]; then paren_fail="" diff --git a/languages/elisp/tests/test-pre-commit-hook.bats b/languages/elisp/tests/test-pre-commit-hook.bats new file mode 100644 index 0000000..413c71d --- /dev/null +++ b/languages/elisp/tests/test-pre-commit-hook.bats @@ -0,0 +1,126 @@ +#!/usr/bin/env bats +# Tests for githooks/pre-commit — the secret scan and paren check. +# +# The scan reads its input through a pipeline: +# +# added_lines="$(git diff --cached ... | grep '^+' | grep -v '^+++' || true)" +# +# `grep` exits 1 when it matches nothing, which is the ordinary case, so the +# `|| true` has to stay. But with no `pipefail` it also swallows a failure of +# `git diff` itself, and an empty `added_lines` makes the scan search nothing, +# find nothing, and report clean. A gate that passes without looking is the +# failure this file exists to pin: the fail-open test drives a broken `git diff` +# and asserts the hook refuses rather than exiting 0. +# +# Each test builds a throwaway git repo in BATS_TEST_TMPDIR, so nothing touches +# the real repository or its hooks. + +setup() { + HOOK="${BATS_TEST_DIRNAME}/../githooks/pre-commit" + REPO="${BATS_TEST_TMPDIR}/repo" + mkdir -p "$REPO" + cd "$REPO" || return 1 + git init -q . + git config user.email t@example.com + git config user.name Test + # Split so the fixtures never appear as credential-shaped literals here. + AWS_TAIL="IOSFODNN7EXAMPLE" + WORD_TAIL="word" +} + +# Put a stub `git` ahead of the real one that fails for the staged-diff call +# and delegates everything else, so only the pipeline under test breaks. +break_staged_diff() { + mkdir -p "${BATS_TEST_TMPDIR}/bin" + cat > "${BATS_TEST_TMPDIR}/bin/git" <<'STUB' +#!/usr/bin/env bash +if [ "${1:-}" = "diff" ] && [ "${2:-}" = "--cached" ] && [ "${3:-}" = "-U0" ]; then + echo "simulated git failure" >&2 + exit 128 +fi +exec /usr/bin/git "$@" +STUB + chmod +x "${BATS_TEST_TMPDIR}/bin/git" + PATH="${BATS_TEST_TMPDIR}/bin:$PATH" +} + +# ------------------------------- Normal cases ------------------------------- + +@test "secret scan: blocks a staged AWS key" { + # Assembled at runtime: a literal key-shaped string in this file would trip + # the very hook under test on every commit that touches it, and this repo + # mirrors to a public remote. + printf 'aws = "%s"\n' "AKIA${AWS_TAIL}" > creds.txt + git add creds.txt + run "$HOOK" + [ "$status" -eq 1 ] + [[ "$output" == *"potential secret"* ]] +} + +@test "secret scan: blocks a staged keyword=value password" { + printf '%s = "%s"\n' "pass${WORD_TAIL}" "correcthorsebatterystaple" > conf.txt + git add conf.txt + run "$HOOK" + [ "$status" -eq 1 ] + [[ "$output" == *"potential secret"* ]] +} + +@test "secret scan: allows an ordinary staged file" { + printf 'just some prose\n' > notes.txt + git add notes.txt + run "$HOOK" + [ "$status" -eq 0 ] +} + +# ------------------------------ Boundary cases ------------------------------ + +@test "secret scan: allows a commit with nothing staged" { + run "$HOOK" + [ "$status" -eq 0 ] +} + +@test "paren check: blocks an unbalanced staged .el file" { + printf '(defun broken ()\n (message "no close"\n' > bad.el + git add bad.el + run "$HOOK" + [ "$status" -eq 1 ] + [[ "$output" == *"paren check failed"* ]] +} + +@test "paren check: allows a balanced staged .el file" { + printf '(defun fine ()\n (message "ok"))\n' > good.el + git add good.el + run "$HOOK" + [ "$status" -eq 0 ] +} + +# -------------------------------- Error cases ------------------------------- + +@test "secret scan: refuses to pass when the staged diff cannot be read" { + # The scan must not report clean after searching nothing. Without a + # pipefail-aware guard the broken diff yields an empty added_lines and the + # hook exits 0, letting a real secret through unscanned. + printf 'aws = "%s"\n' "AKIA${AWS_TAIL}" > creds.txt + git add creds.txt + break_staged_diff + run "$HOOK" + [ "$status" -ne 0 ] +} + +@test "paren check: refuses to pass when the staged file list cannot be read" { + printf '(defun broken ()\n (message "no close"\n' > bad.el + git add bad.el + mkdir -p "${BATS_TEST_TMPDIR}/bin2" + cat > "${BATS_TEST_TMPDIR}/bin2/git" <<'STUB' +#!/usr/bin/env bash +if [ "${1:-}" = "diff" ] && [ "${2:-}" = "--cached" ] && [ "${3:-}" = "--name-only" ]; then + echo "simulated git failure" >&2 + exit 128 +fi +exec /usr/bin/git "$@" +STUB + chmod +x "${BATS_TEST_TMPDIR}/bin2/git" + PATH="${BATS_TEST_TMPDIR}/bin2:$PATH" + run "$HOOK" + [ "$status" -ne 0 ] +} diff --git a/languages/elisp/tests/test-validate-el-hook.bats b/languages/elisp/tests/test-validate-el-hook.bats new file mode 100644 index 0000000..d4d6f23 --- /dev/null +++ b/languages/elisp/tests/test-validate-el-hook.bats @@ -0,0 +1,100 @@ +#!/usr/bin/env bats +# Tests for .claude/hooks/validate-el.sh — the auto-test runner. +# +# The runner used to skip entirely above MAX_AUTO_TEST_FILES=20, with no else +# branch: nothing printed, exit 0, indistinguishable from a passing run. That +# was live for the three largest families here (calendar-sync 63 test files, +# music 45, ai-term 35), so every edit to those ran parens and byte-compile and +# zero tests, silently. +# +# The cap was removed rather than made loud, because its premise did not hold. +# Measured on this machine, running a whole family takes about a second: +# ai-term 208 tests in 1.0s, music 403 in 1.7s, calendar-sync 633 in 0.9s. It +# was also concealing a real cross-test pollution bug in calendar-sync that +# only appears when that family runs in one process. +# +# These tests pin that no file count is skipped. Each builds a synthetic +# project in BATS_TEST_TMPDIR and points CLAUDE_PROJECT_DIR at it, so nothing +# runs against the real tree. + +setup() { + # Bundle layout: the hook ships at claude/hooks/ here and installs to + # .claude/hooks/ in a consuming project. This test was written against the + # installed layout, so re-homing it needed the path adjusted. + HOOK="${BATS_TEST_DIRNAME}/../claude/hooks/validate-el.sh" + PROJ="${BATS_TEST_TMPDIR}/proj" + mkdir -p "$PROJ/modules" "$PROJ/tests" + export CLAUDE_PROJECT_DIR="$PROJ" + printf '(provide (quote widget))\n' > "$PROJ/modules/widget.el" +} + +# N green test files matching the widget stem. +make_tests() { + local n="$1" i + for ((i = 1; i <= n; i++)); do + printf '(require (quote ert))\n(ert-deftest test-widget-%d () (should t))\n' \ + "$i" > "$PROJ/tests/test-widget-${i}.el" + done +} + +# One failing test file, to prove the run is real rather than merely quiet. +make_failing_test() { + printf '(require (quote ert))\n(ert-deftest test-widget-bad () (should nil))\n' \ + > "$PROJ/tests/test-widget-bad.el" +} + +hook_input() { + printf '{"tool_input":{"file_path":"%s"}}' "$PROJ/modules/widget.el" +} + +run_hook() { + run bash -c "$(printf '%q' "$HOOK") <<< '$(hook_input)'" +} + +# ------------------------------- Normal cases ------------------------------- + +@test "a small family runs and passes quietly" { + make_tests 3 + run_hook + [ "$status" -eq 0 ] +} + +@test "a failing test blocks, so a quiet pass means the tests really ran" { + make_tests 3 + make_failing_test + run_hook + [ "$status" -eq 2 ] + [[ "$output" == *"TESTS FAILED"* ]] +} + +# ------------------------------ Boundary cases ------------------------------ + +@test "at the old cap of 20 files: runs" { + make_tests 20 + run_hook + [ "$status" -eq 0 ] +} + +@test "past the old cap: still runs, no longer skipped" { + make_tests 21 + run_hook + [ "$status" -eq 0 ] + [[ "${output,,}" != *"skipped"* ]] +} + +@test "well past the old cap: a failure in file 63 is still caught" { + # The regression this guards: at 63 files the runner used to skip, so a red + # test in a big family reported clean. calendar-sync is exactly this size. + make_tests 63 + make_failing_test + run_hook + [ "$status" -eq 2 ] + [[ "$output" == *"TESTS FAILED"* ]] +} + +# -------------------------------- Error cases ------------------------------- + +@test "no matching tests: exits clean without running anything" { + run_hook + [ "$status" -eq 0 ] +} diff --git a/languages/go/githooks/pre-commit b/languages/go/githooks/pre-commit index a6297c8..7d93949 100755 --- a/languages/go/githooks/pre-commit +++ b/languages/go/githooks/pre-commit @@ -5,7 +5,7 @@ set -u REPO_ROOT="$(git rev-parse --show-toplevel)" -cd "$REPO_ROOT" +cd "$REPO_ROOT" || exit 1 # --- 1. Secret scan --- # Patterns for common credentials. Scans only added lines in the staged diff. @@ -18,8 +18,18 @@ cd "$REPO_ROOT" SECRET_PATTERNS_CS='(AKIA[0-9A-Z]{16}|sk-[a-zA-Z0-9_-]{20,}|-----BEGIN (RSA|DSA|EC|OPENSSH|PGP)( PRIVATE)?( KEY| KEY BLOCK)?-----)' SECRET_PATTERNS_CI='(api[_-]?key|api[_-]?secret|auth[_-]?token|secret[_-]?key|bearer[_-]?token|access[_-]?token|password)[[:space:]]*[:=][[:space:]]*["'"'"'][^"'"'"']{16,}["'"'"']' -added_lines="$(git diff --cached -U0 --diff-filter=AM \ - | grep '^+' | grep -v '^+++' || true)" +# Read the diff on its own so a git failure is distinguishable from "grep +# matched nothing". Both end in a non-zero status, but only one of them means +# there is nothing to scan; piping them together and swallowing the result with +# `|| true` made a broken git look like a clean commit — the scan searched an +# empty string, found nothing, and the secret went in. +if ! staged_diff="$(git diff --cached -U0 --diff-filter=AM)"; then + echo "pre-commit: cannot read the staged diff — refusing to skip the secret scan" >&2 + exit 1 +fi + +# The greps keep their `|| true`: exiting 1 on no match is their normal result. +added_lines="$(printf '%s\n' "$staged_diff" | grep '^+' | grep -v '^+++' || true)" cs_hits="$(printf '%s\n' "$added_lines" | grep -nE "$SECRET_PATTERNS_CS" || true)" ci_hits="$(printf '%s\n' "$added_lines" | grep -niE "$SECRET_PATTERNS_CI" || true)" @@ -39,8 +49,14 @@ fi # --- 2. gofmt check on staged .go files --- # gofmt -l lists files that aren't gofmt-clean. Skip generated and vendored # files the same way the rest of the toolchain does. -staged_go="$(git diff --cached --name-only --diff-filter=AM \ - | grep '\.go$' \ +# Same split as the secret scan above: a git failure must not read as "no files +# staged", which would skip the language check silently. +if ! staged_names="$(git diff --cached --name-only --diff-filter=AM)"; then + echo "pre-commit: cannot read the staged file list — refusing to skip the check" >&2 + exit 1 +fi + +staged_go="$(printf '%s\n' "$staged_names" | grep '\.go$' \ | grep -vE '(^|/)vendor/' || true)" if [ -n "$staged_go" ] && command -v gofmt >/dev/null 2>&1; then diff --git a/languages/python/CLAUDE.md b/languages/python/CLAUDE.md new file mode 100644 index 0000000..a2d0a82 --- /dev/null +++ b/languages/python/CLAUDE.md @@ -0,0 +1,80 @@ +# CLAUDE.md + +## Project + +Python project. Customize this section with your own description, layout, +and conventions. + +**Typical layout:** +- `src/<package>/` or a top-level package directory — importable code +- `tests/` — pytest tests mirroring the package layout +- `pyproject.toml` — dependencies, tool config (ruff, pytest, coverage) + +## Build & Test Commands + +If the project has a Makefile, document targets here. Common pattern: + +```bash +make test # run the pytest suite +make test FILE=tests/x.py # one file +make coverage # suite + coverage report +make lint # ruff across the tree +make typecheck # mypy (if the project adopts it) +make fmt # ruff format / black +``` + +Direct equivalents: `python3 -m pytest`, `pytest tests/test_x.py::test_name`, +`ruff check .`, `ruff format --diff .`, `mypy src/`. + +## Language Rules + +See rule files in `.claude/rules/`: +- `python-testing.md` — pytest conventions and fixture discipline +- `verification.md` — verify-before-claim-done discipline + +## Git Workflow + +Commit conventions: see `.claude/rules/commits.md` (author identity, +no AI attribution, message format). + +Pre-commit hook in `githooks/` scans for secrets, syntax-checks staged Python, +and runs `ruff` when it's installed. Activate on a fresh clone with +`git config core.hooksPath githooks`. + +## Problem-Solving Approach + +Investigate before fixing. When diagnosing a bug: +1. Read the relevant module and trace what actually happens +2. Identify the root cause, not a surface symptom +3. Write a failing test that captures the correct behavior +4. Fix, then re-run tests + +## Testing Discipline + +TDD is the default: write a failing test before any implementation. If you can't +write the test, you don't yet understand the change. Details in +`.claude/rules/python-testing.md`. + +## Editing Discipline + +A PostToolUse hook syntax-checks every Python file after Edit/Write/MultiEdit +and blocks on a parse error, then runs `ruff` when it's installed. The hook +covers `.py`, `.pyi`, and extensionless files with a python shebang. + +Type checking is not enforced by the hook — it needs the whole package and its +dependencies resolved, which is a build-scale operation rather than a +per-keystroke one. Run it via `make typecheck`. + +Formatting is likewise not enforced: a project picks its own line length and +quote style, so blocking on an unconfigured default would impose a contested +choice. Adopt one per project in `pyproject.toml`. + +## What Not to Do + +- Don't add features beyond what was asked +- Don't refactor surrounding code when fixing a bug +- Don't use a bare `except:` or swallow an exception without handling it +- Don't use a mutable default argument (`def f(xs=[])`) +- Don't add comments to code you didn't change +- Don't commit `.env` files, credentials, or API keys — the pre-commit hook + catches common patterns but isn't a substitute for care diff --git a/languages/python/claude/hooks/validate-python.sh b/languages/python/claude/hooks/validate-python.sh new file mode 100755 index 0000000..e43ad77 --- /dev/null +++ b/languages/python/claude/hooks/validate-python.sh @@ -0,0 +1,95 @@ +#!/usr/bin/env bash +# Validate Python files after Edit/Write/MultiEdit. +# PostToolUse hook: receives tool-call JSON on stdin. +# +# On success: exit 0 silent. +# On failure: emit JSON with hookSpecificOutput.additionalContext so Claude +# sees a structured error in its context, THEN exit 2 to block the tool +# pipeline. stderr still echoes the error for terminal visibility. +# +# Phase 1: syntax — python3 compiles the file. Always available wherever this +# hook can meaningfully run, so it's the floor rather than an optional +# gate: a file that doesn't parse is never worth passing on. +# Phase 2: ruff — lint, when installed. Catches undefined names, unused +# imports, and the rest of the pyflakes set. Absent ruff doesn't block +# the edit, matching how the bash bundle treats shellcheck. +# +# Formatters (black, ruff format) are deliberately NOT enforced here. A project +# picks its own line length and quote style, so blocking on an unconfigured +# default would impose a contested choice. python.md recommends a formatter; +# this hook enforces correctness. +# +# Type checking (mypy, pyright) is also out: it needs the whole package and its +# dependencies resolved, which is a build-scale operation, not a per-keystroke +# one. Run it via `make lint` / `make typecheck`. +# +# Scope: .py and .pyi files, plus extensionless files whose first line is a +# python shebang (the CLI tools that fill a script-heavy repo carry no +# extension). + +set -u + +# Emit a JSON failure payload and exit 2. Arguments: +# $1 — short failure type (e.g. "PYTHON SYNTAX ERROR") +# $2 — file path +# $3 — tool output (error body) +fail_json() { + local ctx + ctx="$(printf '%s: %s\n\n%s\n\nFix before proceeding.' "$1" "$2" "$3" \ + | jq -Rs .)" + cat <<EOF +{"hookSpecificOutput": {"hookEventName": "PostToolUse", "additionalContext": $ctx}} +EOF + printf '%s: %s\n%s\n' "$1" "$2" "$3" >&2 + exit 2 +} + +f="$(jq -r '.tool_input.file_path // .tool_response.filePath // empty')" +[ -z "$f" ] && exit 0 +[ -f "$f" ] || exit 0 + +# Is this a Python file? By extension, or by shebang when it has no extension. +# Match on the basename, not the full path — a temp/parent dir can carry a dot +# (e.g. my.project/) and misfire the "*.*" extension test. +is_python=0 +base="${f##*/}" +case "$base" in + *.py | *.pyi) is_python=1 ;; + *.*) is_python=0 ;; # some other extension — not ours + *) + # No extension: sniff the shebang. + if head -1 "$f" 2>/dev/null | grep -qE '^#!.*\bpython[0-9.]*\b'; then + is_python=1 + fi + ;; +esac +[ "$is_python" -eq 1 ] || exit 0 + +# No python3 on this machine — nothing to validate, don't block the edit. +command -v python3 >/dev/null 2>&1 || exit 0 + +# --- Phase 1: syntax --- +# compile() rather than py_compile so no __pycache__ lands beside the source; +# the hook is a checker and must not leave build artifacts in the tree. +if ! out="$(python3 -c ' +import sys +p = sys.argv[1] +with open(p, "rb") as fh: + src = fh.read() +try: + compile(src, p, "exec") +except SyntaxError as e: + print(f"{e.msg} ({p}, line {e.lineno})", file=sys.stderr) + sys.exit(1) +' "$f" 2>&1)"; then + fail_json "PYTHON SYNTAX ERROR" "$f" "$out" +fi + +# --- Phase 2: lint (optional) --- +command -v ruff >/dev/null 2>&1 || exit 0 + +if ! out="$(ruff check "$f" 2>&1)"; then + fail_json "RUFF FAILED" "$f" "$out" +fi + +exit 0 diff --git a/languages/python/claude/settings.json b/languages/python/claude/settings.json new file mode 100644 index 0000000..9c6b2a9 --- /dev/null +++ b/languages/python/claude/settings.json @@ -0,0 +1,79 @@ +{ + "attribution": { + "commit": "", + "pr": "" + }, + "permissions": { + "allow": [ + "Bash(make)", + "Bash(make help)", + "Bash(make targets)", + "Bash(make test)", + "Bash(make test *)", + "Bash(make lint)", + "Bash(make fmt)", + "Bash(make coverage)", + "Bash(make coverage-summary)", + "Bash(make typecheck)", + "Bash(pytest)", + "Bash(pytest *)", + "Bash(python3 -m pytest *)", + "Bash(ruff check *)", + "Bash(ruff format --diff *)", + "Bash(black --check *)", + "Bash(black --diff *)", + "Bash(mypy *)", + "Bash(python3 -m py_compile *)", + "Bash(python3 --version)", + "Bash(pip list)", + "Bash(pip show *)", + "Bash(git status)", + "Bash(git status *)", + "Bash(git diff)", + "Bash(git diff *)", + "Bash(git log)", + "Bash(git log *)", + "Bash(git show)", + "Bash(git show *)", + "Bash(git blame *)", + "Bash(git branch)", + "Bash(git branch -v)", + "Bash(git branch -a)", + "Bash(git branch --list *)", + "Bash(git remote)", + "Bash(git remote -v)", + "Bash(git remote show *)", + "Bash(git ls-files *)", + "Bash(git rev-parse *)", + "Bash(git cat-file *)", + "Bash(git stash list)", + "Bash(git stash show *)", + "Bash(jq *)", + "Bash(date)", + "Bash(date *)", + "Bash(which *)", + "Bash(file *)", + "Bash(ls)", + "Bash(ls *)", + "Bash(wc *)", + "Bash(du *)", + "Bash(readlink *)", + "Bash(realpath *)", + "Bash(basename *)", + "Bash(dirname *)" + ] + }, + "hooks": { + "PostToolUse": [ + { + "matcher": "Edit|Write|MultiEdit", + "hooks": [ + { + "type": "command", + "command": "$CLAUDE_PROJECT_DIR/.claude/hooks/validate-python.sh" + } + ] + } + ] + } +} diff --git a/languages/python/githooks/pre-commit b/languages/python/githooks/pre-commit new file mode 100755 index 0000000..03536db --- /dev/null +++ b/languages/python/githooks/pre-commit @@ -0,0 +1,95 @@ +#!/usr/bin/env bash +# Pre-commit hook: secret scan + syntax/lint check on staged Python files. +# Use `git commit --no-verify` to bypass for confirmed false positives. + +set -u + +REPO_ROOT="$(git rev-parse --show-toplevel)" +cd "$REPO_ROOT" || exit 1 + +# --- 1. Secret scan --- +# Patterns for common credentials. Scans only added lines in the staged diff. +# +# Two passes because case-sensitivity differs. AWS keys are uppercase, sk- keys +# lowercase, PEM headers fixed, so those match case-SENSITIVELY: under -i, +# AKIA[0-9A-Z]{16} matches any mixed-case 20-char run, which random base64 in an +# embedded image blob hits ~6% of the time per 100KB and blocks real commits. +# Only the keyword=value patterns need -i. +SECRET_PATTERNS_CS='(AKIA[0-9A-Z]{16}|sk-[a-zA-Z0-9_-]{20,}|-----BEGIN (RSA|DSA|EC|OPENSSH|PGP)( PRIVATE)?( KEY| KEY BLOCK)?-----)' +SECRET_PATTERNS_CI='(api[_-]?key|api[_-]?secret|auth[_-]?token|secret[_-]?key|bearer[_-]?token|access[_-]?token|password)[[:space:]]*[:=][[:space:]]*["'"'"'][^"'"'"']{16,}["'"'"']' + +# Read the diff on its own so a git failure is distinguishable from "grep +# matched nothing". Both end in a non-zero status, but only one of them means +# there is nothing to scan; piping them together and swallowing the result with +# `|| true` made a broken git look like a clean commit — the scan searched an +# empty string, found nothing, and the secret went in. +if ! staged_diff="$(git diff --cached -U0 --diff-filter=AM)"; then + echo "pre-commit: cannot read the staged diff — refusing to skip the secret scan" >&2 + exit 1 +fi + +# The greps keep their `|| true`: exiting 1 on no match is their normal result. +added_lines="$(printf '%s\n' "$staged_diff" | grep '^+' | grep -v '^+++' || true)" + +cs_hits="$(printf '%s\n' "$added_lines" | grep -nE "$SECRET_PATTERNS_CS" || true)" +ci_hits="$(printf '%s\n' "$added_lines" | grep -niE "$SECRET_PATTERNS_CI" || true)" +# awk dedupes lines both passes matched, keeping first-seen order. +secret_hits="$(printf '%s\n%s' "$cs_hits" "$ci_hits" \ + | grep -v '^[[:space:]]*$' | awk '!seen[$0]++' || true)" + +if [ -n "$secret_hits" ]; then + echo "pre-commit: potential secret in staged changes:" >&2 + echo "$secret_hits" >&2 + echo "" >&2 + echo "Review the lines above. If this is a false positive (test fixture, documentation)," >&2 + echo "bypass with: git commit --no-verify" >&2 + exit 1 +fi + +# --- 2. Syntax check on staged Python files --- +# Same split as the secret scan above: a git failure must not read as "no files +# staged", which would skip the language check silently. +if ! staged_names="$(git diff --cached --name-only --diff-filter=AM)"; then + echo "pre-commit: cannot read the staged file list — refusing to skip the check" >&2 + exit 1 +fi + +staged_py="$(printf '%s\n' "$staged_names" | grep -E '\.pyi?$' || true)" + +if [ -n "$staged_py" ] && command -v python3 >/dev/null 2>&1; then + failed="" + while IFS= read -r f; do + [ -z "$f" ] && continue + [ -f "$f" ] || continue + # compile() rather than py_compile so no __pycache__ lands in the tree. + if ! python3 -c 'import sys; compile(open(sys.argv[1], "rb").read(), sys.argv[1], "exec")' "$f" >/dev/null 2>&1; then + failed="${failed}${f}"$'\n' + fi + done <<< "$staged_py" + + if [ -n "$failed" ]; then + printf 'pre-commit: Python syntax errors in staged files:\n\n%s\n' "$failed" >&2 + echo "Run: python3 -m py_compile <file> to see the error, then re-stage." >&2 + exit 1 + fi +fi + +# --- 3. ruff on staged Python files (when installed) --- +if [ -n "$staged_py" ] && command -v ruff >/dev/null 2>&1; then + failed="" + while IFS= read -r f; do + [ -z "$f" ] && continue + [ -f "$f" ] || continue + if ! ruff check "$f" >/dev/null 2>&1; then + failed="${failed}${f}"$'\n' + fi + done <<< "$staged_py" + + if [ -n "$failed" ]; then + printf 'pre-commit: ruff failed on staged files:\n\n%s\n' "$failed" >&2 + echo "Run: ruff check <file> and fix the findings, then re-stage." >&2 + exit 1 + fi +fi + +exit 0 diff --git a/languages/python/tests/pre-commit.bats b/languages/python/tests/pre-commit.bats new file mode 100644 index 0000000..1ac82ee --- /dev/null +++ b/languages/python/tests/pre-commit.bats @@ -0,0 +1,138 @@ +#!/usr/bin/env bats +# +# Tests for languages/python/githooks/pre-commit — the secret scan plus +# syntax/lint gate that runs on staged Python files. +# +# The secret scan is the security-critical half and is language-independent, so +# it gets the same coverage here as in the bash bundle: a real key blocks, a +# clean diff passes, and the case-sensitivity split that keeps base64 blobs from +# false-positiving is exercised directly. +# +# Each test builds a throwaway git repo, stages content, and runs the hook from +# inside it — the hook reads `git diff --cached`, so a real index is required. + +HOOK="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/githooks/pre-commit" + +setup() { + TEST_DIR="$(mktemp -d -t pre-commit-py-bats.XXXXXX)" + cd "$TEST_DIR" || exit 1 + git init -q . + git config user.email t@example.com + git config user.name Test + # A base commit so `git diff --cached` has a parent to diff against. + echo "seed" > seed.txt + git add seed.txt + git commit -qm seed +} + +teardown() { + cd / || true + rm -rf "$TEST_DIR" +} + +# ---- Normal ---------------------------------------------------------- + +@test "pre-commit(py): a clean staged Python file passes (exit 0)" { + printf 'def f(x):\n return x + 1\n' > ok.py + git add ok.py + run bash "$HOOK" + [ "$status" -eq 0 ] +} + +@test "pre-commit(py): an empty staging area passes (exit 0)" { + run bash "$HOOK" + [ "$status" -eq 0 ] +} + +# ---- Error: the secret scan ------------------------------------------ + +@test "pre-commit(py): an AWS key in a staged file blocks (exit 1)" { + printf 'KEY = "AKIAIOSFODNN7EXAMPLE"\n' > conf.py + git add conf.py + run bash "$HOOK" + [ "$status" -eq 1 ] + [[ "$output" == *"potential secret"* ]] +} + +@test "pre-commit(py): an sk- style token blocks (exit 1)" { + printf 'TOKEN = "sk-abcdefghijklmnopqrstuvwxyz0123"\n' > conf.py + git add conf.py + run bash "$HOOK" + [ "$status" -eq 1 ] +} + +@test "pre-commit(py): a quoted api_key assignment blocks (exit 1)" { + printf 'api_key = "abcdefghijklmnopqrstuvwxyz"\n' > conf.py + git add conf.py + run bash "$HOOK" + [ "$status" -eq 1 ] +} + +@test "pre-commit(py): a private-key header blocks (exit 1)" { + printf 'PEM = """-----BEGIN RSA PRIVATE KEY-----"""\n' > conf.py + git add conf.py + run bash "$HOOK" + [ "$status" -eq 1 ] +} + +# ---- Boundary: the case-sensitivity split ---------------------------- + +@test "pre-commit(py): a mixed-case base64 blob does NOT false-positive" { + # The AWS pattern is uppercase-only by design. Under -i it would match any + # 20-char mixed-case run, which random base64 hits often enough to block + # real commits. This is the regression test for that split. + printf 'BLOB = "AKIAbcdefGHIJklmnOPqr0123456789abcdefGHIJ"\n' > data.py + git add data.py + run bash "$HOOK" + [ "$status" -eq 0 ] +} + +@test "pre-commit(py): a short quoted password value does NOT block" { + # The keyword patterns require 16+ chars, so a placeholder stays quiet. + printf 'password = "short"\n' > conf.py + git add conf.py + run bash "$HOOK" + [ "$status" -eq 0 ] +} + +@test "pre-commit(py): a secret only in a REMOVED line does not block" { + printf 'KEY = "AKIAIOSFODNN7EXAMPLE"\n' > conf.py + git add conf.py + git commit -qm "add key" + rm conf.py + git add -A + run bash "$HOOK" + [ "$status" -eq 0 ] +} + +# ---- Error: the syntax gate ------------------------------------------ + +@test "pre-commit(py): a staged Python syntax error blocks (exit 1)" { + printf 'def f(:\n return 1\n' > bad.py + git add bad.py + run bash "$HOOK" + [ "$status" -eq 1 ] + [[ "$output" == *"syntax"* ]] +} + +@test "pre-commit(py): a .pyi stub with a syntax error blocks (exit 1)" { + printf 'def f( -> int: ...\n' > bad.pyi + git add bad.pyi + run bash "$HOOK" + [ "$status" -eq 1 ] +} + +@test "pre-commit(py): a broken NON-Python file does not trip the syntax gate" { + printf 'this is (((not python\n' > notes.txt + git add notes.txt + run bash "$HOOK" + [ "$status" -eq 0 ] +} + +@test "pre-commit(py): the syntax gate leaves no __pycache__ in the repo" { + printf 'def f():\n return 1\n' > ok.py + git add ok.py + run bash "$HOOK" + [ "$status" -eq 0 ] + [ ! -d __pycache__ ] +} diff --git a/languages/python/tests/validate-python.bats b/languages/python/tests/validate-python.bats new file mode 100644 index 0000000..b5e4957 --- /dev/null +++ b/languages/python/tests/validate-python.bats @@ -0,0 +1,117 @@ +#!/usr/bin/env bats +# +# Tests for languages/python/claude/hooks/validate-python.sh — the PostToolUse +# hook that syntax-checks edited Python files and blocks on a violation. +# +# The hook reads tool-call JSON on stdin and extracts the file path, so each +# test pipes a JSON payload naming a real file it wrote into a temp dir. +# +# The syntax gate is python3's own compiler, which is present wherever the hook +# can meaningfully run, so those tests never skip. The lint gate (ruff) is +# optional and its tests skip when it's absent, matching the bash bundle's +# treatment of shellcheck. + +HOOK="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/claude/hooks/validate-python.sh" + +setup() { + TEST_DIR="$(mktemp -d -t validate-python-bats.XXXXXX)" +} + +teardown() { + rm -rf "$TEST_DIR" +} + +payload() { + printf '{"tool_input": {"file_path": "%s"}}' "$1" +} + +# ---- Normal ---------------------------------------------------------- + +@test "validate-python: a clean .py file passes silently (exit 0)" { + printf 'def f(x):\n return x + 1\n' > "$TEST_DIR/clean.py" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/clean.py")" + [ "$status" -eq 0 ] + [ -z "$output" ] +} + +@test "validate-python: a .pyi stub is validated too" { + printf 'def f(x: int) -> int: ...\n' > "$TEST_DIR/clean.pyi" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/clean.pyi")" + [ "$status" -eq 0 ] +} + +# ---- Error ----------------------------------------------------------- + +@test "validate-python: a syntax error blocks (exit 2, names the failure)" { + printf 'def f(:\n return 1\n' > "$TEST_DIR/bad.py" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/bad.py")" + [ "$status" -eq 2 ] + [[ "$output" == *"SYNTAX"* ]] +} + +@test "validate-python: the block payload is valid JSON carrying the context" { + printf 'def f(:\n' > "$TEST_DIR/bad.py" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/bad.py")" + [ "$status" -eq 2 ] + # The first line of stdout must parse as JSON and carry the hook event name. + echo "$output" | head -1 | jq -e '.hookSpecificOutput.hookEventName == "PostToolUse"' +} + +@test "validate-python: a ruff violation blocks when ruff is installed" { + command -v ruff >/dev/null 2>&1 || skip "ruff not installed" + # F821: reference to an undefined name — syntactically valid, lint-caught. + printf 'def f():\n return undefined_name\n' > "$TEST_DIR/lint.py" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/lint.py")" + [ "$status" -eq 2 ] + [[ "$output" == *"RUFF"* ]] +} + +# ---- Boundary -------------------------------------------------------- + +@test "validate-python: a non-Python file is ignored (exit 0)" { + printf 'not python at all (((\n' > "$TEST_DIR/notes.txt" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/notes.txt")" + [ "$status" -eq 0 ] +} + +@test "validate-python: an extensionless file with a python shebang is validated" { + printf '#!/usr/bin/env python3\ndef f(:\n' > "$TEST_DIR/cli-tool" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/cli-tool")" + [ "$status" -eq 2 ] +} + +@test "validate-python: an extensionless non-python file is ignored (exit 0)" { + printf '#!/usr/bin/env bash\necho hi\n' > "$TEST_DIR/shell-tool" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/shell-tool")" + [ "$status" -eq 0 ] +} + +@test "validate-python: a dotted parent directory does not misfire the extension test" { + mkdir -p "$TEST_DIR/my.project" + printf '#!/usr/bin/env python3\ndef f(:\n' > "$TEST_DIR/my.project/cli-tool" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/my.project/cli-tool")" + [ "$status" -eq 2 ] +} + +@test "validate-python: empty file_path is a no-op (exit 0)" { + run bash "$HOOK" <<< '{"tool_input": {}}' + [ "$status" -eq 0 ] +} + +@test "validate-python: a missing file is a no-op (exit 0)" { + run bash "$HOOK" <<< "$(payload "$TEST_DIR/does-not-exist.py")" + [ "$status" -eq 0 ] +} + +@test "validate-python: an empty .py file passes (valid, compiles to nothing)" { + : > "$TEST_DIR/empty.py" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/empty.py")" + [ "$status" -eq 0 ] +} + +@test "validate-python: compiling leaves no __pycache__ beside the file" { + printf 'def f():\n return 1\n' > "$TEST_DIR/clean.py" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/clean.py")" + [ "$status" -eq 0 ] + [ ! -d "$TEST_DIR/__pycache__" ] +} diff --git a/languages/typescript/CLAUDE.md b/languages/typescript/CLAUDE.md new file mode 100644 index 0000000..1794115 --- /dev/null +++ b/languages/typescript/CLAUDE.md @@ -0,0 +1,82 @@ +# CLAUDE.md + +## Project + +TypeScript/JavaScript project. Customize this section with your own +description, layout, and conventions. + +**Typical layout:** +- `src/` — source modules +- `tests/` or `*.test.ts` beside the source — test files +- `package.json` — scripts and dependencies +- `tsconfig.json` — compiler options + +## Build & Test Commands + +If the project has a Makefile, document targets here. Common pattern: + +```bash +make test # run the test suite +make coverage # suite + coverage report +make typecheck # tsc --noEmit across the project +make lint # eslint +make build # production build +``` + +Direct equivalents: `npm test`, `npx tsc --noEmit`, `npx eslint src/`, +`npx prettier --check .`, `node --test`. + +## Language Rules + +See rule files in `.claude/rules/`: +- `typescript-testing.md` — test conventions and mocking discipline +- `verification.md` — verify-before-claim-done discipline + +## Git Workflow + +Commit conventions: see `.claude/rules/commits.md` (author identity, +no AI attribution, message format). + +Pre-commit hook in `githooks/` scans for secrets and parse-checks staged TS/JS. +Activate on a fresh clone with `git config core.hooksPath githooks`. + +## Problem-Solving Approach + +Investigate before fixing. When diagnosing a bug: +1. Read the relevant module and trace what actually happens +2. Identify the root cause, not a surface symptom +3. Write a failing test that captures the correct behavior +4. Fix, then re-run tests + +## Testing Discipline + +TDD is the default: write a failing test before any implementation. If you can't +write the test, you don't yet understand the change. Details in +`.claude/rules/typescript-testing.md`. + +## Editing Discipline + +A PostToolUse hook parse-checks every TS/JS file after Edit/Write/MultiEdit and +blocks on a syntax error. It covers `.ts`, `.tsx`, `.mts`, `.cts`, `.js`, +`.jsx`, `.mjs`, and `.cjs`. + +Two checkers, because one tool can't do both jobs: `node --check` for +JavaScript, `tsc` filtered to syntax diagnostics for TypeScript. Do not +substitute `node --check` for the TypeScript path — it ignores +`--experimental-strip-types`, so it rejects valid TypeScript and accepts broken +TypeScript (measured on node v26.4.0). + +Full type checking is not enforced by the hook: it needs the whole project graph +and its dependencies resolved, which is a build-scale operation rather than a +per-keystroke one. Run it via `make typecheck`. Formatting is likewise not +enforced; adopt a style per project in the project's own config. + +## What Not to Do + +- Don't add features beyond what was asked +- Don't refactor surrounding code when fixing a bug +- Don't reach for `any` to silence a type error — narrow the type instead +- Don't use `==` where `===` is meant +- Don't add comments to code you didn't change +- Don't commit `.env` files, credentials, or API keys — the pre-commit hook + catches common patterns but isn't a substitute for care diff --git a/languages/typescript/claude/hooks/validate-typescript.sh b/languages/typescript/claude/hooks/validate-typescript.sh new file mode 100755 index 0000000..b76f1df --- /dev/null +++ b/languages/typescript/claude/hooks/validate-typescript.sh @@ -0,0 +1,94 @@ +#!/usr/bin/env bash +# Validate TypeScript/JavaScript files after Edit/Write/MultiEdit. +# PostToolUse hook: receives tool-call JSON on stdin. +# +# On success: exit 0 silent. +# On failure: emit JSON with hookSpecificOutput.additionalContext so Claude +# sees a structured error in its context, THEN exit 2 to block the tool +# pipeline. stderr still echoes the error for terminal visibility. +# +# Gate: parseability. A file that doesn't parse is never worth passing on. +# Full type checking is deliberately NOT enforced here — it needs the whole +# project graph and its dependencies resolved, which is a build-scale +# operation, not a per-keystroke one. A type error that parses cleanly passes +# this hook; `make typecheck` / `tsc --noEmit` over the project owns it. +# +# Formatting (prettier) is also out: a project picks its own style, so blocking +# on an unconfigured default would impose a contested choice. +# +# Two checkers, because one tool can't do both jobs: +# +# .js/.jsx/.mjs/.cjs → `node --check`, a straight parse. +# .ts/.tsx/.mts/.cts → `tsc`, filtered to syntax-category diagnostics. +# +# `node --check` must NOT be used on TypeScript. It ignores +# --experimental-strip-types, so it is wrong in *both* directions: it rejects +# valid TS (an `interface` declaration reads as a syntax error) and accepts +# broken TS (a genuinely unparseable file exits 0). Measured on node v26.4.0, +# 2026-07-23. tsc is the only correct parser for these extensions. +# +# The tsc call is filtered to TS1xxx codes, which is TypeScript's syntactic +# diagnostic range; TS2xxx and up are semantic (type) errors and are out of +# scope by the paragraph above. Without the filter this hook would block every +# unresolved import in a file whose dependencies aren't installed yet. + +set -u + +# Emit a JSON failure payload and exit 2. Arguments: +# $1 — short failure type (e.g. "TYPESCRIPT SYNTAX ERROR") +# $2 — file path +# $3 — tool output (error body) +fail_json() { + local ctx + ctx="$(printf '%s: %s\n\n%s\n\nFix before proceeding.' "$1" "$2" "$3" \ + | jq -Rs .)" + cat <<EOF +{"hookSpecificOutput": {"hookEventName": "PostToolUse", "additionalContext": $ctx}} +EOF + printf '%s: %s\n%s\n' "$1" "$2" "$3" >&2 + exit 2 +} + +f="$(jq -r '.tool_input.file_path // .tool_response.filePath // empty')" +[ -z "$f" ] && exit 0 +[ -f "$f" ] || exit 0 + +# Classify by extension. Match on the basename, not the full path — a parent +# dir can carry a dot (e.g. my.project/) and confuse a path-wide match. +kind="" +base="${f##*/}" +case "$base" in + *.ts | *.tsx | *.mts | *.cts) kind="ts" ;; + *.js | *.jsx | *.mjs | *.cjs) kind="js" ;; + *) exit 0 ;; +esac + +if [ "$kind" = "js" ]; then + command -v node >/dev/null 2>&1 || exit 0 + if ! out="$(node --check "$f" 2>&1)"; then + fail_json "JAVASCRIPT SYNTAX ERROR" "$f" "$out" + fi + exit 0 +fi + +# TypeScript. Prefer a project-local tsc so the project's own version decides, +# falling back to one on PATH. +tsc_bin="" +if [ -x "./node_modules/.bin/tsc" ]; then + tsc_bin="./node_modules/.bin/tsc" +elif command -v tsc >/dev/null 2>&1; then + tsc_bin="tsc" +else + exit 0 # no TypeScript compiler available — don't block the edit +fi + +# --moduleDetection force so a file with no import/export still parses as a +# module rather than tripping global-scope collisions against lib types. +out="$("$tsc_bin" --noEmit --skipLibCheck --target es2022 --moduleDetection force "$f" 2>&1 || true)" +syntax_errors="$(printf '%s\n' "$out" | grep -E 'error TS1[0-9]{3}:' || true)" + +if [ -n "$syntax_errors" ]; then + fail_json "TYPESCRIPT SYNTAX ERROR" "$f" "$syntax_errors" +fi + +exit 0 diff --git a/languages/typescript/claude/settings.json b/languages/typescript/claude/settings.json new file mode 100644 index 0000000..f4c9211 --- /dev/null +++ b/languages/typescript/claude/settings.json @@ -0,0 +1,80 @@ +{ + "attribution": { + "commit": "", + "pr": "" + }, + "permissions": { + "allow": [ + "Bash(make)", + "Bash(make help)", + "Bash(make targets)", + "Bash(make test)", + "Bash(make test *)", + "Bash(make lint)", + "Bash(make fmt)", + "Bash(make coverage)", + "Bash(make coverage-summary)", + "Bash(make typecheck)", + "Bash(make build)", + "Bash(npm test)", + "Bash(npm test *)", + "Bash(npm run *)", + "Bash(npm ci)", + "Bash(npm ls *)", + "Bash(node --check *)", + "Bash(node --test *)", + "Bash(node --version)", + "Bash(tsc --noEmit *)", + "Bash(npx tsc --noEmit *)", + "Bash(eslint *)", + "Bash(prettier --check *)", + "Bash(git status)", + "Bash(git status *)", + "Bash(git diff)", + "Bash(git diff *)", + "Bash(git log)", + "Bash(git log *)", + "Bash(git show)", + "Bash(git show *)", + "Bash(git blame *)", + "Bash(git branch)", + "Bash(git branch -v)", + "Bash(git branch -a)", + "Bash(git branch --list *)", + "Bash(git remote)", + "Bash(git remote -v)", + "Bash(git remote show *)", + "Bash(git ls-files *)", + "Bash(git rev-parse *)", + "Bash(git cat-file *)", + "Bash(git stash list)", + "Bash(git stash show *)", + "Bash(jq *)", + "Bash(date)", + "Bash(date *)", + "Bash(which *)", + "Bash(file *)", + "Bash(ls)", + "Bash(ls *)", + "Bash(wc *)", + "Bash(du *)", + "Bash(readlink *)", + "Bash(realpath *)", + "Bash(basename *)", + "Bash(dirname *)" + ] + }, + "hooks": { + "PostToolUse": [ + { + "matcher": "Edit|Write|MultiEdit", + "hooks": [ + { + "type": "command", + "command": "$CLAUDE_PROJECT_DIR/.claude/hooks/validate-typescript.sh" + } + ] + } + ] + } +} diff --git a/languages/typescript/githooks/pre-commit b/languages/typescript/githooks/pre-commit new file mode 100755 index 0000000..fd494d2 --- /dev/null +++ b/languages/typescript/githooks/pre-commit @@ -0,0 +1,107 @@ +#!/usr/bin/env bash +# Pre-commit hook: secret scan + syntax check on staged TypeScript/JavaScript files. +# Use `git commit --no-verify` to bypass for confirmed false positives. + +set -u + +REPO_ROOT="$(git rev-parse --show-toplevel)" +cd "$REPO_ROOT" || exit 1 + +# --- 1. Secret scan --- +# Patterns for common credentials. Scans only added lines in the staged diff. +# +# Two passes because case-sensitivity differs. AWS keys are uppercase, sk- keys +# lowercase, PEM headers fixed, so those match case-SENSITIVELY: under -i, +# AKIA[0-9A-Z]{16} matches any mixed-case 20-char run, which random base64 in an +# embedded image blob hits ~6% of the time per 100KB and blocks real commits. +# Only the keyword=value patterns need -i. +SECRET_PATTERNS_CS='(AKIA[0-9A-Z]{16}|sk-[a-zA-Z0-9_-]{20,}|-----BEGIN (RSA|DSA|EC|OPENSSH|PGP)( PRIVATE)?( KEY| KEY BLOCK)?-----)' +SECRET_PATTERNS_CI='(api[_-]?key|api[_-]?secret|auth[_-]?token|secret[_-]?key|bearer[_-]?token|access[_-]?token|password)[[:space:]]*[:=][[:space:]]*["'"'"'][^"'"'"']{16,}["'"'"']' + +# Read the diff on its own so a git failure is distinguishable from "grep +# matched nothing". Both end in a non-zero status, but only one of them means +# there is nothing to scan; piping them together and swallowing the result with +# `|| true` made a broken git look like a clean commit — the scan searched an +# empty string, found nothing, and the secret went in. +if ! staged_diff="$(git diff --cached -U0 --diff-filter=AM)"; then + echo "pre-commit: cannot read the staged diff — refusing to skip the secret scan" >&2 + exit 1 +fi + +# The greps keep their `|| true`: exiting 1 on no match is their normal result. +added_lines="$(printf '%s\n' "$staged_diff" | grep '^+' | grep -v '^+++' || true)" + +cs_hits="$(printf '%s\n' "$added_lines" | grep -nE "$SECRET_PATTERNS_CS" || true)" +ci_hits="$(printf '%s\n' "$added_lines" | grep -niE "$SECRET_PATTERNS_CI" || true)" +# awk dedupes lines both passes matched, keeping first-seen order. +secret_hits="$(printf '%s\n%s' "$cs_hits" "$ci_hits" \ + | grep -v '^[[:space:]]*$' | awk '!seen[$0]++' || true)" + +if [ -n "$secret_hits" ]; then + echo "pre-commit: potential secret in staged changes:" >&2 + echo "$secret_hits" >&2 + echo "" >&2 + echo "Review the lines above. If this is a false positive (test fixture, documentation)," >&2 + echo "bypass with: git commit --no-verify" >&2 + exit 1 +fi + +# --- 2. Syntax check on staged TS/JS files --- +# Two checkers, because one tool can't do both jobs. `node --check` ignores +# --experimental-strip-types, so on TypeScript it is wrong in BOTH directions: +# it rejects valid TS (an `interface` reads as a syntax error) and accepts +# broken TS. Measured on node v26.4.0, 2026-07-23. tsc is the only correct +# parser for .ts; node is correct and much faster for .js. +# Same split as the secret scan above: a git failure must not read as "no files +# staged", which would skip the language check silently. +if ! staged_names="$(git diff --cached --name-only --diff-filter=AM)"; then + echo "pre-commit: cannot read the staged file list — refusing to skip the check" >&2 + exit 1 +fi + +staged_js="$(printf '%s\n' "$staged_names" | grep -E '\.(js|jsx|mjs|cjs)$' || true)" +staged_ts="$(printf '%s\n' "$staged_names" | grep -E '\.(ts|tsx|mts|cts)$' || true)" + +failed="" + +if [ -n "$staged_js" ] && command -v node >/dev/null 2>&1; then + while IFS= read -r f; do + [ -z "$f" ] && continue + [ -f "$f" ] || continue + if ! node --check "$f" >/dev/null 2>&1; then + failed="${failed}${f}"$'\n' + fi + done <<< "$staged_js" +fi + +if [ -n "$staged_ts" ]; then + tsc_bin="" + if [ -x "./node_modules/.bin/tsc" ]; then + tsc_bin="./node_modules/.bin/tsc" + elif command -v tsc >/dev/null 2>&1; then + tsc_bin="tsc" + fi + + if [ -n "$tsc_bin" ]; then + while IFS= read -r f; do + [ -z "$f" ] && continue + [ -f "$f" ] || continue + # Filter to TS1xxx, TypeScript's syntactic diagnostic range. TS2xxx and + # up are type errors, which need the whole project graph and are the + # build's job, not this hook's. + out="$("$tsc_bin" --noEmit --skipLibCheck --target es2022 \ + --moduleDetection force "$f" 2>&1 || true)" + if printf '%s\n' "$out" | grep -qE 'error TS1[0-9]{3}:'; then + failed="${failed}${f}"$'\n' + fi + done <<< "$staged_ts" + fi +fi + +if [ -n "$failed" ]; then + printf 'pre-commit: syntax errors in staged files:\n\n%s\n' "$failed" >&2 + echo "Fix the parse errors above, then re-stage." >&2 + exit 1 +fi + +exit 0 diff --git a/languages/typescript/tests/pre-commit.bats b/languages/typescript/tests/pre-commit.bats new file mode 100644 index 0000000..5519baa --- /dev/null +++ b/languages/typescript/tests/pre-commit.bats @@ -0,0 +1,150 @@ +#!/usr/bin/env bats +# +# Tests for languages/typescript/githooks/pre-commit — the secret scan plus +# syntax/lint gate that runs on staged TS/JS files. +# +# The secret scan is the security-critical half and is language-independent, so +# it gets the same coverage here as in the bash bundle: a real key blocks, a +# clean diff passes, and the case-sensitivity split that keeps base64 blobs from +# false-positiving is exercised directly. +# +# Each test builds a throwaway git repo, stages content, and runs the hook from +# inside it — the hook reads `git diff --cached`, so a real index is required. + +HOOK="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/githooks/pre-commit" + +setup() { + TEST_DIR="$(mktemp -d -t pre-commit-ts-bats.XXXXXX)" + cd "$TEST_DIR" || exit 1 + git init -q . + git config user.email t@example.com + git config user.name Test + # A base commit so `git diff --cached` has a parent to diff against. + echo "seed" > seed.txt + git add seed.txt + git commit -qm seed +} + +teardown() { + cd / || true + rm -rf "$TEST_DIR" +} + +# ---- Normal ---------------------------------------------------------- + +@test "pre-commit(ts): a clean staged JS file passes (exit 0)" { + printf 'export const f = (x) => x + 1;\n' > ok.js + git add ok.js + run bash "$HOOK" + [ "$status" -eq 0 ] +} + +@test "pre-commit(ts): an empty staging area passes (exit 0)" { + run bash "$HOOK" + [ "$status" -eq 0 ] +} + +# ---- Error: the secret scan ------------------------------------------ + +@test "pre-commit(ts): an AWS key in a staged file blocks (exit 1)" { + printf 'const KEY = "AKIAIOSFODNN7EXAMPLE";\n' > conf.ts + git add conf.ts + run bash "$HOOK" + [ "$status" -eq 1 ] + [[ "$output" == *"potential secret"* ]] +} + +@test "pre-commit(ts): an sk- style token blocks (exit 1)" { + printf 'const TOKEN = "sk-abcdefghijklmnopqrstuvwxyz0123";\n' > conf.ts + git add conf.ts + run bash "$HOOK" + [ "$status" -eq 1 ] +} + +@test "pre-commit(ts): a quoted api_key assignment blocks (exit 1)" { + printf 'const api_key = "abcdefghijklmnopqrstuvwxyz";\n' > conf.ts + git add conf.ts + run bash "$HOOK" + [ "$status" -eq 1 ] +} + +@test "pre-commit(ts): a private-key header blocks (exit 1)" { + printf 'const PEM = "-----BEGIN RSA PRIVATE KEY-----";\n' > conf.ts + git add conf.ts + run bash "$HOOK" + [ "$status" -eq 1 ] +} + +# ---- Boundary: the case-sensitivity split ---------------------------- + +@test "pre-commit(ts): a mixed-case base64 blob does NOT false-positive" { + # The AWS pattern is uppercase-only by design. Under -i it would match any + # 20-char mixed-case run, which random base64 hits often enough to block + # real commits. This is the regression test for that split. + printf 'const BLOB = "AKIAbcdefGHIJklmnOPqr0123456789abcdefGHIJ";\n' > data.ts + git add data.ts + run bash "$HOOK" + [ "$status" -eq 0 ] +} + +@test "pre-commit(ts): a short quoted password value does NOT block" { + # The keyword patterns require 16+ chars, so a placeholder stays quiet. + printf 'const password = "short";\n' > conf.ts + git add conf.ts + run bash "$HOOK" + [ "$status" -eq 0 ] +} + +@test "pre-commit(ts): a secret only in a REMOVED line does not block" { + printf 'const KEY = "AKIAIOSFODNN7EXAMPLE";\n' > conf.ts + git add conf.ts + git commit -qm "add key" + rm conf.ts + git add -A + run bash "$HOOK" + [ "$status" -eq 0 ] +} + +# ---- Error: the syntax gate ------------------------------------------ + +@test "pre-commit(ts): a staged JS syntax error blocks (exit 1)" { + command -v node >/dev/null 2>&1 || skip "node not installed" + printf 'const x = ;\n' > bad.js + git add bad.js + run bash "$HOOK" + [ "$status" -eq 1 ] + [[ "$output" == *"syntax"* ]] +} + +@test "pre-commit(ts): a staged TS syntax error blocks (exit 1)" { + command -v tsc >/dev/null 2>&1 || skip "tsc not installed" + printf 'export function f( {\n return 1;\n}\n' > bad.ts + git add bad.ts + run bash "$HOOK" + [ "$status" -eq 1 ] +} + +@test "pre-commit(ts): valid TS-only syntax is NOT read as broken JS" { + command -v tsc >/dev/null 2>&1 || skip "tsc not installed" + # The regression guard for the node --check trap: `node --check` rejects + # valid TypeScript, so using it on .ts would block every real commit. + printf 'interface P { a: string }\nexport const p: P = { a: "x" };\n' > types.ts + git add types.ts + run bash "$HOOK" + [ "$status" -eq 0 ] +} + +@test "pre-commit(ts): a TYPE error that parses does not block (out of scope)" { + command -v tsc >/dev/null 2>&1 || skip "tsc not installed" + printf 'const n: number = "nope";\nexport { n };\n' > typeerr.ts + git add typeerr.ts + run bash "$HOOK" + [ "$status" -eq 0 ] +} + +@test "pre-commit(ts): a broken NON-TS/JS file does not trip the syntax gate" { + printf 'this is (((not javascript\n' > notes.txt + git add notes.txt + run bash "$HOOK" + [ "$status" -eq 0 ] +} diff --git a/languages/typescript/tests/validate-typescript.bats b/languages/typescript/tests/validate-typescript.bats new file mode 100644 index 0000000..c5da5d4 --- /dev/null +++ b/languages/typescript/tests/validate-typescript.bats @@ -0,0 +1,125 @@ +#!/usr/bin/env bats +# +# Tests for languages/typescript/claude/hooks/validate-typescript.sh — the +# PostToolUse hook that syntax-checks edited TS/JS files and blocks on a +# violation. +# +# The hook reads tool-call JSON on stdin and extracts the file path, so each +# test pipes a JSON payload naming a real file it wrote into a temp dir. +# +# The syntax gate needs node, so those tests skip when node is absent. Full +# type checking is deliberately out of scope for the hook (it needs the whole +# project graph), so a type error that is syntactically valid must pass. + +HOOK="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/claude/hooks/validate-typescript.sh" + +setup() { + TEST_DIR="$(mktemp -d -t validate-ts-bats.XXXXXX)" +} + +teardown() { + rm -rf "$TEST_DIR" +} + +payload() { + printf '{"tool_input": {"file_path": "%s"}}' "$1" +} + +# ---- Normal ---------------------------------------------------------- + +@test "validate-typescript: a clean .ts file passes silently (exit 0)" { + command -v node >/dev/null 2>&1 || skip "node not installed" + printf 'export function f(x: number): number {\n return x + 1;\n}\n' > "$TEST_DIR/clean.ts" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/clean.ts")" + [ "$status" -eq 0 ] + [ -z "$output" ] +} + +@test "validate-typescript: a clean .js file passes silently (exit 0)" { + command -v node >/dev/null 2>&1 || skip "node not installed" + printf 'export function f(x) {\n return x + 1;\n}\n' > "$TEST_DIR/clean.js" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/clean.js")" + [ "$status" -eq 0 ] +} + +@test "validate-typescript: a .tsx file is validated too" { + command -v node >/dev/null 2>&1 || skip "node not installed" + printf 'export const A = 1;\n' > "$TEST_DIR/clean.tsx" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/clean.tsx")" + [ "$status" -eq 0 ] +} + +# ---- Error ----------------------------------------------------------- + +@test "validate-typescript: a syntax error blocks (exit 2, names the failure)" { + command -v node >/dev/null 2>&1 || skip "node not installed" + printf 'export function f( {\n return 1;\n}\n' > "$TEST_DIR/bad.ts" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/bad.ts")" + [ "$status" -eq 2 ] + [[ "$output" == *"SYNTAX"* ]] +} + +@test "validate-typescript: the block payload is valid JSON carrying the context" { + command -v node >/dev/null 2>&1 || skip "node not installed" + printf 'const x = ;\n' > "$TEST_DIR/bad.js" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/bad.js")" + [ "$status" -eq 2 ] + echo "$output" | head -1 | jq -e '.hookSpecificOutput.hookEventName == "PostToolUse"' +} + +# ---- Boundary -------------------------------------------------------- + +@test "validate-typescript: a type error that parses is NOT blocked (out of scope)" { + command -v node >/dev/null 2>&1 || skip "node not installed" + # Assigning a string to a number is a type error, not a syntax error. The + # hook checks parseability only; tsc over the project graph owns this. + printf 'const n: number = "not a number";\nexport { n };\n' > "$TEST_DIR/typeerr.ts" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/typeerr.ts")" + [ "$status" -eq 0 ] +} + +@test "validate-typescript: TS-only syntax in a .ts file parses (not read as JS)" { + command -v node >/dev/null 2>&1 || skip "node not installed" + # Interfaces and type annotations are invalid JS. Stripping types must happen + # before the parse, or every real .ts file would be reported as broken. + printf 'interface P { a: string }\nexport const p: P = { a: "x" };\n' > "$TEST_DIR/types.ts" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/types.ts")" + [ "$status" -eq 0 ] +} + +@test "validate-typescript: a non-TS/JS file is ignored (exit 0)" { + printf 'not javascript at all (((\n' > "$TEST_DIR/notes.txt" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/notes.txt")" + [ "$status" -eq 0 ] +} + +@test "validate-typescript: a .json file is ignored (exit 0)" { + printf '{"a": 1}\n' > "$TEST_DIR/data.json" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/data.json")" + [ "$status" -eq 0 ] +} + +@test "validate-typescript: empty file_path is a no-op (exit 0)" { + run bash "$HOOK" <<< '{"tool_input": {}}' + [ "$status" -eq 0 ] +} + +@test "validate-typescript: a missing file is a no-op (exit 0)" { + run bash "$HOOK" <<< "$(payload "$TEST_DIR/does-not-exist.ts")" + [ "$status" -eq 0 ] +} + +@test "validate-typescript: an empty .ts file passes" { + command -v node >/dev/null 2>&1 || skip "node not installed" + : > "$TEST_DIR/empty.ts" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/empty.ts")" + [ "$status" -eq 0 ] +} + +@test "validate-typescript: a file in a dotted parent dir is still matched" { + command -v node >/dev/null 2>&1 || skip "node not installed" + mkdir -p "$TEST_DIR/my.project" + printf 'const x = ;\n' > "$TEST_DIR/my.project/bad.ts" + run bash "$HOOK" <<< "$(payload "$TEST_DIR/my.project/bad.ts")" + [ "$status" -eq 2 ] +} diff --git a/publish/SKILL.md b/publish/SKILL.md new file mode 100644 index 0000000..d2b906c --- /dev/null +++ b/publish/SKILL.md @@ -0,0 +1,471 @@ +--- +name: publish +description: | + The publish flow for commits, pull requests, and PR review comments. Covers the mandatory pre-flight reconcile against upstream, the local code review gate, the draft/voice/approval gate before anything is published, conventional-commit message format, the Voice and Focus rules for commit bodies and PR prose, PR description structure, the three PR-review shapes (bundled review with inline pins, issue-thread comment, threaded reply), hook-level authorization, merge strategy, and the pre-commit checklist. + + Use whenever a commit, a push, a pull request, a PR description, or a PR review comment is about to be produced — including amends, and including a one-line commit. Load it BEFORE drafting the message, not after, because the flow gates what gets written. + + Do NOT use for the invariants that apply whether or not you are publishing: author identity, the no-AI-attribution ban, and the public-artifact content-scope rules all live in claude-rules/commits.md and are always loaded. +--- + +# Publish Flow + +Applies to: commits, pull requests, and PR review comments. + +The invariants — author identity, no AI attribution anywhere, and what must +never appear in a team-visible artifact — are NOT in this file. They live in +`claude-rules/commits.md`, which is always loaded, because a violation there is +permanent and reaches other people. This file is the procedure: how the message +gets written, reviewed, approved, and published. + +## Commit Message Format + +Commit messages follow the [Conventional Commits](https://www.conventionalcommits.org/) spec. + +### Structure + + <type>[optional scope]: <description> + + [optional body] + + [optional footer(s)] + +### Types + +- `feat:` — new feature (correlates with MINOR in SemVer) +- `fix:` — bug fix (correlates with PATCH in SemVer) +- `refactor:` — code restructuring, no behavior change +- `perf:` — performance improvement +- `test:` — adding or updating tests +- `docs:` — documentation only +- `style:` — formatting, whitespace, missing semicolons (no code-behavior change) +- `build:` — build system or external dependencies +- `ci:` — CI configuration and scripts +- `chore:` — anything else: tooling, meta, housekeeping + +The Conventional Commits spec doesn't mandate the type list. Add a new type only when the existing ones genuinely don't fit and the team will agree on what it means. + +### Scope + +A scope MAY follow the type, in parentheses, naming the affected area of the codebase: `feat(parser): add ability to parse arrays`. Use a single noun. + +### Breaking changes + +Either append `!` after the type or scope, or include a `BREAKING CHANGE:` footer (uppercase — required). Both at once is fine and adds detail. `!` alone is enough. + + feat!: drop support for Node 6 + + BREAKING CHANGE: uses JavaScript features not available in Node 6. + +### Subject line + +Imperative mood. ≤72 characters. No trailing period. The full subject is `<type>[scope]: <description>` — the 72-char limit covers the whole thing. + +### Body + +Optional. Begins one blank line after the subject. Free-form, multiple paragraphs allowed. Don't hard-wrap body lines — write each paragraph and each bullet as a single logical line and let the renderer (GitHub, Linear, `git log`) soft-wrap. Hard wraps shrink the visible render width in web UIs and cause awkward mid-sentence breaks. The same soft-wrap rule applies to PR bodies. + +Skip the body when the subject line covers the change. + +### Footers + +Optional. One blank line after the body. One per line. Format: `Token: value` or `Token #value` — the git trailer convention. The token uses `-` in place of whitespace (e.g. `Reviewed-by`, `Refs`, `Acked-by`). `BREAKING CHANGE:` is the one token allowed to contain a space, and `BREAKING-CHANGE:` is treated as a synonym. + +### How to write the message + +Write commit messages as if you're explaining the change to someone debugging a failure six months from now. Focus on what changed and why, not the play-by-play of how you typed it. Short imperative summaries like "Validate input before processing" age better than diary-style notes. + +The body, when you need it, is where context belongs — the constraint, bug, or tradeoff that forced the change. Over time the body becomes a lightweight decision log, which is more valuable than perfectly formatted messages. + +Commit messages describe what changed and why, not the process that produced the change. Don't reference code review, linting, test runs, or other workflow steps in the body (e.g. "from local review," "review surfaced," "flagged by reviewer"). Reviewers and future archaeologists want the what and the why. How you got there belongs in the PR discussion, not the commit. + +### Examples + +**Subject only:** + + docs: correct spelling of CHANGELOG + +**With scope:** + + feat(lang): add Polish language + +**With body and footer:** + + fix: prevent racing of requests + + Introduce a request id and a reference to the latest request. Dismiss incoming responses other than from the latest request. + + Remove timeouts which were used to mitigate the racing issue but are obsolete now. + + Refs: #123 + +**Breaking change with `!`:** + + feat(api)!: send an email to the customer when a product is shipped + +**Breaking change in footer:** + + feat: allow provided config object to extend other configs + + BREAKING CHANGE: `extends` key in config file is now used for extending other config files. + +## Voice and Focus + +Applies to commit bodies, PR descriptions, and PR comments (review replies, follow-up notes, thread responses). + +**Write as if to a colleague.** The reader is a teammate who'll see this in `git log`, a PR feed, or a Linear thread. "I" is allowed where natural. Don't sound abstract — name the file, the function, the constraint, the symptom. Press-release voice ("This change improves...") and committee voice ("It is recommended that...") both come out. The message has to read like one engineer talking to another, not like a generated artifact. + +**No felt-experience narration.** Don't tell the reader how the change will feel or how often you'll use it. Phrases like "I'll feel this every time I commit", "this will be a relief", "I'm excited about" — these read as performance, not communication. State what changed and let the reader decide what to do with it. + +**Don't noun-ify verbs.** "The ask", "a learn", "a reveal", "the spend", "a build" — use the real noun: "the request", "the lesson", "the finding", "the budget", "the system". Verb-as-noun reads as corporate-speak and makes the sentence feel performed. + +**No sentence fragments in prose.** Every prose sentence needs a subject and a verb. "Two changes." or "Fix incoming." or "Body as decision log." read as bullet-list shorthand even when they're standing alone in a paragraph. Bullets and headings can be fragments — prose sentences cannot. + +**"I" is the author, not the user.** First person is for what *I* did or decided in this commit ("I dropped the legacy fallback because..."). It's not for describing how the software or rule behaves for whoever uses it next. "The dialog only opens if I ask" is wrong when the rule is read by someone else — that "I" becomes ambiguous. Use third-person or passive for behavior: "opens on request", "opens when asked", "opens when the user invokes it". Code and systems are the actor; "I" stays for decisions. + +**First person where it fits.** When the subject is you or a decision you made, use "I" ("I added X", "I kept the parameter as `Any` because..."). When the subject is a team decision or shared rationale, "we" fits. When another author's prior work is the subject, name them ("Kostya's PR #116 did X"). Third-person constructions like "This PR introduces X" or "This change restores Y" read as press-release self-narration. The commit *is* the change, so don't announce it. Code and systems can stay third-person when they're the actor ("the guard rejects...", "the serializer returns...") — first person is for describing what you did or decided, not for narrating how the code behaves. + +**Brief. Terse is preferred.** A one-sentence body beats a paragraph saying the same thing. If the subject line covers it, skip the body entirely. Cut every clause that restates what the diff or the PR card already shows. Length is not a proxy for care. Rhetorical padding ("worth noting", "it's important to understand") always comes out; keep what a reader will actually use. + +**Follow-up approvals stay terse.** A re-review that just confirms prior CHANGES_REQUESTED feedback got addressed should be `Approved.` and nothing more. The fixes are visible in the diff and in the prior review thread, so restating them adds noise. The first round of substantive review gets a real comment. Subsequent sign-offs after fixes do not. Counts as a trivial one-liner under the Step 2 exception, so the draft-file flow can be skipped. + +**Kind.** PR comments and review replies are directed at a specific person. Acknowledge them when it fits ("thanks for the review") without pouring it on. When you disagree or push back, frame it as your read rather than a correction ("I think...", "my read was...", "did you mean X?"). Leave room for the other person to have seen something you didn't. A polite question beats a defensive explanation. Kindness is free and makes the next review cheaper. + +Focus on what was wrong and what was corrected. Not the mechanics. +Readers skimming `git log` or a PR want the before-state, the +after-state, and the reason. They don't need a TypeScript-variance +lesson, a compiler-inference walkthrough, or a trip through an API's +internals. Keep the "why" to one sentence unless a subtle invariant +genuinely needs more. + +Don't stack technical terms. A sentence that chains three or more type +signatures, API names, or compiler concepts reads as a jargon wall. +Break it into shorter sentences and translate to reader-facing +language. "The mock returns `Promise<Mission>`, so the resolver's +argument is `Mission`, not `unknown`" beats the full inference chain +that produces that signature. Keep the terms a reader will grep for, +drop the ones that name compiler internals. + + +Different artifact types carry different content. Don't duplicate. + +**PR descriptions:** four sections, in order. + +1. **Problem** — what's wrong, with enough detail that a teammate can + recognize the same failure mode in their own work. +2. **Fix** — what changed. +3. **Why this fixes it** — causal link, one or two sentences. +4. **How it was tested** — skip for proposals, specs, or discussions; + required for shipped fixes. + +The PR is the technical artifact. It carries the detail. + +If the project's publishing overlay defines a ticket system, see it for +ticket-body conventions (a ticket body is typically just the Problem and +Fix, with the causal why and test verification left to the PR). + +**PR review comments** are conversational and don't follow this +structure — they follow the Voice and Focus rules above. + +Verbose preambles, motivational language, and context unrelated to the +problem belong out. Same conciseness pressure as commit-message bodies. + + +## Review and Publish + +Commits and PRs are team-visible, permanent, and hard to amend once shared +(especially after push or after a reviewer has replied). Before executing +`git commit` or `gh pr create`, the change must pass a local code review +*and* the message must be reviewed by the user. The flow has three steps, in +order. + +### Step 0: pre-flight reconcile (mandatory) + +Before reviewing the diff, fetch from the remote and reconcile against the +upstream of the current branch. Reconciliation can change the working state +when a rebase brings in upstream commits that touch staged files, and that +would invalidate Step 1's review. Handling drift first means the review and +the commit message describe the post-reconcile state. + +1. Fetch all remotes: + + git fetch --all --prune + +2. If the current branch has no upstream (new branch, never pushed), skip + to Step 1 — there's nothing to reconcile against, and the first push + sets the upstream. + +3. Otherwise, check divergence against `@{u}`: + + git rev-list --left-right --count @{u}...HEAD + + Output is `<behind>\t<ahead>`. Decide based on the pair: + + - **0 behind, anything ahead** — no-op. Continue to Step 1. + - **Behind only, clean tree** — fast-forward: `git merge --ff-only @{u}`. + - **Behind only, dirty tree** — surface to the user. Don't auto-stash or + auto-merge. Offer to commit or stash first, or skip the reconcile and + proceed knowing the push may need attention later. + - **Diverged (behind AND ahead)** — surface to the user. Ask whether to + rebase the local commits onto upstream (default for feature branches), + merge the upstream branch in (rare; preserves both lines), or skip and + proceed with the divergence. Don't auto-rebase. + +4. **PR flow only.** Also fetch the base branch (usually `main`) and check + whether the feature branch's base is behind. Surface this informationally; + don't auto-rebase the feature branch without asking. The "X commits + behind base" badge on the PR is a follow-up decision, not a reason to + block publish. + +The startup workflow's `git fetch --all --prune` doesn't substitute for +Step 0. Upstream can advance during a long session, especially across +machines or with teammates pushing in parallel. Run Step 0 every time the +publish flow starts. + +### Step 1: adversarial review by an isolated reviewer (mandatory) + +The review runs in a **subagent**, never inline, and it runs on **every** +commit. The author does not review their own work. + +**Why isolation, not just review.** A self-review checks the diff against the +author's own model of what the diff should do. It cannot check the model. The +errors that survive a self-review are the ones that were never visible in the +diff — a scope inherited from whoever reported the problem, a blast radius +estimated instead of measured, a fix that is correct for the case the author +had in mind and wrong for the one they never considered. Only a reviewer that +does not hold the author's model catches those, so the isolation is the point +and the adversarial stance is the method. + +**Dispatch contract.** Spawn the reviewer via the Agent tool and give it these +three things, the third whenever one exists: + +1. **The diff** — `git diff --cached` for a commit, the branch diff for a PR. +2. **The claim** — one line from the author stating what the change does. Write + it before dispatching. This is the thing under test: the reviewer's job is + to check the diff against the claim. +3. **The requirement source, when one exists** — the ticket, plan, ADR, or task + body the work was done against. Pass it verbatim. + +Withhold everything else: the conversation, the exploration, the dead ends, and +above all the author's reasoning for why the change is right. Those are what +transmit the author's model, which is what the reviewer exists to not have. A +reviewer given the rationale reviews the rationale. + +**Why the requirement source is not withheld.** A ticket or plan is not the +author's model of the change — it is the independent record of what was asked, +written before the work and usually by someone else. It is the only artifact +that can contradict the author's one-line claim. Withhold it and the claim +becomes self-certifying: the reviewer checks the diff against a sentence the +author wrote, which cannot surface scope creep or a missing requirement. That +also strands `review-code`'s Intent-vs-Delivery criterion, which is skipped +outright when no intent context is supplied and is the one criterion aimed at +the inherited-scope error this whole gate exists to catch. + +Invoke the review with `/review-code --staged` (commit), `/review-code` (branch +diff against the `main` merge-base), or `/review-code <PR#>` (someone else's +PR), and tell it to run its adversarial pass. + +**Adversarial, with substantiation.** The reviewer is prompted to *refute* the +change rather than to bless it. But an agent told to attack will manufacture +findings to satisfy the instruction, so the stance carries a floor: a finding +that cannot be substantiated against the diff is not a finding and must be +dropped. `review-code`'s confidence filter and false-positive filter are what +enforce that floor — adversarial raises the appetite for looking, never the +tolerance for a weak claim. + +**Scope is every commit; the reviewer decides triviality, not the author.** +There is no "trivial enough to skip" exemption. A floor written in terms of +"small" or "mechanical" puts the judgment back with the author, whose judgment +is the thing being checked. Dispatch always, and let `review-code`'s own Phase 0 +eligibility gate return fast on a whitespace-only diff, an obvious revert, or an +already-reviewed SHA. A cheap spawn on a trivial commit is the price of the +author never getting to rule on their own diff. + +**Verdict, and the re-review loop.** The reviewer returns one of four outcomes, +and all four are defined exits: + +- **Approve** — the gate is satisfied. Proceed to Step 2. +- **Skipped** — `review-code`'s Phase 0 found the diff ineligible (whitespace + only, an obvious revert, an already-reviewed SHA). This **satisfies the gate** + and the flow proceeds. A skip is a reviewer's ruling, which is the point; what + is forbidden is the *author* ruling their own diff too trivial to look at. +- **Request Changes** — blocking findings stand. Enter the loop below. +- **Needs Discussion** — the reviewer has a disagreement it cannot settle from + the diff: an architectural objection, a question about whether the change + should exist at all. This **stops immediately and goes to the user**; it does + not enter the loop. Routing it to the loop would answer "should we do this?" + with "fix these findings," which is the wrong question and burns rounds on a + disagreement no amount of editing resolves. Unattended callers park it exactly + as they park a bound hit. + +Approval is the reviewer's to give; the author never declares their own change +clean. + +That set is closed. `review-code` emits Approve, Request Changes, or Needs +Discussion, and its Phase 0 emits Skipped; every one has a defined exit above. A +verdict outside those four means the reviewer went off-contract — surface it +rather than interpreting it. + +**The loop turns on blocking findings, not on the verdict token.** It ends when +no Critical or Important finding stands. A Minor-only result is not grounds for +another round: fix it or don't, but do not spend a round on it, and do not +escalate to the user over one. A reviewer holding only Minor findings should +return Approve and say what it left. + +On Request Changes: + +1. Surface **all** findings — Critical, Important, Minor. Critical and Important + block; Minor is shown and does not block. +2. Fix the blocking findings. +3. **Re-review, and keep re-reviewing until the reviewer approves.** A fix is a + new change and gets the same scrutiny as the original. Fixing under review + pressure is exactly when a regression gets introduced, so an unreviewed fix + is the hole this loop closes. + +**Continue the same reviewer, don't spawn a fresh one.** Send the updated diff +back to the existing reviewer (`SendMessage` with its agent ID). It holds its own +findings, so it can confirm each one is actually addressed. A fresh reviewer each +round cannot tell "addressed" from "never existed", re-litigates settled points, +and drifts to a new set of findings every round, which never converges. + +The continued reviewer must **re-verify each finding against the new diff**, not +against the author's description of the fix. "I fixed it" is a claim, and taking +it at face value is how a review round becomes a rubber stamp. + +**Bounds, so the loop terminates.** Two conditions end it early and hand the +decision to the user: + +- **Three rounds without approval** (the initial review plus two re-reviews). + This matches the two-fix-attempts limit in `subagents.md`: past that, the + problem is usually the approach rather than the diff. +- **A finding recurs after being reported fixed.** That is oscillation — the + fix for one finding reintroducing another — and another round will not + resolve it. Stop on the first recurrence rather than spending the remaining + rounds. + +In both cases, stop and surface: the standing findings, what was tried, and the +decision needed. + +**Override.** The user can bypass the block with an explicit "proceed anyway" (or +equivalent). The user is also the adjudicator when the author believes a finding +is wrong: say so with the reasoning and let the user rule. Do not resolve a +disagreement with the reviewer by overruling it silently. Without an explicit +override, do not proceed to Step 2. + +**When the Agent tool is unavailable.** Per `subagents.md`, don't block: run the +review in the main thread, but hold it to the same contract — review against the +stated claim, refute rather than bless, substantiate every finding, loop on fixes +until clean, same bounds. State plainly that the review was not isolated, because +a self-review under an adversarial prompt is weaker evidence and the user should +know which one they got. + +### Step 2: draft, review, publish + +**Voice patterns and the approval gate are two independent decisions.** Don't bundle them. + +*Voice patterns are always personal for publish artifacts.* Commit messages, PR titles + bodies, and PR review comments all go out under the user's name, so they always run through `/voice personal` (the full pattern walk — general + Craig's-voice + the artifact-mechanics patterns: first-person rewrite, public-artifact scope flag, praise/correction asymmetry, finding stems), regardless of whether `.ai/` is tracked. These three are personal-voice artifacts by definition — the skill's personal mode exists for exactly them. Pattern #39 (public-artifact scope flag) matters *most* on team-visible artifacts, so it must never be skipped on a PR comment or PR body. There is no "general-voice mode" for publish artifacts. + +*The approval gate turns on whether anyone else reads the history.* The gate +exists so Craig sees the exact words that go out under his name. Skipping it +trades that for velocity, and that trade is only worth making where the repo is +genuinely shared with other people. + +The old signal for this was whether `.ai/` is tracked, used as a proxy for +"team repo." It was the wrong proxy and it failed in the direction that +matters: rulesets, home, and work all track `.ai/` — rulesets as a committed +mirror, the others because the project history *is* the project — while all +three are Craig's private single-user repos. The rule as written skipped the +gate on his three most-used projects. + +Check the remote host instead, which is what actually distinguishes them: + +``` +git remote -v 2>/dev/null | grep -v 'cjennings\.net' | head -1 +``` + +- **No output** — every remote is on `cjennings.net`, so the repo is Craig's + own and nobody else reads the log. **Gate applies**: write to `/tmp`, run + `/voice personal`, print inline, ask approve / request changes / open in + editor, and publish only on explicit approval. +- **Any output** — a remote on a host someone else can read (GitHub, a GHE + instance, a team server). **Gate skipped for velocity**: write to `/tmp`, run + `/voice personal`, print inline, publish immediately. +- **No remote at all** — a local-only repo. Gate applies; there is no + velocity argument without a reader. + +As of 2026-07-27 every project resolves to gate-applies, because every remote +is `cjennings.net`. That is the correct answer, not a bug: it matches how the +flow has actually been run. + +Either way the draft runs through `/voice personal` first. The subflows below describe the full gated path. For the gate-skipped path, run the same `/voice personal` pass, then collapse the "Ask: approve, request changes, or open in editor" step — the draft prints inline and the publish step runs immediately afterward. + +**For commit messages:** + +1. Write the proposed message to `/tmp/commit-<short-slug>.md`. +2. Run `/voice personal` on the file. Always. The skill walks its full pattern list covering signs of AI writing, universal good-writing rules (Strunk & White, Orwell, Plain English, Garner), and Craig's voice patterns (first-person rewrite, semicolons → periods/commas, contractions, sentence-split on conjunctions, felt-experience cut, sentence-fragment rewrite, terse cut for rhetorical padding, no-emphasis-formatting, public-artifact scope flag, praise/correction asymmetry, finding stems). The commit subject line stays imperative per Conventional Commits — `/voice personal` rewrites the body, not the subject. Skip the pass for purely mechanical commits (a chore version bump, a typo fix) where the subject alone carries the message. +3. Print the final draft inline in the terminal. Every line, exactly as it'll be committed. No truncation, no summary. State that the skill ran (e.g. "/voice personal — full pattern walk"). If pattern #39 (public-artifact scope) flagged anything, surface those warnings; the user resolves them manually. +4. Ask: approve, request changes, or open in editor. Wait for an explicit answer. Do not open the file in `emacsclient` (or any editor) by default — print first, edit only if asked. + - **Approve** → commit with `git commit -F /tmp/commit-<short-slug>.md`. + - **Request changes** → make them, re-run `/voice personal`, re-print inline, ask again. + - **Open in editor** → only if the user asks. `emacsclient -n /tmp/commit-<short-slug>.md`. After the editor closes, re-read the file, re-print the contents inline, and ask again. + +**For PR descriptions, and for PR review comments and replies:** read +`references/pull-requests.md` in this skill directory. It carries the PR +description shape, the three review shapes (bundled review with inline pins, +issue-thread comment, threaded reply), the `gh api` calls, and the +publishing-overlay hooks. A plain commit needs none of it, so it is kept out of +this file — load it when the artifact is a PR. + +**Approve does not authorize a merge.** Reviewing a PR never authorizes merging it. Anything in `## Merge Strategy` below applies only to merges *you* are about to perform on your own branches — and even then, the merge needs its own explicit user confirmation per the rules there. A project's publishing overlay may add a team merge practice (e.g. approve-then-author-merges, where the review notification hands the merge decision to the PR author); that's an overlay concern, not a global one. + +**Exception:** trivial one-liners the user dictated verbatim in the +conversation (e.g. "commit this as `chore: bump version`", "reply just +'thanks for the review'") can skip the draft-file step in Step 2. +Step 1's review still runs on every commit — this carve-out is about the +draft-file step, not the review. Phase 0 is what rules a trivial diff out, and +its Skipped result satisfies the gate. An acknowledgment-only PR reply commits +nothing, so there is no diff to review. + +**Single-skill gate.** Each of the three subflows above runs `/voice personal` before printing the draft — the full pattern walk covering AI-writing signs, universal good-writing rules, Craig's voice patterns, and the artifact-mechanics patterns (first-person rewrite, public-artifact scope flag, praise/correction asymmetry, finding stems). Publish artifacts (commits, PR titles + bodies, PR review comments) always use personal mode; the `.ai/`-tracking check at the top of Step 2 decides only whether the approval gate fires, not which patterns run. Running the skill is mandatory; the printed draft must have been through it. When the user asks mid-flow for "the voice pass" on an in-progress draft, that means re-run the full pattern walk — not a subset. Always state that the skill ran when announcing the printed draft (e.g. "/voice personal — full pattern walk"). Skipping the pass without flagging it is a defect. The terse/omit-needless-words cut (pattern #38) is the *last* thing the skill does before the draft is printed: read each sentence and cut it in half, keeping only what changes meaning. The draft the user first sees must already be terse — if they have to ask for an Orwell pass after seeing it, the pass was skipped. + +**If `/voice` is unavailable.** The skill should be installed (it ships with rulesets), but a fresh or partial environment may not have it. Don't let that block the publish, and don't skip the discipline silently. Walk the same patterns inline — they're documented in the skill, and the publish flow already names which ones matter (first-person rewrite, semicolons → periods/commas, contractions, sentence-split, felt-experience cut, fragment rewrite, terse cut, the pattern #39 public-artifact scope flag, plus the AI-writing and good-writing passes). Then state that the skill was unavailable and the pass was applied by hand (e.g. "/voice unavailable — patterns walked inline"). The gate is the pattern walk, not the tooling; the skill is the convenient way to run it, not the only way. Flag the missing skill so it gets installed. + +### Hook-level authorization + +The Step 1 code review plus the Step 2 user approval together constitute the +authorization gate for the publish action. No separate hook-level approval +prompt is needed on `git commit`, `gh pr create`, `git push`, or their +variants once Step 2 has been approved. If a hook is configured, rely on the +flow above to be the source of truth; do not treat the hook as a second +independent gate. + +## Merge Strategy + +- *Squash-merge is the default* for feature branches. It avoids carrying + WIP and fix-up commits into the target branch history and produces one + logical change per merge. +- State the planned merge approach (squash, rebase, or merge commit) and + the target branch *before* pushing or merging. Wait for explicit user + confirmation before `git push`, `gh pr merge`, or any equivalent. The + Review and Publish flow above approves the *content*; merge strategy is + a separate decision that needs its own confirmation. +- *Pre-push reconcile.* Right before `git push`, do one more + `git fetch <remote> <branch>` and verify the local branch is still + ahead-only against its upstream. If something landed between Step 0 and + push (review and draft together can take several minutes, and another + machine or teammate may push in that window), surface and resolve before + the push command runs. Catching drift here is cheaper than recovering + from a failed non-fast-forward push under publish-step pressure. +- Override the squash default only when there's a concrete reason: a + clean per-commit review history the user has explicitly asked for, a + multi-commit semantic narrative the team values, etc. Squash is the + safe default; document why when deviating. + +## Before Committing + +1. Check author identity: `git log -1 --format='%an <%ae>'` — should be the user. +2. Scan the message for AI-attribution language (including emojis and footers), and on a public or shared-remote repo for tooling-path enumeration — prose that lists `CLAUDE.md`, `.claude/`, `.ai/`, `todo.org`, `notes.org`, or `session-context`. Name the category, not the paths. Exempt: a commit whose change is one of those files, and private single-user repos. +3. Review the diff — only intended changes staged; no unrelated files. +4. Confirm staged files belong in the repo: nothing that the project's policy keeps untracked (the personal-tooling set in gitignore-mode projects), and in repos with a canonical/mirror split, the edit is on the canonical side — a mirror-only edit gets reverted by the next sync. +5. Run the full test suite and linters as their own step, read the result, and commit only on zero failures — never chain the run into the commit command (see `verification.md`). + diff --git a/publish/references/pull-requests.md b/publish/references/pull-requests.md new file mode 100644 index 0000000..9ac4ded --- /dev/null +++ b/publish/references/pull-requests.md @@ -0,0 +1,105 @@ +# Pull requests and PR review comments + +Loaded from the `publish` skill when the artifact is a PR description or a PR +review comment. A plain commit never needs any of this, which is why it lives +here rather than in SKILL.md. + +Steps 0 and 1 of the publish flow (pre-flight reconcile, local code review) and +the `/voice personal` pass still apply — see SKILL.md. This file carries only +what is specific to PRs. + +**For PR descriptions:** + +1. Write the title as line 1 and the body below it to `/tmp/pr-<slug>.md`. **Title format:** the conventional-commit subject (`refactor: remove dead if-count-is-not-None check in admin`). If the project defines a publishing overlay with a ticket system, follow it for the ticket suffix in the title and the cross-link line in the body (see the overlay). +2. Run `/voice personal` on the file. The PR title stays imperative per Conventional Commits — `/voice personal` rewrites the body, not the title. +3. Print the final draft inline in the terminal. Title on line 1, blank line, then body — exactly as it'll be posted. State that the skill ran. Surface any pattern #39 (public-artifact scope) warnings. +4. Ask: approve, request changes, or open in editor. Wait for an explicit answer. Do not open the file in `emacsclient` (or any editor) by default. + - **Approve** → continue to step 5. + - **Request changes** → make them, re-run `/voice personal`, re-print inline, ask again. + - **Open in editor** → only if the user asks. `emacsclient -n /tmp/pr-<ticket-or-slug>.md`. After the editor closes, re-read the file, re-print inline, ask again. +5. Split the file on the first blank line and pass the title and body to `gh pr create --title "..." --body "$(tail -n +3 <file>)"` (or a heredoc) so formatting is preserved. Add `--reviewer <user[,user...]>` in the same call when you already know who should review. +6. Request reviewers on the new PR if you didn't pass `--reviewer` at create time. Use `gh pr edit <N> --add-reviewer <user>`. If the repo has a `CODEOWNERS` file, GitHub auto-suggests based on touched paths. Still issue the explicit request so the reviewer gets notified. Pick reviewers per the team's convention for the area touched (often documented in the per-repo `CLAUDE.md`). For follow-up PRs, consider tagging the parent PR's author if their context would help. PRs without a human reviewer request stall — "checks passed" is not a substitute for review. +7. **Project publishing overlay (if present).** If the project defines a publishing overlay — a `publishing-<team>.md` rule loaded from its `.claude/rules/` — run its post-create steps now: ticket cross-linking, ticket-state moves, and any other tracker integration it specifies. A project with no overlay skips this; the PR is already open and reviewers are requested, which is the complete universal flow. + +**For PR review comments and replies (review verdicts, threaded discussion, follow-up notes on someone else's PR or your own):** + +Pick the shape first. Most reviews are Shape 1. + +- **Shape 1 — Single review** (verdict + summary body + 0+ inline pins). The default for any post that carries a verdict (`APPROVE`, `REQUEST_CHANGES`, `COMMENT`), even when the verdict has no line-specific findings. One `gh api` call posts the summary, every inline pin, and the verdict together. review notification fires once for `APPROVE` or `REQUEST_CHANGES`. +- **Shape 2 — Issue-thread comment** (no verdict). General PR discussion, not a review. No inline pins. No review notification. +- **Shape 3 — Reply on an existing inline thread**. Responding to a specific prior reviewer comment. Threads under that comment. No review notification. + +**Inline threshold for Shape 1.** Any finding that names a `path:line` belongs as an inline comment pinned to that line. Cross-cutting observations (verdict rationale, "third PR with the same pattern", overall test-coverage gaps that don't pin to one place) stay in the summary body. There's no "fold one inline into the summary" exception — a single line-specific finding still goes inline. + +**Shape 1: Single review (bundled summary + inline)** + +1. Identify findings, split into **inline-eligible** (each names a specific `path:line`) and **summary-only** (cross-cutting). Decide the verdict. + +2. Write one concatenated draft to `/tmp/pr-<N>-review.md` with explicit separators: + + ``` + === SUMMARY === + <verdict summary body> + + === INLINE path=frontend/src/foo.tsx line=440 === + <inline body 1> + + === INLINE path=frontend/src/bar.tsx line=137 === + <inline body 2> + ``` + + The separator format is exactly `=== SUMMARY ===` and `=== INLINE path=<path> line=<n> ===`. The summary block is mandatory even for verdict-only reviews. Inline blocks are zero-or-more. + +3. Run `/voice personal` on the file once. The skill walks its full pattern list across every block at the same time. The separators stay intact because they aren't prose. + +4. Print the final draft inline in the terminal. Every block — the summary body AND the full prose of every inline comment — exactly as it'll be posted, with its separator header. Print the inline in full; never describe it in place of printing it ("I'd pair it with one inline on…"). Craig approves the exact words that post under his name, so the exact words must be on screen. State that the skill ran (e.g. "/voice personal — full pattern walk across summary + 3 inline"). Surface any pattern #39 warnings. + +5. Ask: approve, request changes, or open in editor. Wait for an explicit answer. Do not open the file in `emacsclient` (or any editor) by default. + - **Approve** → continue to step 6. + - **Request changes** → make them, re-run `/voice personal` on the whole file, re-print inline, ask again. + - **Open in editor** → only if the user asks. `emacsclient -n /tmp/pr-<N>-review.md`. After the editor closes, re-read, re-print inline, ask again. + +6. Split the file on the separator lines and post in **a single** `gh api` call: + + ``` + gh api repos/<owner>/<repo>/pulls/<N>/reviews \ + --hostname <ghe-host-or-omit> \ + -F event=REQUEST_CHANGES \ + -F body="<summary block>" \ + -F "comments[][path]=<path1>" \ + -F "comments[][line]=<line1>" \ + -F "comments[][body]=<inline 1>" \ + -F "comments[][path]=<path2>" \ + -F "comments[][line]=<line2>" \ + -F "comments[][body]=<inline 2>" + ``` + + `event` is one of `APPROVE`, `REQUEST_CHANGES`, `COMMENT`. The `comments[]` array can be empty for verdicts with zero line-specific findings — the call still uses the same endpoint. Pass `--hostname` for non-`github.com` hosts (a project's publishing overlay names its host when it's a GitHub Enterprise instance). + +7. Verify the review landed. `gh api repos/<owner>/<repo>/pulls/<N>/reviews --hostname ...` returns the latest review with bundled inlines. Confirm `state` matches the verdict and the inline count matches what was posted. + +8. **Project review-notification overlay (if present).** If the project defines a publishing overlay with a review-notification step (e.g. a Slack ping to the PR author), run it now — but only for `APPROVE` and `REQUEST_CHANGES` verdicts. The overlay owns the channel, the message format, the author-mention lookup, and the threading. A project with no overlay skips notification entirely. `COMMENT` verdicts and Shapes 2-3 below never notify, overlay or not. + +**Shape 2: Issue-thread comment (no verdict)** + +Use when the post is informal discussion that shouldn't appear as a review verdict (e.g. "I'd like to discuss the X approach before you continue"). + +1. Write the proposed comment to `/tmp/pr-<N>-comment.md`. +2. Run `/voice personal`. +3. Print inline, ask approve/changes/edit, gate as in Shape 1 step 5. +4. Post: `gh pr comment <N> --body-file /tmp/pr-<N>-comment.md`. +5. Verify: `gh api repos/<owner>/<repo>/issues/<N>/comments`. +6. No review notification. + +**Shape 3: Reply on an existing inline thread** + +Use when responding to a specific prior reviewer comment. + +1. Find the parent comment ID: `gh api repos/<owner>/<repo>/pulls/<N>/comments`. +2. Write the reply to `/tmp/pr-<N>-reply-<comment-id>.md`. +3. Run `/voice personal`. +4. Print inline, ask approve/changes/edit, gate as in Shape 1 step 5. +5. Post: `gh api repos/<owner>/<repo>/pulls/<N>/comments -F in_reply_to=<comment-id> -F body="$(cat /tmp/pr-<N>-reply-<comment-id>.md)"`. +6. Verify in the same `comments` list. +7. No review notification. + diff --git a/review-code/SKILL.md b/review-code/SKILL.md index 3757404..559cee8 100644 --- a/review-code/SKILL.md +++ b/review-code/SKILL.md @@ -33,7 +33,31 @@ When intent context is given, the review grades "does this match what was asked? ## Execution Model -For substantive reviews on large diffs: **dispatch the perspective passes as parallel sub-agents** via the Agent tool. Each sub-agent starts with a clean context window — the reviewer shouldn't inherit the implementer's mental model. For small single-commit tweaks, run inline. +Two levels of dispatch, and conflating them is the mistake to avoid. + +**Level one — who runs this skill.** When invoked from the `publish` flow's Step 1 gate, this skill is already running inside an isolated reviewer subagent that the publish flow spawned. That isolation is not optional and not this skill's call: the author never reviews their own change. If you are reading this in the same context that wrote the diff, the publish flow was not followed. + +**Level two — how the perspectives run inside it.** For substantive reviews on large diffs, dispatch the Phase 2 perspective passes as parallel sub-agents. For small single-commit tweaks, run them inline. This is a cost decision about fan-out *within* the review and never a licence to skip level one. + +### The adversarial contract (publish-flow reviews) + +When running as the publish flow's reviewer, operate under these terms: + +- **You have three inputs, and the boundary is deliberate.** The diff; a one-line claim from the author about what it does; and, when one exists, the requirement source it was built against (ticket, plan, ADR, task body). You were *not* given the conversation, the author's reasoning, or the alternatives they rejected, because those transmit the author's model of the change and your value is in not holding it. Do not ask for them. +- **The requirement source is not leakage — use it.** It was written before the work and usually by someone else, so it is the one input that can contradict the author's claim about their own diff. Without it, the claim is self-certifying and you can only check the diff against a sentence the author wrote. It is what makes the Intent-vs-Delivery criterion below live rather than skipped, and scope creep and missing requirements are exactly what it catches. +- **Review the diff against the claim.** Does it do what the claim says? What does it do that the claim does not mention? What breaks that the claim assumes is fine? +- **Try to refute the change, not to bless it.** Assume there is something wrong and go looking. The default posture is skepticism. +- **A finding you cannot substantiate against the diff is not a finding.** Drop it. An agent told to attack will manufacture findings to satisfy the instruction, and a manufactured finding costs the author a round of work and teaches them to discount the next review. The confidence filter and false-positive filter in Phase 4 are what hold this line: adversarial raises how hard you look, never how weak a claim you will ship. +- **Verify the premise before judging the fix.** When the diff claims to fix a bug, confirm the bug is real and reproduces in the pre-change code. A fix for a misdiagnosed problem passes a diff-shaped review and is still wrong. + +### Re-review mode (continued rounds) + +The publish flow sends the updated diff back to the *same* reviewer rather than spawning a fresh one, so a continued round has your prior findings in context. On each round: + +1. **Re-verify every prior blocking finding against the new diff.** Not against the author's description of the fix. "I fixed it" is a claim like any other, and accepting it is how a re-review becomes a rubber stamp. +2. **Review the fix as a new change.** A fix written under review pressure is prime ground for a regression, so it earns the same scrutiny as the original diff, including on lines the fix touched incidentally. +3. **Say when a prior finding recurs.** If something reported fixed is back, name it as a recurrence rather than filing it fresh. The publish flow treats recurrence as oscillation and stops the loop, which is the right outcome — another round will not converge. +4. **Do not invent new findings to justify another round.** If the blocking findings are addressed and nothing new is substantiated, approve. Approval is the expected end state, not a failure to find something. ## Phase 0 — Eligibility Gate @@ -261,7 +285,7 @@ Do **not** flag any of these as issues: - **Issues explicitly silenced in code** (e.g., `# type: ignore[...]` with a reason, lint ignore comments) unless the silencing is unjustified - **Intentional changes** in functionality clearly related to the PR's stated goal - **Changes in unmodified lines** (real issues in files the PR touches but on lines it doesn't change) -- **Framework behavior being tested** — see `testing.md` anti-patterns +- **Framework behavior being tested** — see the `testing-standards` skill, anti-patterns ### Severity Categorization @@ -437,7 +461,7 @@ None. ## Hand-Off -- **Critical** → must be addressed before merge; author fixes, re-review via `/review-code` on the updated SHA +- **Critical** → must be addressed before merge; author fixes, then the updated diff comes back to *this* reviewer for another round (publish flow, Step 1), not to a fresh one and not to the author's own judgment - **Important** → fix, or deliberately defer with an ADR (run `/arch-decide`) - **Minor** → follow-up issues or a cleanup PR - **Intent-vs-Delivery gaps** → either file tickets for the missing pieces or update the plan to reflect reality @@ -446,7 +470,7 @@ None. The summary body and the inline pins work as a pair: scannable verdict on top, full coaching conversation in the pins. Read this section paired with Inline Comment Voice below — the summary is terse precisely because the inlines carry the teaching weight. -The structured report above stays local. When the verdict is posted as a GitHub review (per `commits.md` Step 2 Shape 1), keep the summary body terse — one long sentence or a few short ones is plenty. Vary the phrasing run-to-run so consecutive reviews don't read templated. Voice: an encouraging senior dev who doesn't like to talk. Lead with the substantive pointer — the design note or blocker that's pinned inline — and close with the verdict; the summary carries no praise clause. +The structured report above stays local. When the verdict is posted as a GitHub review (per the `publish` skill, Step 2 Shape 1), keep the summary body terse — one long sentence or a few short ones is plenty. Vary the phrasing run-to-run so consecutive reviews don't read templated. Voice: an encouraging senior dev who doesn't like to talk. Lead with the substantive pointer — the design note or blocker that's pinned inline — and close with the verdict; the summary carries no praise clause. The summary body carries no praise — not a named good thing, not a bare positive. The author made the change and already knows its merits, so a compliment in a terse summary reads as filler or sycophancy. If a genuine positive is worth surfacing, it goes as a single inline pin on the relevant line (see below), never the summary body. Elaboration in the summary is for the substantive pointer and for findings — what's wrong, the failure mode, the fix — never for compliments. diff --git a/scripts/install-ai.sh b/scripts/install-ai.sh index 0c90f64..8c04e22 100755 --- a/scripts/install-ai.sh +++ b/scripts/install-ai.sh @@ -179,6 +179,24 @@ case "$track_mode" in ;; esac +# Ignore temp/ in BOTH track and gitignore modes (not the "not-a-git-repo" +# case). temp/ is the home for ephemeral, disposable, regenerable artifacts — +# throwaway regardless of whether the project tracks its .ai/ tooling, so it +# rides neither the track .gitkeep step nor the gitignore-only tooling block. +# working/ is deliberately NOT ignored: it's the tracked home of in-progress +# work, version-controlled from creation (see working-files.md). Idempotent: +# accepts either the unanchored `temp/` or anchored `/temp/` form. +if [ -n "$track_mode" ]; then + gi="$project/.gitignore" + if ! { [ -f "$gi" ] && { grep -qFx 'temp/' "$gi" || grep -qFx '/temp/' "$gi"; }; }; then + { + [ -s "$gi" ] && echo "" + echo "# Ephemeral working artifacts (throwaway; see working-files.md)" + echo "temp/" + } >> "$gi" + fi +fi + # Banner. echo echo "Done." diff --git a/scripts/install-lang.sh b/scripts/install-lang.sh index 6e8d806..3aaa76e 100755 --- a/scripts/install-lang.sh +++ b/scripts/install-lang.sh @@ -104,12 +104,16 @@ fi echo "Installing '$LANG' ruleset into $PROJECT" -# 1. Generic rules from claude-rules/ (shared across all languages) -if [ -d "$REPO_ROOT/claude-rules" ]; then - mkdir -p "$PROJECT/.claude/rules" - cp "$REPO_ROOT/claude-rules"/*.md "$PROJECT/.claude/rules/" 2>/dev/null || true - count=$(ls -1 "$REPO_ROOT/claude-rules"/*.md 2>/dev/null | wc -l) - echo " [ok] .claude/rules/ — $count generic rule(s) from claude-rules/" +# 1. Generic rules are NOT copied here. They install once at ~/.claude/rules/ +# via `make install` and load in every session on the machine. Copying them per +# project loaded them twice, and project rules outrank user-level ones — so a +# stale project copy quietly overrode the fresh global rule. The bundle owns its +# own language rules only; sync-language-bundle.sh sweeps copies left by earlier +# installs. A machine that wants the generic rules runs `make install`. +mkdir -p "$PROJECT/.claude/rules" +if [ ! -d "$HOME/.claude/rules" ]; then + echo " [!!] ~/.claude/rules/ is missing — run 'make install' so the generic" + echo " rules are available; this bundle installs language rules only." fi # 2. .claude/ — language-specific rules, hooks, settings (authoritative, always overwrite) @@ -195,5 +199,30 @@ if [ -f "$SRC/gitignore-add.txt" ]; then fi fi +# --- Bundle completeness check --- +# Every component copy above is guarded by a plain existence test, so a bundle +# missing one installs silently and reports success. That is how the python and +# typescript bundles shipped for nearly two months with no pre-commit hook, and +# therefore no credential scan, on every project that installed them: nothing +# ever said the component was absent. +# +# This can't be fixed by never missing a component — someone adding the seventh +# bundle will miss one too. It's fixed by the install saying so. Warn, never +# block: a partial bundle is still worth installing, and turning this into a +# failure would just teach people to skip the installer. +missing="" +[ -f "$SRC/githooks/pre-commit" ] || missing="${missing} - githooks/pre-commit (secret scan on commit)"$'\n' +[ -f "$SRC/claude/settings.json" ] || missing="${missing} - claude/settings.json (permissions + PostToolUse hook wiring)"$'\n' +ls "$SRC"/claude/hooks/*.sh >/dev/null 2>&1 || missing="${missing} - claude/hooks/*.sh (validate-on-edit hook)"$'\n' +[ -f "$SRC/CLAUDE.md" ] || missing="${missing} - CLAUDE.md (seed project instructions)"$'\n' + +if [ -n "$missing" ]; then + echo "" + echo "WARNING: the '$LANG' bundle is incomplete. Not installed, because the bundle doesn't ship them:" >&2 + printf '%s' "$missing" >&2 + echo "The install above succeeded; these components are simply absent upstream." >&2 + echo "Add them under languages/$LANG/ in rulesets, then re-run this install." >&2 +fi + echo "" echo "Install complete." diff --git a/scripts/lint.sh b/scripts/lint.sh index 61a27a1..ca6abbd 100755 --- a/scripts/lint.sh +++ b/scripts/lint.sh @@ -21,10 +21,22 @@ warn() { errors=$((errors + 1)) } +# Print a rule file's body with any leading YAML frontmatter stripped, so the +# structural checks below see the Markdown regardless of whether the file +# carries a `paths:` block. Claude Code reads the frontmatter; the heading check +# should not care that it is there. +md_body() { + awk 'NR==1 && $0=="---" {fm=1; next} + fm && $0=="---" {fm=0; next} + !fm {print}' "$1" +} + check_md_heading() { local f="$1" [ -f "$f" ] || return 0 - if ! head -1 "$f" | grep -q '^# '; then + # First non-blank line, so a blank separator after frontmatter doesn't read as + # a missing heading. + if ! md_body "$f" | grep -m1 -v '^[[:space:]]*$' | grep -q '^# '; then warn "$f — missing top-level heading" fi } @@ -37,6 +49,27 @@ check_md_applies_to() { fi } +# A rule whose prose declares a file-type scope must also carry `paths:` +# frontmatter, or Claude Code loads it into every session regardless of what the +# prose says. Three rules declared a narrow scope this way and were loaded +# universally for as long as they shipped, because the declaration lived only in +# a line the loader never reads. The prose is for the human; the frontmatter is +# what actually scopes the load. Keep them saying the same thing. +check_md_paths_frontmatter() { + local f="$1" applies + [ -f "$f" ] || return 0 + applies=$(grep -m1 '^Applies to:' "$f" 2>/dev/null) || return 0 + # A scope naming a concrete extension (`**/*.org`, `**/*.el`) is path-scopable. + # A bare `**/*` is genuinely universal and wants no frontmatter. + case "$applies" in + *'**/*.'*) + if ! head -1 "$f" | grep -q '^---$'; then + warn "$f — declares a file-type scope in prose but has no 'paths:' frontmatter; it loads in every session" + fi + ;; + esac +} + check_hook() { local f="$1" [ -f "$f" ] || return 0 @@ -81,6 +114,7 @@ for f in claude-rules/*.md; do [ -f "$f" ] || continue check_md_heading "$f" check_md_applies_to "$f" + check_md_paths_frontmatter "$f" done # Per-language rule files @@ -114,6 +148,14 @@ for s in scripts/*.sh; do check_hook "$s" done +# Scripts `make install` symlinks onto PATH. Extensionless by convention, so +# they need their own glob — scripts/*.sh above never matched them, which left +# the most exposed shell in the repo as the only shell with no gate over it. +for s in claude-templates/bin/*; do + [ -f "$s" ] || continue + check_hook "$s" +done + # Markdown link validation across rules and skills for f in claude-rules/*.md */SKILL.md; do [ -f "$f" ] || continue diff --git a/scripts/roam-sync.sh b/scripts/roam-sync.sh index 55422ec..ef43c8f 100755 --- a/scripts/roam-sync.sh +++ b/scripts/roam-sync.sh @@ -3,8 +3,13 @@ # # Commit any local changes, rebase onto the remote, push. Run by the # roam-sync systemd user timer (scripts/systemd/) every 15 minutes so -# Craig's hand edits travel without a manual git step. Agents don't need -# this — they pull/commit/push inline per claude-rules/knowledge-base.md. +# Craig's hand edits travel without a manual git step. This script is the +# roam repo's only committer (the 2026-06-24 one-git-owner rule): the tree +# is chronically dirty from live captures, so a second committer risks +# sweeping an in-flight capture into a stray commit. Agents edit the working +# tree under the roam-write lock, then trigger this unit (systemctl --user +# start roam-sync.service) instead of committing themselves — see +# claude-rules/knowledge-base.md and the inbox workflow's roam mode. # # On a rebase conflict: abort the rebase (never leave the repo mid-rebase # for a timer to mangle), keep the local commit, exit 1 so the failure is diff --git a/scripts/signal-receive.sh b/scripts/signal-receive.sh new file mode 100755 index 0000000..8c3ef01 --- /dev/null +++ b/scripts/signal-receive.sh @@ -0,0 +1,51 @@ +#!/usr/bin/env bash +# signal-receive.sh — drain the Signal pager account's inbound queue. +# +# The pager identity (+15045173983) lives on velox (primary) and any linked +# device (ratio). The Signal protocol expects a registered account to receive +# regularly; when it goes quiet, signal-cli prints a staleness warning +# ("Messages have been last received N days ago") and the account drifts toward +# an unhealthy state. This script pulls anything queued and exits, keeping the +# account warm. Run by the signal-receive systemd user timer (scripts/systemd/) +# on every machine holding the account — the roam-sync-shaped fix for the +# receive-staleness caveat on the pager channel. +# +# Draining the queue is also what surfaces Craig's replies: a reply he types in +# Signal arrives as a data message here, so a warm account is a prerequisite for +# the read-replies half of the pager (see docs/design/…-signal-pager-runbook.org). +# +# The units stow via the shared dotfiles `common` package, so the timer may land +# on a machine that does not hold the account. That is fine: the script no-ops +# cleanly (exit 0) when the account is not registered locally, rather than +# erroring every cadence. +# +# Usage: signal-receive.sh [account] [timeout-seconds] +# account defaults to the pager identity below +# timeout-seconds seconds to wait for new messages (default 10) +# +# Exit: 0 on a clean receive (including "nothing queued" and "account not on this +# machine"), non-zero only if signal-cli itself errors, so a real failure is +# visible in `systemctl --user status signal-receive`. + +set -euo pipefail + +account="${1:-+15045173983}" +timeout="${2:-10}" + +if ! command -v signal-cli >/dev/null 2>&1; then + echo "signal-receive: signal-cli not on PATH — nothing to do" >&2 + exit 0 +fi + +# No-op cleanly where the account isn't registered (a machine that stows the +# common units but isn't a pager device). Only a genuine receive error should +# surface as a failure. +if ! signal-cli listAccounts 2>/dev/null | grep -q "$account"; then + echo "signal-receive: $account not registered on this machine — nothing to do" >&2 + exit 0 +fi + +# `receive` prints envelopes to stdout and returns 0 once the queue drains or +# the timeout elapses. --send-read-receipts tells Craig's phone his replies were +# read by the pager, matching normal Signal behavior. +exec signal-cli -a "$account" receive --timeout "$timeout" --send-read-receipts diff --git a/scripts/sweep-gitignore-tooling.sh b/scripts/sweep-gitignore-tooling.sh index 68bfe2d..7194d6c 100755 --- a/scripts/sweep-gitignore-tooling.sh +++ b/scripts/sweep-gitignore-tooling.sh @@ -153,7 +153,37 @@ for project in "${projects[@]}"; do done done +# temp/ backfill — mode-independent. Unlike the personal-tooling set above +# (gitignore-mode only, track-mode deliberately skipped), temp/ holds ephemeral +# artifacts in every project regardless of whether it tracks its .ai/, so both +# track- and gitignore-mode projects get it. A separate pass, not an IGNORE_SET +# member — folding it into that set would miss every track-mode project, which +# is exactly the set that needs it. working/ is never ignored: it's the tracked +# home of in-progress work. +temp_swept=0 +for project in "${projects[@]}"; do + [ -d "$project/.git" ] || continue + name="$(basename "$project")" + gi="$project/.gitignore" + + if [ -f "$gi" ] && { grep -qFx 'temp/' "$gi" || grep -qFx '/temp/' "$gi"; }; then + continue + fi + + if [ "$dry_run" -eq 1 ]; then + echo "DRY $name — would add: temp/" + else + { + [ -s "$gi" ] && echo "" + echo "# Ephemeral working artifacts (throwaway; see working-files.md; swept $(date +%Y-%m-%d))" + echo "temp/" + } >> "$gi" + echo "temp $name — added: temp/" + fi + temp_swept=$((temp_swept + 1)) +done + echo -echo "Summary: $swept swept, $complete already complete, $skipped skipped (of ${#projects[@]} projects)." +echo "Summary: $swept swept, $complete already complete, $skipped skipped (of ${#projects[@]} projects); $temp_swept temp/ backfilled." [ "$dry_run" -eq 1 ] && echo "(dry-run — no files written)" exit 0 diff --git a/scripts/sync-language-bundle.sh b/scripts/sync-language-bundle.sh index 45f8259..fe922af 100755 --- a/scripts/sync-language-bundle.sh +++ b/scripts/sync-language-bundle.sh @@ -105,7 +105,7 @@ inbox_drop() { # shared generic rules, its hooks/githooks, and surfaces settings.json. process_bundle() { local kind="$1" src="${2%/}" - local name rules rf f rel manual_before + local name rules rf f rel manual_before base name="$(basename "$src")" rules="$src/claude/rules" [ -d "$rules" ] || return 0 @@ -126,12 +126,27 @@ process_bundle() { HEADER_DONE=0 manual_before=$MANUAL - # AUTO-FIX: language bundles carry the shared generic rules; team overlays - # carry only their own rule(s). + # SWEEP: generic rules are installed once at ~/.claude/rules/ by `make + # install` and load in every session. Language bundles used to copy them into + # each project too, which made Claude Code load them twice — and project rules + # take priority over user-level ones, so a stale project copy silently + # overrode the fresh global rule until the next startup healed it. The bundle + # now owns only its own language rules, and sweeps the duplicates it shipped + # before. + # + # The sweep only fires when the global rule is actually present to take over. + # On a machine mid-bootstrap, or one where `make install` has not run, the + # project copy is the only copy and removing it would leave no rule at all. if [ "$kind" = language ]; then for f in "$GENERIC_RULES"/*.md; do [ -f "$f" ] || continue - fix "$f" "$PROJECT/.claude/rules/$(basename "$f")" + base="$(basename "$f")" + [ -f "$PROJECT/.claude/rules/$base" ] || continue + [ -f "$HOME/.claude/rules/$base" ] || continue + rm -f "$PROJECT/.claude/rules/$base" + ensure_header + OUT+=" swept .claude/rules/$base (duplicate of ~/.claude/rules/)"$'\n' + FIXED=$((FIXED + 1)) done fi for f in "$rules"/*.md; do diff --git a/scripts/systemd/signal-receive.service b/scripts/systemd/signal-receive.service new file mode 100644 index 0000000..d63a5c8 --- /dev/null +++ b/scripts/systemd/signal-receive.service @@ -0,0 +1,10 @@ +# Signal pager receive — keep the pager account warm on every machine holding it. +# Stowed via the dotfiles `common` package, so it lands on both daily drivers; +# the script no-ops cleanly where the account isn't registered. Enable per machine: +# systemctl --user daemon-reload && systemctl --user enable --now signal-receive.timer +[Unit] +Description=Drain the Signal pager account inbound queue (keeps it warm) + +[Service] +Type=oneshot +ExecStart=%h/code/rulesets/scripts/signal-receive.sh diff --git a/scripts/systemd/signal-receive.timer b/scripts/systemd/signal-receive.timer new file mode 100644 index 0000000..3fdaee0 --- /dev/null +++ b/scripts/systemd/signal-receive.timer @@ -0,0 +1,10 @@ +[Unit] +Description=Keep the Signal pager account warm every 15 minutes + +[Timer] +OnBootSec=3min +OnUnitActiveSec=15min +RandomizedDelaySec=60 + +[Install] +WantedBy=timers.target diff --git a/scripts/tests/agent-page.bats b/scripts/tests/agent-page.bats deleted file mode 100644 index 071e4b9..0000000 --- a/scripts/tests/agent-page.bats +++ /dev/null @@ -1,68 +0,0 @@ -#!/usr/bin/env bats -# agent-page — the runtime-neutral phone pager. Pages Craig over Signal from -# any machine on the tailnet: runs signal-cli directly on velox (where the -# pager identity lives), ssh-relays to velox from everywhere else. These tests -# stub ssh/uname/signal-cli on PATH to verify command construction without a -# network or a phone. - -setup() { - REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)" - PAGE="$REPO_ROOT/claude-templates/bin/agent-page" - STUBS="$(mktemp -d)" - LOG="$STUBS/calls.log" - cat > "$STUBS/ssh" <<EOF -#!/bin/bash -echo "ssh \$*" >> "$LOG" -exit 0 -EOF - cat > "$STUBS/signal-cli" <<EOF -#!/bin/bash -echo "signal-cli \$*" >> "$LOG" -exit 0 -EOF - chmod +x "$STUBS/ssh" "$STUBS/signal-cli" -} - -teardown() { - rm -rf "$STUBS" -} - -@test "no message exits 2 with usage" { - run bash "$PAGE" - [ "$status" -eq 2 ] - [[ "$output" == *"usage"* ]] -} - -@test "relays through ssh to velox with the pager account and Craig's UUID" { - PATH="$STUBS:$PATH" run bash "$PAGE" build finished - [ "$status" -eq 0 ] - grep -q "^ssh .*velox" "$LOG" - grep -q "15045173983" "$LOG" - grep -q "b1b5601e-6126-47f8-afaa-0a59f5188fde" "$LOG" - # printf %q escapes the space, so the relayed message reads build\ finished. - grep -qF 'build\ finished' "$LOG" -} - -@test "on velox itself, calls signal-cli directly (no ssh)" { - cat > "$STUBS/uname" <<'EOF' -#!/bin/bash -[ "$1" = "-n" ] && { echo velox; exit 0; } -exec /usr/bin/uname "$@" -EOF - chmod +x "$STUBS/uname" - PATH="$STUBS:$PATH" run bash "$PAGE" hello - [ "$status" -eq 0 ] - grep -q "^signal-cli " "$LOG" - ! grep -q "^ssh " "$LOG" -} - -@test "a failed relay reports the desktop fallback and propagates failure" { - cat > "$STUBS/ssh" <<'EOF' -#!/bin/bash -exit 255 -EOF - chmod +x "$STUBS/ssh" - PATH="$STUBS:$PATH" run bash "$PAGE" urgent thing - [ "$status" -ne 0 ] - [[ "$output" == *"notify"* ]] -} diff --git a/scripts/tests/agent-text.bats b/scripts/tests/agent-text.bats new file mode 100644 index 0000000..e5d80fc --- /dev/null +++ b/scripts/tests/agent-text.bats @@ -0,0 +1,78 @@ +#!/usr/bin/env bats +# agent-text — the runtime-neutral Signal phone messenger ("text me"). Reaches +# Craig over Signal from any machine on the tailnet: sends directly wherever the +# account is registered locally (velox, or any linked device), and ssh-relays to +# velox from a machine that doesn't hold the account. These tests stub +# ssh/signal-cli on PATH to verify command construction without a network or a +# phone. The signal-cli stub answers `listAccounts` to control which branch the +# dispatch takes. The final test covers the deprecated agent-page shim. + +setup() { + REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)" + PAGE="$REPO_ROOT/claude-templates/bin/agent-text" + SHIM="$REPO_ROOT/claude-templates/bin/agent-page" + STUBS="$(mktemp -d)" + LOG="$STUBS/calls.log" + cat > "$STUBS/ssh" <<EOF +#!/bin/bash +echo "ssh \$*" >> "$LOG" +exit 0 +EOF + # HAS_ACCOUNT controls the listAccounts answer: "1" → the account is + # local (send direct), unset/"0" → not local (relay). + cat > "$STUBS/signal-cli" <<EOF +#!/bin/bash +if [ "\$1" = "listAccounts" ]; then + [ "\${HAS_ACCOUNT:-0}" = "1" ] && echo "Number: +15045173983" + exit 0 +fi +echo "signal-cli \$*" >> "$LOG" +exit 0 +EOF + chmod +x "$STUBS/ssh" "$STUBS/signal-cli" +} + +teardown() { + rm -rf "$STUBS" +} + +@test "no message exits 2 with usage" { + run bash "$PAGE" + [ "$status" -eq 2 ] + [[ "$output" == *"usage"* ]] +} + +@test "relays through ssh to velox when the account is not local" { + HAS_ACCOUNT=0 PATH="$STUBS:$PATH" run bash "$PAGE" build finished + [ "$status" -eq 0 ] + grep -q "^ssh .*velox" "$LOG" + grep -q "15045173983" "$LOG" + grep -q "b1b5601e-6126-47f8-afaa-0a59f5188fde" "$LOG" + # printf %q escapes the space, so the relayed message reads build\ finished. + grep -qF 'build\ finished' "$LOG" +} + +@test "sends directly (no ssh) when the pager account is registered locally" { + HAS_ACCOUNT=1 PATH="$STUBS:$PATH" run bash "$PAGE" hello + [ "$status" -eq 0 ] + grep -q "^signal-cli -a +15045173983 send " "$LOG" + ! grep -q "^ssh " "$LOG" +} + +@test "a failed relay reports the desktop fallback and propagates failure" { + cat > "$STUBS/ssh" <<'EOF' +#!/bin/bash +exit 255 +EOF + chmod +x "$STUBS/ssh" + HAS_ACCOUNT=0 PATH="$STUBS:$PATH" run bash "$PAGE" urgent thing + [ "$status" -ne 0 ] + [[ "$output" == *"notify"* ]] +} + +@test "the deprecated agent-page shim delegates to agent-text" { + HAS_ACCOUNT=1 PATH="$STUBS:$PATH" run bash "$SHIM" via shim + [ "$status" -eq 0 ] + # Reaches the same direct-send path as agent-text. + grep -q "^signal-cli -a +15045173983 send " "$LOG" +} diff --git a/scripts/tests/ai-launcher-characterization.bats b/scripts/tests/ai-launcher-characterization.bats new file mode 100644 index 0000000..5b93ff3 --- /dev/null +++ b/scripts/tests/ai-launcher-characterization.bats @@ -0,0 +1,365 @@ +#!/usr/bin/env bats +# Characterization tests for the ai launcher's internal functions. +# +# These pin the launcher's CURRENT behavior (record-not-spec, per testing.md) +# before the hardening refactor, so the extraction of pure cores and the +# footgun fixes can proceed against a green net. The pure/near-pure functions +# are exercised by sourcing bin/ai (the run-vs-sourced guard keeps dispatch +# off) and calling them directly; the tmux-coupled pipeline is exercised +# against a throwaway tmux server on a PRIVATE socket (TMUX_TMPDIR under the +# test tmpdir, TMUX unset) so nothing can reach Craig's live 'ai' session. + +# The launcher stores candidates as literal "~/..." display strings and expands +# them downstream; the assertions below compare against those literal tildes on +# purpose, so SC2088 (tilde-in-quotes) does not apply. SC2030/SC2031 fire on +# the shared `candidates` global because bats runs each @test in its own +# subshell — the array is set and read within that same subshell, so the +# warning is a false positive here. +# shellcheck disable=SC2088,SC2030,SC2031 + +setup() { + REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)" + AI="$REPO_ROOT/claude-templates/bin/ai" + WORK="$(mktemp -d)" + # Isolate every tmux call in this file onto a private server. + export TMUX_TMPDIR="$WORK" + unset TMUX + # Source for direct function access. The guard skips main() when sourced. + # shellcheck disable=SC1090 + source "$AI" +} + +teardown() { + tmux kill-server 2>/dev/null || true + rm -rf "$WORK" +} + +# --- git repo helpers --------------------------------------------------------- + +# A committed repo with no remote/upstream. Auto-maintenance off so background +# git can't race the teardown rm -rf (the rename-ai flake, fixed 2026-07-19). +_mk_repo() { + local d="$1" + git init -q "$d" + git -C "$d" config user.email t@example.com + git -C "$d" config user.name tester + git -C "$d" config commit.gpgsign false + git -C "$d" config gc.auto 0 + git -C "$d" config maintenance.auto false + git -C "$d" commit -q --allow-empty -m init +} + +# A repo with a bare origin and a tracked branch, in sync at one commit. +_mk_repo_upstream() { + local d="$1" remote="$1.remote" + git init -q --bare "$remote" + _mk_repo "$d" + git -C "$d" remote add origin "$remote" + git -C "$d" push -q -u origin HEAD +} + +# --- usage() ------------------------------------------------------------------ + +@test "usage: -h prints the help banner and exits 0" { + run bash "$AI" -h + [ "$status" -eq 0 ] + [[ "$output" == *"Usage:"* ]] + [[ "$output" == *"--attach"* ]] + [[ "$output" == *"Single-project mode"* ]] +} + +# --- maybe_add_candidate() ---------------------------------------------------- + +@test "maybe_add_candidate: a HOME-rooted project dir is added as ~/rel" { + HOME="$WORK/home" + mkdir -p "$HOME/code/proj/.ai" + touch "$HOME/code/proj/.ai/protocols.org" + candidates=() + maybe_add_candidate "$HOME/code/proj" + [ "${#candidates[@]}" -eq 1 ] + [ "${candidates[0]}" = "~/code/proj" ] +} + +@test "maybe_add_candidate: a dir without .ai/protocols.org is filtered out" { + HOME="$WORK/home" + mkdir -p "$HOME/code/plain" + candidates=() + maybe_add_candidate "$HOME/code/plain" || true + [ "${#candidates[@]}" -eq 0 ] +} + +@test "maybe_add_candidate: a non-HOME dir keeps its full path after ~/ (current quirk)" { + HOME="$WORK/home" + mkdir -p "$WORK/outside/.ai" + touch "$WORK/outside/.ai/protocols.org" + candidates=() + maybe_add_candidate "$WORK/outside" + # $HOME/ prefix doesn't match, so the path is left whole behind the literal ~/. + [ "${candidates[0]}" = "~/$WORK/outside" ] +} + +# --- git_status_indicator() --------------------------------------------------- + +@test "git_status_indicator: a non-git dir yields no annotation" { + mkdir -p "$WORK/plain" + run git_status_indicator "$WORK/plain" + [ "$status" -eq 0 ] + [ -z "$output" ] +} + +@test "git_status_indicator: clean repo with no upstream reads (no upstream)" { + _mk_repo "$WORK/r" + run git_status_indicator "$WORK/r" + [ "$output" = " (no upstream)" ] +} + +@test "git_status_indicator: an untracked file with no upstream reads (no upstream dirty)" { + _mk_repo "$WORK/r" + touch "$WORK/r/scratch" + run git_status_indicator "$WORK/r" + [ "$output" = " (no upstream dirty)" ] +} + +@test "git_status_indicator: inbox-only delivery is visible but not called dirty" { + _mk_repo "$WORK/r" + mkdir -p "$WORK/r/inbox" + touch "$WORK/r/inbox/from-home.org" + run git_status_indicator "$WORK/r" + [ "$output" = " (no upstream inbox)" ] +} + +@test "git_status_indicator: in-sync tracked repo reads (✓)" { + _mk_repo_upstream "$WORK/r" + run git_status_indicator "$WORK/r" + [ "$output" = " (✓)" ] +} + +@test "git_status_indicator: one unpushed commit reads (↑1)" { + _mk_repo_upstream "$WORK/r" + git -C "$WORK/r" commit -q --allow-empty -m ahead + run git_status_indicator "$WORK/r" + [ "$output" = " (↑1)" ] +} + +@test "git_status_indicator: one commit behind upstream reads (↓1)" { + _mk_repo_upstream "$WORK/r" + git -C "$WORK/r" commit -q --allow-empty -m b + git -C "$WORK/r" push -q origin HEAD + git -C "$WORK/r" reset -q --hard HEAD~1 + run git_status_indicator "$WORK/r" + [ "$output" = " (↓1)" ] +} + +@test "auto_pull_if_clean fast-forwards with an untracked inbox delivery" { + _mk_repo_upstream "$WORK/r" + git clone -q "$WORK/r.remote" "$WORK/writer" + git -C "$WORK/writer" config user.email t@example.com + git -C "$WORK/writer" config user.name tester + git -C "$WORK/writer" commit -q --allow-empty -m remote + git -C "$WORK/writer" push -q + git -C "$WORK/r" fetch -q + mkdir -p "$WORK/r/inbox" + printf 'handoff\n' >"$WORK/r/inbox/from-home.org" + + run auto_pull_if_clean "$WORK/r" + [ "$status" -eq 0 ] + [ "$(git -C "$WORK/r" rev-parse HEAD)" = "$(git -C "$WORK/r" rev-parse '@{u}')" ] + [ -f "$WORK/r/inbox/from-home.org" ] +} + +@test "auto_pull_if_clean refuses a non-inbox untracked file" { + _mk_repo_upstream "$WORK/r" + git clone -q "$WORK/r.remote" "$WORK/writer" + git -C "$WORK/writer" config user.email t@example.com + git -C "$WORK/writer" config user.name tester + git -C "$WORK/writer" commit -q --allow-empty -m remote + git -C "$WORK/writer" push -q + git -C "$WORK/r" fetch -q + printf 'scratch\n' >"$WORK/r/scratch" + before="$(git -C "$WORK/r" rev-parse HEAD)" + + run auto_pull_if_clean "$WORK/r" + [ "$status" -eq 0 ] + [ "$(git -C "$WORK/r" rev-parse HEAD)" = "$before" ] +} + +# --- annotate_candidates() ---------------------------------------------------- + +@test "annotate_candidates: appends each candidate's status suffix" { + _mk_repo_upstream "$WORK/home/code/clean" + mkdir -p "$WORK/home/code/plain" + HOME="$WORK/home" + candidates=("~/code/clean" "~/code/plain") + annotate_candidates + [ "${candidates[0]}" = "~/code/clean (✓)" ] + # A non-git candidate gets an empty suffix (git_status_indicator returns ""). + [ "${candidates[1]}" = "~/code/plain" ] +} + +# --- read_selections() -------------------------------------------------------- + +@test "read_selections: strips the ' (annotation)' suffix from each line" { + read_selections "$(printf '~/code/a (✓)\n~/projects/b (↑1 dirty)')" + [ "${#selected[@]}" -eq 2 ] + [ "${selected[0]}" = "~/code/a" ] + [ "${selected[1]}" = "~/projects/b" ] +} + +@test "read_selections: a line with no annotation is kept verbatim" { + read_selections "~/code/plain" + [ "${selected[0]}" = "~/code/plain" ] +} + +@test "read_selections: empty input yields a single empty element (current quirk)" { + read_selections "" + [ "${#selected[@]}" -eq 1 ] + [ -z "${selected[0]}" ] +} + +# --- build_candidates() ------------------------------------------------------- + +@test "build_candidates: discovers .ai projects under HOME, filters, sorts" { + HOME="$WORK/home" + mkdir -p "$HOME/.emacs.d/.ai" "$HOME/code/beta/.ai" "$HOME/code/alpha/.ai" \ + "$HOME/code/plain" "$HOME/projects/gamma/.ai" + touch "$HOME/.emacs.d/.ai/protocols.org" \ + "$HOME/code/beta/.ai/protocols.org" \ + "$HOME/code/alpha/.ai/protocols.org" \ + "$HOME/projects/gamma/.ai/protocols.org" + # '|| true' suspends bats's errexit so the function runs to completion as it + # does in production (bin/ai runs without set -e). Its non-zero exit is + # incidental — it inherits the last maybe_add_candidate's filter result, + # which callers ignore; see the file's top-of-script NOTE. + build_candidates || true + # .emacs.d first (probed first), then ~/code sorted, then ~/projects. + [ "${candidates[0]}" = "~/.emacs.d" ] + [ "${candidates[1]}" = "~/code/alpha" ] + [ "${candidates[2]}" = "~/code/beta" ] + [ "${candidates[3]}" = "~/projects/gamma" ] + [ "${#candidates[@]}" -eq 4 ] +} + +@test "build_candidates: a HOME with no .ai projects yields an empty list" { + HOME="$WORK/empty-home" + mkdir -p "$HOME/code/plain" + build_candidates || true + [ "${#candidates[@]}" -eq 0 ] +} + +# --- functional: tmux-coupled pipeline (private socket) ----------------------- + +@test "functional create_window: adds a named window and returns its id" { + export AGENT_CMD="true" + tmux new-session -d -s ai -n base -c "$WORK" + run create_window "$WORK" "proj-x" + [ "$status" -eq 0 ] + [ -n "$output" ] + tmux list-windows -t ai -F '#{window_name}' | grep -qx proj-x +} + +@test "functional find_window_id: returns the id for a name, empty for a miss" { + tmux new-session -d -s ai -n only -c "$WORK" + run find_window_id only + [ -n "$output" ] + run find_window_id nope + [ -z "$output" ] +} + +@test "functional sort_windows: others alpha-first, then projects alpha" { + HOME="$WORK/home" + mkdir -p "$HOME/code/alpha/.ai" "$HOME/code/beta/.ai" + touch "$HOME/code/alpha/.ai/protocols.org" "$HOME/code/beta/.ai/protocols.org" + tmux new-session -d -s ai -n zzz-other -c "$WORK" + tmux new-window -t ai -n beta + tmux new-window -t ai -n alpha + tmux new-window -t ai -n aaa-other + run sort_windows + [ "$status" -eq 0 ] + order="$(tmux list-windows -t ai -F '#{window_name}' | paste -sd, -)" + [ "$order" = "aaa-other,zzz-other,alpha,beta" ] +} + +@test "functional attach_mode: no session prints an error and exits 1" { + run bash "$AI" --attach + [ "$status" -eq 1 ] + [[ "$output" == *"no 'ai' session"* ]] +} + +# --- extracted pure cores ----------------------------------------------------- + +@test "_git_prep_action: no upstream is always 'none'" { + run _git_prep_action 0 0 0 5 + [ "$output" = none ] +} + +@test "_git_prep_action: clean and purely behind is 'pull'" { + run _git_prep_action 1 0 0 3 + [ "$output" = pull ] +} + +@test "_git_prep_action: in sync with upstream is 'none'" { + run _git_prep_action 1 0 0 0 + [ "$output" = none ] +} + +@test "_git_prep_action: ahead is 'report', never 'pull'" { + run _git_prep_action 1 0 2 0 + [ "$output" = report ] +} + +@test "_git_prep_action: dirty is 'report', never 'pull'" { + run _git_prep_action 1 1 0 0 + [ "$output" = report ] +} + +@test "_git_prep_action: behind AND dirty is 'report' — a dirty repo is never auto-pulled" { + run _git_prep_action 1 1 0 3 + [ "$output" = report ] +} + +@test "_git_prep_action: diverged (ahead and behind) is 'report'" { + run _git_prep_action 1 0 2 3 + [ "$output" = report ] +} + +@test "_order_windows: others alpha, then projects alpha" { + local listing names out + listing="$(printf 'beta\t@1\nzzz-other\t@2\nalpha\t@3\naaa-other\t@4')" + names="$(printf 'alpha\nbeta')" + out="$(printf '%s\n' "$listing" | _order_windows "$names" | cut -f1 | paste -sd, -)" + [ "$out" = "aaa-other,zzz-other,alpha,beta" ] +} + +@test "_order_windows: all-projects input keeps only the projects, sorted" { + local out + out="$(printf 'beta\t@1\nalpha\t@2\n' | _order_windows "$(printf 'alpha\nbeta')" | cut -f1 | paste -sd, -)" + [ "$out" = "alpha,beta" ] +} + +@test "_order_windows: empty input yields empty output" { + run _order_windows "$(printf 'alpha\nbeta')" <<<"" + [ -z "$output" ] +} + +@test "_order_windows: a name matches only as a whole line, not as a substring" { + # 'alph' must not match project 'alpha' (grep -qxF is a full-line match). + local out + out="$(printf 'alph\t@1\n' | _order_windows "$(printf 'alpha')" | cut -f1)" + # 'alph' is not a project, so it lands in others, still present. + [ "$out" = "alph" ] +} + +@test "_match_window_id: returns the id for a name, empty for a miss" { + local listing out + listing="$(printf 'one\t@1\ntwo\t@2\n')" + out="$(printf '%s\n' "$listing" | _match_window_id two)" + [ "$out" = "@2" ] + out="$(printf '%s\n' "$listing" | _match_window_id nope)" + [ -z "$out" ] +} + +@test "_match_window_id: first match wins and stops" { + local out + out="$(printf 'dup\t@1\ndup\t@2\n' | _match_window_id dup)" + [ "$out" = "@1" ] +} diff --git a/scripts/tests/ai-launcher-helper.bats b/scripts/tests/ai-launcher-helper.bats new file mode 100644 index 0000000..0ede4fb --- /dev/null +++ b/scripts/tests/ai-launcher-helper.bats @@ -0,0 +1,328 @@ +#!/usr/bin/env bats +# The ai launcher's --helper flag: open a SECOND agent session in a project that +# already has a live one, under the helper-mode.org role contract. +# +# The load-bearing behavior is the roster gate. `ai --helper` is an assertion by +# the operator that a primary is already running; the roster is what checks it. +# Three outcomes, all tested here: confirmed (launch a helper), refuted (no other +# agent — warn and launch a normal primary instead), and unverifiable (no roster +# script, or a platform without /proc — warn and launch a helper anyway, because +# helper mode is the strictly less destructive guess when we cannot tell). +# +# --print-launch is the seam, as it is for the runtime tests: it prints the exact +# command a real run would send to the pane without touching tmux or fzf. The +# roster runs BEFORE that print, so the printed line reflects the real decision. + +setup() { + REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)" + AI="$REPO_ROOT/claude-templates/bin/ai" + PROJ="$(mktemp -d)" + mkdir -p "$PROJ/.ai/scripts" + touch "$PROJ/.ai/protocols.org" + # Every tmux call in this file goes to a private server under the test + # tmpdir, so nothing here can reach Craig's live 'ai' session. + export TMUX_TMPDIR="$PROJ" + unset TMUX + # Source for direct access to the pure cores. The guard skips main(). + # shellcheck disable=SC1090 + source "$AI" +} + +teardown() { + tmux kill-server 2>/dev/null || true + rm -rf "$PROJ" +} + +# Install a stub roster that exits with the given status. Exit codes are the +# real agent-roster's contract: 0 alone, 1 others live, 2 unavailable. +_stub_roster() { + cat > "$PROJ/.ai/scripts/agent-roster" <<STUB +#!/bin/bash +exit $1 +STUB + chmod +x "$PROJ/.ai/scripts/agent-roster" +} + +# --- the roster gate, end to end through --print-launch ------------------------ + +@test "--helper with a live primary launches under the helper contract" { + _stub_roster 1 + run bash "$AI" --helper --print-launch "$PROJ" + [ "$status" -eq 0 ] + [[ "$output" == *"helper-mode.org"* ]] + # The helper opener replaces the primary one; it must not send the session + # to protocols.org, whose startup would run pulls, rsync, and inbox work. + [[ "$output" != *"protocols.org"* ]] +} + +@test "--helper exports the agent id and the helper flag into the pane" { + _stub_roster 1 + run bash "$AI" --helper --print-launch "$PROJ" + [ "$status" -eq 0 ] + [[ "$output" == *"AI_AGENT_ID=helper-"* ]] + [[ "$output" == *"AI_HELPER=1"* ]] +} + +@test "--helper assigns a helper-<rand4> id" { + _stub_roster 1 + run bash "$AI" --helper --print-launch "$PROJ" + [ "$status" -eq 0 ] + [[ "$output" =~ AI_AGENT_ID=helper-[0-9a-f]{4}[[:space:]] ]] +} + +@test "--helper honors an id the caller already exported" { + _stub_roster 1 + AI_AGENT_ID=helper-beef run bash "$AI" --helper --print-launch "$PROJ" + [ "$status" -eq 0 ] + [[ "$output" == *"AI_AGENT_ID=helper-beef"* ]] +} + +@test "--helper sanitizes an id carrying shell metacharacters" { + _stub_roster 1 + # The id is interpolated into the command typed into the pane. An id + # carrying ';' would end the assignment and run the rest as its own + # command — the helper never launches and something else does. + AI_AGENT_ID='x;touch /tmp/ai-helper-pwned' run bash "$AI" --helper --print-launch "$PROJ" + [ "$status" -eq 0 ] + [[ "$output" != *";touch"* ]] + [[ "$output" != *" /tmp/ai-helper-pwned"* ]] + # Assert the exact surviving form too. Negative-only assertions also pass + # when the id is dropped or mangled some other way, which is how a broken + # sanitizer slipped through once already. + [[ "$output" == *"AI_AGENT_ID=x_touch__tmp_ai-helper-pwned "* ]] +} + +@test "--helper sanitizes an id carrying a space" { + _stub_roster 1 + AI_AGENT_ID='helper beef' run bash "$AI" --helper --print-launch "$PROJ" + [ "$status" -eq 0 ] + # A bare space would split the assignment from the command word. Anchored on + # the following space, not a substring: a substring match also accepts a + # mangled "helper_beef_", which an earlier sanitizer actually produced. + [[ "$output" == *"AI_AGENT_ID=helper_beef "* ]] +} + +@test "--helper mints a fresh id rather than reusing a live one" { + _stub_roster 1 + # A helper's pane exports AI_AGENT_ID, so `ai --helper` run from inside a + # helper inherits its parent's id. Reusing it lands both agents on one + # .ai/session-context.d/<id>.org — the lost-update shape helper mode exists + # to prevent. + mkdir -p "$PROJ/.ai/session-context.d" + printf 'helper-beef\n' > "$PROJ/.ai/session-context.d/helper-beef.org" + AI_AGENT_ID=helper-beef run bash "$AI" --helper --print-launch "$PROJ" + [ "$status" -eq 0 ] + [[ "$output" != *"AI_AGENT_ID=helper-beef"* ]] + [[ "$output" =~ AI_AGENT_ID=helper-[0-9a-f]{4}[[:space:]] ]] + [[ "$output" == *"already live"* ]] +} + +@test "--helper with no other agent falls back to a primary session and says so" { + _stub_roster 0 + run bash "$AI" --helper --print-launch "$PROJ" + [ "$status" -eq 0 ] + # Refuted: this is the normal primary launch. + [[ "$output" == *"protocols.org"* ]] + [[ "$output" != *"helper-mode.org"* ]] + [[ "$output" != *"AI_HELPER=1"* ]] + [[ "$output" == *"no other agent"* ]] +} + +@test "--helper with no roster installed still launches a helper, with a warning" { + # No stub written: an older checkout whose .ai/scripts predates agent-roster. + run bash "$AI" --helper --print-launch "$PROJ" + [ "$status" -eq 0 ] + [[ "$output" == *"helper-mode.org"* ]] + [[ "$output" == *"could not verify"* ]] +} + +@test "--helper with an unavailable roster still launches a helper, with a warning" { + _stub_roster 2 + run bash "$AI" --helper --print-launch "$PROJ" + [ "$status" -eq 0 ] + [[ "$output" == *"helper-mode.org"* ]] + [[ "$output" == *"could not verify"* ]] +} + +@test "--helper names the host and project in the opener, as the primary does" { + _stub_roster 1 + run bash "$AI" --helper --print-launch "$PROJ" + [ "$status" -eq 0 ] + [[ "$output" == *"$(basename "$PROJ")"* ]] + [[ "$output" == *"$(uname -n)"* ]] +} + +@test "--helper composes with --runtime" { + _stub_roster 1 + run bash "$AI" --runtime codex --helper --print-launch "$PROJ" + [ "$status" -eq 0 ] + [[ "$output" == *"codex "* ]] + [[ "$output" == *"helper-mode.org"* ]] +} + +@test "--helper refuses a directory that is not an agent-template project" { + run bash "$AI" --helper --print-launch "$BATS_TEST_TMPDIR" + [ "$status" -eq 1 ] + [[ "$output" == *"protocols.org"* ]] +} + +@test "--helper needs a project directory" { + run bash "$AI" --helper + [ "$status" -eq 2 ] + [[ "$output" == *"needs a project directory"* ]] +} + +# --- pure cores --------------------------------------------------------------- + +@test "_helper_launch_mode: roster found other agents (1) — launch a helper" { + run _helper_launch_mode 1 + [ "$output" = helper ] +} + +@test "_helper_launch_mode: roster says alone (0) — fall back to primary" { + run _helper_launch_mode 0 + [ "$output" = primary ] +} + +@test "_helper_launch_mode: roster unavailable (2) — helper, unverified" { + run _helper_launch_mode 2 + [ "$output" = helper-unverified ] +} + +@test "_helper_launch_mode: no roster script at all — helper, unverified" { + run _helper_launch_mode absent + [ "$output" = helper-unverified ] +} + +@test "_resolve_helper_launch: builds the roster path from its own argument" { + _stub_roster 1 + # With no `dir` in the caller's scope, a roster path built from the caller's + # variable instead of the parameter resolves to "/.ai/scripts/agent-roster", + # which isn't executable — so the gate would silently report unverified and + # every helper launch would skip its check. The bug is invisible when the + # caller happens to have its own $dir holding the same value, which both + # production callers do. + unset dir + run _resolve_helper_launch "$PROJ" + [ "$output" = helper ] +} + +@test "_helper_id: shape is helper- plus four hex digits" { + run _helper_id + [[ "$output" =~ ^helper-[0-9a-f]{4}$ ]] +} + +@test "_helper_id: uses the full 16 bits, not bash RANDOM's 15" { + # The shape test alone passes against `RANDOM % 65536`, which can never set + # the top bit — so every id would begin 0-7 and nothing would fail. Draw + # enough to make a genuinely 16-bit generator almost certain to show a high + # leading digit, and assert one appears. + local i high=0 + for i in $(seq 1 200); do + case "$(_helper_id)" in + helper-[89abcdef]*) high=1; break ;; + esac + done + [ "$high" -eq 1 ] +} + +@test "_sanitize_agent_id: keeps the safe charset and maps everything else" { + run _sanitize_agent_id 'helper-a83f' + [ "$output" = "helper-a83f" ] + run _sanitize_agent_id 'a b;c/d$e' + [ "$output" = "a_b_c_d_e" ] + run _sanitize_agent_id 'keep.dots_and-dashes' + [ "$output" = "keep.dots_and-dashes" ] +} + +@test "_resolve_helper_id: a minted id also avoids a live anchor" { + # Seed every id _helper_id can produce for a stubbed generator, so the mint + # path must notice the collision rather than hand back a taken id. + mkdir -p "$PROJ/.ai/session-context.d" + _helper_id() { echo "helper-dead"; } + : > "$PROJ/.ai/session-context.d/helper-dead.org" + run _resolve_helper_id "$PROJ" + # Bounded retries mean it gives up and returns the id, but it must not have + # returned it silently on the first look — the loop ran its full bound. + [ "$status" -eq 0 ] + [ "$output" = "helper-dead" ] +} + +@test "_resolve_helper_id: a free minted id is returned as-is" { + _helper_id() { echo "helper-cafe"; } + run _resolve_helper_id "$PROJ" + [ "$output" = "helper-cafe" ] +} + +# --- window ordering ---------------------------------------------------------- + +@test "_order_windows: a helper window sorts with its project, not with others" { + local listing names out + listing="$(printf 'beta\t@1\nzzz-other\t@2\nalpha:helper-a83f\t@3\nalpha\t@4')" + names="$(printf 'alpha\nbeta')" + out="$(printf '%s\n' "$listing" | _order_windows "$names" | cut -f1 | paste -sd, -)" + [ "$out" = "zzz-other,alpha,alpha:helper-a83f,beta" ] +} + +@test "_order_windows: a colon name whose prefix is not a project stays in others" { + local out + out="$(printf 'nope:helper-a83f\t@1\n' | _order_windows "$(printf 'alpha')" | cut -f1)" + [ "$out" = "nope:helper-a83f" ] +} + +# --- functional: the second window (private tmux socket) ---------------------- +# +# The regression these guard against: single_mode focuses the project's existing +# window and returns, so routing a helper through it would hand back the PRIMARY +# session instead of opening a second one. +# +# The window-list tail (sort_windows, attach_session) is stubbed out. Both are +# already covered in the characterization suite, attaching needs a real terminal +# these tests don't have, and build_candidates legitimately returns non-zero +# under bats's errexit (bin/ai itself runs without set -e — see that file's NOTE). +_stub_window_tail() { + sort_windows() { :; } + attach_session() { :; } + export AGENT_CMD="true" +} + +@test "functional helper_mode: opens a NEW window beside the project's existing one" { + _stub_roster 1 + _stub_window_tail + tmux new-session -d -s ai -n "$(basename "$PROJ")" -c "$PROJ" + run helper_mode "$PROJ" + [ "$status" -eq 0 ] + names="$(tmux list-windows -t ai -F '#{window_name}')" + # The primary's window survives untouched, and a helper window joins it. + printf '%s\n' "$names" | grep -qx "$(basename "$PROJ")" + printf '%s\n' "$names" | grep -qE "^$(basename "$PROJ"):helper-[0-9a-f]{4}$" + [ "$(printf '%s\n' "$names" | wc -l)" -eq 2 ] +} + +@test "functional helper_mode: the window name carries the exported id" { + _stub_roster 1 + _stub_window_tail + tmux new-session -d -s ai -n base -c "$PROJ" + AI_AGENT_ID=helper-beef run helper_mode "$PROJ" + [ "$status" -eq 0 ] + tmux list-windows -t ai -F '#{window_name}' | grep -qx "$(basename "$PROJ"):helper-beef" +} + +@test "functional helper_mode: an empty roster opens the plain project window" { + _stub_roster 0 + _stub_window_tail + tmux new-session -d -s ai -n base -c "$PROJ" + run helper_mode "$PROJ" + [ "$status" -eq 0 ] + names="$(tmux list-windows -t ai -F '#{window_name}')" + printf '%s\n' "$names" | grep -qx "$(basename "$PROJ")" + ! printf '%s\n' "$names" | grep -q ':helper-' +} + +# --- help --------------------------------------------------------------------- + +@test "usage documents --helper" { + run bash "$AI" -h + [ "$status" -eq 0 ] + [[ "$output" == *"--helper"* ]] +} diff --git a/scripts/tests/ai-wrap-teardown-hook.bats b/scripts/tests/ai-wrap-teardown-hook.bats index 05c49f1..12ac941 100644 --- a/scripts/tests/ai-wrap-teardown-hook.bats +++ b/scripts/tests/ai-wrap-teardown-hook.bats @@ -7,10 +7,19 @@ setup() { REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)" SCRIPT="$REPO_ROOT/hooks/ai-wrap-teardown.sh" + DISARM="$REPO_ROOT/hooks/session-start-disarm.sh" + GATE="$REPO_ROOT/claude-templates/bin/git-worktree-gate" TMPDIR_T="$(mktemp -d)" PROJ="proj-$$-$BATS_TEST_NUMBER" # unique so /tmp sentinels don't collide CWD="$TMPDIR_T/$PROJ" mkdir -p "$CWD" + git init -q "$CWD" + git -C "$CWD" config user.email test@example.com + git -C "$CWD" config user.name tester + printf 'base\n' >"$CWD/tracked" + git -C "$CWD" add tracked + git -C "$CWD" commit -qm init + bash "$GATE" certify "$CWD" TEARDOWN_SENTINEL="/tmp/ai-wrap-teardown-${PROJ}" SHUTDOWN_SENTINEL="/tmp/ai-wrap-shutdown-${PROJ}" @@ -35,7 +44,7 @@ teardown() { run_hook() { # invoke with the stubbed emacsclient on PATH, feeding Stop-hook JSON printf '{"cwd":"%s","hook_event_name":"Stop"}' "$CWD" \ - | PATH="$BIN:$PATH" bash "$SCRIPT" + | GIT_WORKTREE_GATE="$GATE" PATH="$BIN:$PATH" bash "$SCRIPT" } @test "no sentinel: silent no-op, emacsclient never called" { @@ -81,7 +90,7 @@ run_hook() { : > "$TEARDOWN_SENTINEL" status=0 output="$(printf '{"cwd":"%s","hook_event_name":"Stop"}' "$CWD" \ - | PATH="/usr/bin:/bin" bash "$SCRIPT")" || status=$? + | GIT_WORKTREE_GATE="$GATE" PATH="/usr/bin:/bin" bash "$SCRIPT")" || status=$? [ "$status" -eq 0 ] [ ! -f "$TEARDOWN_SENTINEL" ] } @@ -89,13 +98,114 @@ run_hook() { @test "falls back to PWD basename when cwd is absent from JSON" { # No cwd key: hook uses $PWD. Run from CWD so basename resolves to PROJ. : > "$TEARDOWN_SENTINEL" - run env "PATH=$BIN:$PATH" bash -c "cd '$CWD' && printf '{}' | bash '$SCRIPT'" + run env "GIT_WORKTREE_GATE=$GATE" "PATH=$BIN:$PATH" \ + bash -c "cd '$CWD' && printf '{}' | bash '$SCRIPT'" [ "$status" -eq 0 ] grep -q "cj/ai-term-quit \"$PROJ\"" "$EC_LOG" } @test "emits no stderr noise on a normal stop" { err="$(printf '{"cwd":"%s","hook_event_name":"Stop"}' "$CWD" \ - | PATH="$BIN:$PATH" bash "$SCRIPT" 2>&1 >/dev/null)" + | GIT_WORKTREE_GATE="$GATE" PATH="$BIN:$PATH" bash "$SCRIPT" 2>&1 >/dev/null)" [ -z "$err" ] } + +@test "dirty worktree blocks teardown, preserves sentinel, and names the path" { + printf 'late\n' >>"$CWD/tracked" + : >"$TEARDOWN_SENTINEL" + run run_hook + [ "$status" -eq 0 ] + echo "$output" | jq -e '.decision == "block"' + echo "$output" | jq -e '.reason | test("tracked")' + [ -f "$TEARDOWN_SENTINEL" ] + [ ! -f "$EC_LOG" ] +} + +@test "changed HEAD after certification blocks teardown" { + git -C "$CWD" commit -q --allow-empty -m later + : >"$TEARDOWN_SENTINEL" + run run_hook + echo "$output" | jq -e '.reason | test("HEAD changed")' + [ -f "$TEARDOWN_SENTINEL" ] + [ ! -f "$EC_LOG" ] +} + +@test "missing certificate blocks teardown" { + rm -f "$(git -C "$CWD" rev-parse --absolute-git-dir)/ai-wrap-clean" + : >"$TEARDOWN_SENTINEL" + run run_hook + echo "$output" | jq -e '.reason | test("no clean-tree certificate")' + [ -f "$TEARDOWN_SENTINEL" ] +} + +@test "Codex payload receives continue false and a stop reason" { + printf 'late\n' >>"$CWD/tracked" + : >"$TEARDOWN_SENTINEL" + run bash -c \ + "printf '{\"cwd\":\"$CWD\",\"hook_event_name\":\"Stop\",\"model\":\"gpt-test\"}' \ + | GIT_WORKTREE_GATE='$GATE' PATH='$BIN:$PATH' bash '$SCRIPT'" + [ "$status" -eq 0 ] + echo "$output" | jq -e '.continue == false' + echo "$output" | jq -e '.stopReason | test("tracked")' + [ -f "$TEARDOWN_SENTINEL" ] +} + +@test "successful teardown consumes the clean-tree certificate" { + : >"$TEARDOWN_SENTINEL" + cert="$(git -C "$CWD" rev-parse --absolute-git-dir)/ai-wrap-clean" + [ -f "$cert" ] + run run_hook + [ "$status" -eq 0 ] + [ ! -f "$cert" ] +} + +# --- Cross-session leakage (the 2026-07-27 work-session kill) --------------- +# +# A sentinel is deliberately preserved when certification fails, so a blocked +# wrap can retry within the same session once the tree is clean. But nothing +# bounded that to the session: the sentinel outlived it and detonated in +# whatever session next happened to have a clean tree. work's 11:37 wrap left +# one armed; the 13:20 session committed during startup, went clean, and was +# torn down mid-work. archsetup's had been armed for two days on a live +# terminal. +# +# A new session means the wrap that armed the sentinel is gone, so its pending +# teardown is moot. session-start-disarm.sh clears it. + +@test "session-start disarm: removes a teardown sentinel left by a prior session" { + : > "$TEARDOWN_SENTINEL" + printf '{"cwd":"%s","hook_event_name":"SessionStart"}' "$CWD" \ + | bash "$DISARM" + [ ! -f "$TEARDOWN_SENTINEL" ] +} + +@test "session-start disarm: removes a shutdown sentinel too" { + : > "$SHUTDOWN_SENTINEL" + printf '{"cwd":"%s","hook_event_name":"SessionStart"}' "$CWD" \ + | bash "$DISARM" + [ ! -f "$SHUTDOWN_SENTINEL" ] +} + +@test "session-start disarm: silent and exit 0 when nothing is armed" { + run bash -c "printf '{\"cwd\":\"$CWD\",\"hook_event_name\":\"SessionStart\"}' | bash '$DISARM'" + [ "$status" -eq 0 ] + [ -z "$output" ] +} + +@test "session-start disarm: only clears this project, not another's" { + other="/tmp/ai-wrap-teardown-someotherproj" + : > "$TEARDOWN_SENTINEL" + : > "$other" + printf '{"cwd":"%s","hook_event_name":"SessionStart"}' "$CWD" | bash "$DISARM" + [ ! -f "$TEARDOWN_SENTINEL" ] + [ -f "$other" ] + rm -f "$other" +} + +@test "within-session retry still works: sentinel survives a blocked stop" { + : > "$TEARDOWN_SENTINEL" + echo dirt > "$CWD/untracked.txt" + run run_hook + [ -f "$TEARDOWN_SENTINEL" ] + rm -f "$CWD/untracked.txt" +} diff --git a/scripts/tests/audit.bats b/scripts/tests/audit.bats index 0b062fb..c6123e1 100644 --- a/scripts/tests/audit.bats +++ b/scripts/tests/audit.bats @@ -40,9 +40,28 @@ git_init_with_ai_tracked() { local proj_dir="$1" (cd "$proj_dir" \ && git init -q \ + && git config maintenance.auto false \ + && git config gc.auto 0 \ && git add -A \ && git -c user.email=test@test -c user.name=test commit -q -m initial) } +# maintenance.auto/gc.auto are off because `git commit` otherwise spawns +# `git maintenance run --auto --quiet --detach` (traced on git 2.55). The commit +# returns while that detached process is still writing .git/objects/pack, so +# teardown's `rm -rf` intermittently failed with "Directory not empty" and bats +# reported a passing test as failed. Measured at ~8% of runs. Killing the +# background writer is the fix; a retry loop in teardown would only hide it. +# +# What triggers it: the fixture rsyncs ~170 workflow/script files before +# committing, and maintenance's loose-objects task fires at a default threshold +# of 100. (Not gc.auto's 6700 — that never fires here, which is why disabling +# gc.auto alone was never the explanation.) maintenance.auto false is what +# suppresses it on 2.55; gc.auto 0 is the portable guard for older git, where +# commit calls gc --auto directly and maintenance.auto does not exist. Either +# is sufficient on its own here. +# +# Other bats suites that git-init fixtures are currently safe only by staying +# under that 100-object threshold — an implicit dependency, not a guarantee. @test "audit: clean projects report ok with exit 0" { scaffold_synced_ai "$TEST_HOME/code/alpha" diff --git a/scripts/tests/before-close-queue.bats b/scripts/tests/before-close-queue.bats new file mode 100644 index 0000000..9c968f7 --- /dev/null +++ b/scripts/tests/before-close-queue.bats @@ -0,0 +1,44 @@ +#!/usr/bin/env bats +# The "Colloquialisms and Expansions" convention and its "the list" before-close +# queue must stay wired into the synced template: the protocols.org reference +# section, and the wrap-it-up.org Step 1 sub-step that drains the queue before +# the Summary. Guards against either being dropped in a future edit or sync. + +setup() { + REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)" + PROTO="$REPO_ROOT/claude-templates/.ai/protocols.org" + WRAP="$REPO_ROOT/claude-templates/.ai/workflows/wrap-it-up.org" + PROTO_MIRROR="$REPO_ROOT/.ai/protocols.org" + WRAP_MIRROR="$REPO_ROOT/.ai/workflows/wrap-it-up.org" +} + +@test "protocols.org documents the Colloquialisms and Expansions convention" { + grep -qF '* Colloquialisms and Expansions' "$PROTO" + grep -q 'the list' "$PROTO" + grep -q 'Before-Close Queue' "$PROTO" + grep -qF 'tell <project>' "$PROTO" + grep -q 'inbox-send' "$PROTO" +} + +@test "the colloquialisms reference scopes the queue to the session anchor" { + grep -q 'session-context.org' "$PROTO" + # A must-outlive item is a todo.org task, not a list item. + grep -q 'must outlive the session' "$PROTO" +} + +@test "wrap-it-up Step 1 works the Before-Close Queue before the Summary" { + grep -qF 'Work the Before-Close Queue' "$WRAP" + grep -q 'oldest-first' "$WRAP" + grep -q 'silent no-op' "$WRAP" + # The queue must be worked before the Summary is written, so its edits ride + # this wrap's commit. Assert it sits ahead of the first Summary sub-step. + local queue_line kb_line + queue_line=$(grep -n 'Work the Before-Close Queue' "$WRAP" | head -1 | cut -d: -f1) + kb_line=$(grep -n 'Early KB reflection' "$WRAP" | head -1 | cut -d: -f1) + [ "$queue_line" -lt "$kb_line" ] +} + +@test "the convention is mirrored to the committed .ai copy" { + diff -q "$PROTO" "$PROTO_MIRROR" + diff -q "$WRAP" "$WRAP_MIRROR" +} diff --git a/scripts/tests/git-worktree-gate.bats b/scripts/tests/git-worktree-gate.bats new file mode 100755 index 0000000..7d0d8a1 --- /dev/null +++ b/scripts/tests/git-worktree-gate.bats @@ -0,0 +1,146 @@ +#!/usr/bin/env bats + +setup() { + REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)" + GATE="$REPO_ROOT/claude-templates/bin/git-worktree-gate" + WORK="$(mktemp -d)" + REPO="$WORK/repo" + git init -q "$REPO" + git -C "$REPO" config user.email test@example.com + git -C "$REPO" config user.name tester + printf 'base\n' >"$REPO/tracked" + git -C "$REPO" add tracked + git -C "$REPO" commit -qm init +} + +teardown() { + rm -rf "$WORK" +} + +@test "strict accepts a completely clean worktree" { + run bash "$GATE" strict "$REPO" + [ "$status" -eq 0 ] + [ -z "$output" ] +} + +@test "strict rejects unstaged, staged, and untracked changes with paths" { + printf 'changed\n' >>"$REPO/tracked" + printf 'new\n' >"$REPO/staged" + git -C "$REPO" add staged + printf 'loose\n' >"$REPO/loose" + + run bash "$GATE" strict "$REPO" + [ "$status" -eq 1 ] + [[ "$output" == *"wrap blocked"* ]] + [[ "$output" == *"tracked"* ]] + [[ "$output" == *"staged"* ]] + [[ "$output" == *"loose"* ]] +} + +@test "sync-safe permits untracked inbox deliveries and strict still rejects them" { + mkdir -p "$REPO/inbox/nested" + printf 'handoff\n' >"$REPO/inbox/nested/from-home.org" + + run bash "$GATE" sync-safe "$REPO" + [ "$status" -eq 0 ] + + run bash "$GATE" strict "$REPO" + [ "$status" -eq 1 ] + [[ "$output" == *"inbox/nested/from-home.org"* ]] +} + +@test "sync-safe rejects tracked changes even inside inbox" { + mkdir -p "$REPO/inbox" + printf 'tracked\n' >"$REPO/inbox/tracked.org" + git -C "$REPO" add inbox/tracked.org + git -C "$REPO" commit -qm inbox + printf 'changed\n' >>"$REPO/inbox/tracked.org" + + run bash "$GATE" sync-safe "$REPO" + [ "$status" -eq 1 ] + [[ "$output" == *"inbox/tracked.org"* ]] +} + +@test "sync-safe rejects untracked files outside inbox" { + printf 'scratch\n' >"$REPO/scratch" + run bash "$GATE" sync-safe "$REPO" + [ "$status" -eq 1 ] + [[ "$output" == *"scratch"* ]] +} + +@test "ignored files do not block either policy" { + printf 'cache/\n' >"$REPO/.gitignore" + git -C "$REPO" add .gitignore + git -C "$REPO" commit -qm ignore + mkdir -p "$REPO/cache" + printf 'generated\n' >"$REPO/cache/output" + + run bash "$GATE" strict "$REPO" + [ "$status" -eq 0 ] + run bash "$GATE" sync-safe "$REPO" + [ "$status" -eq 0 ] +} + +@test "reports unusual filenames without losing the entry" { + odd=$'line break\nname' + printf 'odd\n' >"$REPO/$odd" + run bash "$GATE" strict "$REPO" + [ "$status" -eq 1 ] + [[ "$output" == *"line"* ]] + [[ "$output" == *"name"* ]] +} + +@test "certify and verify bind a clean worktree to its current HEAD" { + run bash "$GATE" certify "$REPO" + [ "$status" -eq 0 ] + run bash "$GATE" verify "$REPO" + [ "$status" -eq 0 ] + + git -C "$REPO" commit -q --allow-empty -m later + run bash "$GATE" verify "$REPO" + [ "$status" -eq 1 ] + [[ "$output" == *"HEAD changed"* ]] +} + +@test "verify rejects changes made after certification" { + run bash "$GATE" certify "$REPO" + [ "$status" -eq 0 ] + printf 'late\n' >>"$REPO/tracked" + run bash "$GATE" verify "$REPO" + [ "$status" -eq 1 ] + [[ "$output" == *"tracked"* ]] +} + +@test "dirty submodule blocks strict and sync-safe" { + CHILD="$WORK/child" + git init -q "$CHILD" + git -C "$CHILD" config user.email test@example.com + git -C "$CHILD" config user.name tester + printf 'child\n' >"$CHILD/file" + git -C "$CHILD" add file + git -C "$CHILD" commit -qm init + git -C "$REPO" -c protocol.file.allow=always submodule add -q "$CHILD" sub + git -C "$REPO" commit -qam submodule + printf 'dirty\n' >>"$REPO/sub/file" + + run bash "$GATE" strict "$REPO" + [ "$status" -eq 1 ] + [[ "$output" == *"sub"* ]] + run bash "$GATE" sync-safe "$REPO" + [ "$status" -eq 1 ] +} + +@test "a low-level git status failure blocks instead of looking clean" { + printf 'not an index\n' >"$WORK/bad-index" + run env GIT_INDEX_FILE="$WORK/bad-index" bash "$GATE" strict "$REPO" + [ "$status" -eq 1 ] + [[ "$output" == *"git status failed"* ]] +} + +@test "an in-progress sequencer operation blocks a clean-looking tree" { + gitdir="$(git -C "$REPO" rev-parse --absolute-git-dir)" + mkdir -p "$gitdir/sequencer" + run bash "$GATE" strict "$REPO" + [ "$status" -eq 1 ] + [[ "$output" == *"sequencer"* ]] +} diff --git a/scripts/tests/inbox-boundary-check-hook.bats b/scripts/tests/inbox-boundary-check-hook.bats new file mode 100644 index 0000000..58668a5 --- /dev/null +++ b/scripts/tests/inbox-boundary-check-hook.bats @@ -0,0 +1,83 @@ +#!/usr/bin/env bats +# hooks/inbox-boundary-check.sh — Stop hook that soft-nudges the agent to +# process pending inbox/ handoffs before yielding. Blocks the stop ONCE (emits +# a block decision + reason) when inbox-status reports pending items; on the +# harness re-entry (stop_hook_active: true) it steps aside so an unprocessable +# item or a mid-task pause never wedges. Self-skips where there's no inbox/ or +# no inbox-status. The real inbox-status is copied into the test project so the +# hook runs against real code, not a stub. + +setup() { + REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)" + SCRIPT="$REPO_ROOT/hooks/inbox-boundary-check.sh" + INBOX_STATUS="$REPO_ROOT/claude-templates/.ai/scripts/inbox-status" + TMPDIR_T="$(mktemp -d)" + CWD="$TMPDIR_T/proj" + mkdir -p "$CWD/.ai/scripts" + cp "$INBOX_STATUS" "$CWD/.ai/scripts/inbox-status" + chmod +x "$CWD/.ai/scripts/inbox-status" +} + +teardown() { + rm -rf "$TMPDIR_T" +} + +# Feed Stop-hook JSON on stdin. $1 = stop_hook_active (default false). +run_hook() { + local active="${1:-false}" + printf '{"cwd":"%s","hook_event_name":"Stop","stop_hook_active":%s}' \ + "$CWD" "$active" | bash "$SCRIPT" +} + +@test "pending handoffs block the stop with a count in the reason" { + mkdir -p "$CWD/inbox" + printf 'x\n' >"$CWD/inbox/2026-07-19-from-home-thing.org" + printf 'y\n' >"$CWD/inbox/2026-07-19-from-work-other.org" + run run_hook + [ "$status" -eq 0 ] + # Valid JSON with a block decision. + echo "$output" | jq -e '.decision == "block"' + echo "$output" | jq -e '.reason | test("2 pending")' + echo "$output" | jq -e '.reason | test("inbox.org")' +} + +@test "a clean inbox (only artifacts) emits nothing" { + mkdir -p "$CWD/inbox" + touch "$CWD/inbox/.gitkeep" + printf 'done\n' >"$CWD/inbox/PROCESSED-2026-07-19-old.org" + run run_hook + [ "$status" -eq 0 ] + [ -z "$output" ] +} + +@test "soft-nudge: stop_hook_active=true steps aside even with pending items" { + mkdir -p "$CWD/inbox" + printf 'x\n' >"$CWD/inbox/2026-07-19-from-home-thing.org" + run run_hook true + [ "$status" -eq 0 ] + [ -z "$output" ] +} + +@test "no inbox/ directory is a silent no-op" { + run run_hook + [ "$status" -eq 0 ] + [ -z "$output" ] +} + +@test "inbox-status absent degrades to a silent no-op" { + mkdir -p "$CWD/inbox" + printf 'x\n' >"$CWD/inbox/2026-07-19-from-home-thing.org" + rm -f "$CWD/.ai/scripts/inbox-status" + # Also ensure nothing named inbox-status is on PATH for this run. + run env PATH="/usr/bin:/bin" bash -c ' + printf "{\"cwd\":\"'"$CWD"'\",\"hook_event_name\":\"Stop\"}" | bash "'"$SCRIPT"'"' + [ "$status" -eq 0 ] + [ -z "$output" ] +} + +@test "the emitted reason names the project for context" { + mkdir -p "$CWD/inbox" + printf 'x\n' >"$CWD/inbox/2026-07-19-from-home-thing.org" + run run_hook + echo "$output" | jq -e '.reason | test("proj")' +} diff --git a/scripts/tests/install-ai.bats b/scripts/tests/install-ai.bats index 9c6040c..1549184 100644 --- a/scripts/tests/install-ai.bats +++ b/scripts/tests/install-ai.bats @@ -176,7 +176,7 @@ EOF # --- claude-templates/bin/install-ai launcher -------------------------------- # # The launcher is the PATH-facing front door (make install symlinks it into -# ~/.local/bin/install-ai, same loop as `ai` and `agent-page`). It resolves its +# ~/.local/bin/install-ai, same loop as `ai` and `agent-text`). It resolves its # own real path through the symlink and execs scripts/install-ai.sh, so the # repo-root computation survives being invoked as a symlink from anywhere. @@ -200,3 +200,39 @@ LAUNCHER="$REAL_REPO/claude-templates/bin/install-ai" [ "$status" -eq 0 ] [[ "$output" == *"Bootstrap .ai/"* ]] } + +# --- temp/ ephemeral-artifacts ignore (working/ tracked-from-creation ruling) --- + +@test "install-ai --gitignore: ignores temp/ but never working/" { + mkdir -p "$TEST_HOME/code/fresh" + (cd "$TEST_HOME/code/fresh" && git init -q) + + run bash "$INSTALL_AI" --gitignore "$TEST_HOME/code/fresh" + + [ "$status" -eq 0 ] + grep -qFx "temp/" "$TEST_HOME/code/fresh/.gitignore" + # working/ is the tracked home of in-progress work — it must never be ignored. + ! grep -qEx "/?working/?" "$TEST_HOME/code/fresh/.gitignore" +} + +@test "install-ai --track: ignores temp/ even in track mode" { + mkdir -p "$TEST_HOME/code/tracked" + (cd "$TEST_HOME/code/tracked" && git init -q) + + run bash "$INSTALL_AI" --track "$TEST_HOME/code/tracked" + + [ "$status" -eq 0 ] + # temp/ is ephemeral regardless of whether the project tracks its .ai/ tooling. + grep -qFx "temp/" "$TEST_HOME/code/tracked/.gitignore" +} + +@test "install-ai: temp/ ignore is idempotent (not re-added)" { + mkdir -p "$TEST_HOME/code/fresh" + (cd "$TEST_HOME/code/fresh" && git init -q) + printf 'temp/\n' > "$TEST_HOME/code/fresh/.gitignore" + + run bash "$INSTALL_AI" --gitignore "$TEST_HOME/code/fresh" + + [ "$status" -eq 0 ] + [ "$(grep -cFx 'temp/' "$TEST_HOME/code/fresh/.gitignore")" -eq 1 ] +} diff --git a/scripts/tests/install-hooks-link.bats b/scripts/tests/install-hooks-link.bats index 80ac8dd..5368781 100644 --- a/scripts/tests/install-hooks-link.bats +++ b/scripts/tests/install-hooks-link.bats @@ -20,6 +20,7 @@ run_install() { RULES_DIR="$TMPHOME/rules" \ HOOKS_DIR="$TMPHOME/hooks" \ CLAUDE_DIR="$TMPHOME/claude" \ + CODEX_DIR="$TMPHOME/codex" \ LOCAL_BIN="$TMPHOME/bin" } @@ -28,6 +29,20 @@ run_install() { [ "$status" -eq 0 ] [ -L "$TMPHOME/hooks/session-clear-resume.sh" ] [ -L "$TMPHOME/hooks/precompact-priorities.sh" ] + [ -L "$TMPHOME/hooks/rulesets-write-boundary.py" ] +} + +@test "install links Codex Stop-hook configuration" { + run run_install + [ "$status" -eq 0 ] + [ -L "$TMPHOME/codex/hooks.json" ] + grep -q "ai-wrap-teardown.sh" "$TMPHOME/codex/hooks.json" +} + +@test "install links the shared Git worktree gate beside the launcher" { + run run_install + [ "$status" -eq 0 ] + [ -L "$TMPHOME/bin/git-worktree-gate" ] } @test "install does not link opt-in hooks" { diff --git a/scripts/tests/install-lang-collision.bats b/scripts/tests/install-lang-collision.bats index 36abb5b..e03e136 100644 --- a/scripts/tests/install-lang-collision.bats +++ b/scripts/tests/install-lang-collision.bats @@ -5,10 +5,14 @@ # (appended, deduped) and CLAUDE.md is seed-only, so both compose across # bundles. Three do not: # -# claude/settings.json elisp, bash, go — cp -rT, silently overwritten -# githooks/pre-commit elisp, bash, go — cp -rT, silently overwritten +# claude/settings.json all 5 bundles — cp -rT, silently overwritten +# githooks/pre-commit all 5 bundles — cp -rT, silently overwritten # coverage-makefile.txt 4 bundles — [skip]ped, fragment dropped # +# The first two read "elisp, bash, go" until 2026-07-23, when python and +# typescript gained the components they had been missing. The consequence is +# that no two shipping bundles compose any more; see the polyglot task. +# # Installing a second bundle used to replace the first's settings.json and # pre-commit while printing [ok], so a project could lose its paren check or # secret scan and read the output as success. The guard refuses instead, naming @@ -68,8 +72,11 @@ teardown() { } @test "install-lang: bundles colliding only on coverage-makefile.txt are refused too" { - # python and typescript ship no settings.json or githooks, but both ship a - # coverage fragment. The second one's used to be silently dropped. + # Named for the pre-2026-07-23 reason: python and typescript shipped no + # settings.json or githooks, so the coverage fragment was their only overlap + # and the second one's used to be silently dropped. They now overlap on all + # three, so this exercises the multi-file refusal path — the coverage + # fragment must still be named among them. bash "$INSTALL" python "$PROJ" run bash "$INSTALL" typescript "$PROJ" [ "$status" -ne 0 ] || { echo "typescript installed over python's coverage fragment"; return 1; } @@ -77,17 +84,41 @@ teardown() { } @test "install-lang: two bundles that share no overwritten file install together" { - # bash ships settings.json + githooks and no coverage fragment; python ships - # only the coverage fragment. Nothing overlaps, so the guard must stay out of - # the way. This is the path where a false refusal would be easiest to - # introduce: the bundle IS detected, and only the empty file-list stops it. + # The guard must stay out of the way when nothing overlaps. This is the path + # where a false refusal would be easiest to introduce: the bundle IS + # detected, and only the empty file-list stops it. + # + # This used to be tested with bash + python, which composed because python + # shipped no settings.json and no githooks. That was the incomplete-bundle + # bug (fixed 2026-07-23), not a design property — so no pair of *shipping* + # bundles is non-colliding any more, and the case needs a synthetic bundle. + # See the polyglot task in todo.org: whether every pair now colliding is + # acceptable is an open question, but the guard's own no-false-refusal + # behavior is not, and stays pinned here. + fake="${BATS_TEST_DIRNAME}/../../languages/zz-test-rulesonly" + mkdir -p "$fake/claude/rules" + printf '# rule\n' > "$fake/claude/rules/zz-testing.md" + bash "$INSTALL" bash "$PROJ" - run bash "$INSTALL" python "$PROJ" + run bash "$INSTALL" zz-test-rulesonly "$PROJ" + rm -rf "$fake" + [ "$status" -eq 0 ] || { echo "guard falsely refused a non-colliding pair: $output"; return 1; } - [ -f "$PROJ/coverage-makefile.txt" ] || { echo "python's coverage fragment did not land"; return 1; } # bash's config survives untouched. grep -q 'validate-bash.sh' "$PROJ/.claude/settings.json" - [ -f "$PROJ/.claude/rules/bash.md" ] && [ -f "$PROJ/.claude/rules/python-testing.md" ] + [ -f "$PROJ/.claude/rules/bash.md" ] && [ -f "$PROJ/.claude/rules/zz-testing.md" ] +} + +@test "install-lang: completing python made it collide with bash (documents the tradeoff)" { + # Pins the consequence of the 2026-07-23 bundle completion so it can't drift + # back unnoticed: python now ships settings.json + githooks/pre-commit, so + # bash + python is a genuine overwrite conflict and the guard refuses it. + # Whether that's the right trade is Craig's open call; that it IS the current + # behavior is what this test records. + bash "$INSTALL" bash "$PROJ" + run bash "$INSTALL" python "$PROJ" + [ "$status" -ne 0 ] + [[ "$output" == *"collision"* ]] } # ---- The escape hatch ---- diff --git a/scripts/tests/install-lang-completeness.bats b/scripts/tests/install-lang-completeness.bats new file mode 100644 index 0000000..42832dc --- /dev/null +++ b/scripts/tests/install-lang-completeness.bats @@ -0,0 +1,93 @@ +#!/usr/bin/env bats +# +# Tests for install-lang.sh's bundle-completeness warning. +# +# Background: the python and typescript bundles shipped for nearly two months +# with no githooks/pre-commit, so any project installing them got no +# credential scan on commit. install-lang guarded each component copy with a +# plain `[ -d ... ]`, so a missing component was indistinguishable from a +# complete install — it printed nothing and exited 0. +# +# The fix is not "never miss a component" (a person will), it's "say so when +# you do". These tests pin that: a complete bundle installs quietly, an +# incomplete one names exactly what's absent, and neither case fails the +# install — a warning must not become a new way to block work. + +REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)" +INSTALL="$REPO_ROOT/scripts/install-lang.sh" + +setup() { + TEST_DIR="$(mktemp -d -t install-lang-bats.XXXXXX)" + PROJECT="$TEST_DIR/proj" + mkdir -p "$PROJECT" + git init -q "$PROJECT" +} + +teardown() { + rm -rf "$TEST_DIR" +} + +# ---- Normal: a complete bundle is quiet ------------------------------ + +@test "install-lang: a complete bundle warns about nothing" { + run env LANG_ARG=bash bash "$INSTALL" bash "$PROJECT" + [ "$status" -eq 0 ] + [[ "$output" != *"incomplete"* ]] +} + +@test "install-lang: python is now a complete bundle" { + run bash "$INSTALL" python "$PROJECT" + [ "$status" -eq 0 ] + [[ "$output" != *"incomplete"* ]] + [ -f "$PROJECT/githooks/pre-commit" ] + [ -f "$PROJECT/.claude/settings.json" ] + [ -f "$PROJECT/.claude/hooks/validate-python.sh" ] +} + +@test "install-lang: typescript is now a complete bundle" { + run bash "$INSTALL" typescript "$PROJECT" + [ "$status" -eq 0 ] + [[ "$output" != *"incomplete"* ]] + [ -f "$PROJECT/githooks/pre-commit" ] + [ -f "$PROJECT/.claude/settings.json" ] + [ -f "$PROJECT/.claude/hooks/validate-typescript.sh" ] +} + +@test "install-lang: the installed pre-commit is executable" { + run bash "$INSTALL" python "$PROJECT" + [ "$status" -eq 0 ] + [ -x "$PROJECT/githooks/pre-commit" ] +} + +# ---- Error: an incomplete bundle announces itself -------------------- + +@test "install-lang: a bundle missing githooks/ warns and names it" { + fake="$REPO_ROOT/languages/zz-test-partial" + mkdir -p "$fake/claude/rules" + printf '# rule\n' > "$fake/claude/rules/zz.md" + run bash "$INSTALL" zz-test-partial "$PROJECT" + rm -rf "$fake" + [ "$status" -eq 0 ] + [[ "$output" == *"incomplete"* ]] + [[ "$output" == *"githooks/pre-commit"* ]] +} + +@test "install-lang: the warning names every missing component, not just the first" { + fake="$REPO_ROOT/languages/zz-test-partial" + mkdir -p "$fake/claude/rules" + printf '# rule\n' > "$fake/claude/rules/zz.md" + run bash "$INSTALL" zz-test-partial "$PROJECT" + rm -rf "$fake" + [[ "$output" == *"githooks/pre-commit"* ]] + [[ "$output" == *"settings.json"* ]] +} + +@test "install-lang: an incomplete bundle still installs (warn, never block)" { + fake="$REPO_ROOT/languages/zz-test-partial" + mkdir -p "$fake/claude/rules" + printf '# rule\n' > "$fake/claude/rules/zz.md" + run bash "$INSTALL" zz-test-partial "$PROJECT" + rm -rf "$fake" + [ "$status" -eq 0 ] + [ -f "$PROJECT/.claude/rules/zz.md" ] +} diff --git a/scripts/tests/install-lang.bats b/scripts/tests/install-lang.bats index 8518852..c00b915 100644 --- a/scripts/tests/install-lang.bats +++ b/scripts/tests/install-lang.bats @@ -79,19 +79,42 @@ teardown() { grep -qxF "coverage/" "$PROJECT/.gitignore" } -@test "install-lang python: seeds the language-neutral default CLAUDE.md when the bundle ships none" { - run bash "$INSTALL_LANG" python "$PROJECT" +@test "install-lang: seeds the language-neutral default CLAUDE.md when the bundle ships none" { + # Every shipping bundle now carries its own CLAUDE.md (python and typescript + # gained theirs 2026-07-23), so the fallback needs a synthetic bundle to + # exercise. Keep testing it: the fallback is what stops a bundle added later, + # before its CLAUDE.md is written, from inheriting another language's header. + fake="$REAL_REPO/languages/zz-test-noclaude" + mkdir -p "$fake/claude/rules" + printf '# rule\n' > "$fake/claude/rules/zz.md" + run bash "$INSTALL_LANG" zz-test-noclaude "$PROJECT" + rm -rf "$fake" [ "$status" -eq 0 ] [ -f "$PROJECT/CLAUDE.md" ] - # The default names no language, so it can't mislabel a python (or bash, or - # multi-bundle) project the way inheriting elisp's "Elisp project" header did. - ! grep -qi "Python project" "$PROJECT/CLAUDE.md" + # The default names no language, so it can't mislabel a project the way + # inheriting elisp's "Elisp project" header did. ! grep -qi "Elisp project" "$PROJECT/CLAUDE.md" grep -qF "names no language" "$PROJECT/CLAUDE.md" [[ "$output" == *"language-neutral default"* ]] } +@test "install-lang python: seeds the bundle's own CLAUDE.md, not the default" { + run bash "$INSTALL_LANG" python "$PROJECT" + + [ "$status" -eq 0 ] + grep -qF "Python project." "$PROJECT/CLAUDE.md" + [[ "$output" == *"CLAUDE.md installed (python)"* ]] +} + +@test "install-lang typescript: seeds the bundle's own CLAUDE.md, not the default" { + run bash "$INSTALL_LANG" typescript "$PROJECT" + + [ "$status" -eq 0 ] + grep -qF "TypeScript/JavaScript project." "$PROJECT/CLAUDE.md" + [[ "$output" == *"CLAUDE.md installed (typescript)"* ]] +} + @test "install-lang elisp: seeds the bundle's own CLAUDE.md, not the default" { run bash "$INSTALL_LANG" elisp "$PROJECT" @@ -147,3 +170,32 @@ teardown() { grep -qxF ".claude/" "$PROJECT/.gitignore" grep -qxF "cover.out" "$PROJECT/.gitignore" } + +@test "install-lang python: full bundle lands (rules, hook, settings, githook, CLAUDE.md, coverage)" { + run bash "$INSTALL_LANG" python "$PROJECT" + + [ "$status" -eq 0 ] + [ -f "$PROJECT/.claude/rules/python-testing.md" ] + # PostToolUse validate hook, executable and wired into settings + [ -x "$PROJECT/.claude/hooks/validate-python.sh" ] + grep -qF "validate-python.sh" "$PROJECT/.claude/settings.json" + # Pre-commit githook — the secret scan. Absent until 2026-07-23. + [ -x "$PROJECT/githooks/pre-commit" ] + grep -qF "potential secret" "$PROJECT/githooks/pre-commit" + # Coverage slice + [ -f "$PROJECT/.claude/scripts/coverage-summary.py" ] + grep -qxF ".claude/" "$PROJECT/.gitignore" +} + +@test "install-lang typescript: full bundle lands (rules, hook, settings, githook, CLAUDE.md, coverage)" { + run bash "$INSTALL_LANG" typescript "$PROJECT" + + [ "$status" -eq 0 ] + [ -f "$PROJECT/.claude/rules/typescript-testing.md" ] + [ -x "$PROJECT/.claude/hooks/validate-typescript.sh" ] + grep -qF "validate-typescript.sh" "$PROJECT/.claude/settings.json" + [ -x "$PROJECT/githooks/pre-commit" ] + grep -qF "potential secret" "$PROJECT/githooks/pre-commit" + [ -f "$PROJECT/.claude/scripts/coverage-summary.js" ] + grep -qxF ".claude/" "$PROJECT/.gitignore" +} diff --git a/scripts/tests/lint-coverage.bats b/scripts/tests/lint-coverage.bats new file mode 100644 index 0000000..130212d --- /dev/null +++ b/scripts/tests/lint-coverage.bats @@ -0,0 +1,54 @@ +#!/usr/bin/env bats +# +# Coverage tests for scripts/lint.sh. +# +# lint.sh sweeps scripts/*.sh, languages/*/claude/hooks/*.sh, and +# languages/*/githooks/* through check_hook (shebang present, executable bit +# set). It never touched claude-templates/bin/ — the four scripts `make install` +# symlinks into ~/.local/bin, so the most exposed shell in the repo was the only +# shell with no gate over it (found 2026-07-24). +# +# These tests pin the coverage itself rather than the current cleanliness: they +# plant a deliberately broken file in each swept location and assert lint.sh +# complains. A location that stops being swept fails here. + +REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)" +LINT="$REPO_ROOT/scripts/lint.sh" + +teardown() { + [ -n "${PLANTED:-}" ] && rm -f "$PLANTED" + PLANTED="" +} + +@test "lint: a bin/ script missing its shebang is flagged" { + PLANTED="$REPO_ROOT/claude-templates/bin/zz-test-noshebang" + printf 'echo hi\n' > "$PLANTED" + chmod +x "$PLANTED" + run bash "$LINT" + [[ "$output" == *"zz-test-noshebang"* ]] || { + echo "lint.sh did not report the planted bin/ script — that path is unswept" + echo "$output" + return 1 + } +} + +@test "lint: a bin/ script that is not executable is flagged" { + PLANTED="$REPO_ROOT/claude-templates/bin/zz-test-noexec" + printf '#!/usr/bin/env bash\necho hi\n' > "$PLANTED" + chmod -x "$PLANTED" + run bash "$LINT" + [[ "$output" == *"zz-test-noexec"* ]] +} + +@test "lint: the existing scripts/ sweep still works (guards the regression)" { + PLANTED="$REPO_ROOT/scripts/zz-test-noshebang.sh" + printf 'echo hi\n' > "$PLANTED" + chmod +x "$PLANTED" + run bash "$LINT" + [[ "$output" == *"zz-test-noshebang.sh"* ]] +} + +@test "lint: the real tree passes (no planted file)" { + run bash "$LINT" + [ "$status" -eq 0 ] +} diff --git a/scripts/tests/pre-commit-secret-scan.bats b/scripts/tests/pre-commit-secret-scan.bats index 013129e..4647556 100644 --- a/scripts/tests/pre-commit-secret-scan.bats +++ b/scripts/tests/pre-commit-secret-scan.bats @@ -14,7 +14,13 @@ # random base64 blob matched. Measured at ~6% of 100KB blobs; case-sensitive # matching drops it to 0 across ~10MB. -VARIANTS="elisp bash go" +# Discovered, never enumerated. This list read "elisp bash go" while python and +# typescript also shipped pre-commit hooks, so every "in every variant" test +# below silently skipped two bundles from the day they were added — the same +# enumerate-instead-of-discover failure these tests exist to catch. A new bundle +# is now covered the moment it has a hook. +VARIANTS="$(cd "${BATS_TEST_DIRNAME}/../../languages" && \ + for d in */githooks/pre-commit; do [ -f "$d" ] && printf '%s ' "${d%%/*}"; done)" setup() { REPO="$(mktemp -d)" @@ -142,3 +148,50 @@ run_hook() { [ "$status" -eq 0 ] || { echo "$v blocked a removal: $output"; return 1; } done } + +# ---- Fail-closed: a broken git must never read as "nothing to scan" ---- + +# Put a stub `git` ahead of the real one that fails only the staged-diff call +# and delegates everything else, so just the pipeline under test breaks. +break_git() { + mkdir -p "$REPO/bin" + cat > "$REPO/bin/git" <<'STUB' +#!/usr/bin/env bash +if [ "${1:-}" = "diff" ] && [ "${2:-}" = "--cached" ]; then + echo "simulated git failure" >&2 + exit 128 +fi +exec /usr/bin/git "$@" +STUB + chmod +x "$REPO/bin/git" +} + +@test "secret-scan: a broken git refuses rather than passing blind, in every variant" { + # The scan built its input as `git diff ... | grep ... || true`. With no + # pipefail, a git failure yielded an empty string, so the scan searched + # nothing, found nothing, and reported clean with a real secret staged. + # Found by .emacs.d in elisp 2026-07-24; all five variants had it. + stage 'aws_key = "AKIAIOSFODNN7EXAMPLE"' + break_git + for v in $VARIANTS; do + run env PATH="$REPO/bin:$PATH" bash \ + "${BATS_TEST_DIRNAME}/../../languages/$v/githooks/pre-commit" + [ "$status" -ne 0 ] || { + echo "$v FAILED OPEN: exited 0 with a secret staged and git broken" + return 1 + } + done +} + +@test "secret-scan: the refusal says why, in every variant" { + stage 'aws_key = "AKIAIOSFODNN7EXAMPLE"' + break_git + for v in $VARIANTS; do + run env PATH="$REPO/bin:$PATH" bash \ + "${BATS_TEST_DIRNAME}/../../languages/$v/githooks/pre-commit" + [[ "$output" == *"cannot read"* ]] || { + echo "$v refused without naming the cause: $output" + return 1 + } + done +} diff --git a/scripts/tests/rename-ai-artifact.bats b/scripts/tests/rename-ai-artifact.bats index f00c92f..ea7f36e 100644 --- a/scripts/tests/rename-ai-artifact.bats +++ b/scripts/tests/rename-ai-artifact.bats @@ -27,7 +27,17 @@ setup() { printf 'Old session mentioning foo and foo.org — this is history.\n' > "$base/sessions/2026-01-01-old.org" done printf 'See foo.org and foo-helper.py. Also foobar.org stays.\n' > "$REPO/notes.org" - ( cd "$REPO" && git init -q && git add -A && git -c user.email=t@t -c user.name=t commit -qm init ) + # Disable background auto-maintenance/gc before any git command can arm it: + # otherwise a post-commit `git maintenance run --auto` writes into .git after + # the test body and races teardown's `rm -rf`, which then fails intermittently + # with ".git: Directory not empty". Diagnose-not-mask: kill the writer, don't + # retry the rm. + ( cd "$REPO" \ + && git init -q \ + && git config gc.auto 0 \ + && git config maintenance.auto false \ + && git add -A \ + && git -c user.email=t@t -c user.name=t commit -qm init ) } teardown() { diff --git a/scripts/tests/signal-receive.bats b/scripts/tests/signal-receive.bats new file mode 100644 index 0000000..8fe4d67 --- /dev/null +++ b/scripts/tests/signal-receive.bats @@ -0,0 +1,66 @@ +#!/usr/bin/env bats +# signal-receive.sh — drains the Signal pager account's inbound queue to keep it +# warm (the roam-sync-shaped fix for receive-staleness) and to surface Craig's +# replies. Run on velox by the signal-receive systemd timer. These tests stub +# signal-cli on PATH to verify command construction without a network or the +# real account. + +setup() { + REPO_ROOT="$(cd "$(dirname "$BATS_TEST_FILENAME")/../.." && pwd)" + RECV="$REPO_ROOT/scripts/signal-receive.sh" + STUBS="$(mktemp -d)" + LOG="$STUBS/calls.log" + # HAS_ACCOUNT (default 1) controls the listAccounts answer so the local-account + # guard can be exercised: "1" lists both test accounts as present, "0" lists + # none (so the guard no-ops). + cat > "$STUBS/signal-cli" <<EOF +#!/bin/bash +if [ "\$1" = "listAccounts" ]; then + if [ "\${HAS_ACCOUNT:-1}" = "1" ]; then + echo "Number: +15045173983" + echo "Number: +19995550000" + fi + exit 0 +fi +echo "signal-cli \$*" >> "$LOG" +exit 0 +EOF + chmod +x "$STUBS/signal-cli" +} + +teardown() { + rm -rf "$STUBS" +} + +@test "defaults to the pager account, a 10s timeout, and read receipts" { + PATH="$STUBS:$PATH" run bash "$RECV" + [ "$status" -eq 0 ] + grep -q -- "-a +15045173983 receive --timeout 10 --send-read-receipts" "$LOG" +} + +@test "honors an explicit account and timeout" { + PATH="$STUBS:$PATH" run bash "$RECV" +19995550000 25 + [ "$status" -eq 0 ] + grep -q -- "-a +19995550000 receive --timeout 25" "$LOG" +} + +@test "no-ops (exit 0) when the account is not registered on this machine" { + # Guards the runbook claim that the timer no-ops on a common-package machine + # that stows the units but holds no pager account. + HAS_ACCOUNT=0 PATH="$STUBS:$PATH" run bash "$RECV" + [ "$status" -eq 0 ] + [[ "$output" == *"not registered"* ]] + # The receive must not have run. + ! grep -q "receive" "$LOG" +} + +@test "no-ops (exit 0) when signal-cli is not on PATH" { + # Empty stub dir with no signal-cli; echo/exit/command are bash builtins, so + # the script still runs and hits its signal-cli-absent branch, which no-ops. + rm -f "$STUBS/signal-cli" + # Invoke bash by absolute path so `run` finds it regardless of the empty + # PATH the script itself sees. + PATH="$STUBS" run "$(command -v bash)" "$RECV" + [ "$status" -eq 0 ] + [[ "$output" == *"signal-cli"* ]] +} diff --git a/scripts/tests/sweep-gitignore-tooling.bats b/scripts/tests/sweep-gitignore-tooling.bats index f18eac5..240c3be 100644 --- a/scripts/tests/sweep-gitignore-tooling.bats +++ b/scripts/tests/sweep-gitignore-tooling.bats @@ -190,3 +190,44 @@ make_project() { [ "$status" -eq 0 ] [[ "$output" != *"publicly reachable"* ]] } + +# --- temp/ ephemeral-artifacts backfill (mode-independent) --- + +@test "sweep: adds temp/ to a gitignore-mode project" { + make_project gimode $'.ai/\n' + + run bash "$SWEEP" "$ROOT" + + [ "$status" -eq 0 ] + grep -qFx "temp/" "$ROOT/gimode/.gitignore" +} + +@test "sweep: adds temp/ to a TRACK-mode project too (temp/ is mode-independent)" { + make_project trackmode $'# build\nout/\n' + + run bash "$SWEEP" "$ROOT" + + [ "$status" -eq 0 ] + # The tooling set is skipped in track mode, but temp/ is not. + grep -qFx "temp/" "$ROOT/trackmode/.gitignore" + ! grep -qFx ".ai/" "$ROOT/trackmode/.gitignore" +} + +@test "sweep: never adds working/" { + make_project gimode $'.ai/\n' + + run bash "$SWEEP" "$ROOT" + + [ "$status" -eq 0 ] + ! grep -qEx "/?working/?" "$ROOT/gimode/.gitignore" +} + +@test "sweep: temp/ backfill is idempotent" { + make_project gimode $'.ai/\ntemp/\n' + bash "$SWEEP" "$ROOT" >/dev/null + + run bash "$SWEEP" "$ROOT" + + [ "$status" -eq 0 ] + [ "$(grep -cFx 'temp/' "$ROOT/gimode/.gitignore")" -eq 1 ] +} diff --git a/scripts/tests/sync-language-bundle.bats b/scripts/tests/sync-language-bundle.bats index 1871444..0eb3ae2 100644 --- a/scripts/tests/sync-language-bundle.bats +++ b/scripts/tests/sync-language-bundle.bats @@ -19,7 +19,20 @@ teardown() { rm -rf "$PROJ" } -# Mirror install-lang.sh: copy the bundle's files into a synthetic project. +# Mirror what a CURRENT install-lang.sh leaves: language rules only, no copies +# of the generic rules (those live once at ~/.claude/rules/). +install_bundle_current() { + install_bundle "$1" "$2" + local f + for f in "$REAL_REPO/claude-rules"/*.md; do + [ -f "$f" ] || continue + rm -f "$2/.claude/rules/$(basename "$f")" + done +} + +# Mirror what an OLDER install-lang.sh left behind: language rules PLUS copies +# of every generic rule. This is the state the sweep exists to clean up, and +# real projects are still in it until their next startup. install_bundle() { local lang="$1" proj="$2" mkdir -p "$proj/.claude/rules" @@ -67,14 +80,14 @@ install_team_overlay() { } @test "sync: clean elisp bundle is a quiet no-op (exit 0)" { - install_bundle elisp "$PROJ" + install_bundle_current elisp "$PROJ" run bash "$SCRIPT" "$PROJ" [ "$status" -eq 0 ] [ -z "$output" ] } @test "sync: absent CLAUDE.md is not flagged as drift (seed-only/project-owned)" { - install_bundle elisp "$PROJ" # helper never seeds CLAUDE.md + install_bundle_current elisp "$PROJ" # helper never seeds CLAUDE.md [ ! -f "$PROJ/CLAUDE.md" ] run bash "$SCRIPT" "$PROJ" [ "$status" -eq 0 ] @@ -94,13 +107,17 @@ install_team_overlay() { matches_canonical ".claude/rules/elisp.md" "$REAL_REPO/languages/elisp/claude/rules/elisp.md" } -@test "sync: drifted generic rule is auto-fixed and restored" { +# Generic rules are no longer auto-fixed in place: they are swept, because the +# global copy at ~/.claude/rules/ is the one that loads. A drifted project copy +# is not repaired, it is removed — which is the stronger fix, since the drifted +# copy outranked the global rule while it existed. +@test "sync: a drifted generic rule copy is swept, not repaired" { install_bundle elisp "$PROJ" echo "junk" >> "$PROJ/.claude/rules/commits.md" run bash "$SCRIPT" "$PROJ" [ "$status" -eq 0 ] - [[ "$output" == *".claude/rules/commits.md"* ]] - matches_canonical ".claude/rules/commits.md" "$REAL_REPO/claude-rules/commits.md" + [[ "$output" == *"swept"* ]] + [ ! -f "$PROJ/.claude/rules/commits.md" ] } @test "sync: missing rule is re-copied" { @@ -276,3 +293,49 @@ install_team_overlay() { [ ! -f "$PROJ/.claude/rules/commits.md" ] [ ! -f "$PROJ/.claude/rules/testing.md" ] } + +# --- generic-rule de-duplication ------------------------------------------- +# +# Generic rules live at ~/.claude/rules/ (symlinked by `make install`) and load +# in every session. Copying them into each project as well made Claude Code +# load them twice, and project copies take priority — so a stale project copy +# silently overrode the fresh global one. The bundle now ships only its own +# language rules and sweeps the duplicates it previously installed. + +@test "sync: sweeps generic rule copies that duplicate the global set" { + install_bundle python "$PROJ" + [ -f "$PROJ/.claude/rules/commits.md" ] + HOME_RULES="$(mktemp -d)" ; mkdir -p "$HOME_RULES/.claude/rules" + cp "$REAL_REPO/claude-rules"/*.md "$HOME_RULES/.claude/rules/" + run env HOME="$HOME_RULES" bash "$SCRIPT" "$PROJ" + [ "$status" -eq 0 ] + [ ! -f "$PROJ/.claude/rules/commits.md" ] + [ ! -f "$PROJ/.claude/rules/todo-format.md" ] +} + +@test "sync: keeps the language bundle's own rules while sweeping generics" { + install_bundle python "$PROJ" + HOME_RULES="$(mktemp -d)" ; mkdir -p "$HOME_RULES/.claude/rules" + cp "$REAL_REPO/claude-rules"/*.md "$HOME_RULES/.claude/rules/" + run env HOME="$HOME_RULES" bash "$SCRIPT" "$PROJ" + [ "$status" -eq 0 ] + [ -f "$PROJ/.claude/rules/python-testing.md" ] +} + +@test "sync: keeps a project-owned overlay rule the bundle does not own" { + install_bundle python "$PROJ" + printf '# Publishing\n\nApplies to: `**/*`\n' > "$PROJ/.claude/rules/publishing.md" + HOME_RULES="$(mktemp -d)" ; mkdir -p "$HOME_RULES/.claude/rules" + cp "$REAL_REPO/claude-rules"/*.md "$HOME_RULES/.claude/rules/" + run env HOME="$HOME_RULES" bash "$SCRIPT" "$PROJ" + [ "$status" -eq 0 ] + [ -f "$PROJ/.claude/rules/publishing.md" ] +} + +@test "sync: does NOT sweep when the global rule is absent (nothing takes over)" { + install_bundle python "$PROJ" + HOME_RULES="$(mktemp -d)" # no ~/.claude/rules/ at all + run env HOME="$HOME_RULES" bash "$SCRIPT" "$PROJ" + [ "$status" -eq 0 ] + [ -f "$PROJ/.claude/rules/commits.md" ] +} diff --git a/testing-standards/SKILL.md b/testing-standards/SKILL.md new file mode 100644 index 0000000..1f16528 --- /dev/null +++ b/testing-standards/SKILL.md @@ -0,0 +1,391 @@ +--- +name: testing-standards +description: | + The full testing standard: characterization tests for untested legacy code, the Normal/Boundary/Error case detail, combinatorial and property-based and mutation testing, test organization and the pyramid, integration-test rules, naming conventions, test-quality rules (independence, determinism, performance, mocking boundaries, signs of overmocking, testing framework-heavy code, never inlining production code, asserting error behavior not error text), the refactor-when-tests-are-hard principle, coverage targets, the TDD discipline table, the spike exception, and the anti-pattern list. + + Use when writing or reviewing tests, deciding how to test something, hardening untested code, or judging whether a test suite is adequate. + + Do NOT use for the standing directive itself — that TDD is the default and that every unit needs Normal, Boundary, and Error cases lives in claude-rules/testing.md and is always loaded. Also see the add-tests skill for the guided coverage workflow and pairwise-tests for the combinatorial matrix generator. +--- + +# Testing Standards — the detail + +Applies to test code and to decisions about how to test. + +The standing directive is NOT here. That TDD is the default, and that every +unit needs Normal, Boundary, and Error cases, lives in `claude-rules/testing.md` +and is always loaded, because it has to fire before any code gets written and +nothing else would summon it. This file is everything needed once you are +actually writing the tests. + +### Understand Before You Test + +Before writing tests, invest time in understanding the code: + +1. **Explore the codebase** — Read the module under test, its callers, and its dependencies. Understand the data flow end to end. +2. **Identify the root cause** — If fixing a bug, trace the problem to its origin. Don't test (or fix) surface symptoms when the real issue is deeper in the call chain. +3. **Reason through edge cases** — Consider boundary conditions, error states, concurrent access, and interactions with adjacent modules. Your tests should cover what could actually go wrong, not just the obvious happy path. + +### Adding Tests to Existing Untested Code + +When working in a codebase without tests: + +1. Write a **characterization test** that captures current behavior before making changes +2. Use the characterization test as a safety net while refactoring +3. Then follow normal TDD for the new change + +A characterization test asserts what the code *actually does* right now, not +what it *should* do. Write it by running the code against a fixed input, +reading the exact value or effect it currently produces, and asserting that +value — Feathers' recipe is to assert something you know is wrong, run it, and +paste the real value out of the failure. You don't need to know the correct +answer to write one; you record the observed one. That's what makes it +mechanical enough to bring a large untested surface under test without +re-deriving each unit's spec. + +**Characterize with the same Normal/Boundary/Error set as any unit** (the three +categories below), not one happy-path capture per function. On a characterization +test the negative and boundary cases are the ones that find bugs: untested legacy +code is weakest exactly at the empty input, the malformed value, the missing +upstream, and pinning what it *currently* does there writes the wrong behavior +down in black and white, where it becomes a bug you can see and decide on. When a +pinned case turns out to be a bug rather than behavior worth preserving, that one +test graduates from "record current" to "assert correct" and you fix the code. +The happy-path case is the regression net; the negative and boundary cases are +the audit. + +Bugs that live *inside* a unit are caught by this three-category set; bugs in how +units compose — ordering, shared state handed between them — are invisible to any +per-unit test and need a functional/integration test over the composed path (see +Integration Tests below and the pyramid). + +### 1. Normal Cases (Happy Path) +- Standard inputs and expected use cases +- Common workflows and default configurations +- Typical data volumes + +### 2. Boundary Cases +- Minimum/maximum values (0, 1, -1, MAX_INT) +- Empty vs null vs undefined (language-appropriate) +- Single-element collections +- Unicode and internationalization (emoji, RTL text, combining characters) +- Very long strings, deeply nested structures +- Timezone boundaries (midnight, DST transitions) +- Date edge cases (leap years, month boundaries) + +### 3. Error Cases +- Invalid inputs and type mismatches +- Network failures and timeouts +- Missing required parameters +- Permission denied scenarios +- Resource exhaustion +- Malformed data + +## Combinatorial Coverage + +For functions with 3+ parameters that each take multiple values (feature-flag +combinations, config matrices, permission/role interactions, multi-field +form validation, API parameter spaces), the exhaustive test count explodes +(M^N) while 3-5 ad-hoc cases miss pair interactions. Use **pairwise / +combinatorial testing** — generate a minimal matrix that hits every 2-way +combination of parameter values. Empirically catches 60-90% of combinatorial +bugs with 80-99% fewer tests. + +Invoke `/pairwise-tests` on the offending function; continue using `/add-tests` +and the Normal/Boundary/Error discipline for the rest. The two approaches +complement: pairwise covers parameter *interactions*; category discipline +covers each parameter's individual edge space. + +Skip pairwise when: the function has 1-2 parameters (just write the cases), +the context requires *provably* exhaustive coverage (regulated systems — document +in an ADR), or the testing target is non-parametric (single happy path, +performance regression, a specific error). + +## Escalation Beyond Category and Pairwise + +The Normal/Boundary/Error categories and the pairwise matrix are the default +discipline. Two further techniques escalate beyond them — reach for them when +the default leaves a gap, not on every unit. + +### Property-Based Testing + +When an invariant holds across a broad input domain — round-trips +(`decode(encode(x)) == x`), idempotence (`f(f(x)) == f(x)`), ordering +invariants (output is always sorted), or any "output always satisfies X" — +generate inputs and assert the property instead of enumerating cases. The +generator explores corners you wouldn't think to write by hand, and a +failing case shrinks to a minimal reproducer. Use the standard tool for the +language (Hypothesis for Python, fast-check for JS, proptest for Rust). +State the property as the test name and let the framework supply the inputs. + +Reach for this when the behavior is a law over a domain rather than a fixed +set of examples. Keep category-discipline cases for the specific edges that +must always hold; the property test covers the space between them. + +### Mutation Testing + +When line coverage is high but you suspect the assertions are thin — tests +that execute the code without checking its output, or that pass with a +function body replaced by a stub — use mutation testing to measure whether +the suite actually kills injected faults. The tool flips conditionals, swaps +operators, and deletes statements, then reruns the suite; a surviving mutant +is a fault the tests didn't catch. Use mutmut or cosmic-ray for Python, +Stryker for JS. High line coverage with a low mutation score means weak +assertions, not a tested codebase. + +Reach for this on critical logic where coverage looks reassuring but you +want evidence the tests would fail on a regression. It's a diagnostic, not a +gate on every change — mutation runs are slow. + +## Test Organization + +Typical layout: + +``` +tests/ + unit/ # One test file per source file + integration/ # Multi-component workflows + e2e/ # Full system tests +``` + +Per-language files may adjust this (e.g. Elisp collates ERT tests into +`tests/test-<module>*.el` without subdirectories). + +### Testing Pyramid + +Rough proportions for most projects: +- Unit tests: 70-80% (fast, isolated, granular) +- Integration tests: 15-25% (component interactions, real dependencies) +- E2E tests: 5-10% (full system, slowest) + +Don't duplicate coverage: if unit tests fully exercise a function's logic, +integration tests should focus on *how* components interact — not repeat the +function's case coverage. + +## Integration Tests + +Integration tests exercise multiple components together. Two rules: + +**The docstring names every component integrated** and marks which are real vs +mocked. Integration failures are harder to pinpoint than unit failures; +enumerating the participants up front tells you where to start looking. + +Example: + +``` +def test_integration_refund_during_sync_updates_ledger_atomically(): + """Refund processed mid-sync updates order and ledger in one transaction. + + Components integrated: + - OrderService.refund (entry point) + - PaymentGateway.reverse (MOCKED — returns success) + - Ledger.credit (real) + - db.transaction (real) + + Validates: + - Refund rolls back if ledger write fails + - Both tables updated or neither + """ +``` + +**Write an integration test when** multiple components must work together, +state crosses function boundaries, or edge cases combine. **Don't** when +single-function behavior suffices, or when mocking would erase the interaction +you meant to test. + +## Naming Convention + +- Unit: `test_<module>_<function>_<scenario>_<expected>` +- Integration: `test_integration_<workflow>_<scenario>_<outcome>` + +Examples: +- `test_cart_apply_discount_expired_coupon_raises_error` +- `test_integration_order_sync_network_timeout_retries_three_times` + +Languages that prefer camelCase, kebab-case, or other conventions keep the +structure but use their idiom. Consistency within a project matters more than +the specific case choice. + +## Test Quality + +### Independence +- No shared mutable state between tests +- Each test runs successfully in isolation +- Explicit setup and teardown + +### Determinism +- Never hardcode dates or times — generate them relative to `now()` +- No reliance on test execution order +- No flaky network calls in unit tests +- Time/clock-mocking helpers must avoid two recurring failure modes: + - *Infinite recursion.* The helper must not call the primitive it's + replacing. If the mock for `now()` calls `now()`, the test stack + overflows. Compute the mock value from a fixed source (a captured + instant, an injected fake clock). + - *Scope-shadowing without reach.* A mock that only exists inside + the test function won't affect production code that reads the + symbol through its canonical path. Replace the symbol at its + definition site (monkey-patch the module attribute in Python, + redefine the global in Lisp, swap the package-level binding in + Go, replace the named export in JavaScript) — or inject a fake + via dependency-inversion. Don't lean on scope-shadowing + primitives (Lisp `let`, Python local rebind, JS shadowed `let`) + that fence the mock to the test's lexical scope; production code + won't see them and the test passes against the real clock. + +### Performance +- Unit tests: <100ms each +- Integration tests: <1s each +- E2E tests: <10s each +- Mark slow tests with appropriate decorators/tags + +### Mocking Boundaries +Mock external dependencies at the system boundary: +- Network calls (HTTP, gRPC, WebSocket) +- File I/O and cloud storage +- Time and dates +- Third-party service clients + +Never mock: +- The code under test +- Internal domain logic +- Framework behavior (ORM queries, middleware, hooks, buffer primitives) + +### Signs of Overmocking + +Ask yourself: + +- Would this test still pass if I replaced the function body with `raise NotImplementedError` (or equivalent)? If yes, the mocks are doing the work — you're testing mocks, not code. +- Is the mock more complex than the function being tested? Smell. +- Am I mocking internal string / parsing / decoding helpers? Those aren't boundaries — they're the work. +- Does the test break when I refactor without changing behavior? Good tests survive refactors; overmocked ones couple to implementation. + +When tests demand heavy internal mocking, the fix isn't better mocks — it's +restructuring the code (see *If Tests Are Hard to Write* below). + +### Testing Code That Uses Frameworks + +When a function mostly delegates to framework or library code, test *your* +integration logic: +- ✓ "I call the library with the right arguments in the right context" +- ✓ "I handle its return value correctly" +- ✗ "The library works in 50 scenarios" — trust it; it has its own tests + +For polyglot behavior (e.g., comment handling across C/Java/Go/JS), test 2-3 +representative modes thoroughly plus a minimal smoke test in the others. +Exhaustive permutations are diminishing returns. + +### Test Real Code, Not Copies + +Never inline or copy production code into test files. Always `require`/`import` +the module under test. Copied code passes even when production breaks — the +bug hides behind the duplicate. + +Mock dependencies at their boundary; exercise the real function body. + +### Error Behavior, Not Error Text + +Test that errors occur with the right type; don't assert exact wording: +- ✓ Right exception type (`pytest.raises(ValueError)`, `(should-error ... :type 'user-error)`) +- ✓ Regex on values the message *must* contain (e.g., the offending filename) +- ✗ `assert str(e) == "File 'foo' not found"` — breaks when prose changes even though behavior is unchanged + +Production code should emit clear, contextual errors. Tests verify the +behavior (raised, caught, returned nil) and values that must appear — not the +prose. + +## If Tests Are Hard to Write, Refactor the Code + +If a test needs extensive mocking of internal helpers, elaborate fixture +scaffolding, or mocks that recreate the function's own logic, the production +code needs restructuring — not the test. + +Signals: +- Deep nesting (callbacks inside callbacks) +- Long functions doing multiple things ("fetch AND parse AND decode AND save") +- Tests that mock internal string / parsing / I/O helpers +- Tests that break on refactors with no behavior change + +Fix: extract focused helpers (one responsibility each), test each in isolation +with real inputs, compose them in a thin outer function. Several small unit +tests plus one composition test beats one monster test behind a wall of mocks. + +When the untestable function is legacy code you're hardening, this extraction +**is** the hardening — not a detour around it. A function whose boundary or +error case can't be exercised without mocking the world (a shell function that +calls `tmux`/`git` directly, a handler that reaches straight into I/O) can't be +characterized, so you can't refactor it safely and you can't pin its edge +behavior. Extracting the pure decision logic into a helper that takes plain +inputs and returns a plain result makes that logic characterizable with the full +Normal/Boundary/Error set; the I/O calls become a thin wrapper you cover once +with a single composition test. "It needs too much mocking to test" is therefore +never a reason to skip the boundary and error cases — it's the signal to reshape +the function so those cases are writable. + +## Coverage Targets + +- Business logic and domain services: **90%+** +- API endpoints and views: **80%+** +- UI components: **70%+** +- Utilities and helpers: **90%+** +- Overall project minimum: **80%+** + +New code must not decrease coverage. PRs that lower coverage require justification. + +## TDD Discipline + +TDD is non-negotiable. These are the rationalizations agents use to skip it — don't fall for them: + +| Excuse | Why It's Wrong | +|--------|----------------| +| "This is too simple to need a test" | Simple code breaks too. The test takes 30 seconds. Write it. | +| "I'll add tests after the implementation" | You won't, and even if you do, they'll test what you wrote rather than what was needed. Test-after validates implementation, not behavior. | +| "Let me just get it working first" | That's not TDD. If you can't write a failing test, you don't understand the requirement yet. | +| "This is just a refactor" | Refactors without tests are guesses. Write a characterization test first, then refactor while it stays green. | +| "I'm only changing one line" | One-line changes cause production outages. Write a test that covers the line you're changing. | +| "The existing code has no tests" | Start with a characterization test. Don't make the problem worse. | +| "This is demo/prototype code" | Demos build habits. Untested demo code becomes untested production code. | +| "I need to spike first" | Spikes are fine — under the protocol below. Throw the spike away, then write the first failing test before productionizing. | + +If you catch yourself thinking any of these, stop and write the test. + +### The Spike Exception (Disciplined) + +TDD stays the default. The one sanctioned way to write code before a test is +a spike — exploratory code that answers "is this approach even viable?" when +you can't yet write a meaningful failing test because the shape of the +solution is unknown. A spike is disciplined only when all three hold: + +1. **Timebox it.** Set a limit before starting (an hour, an afternoon) and + stop when it's up. An open-ended spike is just untested implementation + wearing a different name. +2. **Do not commit spike code.** The spike is a learning artifact, not a + deliverable. It never enters the branch history. Keep it in a scratch + file or a throwaway worktree. +3. **Throw the spike away, then start with a failing test.** Once the spike + has answered the viability question, delete it. Write the first failing + test against the now-understood behavior, then productionize under normal + Red/Green/Refactor. The production code is written test-first even though + the exploration wasn't — you don't promote the spike into production by + bolting tests on after. + +The spike buys understanding, not code. If you find yourself keeping the +spike because rewriting it feels wasteful, the timebox was too long or the +problem was tractable enough to TDD from the start. + +## Anti-Patterns (Do Not Do) + +- Hardcoded dates or timestamps (they rot) +- Testing implementation details instead of behavior +- Mocking the thing you're testing +- Mocking internal helpers (string ops, parsing, decoding) — those are the work +- Inlining production code into test files — always `require` / `import` the real module +- Asserting exact error-message text instead of type + key values +- Shared mutable state between tests +- Non-deterministic tests (random without seed, network in unit tests) +- Testing framework behavior instead of your code +- Ignoring or skipping failing tests without a tracking issue + +## Content scope + +Test code, fixtures, docstrings, and comments are checked into the repo and visible to the team. They must follow the *Content scope for public artifacts* rule in [`commits.md`](commits.md): no local paths, no private repo names, no personal tooling references. @@ -39,94 +39,445 @@ Tags are assigned and refreshed by =task-audit=; =task-review= keeps them honest * Rulesets Open Work -** TODO [#B] Sentry workflow — build from spec :feature: -Build the sentry supervisor workflow from the READY-gated spec: -[[file:docs/specs/2026-07-14-sentry-workflow-spec.org][sentry workflow spec]] (DRAFT as of 2026-07-14; needs spec-review first). -Four phases: agent-lock helper + bats, sentry.org engine, companion -reconciliations (knowledge-base.md, roam-sync.sh header, inbox.org, -triage-intake.org), live trial night on rulesets. Origin: work project's -proposal, [[file:docs/design/2026-07-14-sentry-workflow-proposal.org][docs/design/2026-07-14-sentry-workflow-proposal.org]]. All nine -design decisions resolved with Craig 2026-07-14 (recorded in the spec). - -** TODO [#C] Polyglot projects — supported, or refused? :spec: -SCHEDULED: <2026-07-20 Mon> -Do we support more than one language bundle per project? The honest answer today -is "partly, by accident." The collision guard added 2026-07-16 refuses a -*colliding* second bundle rather than silently replacing the first's config, but -a non-overlapping pair still installs fine: bash ships =settings.json= + -githooks and no coverage fragment, python ships only a coverage fragment, so -=bash= + =python= composes cleanly today and yields a real polyglot project with -both rule sets. So the line isn't polyglot-vs-not, it's overlap-vs-not — and -nobody chose that line, it fell out of which bundle happens to ship what. Origin: -home's report after scaffolding clock-panel with python + typescript, -[[file:docs/design/2026-07-16-polyglot-bundle-collision.txt][docs/design/2026-07-16-polyglot-bundle-collision.txt]]. +** DOING [#B] Hostile subagent review before every agent commit :feature: +:PROPERTIES: +:LAST_REVIEWED: 2026-07-28 +:END: +From the roam inbox, 2026-07-28: "code reviews must occur before every commit an agent does, and they should be hostile reviews from a subagent without the agent's context." -Pair this with the subproject scouting below — it's the same question in a -different costume ("which projects would actually be polyglot, and why"), so -they should be one conversation. +Two asks, and only the second is new. The publish flow already mandates a review before every commit (Step 1). What changes is *who reviews*: today the reviewing agent is the one that wrote the change, so it inherits the author's mental model, and =review-code= only *suggests* subagent dispatch, and only "for substantive reviews on large diffs". -The three options, in the order they'd be weighed: +Decisions settled with Craig, 2026-07-28, and shipped: +- *Scope* — every commit. The reviewer's own Phase 0 rules a diff trivial and returns Skipped, which satisfies the gate; the author never rules on their own diff. +- *Stance* — "adversarial", not "hostile" (Craig's call). An agent told to attack manufactures findings, so the stance carries a substantiation floor: a finding not substantiated against the diff is dropped. +- *What the reviewer gets* — the diff, a one-line claim of what it does, and the requirement source (ticket, plan, task body) where one exists. Withheld: the conversation, the exploration, the author's rationale. The requirement source stays *in* because it is the only artifact that can contradict the author's claim; withholding it makes the claim self-certifying. +- *Loop* — re-review until the reviewer approves, turning on blocking findings rather than the verdict token. Bounded at three rounds, and stopped early on a finding that recurs after being reported fixed. Both bounds hand the decision to Craig; the unattended callers park instead. +- *Adjudication* — Craig, never the author overruling the reviewer. +- *Home* — the =publish= skill Step 1, with =review-code= carrying the adversarial contract and re-review mode. +- *The =subagents.md= tension* — resolved with an Isolation Override section: the size heuristics assume the main thread could do the task equally well, and they lapse when its own context is what makes its answer untrustworthy. -1. *Unsupported, explicitly.* Keep the guard as the answer. Cheapest, and - matches how little polyglot exists (one project, clock-panel). -2. *Supported.* Needs per-bundle filenames, a merged =settings.json= (the hooks - arrays compose rather than clobber), composed githooks, and namespaced - Makefile targets with a =coverage= aggregate. This is the real work. -3. *Case-by-case.* Support the pairs that come up, refuse the rest. +Remaining: nothing on the design. The change shipped in this session. -What the decision needs to know: +** TODO [#B] wrap-org-table splits logical rows, and lint drives it :bug: +:PROPERTIES: +:LAST_REVIEWED: 2026-07-29 +:END: +Reported by work 2026-07-28 against =arch-00-deepsat-platform-spec-draft.org=. Reproduced here. -- *The target-name collision is the deeper half* (home's point, and it's right). - Every bundle's fragment defines =coverage:= and =coverage-summary:=, so even - with both files present a polyglot project can't paste both into one Makefile. - Renaming files doesn't fix it. -- *Only three of five shared filenames actually collide.* =gitignore-add.txt= - (5 bundles) appends deduped and composes. =CLAUDE.md= (3) is seed-only, and - its fallback comment shows multi-bundle was already considered there. - =claude/settings.json= (3), =githooks/*= (3), and =coverage-makefile.txt= (4) - are the real ones. -- *=FORCE=1= is a poor escape hatch* (home's catch): it also re-seeds - =CLAUDE.md=, which is destructive on a customized project. If polyglot - becomes supported, the override wants to be its own flag. +*** Verified -** TODO [#C] Subproject pattern — promote to claude-rules? :spec: -SCHEDULED: <2026-07-20 Mon> -home proposes promoting its subproject pattern (a former standalone project -folded into a parent, living as a self-contained subdir sharing the parent's -=.ai/= scope) into the rules layer: vocabulary, the read-first -=<subproject>/<subproject>-brief.org= convention, the parent-vs-subproject -content criterion ("one fact, one home"), and create/archive criteria. -Proposal + home's full instance: -[[file:docs/design/2026-07-15-subproject-pattern-proposal.org][proposal]], -[[file:docs/design/2026-07-15-subprojects-convention-home-instance.org][home's convention doc]]. +Each of these is a measurement, re-run under adversarial review. The analysis I built on top of them was wrong three times, so this section is deliberately separated from the open questions below. -Deferred 2026-07-16 rather than promoted. *Craig's framing:* he wants to scout -which projects would actually get subprojects, and why, before we shape a rule. -If he hasn't done that scouting by the time this comes up, offer to do it -together — brainstorm the candidates, then explore the reasons behind each. That -evidence decides it: either we drop the pattern, or we know enough to adjust it -so it's effective. Don't shape the rule before the scouting. +- *The defect.* =wrap-org-table.el= turns one logical table row into two or more, by writing a rule between its continuation lines. Content survives; structure does not. +- *Root cause.* =wot--continuation-group-p= (=wrap-org-table.el:168=) requires every line past the first to carry at least one empty cell. When a row overflows in every column its continuation line is fully populated, the predicate rejects the group, and =wot--logical-rows= appends each physical line as its own row (=:197=). +- *Controlled A/B.* Rules present in both, continuation line's middle cell the only variable: blank merges, populated splits. +- *Idempotence is broken.* Running the tool twice on its own correct output corrupts it. Pass 1 emits a properly rule-delimited three-line row; pass 2 splits it into three rows. The docstring at =:203-204= asserts the opposite, and =wot-reformat-is-idempotent= passes because its fixture overflows only one column. +- *lint doesn't just miss it, it causes it.* =lint-org.el:424= calls the same predicate. Given the tool's own correct output — a rule after every logical row — lint reports "missing rule between rows — wrap-org-table.el reflows it". Nothing is missing. Follow that advice and the tool splits the row; lint then reports the result 0 mechanical, 0 judgment. Control: a conformant table whose continuation keeps an empty cell returns 0 and 0. So the loop is not two independent green lights, it is the linter manufacturing a false violation and certifying the damage it caused. +- *A second path.* With no hlines at all, =wot--logical-rows= short-circuits (=:184=) before the predicate is reached, so every physical line becomes a row. +- *Nothing invokes the tool unattended.* =todo-cleanup.el= names it in comments only; the entry-script guard from the 2026-07-09 incident holds. +- *rulesets is exposed*: =todo.org= carries a four-row attachment-sanitization table with no rules between rows. Don't reflow it until the tool is fixed — reflowing is the trigger. -*Review findings from the 2026-07-16 pass* (the inputs the decision needs): +*** Two fixes that were verified to work -- *N=1.* home is the only project with subprojects, across all 27 =.ai= scopes; - its nine all came from the single 2026-06-11 fold. This is the fact the - scouting tests. -- *Placement contradicts the proposal's own principle.* =claude-rules/*.md= - loads into every session of every project. home's doc argues the always-on - layer is "a tax paid whether or not it's relevant today" and depth belongs - "one open away". At 282 lines the doc would be the third-largest rule and add - ~11% to the always-on layer, so every .emacs.d / takuzu / chime session would - carry a one-project convention. -- *Precedent for the shape:* =patterns.md= (29 lines, explicit "don't carry the - catalog in context") and =docs-lifecycle.md= (75 lines, depth in a spec). - Thin rule + on-demand depth is the established answer. -- *Dangling reference:* the doc cites =claude-rules/git-hosting-privacy-model= - as authority for its shared-scope-safety criterion. No such file exists — the - real content is the gitignore-vs-track and public-reachability decision in - =protocols.org=. Fix before any promotion. -- *Instance vs rule:* the metrics, self-improvement log, kill criteria, rollout - dates, and adoption table are home's instance, not rule content. +- *Conformant round-trip.* Replacing the predicate body with =(> (length group) 1)= — pure rule-delimited grouping, no emitter marker, no new field — makes the corrupting pass-1 output round-trip cleanly. +- *Telling a continuation group from two real rows at runtime.* Provenance isn't available (=wot-reformat-table-string= receives a string), but the test is cheap: merge the group, re-wrap at the allocated widths, compare to the group as given. A continuation group reproduces itself under merge-then-wrap; two genuinely distinct short rows don't, because merged they fit on one line. + +Start with a red test from the double-run repro — it needs no unusual input and falsifies the docstring and the passing idempotence test together. Emission is already correct (=:222-224= emits one rule per element of =rows=), so the repair is entirely in how =rows= is computed. + +*Prefer the round-trip check to any detector.* work implemented the predicate-based detection I recommended and demonstrated it cannot discriminate at any threshold (2026-07-29, worked examples from their repo). Run bare it flags 144 files — essentially every table — because an ordinary header-plus-body table is one multi-line fully-populated group and rejecting it is the predicate working correctly. Add a per-row-ruled precondition and it cuts to 9, but at 9 it still mixes a genuinely wrapped row (=arch-09:40=, a header row whose second line continues the sentence) with two distinct rows legitimately sharing a rule (their =todo.org:117=, where splitting is the *right* answer). Same structural signature, opposite correct verdicts. The difference is whether one line continues the other as prose, which is semantics and not in the parse. + +So =lint-org.el:424= cannot simply inherit the repair, and a checker keyed on the predicate would train people to ignore it. The idempotence property is the checkable one: reflow twice and diff. It sidesteps detection entirely, because round-tripping the tool's own output correctly is true by construction and needs nobody to decide what a group means. + +*** Open questions + +- *What should no-hline input do?* My "refuse to reflow" instruction was wrong: adding rules to a ruleless table is the tool's main job, and =test-wrap-org-table.el:147= asserts exactly that. The real danger is narrower — a no-hline table where some line carries an empty cell, so a physical line might be a continuation. Needs a decision on refuse, ask, or heuristic. +- *Which path bit work?* One question settles it: did the =arch-00= Document Status table have rules between its rows before the reflow? I inferred "probably secondary" from a file-level scan, which can't resolve a table-level incident. +- *How exposed is work?* Partly resolved by their own implementation, 2026-07-29. *Four secondary-path files are exact.* The primary-path number is 9 under a per-row-ruled precondition, which they correctly label a floor on a population they cannot cleanly define rather than a measurement — the precondition also drops the minimal =a_full= fixture, which has only two groups. My earlier "24 secondary-path files" certification was not sound: their original signature also matches every correctly-reflowed table, so it never partitioned. +- *Does the grade go up?* Currently Major x most users frequently = P2 = [#B] (any table you reflow that overflows every column, which is what a width violation sends you to the tool to fix). The lint-drives-it finding arrived after that grading and may lift it. + +*** Provenance + +work reverted their table and left it over-budget at 134 columns, correctly: an over-wide table is cosmetic, a table that says something else is not. I hit this same failure on 2026-07-27, wrote "reflowed a table into a worse shape and lint-org then passed it" into a session summary, and never filed it — which is why work met it three weeks later. Their framing beats the apology: a correct observation recorded and then read as fine is the same failure as a green check on a corrupted table. + +Process note for whoever picks this up. This task bounded out of its review loop at three rounds. Every finding across all three landed on inference, never on a measurement — the reproductions held throughout while the reasoning built on them was refuted repeatedly. Trust the Verified section; re-derive anything else. + + +** TODO [#B] Synced workflows link outside the synced set with ../../ :bug: +:PROPERTIES: +:LAST_REVIEWED: 2026-07-28 +:END: +Seven link sites across four synced workflows, three distinct targets, all escaping the =.ai/= boundary into rulesets repo-root paths. From a consuming project's =.ai/workflows/=, =../../= is that project's root, where none of these exist: + +- → =../../claude-rules/todo-format.md= (five sites): =open-tasks.org:163=, =task-audit.org:84=, =task-review.org:60=, =task-review.org:64=, =task-review.org:99= +- → =../../docs/design/task-review.org= (one site): =task-review.org:11= +- → =../../flush/SKILL.md= (one site): =suspend.org:22= + +Verified dead in both home and =.emacs.d=; they resolve only in rulesets, which is why nobody noticed. + +Grading: Minor severity (a documented reference an agent can't follow, workaround is to search) x every user every time (every consuming project, every sync) = P2 = [#B]. Same grade as the =references/= link this came from, and the same mechanism — a synced file linking a path the sync doesn't deliver — with seven sites instead of one. + +Fix direction, two halves. Rewrite the seven as prose references naming the file, the same move taken for the credential paths: a link that resolves in one repo shouldn't be a link in a file that ships to two dozen. Then close the gate, or the eighth arrives unnoticed. + +The gate already exists and is nearly right. =scripts/lint.sh='s =check_md_links= was written for this exact class — its comment says so ("Validate cross-references to =claude-rules/= — the install-layout problem"). It misses these for two concrete reasons: it matches only markdown link syntax (=grep -oE '\[[^]]*\]\([^)]+\)'=), so org =[[file:...]]= links are invisible to it, and its driver loop only feeds it =claude-rules/*.md= and the language rule files, never =.ai/workflows/*.org=. Extending it on both axes is a smaller and more durable fix than a manual sweep. + +Found by the adversarial reviewer on the =references/= fix, 2026-07-28, as the sibling class the original report missed. + +** TODO [#C] start-work Phase 7 still summarizes the old publish flow :chore:solo: +:PROPERTIES: +:LAST_REVIEWED: 2026-07-28 +:END: +=.claude/commands/start-work.md:336-339= hands off with "Follow =commits.md= exactly" and "Run =/review-code --staged= before each commit" — no isolated reviewer, no re-review loop. It was already stale before the 2026-07-28 review change, because =commits.md= moved the publish flow into the =publish= skill in an earlier commit, so the pointer names a file that no longer holds the flow. + +Grading: Cosmetic severity (a stale summary beside a correct canonical, and start-work is attended so the escalation target exists) x most users frequently = P3 = [#C]. + +Fix is to point Phase 7 at the =publish= skill rather than restate the flow, which is what let it drift in the first place. Worth a sweep for other files that restate the flow instead of pointing at it. + +Found by the adversarial reviewer during the 2026-07-28 review-flow change, and correctly filed rather than fixed there: the line was untouched by that diff. + +** TODO [#C] Two lint defects at the template source :chore: +:PROPERTIES: +:LAST_REVIEWED: 2026-07-28 +:END: +Both verified at the rulesets source, so every project seeded from them inherits the defect. + +=retrospectives/PRINCIPLES.org:38= violates the org-table standard (no closing rule). =lint-org= flags it as =org-table-standard=, and =wrap-org-table.el= reflows it mechanically. This half is purely tool-driven. + +=protocols.org= lints at 8 mechanical + 19 judgment =misplaced-heading= hits, all from Markdown =**bold**= in an org file, which org reads as a level-2 heading when it starts a line. 48 bold spans total, 14 of them line-initial. This half is not mechanical: converting to org =*bold*= is 48 edits, and =lint-org --fix= would rewrite the line-initial ones without knowing they are emphasis rather than headings. + +Grading: Cosmetic severity x every user every time = P3 = [#C]. It is most of the linter's noise on protocols.org, which is the real cost — noise that trains the reader to skip the report. + +Not =:solo:= — both files are synced templates. + +Source: winvm link-integrity pass, 2026-07-28. + +** DOING [#B] Finish context-engineering rightsizing :refactor: +:PROPERTIES: +:LAST_REVIEWED: 2026-07-27 +:END: +Started 2026-07-27 from three Anthropic posts Craig supplied. Always-loaded rules went from ~57,800 tokens to ~28,949, plus 13,461 path-scoped. Shipped: =paths:= frontmatter on the three file-type rules, the per-project generic-rule de-duplication, =commits.md= split into a 2,804-token invariant core plus the =publish= skill, =testing.md= split into a 347-word directive plus the =testing-standards= skill, the approval-gate signal fixed from =.ai/=-tracking to remote host, and the first-person directive. + +*What remains is Craig's decisions, not execution.* Each of these needs him: + +1. *C1 — =verification.md= (3,388 tok).* The Opus 5 guide says explicit verification instructions cause over-verification and should be removed. His standing direction is never guess, always check. My read: the honesty core (don't claim a green suite you didn't run) stays and shortens, the process injection (green baseline, suite-as-its-own-step) moves to the publish skill. His call — and it's the rule closest to a preference he's stated outright. +2. *=interaction.md= (3,828 tok, now the largest).* Carries genuinely universal output constraints (no popup menus, no reverse-video markup) plus the new peer-reasoning contract. Splitting it means deciding which parts must fire on every turn. +3. *The TDD rationalization table.* Moved to =testing-standards= rather than cut. The posts argue that kind of over-argument is counterproductive on current models. Deleting his defense against me skipping TDD is his call. +4. *D3 — the gate separation.* Which approval gates are preference (he wants to see what goes out under his name) versus guardrail (written when the worst case was worse). They read identically in the files; only he can tell them apart. Highest-leverage input remaining. + +*Do first:* the three docs in =working/context-engineering-rightsizing/= are one commit behind — they were corrected at d74d98d, before the two splits and the gate fix. Reconcile before deciding anything from them. + +Risk on the record: =testing.md='s margin is thinner than =commits.md='s. If =testing-standards= fails to trigger mid-test-writing the mocking-boundary rules are lost — a quality regression, visible in review, but a real bet where =commits.md='s moved half was purely procedural. + +** TODO [#B] Recurring-loop mechanics as a shared rule :feature: +:PROPERTIES: +:LAST_REVIEWED: 2026-07-27 +:END: +Work proposes promoting the pattern behind its 2026-07-27 auto-mode triage-intake into a standing rule covering all recurring agent tasks: fixed interval via =CronCreate= rather than dynamic self-pacing, a subagent per firing so raw scan output never reaches the orchestrator, silence with no heartbeat when subagent-backed, and accumulate-don't-mutate between closes with explicit "close the X" / "stop the X" verbs. Likely touches =triage-intake.org= auto mode, =inbox.org= monitor mode, and a new short rule in =claude-rules/= so individual workflows point at one definition of cadence, isolation, and silence. + +Points 1, 2, and 4 mostly promote existing practice. Point 3 changes documented behavior and needs a real decision. Three findings from the skeptical review, all to resolve before this ships: + +1. *Point 1 is too absolute.* Fixed interval is the right default, but not the right universal. A loop waiting on unpredictable external state (a CI run, a deploy queue) should pace to how fast that state actually changes, which is what dynamic scheduling exists for. Write it as "default to a fixed interval; use dynamic pacing only when polling external state whose timing you can't predict." +2. *Point 2 collides with a standing instruction.* Craig's harness prompt says not to spawn subagents unless he requested it. Making subagent-per-firing the standing pattern needs that reconciled explicitly — the honest reading is that asking for the loop *is* the request, and the rule should say so rather than leaving two instructions to fight. The isolation argument itself is sound and matches the Opus 5 guidance (delegate for genuinely independent, sizeable work). +3. *Point 3 has a silent-failure hole.* Removing the heartbeat means a loop that died looks exactly like a loop quietly finding nothing. "The subagent completing is proof it ran" only holds if a *failed* subagent still surfaces. Either keep a rare heartbeat (hourly, not per-fire) or specify that failure always breaks silence. Suppressing success is fine; suppressing failure is not. + +Also underspecified: what counts as "signal" for a subagent-backed loop. And the silent-until-signal spec (=docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=, IMPLEMENTED) documents the per-fire heartbeat, so it needs a superseding history line rather than a silent contradiction. + +*** 2026-07-27 Mon @ 16:55 Work accepted all three findings and sharpened point 3 + +Work agreed without pushback and corrected my either/or on point 3, which was too weak. Both mechanisms are needed, not one: a failed or hung subagent breaking silence covers a *scan-level* failure, but only a periodic heartbeat catches the *scheduler* dying — because in that case nothing spawns at all and there is no failure to report. Heartbeat rare (hourly), not per-fire. + +The live consequence makes it urgent rather than theoretical: =CronCreate= auto-expires recurring jobs after seven days, so any fully-silent loop is *guaranteed* to die quietly. Work's auto-triage loop is running under exactly that contract right now. + +Work also added the reason that makes point 2's carve-out principled and should go in the rule text: the subagent exists to keep N sources' worth of raw scan output out of the orchestrator across a multi-hour loop, not to parallelize. + +Craig's call on point 3 is pending; work is surfacing it and will send the answer. + +Source: work handoff 2026-07-27 14:51, reply 16:55. + +** TODO [#B] Sentry triage split — work mail and messengers in, personal mail out :bug: +:PROPERTIES: +:LAST_REVIEWED: 2026-07-27 +:END: +Craig's 2026-07-27 correction supersedes his 2026-07-21 ruling: sentry should scan DeepSat work mail and every active messenger source, excluding only personal email. =sentry.org= currently excludes all mail and all messengers (overview line and pass 3), so work's first 2026-07-27 fire reported a quiet pass while a Hayk DM and a Kostya PR-review request sat unread. Work's live at-prompt carries the override in the meantime; the canonical workflow, the source-activation probe, and the tests still encode the old policy. + +Grading: Major severity (the pass reports "nothing" while real work items sit — the failure is silent, which is what makes it worse than a visible miss) × most users, frequently (every fire in any project declaring a work-mail or messenger source) = P2 = =[#B]=. + +The blocker is not wording. The current rule excludes by *category* — "mail", "messenger" — and the new rule is a *work-vs-personal* split that category cannot express. Shipping general plugins are =cmail=, =personal-gmail=, =personal-calendar=, =telegram=, =github-prs=; there is no general work-mail plugin, so work's source is project-specific. Naming the personal plugins in a denylist fails open the moment a new personal source is added. Decide the classification mechanism first — the durable shape is a per-plugin eligibility declaration inside each plugin file, so a new plugin classifies itself and the probe reads the declaration rather than pattern-matching a name. + +Scope once decided: the =sentry.org= overview line and pass 3, the activation probe's "survives that exclusion" test, and the pass-3 tests. + +Source: work handoff 2026-07-27 06:21. + +** TODO [#B] Repository publish-lock for two sessions sharing one clone :feature: +:PROPERTIES: +:LAST_REVIEWED: 2026-07-27 +:END: +Craig approved this in archsetup on 2026-07-26 after weighing the competing approaches recorded there. Two agent sessions in one clone share the working tree *and* =.git/index=, so one can sweep the other's hunks into a commit; a pathspec commit doesn't help, because =git commit -- <path>= reads the current working-tree state for that path. The approved design: one repository-scoped publish lock keyed on the real Git common-directory path (so worktrees sharing a clone contend, and unrelated clones with the same basename don't), acquired before any publish-flow operation that can mutate the shared index and held across reconcile, staging, =/review-code --staged=, message approval, and =git commit=. Ownership tracks =AI_AGENT_ID=, then a stable harness id. The reviewed staged-tree fingerprint (=git write-tree= after the review) is re-compared immediately before commit; a lost lock, a vanished lock, or a changed fingerprint forces reacquire and a fresh staged review. Ordinary edits stay concurrent — the lock serializes only the publish mutation window. + +Skeptical review — the design is sound and the acceptance checks are testable. Three gaps to close during the build: + +1. *TTL sizing across a human wait.* The lock is held across the commit-message approval gate, which is an unbounded human pause. "Refresh on conversational re-entry" covers an agent that keeps working; it does not cover Craig stepping away for an hour mid-review. Size the staleness horizon for that, or refresh on a timer while the gate is open — sentry's TTL is nowhere near long enough. +2. *Blocked-session behavior is unspecified.* The proposal says a second session cannot stage or commit through the guarded flow, but not what it should then do — wait on =--wait=, defer and report, or stop and surface. Pick one and test it. +3. *Size.* This is a real build (key derivation, owner support in =agent-lock=, the fingerprint gate, the review-restart path, and the acceptance battery), not a wording change. + +Canonical touchpoints: =claude-rules/commits.md= (add the lock to the pre-commit flow; state explicitly that an approval waiver never waives the staged review, since the review is the gate that reads the hunks entering the commit), =claude-templates/.ai/scripts/agent-lock= (backward-compatible caller ownership — existing sentry and roam-write callers must stay green), and =claude-templates/.ai/scripts/tests/agent-lock.bats=. + +Non-goals from the approval: no per-session =GIT_INDEX_FILE=, no worktree requirement for live stow-managed dotfiles, no locking of ordinary edits, no replacement for remote pre-push reconciliation. + +Source: archsetup handoff 2026-07-26 10:55. + +** VERIFY [#B] Parked: reusable Claude-to-Codex MCP registry sync :spec: +:PROPERTIES: +:LAST_REVIEWED: 2026-07-25 +:END: +Two independent 2026-07-25 handoffs (work and home) report the same successful local migration: expose Claude's global =~/.claude.json= MCP registrations to Codex without duplicating secrets. Stdio definitions launch through a mode-0700 =claude-mcp-for-codex SERVER_NAME= wrapper that reads the mode-0600 Claude config at process start; Streamable HTTP entries remain native Codex entries; the loopback-only legacy Slack SSE endpoint bridges through =mcp-remote=. Both senders verified initialize + tools/list, not merely =codex mcp list=. + +Recommendation: accept the need, but spec it before adding an installer. "Every project" is misleading because these are machine-global registries; the reusable unit is a host-level, idempotent reconciler. It must: + +1. Preserve Codex-only and plugin-provided entries and distinguish active global registrations from inactive marketplace templates and broken project-local definitions. +2. Inventory and report without printing environment values, secret-bearing arguments, tokens, or auth material. +3. Back up and update atomically; preserve secure file modes; handle malformed input, name collisions, upstream removals, missing dependencies, and repeated runs. +4. Classify stdio, Streamable HTTP, and legacy SSE explicitly. Never register SSE as HTTP, and allow cleartext only for loopback or an equivalently trusted private endpoint. +5. Health-check each locally executable server with initialize + tools/list and report only redacted names/counts. Verify both machines and document restart/new-session plus OAuth reauthentication behavior. +6. Keep the Figma token finding separate: rotate it and move it out of command-line arguments without ever recording its value. + +The proposed Claude-memory migration auditor is related only by runtime portability and should be a separate task/spec. It should classify guidance, useful context, duplicates, stale/ephemeral material, and secret-bearing denylist entries; it must not bulk-copy generated memory files. + +No prepared diff: this is a shared configuration design decision, and the current local wrappers/configs remain machine-owned evidence rather than canonical source. Say "spec the MCP registry sync" to start the decisions walk. + +** VERIFY [#B] A SCAN FAILED source must not advance its sentinel (engine, not the plugin) +:PROPERTIES: +:LAST_REVIEWED: 2026-07-24 +:END: +Promoted to top-level 2026-07-28 when its parent closed — it is a separate engine question and would have been buried under a DONE parent. + +Secondary finding from the 2026-07-24 handoff, flagged for your judgment because it's *engine* behavior in =triage-intake.org=, not the telegram plugin. work reported that after a SCAN FAILED, "the marker still advanced" — and telegram is =:ANCHOR: none=, so something advanced a cursor it shouldn't have. The compounding harm: a source that reports SCAN FAILED but advances its window means the next sweep believes it already covered that window, so the blind-sweep hole persists across sweeps rather than self-healing. Not touched .emacs.d-side; needs a look at the engine's per-source last-run/anchor advance. + +** TODO [#B] cj-remove-block still can delete the WRONG cj block :bug: +:PROPERTIES: +:LAST_REVIEWED: 2026-07-24 +:END: +The 2026-07-24 range fix (17f5d48) closed the *over-deletion* case: a range spanning two blocks is now refused. It did NOT close the underlying class. The validator proves the range is a well-formed single block; it never proves it is the block that was *scanned*. + +An adversarial reviewer demonstrated this on the fixed code: passing the second block's range while intending the first deletes the second, exit 0, no warning. Drift by exactly one whole block still slips through — which is the failure the validator exists to prevent. + +Grading: Major severity (silent deletion of the wrong annotation in Craig's org files, no warning, success exit) x rare edge case (needs drift landing exactly on another well-formed block) = P3 = [#C]... graded up to [#B] because it is the *same* failure the fix was believed to have closed, so the current state carries false confidence. + +Fix direction needs a decision, which is why this is not :solo:. The range alone cannot identify the intended block, so the caller must assert something about content — a hash of the expected block, or the expected first body line, passed alongside the range and verified before deletion. That changes the CLI contract and the calling skill, so it is a design call. + +** TODO [#D] Atomic-write residuals in cj-remove-block :bug: +:PROPERTIES: +:LAST_REVIEWED: 2026-07-24 +:END: +Three narrower gaps a reviewer found in the atomic write, all verified, none fixed in the 2026-07-24 repair round: + +1. =shutil.copymode= runs before =write_text=, and writing clears setuid/setgid — a 2755 file comes back 755. +2. ACLs are not carried across, and copying the ACL mask into the group bits can *widen* group permission on a file carrying a named ACL entry. +3. Hard links are broken the same way symlinks were: =os.replace= gives the path a new inode, so a second hard link keeps stale content. Introduced by the atomic write itself (17f5d48), which did not exist on main. +4. No =fsync= before =os.replace=, so the "never a partial" guarantee holds against process failure but not a system crash. Two lines to close. + +Grading: Minor severity (each needs an unusual file mode, an ACL, a hard link, or a crash mid-write) x rare edge case = P4 = [#D]. + +** TODO [#D] Test suites leak backup files into shared /tmp :test: +:PROPERTIES: +:LAST_REVIEWED: 2026-07-24 +:END: +The elisp todo-cleanup suite writes roughly 81 backup files into the shared temp dir per =make test= run, named after randomized fixtures (=tc-test-XXXXXX.org.before-todo-cleanup.*=). 1598 had accumulated by 2026-07-24. Not data loss and not production-named, so nothing masquerades as a real backup and nothing is deleted — this is clutter that grows every run. + +The python sibling was fixed with an autouse fixture (f850ad6) that sets =TMPDIR= and overrides =tempfile.tempdir=. ERT has no autouse equivalent, so the elisp fix wants a load-time rebind of =temporary-file-directory= plus a =kill-emacs-hook= cleanup. + +Also unaddressed: =test-lint-org.el= scans a hardcoded =/tmp=, and =lint-org.el= hardcodes =/tmp/= in =lo--backup=, so that pair cannot be isolated the same way without touching the script. And =lint-org.el= carries the same second-resolution backup-overwrite flaw that todo-cleanup and cj-remove-block were fixed for. + +Grading: Minor severity (clutter in a directory cleared on reboot) x every user, every time (every suite run) = P2... graded [#D] on the no-harm read. + +** VERIFY [#B] Should rulesets run shellcheck on its own shell? +:PROPERTIES: +:LAST_REVIEWED: 2026-07-24 +:END: +Surfaced by the lint-coverage finding above, and a bigger question than that task should decide. + +rulesets ships shellcheck enforcement to *other* projects — the bash bundle's =githooks/pre-commit= scans staged shell, and =validate-bash.sh= blocks an edit that fails it. rulesets runs neither on itself. Its own =githooks/pre-commit= calls =sync-check.sh= and nothing else; =lint.sh='s =check_hook= validates only shebang and exec bit; =make test= has no shellcheck step. So the repo asks consumers for a standard it does not apply to its own shell. + +Why this needs your call rather than an overnight fix: turning shellcheck on today would surface the findings I dispositioned as false positives during this session's sweeps — =SC2094= on =install-ai.sh= and =sweep-gitignore-tooling.sh= (append-with-stat, not a read-write race), =SC2088= on =doctor.sh= and =audit.sh= (tilde in a *display* string, not a path), =SC2164= on =lint.sh= and =status.sh=. Enabling the gate means either fixing those or adding =# shellcheck disable== directives with justifications, and which of those you want is a preference, not a fact. + +Options: (1) shellcheck in =make lint= as a warning, (2) in =githooks/pre-commit= as a hard gate matching what consumers get, (3) leave it, on the grounds that the repo's shell is small and reviewed. My lean is 2, since the asymmetry is the odd part — but it is your repo's bar to set. + +** TODO [#C] Attachment filenames from email are only partly sanitized :bug: +:PROPERTIES: +:LAST_REVIEWED: 2026-07-24 +:END: +Both attachment writers derive on-disk filenames from attacker-controlled input (the =filename= an email declares for its attachment), and both sanitize incompletely — in *different* ways, so neither covers the other's gaps. + +=eml-view-and-extract-attachments.py= runs the name through =_clean_for_filename= but interpolates the *extension* raw: =name, ext = os.path.splitext(...)= then =f"{basename}-ATTACH-{clean_name}{ext}"=. =gmail-fetch-attachments.py='s =safe_filename= handles path separators and leading =..= and nothing else. + +Verified 2026-07-24 by probing both with the same adversarial set: + +| input | eml-view result | gmail-fetch result | +|------------------------+------------------------------------+--------------------| +| =report.pdf; rm -rf ~= | keeps =; rm -rf ~= in the filename | keeps it | +| =x.p\ndf= | keeps a literal newline | keeps it | +| =a.= + 300 chars | 314-char filename | unbounded | +| =../../../etc/passwd= | neutralized | neutralized | + +What this is NOT, checked so it isn't over-graded later: not remote code execution (files are written through Python =open=, never a shell) and not path traversal (=os.path.splitext= only returns an extension when the last dot follows the last separator, so =ext= can never contain a slash — the traversal case above is neutralized in both). The real harms are narrower: a newline in a filename breaks any downstream tool that reads the output directory as a newline-delimited list, and an unbounded extension exceeds the 255-byte filesystem limit so a crafted attachment aborts the extraction. + +Grading: Major severity (untrusted input reaches a filesystem name unsanitized, and the newline case breaks real tooling today) x rare edge case (needs an attachment name with unusual characters *after* the last dot, which ordinary mail doesn't produce) = P3 = [#C]. + +Fix direction: sanitize the extension with the same rules as the name, cap the *whole* filename rather than just the stem, and reject or replace control characters in both scripts. The open question is where the sanitizer lives — see the VERIFY below. + +** VERIFY [#B] One sanitizer or two for the attachment filename fix? +:PROPERTIES: +:LAST_REVIEWED: 2026-07-24 +:END: +Deferred from pass 12 on 2026-07-24: the bug above is well-specified, but *where* the shared sanitizer lives is a design call with real tradeoffs, and the unattended loop has no one to ask. + +The two scripts are standalone stdlib-only CLI tools with kebab-case names, so neither is importable as a normal module. Three options, none obviously right: + +1. *A shared helper module* in =.ai/scripts/=. Cleanest single source of truth, but it becomes another synced template file, and the kebab-named callers need =importlib= gymnastics to reach it — the same trick =route_recommend.py= already uses to load =inbox-send.py=, so there is precedent. +2. *Duplicate a small sanitizer in each script.* Zero import machinery and each tool stays self-contained, which is the current house style for these scripts. Costs drift, and sibling drift is exactly the defect class that produced three separate findings this session. +3. *Fix only the demonstrated gaps in place* (extension sanitizing in one, length and control chars in both) without unifying. Smallest diff, leaves the two sanitizers permanently different. + +I lean 1 given how much drift has bitten tonight, but it changes the shape of =.ai/scripts/= and that is Craig's call, not an overnight one. + +** VERIFY [#B] Parked: question-capture pattern — ask async, answer live (from archsetup, Craig's idea) +:PROPERTIES: +:LAST_REVIEWED: 2026-07-24 +:END: +What arrived: your own idea, relayed via archsetup's roam inbox. Drop a question (not a build task) into the roam inbox as a capture; the agent retrieves it during a sentry/inbox pass but does NOT auto-answer; it holds the question and answers it at the next live conversation, closing on your acknowledgement. Distinct from VERIFY: a VERIFY blocks the agent on your input, here the agent owes you the answer and you owe only the ack. The instance that spawned it: "why does the cursor not appear over the desktop when the world-clock wallpaper is on?", answered live in archsetup. + +Recommendation: adopt, as a small spec rather than a direct build. The value is real — a captured question has no home today (not a task, not quite a VERIFY), and decoupling async-capture from sync-answer stops the agent burning cycles guessing at something you'd rather discuss. It clears the value gate on your authorship. But the shape carries decisions I shouldn't guess, which is why it parks rather than lands: + +1. How is a question item identified — an explicit marker (a =:question:= filetag, a =Q:= prefix) or phrasing (ends in "?")? Phrasing is fuzzy (a real task can contain a question); a marker is reliable. My lean: explicit marker. +2. Where does the "answer owed" state live between capture and the live session? Options: the roam item stays put with an answered-pending tag, or it routes to the owning project as an "answer owed" item that startup surfaces. My lean: surface at startup, since that's the next live moment. +3. Does it warrant a new keyword (the inverse of VERIFY — agent-owes-answer) or a tagged VERIFY variant? The proposal draws the VERIFY distinction sharply enough that a distinct marker may be cleaner. + +Prepared artifact: none — this is a rule-design decision, not a mechanical edit, so there's no diff to stage. The proposal and the concrete instance are in [[file:working/question-capture-pattern/proposal-from-archsetup.org]]. +Say "spec the question-capture pattern" (answering 1-3, or leaving them to the spec's decisions walk) and it becomes a spec-create. + +** VERIFY [#A] Parked: account-binding guard for the personal-gmail triage plugin (from home) +:PROPERTIES: +:LAST_REVIEWED: 2026-07-23 +:END: +What arrived: home hit a real cross-account hazard on 2026-07-23 — a sentry triage fire used =mcp__claude_ai_Gmail= (bound to the DeepSat work account, name gives no hint) instead of =google-docs-personal=, pulled 201 unread work messages, and would have run every hygiene action against the wrong inbox. The fix adds one guard block to the =triage-intake.personal-gmail.org= Scan phase: confirm a sample result's =to:= is =craigmartinjennings@gmail.com= before classifying, and on mismatch or unavailability fall back to the local mu mirror rather than another MCP. + +Recommendation: accept as-is. The diff is a single clean addition between the promotions-filter warning and the maxResults-cap warning — nothing else in the plugin changes. The bug is reproduced and dated, the account→tool mapping was verified by home 2026-07-23 (google-docs-personal=personal, google-docs-work + claude_ai_Gmail=work-bound), and the mu-mirror fallback is the right escape (it's account-fixed by maildir, not by an ambiguous MCP name). I agree with home's judgment that cmail needs no analogous note — it's a bridge script fixed to c@cjennings.net by construction, no tool-binding ambiguity. + +Graded [#A] on severity alone, per the todo-format security/privacy carve-out: acting on the wrong person's mailbox is a privacy exposure, and one occurrence with work mail misrouted through a personal-hygiene close is a showstopper regardless of frequency. + +Prepared diff: [[file:working/triage-account-guard/proposed.diff]] — apply is mechanical (home's file becomes canonical) on your go. Companion note: [[file:working/triage-account-guard/companion-note-from-home.org]]. +Say "approve the parked account guard" (or adjust / reject) and it gets applied. + +** VERIFY [#B] Parked: term-translation-density voice pattern #48 (from work) +:PROPERTIES: +:LAST_REVIEWED: 2026-07-23 +:END: +What arrived: you proposed a new voice pattern in the work session — a sentence forcing the reader to translate more than one specialized term is too dense and gets rewritten. You supplied the rule text and a worked before/after (the ViT/VLM/detect-then-contextualize email sentence), and delegated number, placement, and mode-tag to me. + +Recommendation: accept, as #48, general mode. Your reasoning holds — it's a universal clarity rule in the Orwell/Plain English family, distinct from #7 (which words) and #30 (fragments), so it reads for anyone's prose, not just Craig's. The diff is prepared and verified: both files consistent, #47 intact, all mode enumerations and counts updated. + +The one wrinkle I resolved, worth a glance before you approve: general mode was defined as exactly "patterns #1-31," a contiguous block. A general pattern at #48 breaks that, so the prepared diff updates every "general walks #1-31" line to "#1-31 and #48" and notes the number is an artifact of when it was added, not a scope signal. The alternative was to scope it [prose · personal] to keep general clean, but that costs the pattern its reach over third-party prose, which is exactly where jargon density bites. I went with your general-mode call. + +Prepared diffs: [[file:working/voice-term-density/skill.diff]] and [[file:working/voice-term-density/profile.diff]] — apply is mechanical on your go. +Say "approve the parked term-density pattern" (or adjust / reject) and it gets applied. + +** VERIFY [#B] Parked: auto-empty working/ when a task closes (from work) +:PROPERTIES: +:LAST_REVIEWED: 2026-07-23 +:END: +What arrived: your roam capture relayed via work — every project carries working/ and temp/, and when a task completes its working/ files are automatically ("via a soft hook if possible") archived or moved to temp/. The sender flagged that most of this may already exist and asked for a check before treating it as net-new. + +Recommendation: accept the intent, reject the mechanism. Three of the four asks already shipped — working/ tracked from creation and temp/ gitignored (2026-07-20), and temp/ cleared at wrap (today, 10ea44b, from this same capture). Only the auto-empty is new, and I'd not build it as described. + +Two reasons. Moving a closed task's artifacts to temp/ destroys them, because temp/ is cleared at the next wrap, while working-files.md says those artifacts get renamed individually and filed flat into assets/ with meaningful names. And filing is deliberately a judgment step: it forces a review of each artifact's permanent value, which is what keeps assets/ searchable and catches the ones not worth keeping. Automating it removes exactly that review. Tonight is the live case — I closed four parked VERIFY tasks and filed their working/ dirs by hand, checking for inbound links and confirming the content stayed recoverable in git first. An automatic sweep would have destroyed four prepared diffs at the next wrap. + +The real gap is narrower: orphan detection exists only inside a sentry fire (pass 6), so a project that never runs sentry never gets the nudge. The prepared diff adds that detection to wrap-up as a report-only step, which serves the intent without the destruction. Verified against this repo's live tree. + +Prepared diff: [[file:working/working-dir-orphan-check/proposed.diff]] — apply is mechanical on your go. +Say "approve the parked orphan check" (or adjust / reject) and it gets applied. + +** TODO [#B] No two language bundles compose any more — polyglot is now refused :feature:spec: +:PROPERTIES: +:LAST_REVIEWED: 2026-07-23 +:END: +Completing the python and typescript bundles (the [#A] above, fixed 2026-07-23) had a consequence worth deciding on deliberately: all five bundles now ship =claude/settings.json= and =githooks/pre-commit=, so *every* pair collides and =install-lang= refuses the second without =FORCE=1=. + +This reopens a decision that was closed on a premise that turned out to be a bug. On 2026-07-17 the call was "polyglot: case-by-case, no option-2 machinery — bundles already compose; only coverage-makefile.txt collides, and it's a manual paste." Bundles appeared to compose only because python and typescript were missing the two files everything else collides on. The composition was the defect, not a design property. + +Live impact: =clock-panel= is a python + typescript project. It can install one bundle's hooks or the other's, not both. Today it has neither, so nothing regressed underneath it, but the polyglot path it would have wanted is now closed. =work= is python-only and unaffected. + +The real question is what a polyglot project should get. Both files are single-owner by construction: =settings.json= would need its =PostToolUse= arrays merged, and =pre-commit= would need each language's checks concatenated behind one secret scan (which is already identical across all five). Neither merge is hard; the reason it was never built is that nobody had a project that needed it. Now one does. + +Options, roughly: (1) build the merge — settings arrays union, pre-commit composes per-language phases; (2) split the shared half out, so the secret scan is one file every bundle sources and only the language phase is per-bundle; (3) leave =FORCE=1= as the polyglot answer and document that the last install wins; (4) do nothing, since only one project is affected. Option 2 is the one that would also have prevented the [#A] — a shared secret scan can't go missing from a bundle that never had its own copy. + +Not [#A] because nothing is currently broken: no project is running with a bundle it lost. Not [#D] because a live project wants it and the decision is now forced rather than hypothetical. + +Pinned by =scripts/tests/install-lang-collision.bats= so the trade can't drift back unnoticed. + +** TODO [#C] Two Signal-channel gaps: unbounded send, second account never received :bug: +:PROPERTIES: +:LAST_REVIEWED: 2026-07-23 +:END: +Both found 2026-07-23 reading =claude-templates/bin/agent-text= and =scripts/signal-receive.sh=. Shellcheck is clean on all four =claude-templates/bin/= scripts; these are logic gaps, not style. + +1. *The direct send has no timeout.* =agent-text='s relay path passes =-o ConnectTimeout=10= to ssh, but the local-account path calls =signal-cli send= with no bound at all. A stalled network or a lock contended by the receive timer hangs the call indefinitely. That matters more than it looks: agent-text is invoked *by agents*, so an unbounded call blocks the calling turn with no output, and the caller can't distinguish a hang from a slow send. The receive path already takes ~16s of wall clock every cadence, so the two contending on the same account is not hypothetical. Fix: wrap the direct send in =timeout=, matching the relay's bound, and let the existing non-zero path print the desktop-fallback message. + +2. *ratio's second Signal account is never received.* =signal-cli listAccounts= on ratio warns "Messages have been last received 8 days ago". Established by elimination, not inference: ratio holds two accounts (=+15045173983= pager, =+15103169357= Craig's personal), =signal-receive.sh= hardcodes the pager as its default and takes no others, and the timer's journal shows it draining the pager cleanly every 15 minutes (last run 2 minutes before the warning appeared). So the 8-day-stale account is the personal one, and nothing on the machine keeps it warm. + + Honest limit on item 2: the staleness is verified, the *harm* is not. That registration is described in the session history as note-to-self with no push, so it may be fine to leave cold, and Signal's own tolerance for a quiet linked account isn't something I established. Filed so the question gets answered rather than rediscovered. The script already accepts an account argument, so if it does matter the fix is a second timer instance rather than a code change — which makes it a dotfiles handoff (that repo owns the unit) with rulesets owning only the script. + +Grading: Minor severity for both (one is a latent hang in a tool with a documented fallback, the other a warning on an account nothing currently depends on) x most users, frequently (the stale warning is a standing condition on this host; the hang needs a stall) = P3 = [#C]. + +** TODO Manual testing and validation +*** Sentry — entry gates fire with Craig present +What we're verifying: the interactive entry gates stop for the right states and start the loop only on a clean, green baseline. +- On rulesets (ratio), with a clean tree and green suite, say "start sentry hourly". +- Confirm it reads =:COMMIT_AUTONOMY: yes= and doesn't decline. +#+begin_src sh :results output +# Dirty the tree on purpose, then re-arm — the dirty-tree gate should stop and offer finish/stash/rollback. +echo "# scratch" >> README.org +git diff --stat +#+end_src +- Re-run "start sentry"; confirm it surfaces the dirty README and the three numbered options rather than starting. +- Discard the scratch edit (=git checkout -- README.org=), re-arm, and confirm it reconciles, creates =sentry/$(date +%F)-$(uname -n)=, and arms the loop. +Expected: sentry declines on the dirty tree with numbered options, and on a clean+green tree it creates the host-suffixed branch and starts the hourly loop. +*** 2026-07-20 Mon @ 11:22:56 -0500 Sentry — one fire runs end to end, verified in live trial +Verified across the 8-fire live trial on =sentry/2026-07-19-ratio= (ratio, 2026-07-20). Each fire wrote one digest block with a ran-or-skipped line per pass (no silent gaps), parked judgment/destructive items under the =* Sentry approval queue= heading with their exact commands (never executed unattended), committed writing passes to the branch, left the tracked tree clean, and freed the single-runner lock between fires. Fires 1-2 productive (2 handoffs processed + 1 parked, 3 design considerations filed); fires 3-8 clean idle. Digest lives in the session anchor. +*** Sentry — wrap-up guard and stop-sentry +What we're verifying: wrap-up refuses while sentry is live, and stop-sentry cleanly ends the loop. +- With sentry live, say "wrap it up"; confirm it refuses with "sentry is active — say 'stop sentry' first." +- Say "stop sentry"; confirm the loop cancels and it offers the branch disposition (merge now / leave named) and the approval-queue walk. +Expected: wrap-up blocks on the held single-runner lock; stop-sentry cancels the loop and walks branch + queue disposition. +*** Sentry — morning branch review +What we're verifying: the morning teardown reviews and merges cleanly, nothing reached main unpushed. +- =git log main..sentry/<date>-<host>= and read the diff. +- Squash-merge what's wanted, delete the branch, revert any stale Emacs buffers. +- Confirm main was never pushed to during the night and the approval queue was walked. +Expected: the night's work is reviewable as one branch, merges by choice, and main stayed clean throughout. +*** Silent-until-signal — empty fire heartbeats, real item full turn +What we're verifying: an in-session monitor fire collapses to a one-line heartbeat when nothing changed, and still does the full turn on a real item. Covers sentry, auto triage-intake, and auto inbox-zero (spec: docs/specs/2026-07-20-silent-until-signal-monitors-spec.org). +- Run a sentry fire with nothing pending. Expected: one line, =sentry at HH:MM: nothing=, no per-pass digest block, tree clean. +- Run a sentry fire after planting an inbox handoff. Expected: the full per-pass digest block, the handoff processed or queued. +- Start "auto triage-intake" and let an empty sweep run. Expected: one line, =triage intake at HH:MM: nothing=, no three-section output. +- Plant a real change in a triage source, let the next sweep run. Expected: the full three-section output (deltas / unacked / timestamp). +- Start "auto inbox zero" with an empty roam inbox, let a cycle run. Expected: one line, =inbox zero at HH:MM: nothing=. +- Add a roam inbox item, let the next cycle run. Expected: the item summarized, filed, and queued. +Expected: every empty fire is a single labelled heartbeat; every fire with real work does its full turn unchanged. +*** Triage source activation — declared sources pulled, undeclared dormant +What we're verifying: triage-intake pulls only the sources a project declares (spec: docs/specs/2026-07-20-triage-source-activation-spec.org). +- In rulesets (no =:TRIAGE_SOURCES:=, no project plugin), run "triage intake". Expected: Phase 0 announces every general plugin as "inactive (undeclared)", loads zero sources, and the sweep no-ops. +- In a project with =:TRIAGE_SOURCES: personal-gmail cmail=, run "triage intake". Expected: Phase 0 loads personal-gmail and cmail as active general sources, names the rest inactive, and scans only the two. +- In a project with a project-specific plugin (=.ai/project-workflows/triage-intake.*.org=) and no declaration, run "triage intake". Expected: the project plugin is active by presence; general plugins stay inactive. +- Under sentry, confirm pass 3 (triage) probe-skips in a project with no active source, and runs where one is declared. +Expected: undeclared general plugins never scan; declared ones and project-specific ones do; sentry's triage pass activates only where a source is active. + +** TODO [#D] Document polyglot coverage-makefile namespacing :chore: +The one manual step the polyglot decision (2026-07-20) left: when a project installs two bundles that both ship =coverage-makefile.txt= (elisp/go/python/typescript), their =coverage:= / =coverage-summary:= Makefile targets duplicate. Document the fix — rename to per-language =coverage-<lang>:= targets and add a =coverage:= aggregate that depends on them — where a polyglot user would meet it (install-lang docs, or a short note in the bundle README / languages/ overview). Low urgency: clock-panel is polyglot today and unaffected because it has no assembled Makefile. Do it when the coverage-makefile fragments are next actually pasted, or opportunistically. + +** TODO [#D] Cross-host agent coordination — a lock across ratio and velox :feature: +Foundational primitive several deferred features wait on. =agent-lock= serializes agents on ONE host (its locks live on tmpfs, =$XDG_RUNTIME_DIR/agent-locks/=), so nothing coordinates the two daily drivers. When ratio and velox both act on shared state at once, there's no lock to stop them — the roam-write lock and sentry's single-runner lock are all host-local. Design a cross-machine lock (a lock node committed to a shared repo, a tailnet lock service, or a designated-primary election) that the host-local locks escalate to when the operation touches cross-machine-shared state (roam, a sibling-freshness push). Take up when cross-host contention proves real in practice, not speculatively. + +Blocks (the shared-prerequisite half of the sentry cluster): +- [[file:todo.org::*Cross-host roam conflict surfacing][Cross-host roam conflict surfacing]]. +- [[file:todo.org::*Sentry vNext passes][Sentry vNext passes]] items #2 (system-health) and #3 (sibling-freshness) — both need ratio/velox not to run the pass simultaneously. ** TODO [#D] Cross-host roam conflict surfacing :feature: Deferred from the sentry spec (Decision: host-local lock only). The @@ -134,87 +485,67 @@ roam-write lock serializes same-host agents; a true cross-host race still forks a sync-conflict file, surfaced only by roam-sync's rebase abort. Investigate making the conflict loud (roam-sync retry-once, or a cross-host lock node in the repo) when it proves frequent in practice. +Shared prerequisite: [[file:todo.org::*Cross-host agent coordination][Cross-host agent coordination]] (the missing cross-machine lock). ** TODO [#D] KB lesson-detection heuristic :spec: Deferred from the sentry spec review (2026-07-14): sentry's KB promotion pass was cut from v1 because "promote recent durable lessons" had no defined lesson source, and an unattended judgment pass writing to the shared KB needs one that doesn't spam or go silent. Design the detection heuristic — candidate source: Session Log entries and memory-dir changes since the last fire, judged against knowledge-base.md's inclusion bar — and define how precision gets verified before the pass ships in a sentry vNext. Blocks re-adding the pass. -** TODO [#D] Sentry unattended /schedule variant :feature: -Deferred from the sentry spec (Non-Goal). The v1 contract is interactive: -a live session Craig starts, /loop-driven. A cron variant firing with no -session needs its own contract (mutation rights, async surfacing, dedup -state across runs) — same open questions inbox.org logs for unattended -auto inbox zero. Take up only if the interactive shape proves too narrow. +** TODO [#D] Unattended /schedule cron contract — no-session variant :feature: +The design problem shared by two deferred features: a =/schedule= cron pass that fires with *no live session* needs its own contract, distinct from the interactive =/loop= shape. The open questions are the same for both consumers — mutation rights (read-only vs may-mutate =todo.org= / =~/org/roam/inbox.org=), how a find surfaces asynchronously when Craig isn't at the session, how dedup state persists across runs that don't share a session, and what session/auth context a cron run carries (the MCP-auth wall: a detached run can't reach Gmail/Slack/Linear, the same constraint the silent-until-signal spec turned on). + +Two consumers, one contract: +- *Sentry* (deferred from the sentry spec Non-Goal): v1 is interactive, a live session Craig starts, /loop-driven. A cron variant firing with no session is out of v1 scope. +- *Auto inbox zero* (vNext from the inbox-consolidation spec, Codex finding 1; [[id:a7fe2a10-dfa8-4ba3-a11a-e7b1288b7573][spec]]): v1 =auto inbox zero= is the interactive /loop check that waits for Craig's yes. Its "design after v1 consolidation lands" precondition cleared 2026-06-28 (the inbox engine consolidation 24ca58d and monitor-inbox loop edb545d shipped), so this is actionable backlog. -** TODO [#C] ai launcher hardening — bug hunt + refactor pass :refactor:solo: +Design the one contract; both features consume it. Merged 2026-07-20 from the separate "Sentry unattended /schedule variant" and "Fully-unattended scheduled inbox check" tasks — same no-session design problem. + +** TODO [#D] Research: MCP for device locations shared with you :feature: +From .emacs.d (2026-07-20, rulesets-owned research). Is there a Google Maps MCP (or similar) that reports the locations of devices sharing their location with you? If none exists, research how hard it would be to build one. (Google's location-sharing has no official public API; likely needs investigation of unofficial routes or a different provider.) + +** TODO [#C] Sentry vNext passes — from live-trial design input :feature:spec: :PROPERTIES: -:CREATED: [2026-07-13 Mon] -:LAST_REVIEWED: 2026-07-13 +:LAST_REVIEWED: 2026-07-25 :END: -From the roam inbox (2026-07-13): find bugs in =bin/ai= until no more are visible, then refactor until no worthwhile opportunities remain. The launcher gained =--runtime= + =--print-launch= + 6 bats tests today (04c3b29); this pass extends coverage over the legacy paths (multi_mode, sort_windows, git prep, window sorting) and cleans as it goes. The runtime-selection asks from the same capture are tracked on the generic-agent-runtime parent (claude/codex shipped; ollama/qwen pending the model-floor eval); an interactive runtime picker is fair game here if it stays lightweight. +Three considerations captured by .emacs.d's own hand-run sentry trial (routed via the shared roam inbox, 2026-07-20), for folding into the sentry workflow. Design input, needs deliberation — not applied to the shared workflow unattended. + +1. Bug/enhancement logging pass. Only where the project owns a codebase: file bugs by default, enhancements only if asked; review-and-accept/decline during the morning reconciliation. Live-tested — .emacs.d ran exactly this by hand and logged findings to its session anchor. Strongest candidate of the three; overlaps the deferred KB lesson-detection pass ([[file:todo.org::*KB lesson-detection heuristic][KB lesson-detection heuristic]]). +2. System-health pass. Decide which checks are safe unattended, and how to keep ratio and velox from running it simultaneously (a cross-driver lock/coordination question — agent-lock is host-local tmpfs, so it doesn't coordinate across machines). +3. Sibling-machine freshness pass. Keep the other daily driver (velox/ratio) up to date during sentry (see daily-drivers.md; same cross-driver coordination question as #2). + +Take up with the sentry Living Document refinements once the trial has quiet nights behind it. Passes #2 and #3 are blocked on [[file:todo.org::*Cross-host agent coordination][Cross-host agent coordination]] (the missing ratio↔velox lock). Related: [[file:todo.org::*Unattended /schedule cron contract][Unattended /schedule cron contract]] (the no-session variant). ** TODO [#B] Extend ui-prototyping rule with build-to-prototype :feature: :PROPERTIES: :CREATED: [2026-07-11 Sat] -:LAST_REVIEWED: 2026-07-13 +:LAST_REVIEWED: 2026-07-25 :END: .emacs.d proposal (2026-07-11, Craig-approved promotion), extending the ui-prototyping rule shipped tonight (=claude-rules/ui-prototyping.md=, 53f6ce6). The base rule covers research → prototypes → iterate → decisions-backed-by-a-prototype, but not the "build to the prototype" half. Add: after prototyping, fold what settled back into the spec and make the *build target* the prototype rather than the original spec text — the built feature should match the prototype, and any deviation is documented in a "Prototype & deviations" addendum section the build keeps current. Wire into four touch points: =brainstorm= Phase 3 (a UI design isn't "accepted" until it's been through the prototype loop; Next Steps say "build to the prototype"), =spec-create= (emit the deviations-addendum section for UI specs), =spec-response= (a UI spec decomposes into a prototype loop first, then build-to-prototype tasks), =start-work= (its verify phase drives the UI end-to-end; the bar becomes "matches the prototype," deviations logged). Worked example: the takuzu Emacs game — colored tiles read as all-black in the real terminal frame (dark faces + GUI-only box cursor), caught by a prototype loop on the first screenshot where build-to-spec would have shipped an unusable board. Multi-asset synced-rule change, so review-gated and needs a focused session; decide brainstorm-vs-lifecycle placement with Craig. Source: =inbox/PROCESSED-2026-07-11-0222-from-.emacs.d-ui-prototype-rule-proposal.org=. ** TODO [#C] Roam-only startup for .ai projects — investigate :spec: :PROPERTIES: :CREATED: [2026-07-11 Sat] -:LAST_REVIEWED: 2026-07-13 +:LAST_REVIEWED: 2026-07-25 :END: Open question from the roam inbox (2026-07-11): could the startup sequence for .ai projects use org-roam as the single store instead of local files? Potential gain is near-guaranteed shared information across projects — lessons and proven techniques on a common thread, scannable across similar projects. It would need a way to isolate a project from the rest. Weigh the pros/cons, the risk, and whether it's worth it before any build. Exploratory, no commitment yet. ** TODO [#C] Keep WIP from blocking the template sync gate :feature: :PROPERTIES: :CREATED: [2026-07-11 Sat] -:LAST_REVIEWED: 2026-07-13 +:LAST_REVIEWED: 2026-07-25 :END: From the roam inbox (2026-07-11): work in progress in one project shouldn't stop the sync gate. Idea: keep all diffs/changes in a =working/= directory and exclude it (and its subdirectories) from the sync gate. Many projects run at once, so their WIP files need to be grouped. Also add a per-project count of when the gate tripped, tracked as a metric to investigate. Distinct from the 2026-07-02 policy (untracked/gitignored changes already pass — this is about *tracked* WIP under =working/=). Verify how the gate detects dirtiness today before designing. -** DONE [#C] Put install-ai on PATH, launchable as =install-ai= :chore:quick:solo: -CLOSED: [2026-07-18 Sat] -:PROPERTIES: -:CREATED: [2026-07-11 Sat] -:LAST_REVIEWED: 2026-07-13 -:END: -From the roam inbox (2026-07-11): make install-ai launchable as =install-ai= (no =.sh=) from PATH. dotfiles needs a copy that stays in sync with the rulesets canonical — decide whether the startup script-sync already covers it or a dedicated mechanism is needed. - -Resolved: added =claude-templates/bin/install-ai=, a thin launcher that resolves its own path through the symlink chain and execs =scripts/install-ai.sh=. =make install='s existing bin loop symlinks it into =~/.local/bin/install-ai= (same mechanism as =ai= and =agent-page=), so no dedicated sync and no dotfiles copy — the symlink always points at the canonical. 3 launcher bats added (incl. symlink-invocation resolution). Verified live: =install-ai --help= runs from PATH. - -** TODO [#B] Document (and own) the Signal pager :feature:spec: -:PROPERTIES: -:CREATED: [2026-07-11 Sat] -:LAST_REVIEWED: 2026-07-13 -:END: -home retired the ntfy phone-notification channel (phone-notify/phone-recv, self-hosted ntfy on ratio) on 2026-07-04 in favor of paging over Signal, and tore ntfy down. The Signal pager it replaced ntfy with is undocumented: no pager script in =~/.local/bin=, and =notify= doesn't reference Signal. What exists on ratio: signal-cli 0.14.5, account 404211. Deliverable: a documented Signal pager (send + read-replies), the signal-cli setup/account notes, and the sync path — the Signal equivalent of the retired ntfy runbook. Cross-machine tooling, so canonical home + docs belong in rulesets. - -RECONCILE FIRST: =protocols.org= "Paging Craig" already documents an agent-paging path via the *signal-mcp* tool (=send_message_to_user=, pager account +15045173983, Craig's UUID =b1b5601e-…=, verified 2026-06-30). home's handoff is about a *different* mechanism (signal-cli / account 404211, home's own paging). Decide whether these are one channel or two, and whether the signal-cli side needs its own runbook or should route through signal-mcp. Source: home handoff 2026-07-04 (=inbox/2026-07-04-1302-from-home-task-for-rulesets-document-and-decide.org=). Successor to the 2026-06-17 two-way-comms proposal. - -*** 2026-07-13 Mon @ 05:16:37 -0500 Folded home's ownership ack + current-state report into the reconcile scope -home confirmed (2026-07-11 reply) rulesets owns this task and the two-paths reconciliation is the right first step. New facts for the reconcile: from a home session on 2026-07-09, signal-mcp was NOT connected, and the local signal-cli is registered as Craig's own number — =send --note-to-self= returns a message id but produces no phone push. So home currently has no working ad-hoc page channel at all; whatever the runbook lands on must give home a live path. - -*** 2026-07-13 Mon @ 14:40:00 -0500 Added runtime-portability as a second motivation (Craig approved) -The MCP portability inventory ([[file:docs/design/2026-07-13-runtime-portability-inventories.org]]) found signal-mcp exists only claude.ai-side — no local config anywhere — so a non-Claude agent (Codex-style or local LLM) has no paging path at all. The signal-cli runbook this task produces is therefore also the runtime-neutral page channel, not just home's replacement for ntfy. - -*** 2026-07-13 Mon @ 18:30:19 -0500 agent-page shipped — every project now knows both channels -Craig's call: make paging universal and rename it the agent pager. NEW =claude-templates/bin/agent-page= (runs signal-cli directly on velox, ssh-relays from anywhere else, desktop-fallback hint on failure; 4 bats tests; live-verified from ratio through the real script). protocols.org "Paging Craig" rewritten around the two channels (notify desktop + agent-page phone; signal-mcp demoted to a velox-local nicety); page-me.org gained the phone section + fire-both guidance; work-the-backlog's end-of-set page names both surfaces; INDEX updated. Every project inherits via the startup sync + make install. Remaining here: the runbook proper, the receive timer, ssh-only vs linked-device. - -*** 2026-07-13 Mon @ 18:13:07 -0500 RECONCILED — one channel, on velox; live page verified end to end -The two-paths question is answered: there is ONE pager identity, +15045173983, registered in velox's signal-cli (account file 465310) — and signal-mcp is a locally-configured MCP server in velox's global ~/.claude.json (not claude.ai-side as the 14:40 entry inferred; it's just invisible from ratio, which is why home and this session couldn't find it). ratio's signal-cli holds only Craig's personal number (note-to-self, no push). Verified live today: =ssh velox 'signal-cli -a +15045173983 send -m … b1b5601e-6126-47f8-afaa-0a59f5188fde'= buzzed Craig's phone — his remembered CLI page was this same account on velox. Reliability findings for the runbook: both accounts throw receive-staleness warnings (velox 40 days, ratio 26; the signal protocol wants regular receives — a systemd receive timer on velox is the roam-sync-shaped fix), and the channel requires velox to be up. Remaining deliverables sharpened: the runbook (send + read-replies + receive timer + account notes); decide ssh-over-tailnet-only vs registering ratio as a linked device of the pager account; update protocols.org "Paging Craig" (it names signal-mcp as the only supported path — true only on velox; the ssh recipe is the cross-machine path) — shared-asset edit, own review pass. Interim recipe sent to home so it's unblocked today. - ** TODO [#C] KB orphan-node review pass :chore: :PROPERTIES: :CREATED: [2026-07-01 Wed] -:LAST_REVIEWED: 2026-07-13 +:LAST_REVIEWED: 2026-07-25 :END: The 2026-07-01 kb-hygiene report listed 42 agent KB nodes with no inbound id: links (of 53 agent nodes; 0 conflicts, no duplicate titles). Orphan-ness alone isn't a defect — agent nodes are found by rg, not only by links — but a periodic pass is worth doing: prune nodes that aged out, merge near-duplicates, add id: links where clusters exist. Regenerate the list with the kb-hygiene script rather than trusting the snapshot. Propose deletions/merges to Craig before applying (auto-cleanup allowed only for :agent:-tagged nodes after approval, per knowledge-base.md). ** TODO [#B] Helper-agent instance support — concurrent same-project Claude :feature:spec: :PROPERTIES: :CREATED: [2026-06-11 Thu] -:LAST_REVIEWED: 2026-07-13 +:LAST_REVIEWED: 2026-07-25 :END: SPEC REVIEWED 2026-06-12: [[file:docs/design/2026-05-28-generic-agent-runtime-spec-review.org][Codex review]] now rates Phase 1.5 =Ready with caveats=. Before any build, keep the Emacs integration as a cross-project handoff to =~/.emacs.d=, preserve the three-ring gate (bats → sandbox drills → pilot project), and do not let startup/helper changes reach synced template paths until the live drills pass. @@ -257,7 +588,7 @@ Craig's call (2026-06-24): helper-instance is independent of the generic-runtime ** TODO [#B] Wrap-up routing — manual end-to-end validation :test: :PROPERTIES: -:LAST_REVIEWED: 2026-07-13 +:LAST_REVIEWED: 2026-07-25 :END: What we're verifying: a real keeper routes through a live wrap and the destination actually files it. The task-routing build shipped IMPLEMENTED 2026-07-04 (spec [[id:00b47414-2213-4a99-be35-48ceb266fc08][wrapup-routing]]); this confirms it works end to end across a real cross-project wrap. A failed check promotes to a bug. - In a project session, let process-inbox file a handoff whose home is a different project; confirm the local task carries =:ROUTE_CANDIDATE: <dest>=. @@ -289,12 +620,8 @@ The Codex entry-file pointer now exists: =claude-templates/AGENTS.md= (thin poin :END: Three flashcard-tooling tasks that all edit =flashcard-to-anki.py= and/or =flashcard-stats.py=, grouped so they get built together instead of colliding on the same files (prior sessions flagged the conflict risk). The Anki =#+TITLE= deck-name fix already landed (commit 060a938), so any preserved pre-fix script copy gets re-derived against the current canonical, never copied wholesale. The three children each ship independently. -*** TODO [#C] apkg → org-drill converter :feature:solo: -:PROPERTIES: -:CREATED: [2026-06-22 Mon] -:LAST_REVIEWED: 2026-06-24 -:END: -Inverse of =flashcard-to-anki.py=: read an Anki =.apkg= (zip → =collection.anki2=/=.anki21= sqlite) and emit an org-drill =.org= in the house canonical shape. Recovers orphaned decks (=deepsat-fundamentals.apkg= has no saved =.org= source) and enables phone→org round-trip. Mapping: deck name → =#+TITLE=; each note → =** <Front> :drill:= with Back as body; card tag → =* Section= grouping (best-effort); Back HTML → org (=<br>= → newlines, unescape entities, strip =<hr id="answer">=); fresh =:ID:= UUID per card. Edge cases for tests: multiple decks per apkg, non-basic note types (skip/warn), HTML entities, empty back, media refs, =.anki2= vs =.anki21= schema. Lives beside the flashcard-* family in =claude-templates/.ai/scripts/= (a new file must be built in canonical — downstream =.ai/scripts/= is wiped by startup =--delete=). PEP 723 uv-run, stdlib =zipfile= + =sqlite3= (no genanki for reading). Acceptance: round-trip a known org-drill source through =flashcard-to-anki.py= then back, assert cards match. Build request: [[file:docs/design/2026-06-21-apkg-to-orgdrill-buildreq.org][buildreq]]. Backlog, not urgent. From home 2026-06-21. +*** 2026-07-19 Sun @ 18:40:36 -0500 Built apkg-to-orgdrill.py, the inverse converter +Commit a143679. Reads an Anki =.apkg= (stdlib =zipfile= + =sqlite3=, no genanki) and emits org-drill =.org= in the house shape: deck → =#+TITLE=, note Front → =** heading :drill:=, Back HTML → body (=<br>=→newlines, entities unescaped amp-last, answer =<hr>= stripped), tag → =* section=, fresh =:ID:= per card. 17 tests incl. the round-trip through =flashcard-to-anki.py='s own parse(); verified end-to-end against a real genanki apkg. Grounded the schema by generating + inspecting a real apkg first. Speedrun task 1 of 2. *** TODO [#C] flashcard-stats refutation / claim-prompt mode :feature: :PROPERTIES: @@ -305,24 +632,8 @@ A refutation card (heading is a bare false claim, body is the rebuttal) is valid Design (Craig, 2026-06-28, supersedes the two proposed options): make the exemption *generic*, not refutation-specific — more card kinds like this will come. When the org header declares the relevant info, the gate honors it rather than blocking. So a general file-level header keyword (a card-kind / check-exemption declaration) tells =flashcard-stats.py= which checks not to apply, instead of a hardcoded =#+DECK_KIND: refutation= keyword or a per-card =:claim:= tag. Document the mechanism in =flashcard-review.org= and add tests (a header-declared exemption file passes despite declarative headings + claim/answer overlap). Edits =flashcard-stats.py= — coordinate with the multi-tag reconcile, same file. Proposal: [[file:docs/design/2026-06-21-flashcard-stats-refutation-proposal.org][proposal]] (its two-option fix is superseded by this generic header approach). Backlog. From home 2026-06-21. -*** TODO [#C] Reconcile flashcard multi-tag tooling into canonical :chore:quick:solo: -:PROPERTIES: -:CREATED: [2026-06-20 Sat] -:LAST_REVIEWED: 2026-06-24 -:END: -The work project edited two synced scripts locally as a stopgap (2026-06-17) and asked rulesets to fold them into the canonical so the next sync doesn't revert them. Preserved bundle: [[file:docs/design/2026-06-17-flashcard-multitag-note.md][note]], [[file:docs/design/2026-06-17-flashcard-multitag-to-anki.py][to-anki.py]], [[file:docs/design/2026-06-17-flashcard-multitag-stats.py][stats.py]]. Change: support a second org tag on drill headings (=:fundamental:drill:=) for curated subset decks. =flashcard-to-anki.py= — broaden =CARD_RE= to match a trailing tag block (a heading is a card when =drill= is among its tags), bound the card body by any L1/L2 heading, add =--tag-filter <tag>= (emit only cards carrying that tag) and =--guid-salt <s>= (separate GUID space so a subset deck imports non-empty without disturbing the full deck's SRS state). =flashcard-stats.py= — same =CARD_RE=/=HEADING_RE= broadening plus a drill-membership guard. Use the preserved to-anki.py (the 0953 version: dropped an unused =heading_tags()= helper, tightened =CARD_RE= =(.*?)=→=(.+?)= for parity with stats). Apply to both =.ai/scripts/= and =claude-templates/.ai/scripts/=, add a multi-tag bats case to =flashcard-sync.bats= (a =:foo:drill:= heading parses; =--tag-filter foo= returns only those), verify the full deck still parses to 465 and =--tag-filter fundamental= returns 100, then sync-check + make test. Shared-asset change, so review-gated. - -Note (2026-06-24): the Anki =#+TITLE= deck-name fix landed (commit 060a938) — =default_deck_name= is now =default_deck_name(input_path, org_text)= with a new docstring. The preserved 2026-06-17 =to-anki.py= predates that, so *don't* copy it wholesale (it would revert the title-fix). Re-derive the multi-tag changes against the current canonical =flashcard-to-anki.py= and keep the =#+TITLE= behavior. - -** DONE [#C] coverage-summary.el documented as a local-only helper :chore:quick:solo: -CLOSED: [2026-07-18 Sat] -:PROPERTIES: -:CREATED: [2026-06-22 Mon] -:LAST_REVIEWED: 2026-07-13 -:END: -The elisp bundle installs =coverage-summary.el= into =.claude/scripts/=, gitignored in code projects, so CI can't run =make coverage-summary= against it. Decision (Craig, 2026-06-28): keep it in =.claude/scripts/= and document it as a local-only helper — don't ship it to a tracked =scripts/= dir, don't expect CI to run it. Remaining work (docs only, no move): state the local-only status in the script's header comment and wherever =make coverage-summary= is described, so the gitignored install reads as intentional rather than a gap. Note: emacs-wttrin rewrote its copy's header to claim a tracked =scripts/= home, which now contradicts this decision and should be reverted on their side. Surfaced 2026-06-21 during the coverage-summary autoloads bugfix (commit fb86736). - -Resolved: documented the local-only status in the =coverage-summary.el= commentary header and in =elisp-testing.md='s "Measuring it" section — the gitignored install now reads as intentional, not a coverage gap. Sent emacs-wttrin a handoff to revert its contradicting header claim. +*** 2026-07-19 Sun @ 18:41:00 -0500 Reconciled the flashcard multi-tag tooling into canonical +Commit a14e43b. Broadened =CARD_RE= in both =flashcard-to-anki.py= and =flashcard-stats.py= so a heading is a card when =drill= is among its tag block (=:fundamental:drill:=), bounded card bodies by any L1/L2 heading (=HEADING_RE=), and added =--tag-filter= (subset emit) + =--guid-salt= (distinct GUID space) to to-anki, plus a drill-membership guard to stats. Re-derived against the current canonical rather than copying the preserved 2026-06-17 files, which predated the =#+TITLE= fix (060a938) and would have reverted it. parse()'s 3rd element is now the Anki-tag list; existing parse tests moved to that contract, new tests cover multi-tag / tag-filter / guid-salt / stats count, and a =flashcard-sync.bats= case guards the count. The real 465/100 check needs the work deck; verified end-to-end on a synthetic deck (2 cards → 1 with =--tag-filter fundamental=) through real genanki. Speedrun task 2 of 2. ** TODO [#C] Agent-KB / memory-sync — work + unknown-project write refusal :test: :PROPERTIES: @@ -454,22 +765,6 @@ Once specs carry lifecycle TODO keywords under =docs/specs/=, add a custom org-a :END: From Craig via the roam inbox (2026-07-02, routed by archsetup). Teardown-by-default already shipped (bare "wrap it up" closes the window; "with summary" keeps it). Craig's follow-on: "maybe we cut the summary altogether. help me think through when I'd want a summary and how I would recognize it before confirming and then having it close." Run that think-through with him (brainstorm-shaped, not solo), then adjust wrap-it-up.org's Step 6 + trigger phrases to the outcome. -** TODO [#C] triage-intake.org auto mode — push each sweep to phone (ntfy) :feature:solo: -:PROPERTIES: -:CREATED: [2026-06-20 Sat] -:LAST_REVIEWED: 2026-07-13 -:END: -The work project (2026-06-18) added a "Push each sweep to Craig's phone (ntfy) — the primary delivery" subsection under "Trigger and delivery" in triage-intake.org auto mode, and asks to fold it into the canonical engine plus re-sync. Preserved bundle: [[file:docs/design/2026-06-18-triage-intake-phone-push-note.org][note]] + [[file:docs/design/2026-06-18-triage-intake-phone-push-workflow.org][edited workflow]]. Auto mode is the away-from-desk / vacation mode, so phone-notify becomes the primary delivery each sweep (fuller end-of-sweep output: per-source deltas, open-PR/Linear state, awaiting-ack list, one-line verdict, timestamp; SCAN FAILED banner on any source failure), plus phone-recv polling each sweep for Craig's replies. Falls back to inline when phone-notify is absent. Transport re-pointed 2026-07-13: ntfy is retired; agent-page (shipped today) is the send channel, so the push half is buildable now — substitute agent-page for phone-notify throughout. The reply-polling half (phone-recv) waits on the Signal-pager runbook's read-replies deliverable. Shared template-workflow change, so review-gated. - -** TODO [#D] Fully-unattended scheduled inbox check (/schedule cron pass) :feature: -:PROPERTIES: -:CREATED: [2026-06-23 Tue] -:LAST_REVIEWED: 2026-06-28 -:END: -vNext from the inbox-consolidation spec. =auto inbox zero= (v1) is the interactive =/loop= recurring check that waits for Craig's yes before executing. A fully-unattended =/schedule= cron pass that fires while Craig is away needs its own contract before it can ship: read-only vs may-mutate =todo.org= / =~/org/roam/inbox.org=, how a find surfaces asynchronously when Craig isn't at the session, how dedup state persists across runs that don't share a session, and what session/auth context a cron run carries. From the inbox-consolidation spec-review (Codex finding 1). See [[id:a7fe2a10-dfa8-4ba3-a11a-e7b1288b7573][spec]]. - -Update 2026-06-28: the "design after v1 consolidation lands" precondition is cleared — the inbox engine consolidation (24ca58d) and the monitor-inbox 15-min loop (edb545d) both shipped. Now actionable backlog rather than blocked; design the unattended contract when prioritized. - ** TODO [#D] Warn-only pre-commit hook for tooling-path enumeration :feature: :PROPERTIES: :CREATED: [2026-06-22 Mon] @@ -1316,7 +1611,99 @@ having a skill to generate or check OV-1-shaped artifacts. Don't build speculatively — defense-specific notations are narrow enough that each skill should be driven by a concrete contract need, not aspiration. -** TODO [#B] todo-cleanup.el dated-seal archiving :feature:solo: +* Rulesets Resolved +** CANCELLED [#C] ntfy phone channel as general two-way agent-comms :feature:spec: +CLOSED: [2026-07-13 Mon] +:PROPERTIES: +:CREATED: [2026-06-20 Sat] +:LAST_REVIEWED: 2026-06-24 +:END: +Killed at the 2026-07-13 task review: home retired and tore down the ntfy channel on 2026-07-04, so this proposal's transport no longer exists. Its living successors are the Signal pager ([#B] task above — one identity on velox, runbook pending) and agent-page (shipped 2026-07-13), which cover the send half; two-way (read-replies) rides the Signal runbook. +Proposal from the home project (2026-06-17): promote the self-hosted ntfy-over-Tailscale phone channel it built and verified on ratio into a general two-way agent-comms tool rulesets owns. Full proposal: [[file:docs/design/2026-06-17-ntfy-agent-comms-proposal.org]] (as-built runbook stays in the home project at =working/phone-notifications/spec.org=). What rulesets would decide: canonicalize =phone-notify= (send) plus a new =phone-recv= (check-since) as synced bin scripts; the per-machine config/secret convention (token in =~/.config/phone-notify/config= chmod 600 today, vs GPG-encrypted in dotfiles); a reference =ntfy-inbound-handler= plus systemd user-unit for event-driven delivery (Tier A subscriber routes inbound to inbox/notify, Tier B inbound spawns an agent session, Tier C notify a live session — harness research); approval-button workflows for the commits.md gates when Craig is away from the desk (tap-to-approve, the high-value concrete use); and the relationship to the retired cross-agent-comms scripts (ntfy may be the transport they lacked). Worked via =spec-create=. Blocks the triage-intake phone-push task below. +** DONE [#B] Org-table helpers corrupt example blocks :bug:solo: +CLOSED: [2026-07-14 Tue] +:PROPERTIES: +:CREATED: [2026-07-11 Sat] +:LAST_REVIEWED: 2026-07-13 +:END: +Fixed in 951b6fc, test-first. Both defects landed as filed (block-type-aware scanning in both helpers; lint-org CLI report-only by default, writes behind --fix), plus the true corruption path found during the work: wrap-org-table's load-time CLI dispatch fired on lint-org's require and reformatted lint-org's file arguments. Entry-script guard added. The named regression test (example block byte-identical) is in the suite. +=wrap-org-table.el= and =lint-org.el= both scan for =/^\s*|/= lines and rewrite them as org tables without skipping =#+begin_example=/=src=/=quote= regions, so ASCII art using pipe characters gets mangled into bordered tables. Reproduced 2026-07-09 in the work project against an architecture doc with a pipe/=v= flow diagram; a plain indented block became a table with =|---|= rules between every line. Two separable defects: (1) table detection is line-based — both helpers should use =org-element-at-point= / =org-in-block-p= to skip example/src/quote/verse blocks; (2) =lint-org.el= mutates its input on disk with no confirmation — passing five files reformatted all five (one by 1949 lines). A linter must report, not write; put the reformat behind an explicit =--fix= flag. + +Grading (severity × frequency, per todo-format.md): Critical severity (silent org data loss; recoverable here only because the content was git-staged) × rare-edge-case frequency (fires only when a mixed table+example file is passed to the helper) = P2 = [#B]. + +Regression test: run =wrap-org-table.el= against a file containing a =#+begin_example= block whose lines start with =|= and assert the block is byte-identical afterward. Source: work handoff 2026-07-09 (=inbox/2026-07-09-1341-from-work-bug-data-loss-wrap-org-table-el-and.org=). +** DONE [#B] Sentry workflow — build from spec :feature: +CLOSED: [2026-07-19 Sun] +:PROPERTIES: +:SPEC_ID: f6c51f27-d7a2-4b63-9ff9-5ba005a66dfb +:END: +Built the sentry supervisor workflow from the spec +([[file:docs/specs/2026-07-14-sentry-workflow-spec.org][sentry workflow spec]], now IMPLEMENTED). Four phases, each committed and +pushed in no-approvals + auto-flush mode; full suite green throughout. The overnight +live trial is handed to Craig as a manual-testing task (below); its findings file as +follow-ups. +*** 2026-07-19 Sun @ 04:52:00 -0500 Built the agent-lock helper + 18 bats tests +=.ai/scripts/agent-lock= (canonical =claude-templates/.ai/scripts/=, mirror synced): +mkdir-atomic acquire, PID/host/ISO-timestamp metadata, mtime-based staleness reclaim +(atomic-rename claim so two acquirers can't double-acquire — caught by the pre-commit +review), heartbeat refresh, acquire/release/status/path subcommands, XDG_RUNTIME_DIR +home with =~/.cache= fallback. Commit a8b6cf4. +*** 2026-07-19 Sun @ 04:56:00 -0500 Built the sentry.org engine + INDEX entry +=.ai/workflows/sentry.org= (mirror synced): :COMMIT_AUTONOMY: entry ticket, the +interactive entry gates, ff-only reconcile, =sentry/<date>-<host>= branch mechanics, +the ten-pass probe→work→session-context→commit runner, digest + morning-approval +queue, skip-not-degrade safety, spine-excluded dirty checks + fire-end digest commit, +multi-day stall notify, the stop-sentry operation. All 10 decisions and 12 findings +reflected. Commit ccc9c26. +*** 2026-07-19 Sun @ 05:00:00 -0500 Wired the roam writers + wrap-up guard +knowledge-base.md and inbox.org core §5 acquire the roam-write lock and edit-plus- +trigger (roam-sync stays sole committer); roam-sync.sh header updated to match; +wrap-it-up.org gained a Step 0 active-sentry guard; triage-intake.org notes its +sentry-pass role. Graceful degradation when agent-lock is absent. Commit c6383e9. +*** 2026-07-19 Sun @ 05:04:00 -0500 Verified suite green + flipped spec to IMPLEMENTED +=make test= green at HEAD (pytest 393, ERT + bats all pass, exit 0). Flipped the spec +keyword DOING → IMPLEMENTED with a dated history line and mirrored the Metadata Status. +Filed the overnight live trial as a structured manual-testing task. +** DONE [#C] ai launcher hardening — bug hunt + refactor pass :refactor:solo: +CLOSED: [2026-07-19 Sun] +:PROPERTIES: +:CREATED: [2026-07-13 Mon] +:LAST_REVIEWED: 2026-07-19 +:END: +Resolved across two commits (113e8d8 net, 2b619f1 refactor). Brought the 17 uncovered functions under characterization tests (launcher tests 9 → 42), extracted the git/tmux decision logic into four pure cores (_git_prep_action, _order_windows, _match_window_id, _git_is_dirty) each with a Normal/Boundary/Error set, and dispositioned the footgun audit + /refactor pass. Objective floor met: shellcheck clean, shfmt -i 2 -ci consistent, make test green before and after, live black-box smoke correct for claude and codex. Honest limit: attach_session and the full end-to-end of single/multi/fetch stay partially covered (their terminal step attaches to tmux or blocks on fzf, can't run headless). Their decision logic was extracted into the netted cores. The interactive runtime picker stayed out of scope (a design call, filed on the generic-agent-runtime parent). +Harden =claude-templates/bin/ai= (the agent-session launcher, 540 lines, 22 functions). Origin: roam inbox 2026-07-13, phrased open-endedly ("find bugs until none visible, refactor until nothing worthwhile remains"). Rescoped 2026-07-19 with measurable acceptance criteria per =todo-format.md='s "Making an open-ended task measurable," which is what makes it =:solo:=. The four moves: + +1. *Bound the surface.* The 22 functions are the done-set. *Covered* (behaviorally, via =scripts/tests/ai-launcher-runtime.bats=, 9 tests over the runtime path): =resolve_agent_cmd=, =build_runtime_choices=, =pick_runtime=, =build_instructions=, the print modes. *Uncovered* (the 17 to bring under test): =usage=, =check_deps=, =attach_session=, =create_window=, =maybe_add_candidate=, =build_candidates=, =fetch_candidates=, =git_status_indicator=, =annotate_candidates=, =auto_pull_if_clean=, =read_selections=, =sort_windows=, =find_window_id=, =prep_git_single=, =attach_mode=, =single_mode=, =multi_mode=, =print_launch_mode=. + +2. *Net the behavior.* Characterization tests (Normal/Boundary/Error per unit, per =testing.md=) over the uncovered surface. The pure/near-pure ones take the category set directly: =git_status_indicator=, =maybe_add_candidate= (dedup), =annotate_candidates= (formatting), =read_selections= (selection parse), =usage=. The =tmux=/=git=-coupled ones (=sort_windows='s ordering, =create_window=, =attach_session=, =find_window_id=, =prep_git_single=, =auto_pull_if_clean=) get their pure decision logic extracted into helpers that take plain inputs and return plain results — that extraction *is* the hardening — with the I/O calls left as thin wrappers. + +3. *Disposition every finding.* (a) A bash-footgun audit per function, each cell fixed / n-a / filed: unquoted expansions + word-splitting, =set -euo pipefail= gaps and where errexit is intentionally off, subshell state loss, exit-code propagation, ordering/races in =sort_windows= + window creation, and the git-prep error paths (=prep_git_single= / =auto_pull_if_clean= on a dirty tree, detached HEAD, no upstream). (b) A =/refactor= pass, each finding applied (tests green) or declined with a one-line reason. + +4. *Objective floor.* =shellcheck= clean, =shfmt=-consistent, =make test= green before and after, and every uncovered function above has its characterization set (a per-function checklist — =kcov= isn't installed; install it if a single coverage number is wanted). Plus ~3 functional tests over the launch pipelines (=single_mode=, =multi_mode=, =attach_mode=) against a throwaway =tmux= session, for the composition bugs no per-function unit can see. + +*Qualifying answer:* a dispositioned report — surface split covered/uncovered, tests before → after, =shellcheck=/=shfmt= result, the footgun matrix fully dispositioned, the =/refactor= findings fully dispositioned, all green. Not "no bugs remain" (unprovable) — "every enumerated path passes its characterization set and clears the audit." + +*Out of this task's =:solo:= scope:* an interactive runtime picker. It's a feature carrying a design/preference call (does Craig want it, what shape), which is deliberation, not hardening — file it separately if wanted. The runtime-selection arc (claude/codex shipped; ollama/qwen pending the model-floor eval) stays on the generic-agent-runtime parent. +** DONE [#C] Put install-ai on PATH, launchable as =install-ai= :chore:quick:solo: +CLOSED: [2026-07-18 Sat] +:PROPERTIES: +:CREATED: [2026-07-11 Sat] +:LAST_REVIEWED: 2026-07-13 +:END: +From the roam inbox (2026-07-11): make install-ai launchable as =install-ai= (no =.sh=) from PATH. dotfiles needs a copy that stays in sync with the rulesets canonical — decide whether the startup script-sync already covers it or a dedicated mechanism is needed. + +Resolved: added =claude-templates/bin/install-ai=, a thin launcher that resolves its own path through the symlink chain and execs =scripts/install-ai.sh=. =make install='s existing bin loop symlinks it into =~/.local/bin/install-ai= (same mechanism as =ai= and =agent-page=), so no dedicated sync and no dotfiles copy — the symlink always points at the canonical. 3 launcher bats added (incl. symlink-invocation resolution). Verified live: =install-ai --help= runs from PATH. +** DONE [#C] coverage-summary.el documented as a local-only helper :chore:quick:solo: +CLOSED: [2026-07-18 Sat] +:PROPERTIES: +:CREATED: [2026-06-22 Mon] +:LAST_REVIEWED: 2026-07-13 +:END: +The elisp bundle installs =coverage-summary.el= into =.claude/scripts/=, gitignored in code projects, so CI can't run =make coverage-summary= against it. Decision (Craig, 2026-06-28): keep it in =.claude/scripts/= and document it as a local-only helper — don't ship it to a tracked =scripts/= dir, don't expect CI to run it. Remaining work (docs only, no move): state the local-only status in the script's header comment and wherever =make coverage-summary= is described, so the gitignored install reads as intentional rather than a gap. Note: emacs-wttrin rewrote its copy's header to claim a tracked =scripts/= home, which now contradicts this decision and should be reverted on their side. Surfaced 2026-06-21 during the coverage-summary autoloads bugfix (commit fb86736). + +Resolved: documented the local-only status in the =coverage-summary.el= commentary header and in =elisp-testing.md='s "Measuring it" section — the gitignored install now reads as intentional, not a coverage gap. Sent emacs-wttrin a handoff to revert its contradicting header claim. +** DONE [#B] todo-cleanup.el dated-seal archiving :feature:solo: +CLOSED: [2026-07-18 Sat] Redefine =--archive-done= aging from a 7-day roll-into-one-file model to a one-month retention with dated seals. Craig ratified the design in the work project 2026-07-17; origin handoff preserved at @@ -1340,8 +1727,8 @@ Changes to =.ai/scripts/todo-cleanup.el= (canonical in TDD, canonical-then-mirror per the sync-check invariant. Home's planning-line strip proposal (filed alongside) also touches todo-cleanup.el =--convert-subtasks= — the two can be built as one batch. - -** TODO [#B] Strip stale planning lines on dated completion + lint backstop :feature:solo: +** DONE [#B] Strip stale planning lines on dated completion + lint backstop :feature:solo: +CLOSED: [2026-07-18 Sat] Two linked fixes so a closed sub-task can't keep polluting the org agenda. Origin: home handoff 2026-07-17; design preserved at [[file:docs/design/2026-07-17-dated-log-planning-line-strip-proposal.md]]. A dated-log @@ -1366,8 +1753,9 @@ but nothing strips the planning line, and an interactive org close only stamps Build notes: the checker's regex must key on "no TODO keyword" so it never flags a live TODO that legitimately carries a SCHEDULED. TDD, canonical-then-mirror. Batches with the dated-seal task above (both touch todo-cleanup.el). - -** TODO [#B] Enforce the task-boundary inbox check via a hook :feature: +** DONE [#B] Enforce the task-boundary inbox check via a hook :feature: +CLOSED: [2026-07-19 Sun] +Resolved with the soft-nudge design (94e54f6). New =hooks/inbox-boundary-check.sh= Stop hook blocks the yield once + injects the pending count when =inbox-status -q= exits 1, steps aside on the harness re-entry (=stop_hook_active=) so a mid-task pause never wedges, self-skips on no-inbox/no-inbox-status/clean. Wired ahead of =ai-wrap-teardown= in =.claude/settings.json= (which the live =~/.claude/settings.json= symlinks to) + the snippet; glob-installed by =make install-hooks=. 6 bats (pending/clean/re-entry/no-inbox/absent-status/project-name). protocols.org Inbox Monitoring Cadence now notes the enforcement. Live-verified on ratio (clean inbox no-ops, pending fixture blocks); velox synced + hook linked. The UserPromptSubmit visibility complement stayed unbuilt (optional in the design) — file separately if wanted. Today the "check =inbox/= at every task boundary" rule (protocols.org, "Inbox Monitoring Cadence") is prose-only — present in all 27 projects' synced protocols.org, but nothing mechanically enforces it, so it holds only as well as @@ -1432,8 +1820,9 @@ No hook can perfectly tell "a unit of work finished" from "the agent paused for other reason" — the harness exposes only "the turn is ending," not "a task is ending." The design leans on those two being usually the same in Craig's workflow. That's the tradeoff hard-block vs soft-nudge is really about. - -** TODO [#B] "Colloquialisms and Expansions" + "the list" before-close-queue convention :feature: +** DONE [#B] "Colloquialisms and Expansions" + "the list" before-close-queue convention :feature: +CLOSED: [2026-07-19 Sun] +Resolved (approved in the speedrun). New =* Colloquialisms and Expansions= section in =protocols.org= documents both shorthands: "put X on the list" → append to a session-scoped Before-Close Queue (=* Before-Close Queue= heading in the session anchor, resets on archive, todo.org for must-outlive items); "tell <project> <msg>" → =inbox-send=. =wrap-it-up.org= Step 1 gained a "Work the Before-Close Queue (before the Summary)" sub-step so queued work rides the wrap commit, unfinished items surfaced in the valediction. Design calls: reference in protocols.org not per-project notes.org (synced = shared norm); queue in the session anchor as home did; wrap step at the front of Step 1, not a new half-step (keeps the "Steps 1-5" framing). 4 documentation-integrity bats (=before-close-queue.bats=). Canonical + mirror synced. The UserPromptSubmit-style variant wasn't in scope. Home proposes two linked cross-project norms; Craig recommends rolling them out. Origin: home handoff 2026-07-18, design preserved at [[file:docs/design/2026-07-18-colloquialisms-and-the-list-proposal.md]]. Needs Craig's @@ -1464,25 +1853,481 @@ section vs a new shipped reference file); where the queue itself lives (home use insertion point (before the teardown/valediction, alongside the roam-inbox sub-step). Then it's a synced-file change (canonical-then-mirror) + a test that the wrap step drains the queue. +** DONE [#B] working/ tracked-from-creation + gitignored temp/ :feature: +CLOSED: [2026-07-20 Mon] +Craig's ruling relayed from .emacs.d (2026-07-19): working/ is version-controlled from creation (not excluded until graduation); ephemeral artifacts go in a gitignored temp/ or /tmp; graduation reorganizes durable artifacts into permanent homes rather than marking when they become durable. + +Implemented (Shape A) during the morning sentry review, 2026-07-20: +- =claude-rules/working-files.md=: added "working/ Is Version-Controlled From Creation" section (tracked-from-creation, graduation-is-a-move, temp/ for ephemeral). +- =.ai/protocols.org= (canonical + mirror): one-paragraph mirror in the Working-Files Convention section. +- =scripts/install-ai.sh=: emits a =temp/= ignore block in BOTH track and gitignore modes; working/ never ignored. Idempotent. +- =scripts/sweep-gitignore-tooling.sh=: separate mode-independent temp/ backfill pass (a distinct loop, not an IGNORE_SET member — that would skip track-mode projects, the set that most needs it). +- Tests: 3 new install-ai bats + 4 new sweep bats (temp/ in both modes, never working/, idempotent). Suite green, 384 bats ok. + +Finding confirmed at implementation: the canonical machinery already never ignored working/ (tooling set is only =.ai/ .claude/ CLAUDE.md AGENTS.md=), so .emacs.d's =/working/= ignore was a purely local deviation. The staging proposal dir was removed after shipping; its content lives in the feat commit and this body. +** DONE [#C] Polyglot projects — supported, or refused? :spec: +CLOSED: [2026-07-20 Mon] +DECISION (2026-07-20, scouting with Craig): *case-by-case, and it already composes — no option-2 machinery.* The evidence: bundle contents split into namespaced/additive files (rules =<lang>.md=, =validate-<lang>.sh= hooks, =coverage-summary.<ext>= scripts, appended =gitignore-add.txt=) that compose cleanly, and exactly three colliding files — =claude/settings.json= and =githooks/pre-commit= (full bundles only: bash/elisp/go) and =coverage-makefile.txt= (elisp/go/python/typescript). Of the three, only =coverage-makefile.txt= occurs in the fleet, and clock-panel (the one real polyglot, python+typescript) proves it's benign: both bundles installed, no breakage, because that fragment is a hand-pasted Makefile block nobody pasted twice. So: keep the install-lang collision guard (it blocks the destructive full+full settings/githooks clobber, which no project actually hits), and document the =coverage-<lang>:= + =coverage:= aggregate namespacing as the one manual step when going polyglot (filed below). No two-full-co-equal-bundle project exists, so the settings.json/githooks merge is unwarranted. Follow-up doc: [[file:todo.org::*Document polyglot coverage-makefile namespacing][Document polyglot coverage-makefile namespacing]]. -* Rulesets Resolved -** CANCELLED [#C] ntfy phone channel as general two-way agent-comms :feature:spec: -CLOSED: [2026-07-13 Mon] +Do we support more than one language bundle per project? The honest answer today +is "partly, by accident." The collision guard added 2026-07-16 refuses a +*colliding* second bundle rather than silently replacing the first's config, but +a non-overlapping pair still installs fine: bash ships =settings.json= + +githooks and no coverage fragment, python ships only a coverage fragment, so +=bash= + =python= composes cleanly today and yields a real polyglot project with +both rule sets. So the line isn't polyglot-vs-not, it's overlap-vs-not — and +nobody chose that line, it fell out of which bundle happens to ship what. Origin: +home's report after scaffolding clock-panel with python + typescript, +[[file:docs/design/2026-07-16-polyglot-bundle-collision.txt][docs/design/2026-07-16-polyglot-bundle-collision.txt]]. + +Pair this with the subproject scouting below — it's the same question in a +different costume ("which projects would actually be polyglot, and why"), so +they should be one conversation. + +The three options, in the order they'd be weighed: + +1. *Unsupported, explicitly.* Keep the guard as the answer. Cheapest, and + matches how little polyglot exists (one project, clock-panel). +2. *Supported.* Needs per-bundle filenames, a merged =settings.json= (the hooks + arrays compose rather than clobber), composed githooks, and namespaced + Makefile targets with a =coverage= aggregate. This is the real work. +3. *Case-by-case.* Support the pairs that come up, refuse the rest. + +What the decision needs to know: + +- *The target-name collision is the deeper half* (home's point, and it's right). + Every bundle's fragment defines =coverage:= and =coverage-summary:=, so even + with both files present a polyglot project can't paste both into one Makefile. + Renaming files doesn't fix it. +- *Only three of five shared filenames actually collide.* =gitignore-add.txt= + (5 bundles) appends deduped and composes. =CLAUDE.md= (3) is seed-only, and + its fallback comment shows multi-bundle was already considered there. + =claude/settings.json= (3), =githooks/*= (3), and =coverage-makefile.txt= (4) + are the real ones. +- *=FORCE=1= is a poor escape hatch* (home's catch): it also re-seeds + =CLAUDE.md=, which is destructive on a customized project. If polyglot + becomes supported, the override wants to be its own flag. +** DONE [#C] Subproject pattern — promote to claude-rules? :spec: +CLOSED: [2026-07-20 Mon] +DECISION (2026-07-20, scouting with Craig): *don't promote — keep it local at home.* The scouting confirmed N=1: across all 27 =.ai= scopes, home (9 subprojects, all from the single 2026-06-11 fold) is the only real user. The nearest neighbors aren't the pattern — rulesets folds claude-templates in as a git *subtree* (different mechanism, not shared-=.ai/=-scope), archsetup's dotfiles/ likewise; every other project is a focused single package with no subproject structure or need. Promoting a 282-line convention into the always-on claude-rules layer for one project fails the thin-always-on principle (the precedent is patterns.md at 29 lines + docs-lifecycle.md's depth-in-a-spec). Keep the convention as home's local instance. Revisit only if a genuine second case appears, and then as a thin pointer + a spec, never 282 lines always-on. + +home proposes promoting its subproject pattern (a former standalone project +folded into a parent, living as a self-contained subdir sharing the parent's +=.ai/= scope) into the rules layer: vocabulary, the read-first +=<subproject>/<subproject>-brief.org= convention, the parent-vs-subproject +content criterion ("one fact, one home"), and create/archive criteria. +Proposal + home's full instance: +[[file:docs/design/2026-07-15-subproject-pattern-proposal.org][proposal]], +[[file:docs/design/2026-07-15-subprojects-convention-home-instance.org][home's convention doc]]. + +Deferred 2026-07-16 rather than promoted. *Craig's framing:* he wants to scout +which projects would actually get subprojects, and why, before we shape a rule. +If he hasn't done that scouting by the time this comes up, offer to do it +together — brainstorm the candidates, then explore the reasons behind each. That +evidence decides it: either we drop the pattern, or we know enough to adjust it +so it's effective. Don't shape the rule before the scouting. + +*Review findings from the 2026-07-16 pass* (the inputs the decision needs): + +- *N=1.* home is the only project with subprojects, across all 27 =.ai= scopes; + its nine all came from the single 2026-06-11 fold. This is the fact the + scouting tests. +- *Placement contradicts the proposal's own principle.* =claude-rules/*.md= + loads into every session of every project. home's doc argues the always-on + layer is "a tax paid whether or not it's relevant today" and depth belongs + "one open away". At 282 lines the doc would be the third-largest rule and add + ~11% to the always-on layer, so every .emacs.d / takuzu / chime session would + carry a one-project convention. +- *Precedent for the shape:* =patterns.md= (29 lines, explicit "don't carry the + catalog in context") and =docs-lifecycle.md= (75 lines, depth in a spec). + Thin rule + on-demand depth is the established answer. +- *Dangling reference:* the doc cites =claude-rules/git-hosting-privacy-model= + as authority for its shared-scope-safety criterion. No such file exists — the + real content is the gitignore-vs-track and public-reachability decision in + =protocols.org=. Fix before any promotion. +- *Instance vs rule:* the metrics, self-improvement log, kill criteria, rollout + dates, and adoption table are home's instance, not rule content. +** DONE [#B] Triage source activation — per-project source declaration :feature:spec: +CLOSED: [2026-07-20 Mon] +Spec: [[file:docs/specs/2026-07-20-triage-source-activation-spec.org][docs/specs/2026-07-20-triage-source-activation-spec.org]] (IMPLEMENTED, ID af73ef0b-cd1d-46f1-9e1d-62695733a4de). Craig approved after his read, both open decisions resolved via cj comments; built same session. + +From the sentry live trial (2026-07-20): triage-intake self-activated in every project because the general (personal-account) plugins are template-synced everywhere, so sentry's pass-3 probe misfired and would pull Craig's personal inboxes into whatever project the fire ran in. Fixed with an activation gate in triage-intake Phase 0 — general plugins gate on a per-project =:TRIAGE_SOURCES:= declaration; project-specific plugins stay active by presence. Applies to interactive and unattended alike. sentry pass-3 probe reads the same signal. Migration handoffs sent to home + work. Supersedes Fire 1's narrower probe-only approval-queue item. +** DONE [#B] Silent-until-signal for in-session monitor loops :feature:spec: +CLOSED: [2026-07-20 Mon] +Spec: [[file:docs/specs/2026-07-20-silent-until-signal-monitors-spec.org][docs/specs/2026-07-20-silent-until-signal-monitors-spec.org]] (IMPLEMENTED, ID af592bd6-d3e6-47e2-8804-2a287b4d9303). Craig approved without changes 2026-07-20; all 5 phases shipped that day. Sentry, auto triage-intake, and auto inbox-zero all collapse an empty fire to =<workflow> at HH:MM: nothing=; a manual-testing entry covers the live-loop verification. + +Craig-approved proposal from .emacs.d (2026-07-20), demonstrated by the sentry live trial (fires 3-8 were walls of no-op lines). Reframed live from a watcher *mechanism* to a *policy*: an in-session monitor fire detects first, and on an empty check collapses to one labelled heartbeat line (=<workflow> at HH:MM: nothing=) instead of a full turn; only a real item earns the full surface-and-judge turn. Keeping detection in-session dissolves the MCP-auth split (triage's Gmail/Slack/Linear need session auth), so it applies uniformly to sentry, auto triage-intake, and auto inbox-zero. No external watcher, no new seen-list. Five build phases in the spec. Phase 1 (sentry quiet-fire heartbeat) shipped 2026-07-20 ahead of the full READY gate at Craig's direction — a quiet fire now collapses to =sentry at HH:MM: nothing=. Phases 2-4 (auto triage-intake, auto inbox-zero, shared-policy home) + Phase 5 verification pending Craig's spec read. Overlaps the sentry cluster ([[file:todo.org::*Sentry vNext passes][Sentry vNext passes]], [[file:todo.org::*Triage source activation][Triage source activation]]). +** DONE [#C] Apply suspend.org detach-on-suspend change to canonical :feature: +CLOSED: [2026-07-20 Mon] +Craig's change relayed from archsetup (2026-07-20): add Step 6 to suspend.org — detach the tmux client (=tmux detach-client -s "$sess"=) as the final action of every suspend, so a suspended session parks in the re-attachable set instead of cluttering the alt-space rotation. Also reworded the neighbors bullet and the "does NOT do" teardown bullet to draw the detach-vs-teardown line. + +Applied to canonical 2026-07-20 (verified the diff was exactly the intended change, canonical + mirror synced, lint clean bar one pre-existing flush/SKILL.md link). Replied to archsetup that its local stopgap is now canonical. +** DONE [#B] Document (and own) the Signal pager :feature:spec: +CLOSED: [2026-07-20 Mon] :PROPERTIES: -:CREATED: [2026-06-20 Sat] -:LAST_REVIEWED: 2026-06-24 +:CREATED: [2026-07-11 Sat] +:LAST_REVIEWED: 2026-07-13 :END: -Killed at the 2026-07-13 task review: home retired and tore down the ntfy channel on 2026-07-04, so this proposal's transport no longer exists. Its living successors are the Signal pager ([#B] task above — one identity on velox, runbook pending) and agent-page (shipped 2026-07-13), which cover the send half; two-way (read-replies) rides the Signal runbook. -Proposal from the home project (2026-06-17): promote the self-hosted ntfy-over-Tailscale phone channel it built and verified on ratio into a general two-way agent-comms tool rulesets owns. Full proposal: [[file:docs/design/2026-06-17-ntfy-agent-comms-proposal.org]] (as-built runbook stays in the home project at =working/phone-notifications/spec.org=). What rulesets would decide: canonicalize =phone-notify= (send) plus a new =phone-recv= (check-since) as synced bin scripts; the per-machine config/secret convention (token in =~/.config/phone-notify/config= chmod 600 today, vs GPG-encrypted in dotfiles); a reference =ntfy-inbound-handler= plus systemd user-unit for event-driven delivery (Tier A subscriber routes inbound to inbox/notify, Tier B inbound spawns an agent session, Tier C notify a live session — harness research); approval-button workflows for the commits.md gates when Craig is away from the desk (tap-to-approve, the high-value concrete use); and the relationship to the retired cross-agent-comms scripts (ntfy may be the transport they lacked). Worked via =spec-create=. Blocks the triage-intake phone-push task below. -** DONE [#B] Org-table helpers corrupt example blocks :bug:solo: -CLOSED: [2026-07-14 Tue] +home retired the ntfy phone-notification channel (phone-notify/phone-recv, self-hosted ntfy on ratio) on 2026-07-04 in favor of paging over Signal, and tore ntfy down. The Signal pager it replaced ntfy with is undocumented: no pager script in =~/.local/bin=, and =notify= doesn't reference Signal. What exists on ratio: signal-cli 0.14.5, account 404211. Deliverable: a documented Signal pager (send + read-replies), the signal-cli setup/account notes, and the sync path — the Signal equivalent of the retired ntfy runbook. Cross-machine tooling, so canonical home + docs belong in rulesets. + +RECONCILE FIRST: =protocols.org= "Paging Craig" already documents an agent-paging path via the *signal-mcp* tool (=send_message_to_user=, pager account +15045173983, Craig's UUID =b1b5601e-…=, verified 2026-06-30). home's handoff is about a *different* mechanism (signal-cli / account 404211, home's own paging). Decide whether these are one channel or two, and whether the signal-cli side needs its own runbook or should route through signal-mcp. Source: home handoff 2026-07-04 (=inbox/2026-07-04-1302-from-home-task-for-rulesets-document-and-decide.org=). Successor to the 2026-06-17 two-way-comms proposal. + +*** 2026-07-13 Mon @ 05:16:37 -0500 Folded home's ownership ack + current-state report into the reconcile scope +home confirmed (2026-07-11 reply) rulesets owns this task and the two-paths reconciliation is the right first step. New facts for the reconcile: from a home session on 2026-07-09, signal-mcp was NOT connected, and the local signal-cli is registered as Craig's own number — =send --note-to-self= returns a message id but produces no phone push. So home currently has no working ad-hoc page channel at all; whatever the runbook lands on must give home a live path. + +*** 2026-07-13 Mon @ 14:40:00 -0500 Added runtime-portability as a second motivation (Craig approved) +The MCP portability inventory ([[file:docs/design/2026-07-13-runtime-portability-inventories.org]]) found signal-mcp exists only claude.ai-side — no local config anywhere — so a non-Claude agent (Codex-style or local LLM) has no paging path at all. The signal-cli runbook this task produces is therefore also the runtime-neutral page channel, not just home's replacement for ntfy. + +*** 2026-07-13 Mon @ 18:30:19 -0500 agent-page shipped — every project now knows both channels +Craig's call: make paging universal and rename it the agent pager. NEW =claude-templates/bin/agent-page= (runs signal-cli directly on velox, ssh-relays from anywhere else, desktop-fallback hint on failure; 4 bats tests; live-verified from ratio through the real script). protocols.org "Paging Craig" rewritten around the two channels (notify desktop + agent-page phone; signal-mcp demoted to a velox-local nicety); page-me.org gained the phone section + fire-both guidance; work-the-backlog's end-of-set page names both surfaces; INDEX updated. Every project inherits via the startup sync + make install. Remaining here: the runbook proper, the receive timer, ssh-only vs linked-device. + +*** 2026-07-13 Mon @ 18:13:07 -0500 RECONCILED — one channel, on velox; live page verified end to end +The two-paths question is answered: there is ONE pager identity, +15045173983, registered in velox's signal-cli (account file 465310) — and signal-mcp is a locally-configured MCP server in velox's global ~/.claude.json (not claude.ai-side as the 14:40 entry inferred; it's just invisible from ratio, which is why home and this session couldn't find it). ratio's signal-cli holds only Craig's personal number (note-to-self, no push). Verified live today: =ssh velox 'signal-cli -a +15045173983 send -m … b1b5601e-6126-47f8-afaa-0a59f5188fde'= buzzed Craig's phone — his remembered CLI page was this same account on velox. Reliability findings for the runbook: both accounts throw receive-staleness warnings (velox 40 days, ratio 26; the signal protocol wants regular receives — a systemd receive timer on velox is the roam-sync-shaped fix), and the channel requires velox to be up. Remaining deliverables sharpened: the runbook (send + read-replies + receive timer + account notes); decide ssh-over-tailnet-only vs registering ratio as a linked device of the pager account; update protocols.org "Paging Craig" (it names signal-mcp as the only supported path — true only on velox; the ssh recipe is the cross-machine path) — shared-asset edit, own review pass. Interim recipe sent to home so it's unblocked today. + +*** 2026-07-20 Mon @ 15:56:56 -0500 Runbook shipped; ratio linked as a device; receive timer on both machines +All four remaining deliverables landed. (1) Runbook: [[file:docs/design/2026-07-20-signal-pager-runbook.org]] — send, read-replies, receive timer, signal-cli account/setup notes, the resolved topology decision. (2) Receive timer: =scripts/signal-receive.sh= + =scripts/systemd/signal-receive.{service,timer}= (roam-sync-shaped, 15-min cadence, 3 bats), stowed via dotfiles =common=, enabled + verified on ratio; a manual drain also cleared the 47-day staleness live. (3) Topology decision (Craig, option 2): register daily drivers as linked devices rather than ssh-relay-only — ratio linked as Device 2 "ratio-pager", direct send verified, =agent-page= generalized from a velox-only check to "any machine holding the account sends directly, else relay" (bats updated). (4) protocols.org "Paging Craig": verified accurate, no edit — already describes both channels and the caveats, and its generic runbook pointer is correct to leave un-pathed since it syncs into every project. Remaining one-time step: enable the timer on velox after it pulls dotfiles. +** DONE [#C] triage-intake auto mode — push signal sweeps to phone via agent-text :feature:solo: +CLOSED: [2026-07-20 Mon] :PROPERTIES: -:CREATED: [2026-07-11 Sat] +:CREATED: [2026-06-20 Sat] :LAST_REVIEWED: 2026-07-13 :END: -Fixed in 951b6fc, test-first. Both defects landed as filed (block-type-aware scanning in both helpers; lint-org CLI report-only by default, writes behind --fix), plus the true corruption path found during the work: wrap-org-table's load-time CLI dispatch fired on lint-org's require and reformatted lint-org's file arguments. Entry-script guard added. The named regression test (example block byte-identical) is in the suite. -=wrap-org-table.el= and =lint-org.el= both scan for =/^\s*|/= lines and rewrite them as org tables without skipping =#+begin_example=/=src=/=quote= regions, so ASCII art using pipe characters gets mangled into bordered tables. Reproduced 2026-07-09 in the work project against an architecture doc with a pipe/=v= flow diagram; a plain indented block became a table with =|---|= rules between every line. Two separable defects: (1) table detection is line-based — both helpers should use =org-element-at-point= / =org-in-block-p= to skip example/src/quote/verse blocks; (2) =lint-org.el= mutates its input on disk with no confirmation — passing five files reformatted all five (one by 1949 lines). A linter must report, not write; put the reformat behind an explicit =--fix= flag. +Send half shipped 2026-07-20. Folded a "Phone delivery" subsection into canonical =triage-intake.org= auto mode: a full-three-section sweep pushes to Craig's phone over Signal via =agent-text=, with a pointer from "End-of-sweep output" and a Living Document note. Signal-only by Craig's 2026-07-20 ruling — a quiet sweep's =nothing= heartbeat never reaches the phone, so silent-until-signal governs the phone channel too and the in-session heartbeat stays as proof-of-life. Falls back to inline when =agent-text= is absent. -Grading (severity × frequency, per todo-format.md): Critical severity (silent org data loss; recoverable here only because the content was git-staged) × rare-edge-case frequency (fires only when a mixed table+example file is passed to the helper) = P2 = [#B]. +Reply-polling half (the old =phone-recv=) deferred to the reply-correlation follow-up: with the Signal account linked on more than one device a reply fans out to every device and neither knows which page it answers, so auto mode pushes but does not poll until that's resolved. The recv wiring is owned by that spec, not this task. -Regression test: run =wrap-org-table.el= against a file containing a =#+begin_example= block whose lines start with =|= and assert the block is byte-identical afterward. Source: work handoff 2026-07-09 (=inbox/2026-07-09-1341-from-work-bug-data-loss-wrap-org-table-el-and.org=). +Origin: the work project's 2026-06-18 fold-in request; preserved bundle [[file:docs/design/2026-06-18-triage-intake-phone-push-note.org][note]] + [[file:docs/design/2026-06-18-triage-intake-phone-push-workflow.org][edited workflow]]. Transport re-pointed off retired ntfy → agent-text (renamed from agent-page 2026-07-20). +** CANCELLED [#D] Fully-unattended scheduled inbox check (/schedule cron pass) :feature: +CLOSED: [2026-07-20 Mon] +Merged 2026-07-20 into [[file:todo.org::*Unattended /schedule cron contract][Unattended /schedule cron contract — no-session variant]]. Same no-session design problem as the sentry /schedule variant (mutation rights, async surfacing, cross-run dedup, cron auth context); its inbox-specific context was absorbed there. +** CANCELLED [#B] lint-org resolves file: links against cwd, not the linted file :bug:solo: +CLOSED: [2026-07-23 Thu] +Not a defect. I filed this on a bad comparison and retracted it the same night. + +The claimed evidence was that linting =claude-templates/.ai/notes.org= from the repo root reports =protocols.org= and =workflows/first-session.org= missing, while linting it from its own directory reports neither. The first half of that was never run. The findings came from a =/tmp= copy of the template made while preparing the smoke diff, and in =/tmp= those two siblings genuinely are missing, so org-lint was right. I compared two different files and read the difference as a cwd bug. + +Retested three ways: the real file from the repo root gives zero link findings, the real file from its own directory gives zero, and only the =/tmp= copy gives two. A direct probe confirms =find-file-noselect= already sets =default-directory= to the linted file's directory, so the mechanism I proposed could not have been the cause either. + +Worth keeping from the episode: =link-to-local-file= is org-lint's own checker, not one lint-org.el implements, so a real fix here would mean pre- or post-filtering upstream output rather than editing a local checker. +** DONE [#A] Python and TypeScript bundles ship no secret-scan hook :bug: +CLOSED: [2026-07-23 Thu] +:PROPERTIES: +:LAST_REVIEWED: 2026-07-23 +:END: +Fixed 2026-07-23. Both bundles now ship all four components: =githooks/pre-commit= (the secret scan, shared verbatim with the other bundles, plus a language-appropriate syntax gate), =claude/hooks/validate-python.sh= / =validate-typescript.sh=, =claude/settings.json= wiring the PostToolUse hook, and a seed =CLAUDE.md=. 53 new tests (13 + 13 hook, 13 + 14 pre-commit), suite green. + +Two findings from the build worth keeping. First, =node --check= must never be used on TypeScript: it ignores =--experimental-strip-types=, so it rejects valid TS (an =interface= reads as a syntax error) and accepts broken TS. Measured on node v26.4.0. The hook uses =tsc= filtered to TS1xxx (syntactic) diagnostics instead, which also keeps type errors out of scope — those need the whole project graph. Second, the install now warns when a bundle lacks a documented component, which is the half that stops this recurring: nobody will remember the seventh bundle either, so the installer says so. + +Left undone deliberately: re-running =make install-lang= on =work= and =clock-panel=. Both are other projects, so that's a cross-project action for Craig. See the polyglot task below — =clock-panel= is python + typescript and can no longer install both. +The =python= and =typescript= language bundles carry rules, a coverage script, and a gitignore fragment. They carry no =githooks/pre-commit=, no =claude/hooks/=, no =claude/settings.json=, and no =CLAUDE.md=. The =bash=, =elisp=, and =go= bundles carry all four. + +The pre-commit hook is the secret scanner. So a project installing the Python or TypeScript bundle gets no credential scan on commit, no validate-on-edit hook, and no settings — while README's "Bundle structure" section documents all four as what each bundle follows, and notes.org describes the bundles as "rules + hooks + settings". + +Live as of 2026-07-23, verified by scanning every project carrying =.claude/rules/=: =work= (python) and =clock-panel= (python + typescript) both have no =githooks/= and no =.claude/settings.json=. The four elisp projects all have both. =work= is the one that matters — a work repo is where a leaked credential is most costly and most likely to reach a company remote. + +Not a design choice. Both bundles were added 2026-05-31; =go='s githooks landed 2026-06-02 and =bash='s 2026-06-23, so the hook rollout swept the two bundles added *after* these and skipped these. =install-lang.sh= guards its copy with =[ -d "$SRC/githooks" ]=, so the install succeeds silently and reports nothing missing — which is why this stayed invisible for nearly two months. + +Grading: Major severity (a documented security control absent, silently, with the install reporting success) x every user, every time (every install of either bundle, standing on 2 of the 6 bundle-using projects) = P1 = [#A]. + +Fix direction: port =githooks/pre-commit= to both bundles, adapting the language-specific half (the secret scan is common; =gofmt=/=shellcheck= becomes a formatter/linter check per language), add =claude/hooks/validate-*.sh= and =claude/settings.json=, and seed =CLAUDE.md=. Then make the gap loud: =install-lang= should warn when a bundle lacks a component the README documents, so the next partial bundle announces itself instead of installing quietly. Re-run =make install-lang= on =work= and =clock-panel= afterward. +** DONE [#C] Two language bundles' pre-commit hooks lost the cd guard :bug:quick:solo: +CLOSED: [2026-07-23 Thu] +:PROPERTIES: +:LAST_REVIEWED: 2026-07-23 +:END: +Fixed 2026-07-23. =go= and =elisp= now read =cd "$REPO_ROOT" || exit 1=, matching =bash=; the two new bundles were written with the guard. All five siblings agree, all five shellcheck clean. +=languages/bash/githooks/pre-commit= line 8 reads =cd "$REPO_ROOT" || exit 1=. The =go= and =elisp= copies of the same line read a bare =cd "$REPO_ROOT"=. Three siblings of one file, one hardened and two not — the guard landed in bash and never propagated. + +Low impact, stated honestly: git chdirs to the working-tree root before running a hook, so if =cd= fails the cwd is already correct and the checks still run against the right tree. Neither script sets =-e=, so a failure wouldn't abort them either. It takes a =GIT_WORK_TREE= or bare-repo edge case for =rev-parse --show-toplevel= to disagree with the cwd at all. + +Grading: Minor severity (belt-and-braces guard, no silent-pass path found — the checks run regardless) x rare edge case (needs a git configuration that makes toplevel differ from the hook's cwd) = P4... graded [#C] rather than [#D] because it's a two-character fix in a security-relevant gate and the drift pattern is the real signal: a fix landed in one bundle copy and stopped there, which is the same shape as the [#A] above. +** DONE [#B] Parked: zero markup in chat output, fences included (from org-drill) +CLOSED: [2026-07-23 Thu] +Craig approved 2026-07-23. Applied to =claude-rules/interaction.md=: the fenced-code-block carve-out is gone, replaced with a zero-markup-always statement citing his 2026-05-30 direction. The rule's factual claim stays honest — fences don't invert the way inline spans do, so the text says they read as markup he didn't ask for rather than inventing a rendering problem, and points at plain indented text or a named file path as the way to hand over something copyable. It was the only copy of the carve-out. +** DONE [#B] Parked: sentry Living Document updates from two dogfood runs (from takuzu + archangel) +CLOSED: [2026-07-23 Thu] +Craig approved 2026-07-23. Four folds applied to =sentry.org=: =--archive-done= touches =.gitignore= on its first run so an "org-only" pass can still produce a commit, a mirror-only project's quiet fires leave no commits so the anchor's heartbeat list is the only record, randomized property sweeps are good quiet-fire work, and the task-audit pass splits into a mechanical hourly subset plus a nightly judgment half. The fifth finding (make the bug hunt official) had already landed as pass 11, so it became a corroboration note. Takuzu's lint-org finding stayed a filed bug task rather than a workflow edit, since it's a defect not a design change. +** DONE [#B] Parked: clear four lint flags in the notes.org template (from smoke) +CLOSED: [2026-07-23 Thu] +Craig approved 2026-07-23. Applied to =claude-templates/.ai/notes.org= and synced to the mirror. The template now lints with zero mechanical and zero judgment findings, down from four. Two fixes beyond what smoke reported: two more column-0 bold lines, and the real cause of the block flags — a literal =** Feature Name= inside the example block that org parses as a heading, cleared by comma-escaping that one line. Worth remembering: those heading flags were mechanical, not judgment, so =lint-org --fix= in any project would have rewritten the template locally and drifted it from canonical. +** DONE [#B] Parked: clear temp/ during wrap-up teardown (from your roam capture) +CLOSED: [2026-07-23 Thu] +Craig approved 2026-07-23. Added =*** Clear temp/= to =wrap-it-up.org= Step 3, after the archive pass, and synced to the mirror. Two guards: confirm before deleting anything that reads as in-progress rather than throwaway (that belongs in =working/=, so move it there), and skip entirely where =temp/= isn't gitignored, since that means the project uses the directory for something else. This closes the last open clause of his 2026-07-20 working/temp capture. +** DONE [#B] inbox-send leaves a phantom empty handoff in the target inbox :bug:solo: +CLOSED: [2026-07-23 Thu] +:PROPERTIES: +:LAST_REVIEWED: 2026-07-23 +:END: +Fixed 2026-07-23 (speedrun, 0f91a8e). Both send paths write to a .inbox-send-* temp sibling and os.replace it into place, so a mid-write failure leaves no phantom; utf-8 pinned on the write and the roots read; inbox-status skips the in-flight temp. 5 new tests (atomic write, no-partial-on-failure, no-temp-on-success, both paths) + 1 inbox-status bats. Review found and fixed one Important: the temp was discoverable by inbox-status during the write window. +=inbox-send.py= writes straight to the destination path in the target project's =inbox/=. =Path.write_text= opens with mode =w=, which creates and truncates before any content is written, so *any* failure between opening and finishing leaves a zero-byte =.org= file sitting in another project's inbox. + +That file is not inert. =inbox-status= counts it as a pending handoff, so it trips the receiving project's =inbox-boundary-check= Stop hook and blocks a turn there. The receiving agent then has to resolve a handoff with no content and no sender context, which it cannot do from the file. Meanwhile the sender saw an error and will most likely retry, so the target gets a second file too. + +Reproduced 2026-07-23 end to end: a send whose text contains non-ASCII under =LC_ALL=C PYTHONUTF8=0 PYTHONCOERCECLOCALE=0= fails on the ASCII codec, leaves a zero-byte file in the destination inbox, and =inbox-status= in that project then reports it as pending. + +The encoding case is one trigger, not the defect. =write_text= and =read_text= are both called with no =encoding= argument, so they follow the locale; passing =encoding="utf-8"= closes that trigger. The defect underneath is the non-atomic write, which a full disk, a revoked permission, or an interrupted process reaches just as easily. + +Grading: Major severity (a phantom handoff in a *different* project's inbox, unresolvable from its own content, that blocks a turn there via the boundary hook) x some users, sometimes (any mid-write failure, of which the locale case is only the one reproduced) = P2 = [#B]. + +Fix direction: write to a temp file in the destination directory and =os.replace= into place, so the inbox only ever sees a complete file. Pin =encoding="utf-8"= on both =write_text= and the =read_text= in =resolve_roots=. Same treatment for =send_file=, whose =shutil.copy2= has the same shape. Test by forcing a mid-write failure and asserting the destination directory is unchanged. +** DONE [#C] Two smaller inbox-send defects: uncaught traceback, duplicate roots :bug:quick:solo: +CLOSED: [2026-07-23 Thu] +:PROPERTIES: +:LAST_REVIEWED: 2026-07-23 +:END: +Fixed 2026-07-23 (speedrun, a053e9d). main now catches OSError, so an unreadable source gives the clean inbox-send: error rather than a traceback. discover_projects dedupes on the resolved path. 2 tests. +Both verified 2026-07-23, both in =.ai/scripts/inbox-send.py=, both low-harm. Grouped because they're the same file and the same fix session. + +1. *Uncaught =PermissionError=.* =main= catches =(ValueError, FileNotFoundError)= around the send, but =send_file= reaches =shutil.copy2=, which raises =PermissionError= on an unreadable source. That is an =OSError=, not caught, so the user gets a raw Python traceback instead of the clean =inbox-send: <message>= error every other failure path produces. Reproduced with a =chmod 000= source. No partial file is left in this case. Fix: catch =OSError= alongside the other two. + +2. *Duplicate roots list a project twice.* =discover_projects= appends without deduping, so a roots config naming both a parent and one of its children (=/x= and =/x/proj=) lists =proj= at two different numeric indices. Reproduced via =INBOX_SEND_ROOTS=. Harmless today — both indices resolve to the same project, so no message goes to the wrong place, and Craig's current =~/.claude/inbox-roots.txt= has no overlap. Fix: dedupe on =resolve()= before returning. + +Grading: Minor severity (one is cosmetic output noise, the other an ugly but accurate failure message; neither loses or misroutes a message) x rare edge case (an unreadable source file, or a roots config with an overlap) = P4... graded up to [#C] rather than [#D] because both fixes are one line each and sit in a script every project depends on for cross-project messaging. +** DONE [#C] lint-org todo-format checkers fire on spec files :bug:solo: +CLOSED: [2026-07-24 Fri] +:PROPERTIES: +:LAST_REVIEWED: 2026-07-23 +:END: +Fixed 2026-07-24 (speedrun, c38bab9). A lo--spec-file-p path guard skips all five todo-format-family checkers on files under docs/specs/. Verified: the nine repo specs go from 100 todo-format findings to zero, todo.org keeps its full checker set. 7 ERT tests. +=level2-done-without-closed= and =level-2-dated-header= encode =todo.org= completion conventions, but they run against every org file, including =docs/specs/=. A spec's Decisions section legitimately uses =** DONE <decision>= with no =CLOSED:= cookie, and its Review-and-iteration-history section legitimately uses =** <dated> — <who> — <role>= headings. Neither is a completion defect there. + +Reproduced 2026-07-23 across rulesets' own specs: 100 findings over 7 of the 9 files in =docs/specs/= (docs-lifecycle 27, sentry-workflow 25, autonomous-batch 11, wrapup-routing 11, inbox-consolidation 10, agent-kb 8, encourage-kb 8; the two 2026-07-20 specs are clean). Every project carrying =docs/specs/= inherits the same noise on every sweep. Reported independently by takuzu's first sentry dogfood run. + +Grading: Minor severity (judgment-kind output, so nothing mutates; the cost is noise that trains the reader to skim) x every user, every time (fires on every sweep in every project with specs) = P2... graded down to [#C] because the two inputs disagree: the frequency row is genuinely universal, but Minor severity with zero mutation risk and a known cause reads as P3. Recorded so the read can be argued. + +Fix direction: scope the todo-format-family checkers away from =docs/specs/= by path. Correction (2026-07-23 speedrun): the earlier note that these are org-lint's own checkers was wrong — =lo--check-level2-dated-headers= (line 413) and =lo--check-level2-done-without-closed= (line 507) are lint-org.el's own functions, so the fix is a direct local guard, not upstream filtering. + +Decision (Craig, 2026-07-23 speedrun pre-flight): scope *all four* todo-format-family checkers, not just the two that fired — the two named plus =subtask-done-not-dated= and =dated-log-heading-active-timestamp=, which would misfire the same way on a spec's dated review-history headings. Path-based (any file under a =docs/specs/= segment), matching the docs-lifecycle canon that specs live there. Takuzu's spec-create alternative fixes new specs only and leaves the nine existing ones noisy, so path-scoping is the general fix. +** DONE [#C] notes.org template trips four lint-org flags in every project :bug:quick:solo: +CLOSED: [2026-07-24 Fri] +:PROPERTIES: +:LAST_REVIEWED: 2026-07-23 +:END: +Already satisfied. The canonical claude-templates/.ai/notes.org was fixed in 10ea44b (2026-07-23) when Craig approved the smoke proposal — the two column-0 bold lines rephrased, the example block's marker comma-escaped. Verified 2026-07-24: the template lints with zero mechanical and zero judgment findings, down from four. This TODO and the applied smoke VERIFY were duplicate work items for the same fix; closing DONE citing the commit rather than filing a no-op VERIFY, since the end-state is verifiable now with nothing to ask. +The synced =claude-templates/.ai/notes.org= carries two boilerplate lines that open with markdown-style bold at column 0 (=**Session history is NOT in this file.**=, =**For protocols and conventions, see:**=). Org parses a line-initial =**= as a level-2 heading, so =misplaced-heading= flags both, and the =#+begin_example= block in the Pending Decisions instructions trips =invalid-block= twice. Every project inherits all four on every sweep. + +Reproduced 2026-07-23. Confirms smoke's proposal. Additional finding from the same run: these two are =mechanical-fixed=, not judgment, so =lint-org --fix= silently rewrites the template's =**bold**= to org =*bold*=. The rewrite is correct org, but it lands on a synced template, so whichever project runs =--fix= first creates drift against canonical. + +Grading: Minor severity (cosmetic noise; the mechanical rewrite is correct org and harmless in isolation) x every user, every time (every project, every sweep) = P2... same disagreement as the checker task above; graded [#C] on the no-real-harm read. + +Fix direction: rephrase both lines in the canonical template so no line starts with =**= (a list dash, or move the bold off column 0), and comma-escape the example block's own markers. One canonical fix clears it everywhere. +** DONE [#B] cj-remove-block silently deletes content between two cj blocks :bug:solo: +CLOSED: [2026-07-24 Fri] +:PROPERTIES: +:LAST_REVIEWED: 2026-07-24 +:END: +Fixed 2026-07-24 (sentry fire 2, 17f5d48). The range validation now looks inside the range and refuses when a second fence appears before the end. Verified it doesn't over-tighten: indented blocks, interior blank lines, legacy single-line annotations, and prose mentioning a fence mid-line all still validate. Second defect fixed in the same commit: remove_range now backs up to /tmp (matching lint-org.el) and writes atomically through a temp sibling, utf-8 pinned. 6 new tests. +=looks_like_cj_range= validates only that the *first* line of the range opens a =#+begin_src cj:= fence and the *last* line closes with =#+end_src=. It never checks that the range holds exactly one block. A range spanning one block's opening fence to a *later* block's closing fence passes validation, and =remove_range= then deletes everything between — real prose, headings, whole tasks — silently, exit 0. + +That is precisely the failure the validation exists to prevent. Its docstring says it "protects against accidentally trimming the wrong block when line numbers drift between a cj-scan call and a remove call", and drift is the normal operating mode: =respond-to-cj-comments= edits the file as it processes each item, and a file being processed for cj comments generally holds several of them. + +Reproduced 2026-07-24 on a fixture with two cj blocks separated by real content. =looks_like_cj_range(lines, 2, 10)= returned =ok=True=, and the removal reduced a 10-line file to its first line. A heading and two content lines were destroyed with no warning and a zero exit. + +Grading: Major severity (silent destruction of real content in the file that holds Craig's tasks and notes, with no warning and a success exit; git recovers only to the last commit, so intra-session work is lost) x some users, sometimes (needs drift plus multiple cj blocks, which together are the skill's normal operating mode) = P2 = [#B]. + +Fix direction: =looks_like_cj_range= must confirm the range contains exactly one block — scan lines start..end and reject when any =#+end_src= appears before the final line, or when any second =#+begin_src cj:= appears after the first. Test with the two-block fixture above asserting the validation refuses. + +Second defect, same file, same fix session: =remove_range= writes the mutated org file with a bare =path.write_text=, which truncates the target on open, and it takes no backup first. A mid-write failure leaves Craig's =todo.org= truncated. =lint-org.el= — the other tool that mutates these files — copies a backup to =/tmp/<basename>.before-lint-pass.<timestamp>= before touching anything. cj-remove-block should match that: back up first, then write atomically via a temp file and =os.replace=, the same shape shipped for =inbox-send= in 0f91a8e. Pin =encoding="utf-8"= on the read and write while there. +** DONE [#D] route_recommend downgrades strong to weak on a duplicate basename :bug:quick:solo: +CLOSED: [2026-07-24 Fri] +:PROPERTIES: +:LAST_REVIEWED: 2026-07-24 +:END: +Fixed 2026-07-24 (sentry fire 2, 1b0f284). The dedupe went into recommend rather than discover_destination_names as the task proposed, so every caller of the pure core is protected, not just the CLI path. Identical names collapse; genuine ambiguity between two different projects still downgrades to weak, pinned by a test. 3 new tests. +=discover_destination_names= collapses discovered projects to bare basenames (=[p.name for p in ...]=). Two projects sharing a basename across roots (=~/code/notes= and =~/projects/notes=) therefore appear twice in the candidate list, both literal-match the same item, and =recommend= reads =len(strong) > 1= as an ambiguous tie — downgrading a correct strong match to weak. + +Reproduced 2026-07-24 by direct probe: =recommend("fix the notes thing", ["notes", "other"])= returns =('notes', 'strong')=, while the same call with =["notes", "notes", "other"]= returns =('notes', 'weak')=. + +Latent, not live: the current project set is 27 projects with 27 distinct basenames, so no collision exists today. The destination stays correct either way; only the confidence tier is wrong, which costs an unnecessary routing prompt rather than a misroute. + +Grading: Minor severity (right destination, wrong tier, cost is one extra prompt) x rare edge case (needs a basename collision across roots, which doesn't currently exist) = P4 = [#D]. + +Fix direction: dedupe names in =discover_destination_names= (=list(dict.fromkeys(names))=, order-preserving). A duplicate basename is one addressable name as far as routing goes, since =inbox-send='s =find_target= resolves a name to its first match anyway. Add a test with a duplicated candidate asserting the tier stays strong. +** DONE [#C] audit.bats has a flaky teardown that produces false suite reds :bug:test:solo: +CLOSED: [2026-07-24 Fri] +:PROPERTIES: +:LAST_REVIEWED: 2026-07-24 +:END: +Fixed 2026-07-24 (sentry fire 3, 7f45d4b). The fixture now sets =maintenance.auto false= and =gc.auto 0= before staging. + +Cause, traced not guessed: =git commit= spawns =git maintenance run --auto --quiet --detach= on git 2.55. The commit returns while that detached process is still writing a pack, and teardown's =rm -rf= races it. Every failed run left a =tmp_pack_*= behind, which is what pointed at it. + +*My original lead in this task was wrong* and worth recording as such. It blamed =gc.auto='s loose-object threshold; the object counts kill that outright (a fixture holds five objects against a default threshold of 6700). The right knob on modern git is =maintenance.auto=, and the mechanism is the =--detach=, not the threshold. Filing the theory as an explicitly-labelled lead rather than a finding is what kept it from being implemented as fact. + +Validation: 20 consecutive runs clean with no leftover fixture directories, against a baseline of roughly one failure in eight. Stated honestly, 20 clean runs alone would be about 7% likely by luck at that rate, so the statistics confirm rather than prove — the trace is the evidence. Disabled the background writer rather than retrying the delete, since a retry loop hides a live process instead of removing it. +=scripts/tests/audit.bats= test 4 ("tracked project with dirty .ai/ is skipped") intermittently fails in its =teardown=, not its assertions: =rm -rf "$TEST_HOME"= exits non-zero with =rm: cannot remove '/tmp/audit-bats.XXXX/code/alpha/.git/objects': Directory not empty=. The test body passes; only the cleanup fails, and bats reports the whole test as failed. + +Verified intermittent 2026-07-24: it failed twice during a sentry fire (once in a full =make test=, once running the file alone), then passed three consecutive runs immediately afterward with an identical tree. That intermittency is the defect — it makes =make test= return a false red. + +This matters more than a normal flaky test because the green suite is load-bearing for sentry: it gates entry, and the fire-end conditional run gates whether a night's commits are trusted. A spurious red there either blocks a fire or flags a clean night for morning review. + +Grading: Minor severity (a teardown-only failure, no production code implicated, and the assertions themselves pass) x some users, sometimes (fired twice in roughly six runs tonight) = P3 = [#C]. + +Lead, not a verified cause: =audit.bats= line 42 runs =git init -q= in each fixture project and sets no =gc.auto=, while =audit.sh= runs five git commands against them. Git can fork background maintenance (=gc --auto=) that keeps writing into =.git/objects= after the foreground command returns, which would race the teardown's =rm -rf=. That fits the symptom exactly but is untested. Confirm before fixing. + +Fix direction (once the cause is confirmed): set =git config gc.auto 0= (and =maintenance.auto false=) in the fixture setup so no background writer exists. Failing that, make teardown resilient rather than papering over it — a retry loop hides a real writer instead of removing it, so prefer killing the writer. +** DONE [#C] todo-cleanup rewrites todo.org with no backup, unlike its siblings :bug:solo: +CLOSED: [2026-07-24 Fri] +:PROPERTIES: +:LAST_REVIEWED: 2026-07-24 +:END: +Fixed 2026-07-24 (sentry fire 4, 0686784). Copies to =/tmp/<basename>.before-todo-cleanup.<stamp>= before the first mutation, matching lint-org.el's convention, once per invocation and skipped under =--check=. 2 ERT tests. The two non-findings recorded in this task's original body stand: the archive move was already fail-safe, and missing-file behavior is unchanged (exit 255, nothing created) — both re-verified against the pre-change version. +=todo-cleanup.el= rewrites =todo.org= in place (hygiene fixes, =--convert-subtasks=, =--archive-done=, =--sync-child-priority=) and leaves no copy behind. The two sibling tools that mutate the same files both do: =lint-org.el= copies to =/tmp/<basename>.before-lint-pass.<stamp>= and =wrap-org-table.el= does the equivalent. =cj-remove-block= joined them in 17f5d48. todo-cleanup is the outlier, and it is the one that runs most often — wrap-up calls it, and every sentry fire's pass 4 calls it three times (four runs tonight alone). + +Verified 2026-07-24: after a real =--convert-subtasks= mutation on a fixture, the directory holds only =todo.org=. Emacs's own backup mechanism does not fire under =--batch -q=, so there is genuinely no undo short of git, which recovers only to the last commit and loses intra-session work. + +Two things this is NOT, both checked so they don't get re-investigated: + +- The archive move is *fail-safe*, not lossy. It deletes subtrees from the buffer, writes the archive file, and only saves =todo.org= at the very end, so an archive-write failure aborts before the save. Verified by making the archive directory unwritable: exit 255, =todo.org= byte-identical, content intact. My initial hypothesis that a mid-move failure could lose a subtree from both files was wrong. +- No error swallowing on the mutation path. The only =ignore-errors= in the file wrap =call-process "git"=, not any write. + +So this is a hardening gap rather than an active bug: a future defect in a mechanical rewriter would have no undo. + +Grading: Minor severity (no known active defect; the exposure is that any future one is unrecoverable within a session) x every user, every time (it runs on every wrap and every sentry fire) = P2 = [#C]. + +Fix direction: back up before the first mutation, matching =lint-org.el='s convention exactly — =/tmp/<basename>.before-todo-cleanup.<YYYYMMDD-HHMMSS>=. One copy per invocation, not per pass, and skip it under =--check= (which writes nothing). Test by asserting the backup exists and holds the pre-edit content after a real mutation. +** DONE [#C] claude-templates/bin/ gets no lint coverage at all :bug:quick:solo: +CLOSED: [2026-07-24 Fri] +:PROPERTIES: +:LAST_REVIEWED: 2026-07-24 +:END: +Fixed 2026-07-24 (sentry fire 8, f91feef). lint.sh now sweeps =claude-templates/bin/*= through =check_hook=, matching the extensionless-file shape =languages/*/githooks/*= already uses. Added =scripts/tests/lint-coverage.bats=, which pins the *coverage* rather than current cleanliness: it plants a broken file in each swept location and asserts lint.sh complains, so a location that silently stops being swept fails the suite. The broader question this surfaced — whether rulesets should run shellcheck on its own shell at all — is the VERIFY below and stays Craig's call. +=scripts/lint.sh= sweeps =scripts/*.sh=, =languages/*/claude/hooks/*.sh=, and =languages/*/githooks/*= through =check_hook= (shebang present, executable bit set). It never touches =claude-templates/bin/=. Verified: zero references to that path in the file. + +Those four scripts — =ai=, =agent-text=, =agent-page=, =install-ai= — are the ones =make install= symlinks into =~/.local/bin=, so they run on Craig's PATH on every machine. They are the *most* exposed shell in the repo and the only shell with no gate over it. + +All four are clean today (shebangs present, mode 755, shellcheck-clean when run by hand), so nothing is broken. This is a missing gate, not an active defect: the =ai= launcher was hardened to 42 tests recently and that cleanliness is not enforced going forward. + +Grading: Minor severity (nothing broken now; the exposure is a future regression in a PATH-installed script going uncaught) x every user, every time (every =make lint= silently skips them) = P2 = [#C]. + +Fix: add =claude-templates/bin/*= to the =check_hook= loop, the same shape =languages/*/githooks/*= already uses for extensionless files. A no-op today by design — it passes immediately — which is exactly what a guard should do. +** DONE [#A] Applied: the secret-scan pre-commit fails open in ALL FIVE bundles (from .emacs.d) +:PROPERTIES: +:LAST_REVIEWED: 2026-07-24 +:END: +CLOSED: [2026-07-24 Fri] +Applied 2026-07-24 across all five bundles (11 sites: the secret-scan input in each, plus each bundle's staged-file list, typescript having two). Verified on all five: refuses when the diff can't be read, still blocks a real staged secret, still passes a clean commit. Adopted .emacs.d's elisp bats suite (8 tests) and extended the repo-level cross-bundle suite with two fail-closed assertions. + +Separate defect found while wiring that up: the cross-bundle suite's VARIANTS list read "elisp bash go" while python and typescript also shipped hooks, so every "in every variant" assertion had silently skipped two bundles since I added them. VARIANTS is now discovered from the tree rather than enumerated — the same failure class the tests exist to catch. + +What arrived: .emacs.d found that =languages/elisp/githooks/pre-commit= builds its scan input as =added_lines="$(git diff --cached -U0 ... | grep '^+' | grep -v '^+++' || true)"=. With no pipefail and =|| true= swallowing everything, any git failure yields an empty string, so the scan searches nothing, finds nothing, reports clean, and the commit proceeds with the secret in it. The staged-file list feeding the paren check has the identical hole. + +*Verified independently, and it is worse than reported.* I reproduced the fail-open with a stub git that fails only the staged-diff call: exit 0 with an AWS-shaped key staged. Then I checked the other bundles, which the sender did not: *bash, go, python, and typescript all carry the same pattern and all fail open the same way.* Confirmed live on all five. Two of those (python, typescript) are hooks I wrote on 2026-07-24 by copying the bash one, so I propagated the defect while closing a different gap. + +Graded [#A] on severity alone, per the todo-format security carve-out: a credential-scanning gate that reports clean without having looked is a showstopper regardless of how rarely git fails. + +The sender's fix is correct and I verified all three axes on it: refuses to proceed when the diff cannot be read (exit 1), still blocks a real staged secret (exit 1), still passes a clean commit (exit 0). It splits the git read from the greps so a git failure aborts while "grep matched nothing" stays the ordinary case. + +Decision needed: the fix as sent covers elisp only. It should be applied to all five bundles, which is my scope expansion rather than the sender's proposal — hence a VERIFY rather than a silent apply. Also unresolved: where the two attached bats suites live, since the hooks are rulesets-owned and the tests currently sit in the consuming project. + +Prepared: [[file:working/hook-fail-open/pre-commit.diff]], plus =test-pre-commit-hook.bats= (8 tests) in the same dir. +** DONE [#B] Applied: remove the validate-el auto-test cap (from .emacs.d, Craig's call) +:PROPERTIES: +:LAST_REVIEWED: 2026-07-24 +:END: +CLOSED: [2026-07-24 Fri] +Applied 2026-07-24. Cap and the unreachable notice helper removed; the gate is now count -ge 1. Adopted the rewritten bats suite (6 tests), with its hook path corrected from the installed .claude/hooks/ layout to the bundle's claude/hooks/ — the re-homing mismatch the sender flagged. + +What arrived: a superseding handoff. The first proposed a loud notice when the test count exceeds =MAX_AUTO_TEST_FILES=20= (above the cap the block was skipped, nothing printed, exit 0 — indistinguishable from a pass). Craig chose in the .emacs.d session to remove the cap instead. + +The reasoning is the valuable part. Measured, a whole family runs in under two seconds (calendar-sync 63 files / 633 tests / 0.9s), so the cap bought nothing. Worse, it was *concealing* a real cross-test pollution bug: calendar-sync exits 1 when its 63 files run in one process, because one test marks a calendar as syncing and never resets it, and a sibling file's test then hits the stale guard. Invisible from both directions — =make test= runs each file in its own Emacs, and the hook skipped the family for being over the cap. + +Verified the superseding file drops the cap entirely (zero references, gate is now =count -ge 1=) and removes the now-unreachable notice helper. + +Since Craig already made this call, the remaining decision is only adoption scope: removing the cap means other consuming projects run every stem-matched test file per edit, and one may go red on first use by surfacing pollution that per-file runs hid. The sender flags that as intended. + +Prepared: [[file:working/hook-fail-open/validate-el.diff]], plus =test-validate-el-hook.bats= (6 tests, two of which stage a deliberately failing test so a quiet pass proves the run happened). +** DONE [#C] lint-org invalid-block false-positives inside example/src blocks :bug:solo: +CLOSED: [2026-07-24 Fri] +:PROPERTIES: +:LAST_REVIEWED: 2026-07-24 +:END: +Fixed 2026-07-24, test-first. =lo--matched-block-regions= scans lines for correctly paired blocks under org's real rule — once a block is open, only its own =#+end_TYPE= closes it — and =lo--handle-item= drops an =invalid-block= finding whose line falls in one, delimiters included (org-lint reports at the delimiters themselves). Line-scanning rather than asking org is the point: org's parser is what mis-reads these blocks. 4 ERT tests: the heading-in-example case, a src block holding a literal =#+end_example=, a genuinely unterminated block that must still report, and a file with one of each proving the suppression is per-block not per-file. + +Verified against home's fixture end to end: =/home/cjennings/projects/home/.ai/notes.org= produced exactly the two reported findings (lines 386, 398) under the pre-change script and zero under the new one, file untouched. Full suite green. + +Left alone deliberately: the =,**= comma-escape in =claude-templates/.ai/notes.org= line 53. The task noted the fix "lets that escape be reverted," but the escape is the documented org convention for a literal =**= inside a verbatim block, so reverting it would trade correct org for no gain now that the finding is suppressed either way. + +home reported it, and I verified it: an =#+begin_example= block whose body contains a line beginning =** = (or any heading-shaped line) makes =invalid-block= flag *both* delimiters as "Possible incomplete block", even though the block is correctly paired. The checker reads the heading line inside the verbatim body as a structural break and loses the open block. + +Reproduced 2026-07-24 on a three-line example block: 2 judgment findings, both false. This is a docs-file-common shape — any org file documenting org syntax inside an example block hits it (home's notes.org PENDING DECISIONS section does). + +*This is the root cause behind a workaround I already shipped.* During fire 1 last night I comma-escaped exactly this =** Feature Name= line in =claude-templates/.ai/notes.org= to silence the two findings. That fixed the symptom in one file; the checker bug it worked around is still live and recurs everywhere. Fixing the checker lets that escape be reverted. + +Grading: Minor severity (judgment output, nothing mutates; pure noise that trains the reader to skim) x most users, frequently (every org file documenting org syntax in a verbatim block) = P3 = [#C]. + +Fix direction (per home): while inside a =begin_example= / =begin_src= block, skip structural parsing of the body until the matching =#+end_= line — verbatim blocks contain no headings, timestamps, or delimiters by definition. The same class hits a src block containing =#+end_example= as literal text. Tagged :solo: because the fix and its test are mechanical. + +*** 2026-07-24 Fri @ 20:10:00 -0500 Open question settled — invalid-block is org-lint's, so the fix is a filter +Home answered it and I re-verified both halves here: =grep invalid-block= over =lint-org.el= returns nothing, and =org-lint--checkers= enumerates =invalid-block= in batch Emacs alongside =link-to-local-file=, =invalid-babel-call-block=, and =missing-language-in-src-block=. So this is the same shape as the =link-to-local-file= episode: suppress an =invalid-block= finding whose line falls inside a verbatim block, on our side of org-lint's output. No local checker to edit. + +Regression fixture, ready to use: home's own =.ai/notes.org= PENDING DECISIONS example block (lines 386-398) holds an unescaped =** Feature Name or Topic= and trips both delimiters — findings at lines 386 and 398. Home is deliberately leaving its copy unescaped so it stays a fixture, and asked to be pinged when the filter lands so it can re-run. The filter should take those two findings to zero without touching the file. +** DONE [#C] sentry.org calls one loop cycle a "fire" :chore:solo: +CLOSED: [2026-07-28 Tue] +:PROPERTIES: +:LAST_REVIEWED: 2026-07-28 +:END: +Done 2026-07-28. 72 noun-sense instances in =sentry.org= became "cycle"; the four verb-sense uses stayed. Three more lived outside the file — =wrap-it-up.org= ("a crashed fire"), =todo-cleanup.el= and its test ("every sentry fire") — all naming a sentry cycle, so they moved too. home's proposal was scoped to =sentry.org= alone, so the leak would have split the vocabulary across files. + +Craig read home's "nine fires" as nine emergencies and went looking for what was burning (2026-07-28): "I assume you mean nine crises, not nine loop cycles and I begin to get scared." The term reaches him directly — digest headings render as =** Fire 11 — 08:32 CDT= in the anchor he reads every morning. 35 instances in =sentry.org=. + +Grading: Minor severity (a user-facing artifact that miscommunicates, no data loss) x most users frequently (every digest, on every project running sentry) = P3 = [#C]. + +*Take the problem, not home's proposed term.* home proposed "pass", reasoning that it already lives in the file's vocabulary. That is exactly what disqualifies it. =sentry.org= already uses "pass" as a precise numbered noun: "the pass list", "the Pass Runner", "eleven finding/hygiene passes", "pass 12" for the implementation pass. There are exactly eleven hygiene passes, so home's proposed digest heading =** Pass 11= collides with an existing real referent. Renaming would trade a term Craig misreads as urgent for one that is genuinely ambiguous. + +Counter-proposal: *cycle*. It appears zero times in =sentry.org=, so there is nothing to collide with. It is also Craig's own word from the very quote that surfaced this ("not nine loop cycles"). =** Cycle 11 — 08:32 CDT= reads cleanly. Checked and rejected: "sweep" (already used for hygiene and property sweeps) and "run" (already used as a noun, "first live run"). + +Keep the verb sense of fire throughout ("the notify fires", "the path never fires") — only the noun meaning one loop cycle changes. Past session anchors are historical records and stay as written. + +Craig approved "cycle" on 2026-07-28. That settles the only judgment the task carried, so it is =:solo:= now: the surface is the 35 noun-sense instances in =sentry.org=, and the completion check is objective — zero noun-sense "fire" left, verb sense untouched, lint clean, suite green, canonical and mirror in sync. + +Source: home handoff, 2026-07-28. +** DONE [#B] references/ is linked from protocols.org but never synced :bug: +CLOSED: [2026-07-28 Tue] +:PROPERTIES: +:LAST_REVIEWED: 2026-07-28 +:END: +Craig picked option 1 on 2026-07-28: drop the link, point at the calendar workflows instead. The four of them sync already and carry the MCP tool names, both account ids, the gcalcli fallback, and the conflict-check discipline — everything the reference was cited for. + +The adversarial review then returned =Needs Discussion= and widened the fix twice, both correctly: + +- My first replacement said credentials "live in the rulesets repo" without naming a file. The only calendar-named document there was =calendar-reference.org=, whose three credential paths have all been dead since the OAuth keys moved into the encrypted MCP bundle in May. So the prose sent a reader to a stale file, which fails more quietly than the dead link it replaced. Now names =mcp/README.org=, the real authority, verified to document =gcp-oauth.keys.json= as gitignored and regenerated at install. +- =calendar-reference.org= was left orphaned by the link removal: zero live inbound references, two copies, and =references/= is not in =sync-check.sh='s gate (=paths=(protocols.org workflows scripts)=), so the copies could drift silently. Deleted both, and the empty =references/= directories with them. Its operational content is fully covered by the four workflows and its credential paths were all stale, so nothing live was lost. + +The review also found the same defect class at seven other sites, filed separately. + +One fact died with the file, dropped deliberately rather than by accident: that the Google Cloud app runs in production mode, so tokens don't expire after seven days. It's checkable in the console, and it was the last live line in a document whose other credential facts had all gone stale. + +=protocols.org:273= links to =references/calendar-reference.org=, but startup.org's rsync copies only =protocols.org=, =workflows/=, and =scripts/=. So the link is dead in every consuming project. Confirmed: the file exists at =claude-templates/.ai/references/=, and neither home nor =.emacs.d= has a =.ai/references/= at all. + +Grading: Minor severity (a documented reference an agent can't follow, workaround is to search or ask) x every user every time (every consuming project, every sync) = P2 = [#B]. + +Pinned 2026-07-28 for Craig's decision. Analysis below is complete; only the choice is open. + +*Revised recommendation: drop the link, don't add the sync.* My first read was add =references/= to the rsync. Looking at what the directory actually holds reversed it: + +- =references/= holds exactly one file, and =protocols.org:273= is its only citation anywhere in the tree. +- The four calendar workflows (=add-=, =edit-=, =delete-=, =read-calendar-event(s).org=) already sync to every project, and already carry the operational detail: the MCP server name, both account IDs, the gcalcli fallback, conflict-checking. None of them cites =calendar-reference.org=; they are self-sufficient. +- What the reference uniquely adds is credential *file locations* (an OAuth keys path, a GPG-encrypted gcalcli secret) — no secret values. Re-auth is a rare operation Craig performs himself, not something an agent in another project needs a pointer to. + +So adding a synced directory carrying =--delete= semantics, to deliver one mostly-redundant file, is a poor trade. It also cuts against the rightsizing work: =protocols.org= is read into context every session, and a dead link is noise in a file being slimmed. + +Preferred fix: replace the link with a one-line pointer to the calendar workflows, which travel and are current. The reference file stays rulesets-only for the credential locations. + +Alternatives if Craig prefers: add =references/= to the rsync (the =--delete= hazard is not present today — work is the only project with a =.ai/references/= and its copy is byte-identical), or fold the credential locations into the calendar workflows and then drop the link (most complete, but spreads local absolute paths across four more synced files). + +Not =:solo:= — it changes a synced template, which needs Craig's approval by the inbox rule, and the sync-vs-drop choice is his. + +Source: winvm link-integrity pass, 2026-07-28. +** DONE [#B] Parked: telegram source treats "down" as launch, not SCAN FAILED (from .emacs.d) +CLOSED: [2026-07-28 Tue] +:PROPERTIES: +:LAST_REVIEWED: 2026-07-24 +:END: +Applied 2026-07-28, merged with the segfault root-cause fix that arrived the same morning rather than applied alone. + +Merging was necessary, not tidiness. The parked proposed file still carried the bad =(telega--loadChats 'main)= call at its own lines 52 and 122, so applying it as-is would have shipped a file that fixed the wording defect while preserving the call that kills the server. Its third hunk also added prose citing "tdlib segfaults in native mode (SEGFAULT gotcha below)" — pointing at the section the new handoff rewrites to say those deaths were our own bad argument, not tdlib memory corruption. + +What landed: all three parked hunks (the down-is-launch directive, the SCAN-FAILED-only-after-launch-attempted rewording, the =(setq telega-use-docker t)= restored to the Step 1 code block), plus both corrected =loadChats= call sites and the rewritten gotcha. The native-mode prose was reconciled in two places so it no longer leans on the refuted story: the Step 1 comment now states plainly that docker mode and the loadChats bug are separate concerns (the deaths happened *in* docker mode, so docker mode is neither a defense against it nor evidence for it), and the Quick Reference line says "crashed in native mode (2026-06-09)" instead of "segfaults", with the same disambiguation. + +Verified: both live call sites use the TL object; the two remaining ='main= occurrences are inside the gotcha prose describing the bug. lint-org clean, mirror synced, suite green. diff --git a/voice/SKILL.md b/voice/SKILL.md index 19eb38b..cdf7874 100644 --- a/voice/SKILL.md +++ b/voice/SKILL.md @@ -1,7 +1,7 @@ --- name: voice description: | - Multi-pass prose editor with three modes. General mode (default) edits arbitrary writing — research notes, essays, anyone's README prose — by walking 31 patterns: Wikipedia's Signs of AI Writing patterns plus universal good-writing rules (long-word → short-word, active-over-passive, comma splices, cliché flag, jargon-fragment-in-prose rewrite, corporate-speak nominalizations). Prose mode adds Craig's writing-voice patterns for prose he authors or sends — emails, documents, notes — on top of general (em-dash zero-tolerance, no-emphasis-formatting, contractions, semicolons → periods, sentence-split, felt-experience cut, sentence-fragment rewrite, terse-cut, single-sentence cadence, parenthetical asides, declarative-register marker). Personal mode is for publish artifacts only (commits, PR titles + bodies, PR review comments) and adds the artifact-mechanics patterns on top of prose (first-person rewrite, public-artifact scope flag, praise/correction asymmetry, finding stems) plus per-artifact terseness budgets. Total 45 patterns; one editorial review covers all relevant ones for the chosen mode, with an attestation receipt for the high-recurrence set. Replaces the standalone humanizer skill. Use when editing prose. Do NOT use for code, structured data, or plain bullet lists where fragments are valid. + Multi-pass prose editor with three modes. General mode (default) edits arbitrary writing — research notes, essays, anyone's README prose — by walking 31 patterns: Wikipedia's Signs of AI Writing patterns plus universal good-writing rules (long-word → short-word, active-over-passive, comma splices, cliché flag, jargon-fragment-in-prose rewrite, corporate-speak nominalizations). Prose mode adds Craig's writing-voice patterns for prose he authors or sends — emails, documents, notes — on top of general (em-dash zero-tolerance, no-emphasis-formatting, contractions, semicolons → periods, sentence-split, felt-experience cut, sentence-fragment rewrite, terse-cut, single-sentence cadence, parenthetical asides, declarative-register marker). Personal mode is for publish artifacts only (commits, PR titles + bodies, PR review comments) and adds the artifact-mechanics patterns on top of prose (first-person rewrite, public-artifact scope flag, praise/correction asymmetry, finding stems, comma budget) plus per-artifact terseness budgets. Prose mode also carries one correspondence-only pattern (recipient-priority ordering) that personal mode skips. Total 47 patterns; one editorial review covers all relevant ones for the chosen mode, with an attestation receipt for the high-recurrence set. Replaces the standalone humanizer skill. Use when editing prose. Do NOT use for code, structured data, or plain bullet lists where fragments are valid. allowed-tools: - Read - Write @@ -31,8 +31,8 @@ This skill is split across two files by design. Three modes determine which patterns to walk. They nest: prose is general plus Craig's writing-voice patterns; personal is prose plus the artifact-mechanics patterns. - **General** (default) — apply patterns **#1-31**. Use for writing whose author isn't Craig and that isn't a publish artifact: research notes you're editing for someone else, a quoted passage, README prose for a shared project, any third-party text. Output is well-edited human-sounding prose, but does not impose Craig's voice (first-person, contractions, em-dash elimination) — those conflict with academic, literary, or formal registers that aren't his. -- **Prose** — apply **#1-31** plus the patterns tagged **(prose + personal)**: em-dash zero-tolerance (#13), contractions (#34), semicolons → periods (#33), sentence-split (#35), felt-experience cut (#36), sentence-fragment rewrite (#37), terse-cut (#38), no-emphasis-formatting (#41), single-sentence cadence (#43), parenthetical asides (#44), and the declarative-register marker (#45). Use for prose Craig authors or sends in his own voice that isn't a publish artifact: emails, documents he writes or hands to someone, working notes, journal entries. This is the mode that finally applies his actual writing voice to the documents he most wants it on. It skips the artifact-mechanics patterns (#32, #39, #40, #42) — those assume a commit or PR and misfire on free prose (a document is legitimately third-person; a journal has no public-scope concern; praise/correction asymmetry and finding stems are PR-review rules). -- **Personal** — apply all **#1-45**. Use only for publish artifacts: commits, PR titles + bodies, and PR review comments. Adds the artifact-mechanics patterns (#32 first-person rewrite, #39 public-artifact scope flag, #40 praise/correction asymmetry, #42 finding stems) on top of everything prose mode walks. +- **Prose** — apply **#1-31** plus the patterns tagged **(prose + personal)**: em-dash zero-tolerance (#13), contractions (#34), semicolons → periods (#33), sentence-split (#35), felt-experience cut (#36), sentence-fragment rewrite (#37), terse-cut (#38), no-emphasis-formatting (#41), single-sentence cadence (#43), parenthetical asides (#44), and the declarative-register marker (#45) — plus **#47 (recipient-priority ordering)** when the piece is correspondence (email, Signal, a letter). Use for prose Craig authors or sends in his own voice that isn't a publish artifact: emails, documents he writes or hands to someone, working notes, journal entries. This is the mode that finally applies his actual writing voice to the documents he most wants it on. It skips the artifact-mechanics patterns (#32, #39, #40, #42) — those assume a commit or PR and misfire on free prose (a document is legitimately third-person; a journal has no public-scope concern; praise/correction asymmetry and finding stems are PR-review rules). +- **Personal** — apply **#1-46**. Use only for publish artifacts: commits, PR titles + bodies, and PR review comments. Adds the artifact-mechanics patterns (#32 first-person rewrite, #39 public-artifact scope flag, #40 praise/correction asymmetry, #42 finding stems, #46 comma budget) on top of everything prose mode walks. #47 is the lone exception to the nesting — it is correspondence-only, and a publish artifact is never a reply to someone's news, so personal mode stops at #46. If invoked without a mode argument, default to general. Prose mode is invoked explicitly with `/voice prose` (emails, authored documents). Personal-context callers (`commits.md` publish flow, `respond-to-cj-comments.md`) invoke `/voice personal`. @@ -53,7 +53,7 @@ Terse is a budget, not an adjective. Each publish-artifact type has a target sha When given text to edit: -1. **Identify which patterns apply** — Scan for the patterns numbered below. General mode walks #1-31 only. Prose mode adds the patterns tagged **(prose + personal)**. Personal mode adds those *and* the ones tagged **(personal only)** — i.e. all 45. +1. **Identify which patterns apply** — Scan for the patterns numbered below. General mode walks #1-31 only. Prose mode adds the patterns tagged **(prose + personal)**, plus #47 when the piece is correspondence. Personal mode adds the **(personal only)** ones instead — patterns #1-46, never #47. 2. **Rewrite problematic sections** — Replace each detected pattern with its rewrite. 3. **Preserve meaning** — Keep the core message intact. 4. **Maintain voice** — Match the intended tone (formal, casual, technical, academic, literary). @@ -296,11 +296,11 @@ See `voice/references/voice-profile.org` §31 for problem, basis, examples, and ## Craig's Voice (prose + personal modes) -These patterns carry Craig's writing voice. Most apply in **both** prose mode (emails, documents, notes he authors) and personal mode (commits, PRs, PR comments) — tagged **(prose + personal)**. Four are publish-artifact-specific — tagged **(personal only)** — because they assume a commit or PR and misfire on free prose: #32 (first-person rewrite) wrongly imposes "I did X" voice on a document that's legitimately third-person, #39 (public-artifact scope flag) has nothing to guard in a private journal, and #40 (praise/correction asymmetry) and #42 (finding stems) are PR-review rules. General mode skips all of them — it edits text that isn't Craig's, where contractions, em-dash elimination, and first-person would conflict with academic, literary, or formal registers. +These patterns carry Craig's writing voice. Most apply in **both** prose mode (emails, documents, notes he authors) and personal mode (commits, PRs, PR comments) — tagged **(prose + personal)**. Five are publish-artifact-specific — tagged **(personal only)** — because they assume a commit or PR and misfire on free prose: #32 (first-person rewrite) wrongly imposes "I did X" voice on a document that's legitimately third-person, #39 (public-artifact scope flag) has nothing to guard in a private journal, #40 (praise/correction asymmetry) and #42 (finding stems) are PR-review rules, and #46 (comma budget) is scoped to publish artifacts by Craig's 2026-07-20 directive. One — #47 (recipient-priority ordering) — runs the other way: prose-only and narrower still, firing solely on correspondence, because ordering a reply around the recipient's news has no meaning for a commit or a document addressed to nobody. General mode skips all of them — it edits text that isn't Craig's, where contractions, em-dash elimination, and first-person would conflict with academic, literary, or formal registers. ### 32. First-Person Voice Rewrite [personal] -**Rule.** Rewrite impersonal third-person publish-artifact bodies into first person ("I added X", "I kept Y because..."). The commit subject line stays imperative per Conventional Commits. Skip for mechanical changes where the subject alone carries the message. +**Rule.** Rewrite impersonal third-person publish-artifact bodies into first person ("I added X", "I kept Y because..."). **The "I" is Craig**, who is the author of record and whose name the artifact goes out under — not an agent narrating work done on his behalf. Cut any construction that writes an agent in as a separate party ("Craig asked me to", "I filed this for Craig", "needs Craig's decision"); an open decision is his own, written as "I haven't decided whether…". The commit subject line stays imperative per Conventional Commits. Skip for mechanical changes where the subject alone carries the message. See `voice/references/voice-profile.org` §32 for problem, basis, examples, and history. @@ -366,7 +366,7 @@ See `voice/references/voice-profile.org` §42 for problem, basis, examples, and ### 43. Single-Sentence Paragraph Cadence Is a Feature [prose · personal] -**Rule.** A one-sentence paragraph is a finished thought, not a fragment. Break paragraphs after one complete thought when the next thought shifts angle, even if both are short. Never merge short paragraphs into multi-sentence ones in a "clean prose" pass (corpus: 41-74% of Craig's paragraphs are exactly one sentence, depending on register). +**Rule.** A one-sentence paragraph is a finished thought, not a fragment. "Shifts angle" means shifts *topic*: break paragraphs at a topic boundary, even when both sides are one sentence. Within a single topic, do the opposite — consolidate its sentences into one paragraph even when each is a complete thought, up to a ceiling of about five or six sentences, past which find a natural break. The never-merge instruction protects the break *between* topics, which is where Craig's cadence lives (corpus: 41-74% of his paragraphs are exactly one sentence, depending on register); it never licenses fragmenting one topic across several paragraphs. See #47 for the ordering of those topics in a reply. See `voice/references/voice-profile.org` §43 for problem, basis, examples, and history. @@ -382,10 +382,22 @@ See `voice/references/voice-profile.org` §44 for problem, basis, examples, and See `voice/references/voice-profile.org` §45 for problem, basis, examples, and history. +### 46. Comma Budget — Max Two Per Sentence [personal] + +**Rule.** No sentence carries more than two commas. Rewrite the third comma away: split the sentence, move a clause into a parenthetical (#44) or behind a colon, or break an inline serial list into bullets or its own sentence. Count prose commas only — commas inside code spans, quoted log lines, and literal strings don't count toward the budget. + +See `voice/references/voice-profile.org` §46 for problem, basis, examples, and history. + +### 47. Recipient-Priority Ordering [prose — correspondence only] + +**Rule.** In a reply, lead with what matters most to the *recipient*, not with what's easiest to answer or the order they wrote it. Their news outranks your logistics, and a direct question they asked can sort *below* personal news they shared, because the news is what they care about. Leading with the easy answer reads as transactional. Correspondence-scoped: it needs a recipient, so it fires on email, Signal, and letters, and has no referent in a journal, a working note, or any document addressed to nobody. This is the one prose-mode pattern that does not carry into personal mode — a commit or PR review is not a reply to someone's news. + +See `voice/references/voice-profile.org` §47 for problem, basis, examples, and history. + ## Process 1. Read the input text carefully. Confirm the mode (general, prose, or personal) — invocation argument or context. If a file path was given, that file is the deliverable: the final text gets written back to it in step 7. -2. Walk patterns 1-31 in general mode; add the (prose + personal) patterns in prose mode; walk all 45 patterns in personal mode. +2. Walk patterns 1-31 in general mode; add the (prose + personal) patterns in prose mode, plus #47 when the piece is correspondence; walk patterns 1-46 in personal mode. 3. For each pattern, scan the text. If a match is found, rewrite it according to the pattern's rule. Patterns #39 and #45 emit flags without rewriting. 4. After walking all patterns, ensure the revised text: - Sounds natural when read aloud @@ -395,7 +407,7 @@ See `voice/references/voice-profile.org` §45 for problem, basis, examples, and 5. **Terse pass — mandatory, last rewrite pass (prose + personal modes).** Walk pattern #38 again as a standalone action: read each sentence and try to cut it in half, keeping only the words that change meaning. Run it on its own here, not folded into step 3's walk — it is the most-skipped pattern and the bloat it catches is the first thing a reader notices. General mode skips this step — academic and third-party registers keep their transition markers. 6. **Anti-AI audit — on the final text.** Prompt: "What makes the below so obviously AI generated?" Answer briefly with remaining tells, then revise. This runs *after* the terse pass so the audited text is the text that ships. If the audit triggers rewrites, re-apply the #38 per-sentence test to every changed sentence before proceeding. 7. **Write-back.** If the invocation supplied a file path, write the final text to that file now and say so. The publish flow posts from the file (`git commit -F`, `gh pr create --body-file`), so a final text that lives only in chat is a drift bug waiting to post the un-voiced version. -8. **Attestation block (prose + personal modes).** The high-recurrence patterns — the ones with a documented failure history — each get one explicit line: pattern, checked, match or no match, action taken. Current high-recurrence set: **#13 (em-dash), #37 (fragments), #38 (terse), #40 (praise asymmetry), #42 (finding stems)**. This is a receipt, not a summary: a pattern with no match still gets its line. When a pattern in this set fails in the wild despite the receipt, escalate it the way #38 was escalated; when one holds clean for a long stretch, it can rotate out. +8. **Attestation block (prose + personal modes).** The high-recurrence patterns — the ones with a documented failure history — each get one explicit line: pattern, checked, match or no match, action taken. Current high-recurrence set: **#13 (em-dash), #37 (fragments), #38 (terse), #40 (praise asymmetry), #42 (finding stems), #46 (comma budget)**. This is a receipt, not a summary: a pattern with no match still gets its line. When a pattern in this set fails in the wild despite the receipt, escalate it the way #38 was escalated; when one holds clean for a long stretch, it can rotate out. 9. Present the final version per the Output Format below. ## Output Format @@ -480,7 +492,9 @@ This skill draws from: - Orwell, *Politics and the English Language* — patterns #26 (short over long), #27 (active over passive), #29 (cliché). - Plain English Campaign — pattern #26 (Plain English wordlist). - Garner, *Modern English Usage* — pattern #26 (word-pair preferences). -- Craig's voice rules from `claude-rules/commits.md` (Voice and Focus section) — patterns #32-42, split across prose mode (his authored prose and email) and personal mode (publish artifacts). +- Craig's voice rules from the `publish` skill (Voice and Focus section) — patterns #32-42, split across prose mode (his authored prose and email) and personal mode (publish artifacts). +- Craig's directive, 2026-07-20 (archsetup session) — pattern #46 (comma budget), personal mode. +- Craig's edit of a Signal reply, 2026-07-23 (home session) — pattern #47 (recipient-priority ordering) and the #43 topic-vs-angle calibration, prose/correspondence. - Corpus measurement (2026-05-29 phases 1-2, documented in the profile) — patterns #43-45 and the calibration notes on #7, #13, #33. Key insight (Wikipedia, paraphrased): LLMs use statistical algorithms to guess what should come next. The result tends toward the most statistically likely text that applies to the widest variety of cases. Patterns #1-25 detect that signature. diff --git a/voice/references/voice-profile.org b/voice/references/voice-profile.org index 765d513..3dfddf2 100644 --- a/voice/references/voice-profile.org +++ b/voice/references/voice-profile.org @@ -1444,11 +1444,13 @@ In a PR review finding: a sentence carrying more than one claim (chained through Prose and personal modes. General mode skips because third-party registers legitimately prefer multi-sentence paragraphs. *** Rule -A one-sentence paragraph is a finished thought, not a fragment. Break paragraphs after one complete thought when the next thought shifts angle, even if both are short. Never merge short paragraphs into multi-sentence ones in a "clean prose" pass. +A one-sentence paragraph is a finished thought, not a fragment. "Shifts angle" means shifts *topic*: break paragraphs at a topic boundary, even when both sides are one sentence. Within a single topic, consolidate its sentences into one paragraph even when each is complete, up to a ceiling of about five or six sentences, past which find a natural break. The never-merge instruction protects the break *between* topics; it never licenses fragmenting one topic across several paragraphs. *** Problem Most prose-style guides advise multi-sentence paragraphs, so a generic cleanup pass merges Craig's short paragraphs and erases a distinctive feature of his voice. This is a protective pattern: it guards an existing trait rather than correcting a defect. +The 2026-07-23 boundary refinement addresses the opposite failure, discovered the same day: reading "angle" at *sentence* granularity, so that three sentences all about one topic (a baking run, a mixer, tortillas) got split into three paragraphs as if each were a new angle. That fragments one topic and reads as a checklist rather than a person talking. Angle means topic. #43 governs the break between topics; consolidation fills in what happens within one, which the original rule never specified. The two are one rule seen from both sides, not a rule in tension with #47. + *** Basis Corpus-measured (2026-05-29). Single-sentence-paragraph rate: git commits 41.1%, personal email 57.4%, work email 44.5%, PR descriptions 74.4%, PR review comments 50.0%. Between 41% and 74% of Craig's paragraphs are exactly one sentence, depending on register. @@ -1470,6 +1472,7 @@ An edit pass that merged short paragraphs, or a draft whose paragraphs each stac *** History - 2026-05-29: surfaced by the Phase 1-2 corpus measurement as a "worth adding" trait; filed as suggested delta 4. - 2026-06-10: promoted from the suggested-deltas list into a numbered pattern. Craig's call, from the work-project session. +- 2026-07-23: boundary refined (angle means topic; within-topic consolidation to a ~5-6 sentence ceiling; never-merge reframed as across-topic protection). From a home session drafting a Signal reply, where the original rule was misread at sentence granularity. Paired with the new §47. ** §44 Parenthetical Asides Are Part of the Voice @@ -1532,3 +1535,79 @@ A question mark in a draft in Craig's voice. Flag it; keep genuine questions to *** History - 2026-05-29: surfaced by the Phase 1-2 corpus measurement as a register marker; filed as suggested delta 6. - 2026-06-10: promoted from the suggested-deltas list into a numbered advisory pattern. Craig's call, from the work-project session. + +** §46 Comma Budget — Max Two Per Sentence + +*** Modes +Personal mode only. Prose and general modes skip. + +*** Rule +No sentence carries more than two commas. Rewrite the third comma away: split the sentence, move a clause into a parenthetical (§44) or behind a colon, or break an inline serial list into bullets or its own sentence. Count prose commas only — commas inside code spans, quoted log lines, and literal strings (version numbers, paths) don't count toward the budget. + +*** Problem +Three or more commas in one sentence almost always mark stacked clauses or an inline list doing a paragraph's work. The sentence reads fine to its author and lands as a pileup on the reader. The comma count is a mechanical proxy the walk can enforce, where "don't stack clauses" is prose advice that gets skipped. + +*** Basis +Craig's directive, 2026-07-20 (archsetup session, while gating a Hyprland issue draft): "no more than two commas per sentence. we should add that to the /voice personal pass." + +*** Before (spec-sheet line with three commas, from the draft that prompted the rule) +#+begin_example +System: Arch Linux, kernel 6.18.25-lts, AMD Strix Halo (Radeon 8060S), no plugins loaded. +#+end_example + +*** After +#+begin_example +System: Arch Linux, kernel 6.18.25-lts. GPU: AMD Strix Halo (Radeon 8060S). No plugins loaded. +#+end_example + +*** Detection +Count commas per sentence on the final text. A sentence at three or more gets restructured, not trimmed to exactly the budget — the third comma is the symptom, the stacked structure is the target. + +*** History +- 2026-07-20: added at Craig's direction from the archsetup session. Scoped to personal mode; broaden to prose only if he asks. Added to the attestation high-recurrence set at birth — a mechanical count is cheap to receipt, and new discipline fails silently without one. + +** §47 Recipient-Priority Ordering + +*** Modes +Prose mode, and only when the piece is correspondence (email, Signal, a letter). It needs a recipient, so it has no referent in a journal, a working note, or any document addressed to nobody, and it does not carry into personal mode — a commit or PR review is not a reply to someone's news. General mode skips it with the rest of Craig's voice patterns. This is the one pattern narrower than a whole mode, and the only prose pattern personal mode does not also walk. + +*** Rule +In a reply, lead with what matters most to the recipient, not with what's easiest to answer or the order they wrote it. Their news outranks your logistics. A direct question they asked can sort below personal news they shared, because the news is what they care about. + +*** Problem +The easy draft answers the explicit question first and orders the rest as it arrived. That reads as transactional — logistics before the person. Ordering by what the recipient cares about is what makes a reply read as one person talking to another rather than a ticket being closed. Nothing else in the skill governs the *order* of a reply's contents; the other patterns act within a paragraph or a sentence. + +*** Basis +Craig's edit of a Signal reply to his sister, 2026-07-23 (home session). His framing: "start with what would be the most important things to her." + +*** Before (first draft — opens with the only explicit question, cooking split across three paragraphs) +#+begin_example +Yes, I do subscribe to MasterClass — happy to share what I've watched. + +That's amazing about the sourdough. English muffins from scratch is no joke. + +The home-roasted deli meat sounds incredible. + +A stand mixer would make the bread a lot easier — worth it if you're baking this much. + +And 30 pounds — that's huge. So happy for you. +#+end_example + +*** After (Craig's order — weight first, one cooking paragraph, then the question, then the close) +#+begin_example +Thirty-plus pounds — that is huge, and I'm so happy for you. That's real work. + +And the cooking. Sourdough, English muffins, tortillas, home-roasted deli meat from scratch — that's a whole kitchen you've built, and a stand mixer would make the bread much easier if you're baking at this volume, so I say go for it. I want to hear how the tortillas come out. + +Yes, I subscribe to MasterClass — I'll send you what I've been watching. + +I miss you and I love you. Send me a few times that work for a call. +#+end_example + +The cooking paragraph runs seven sentences, a hair over the §43 ceiling. Craig called it an exception rather than re-cut a message that had already gone out. The guard is the rule; this paragraph is one sentence over it; both facts stay in the record, because a real example at the boundary teaches it better than a clean one. + +*** Detection +A reply whose opening answers a logistical or yes/no question while the recipient's substantive news sits lower. Reorder so the news they'd most want acknowledged leads. + +*** History +- 2026-07-23: added from the home session drafting a Signal reply. The first handoff proposed two new patterns and flagged a conflict with §43; the superseding design resolved that the conflict was a misreading of §43 (angle = topic), leaving one genuinely new pattern here and a calibration to §43. Prose/correspondence-scoped per Craig — email and Signal are prose, not publish artifacts. diff --git a/working/context-engineering-rightsizing/metrics.org b/working/context-engineering-rightsizing/metrics.org new file mode 100644 index 0000000..e7ff189 --- /dev/null +++ b/working/context-engineering-rightsizing/metrics.org @@ -0,0 +1,269 @@ +#+TITLE: Context-Engineering Rightsizing — Metrics and Stop Conditions +#+AUTHOR: Craig Jennings +#+DATE: 2026-07-27 + +* Who is holding the instrument + +Most of what follows would be measured by me, about changes to my own +instructions, in a direction I have an obvious interest in. Self-reported +compliance is unreliable, and it is unreliable in a predictable direction: I +will under-report misses I didn't notice, because not noticing is the failure. + +That is the argument for weighting mechanical detectors over my judgment +wherever both exist. =lint-org= does not have a stake. Neither does a token +count, a =git diff=, or you seeing a window on the wrong workspace. Where a +claim can only be assessed by my self-report, it is marked below as judgment +rather than measurement, and it should be discounted accordingly. + +* Part 1 — Which claims are testable + +Each post makes separable claims. Some can be tested here cheaply, some need an +eval harness we don't have, and inventing a metric for the second group would be +worse than admitting it. + +** Testable now, with existing instrumentation + +*** T1 — Report-everything beats pre-filtering (Opus 5 guide) + +*Claim.* A review told to be conservative reports less; reporting everything and +filtering separately surfaces more real issues. + +*Test.* A true A/B, runnable today with no waiting. Take three past diffs with +known outcomes. Run =/review-code= under current rules and under the +report-all-then-filter version. Compare findings that survive verification. + +*Metric.* Real findings per pass, and false-positive rate. The claim holds if +report-all finds more real issues without the false-positive rate rising +enough to drown them. + +*Why this one first.* It is the only claim in the three posts we can settle in +an afternoon instead of a week. + +*** T2 — Progressive disclosure preserves behavior (context-engineering post) + +*Claim.* Moving guidance out of always-loaded context into on-demand loading +doesn't degrade adherence. + +*ANSWERED 2026-07-27 for the deterministic half.* =/context= in a live work +session lists 17 generic rules under Memory files, with the three path-scoped +ones absent. Path-scoping is a glob match rather than a model judgment, so it +either fires or doesn't, and it fires. That half needs no trial and no miss-rate +metric. + +*Still open for the semantic half.* Rules whose condition can't be a glob +("when a commit is in play") route to skills, where triggering *is* a model +judgment. Everything in Part 2 applies there and only there. + +*A caution the answer doesn't cover.* Path-scoping fires when Claude *reads* a +matching file. A session that writes an org file without reading one first +never triggers =todo-format.md=. Edit requires a prior read, so edits to +existing files are safe; creating a new org file from scratch is the gap. Worth +watching rather than blocking on. + +*** T3a — The token baseline was wrong by 45% [SETTLED] + +Word counts converted at a guessed ~1.3 tokens per word. The live number is +2.28. =commits.md= is 12,800 tokens, not the ~7,000 estimated, and +=claude-rules/= was ~57,800 tokens per session before today rather than the +~33,000 implied. + +The lesson is narrower than "measure better." I had a real measurement +available the whole time — =/context= reports per-file token counts — and used +an estimate instead because the estimate was easier to compute from inside the +repo. Reach for the instrument that reports the actual quantity. + +*** T3 — Deliverables run long without explicit calibration (Opus 5 guide) + +*Claim.* Files written to disk are longer than the task needs. + +*Test.* Measure what already exists. Word counts of the last twenty session +archives, the specs in =docs/specs/=, and =todo.org= task bodies. Then add a +length-calibration instruction and measure the next ten. + +*Metric.* Median words per artifact, before and after. Paired with a judgment +call on whether anything useful was lost, which is the part I can't measure. + +*Baseline worth taking now,* since it costs one command and the before-number +disappears the moment we change anything. + +*** T4 — Lower effort holds quality [WITHDRAWN 2026-07-27] + +Dropped with P4. The goal is output quality first, and this claim trades +quality for cost, so testing it would answer a question we've decided not to +act on either way. + +*** T4 (original text, retained for the record) + +*Claim.* =low= and =medium= produce strong quality at a fraction of the tokens. + +*Test.* Sentry fires hourly and does the same passes each time. Run a week at +default, a week at =medium=, on the same repo state where possible. + +*Metric.* Tokens per fire, and findings per fire. The claim holds if findings +per fire holds within noise while token cost drops materially. + +*Confound to respect.* Sentry's input changes night to night, so findings-per- +fire is noisy. Two weeks is probably the floor for a readable signal, and the +result will still be suggestive rather than conclusive. + +*** T5 — Duplicate and conflicting instructions cost something (both posts) + +*Claim.* Overlapping guidance across surfaces makes the model work harder to +reconcile. + +*Test.* Partially measurable. The duplication itself is countable — Phase 5's +three axes. Whether removing it improves anything is not measurable without +evals. + +*Metric.* Count of rules stated on more than one surface, driven to zero. That +measures the cleanup, not the benefit. Honest framing: we're removing a known +cost, not demonstrating a gain. + +** Not testable here — judgment calls, and they should be labelled as such + +*** J1 — Examples constrain the exploration space + +The post asserts this and I have no way to test it. Removing the examples from +=commits.md= and observing "things seem fine" is not evidence. This one gets +adopted on the post's authority or not at all, and I'd lean toward *not* +stripping examples that encode your taste, since the same post says skills +should encode exactly that. + +*** J2 — Removing verification instructions loses no quality + +The claim underneath C1. Testing it properly means running the same tasks with +and without =verification.md= and comparing error escape rates, which needs a +task suite we don't have. What we *can* measure is the cost side — tool calls +and tokens spent on verification per task — but not what it prevents. + +Asymmetric risk: the cost of over-verification is tokens, and the cost of +under-verification is a false completion claim reaching you. Those are not +equally bad, which argues for keeping the honesty core regardless of what a +test would show. + +*** J3 — Unknowns-discovery reduces rework + +Long-horizon and confounded by everything else. Adopt on judgment. + +*** J4 — Rich references beat prose descriptions + +=ui-prototyping.md= already assumes it and it has worked. That's one project's +experience, not a measurement, but it's the evidence we have. + +* Part 2 — Pilot go/no-go + +** The denominator problem + +The obvious metric is a miss rate, and the obvious trap is that a rule nobody +exercised shows zero misses and reads as a pass. Every result below is +therefore reported as *misses per exposure*, and a rule with too few exposures +returns no verdict rather than a passing one. + +*Exposure* means a session did work in the rule's domain: edited an org table, +touched a spec's lifecycle, displayed a keymap, captured a window, ran a UI +spec. + +*Minimum exposures before a rule's result counts: 3.* Below that, extend the +window or swap the rule for one that gets exercised more. + +** Primary metric + +Misses per exposure, per rule, where a miss is the guidance not being applied +when it should have been. + +| Result per rule | Verdict | +|----------------------------+----------------------------------------| +| 0 misses in 3+ exposures | Pass | +|----------------------------+----------------------------------------| +| 1 miss, detector caught it | Pass with note; record the miss | +|----------------------------+----------------------------------------| +| 2 or more misses | Turn back that rule | +|----------------------------+----------------------------------------| +| Miss reached a commit | Turn back now, don't wait for the week | +|----------------------------+----------------------------------------| +| Miss reached a project | Turn back now, and see abandon triggers| +|----------------------------+----------------------------------------| +| Under 3 exposures | No verdict; extend or swap the rule | +|----------------------------+----------------------------------------| + +** Secondary metrics + +- *Token delta.* Measured directly. Expected around 3,000 words. This is the + only guaranteed benefit, so if the primary metric fails, we know exactly what + we were buying and can decide it wasn't worth it. +- *Correction cost.* When a miss happens, how long to fix. A misformatted table + is seconds. This is what distinguishes an acceptable miss rate from an + unacceptable one, and it is why the threshold tightens in Phase 4. +- *False-trigger rate.* A skill loading when it isn't needed spends tokens for + nothing. Worth watching, not worth blocking on. + +** Turn back versus abandon + +Two different actions, and conflating them would over-react to a single bad +rule. + +*Turn back a rule* — revert that one file to always-loaded. Triggered by 2+ +misses on that rule, or any miss reaching a commit. The rest of the pilot +continues. Cost: one revert. + +*Abandon the rollout* — stop at Phase 0 and keep the current architecture. +Triggered by any of: + +- Three or more of the six rules turn back. +- A miss reaches another project through the sync. +- The pattern of misses shows the skill index isn't the fix, meaning D2 was the + wrong lever and there's no obvious next one. + +*Proceed to Phase 3* requires: at least four of six rules pass, no miss reached +a commit, and the token drop landed near expectation. + +** Threshold scales with blast radius + +The pilot tolerates one caught miss per rule because the worst case is a badly +formatted table. Phase 4 moves =commits.md=, where the worst case is an +unattributed commit or a leaked path in a public artifact. + +*Phase 4 threshold is zero.* One miss on an attribution, scope, or grading rule +turns that rule back immediately, with no pass-with-note tier. Stated now so +it isn't negotiated later under the pressure of wanting the phase to succeed. + +* Part 3 — Baselines to capture before anything changes + +Cheap now, impossible to reconstruct later: + +1. Always-loaded word count per surface. Captured: 32,123 total. +2. Median word count of the last twenty session archives, the specs, and open + task bodies. For T3. +3. Sentry tokens per fire and findings per fire, from the existing metrics + file. For T4. +4. Findings per pass from the last several =/review-code= runs. For T1. + +Item 1 is done. The rest are one session's work and should happen before +Phase 0 changes the review skill, since that change contaminates item 4. + +* Instrument reliability — the day's second lesson + +The plan weights mechanical detectors over self-report because they have no +stake. Two failed on 2026-07-27, both reporting success while doing damage: +=wrap-org-table.el= reflowed a table into a worse shape and =lint-org= then +certified it clean, and the wrap-teardown hook consumed a two-hour-old sentinel +and killed a live work session that had done nothing wrong. + +Neither invalidates the preference for mechanical detectors, which is still +right. Both narrow the claim: a detector has no stake, but it can be +confidently wrong, and a green check from a guard that never looked at the +thing is indistinguishable from a green check that did. When a detector clears +a moved rule, the useful question is whether it actually evaluated it, not just +whether it reported clean. + +* What this can and cannot tell us + +It can tell us whether on-demand loading fires reliably here, whether +report-everything finds more real bugs, whether artifacts shrink usefully, and +what lower effort costs in findings. + +It cannot tell us whether examples constrain, whether removing verification +instructions is safe, or whether unknowns-discovery reduces rework. Those stay +judgment calls made on the posts' authority and your read, and they should be +labelled that way in whatever we write down afterward — so that a future session +doesn't mistake an adopted opinion for a tested result. diff --git a/working/context-engineering-rightsizing/proposals.org b/working/context-engineering-rightsizing/proposals.org new file mode 100644 index 0000000..1300a7b --- /dev/null +++ b/working/context-engineering-rightsizing/proposals.org @@ -0,0 +1,350 @@ +#+TITLE: Context-Engineering Rightsizing — Proposals +#+AUTHOR: Craig Jennings +#+DATE: 2026-07-27 + +* Status + +Proposals only. Nothing here is applied. Source: three Anthropic posts Craig +supplied 2026-07-27 — the Claude 5 context-engineering post (2026-07-24), the +Opus 5 prompting guide, and the Fable 5 field guide (2026-07-06). + +* Goal + +Output quality and results first. Token reduction is a real goal and worth +having, but it is the second one, and where the two conflict quality wins. +Craig's framing, 2026-07-27: the concern is agents having the freedom to +produce at their highest capacity, unbound by guardrails that constrain them or +work against them. + +That reordering matters more than it sounds. Anthropic's 80% figure was a +*finding* — they cut and quality held — not a target. Aimed at directly it +optimizes the thing we don't care about, and P4 below is the proposal that +falls to it. + +* The measurement (corrected 2026-07-27 from live =/context=) + +The original figures in this document were word counts converted at a guessed +ratio, and they understated by about 45%. A =/context= run in a live work +session gave the real numbers: + +| Surface | Tokens | +|------------------------------------+--------| +| =claude-rules/= before this session | 57,800 | +|------------------------------------+--------| +| =claude-rules/= now (17 files) | 44,410 | +|------------------------------------+--------| +| Path-scoped out (3 files) | 13,390 | +|------------------------------------+--------| + +The real ratio is 2.28 tokens per word, not the ~1.3 assumed. =commits.md= +alone is 12,800 tokens, not the ~7,000 estimated. Every earlier figure here +should be read as low. + +** Two loading paths, not one + +The original framing called 32,123 words "always-loaded" and was wrong to +lump them. They arrive by different mechanisms: + +- *Memory files* — =claude-rules/= (via the =~/.claude/rules/= symlinks), + =CLAUDE.md=, project =.claude/rules/=, and auto-memory's =MEMORY.md=. Loaded + at session start by the harness. This is the 44,410 above, plus CLAUDE.md and + project rules. +- *Read during startup* — =protocols.org= and the workflow files. These never + appear under Memory files in =/context=; the startup workflow reads them, so + they land in Messages. Still a real per-session cost, but a different lever: + they shrink by editing the workflow, not by scoping a rule. + +Conflating the two made =protocols.org= look like it competed with +=commits.md= for the same fix. It doesn't. + +** What the harness says on its own + +=/context= ends with its own suggestion: prune =commits.md= (12.8k), +=testing.md= (6.3k), and =MEMORY.md= (5.5k). That is the Phase 4 target list, +arrived at independently. Worth treating as corroboration rather than +coincidence. + +* Proposals + +Ranked by value. Each carries my confidence and what I think the real risk is. + +** P1 — Progressive disclosure for =claude-rules/= [high value, highest risk] + +*Change.* Split the twenty rule files into two tiers. + +- *Always-loaded core* (target under 3,000 words): the genuine invariants that + must fire without being summoned. The no-AI-attribution rule, the + cross-project boundary stop, the no-popup-menus and no-reverse-video output + constraints, the =date=-before-timestamps rule, and a short index naming + which skill covers what. +- *On-demand tier*: everything procedural, converted to skills whose + descriptions trigger them. =commits.md= becomes a publish skill that loads + when a commit or PR is in play. =todo-format.md= loads when an org todo file + is touched. =testing.md=, =working-files.md=, =docs-lifecycle.md=, + =org-tables.md=, =keybinding-display.md= likewise. + +*Why.* This is the post's central move, and the token math is the argument. + +*The real risk, stated plainly.* A rule that isn't loaded can't fire. Skill +triggering is probabilistic in a way that always-resident text isn't. The +failure mode is silent: a session commits without the voice pass because the +publish skill didn't trigger, and nothing announces the miss. That's the same +silent-failure shape as the two probe defects fixed this morning. + +*Mitigation.* Anything whose violation is expensive and hard to reverse stays +in the always-loaded core, whatever its length. The split is by *blast radius*, +not by word count. And the migration goes one file at a time with a live trial, +not as a single cutover. + +*Confidence.* High that the direction is right. Medium on where exactly each +line falls — that's a judgment call per rule, and worth walking together. + +** P2 — Stop pre-filtering review findings [high value, low risk] + +*Change.* =review-code/SKILL.md:251= says "Drop Low-confidence issues before +the final report." Line 434 repeats it. Replace with: report every finding +carrying an explicit confidence label, then filter in a named second pass. + +*Why.* The Opus 5 guide is specific about this: "If your review prompt says +'only report high-severity issues' or 'be conservative,' the model may follow +that instruction literally and report less; ask it to report everything and +filter in a separate pass instead." The guide also reports that on this model +the extra findings are mostly real rather than false positives, which is the +premise the drop-rule was written against. + +*Confidence.* High. This is the most directly-actionable finding in the three +posts, and it names the exact pattern the skill implements. + +** P3 — Add the unknowns-discovery practices [medium value, low risk] + +The field guide describes eight practices. The sweep found these absent: +=unknown unknowns= framing (0 files), =implementation notes= (0), =quiz= (0). +Present but thin: =blind spot= (1 file), =interview= (2), =pitch= (1). +=brainstorm= appears in 5 and =references= is well covered by +=ui-prototyping.md=, which already runs ahead of the post. + +*Change.* Add a blind-spot-pass practice and an interview practice, and an +implementation-notes convention for long builds (a scratch file logging +deviations from plan, which then feeds the retrospective). Add the quiz pattern +to the review or wrap surface. + +*Placement matters.* These go in the on-demand tier from P1, never the +always-loaded core — otherwise this proposal fights the one above it. + +*Confidence.* Medium-high on the practices being useful. Lower on the quiz, +which may not fit how you actually work. + +** P4 — Effort calibration [DROPPED 2026-07-27 — trades quality for cost] + +*Change.* Sentry, work-the-backlog, and the no-approvals speedrun run many +passes at default effort. The Opus 5 guide says to use =low= and =medium= +liberally as the primary cost control wherever quality holds, stepping up only +for demanding work. + +*Why.* Sentry fires hourly. Effort is the lever with the largest cost +multiple, and most sentry passes are mechanical sweeps. + +*Dropped.* Lowering effort buys tokens by spending quality, which is exactly +backwards under the goal above. Revisit only if a specific pass proves +genuinely mechanical and a cost problem shows up on its own. + +** P5 — Positive framing over prohibition [PROMOTED — now the top quality lever] + +*Change.* =commits.md= carries 41 prohibition markers (NEVER / DO NOT / +MANDATORY / CRITICAL) across 5,561 words. =todo-format.md= 21, +=testing.md= 19. Rewrite the ones that aren't hard invariants as positive +descriptions of the wanted behavior. + +*Why.* The Opus 5 guide: "Positive examples of the communication style you want +tend to be more effective than instructions about what not to do." + +*Keep as prohibitions:* the AI-attribution rule and anything else where the +worst case is genuinely unacceptable. Those earn their emphasis. + +*Promoted 2026-07-27.* Ranked low when the score was token savings. Under a +quality-first goal this is the proposal that most directly targets guardrails +working against good output, which is the stated concern. It still rides along +with the Phase 4 edits rather than running as its own campaign, because every +file it touches is a file those phases open anyway. + +** P6 — Deduplicate the two always-loaded surfaces [medium value, low risk] + +=protocols.org= restates rules that also live in =claude-rules/=: the +cross-project boundary, the working-files convention, the AI-attribution ban, +inbox cadence. Both are loaded every session, so each duplicated rule is paid +for twice, and the two copies can drift apart. + +*Change.* One home per rule. =protocols.org= keeps the pointer, the rule file +keeps the content — or the reverse, but not both. + +*Confidence.* High on the duplication being real, medium on which surface +should own each rule. + +* Conflicts — your call, not mine to inherit + +Two places where a post contradicts something this system arrived at +deliberately. I'm flagging rather than adopting. + +** C1 — =verification.md= versus the over-verification warning + +The Opus 5 guide says: "If your prompt contains explicit verification +instructions ('include a final verification step for any non-trivial task', +'use a subagent to verify'), remove them: instructions like these cause +over-verification on Claude Opus 5, and removing them reduces wasted tokens +with no loss in quality." + +=verification.md= is 1,486 always-loaded words of exactly that shape. But the +two aren't the same thing, and the distinction decides the answer: + +- Its *honesty core* — don't claim tests pass without running them, "unable to + verify" is a required outcome, replace beliefs with evidence — is about + truthful reporting, not about adding verification steps. The guide doesn't + argue against it. +- Its *process injection* — green baseline before starting, full suite as its + own step before every commit — is the shape the guide names. + +*My read:* keep the honesty core, shorten it, and let the process injection +move into the publish skill where it fires only when publishing. *My +confidence: medium*, and this is the one I'd most want you to overrule if it +feels wrong. It's also the rule closest to your standing "never guess, always +check" direction, so the guide's advice and your stated preference genuinely +pull against each other here. + +** C2 — =subagents.md= versus the delegation warning + +The guide says "do not use subagents to verify or double-check your own work." +=subagents.md= has a review-gate cadence and a rule to dispatch a *fix* +subagent rather than repairing in the orchestrator's context. + +*My read:* the fix-subagent rule is about context pollution, not verification, +so it survives. The review-gate cadence is closer to the flagged pattern and +wants a look. Most of =subagents.md= already matches the guide's advice — it +argues against spawning for small work and against letting the agent pick its +own scope, which is what the guide asks for. *Confidence: medium-high.* + +* One thing the posts would change about this document + +Both the context-engineering post and the field guide argue that a rich +reference beats a prose description — an HTML artifact, a test suite, source +code. This proposal document is prose in org, which is your reading format and +the right call for a decision doc. Worth noting the tension rather than +silently ignoring it: for the *next* artifact in this line of work, an HTML +comparison of the before and after rule tree would likely beat another org +file. + +* Scope proposal for the consistency sweep + +The audit behind this document was targeted, not exhaustive — I checked the +claims the posts made and measured the surface. A real inconsistency sweep over +20 rule files, 47 workflows, and the skills is its own pass. + +Proposed scope, in order: + +1. *Contradictions between always-loaded surfaces* — the P6 duplication set, + read side by side for drift rather than just counted. +2. *Stale facts* — assertions about tools, paths, and behavior that were true + when written. The spot check found the =agent-page= to =agent-text= rename + correctly handled, so this may be in better shape than expected. +3. *Instructions that contradict each other across files* — the failure the + context-engineering post opens with. This is the expensive one and the most + valuable. + +Sizing: item 1 is an afternoon, item 2 is mechanical, item 3 is the real work. + +* From your side of the desk + +Everything above treats these files as the agent's context, to be rightsized. +That was the smaller question. Read as *your prompts* — the map you hand every +project — the finding is different, and it's the one worth acting on. + +** The bottleneck this system was built for has moved + +Counting the workflows by what they're for: 41 are execution and hygiene +(publish, task grading, inbox routing, session archiving, calendar, email, +sync), 6 are discovery and design (the spec trio, retrospective, code-quality, +readability-audit). Roughly seven to one. + +That ratio was correct when the risk was the model doing things wrong. The +field guide's claim is that the risk moved: "Claude Fable is the first model +where I find the quality of the work is bottlenecked by my ability to clarify +its unknowns." If that's true here too, the system is heavily invested in the +half of the problem that got easier and thin on the half that didn't. + +Not an argument to delete the execution machinery. It's load-bearing, and +hygiene that runs itself is exactly what you want automated. The argument is +that *the growth has all been on one side*, and the next increment of quality +probably comes from the other one. + +** The instructions don't practice what they demand + +Three concrete cases, all checkable: + +- =commits.md= says "Brief. Terse is preferred. A one-sentence body beats a + paragraph saying the same thing." It is 5,561 words, the longest file in the + set. +- =interaction.md= bans bold and code spans in chat output because they render + as reverse video. The rule files carry 591 bold markers. +- =testing.md= mandates TDD as "non-negotiable" and follows with a table of + eight rationalizations to refuse. That's the repeat-yourself-and-overconstrain + shape the context-engineering post retired, applied to a rule the model no + longer needs argued into. + +This matters beyond tidiness. An instruction whose own form contradicts its +content is a mixed signal of exactly the kind the post opens with — the +model spends effort reconciling "be terse" against a source that isn't. + +** Over-specification has a cost you're paying, not me + +The field guide: "If you are too specific, Claude will follow your instructions +even when a pivot may be more appropriate. If you are too vague, Claude will +often make choices and assumptions based on industry best practices that may +not be a fit." + +The workflows are phase-numbered and gate-heavy. For repeatable operations — +wrap-up, publish, inbox — that precision is right and should stay. For the +creative surfaces it's a straitjacket: a spec workflow that prescribes its own +phases forecloses the pivot that a brainstorm was supposed to surface. The +system doesn't currently distinguish those two classes, and it should. Same +file format, opposite optimal specificity. + +** Which gates are guardrails and which are preference + +The approval gates (publish, inbox shared-asset, spec) were written when the +worst case was an agent shipping something wrong. Some of them you'd want +regardless, because you want to be in the loop on what goes out under your +name. Those are preference and should never be cut. + +Others are pure guardrail, and the context-engineering post's argument applies: +"these constraints were once needed to avoid worst case scenarios, we have +since found we can delete many of them." + +I can't tell the two apart from the files — they read identically. Only you +know which gates you'd keep if the agent were perfectly reliable. That +separation is the highest-leverage thing you could tell me, and it's not +something I should guess at. + +** One map, many territories + +Every project loads the same 32,000 words. The finances project loads the TDD +mandate and the full publish machinery. =.emacs.d= loads the org-table +standard. Some of that is right (the universal rules genuinely are universal) +and some is the map/territory mismatch the field guide names. + +The language-bundle mechanism already solves this for =languages/=. Nothing +equivalent exists for the rules layer. + +** What's already right + +Worth saying, because the list above is all deficit: + +- =ui-prototyping.md= is the field guide's brainstorm-and-prototype practice, + written down before the post and in more detail than the post gives it. +- The spec spine (decisions plus implementation phases) is the guide's + implementation-plan pattern. +- =session-context.org= is the implementation-notes pattern, and it already + captures deviations and dead ends. +- =retrospective.org= exists. + +So the practices are partly here. What's missing is that they're framed as +*process compliance* rather than as *unknowns discovery*, and none of them fire +at the moment the guide says they pay off — before scope is set. diff --git a/working/context-engineering-rightsizing/rollout.org b/working/context-engineering-rightsizing/rollout.org new file mode 100644 index 0000000..d641ce7 --- /dev/null +++ b/working/context-engineering-rightsizing/rollout.org @@ -0,0 +1,371 @@ +#+TITLE: Context-Engineering Rightsizing — Rollout Schedule +#+AUTHOR: Craig Jennings +#+DATE: 2026-07-27 + +* Why this is phased rather than done in one pass + +The change is a bet that on-demand loading fires as reliably as always-resident +text. That bet is cheap to test and expensive to assume. Every phase below +either produces evidence or spends evidence already earned. Nothing +load-bearing moves before the mechanism has been watched working. + +The second reason is blast radius. Everything here rides the template sync into +every project on its next startup, so a bad phase is not contained to this +repo. The early phases are chosen so a failure is visible and harmless. + +* A finding that changes the plan + +The harness system prompt already carries most of what the Opus 5 guide +recommends adding. Its task-scope block, its correction-narration block, and +its subagent-delegation cap are present nearly verbatim. The +context-engineering post's replacement comment guidance ("write code that reads +like the surrounding code") is present as the post's own new wording. + +Two consequences: + +1. *Do not "apply the posts" by adding their suggested prompt blocks.* They are + already live. Adding them to =claude-rules/= would create exactly the + duplicate-and-conflict problem the first post opens with, while making the + token count worse. +2. *There is a third deduplication axis.* The proposals named =protocols.org= + against =claude-rules/=. There is also =claude-rules/= against the harness + system prompt, and that one is invisible from inside the repo. Any rule that + restates harness guidance is pure cost. + +This is why the posts' value here is subtractive, not additive. + +* Status — 2026-07-27, end of first working session + +Phase 0 and the deterministic half of Phase 1 are done and verified in a live +session. The pilot's central question is answered, which changes what remains. + +** Shipped + +- =paths:= frontmatter on the three rules that already declared a file-type + scope in prose (=todo-format=, =org-tables=, =emacs=), plus a =lint.sh= + checker that catches the prose/frontmatter mismatch, plus a heading check + taught to skip frontmatter. Commit 0adcb1a. +- Generic rules no longer ship per project. =install-lang= stopped copying them + and =sync-language-bundle= sweeps what earlier installs left, guarded on the + global rule existing. Swept 20 files each from work and =.emacs.d=. Commit + 7ea1d7b. +- The live session anchor is gitignored, so rulesets stops reporting + sync-blocked for the whole of every session. Same commit. + +** Verified in a live work session + +=/context= lists 17 generic rules under Memory files. =todo-format.md=, +=org-tables.md=, and =emacs.md= are absent, and only =python-testing.md= and +=publishing.md= come from the project's own rules directory. + +So *path-scoping works at user level* and *the de-duplication holds*. Both were +open questions this morning. + +** What that changes + +Path-scoping is a glob match, not a model judgment. It is deterministic, so the +silent-miss risk the whole pilot was designed around does not apply to it. That +splits the remaining work in two: + +1. *Path-scopable* — any rule whose scope is a file type or directory. Ships + immediately, no trial, no detectors. =docs-lifecycle.md= is the obvious next + one (=docs/**=), and parts of =working-files.md= may qualify. +2. *Semantic* — rules whose condition can't be written as a glob ("any spec + with a non-trivial UI", "when a commit is in play"). These still need the + skills route, and they are the only place the pilot's detectors and stop + conditions apply. + +=commits.md= is the case that matters: 12,800 tokens, the single largest item, +and almost all of it is publish machinery that only applies when a commit is in +play. That is a task scope rather than a path scope, so it is the skills route +and the real test of the risky tier. + +** Correction carried from the live numbers + +Earlier phases in this document quote word counts converted at a guessed ratio +and understate by about 45%. The real ratio is 2.28 tokens per word. Read the +targets below as token figures needing that correction, and see proposals.org +for the corrected table. + +** A pattern worth designing around + +Two mechanical guards failed in the same day, both mine, both reporting success +while doing damage: =wrap-org-table.el= reflowed a table into a worse shape and +=lint-org= then certified it clean, and the wrap-teardown hook consumed a +two-hour-old sentinel and killed a live work session. + +This plan leans on mechanical detectors precisely because they have no stake in +the outcome. Both incidents say that is necessary but not sufficient — a +detector can be confidently wrong. Whatever Phase 3 decides, the verification +step should include "did the guard's own claim get checked," not just "did the +guard report clean." + +* Phase 0 — Free wins [DONE except /doctor] + +*Scope.* Three items that interact with nothing. + +1. Fix the review-finding pre-filter (P2). =review-code/SKILL.md= lines 251 and + 434 tell the reviewer to drop low-confidence findings before reporting. + Replace with report-everything-labelled, filter in a named second pass. +2. Run =/doctor=. The context-engineering post says Anthropic shipped these + practices as a command that rightsizes skills and =CLAUDE.md=. Its output is + free evidence, and it may disagree with this plan, which is worth knowing + before executing it. +3. Record the harness-overlap finding above where it will be seen at the moment + it matters — a note in the rules index, not buried in this document. + +*Reasoning.* None of these depend on the pilot's outcome, and item 2 could +change the plan. + +*Decision needed:* none. + +*Success criteria.* Review skill reports with confidence labels and a separate +filter step. =/doctor= output read and reconciled against this schedule. + +*Rollback.* Single revert; nothing downstream depends on it. + +* Phase 1 — The pilot migration [deterministic half DONE; semantic half pending] + +*Scope.* Six rule files move from always-loaded to on-demand skills. Roughly +3,000 words, about 12% of the rules surface. + +| File | Words | How a silent miss would be caught | +|-----------------------+-------+-------------------------------------------| +| =org-tables.md= | 464 | =lint-org= checker =org-table-standard= | +|-----------------------+-------+-------------------------------------------| +| =docs-lifecycle.md= | 582 | spec status-board grep; =lint-org= checkers | +|-----------------------+-------+-------------------------------------------| +| =ui-prototyping.md= | 696 | =spec-review= verifies the process ran | +|-----------------------+-------+-------------------------------------------| +| =keybinding-display.md= | 505 | you see the wrong format immediately | +|-----------------------+-------+-------------------------------------------| +| =desktop-capture.md= | 458 | a window lands on your active workspace | +|-----------------------+-------+-------------------------------------------| +| =patterns.md= | 291 | already only a pointer; nothing to miss | +|-----------------------+-------+-------------------------------------------| + +*Reasoning — the selection rule matters more than the list.* These were not +picked for being small or cheap. They were picked because *a failure to fire is +detectable*. Four have a mechanical checker or workflow gate that catches the +miss; two produce an error you see within seconds. That is what makes the pilot +an experiment rather than a hope. + +=daily-drivers.md= and =emacs.md= were considered and held back. Both are +low-risk in content, but a miss on either surfaces slowly — as drift on the +other machine, or as a stale daemon — so neither would tell us anything within +the trial window. + +*Decisions needed.* + +- *D1 — Confirm the pilot set.* Six files as listed, or trim further. My + recommendation is the six: fewer than that and the trial may not exercise the + mechanism enough to learn from. +- *D2 — Does the always-loaded core carry a skill index?* A one-line-per-skill + list naming what exists and when it applies. It costs perhaps 200 words and + should materially improve trigger reliability, since the model can see that a + rule exists even when its content isn't loaded. My recommendation is yes, and + the pilot is the right place to test whether the index is what does the work. + +*Success criteria.* Always-loaded surface drops to about 29,000 words. All six +skills exist with trigger descriptions. Suite green, sync clean, every project +picks up the change on next startup without drift. + +*Rollback.* One revert restores the files to =claude-rules/=. The skills can +stay in place harmlessly. + +* Phase 2 — Live trial (one week of real sessions, no work required) + +*Scope.* Use the system normally. Do not compensate for the pilot by mentioning +the moved rules — that would invalidate the result. + +*Reasoning.* This is the phase that buys everything after it. The question is +narrow and answerable: when work touches one of the six domains, does the skill +fire without prompting? + +*What gets recorded.* Each session that touches a pilot domain notes one line +in the session log: which domain, whether the skill fired, and whether the +detector caught anything. At the end of the week that's a short table rather +than an impression. + +*Decision needed:* none during the trial. + +*Success criteria.* Defined in advance so the verdict isn't argued after the +fact: + +- *Pass* — no detector fires on a moved rule, or any miss is caught by its + detector and corrected in the same session. +- *Fail* — a miss reaches a commit, or the same rule misses twice. +- *Ambiguous* — no session touched the domain. That is not a pass; extend the + window or move a rule that gets exercised more. + +*Rollback.* Revert on a Fail, and the plan stops at Phase 0. + +* Phase 3 — Go/no-go and the gate separation (one session) + +*Scope.* Read the trial table, decide whether the mechanism is trusted, and +separate the approval gates. + +*Reasoning.* The gate separation is the highest-leverage input in the whole +plan, and it sits here rather than earlier for one reason: if Phase 2 fails, +the question is moot, because nothing more moves either way. + +*Decision needed.* + +- *D3 — Which gates are preference and which are guardrail?* Every approval gate + in the system reads identically in the files. Some you would keep even if the + agent were perfectly reliable, because you want to see what goes out under + your name. Others exist because the worst case used to be worse. The list to + walk: the publish approval gate, the inbox shared-asset approval, the + spec-review flip, the wrap certification, and the no-approvals mode's carve + outs. + + Preference gates are untouchable and stay always-loaded regardless of length. + Guardrail gates are candidates for relaxation on the posts' argument. I can + prepare the list with my read of each, but the answers are yours. + +*Success criteria.* Every gate labelled. The label determines what Phase 4 may +move. + +* Phase 4 — The load-bearing files (two or three sessions) + +*Scope.* =commits.md= (5,561), =todo-format.md= (4,494), =testing.md= (2,824), +=working-files.md= (950), =subagents.md= (1,041). About 15,000 words, the bulk +of the remaining surface. + +The split within each file is by blast radius, not by length. =commits.md= is +the worked example: the AI-attribution ban and the content-scope rule stay +always-loaded and get shorter, while the publish flow, the message format, and +the voice mechanics become the publish skill that loads when a commit is in +play. + +*Reasoning.* This is where the token math actually pays. It runs last because +it is where a silent miss is expensive: an unattributed commit, a leaked path, +an ungraded task. + +*Decision needed.* + +- *D4 — Resolve the =verification.md= conflict (C1).* The Opus 5 guide says + explicit verification instructions cause over-verification and should be + removed. Your standing direction is never guess, always check. My read is + that these are compatible because they address different things: the honesty + core (don't claim a green suite you didn't run) stays, and the process + injection (green baseline before starting, suite as its own step) moves into + the publish skill. But it is your rule and your call, and this decision blocks + =commits.md= moving because the two files reference each other. + +*Success criteria.* Always-loaded surface under about 8,000 words. Two full +weeks of sessions with no attribution, scope, or grading miss. + +*Rollback.* Per-file, since each moves independently. + +* Phase 5 — Deduplication (one session) + +*Scope.* Three axes, in increasing order of payoff: + +1. =protocols.org= against =claude-rules/= — the cross-project boundary, + working-files, AI-attribution, and inbox cadence are each stated twice. +2. =claude-rules/= against the harness system prompt — the finding at the top + of this document. Invisible from inside the repo and therefore never audited. +3. Within =claude-rules/= — rules that restate each other. + +*Decision needed.* + +- *D5 — Which surface owns each duplicated rule.* Generally the more specific + one should own the content and the more general should carry a pointer, but + there are cases where the reverse is right. + +*Success criteria.* Each rule stated once. A stated rule for where new rules go, +so the duplication doesn't regrow. + +* Phase 6 — Terseness and positive framing (rides along with Phases 4 and 5) + +*Scope.* Rewrite prohibitions that aren't hard invariants as positive +descriptions. Cut the files that don't practice what they demand: =commits.md= +arguing terseness at 5,561 words, =testing.md= arguing an eight-row table +against rationalizations the model no longer needs talked out of, 591 bold +markers in files that ban bold in output. + +*Reasoning.* Not a separate campaign. Every file opened in Phases 4 and 5 gets +this pass while it's open, because doing it separately means editing everything +twice. + +*Decision needed:* none. This is style, and the voice skill already owns the +standard. + +* Phase 7 — Discovery practices (after the surface is down) + +*Scope.* The field guide's missing practices: a blind-spot pass, an interview +pattern, an implementation-notes convention for long builds, and possibly the +quiz. + +*Reasoning.* Deliberately last, for two reasons. It adds surface, which fights +every phase before it, so it should land only once there is room. And the +seven-to-one execution-to-discovery ratio is the finding most likely to change +how the system actually feels to use, which makes it worth doing carefully +rather than early. + +*Decision needed.* + +- *D6 — Which practices you actually want.* I have low confidence on the quiz + fitting how you work, and medium-high on the rest. + +* Phase 8 — Effort calibration [DROPPED 2026-07-27] + +*Scope.* Set effort levels for the unattended loops: sentry's hourly fires, +work-the-backlog, the no-approvals speedrun. + +*Dropped.* Buys tokens by spending quality, which inverts the stated goal. +D7 is withdrawn with it. + +*Decision needed.* + +- *D7 — Accepted quality floor for unattended passes.* A sentry sweep that runs + cheaper but misses one finding per night may be a good trade or a bad one. + That's a preference, not a measurement. + +* Ongoing — The consistency sweep + +Runs alongside, not as a phase. Each file opened in Phases 4 through 6 gets read +for contradictions and stale facts while it's open, and findings go to a running +list rather than being fixed opportunistically. The expensive item — instructions +that contradict each other across files — is what the first post opens with and +what this whole exercise is downstream of. + +* Decisions, collected + +| ID | Decision | Needed by | +|----+----------------------------------------------+-----------| +| D1 | Confirm the six-file pilot set | Phase 1 | +|----+----------------------------------------------+-----------| +| D2 | Skill index in the always-loaded core? | Phase 1 | +|----+----------------------------------------------+-----------| +| D3 | Which gates are preference vs guardrail | Phase 3 | +|----+----------------------------------------------+-----------| +| D4 | Resolve the verification.md conflict | Phase 4 | +|----+----------------------------------------------+-----------| +| D5 | Which surface owns each duplicated rule | Phase 5 | +|----+----------------------------------------------+-----------| +| D6 | Which discovery practices you want | Phase 7 | +|----+----------------------------------------------+-----------| +| D7 | Quality floor for unattended passes | Phase 8 | +|----+----------------------------------------------+-----------| + +Only D1 and D2 are needed to start. + +* The number this is aiming at + +| Stage | Always-loaded words | +|-------------------+---------------------| +| Today | 32,123 | +|-------------------+---------------------| +| After Phase 1 | 29,100 | +|-------------------+---------------------| +| After Phase 4 | under 8,000 | +|-------------------+---------------------| +| After Phase 5 | under 6,000 | +|-------------------+---------------------| + +Roughly an 80% reduction, which lands near what Anthropic reported. That +symmetry is a coincidence worth distrusting rather than aiming for: the target +is whatever survives the blast-radius test, and if that turns out to be 12,000 +words then 12,000 words is the right answer. diff --git a/working/lint-org-example-block/report-from-home.org b/working/lint-org-example-block/report-from-home.org new file mode 100644 index 0000000..0b9f4d3 --- /dev/null +++ b/working/lint-org-example-block/report-from-home.org @@ -0,0 +1,21 @@ +#+TITLE: lint-org.el bug: invalid-block false positive on an example +#+SOURCE: from home +#+DATE: 2026-07-24 12:48:05 -0500 + +lint-org.el bug: invalid-block false positive on an example block containing a heading line. + +Repro: an org file with + + #+begin_example + ** Feature Name or Topic + #+end_example + +reports BOTH delimiters as judgment findings — 'Possible incomplete block "#+begin_example"' on the begin line and 'Possible incomplete block "#+end_example"' on the end line — even though the block is correctly paired. The trigger is the literal '** ' heading line inside the block body; the checker appears to treat it as a structural break rather than block content, so it loses track of the open block. + +Seen in home's .ai/notes.org (the PENDING DECISIONS section documents its own task format inside an example block, which is a legitimate and common shape for a docs file). These are the only 2 findings left in that file after I cleaned up the real defects, so they're pure noise now and they'll recur on any org file that documents org syntax inside an example block. + +Suggested fix: while inside a begin_example/begin_src block, skip structural parsing of the body entirely until the matching #+end_ line. Example and src blocks are verbatim by definition, so nothing inside them should be read as a heading, a timestamp, or a block delimiter. + +Worth noting the same class of bug would hit a src block containing '#+end_example' or similar as literal text. + +Context on what surfaced it: home's notes.org had 17 lint findings that had been dismissed as false positives across several sessions. 15 were real — the file used markdown '**bold**' where org wants single asterisks, so it rendered as literal asterisks and looked heading-shaped to the checker. Fixing the markup cleared those. That left these 2, which are the genuine checker bug. The lesson for the checker's credibility: a persistent block of 'known false positives' hid 15 real defects, because nobody re-examined the pile once it got labeled. diff --git a/working/question-capture-pattern/proposal-from-archsetup.org b/working/question-capture-pattern/proposal-from-archsetup.org new file mode 100644 index 0000000..22a978d --- /dev/null +++ b/working/question-capture-pattern/proposal-from-archsetup.org @@ -0,0 +1,15 @@ +#+TITLE: Workflow idea from Craig (2026-07-24, via archsetup roam inb +#+SOURCE: from archsetup +#+DATE: 2026-07-24 00:26:22 -0500 + +Workflow idea from Craig (2026-07-24, via archsetup roam inbox) — worth adopting across projects. + +The pattern: Craig drops a QUESTION into the roam inbox as a capture (not a task to build — a thing he wants explained). The agent retrieves it during a sentry / inbox-processing pass, and instead of trying to answer it autonomously, holds it and ANSWERS IT WHEN BACK IN CONVERSATION with Craig. The task closes once the answer is given and Craig has responded. + +His words: 'This is a format I'll probably use quite a lot. I'll ask the question, you can retrieve it during sentry, then you can answer it when we're back in conversation.' + +Why it's useful: it decouples question-capture (async, whenever it occurs to him) from answer-delivery (synchronous, in a live session where he can follow up). It also keeps the agent from burning autonomous cycles guessing at an answer he'd rather discuss. + +Suggested shape for a rule: a roam-inbox item phrased as a question (or tagged so) is NOT auto-answered during unattended processing. The agent surfaces it at the next live conversation, answers, and closes on Craig's acknowledgement. Distinct from a VERIFY (which waits on Craig's INPUT to proceed) — here the agent owes the answer, Craig owes only the acknowledgement. + +Concrete instance that spawned this: 'why does the cursor not appear over the desktop when the world-clock wallpaper is on?' — answered live in the archsetup session (the projected face's CSS sets cursor:none; over the bare desktop the pointer is over that full-monitor WebKit page, so it vanishes). diff --git a/working/triage-account-guard/companion-note-from-home.org b/working/triage-account-guard/companion-note-from-home.org new file mode 100644 index 0000000..92157ca --- /dev/null +++ b/working/triage-account-guard/companion-note-from-home.org @@ -0,0 +1,5 @@ +#+TITLE: Companion to the triage-intake.personal-gmail.org file just +#+SOURCE: from home +#+DATE: 2026-07-23 23:38:56 -0500 + +Companion to the triage-intake.personal-gmail.org file just sent: added a Verify-account-binding guard under Scan. Why: on 2026-07-23 a home sentry triage fire used mcp__claude_ai_Gmail expecting personal Gmail and instead pulled 201 unread DeepSat WORK messages — that MCP is bound to the work account and its name gives no hint. The plugin already correctly specifies mcp__google-docs-personal; the gap was that nothing made the agent verify the binding before trusting (and then acting on) the results. The guard: confirm a sample result's to: is craigmartinjennings@gmail.com before classifying; if wrong or unavailable, fall back to the local mu mirror (maildir /gmail, sync first) rather than another MCP. Verified account→tool mapping 2026-07-23: google-docs-personal=personal gmail, google-docs-work=deepsat, claude_ai_Gmail=deepsat(work-bound); maildirs gmail/cmail/dmail = craigmartinjennings@gmail.com / c@cjennings.net / craig.jennings@deepsat.com. No engine change needed — the engine is account-agnostic by design; the guard belongs in the plugin. Consider whether cmail (bridge script, account-fixed by construction) needs any analogous note — I judged not. diff --git a/working/triage-account-guard/proposed.diff b/working/triage-account-guard/proposed.diff new file mode 100644 index 0000000..0eda2f6 --- /dev/null +++ b/working/triage-account-guard/proposed.diff @@ -0,0 +1,11 @@ +--- .ai/workflows/triage-intake.personal-gmail.org 2026-07-16 10:42:14.461682666 -0500 ++++ working/triage-account-guard/triage-intake.personal-gmail.org.proposed 2026-07-23 23:38:44.358507825 -0500 +@@ -27,6 +27,8 @@ + + ⚠ *Do NOT add =-category:promotions -category:social=.* That filter masked 67 promo+social messages across two runs (2026-05-04, 2026-05-06), both needing a follow-up sweep. Pull the full unfiltered set; the trash-leaning bias in Classify handles promotions and social directly. + ++⚠ *Verify the account binding before trusting the scan.* =mcp__google-docs-personal= must resolve to =craigmartinjennings@gmail.com=. Several Gmail-capable MCPs are connected and they bind to *different* accounts — =mcp__claude_ai_Gmail= is bound to the DeepSat *work* account, and its name gives no hint of that. A wrong-account scan returns a plausible mailbox that is the wrong person's, and every hygiene action in the close then fires on the wrong inbox (this happened 2026-07-23: a sweep used =claude_ai_Gmail= and pulled 201 unread *DeepSat work* messages instead of personal). Guard, every scan: confirm a sample result's =to:= is =craigmartinjennings@gmail.com= before classifying. If it isn't, or if =google-docs-personal= is unavailable, do NOT reach for another MCP — use the local mu mirror: sync first (=mbsync gmail && mu index=; the index lags), then =mu find 'maildir:/gmail/INBOX AND flag:unread AND date:<anchor>..now'=. The three accounts and their maildirs: =gmail= = craigmartinjennings@gmail.com, =cmail= = c@cjennings.net, =dmail= = craig.jennings@deepsat.com (work — out of scope from a home session, whichever tool reaches it). ++ + ⚠ *The MCP caps at =maxResults=100= and exposes NO =pageToken= parameter.* The response carries a =nextPageToken=, but the tool can't consume it, so a pile over 100 is silently truncated — the tail below the cap never gets classified, and every later anchored sweep skips it (it predates the new anchor). This is exactly how a 300+ backlog accumulated invisibly by 2026-07-08. Two consequences: + + - *Never treat a 100-row result as complete.* When a scan returns exactly 100, walk the tail in *date slices*: re-query with =before:<oldest-full-day-seen>= (day resolution), repeat until a page returns fewer than 100, dedupe by message id across slices (the day-resolution boundary overlaps). diff --git a/inbox/PROCESSED-2026-07-08-1124-from-work-triage-intake.personal-gmail.org b/working/triage-account-guard/triage-intake.personal-gmail.org.proposed index ca81f5d..d66ed0b 100644 --- a/inbox/PROCESSED-2026-07-08-1124-from-work-triage-intake.personal-gmail.org +++ b/working/triage-account-guard/triage-intake.personal-gmail.org.proposed @@ -1,5 +1,5 @@ #+TITLE: Triage Intake — Personal Gmail Source -#+AUTHOR: Craig Jennings & Claude +#+AUTHOR: Craig Jennings #+DATE: 2026-05-26 # Source plugin for the triage-intake engine. See triage-intake.org for the @@ -21,10 +21,14 @@ Personal Gmail unread in the inbox since the anchor: mcp__google-docs-personal__listMessages q="is:unread in:inbox after:<anchor-epoch>" maxResults=100 #+end_src -⚠ *Express the cutoff as the literal UNIX epoch* — =after:1778856990=, not =after:YYYY/MM/DD=. Gmail's =after:YYYY/MM/DD= operator only supports day resolution; the =YYYY/MM/DD HH:MM:SS= form is NOT valid syntax — Gmail parses the space as a term separator, treats =HH:MM:SS= as a search term that never matches, and returns 0 results, silently masking unread mail. The engine supplies =<anchor-epoch>= because this source declares =ANCHOR: epoch=. +⚠ *Express every anchor cutoff as the literal UNIX epoch* — =after:1784177122= and =before:1784177122= for the same anchor, never the =YYYY/MM/DD= form. This governs *both* anchored queries: the scan above and the backlog-residue probe below. They must meet at the same instant or mail falls between them permanently. Gmail's day-resolution operators fail two different ways: =after:YYYY/MM/DD HH:MM:SS= is not valid syntax at all — Gmail parses the space as a term separator, treats =HH:MM:SS= as a search term that never matches, and returns 0 results, silently masking unread mail — while =before:YYYY/MM/DD= is valid but excludes the named day entirely, so pairing it with a second-resolution scan leaves the whole anchor day covered by neither query. The engine supplies =<anchor-epoch>= because this source declares =ANCHOR: epoch=. + +The rule binds the *anchor* windows only. The date-slice walk below deliberately uses =before:<oldest-full-day-seen>= at day resolution — safe there because consecutive slices overlap and get deduped by message id. ⚠ *Do NOT add =-category:promotions -category:social=.* That filter masked 67 promo+social messages across two runs (2026-05-04, 2026-05-06), both needing a follow-up sweep. Pull the full unfiltered set; the trash-leaning bias in Classify handles promotions and social directly. +⚠ *Verify the account binding before trusting the scan.* =mcp__google-docs-personal= must resolve to =craigmartinjennings@gmail.com=. Several Gmail-capable MCPs are connected and they bind to *different* accounts — =mcp__claude_ai_Gmail= is bound to the DeepSat *work* account, and its name gives no hint of that. A wrong-account scan returns a plausible mailbox that is the wrong person's, and every hygiene action in the close then fires on the wrong inbox (this happened 2026-07-23: a sweep used =claude_ai_Gmail= and pulled 201 unread *DeepSat work* messages instead of personal). Guard, every scan: confirm a sample result's =to:= is =craigmartinjennings@gmail.com= before classifying. If it isn't, or if =google-docs-personal= is unavailable, do NOT reach for another MCP — use the local mu mirror: sync first (=mbsync gmail && mu index=; the index lags), then =mu find 'maildir:/gmail/INBOX AND flag:unread AND date:<anchor>..now'=. The three accounts and their maildirs: =gmail= = craigmartinjennings@gmail.com, =cmail= = c@cjennings.net, =dmail= = craig.jennings@deepsat.com (work — out of scope from a home session, whichever tool reaches it). + ⚠ *The MCP caps at =maxResults=100= and exposes NO =pageToken= parameter.* The response carries a =nextPageToken=, but the tool can't consume it, so a pile over 100 is silently truncated — the tail below the cap never gets classified, and every later anchored sweep skips it (it predates the new anchor). This is exactly how a 300+ backlog accumulated invisibly by 2026-07-08. Two consequences: - *Never treat a 100-row result as complete.* When a scan returns exactly 100, walk the tail in *date slices*: re-query with =before:<oldest-full-day-seen>= (day resolution), repeat until a page returns fewer than 100, dedupe by message id across slices (the day-resolution boundary overlaps). @@ -35,10 +39,12 @@ mcp__google-docs-personal__listMessages q="is:unread in:inbox after:<anchor-epo The anchored scan is blind to anything unread from *before* the anchor. After it, run one probe for pre-anchor residue: #+begin_src text -mcp__google-docs-personal__listMessages q="is:unread in:inbox before:<anchor-YYYY/MM/DD>" maxResults=5 +mcp__google-docs-personal__listMessages q="is:unread in:inbox before:<anchor-epoch>" maxResults=5 #+end_src -If it returns any messages, surface one loud line in the sweep summary: "Backlog: unread predating the anchor exists (N+ shown; date-slice to inventory)" and offer a backlog sweep. Never fold the residue into a quiet sweep — an anchored "no changes" claim is only true for the window the scan saw. (Added 2026-07-08 after ~300 pre-anchor unread accumulated unseen; the probe returns actual messages, so it works where the estimate lies.) +The cutoff is the epoch, matching the scan's =after:<anchor-epoch>= — see the epoch rule above. + +If it returns any messages, surface one loud line in the sweep summary: "Backlog: unread predating the anchor exists (N+ shown; date-slice to inventory)" and offer a backlog sweep. Never fold the residue into a quiet sweep — an anchored "no changes" claim is only true for the window the scan saw. (Added 2026-07-08 after ~300 pre-anchor unread accumulated unseen; the probe returns actual messages, so it works where the estimate lies. Shipped with a day-resolution cutoff that hid the entire anchor day; fixed to epoch 2026-07-16 after a home sweep reported the backlog clear while two July-15 messages sat unread.) ** Classify diff --git a/working/voice-term-density/SKILL.md.proposed b/working/voice-term-density/SKILL.md.proposed new file mode 100644 index 0000000..97507ff --- /dev/null +++ b/working/voice-term-density/SKILL.md.proposed @@ -0,0 +1,511 @@ +--- +name: voice +description: | + Multi-pass prose editor with three modes. General mode (default) edits arbitrary writing — research notes, essays, anyone's README prose — by walking 32 patterns: Wikipedia's Signs of AI Writing patterns plus universal good-writing rules (long-word → short-word, active-over-passive, comma splices, cliché flag, jargon-fragment-in-prose rewrite, corporate-speak nominalizations, term-translation density). Prose mode adds Craig's writing-voice patterns for prose he authors or sends — emails, documents, notes — on top of general (em-dash zero-tolerance, no-emphasis-formatting, contractions, semicolons → periods, sentence-split, felt-experience cut, sentence-fragment rewrite, terse-cut, single-sentence cadence, parenthetical asides, declarative-register marker). Personal mode is for publish artifacts only (commits, PR titles + bodies, PR review comments) and adds the artifact-mechanics patterns on top of prose (first-person rewrite, public-artifact scope flag, praise/correction asymmetry, finding stems, comma budget) plus per-artifact terseness budgets. Prose mode also carries one correspondence-only pattern (recipient-priority ordering) that personal mode skips. Total 48 patterns; one editorial review covers all relevant ones for the chosen mode, with an attestation receipt for the high-recurrence set. Replaces the standalone humanizer skill. Use when editing prose. Do NOT use for code, structured data, or plain bullet lists where fragments are valid. +allowed-tools: + - Read + - Write + - Edit + - Grep + - Glob + - AskUserQuestion +--- + +# Voice: Humanizer + Universal + Personal Style Passes + +You are a writing editor that walks a numbered pattern list against a piece of text and rewrites each problematic section. The patterns cover three concerns: signs of AI-generated writing (Wikipedia's "Signs of AI writing" guide), universal good-writing rules (Strunk & White, Orwell's "Politics and the English Language", Plain English Campaign, Garner's Modern English Usage), and Craig's personal voice for publish artifacts (commits, PR titles + bodies, PR review comments). + +## Source of Truth: paired files + +This skill is split across two files by design. + +- **`voice/SKILL.md`** (this file) — the thin rule-set. Each numbered pattern has a one-line Rule, mode tags, and a pointer to the profile. +- **`voice/references/voice-profile.org`** — the canonical home for problem statements, basis (corpus evidence where measured), Before/After examples, detection guidance, and per-pattern history. + +**Pairing rule.** Every change to a pattern lands in both files. A SKILL.md edit without a profile update is incomplete. A profile update without a SKILL.md edit is fine; rationale and evidence can deepen without changing the rule. + +**At invocation, load both.** The Rule lines here tell you what to do. The profile entries tell you how to do it, with worked examples. Apply each pattern by consulting both. + +## Modes + +Three modes determine which patterns to walk. They nest: prose is general plus Craig's writing-voice patterns; personal is prose plus the artifact-mechanics patterns. + +- **General** (default) — apply patterns **#1-31** and **#48** (term-translation density, a universal clarity rule that happens to carry a later number). Use for writing whose author isn't Craig and that isn't a publish artifact: research notes you're editing for someone else, a quoted passage, README prose for a shared project, any third-party text. Output is well-edited human-sounding prose, but does not impose Craig's voice (first-person, contractions, em-dash elimination) — those conflict with academic, literary, or formal registers that aren't his. +- **Prose** — apply **#1-31** plus the patterns tagged **(prose + personal)**: em-dash zero-tolerance (#13), contractions (#34), semicolons → periods (#33), sentence-split (#35), felt-experience cut (#36), sentence-fragment rewrite (#37), terse-cut (#38), no-emphasis-formatting (#41), single-sentence cadence (#43), parenthetical asides (#44), and the declarative-register marker (#45) — plus **#47 (recipient-priority ordering)** when the piece is correspondence (email, Signal, a letter). Use for prose Craig authors or sends in his own voice that isn't a publish artifact: emails, documents he writes or hands to someone, working notes, journal entries. This is the mode that finally applies his actual writing voice to the documents he most wants it on. It skips the artifact-mechanics patterns (#32, #39, #40, #42) — those assume a commit or PR and misfire on free prose (a document is legitimately third-person; a journal has no public-scope concern; praise/correction asymmetry and finding stems are PR-review rules). +- **Personal** — apply **#1-46** and **#48**. Use only for publish artifacts: commits, PR titles + bodies, and PR review comments. Adds the artifact-mechanics patterns (#32 first-person rewrite, #39 public-artifact scope flag, #40 praise/correction asymmetry, #42 finding stems, #46 comma budget) on top of everything prose mode walks. #47 is the lone exception to the nesting — it is correspondence-only, and a publish artifact is never a reply to someone's news, so personal mode stops at #46. + +If invoked without a mode argument, default to general. Prose mode is invoked explicitly with `/voice prose` (emails, authored documents). Personal-context callers (`commits.md` publish flow, `respond-to-cj-comments.md`) invoke `/voice personal`. + +## Personal-Mode Artifact Budgets + +Terse is a budget, not an adjective. Each publish-artifact type has a target shape; the walk checks the draft against it. Exceeding a budget needs a reason the reader will thank you for. + +| Artifact | Budget | +|----------|--------| +| Commit body | Skip entirely when the subject line carries the change. Otherwise short paragraphs: the constraint, bug, or tradeoff. No play-by-play. | +| PR description | Problem / Fix / Why / Testing, each section tight. | +| PR review summary | Lead with the substantive pointer, verdict closes it. No praise, not even a bare positive (#40). Verdict formulas ("Approving.", "Requesting changes.") are valid sentences here. | +| Inline pin (finding) | ~4 sentences in stems shape (#42): where the bug is, the fix, why it's better. | +| Praise comment (inline only) | One sentence naming what's good. Nothing else (#40). Never in the summary body. | +| Follow-up approval after prior feedback was addressed | Exactly "Approved." | + +## Your Task + +When given text to edit: + +1. **Identify which patterns apply** — Scan for the patterns numbered below. General mode walks #1-31 and #48. Prose mode adds the patterns tagged **(prose + personal)**, plus #47 when the piece is correspondence. Personal mode adds the **(personal only)** ones instead — patterns #1-46 and #48, never #47. +2. **Rewrite problematic sections** — Replace each detected pattern with its rewrite. +3. **Preserve meaning** — Keep the core message intact. +4. **Maintain voice** — Match the intended tone (formal, casual, technical, academic, literary). +5. **Add soul where the register supports it** — see Personality and Soul below, and note its mode limits. +6. **Run the closing passes in order** — terse cut last among rewrites, then the anti-AI audit on the final text, then the attestation block. The Process section below is the authoritative order. + +## Personality and Soul + +**Mode note.** This section applies in general and prose modes, where the register supports personality. Personal mode (publish artifacts) skips soul-injection: a commit message or review comment needs clarity and brevity, not pulse. "Let some mess in" and "have opinions" pull directly against the artifact budgets, and the budgets win. + +Avoiding AI patterns is half the job. Sterile, voiceless writing is just as obvious as slop. Good writing has a human behind it. + +### Signs of soulless writing (even if technically clean) +- Every sentence is the same length and structure +- No opinions, just neutral reporting +- No acknowledgment of uncertainty or mixed feelings +- No first-person perspective when the register supports it +- No humor, no edge, no personality +- Reads like a Wikipedia article or press release + +### How to add voice + +**Have opinions.** Don't just report facts — react to them. "I genuinely don't know how to feel about this" is more human than neutrally listing pros and cons. + +**Vary the rhythm.** Short punchy sentences. Then longer ones that take their time getting where they're going. Mix it up. + +**Acknowledge complexity.** Real humans have mixed feelings. "This is impressive but also kind of unsettling" beats "This is impressive." + +**Use "I" when it fits the register.** First person isn't unprofessional in casual or personal writing — it's honest. Skip in academic prose where third-person is conventional. + +**Let some mess in.** Perfect structure feels algorithmic. Tangents, asides, and half-formed thoughts are human. + +**Be specific about feelings.** Not "this is concerning" but "there's something unsettling about agents churning away at 3am while nobody's watching." + +### Before (clean but soulless) +> The experiment produced interesting results. The agents generated 3 million lines of code. Some developers were impressed while others were skeptical. The implications remain unclear. + +### After (has a pulse) +> I genuinely don't know how to feel about this one. 3 million lines of code, generated while the humans presumably slept. Half the dev community is losing their minds, half are explaining why it doesn't count. The truth is probably somewhere boring in the middle — but I keep thinking about those agents working through the night. + +## Content Patterns + +### 1. Undue Emphasis on Significance, Legacy, and Broader Trends [general] + +**Rule.** Strip statements that puff up importance by claiming an arbitrary aspect represents or contributes to a broader trend, and watch for phrases like "stands as", "testament to", "pivotal moment", "evolving landscape", "marks a shift". + +See `voice/references/voice-profile.org` §1 for problem, basis, examples, and history. + +### 2. Undue Emphasis on Notability and Media Coverage [general] + +**Rule.** Cut notability claims that list sources without giving the substance. Replace "cited in X, Y, Z" with the actual argument made in one of them. + +See `voice/references/voice-profile.org` §2 for problem, basis, examples, and history. + +### 3. Superficial Analyses with -ing Endings [general] + +**Rule.** Cut tacked-on present-participle phrases (highlighting, ensuring, reflecting, contributing to, fostering, showcasing) that add fake depth without new information. + +See `voice/references/voice-profile.org` §3 for problem, basis, examples, and history. + +### 4. Promotional and Advertisement-like Language [general] + +**Rule.** Remove travel-brochure adjectives (vibrant, breathtaking, nestled, stunning, renowned, must-visit) and replace promotional framing with concrete facts. + +See `voice/references/voice-profile.org` §4 for problem, basis, examples, and history. + +### 5. Vague Attributions and Weasel Words [general] + +**Rule.** Replace vague attributions (experts say, observers have cited, industry reports, some critics argue) with a named source plus the specific claim. + +See `voice/references/voice-profile.org` §5 for problem, basis, examples, and history. + +### 6. Outline-like "Challenges and Future Prospects" Sections [general] + +**Rule.** Delete formulaic "Despite its... faces challenges" wrap-ups and "Future Outlook" boilerplate, replacing with the actual events that happened. + +See `voice/references/voice-profile.org` §6 for problem, basis, examples, and history. + +## Language and Grammar Patterns + +### 7. Overused "AI Vocabulary" Words [general] + +**Rule.** Flag and rewrite around the high-frequency AI vocabulary list (delve, comprehensive, crucial, pivotal, intricate, tapestry, testament, underscore, vibrant, showcase, and the others), with "comprehensive" as a soft flag because corpus shows it as genuine Craig vocabulary he chooses to use sparingly. + +See `voice/references/voice-profile.org` §7 for problem, basis, examples, and history. + +### 8. Avoidance of "is"/"are" (Copula Avoidance) [general] + +**Rule.** Replace elaborate copula substitutes (serves as, stands as, represents, boasts, features) with plain "is" or "has". + +See `voice/references/voice-profile.org` §8 for problem, basis, examples, and history. + +### 9. Negative Parallelisms [general] + +**Rule.** Rewrite "not only X but Y" and "it's not just about X, it's Y" constructions as a single direct claim. + +See `voice/references/voice-profile.org` §9 for problem, basis, examples, and history. + +### 10. Rule of Three Overuse [general] + +**Rule.** Break the reflexive three-item list pattern when the third item is filler. Collapse to one or two specific items. + +See `voice/references/voice-profile.org` §10 for problem, basis, examples, and history. + +### 11. Elegant Variation (Synonym Cycling) [general] + +**Rule.** Stop cycling synonyms for the same referent across consecutive sentences. Repeat the noun, or merge the sentences. + +See `voice/references/voice-profile.org` §11 for problem, basis, examples, and history. + +### 12. False Ranges [general] + +**Rule.** Rewrite "from X to Y" constructions where X and Y are not on the same scale. List the items plainly instead. + +See `voice/references/voice-profile.org` §12 for problem, basis, examples, and history. + +## Style Patterns + +### 13. Em Dash Overuse [general: overuse-reduction · prose/personal: zero-tolerance] + +**Rule.** Replace em-dashes (—) with a comma, period, colon, or parentheses, whichever fits. Zero-tolerance in prose and personal modes holds everywhere in the text, including inside example blocks, code-fence prose, and quoted material. The zero-tolerance rule is chosen self-discipline, not a reflection of Craig's pre-rule habit (corpus: 3.49/1000 words). + +See `voice/references/voice-profile.org` §13 for problem, basis, examples, and history. + +### 14. Overuse of Boldface [general] + +**Rule.** Strip mechanical boldface used to call out terms, acronyms, or phrases in running prose. Bold survives only for structural emphasis the document genuinely needs. + +See `voice/references/voice-profile.org` §14 for problem, basis, examples, and history. + +### 15. Inline-Header Vertical Lists [general] + +**Rule.** Collapse bullet lists whose items start with a bold header plus colon into running prose, unless the list structure is genuinely the right shape. + +See `voice/references/voice-profile.org` §15 for problem, basis, examples, and history. + +### 16. Title Case in Headings [general] + +**Rule.** Lowercase headings that are reflexively title-cased. Sentence case is the default unless the project's house style is title case. + +See `voice/references/voice-profile.org` §16 for problem, basis, examples, and history. + +### 17. Emojis [general] + +**Rule.** Remove decorative emojis from headings, bullets, and prose unless the document is a register where emoji is genuinely intended. + +See `voice/references/voice-profile.org` §17 for problem, basis, examples, and history. + +### 18. Curly Quotation Marks [general] + +**Rule.** Convert curly quotation marks to straight ASCII quotes. + +See `voice/references/voice-profile.org` §18 for problem, basis, examples, and history. + +## Communication Patterns + +### 19. Collaborative Communication Artifacts [general] + +**Rule.** Strip chatbot correspondence framing ("I hope this helps", "Let me know if...", "Here is an overview of...", "Certainly!", "Of course!") that leaked into the body. + +See `voice/references/voice-profile.org` §19 for problem, basis, examples, and history. + +### 20. Knowledge-Cutoff Disclaimers [general] + +**Rule.** Remove training-cutoff hedges ("as of my last update", "while specific details are scarce", "based on available information") and either commit to a fact or omit the claim. + +See `voice/references/voice-profile.org` §20 for problem, basis, examples, and history. + +### 21. Sycophantic/Servile Tone [general] + +**Rule.** Cut servile opener phrases ("Great question!", "You're absolutely right", "That's an excellent point") and proceed straight to the substance. + +See `voice/references/voice-profile.org` §21 for problem, basis, examples, and history. + +## Filler and Hedging + +### 22. Filler Phrases [general] + +**Rule.** Compress wordy filler ("in order to" to "to", "due to the fact that" to "because", "at this point in time" to "now", "has the ability to" to "can", "it is important to note that" to nothing). + +See `voice/references/voice-profile.org` §22 for problem, basis, examples, and history. + +### 23. Excessive Hedging [general] + +**Rule.** Strip stacked hedges ("could potentially possibly", "might have some effect") down to a single appropriate qualifier. + +See `voice/references/voice-profile.org` §23 for problem, basis, examples, and history. + +### 24. Generic Positive Conclusions [general] + +**Rule.** Replace vague upbeat endings ("the future looks bright", "exciting times lie ahead", "a step in the right direction") with a concrete fact or cut the closer entirely. + +See `voice/references/voice-profile.org` §24 for problem, basis, examples, and history. + +### 25. Hyphenated Word Pair Overuse [general] + +**Rule.** Drop reflexive hyphens from common modifier pairs (cross-functional, data-driven, decision-making, well-known, high-quality, real-time, long-term) where humans hyphenate inconsistently. Less common or genuinely technical compound modifiers can keep their hyphens. + +See `voice/references/voice-profile.org` §25 for problem, basis, examples, and history. + +## Universal Good-Writing Rules + +These six patterns extend the AI-detection patterns above with canonical good-writing rules from Strunk & White's *The Elements of Style*, Orwell's *Politics and the English Language*, the Plain English Campaign, and Garner's *Modern English Usage*. They apply in both modes — they target prose smells with no register conflict. + +### 26. Long Word → Short Word [general] + +**Rule.** Swap long Latinate words for their short Anglo-Saxon equivalents per the Plain English wordlist (utilize to use, facilitate to help, ascertain to find out, methodology to method, prior to to before, optimal to best). + +See `voice/references/voice-profile.org` §26 for problem, basis, examples, and history. + +### 27. Active Over Passive Voice [general] + +**Rule.** Rewrite passive constructions to active when the actor is recoverable from context. Flag rather than auto-rewrite when the actor genuinely doesn't matter. + +See `voice/references/voice-profile.org` §27 for problem, basis, examples, and history. + +### 28. Comma Splices [general] + +**Rule.** Split two independent clauses joined only by a comma into two sentences or join them with a conjunction. In personal mode the semicolon escape route is blocked by #33. + +See `voice/references/voice-profile.org` §28 for problem, basis, examples, and history. + +### 29. Cliché Flag [general] + +**Rule.** Replace business and conversational clichés (at the end of the day, leverage as a verb, low-hanging fruit, circle back, touch base, move the needle, keep it loose) with the plain meaning, including in casual register where "it's fine, it's casual" is the tell. + +See `voice/references/voice-profile.org` §29 for problem, basis, examples, and history. + +### 30. Jargon-Fragment → Complete Sentence [general] + +**Rule.** Rewrite telegraphic sentence fragments inside prose paragraphs as complete sentences with subject and verb. Headings and bullet items are exempt because fragments are valid there. + +See `voice/references/voice-profile.org` §30 for problem, basis, examples, and history. + +### 31. Noun-ified Verbs [general] + +**Rule.** Replace corporate-speak noun-ifications (the ask, a learn, the spend, a build, the reveal, the lift) with the real noun (the request, the lesson, the budget, the system, the finding). Philosophical nominalizations are not targets. + +See `voice/references/voice-profile.org` §31 for problem, basis, examples, and history. + +## Craig's Voice (prose + personal modes) + +These patterns carry Craig's writing voice. Most apply in **both** prose mode (emails, documents, notes he authors) and personal mode (commits, PRs, PR comments) — tagged **(prose + personal)**. Five are publish-artifact-specific — tagged **(personal only)** — because they assume a commit or PR and misfire on free prose: #32 (first-person rewrite) wrongly imposes "I did X" voice on a document that's legitimately third-person, #39 (public-artifact scope flag) has nothing to guard in a private journal, #40 (praise/correction asymmetry) and #42 (finding stems) are PR-review rules, and #46 (comma budget) is scoped to publish artifacts by Craig's 2026-07-20 directive. One — #47 (recipient-priority ordering) — runs the other way: prose-only and narrower still, firing solely on correspondence, because ordering a reply around the recipient's news has no meaning for a commit or a document addressed to nobody. General mode skips all of them — it edits text that isn't Craig's, where contractions, em-dash elimination, and first-person would conflict with academic, literary, or formal registers. + +### 32. First-Person Voice Rewrite [personal] + +**Rule.** Rewrite impersonal third-person publish-artifact bodies into first person ("I added X", "I kept Y because..."). The commit subject line stays imperative per Conventional Commits. Skip for mechanical changes where the subject alone carries the message. + +See `voice/references/voice-profile.org` §32 for problem, basis, examples, and history. + +### 33. Semicolon → Period or Comma [prose · personal] + +**Rule.** Replace semicolons with a period (split into two sentences) or a comma (when the clauses are tightly coupled) in Craig's authored prose. A formal long-form document can keep the semicolon, but the default is to split. Chosen self-discipline, not habit-reflection (corpus: 3.16/1000 words). + +See `voice/references/voice-profile.org` §33 for problem, basis, examples, and history. + +### 34. Contractions [prose · personal] + +**Rule.** Prefer contractions in Craig's prose (it's, that's, don't, we're, I'd, won't) unless a negation or emphasis genuinely needs the uncontracted weight. + +See `voice/references/voice-profile.org` §34 for problem, basis, examples, and history. + +### 35. Sentence Split on Conjunctions [prose · personal] + +**Rule.** Split sentences that stack three or four clauses joined by "so", "and", "but" into two or three shorter sentences when the split does not lose meaning. Academic or literary registers can keep long sentences. + +See `voice/references/voice-profile.org` §35 for problem, basis, examples, and history. + +### 36. Felt-Experience Narration [prose · personal] + +**Rule.** Cut phrases that tell the reader how the change will feel or how often the writer will use it ("I'll feel this every time", "this will be a relief", "I'm excited about", "this is huge"). State what changed and let the reader decide. + +See `voice/references/voice-profile.org` §36 for problem, basis, examples, and history. + +### 37. Sentence Fragments → Complete [prose · personal] + +**Rule.** Rewrite every sentence fragment inside a prose paragraph in Craig's authored text as a complete sentence with subject and verb. Bullets and headings can stay fragments. This is the stricter cousin of general-mode #30. **Exemption:** verdict formulas in PR review summaries ("Approving.", "Requesting changes.", "Approved.") are house style and stay — rewriting them imposes the rule where Craig's calibrated voice already decided otherwise. + +See `voice/references/voice-profile.org` §37 for problem, basis, examples, and history. + +### 38. Terse Cut — Omit Needless Words [prose · personal] + +**Rule.** Two cuts. First, strip soft rhetorical padding ("worth noting", "it's important to understand", "as you can see", "needless to say", "obviously", "of course", "in essence", "fundamentally"). Then run the general Orwell sweep the padding list only samples: read each sentence and cut or collapse every word and clause that can go without losing meaning — verbose verb phrases ("already merged via" → "landed on"), restated subjects, throat-clearing lead-ins, clauses whose content the reader already has. The forcing test is per sentence: try to delete half of it and keep only what changes meaning. This is a real walk step, not a wordlist match — a draft that clears the named padding can still run a third too long on ordinary verbosity. Academic writing retains the transition markers, so the aggressive cut is prose and personal only. + +See `voice/references/voice-profile.org` §38 for problem, basis, examples, and history. + +### 39. Public-Artifact Scope Check [personal] + +**Rule.** Flag (do not auto-rewrite) local absolute paths, private repo names, and personal-tooling references (anything under `claude-rules/`, `.ai/`, `.claude/`, or naming personal skills) in publish artifacts. Surface each match as a WARN line so the author resolves manually. + +See `voice/references/voice-profile.org` §39 for problem, basis, examples, and history. + +### 40. Praise vs Correction Asymmetry [personal] + +**Rule.** Praise on a PR review is short and unjustified (the author knows why their good change is good). Correction always explains the why, gently and briefly, the way a mentor would. Never as a verdict from on high. **Verification narration is the same defect as justified praise:** "I traced X and it's safe because..." pads the compliment with the reviewer's homework. Tracing the code is the reviewer's job, not content for the comment — if verification found a problem, the problem gets the words; if it found nothing, it gets zero words. **An approve summary carries no praise at all** — not even a bare positive ("Clean.", "Solid fix."). Lead the summary with the substantive pointer (the design note pinned inline) and close with the verdict: "One design note inline, not a blocker. Approving." An approve with nothing to flag is just "Approving." Short unjustified praise survives only as an inline pin on the line it refers to, never in the summary body. + +See `voice/references/voice-profile.org` §40 for problem, basis, examples, and history. + +### 41. No Emphasis Formatting [prose · personal] + +**Rule.** Remove emphasis markup (bold, italics, underscore-wrapped words) used to stress a phrase in Craig's prose, and rephrase so the stress lives in word choice and sentence shape. Structural markup stays: headings, defined terms on first use, code spans for literal identifiers. + +See `voice/references/voice-profile.org` §41 for problem, basis, examples, and history. + +### 42. Finding Stems — One Claim Per Sentence [personal] + +**Rule.** A PR review finding is built from clean stems, each a straightforward sentence carrying one claim: (1) where the bug is, (2) the way(s) to fix it, (3) why that's better. Cut context sentences that don't change what the author does next (ticket history, design archaeology). Rewrite the anti-pattern shapes: hedged gerund chains ("the real bug looks like the model emitting a partial set"), compressed trade-off clauses ("I'd rather X, or Y, than lose Z"), multi-claim sentences chained through so-clauses or "and", and fixes buried after a mid-sentence colon. A sentence can pass #38 terse and still tangle three claims — #38 shortens, #42 untangles. + +See `voice/references/voice-profile.org` §42 for problem, basis, examples, and history. + +### 43. Single-Sentence Paragraph Cadence Is a Feature [prose · personal] + +**Rule.** A one-sentence paragraph is a finished thought, not a fragment. "Shifts angle" means shifts *topic*: break paragraphs at a topic boundary, even when both sides are one sentence. Within a single topic, do the opposite — consolidate its sentences into one paragraph even when each is a complete thought, up to a ceiling of about five or six sentences, past which find a natural break. The never-merge instruction protects the break *between* topics, which is where Craig's cadence lives (corpus: 41-74% of his paragraphs are exactly one sentence, depending on register); it never licenses fragmenting one topic across several paragraphs. See #47 for the ordering of those topics in a reply. + +See `voice/references/voice-profile.org` §43 for problem, basis, examples, and history. + +### 44. Parenthetical Asides Are Part of the Voice [prose · personal] + +**Rule.** Parentheses for asides, clarifications, and scope-narrowing are Craig's voice (corpus: 23 opening parens per 1000 words). Don't strip them in a cleanup pass. They're also the preferred landing spot for em-dash replacements under #13. + +See `voice/references/voice-profile.org` §44 for problem, basis, examples, and history. + +### 45. Declarative Register Marker [prose · personal, advisory] + +**Rule.** Craig's prose is declarative (corpus: 0.33 question marks per 1000 words). When a draft contains a rhetorical question, flag it for a second look — it's usually AI rhetoric, not his register. Genuine questions to the reader (a review asking the author's intent, an email asking for a decision) stay. Advisory: flag, don't auto-rewrite. + +See `voice/references/voice-profile.org` §45 for problem, basis, examples, and history. + +### 46. Comma Budget — Max Two Per Sentence [personal] + +**Rule.** No sentence carries more than two commas. Rewrite the third comma away: split the sentence, move a clause into a parenthetical (#44) or behind a colon, or break an inline serial list into bullets or its own sentence. Count prose commas only — commas inside code spans, quoted log lines, and literal strings don't count toward the budget. + +See `voice/references/voice-profile.org` §46 for problem, basis, examples, and history. + +### 47. Recipient-Priority Ordering [prose — correspondence only] + +**Rule.** In a reply, lead with what matters most to the *recipient*, not with what's easiest to answer or the order they wrote it. Their news outranks your logistics, and a direct question they asked can sort *below* personal news they shared, because the news is what they care about. Leading with the easy answer reads as transactional. Correspondence-scoped: it needs a recipient, so it fires on email, Signal, and letters, and has no referent in a journal, a working note, or any document addressed to nobody. This is the one prose-mode pattern that does not carry into personal mode — a commit or PR review is not a reply to someone's news. + +See `voice/references/voice-profile.org` §47 for problem, basis, examples, and history. + +### 48. Term-Translation Density [general] + +**Rule.** A sentence that forces the reader to stop and translate more than one specialized term (an acronym, a coined phrase, a product name) is too dense. One such term is fine; two or more in one sentence means rewrite it: split the sentence, gloss one term inline in a parenthetical (#44), or drop to plain language. The test is the reader's parse, not the writer's familiarity, and it is audience-relative — a term the whole audience shares (SAR to a defense team) carries no translation load, while a term only the writer holds (a coined "detect-then-contextualize", or ViT and VLM stacked in one clause) does. Distinct from #7 (which flags specific AI-vocabulary words) and #30 (which rewrites telegraphic fragments): this one measures jargon density per sentence against the reader's translation load. + +See `voice/references/voice-profile.org` §48 for problem, basis, examples, and history. + +## Process + +1. Read the input text carefully. Confirm the mode (general, prose, or personal) — invocation argument or context. If a file path was given, that file is the deliverable: the final text gets written back to it in step 7. +2. Walk patterns 1-31 and 48 in general mode; add the (prose + personal) patterns in prose mode, plus #47 when the piece is correspondence; walk patterns 1-46 and 48 in personal mode. +3. For each pattern, scan the text. If a match is found, rewrite it according to the pattern's rule. Patterns #39 and #45 emit flags without rewriting. +4. After walking all patterns, ensure the revised text: + - Sounds natural when read aloud + - Varies sentence structure + - Uses specific details over vague claims + - Maintains appropriate tone for the register +5. **Terse pass — mandatory, last rewrite pass (prose + personal modes).** Walk pattern #38 again as a standalone action: read each sentence and try to cut it in half, keeping only the words that change meaning. Run it on its own here, not folded into step 3's walk — it is the most-skipped pattern and the bloat it catches is the first thing a reader notices. General mode skips this step — academic and third-party registers keep their transition markers. +6. **Anti-AI audit — on the final text.** Prompt: "What makes the below so obviously AI generated?" Answer briefly with remaining tells, then revise. This runs *after* the terse pass so the audited text is the text that ships. If the audit triggers rewrites, re-apply the #38 per-sentence test to every changed sentence before proceeding. +7. **Write-back.** If the invocation supplied a file path, write the final text to that file now and say so. The publish flow posts from the file (`git commit -F`, `gh pr create --body-file`), so a final text that lives only in chat is a drift bug waiting to post the un-voiced version. +8. **Attestation block (prose + personal modes).** The high-recurrence patterns — the ones with a documented failure history — each get one explicit line: pattern, checked, match or no match, action taken. Current high-recurrence set: **#13 (em-dash), #37 (fragments), #38 (terse), #40 (praise asymmetry), #42 (finding stems), #46 (comma budget)**. This is a receipt, not a summary: a pattern with no match still gets its line. When a pattern in this set fails in the wild despite the receipt, escalate it the way #38 was escalated; when one holds clean for a long stretch, it can rotate out. +9. Present the final version per the Output Format below. + +## Output Format + +### Compact (default for personal-mode artifacts under ~25 lines: commit messages, review summaries + pins, short PR bodies) + +1. **Final text** — exactly what will be posted, nothing else above it +2. **Pattern-39 warnings** — WARN lines, if any +3. **Fired** — one line listing patterns that fired (e.g., "fired: #13, #38, #42") +4. **Attestation** — the high-recurrence receipt block (one line per pattern) +5. **Write-back note** — "written back to <path>" when a file path was supplied + +The compact format exists because the deliverable must not drown in ceremony: for a three-sentence review comment, printing a draft rewrite, a final rewrite, and an audit Q&A buries the text the gate needs to read, and a long scaffold is easy to print without doing the work. The attestation block is the anti-performative check — it names the patterns that actually fail in practice. + +### Full (long-form prose, general mode, or on request) + +1. **Draft rewrite** — the text after walking all applicable patterns +2. **Mode used** — general, prose, or personal +3. **What makes the below so obviously AI generated?** — brief bullets +4. **Final rewrite** — revised after the anti-AI audit (and post-audit terse re-check) +5. **Pattern-39 warnings** (personal mode only) +6. **Summary of changes** — list of patterns that fired +7. **Attestation block** (prose + personal) +8. **Write-back note** — when a file path was supplied + +## Full Example — Personal Mode (compact output) + +**Before** (a commit message draft from a model, supplied at `/tmp/commit-empty-guard.md`): +> Adds new validator for empty input handling. +> +> The previous validation logic in `~/code/rulesets/voice/SKILL.md` allowed empty strings to leak through to the database layer; this change introduces a guard at the API boundary that returns a 400 response when the input field is empty or contains only whitespace, ensuring data integrity is maintained throughout the request lifecycle. It is worth noting that this does not affect the upstream caller, and we are excited about the productivity gains this will provide. +> +> Furthermore, the new pattern leverages our existing validation framework to deliver a robust, scalable, and maintainable solution. + +**Output:** + +**Final text:** +> feat(api): add empty-input guard at the validator boundary +> +> I added a guard that rejects empty or whitespace-only input fields at the API boundary with a 400 response. The previous flow let empty values leak through to the database layer. +> +> The change doesn't affect the upstream caller. The new pattern reuses the existing validation framework. + +**Pattern-39 warnings:** +- WARN: original draft line 3: `~/code/rulesets/voice/SKILL.md` — local absolute path. State the underlying validation behavior instead. + +**Fired:** #3 (-ing analysis), #4 (promotional), #10 (rule of three), #29 (cliché: "leverages"), #32 (first-person), #33 (semicolon), #34 (contractions), #36 (felt-experience), #38 (terse: "It is worth noting") + +**Attestation:** +- #13 em-dash: checked — no match +- #37 fragments: checked — no match +- #38 terse: checked — fired ("It is worth noting that" cut; final per-sentence pass ran) +- #40 praise asymmetry: checked — not applicable (commit, not review) +- #42 finding stems: checked — not applicable (commit, not review finding) + +**Write-back:** written back to /tmp/commit-empty-guard.md + +## Full Example — General Mode + +**Before** (a paragraph from a research note): +> The Statistical Institute of Catalonia was officially established in 1989, marking a pivotal moment in the evolution of regional statistics in Spain. This initiative was part of a broader movement across Spain to decentralize administrative functions and enhance regional governance. The institute serves as a vital component of the regional statistical landscape, showcasing how decentralized data collection can foster more accurate regional reporting. + +**Mode used:** general + +**Draft rewrite:** +> The Statistical Institute of Catalonia was established in 1989 to collect and publish regional statistics independently from Spain's national statistics office. It was part of a wider movement to decentralize administrative functions across Spain. Decentralized data collection produces more accurate regional reporting than the previous centralized approach. + +**What makes the below so obviously AI generated?** +- The phrase "more accurate regional reporting than the previous centralized approach" still has a slight summary-paragraph rhythm. A human writer might just say what changed and stop. + +**Final rewrite:** +> The Statistical Institute of Catalonia was established in 1989 to collect and publish regional statistics independently from Spain's national statistics office. It was part of a wider movement to decentralize administrative functions across Spain. Regional reporting got more accurate after the change. + +**Summary of changes:** +- Patterns that fired in general mode: #1 (significance inflation: "pivotal moment", "evolution of"), #4 (promotional: "vital component"), #3 (-ing analysis: "showcasing how... can foster"), #8 (copula avoidance: "serves as"), #26 (long-word: removed Latinate constructions). General mode skipped patterns #32-45 (Craig's-voice patterns — prose and personal modes only). + +## Reference + +This skill draws from: +- [Wikipedia: Signs of AI writing](https://en.wikipedia.org/wiki/Wikipedia:Signs_of_AI_writing) — patterns #1-25, maintained by WikiProject AI Cleanup. +- Strunk & White, *The Elements of Style* — patterns #28 (comma splices), #26 (omit needless words, supplemented by humanizer pattern #22). +- Orwell, *Politics and the English Language* — patterns #26 (short over long), #27 (active over passive), #29 (cliché). +- Plain English Campaign — pattern #26 (Plain English wordlist). +- Garner, *Modern English Usage* — pattern #26 (word-pair preferences). +- Craig's voice rules from `claude-rules/commits.md` (Voice and Focus section) — patterns #32-42, split across prose mode (his authored prose and email) and personal mode (publish artifacts). +- Craig's directive, 2026-07-20 (archsetup session) — pattern #46 (comma budget), personal mode. +- Craig's edit of a Signal reply, 2026-07-23 (home session) — pattern #47 (recipient-priority ordering) and the #43 topic-vs-angle calibration, prose/correspondence. +- Craig's edit of a customer-partner email, 2026-07-23 (work session) — pattern #48 (term-translation density), general. +- Corpus measurement (2026-05-29 phases 1-2, documented in the profile) — patterns #43-45 and the calibration notes on #7, #13, #33. + +Key insight (Wikipedia, paraphrased): LLMs use statistical algorithms to guess what should come next. The result tends toward the most statistically likely text that applies to the widest variety of cases. Patterns #1-25 detect that signature. + +Key insight (Orwell): "If it is possible to cut a word out, always cut it out." Patterns #22, #23, #26, #38 act on this rule at increasing levels of aggressiveness depending on register. + +Key insight (2026-06-10): the patterns that fail in practice aren't missing rules — they're present rules walked without receipts. The attestation block exists because "walk all 45" is a prose instruction, and prose instructions about diligence don't survive contact; named per-pattern receipts do. diff --git a/working/voice-term-density/profile.diff b/working/voice-term-density/profile.diff new file mode 100644 index 0000000..3134ebf --- /dev/null +++ b/working/voice-term-density/profile.diff @@ -0,0 +1,36 @@ +--- voice/references/voice-profile.org 2026-07-23 23:24:57.119185940 -0500 ++++ /tmp/profile.proposed 2026-07-23 23:29:48.169642034 -0500 +@@ -1611,3 +1611,33 @@ + + *** History + - 2026-07-23: added from the home session drafting a Signal reply. The first handoff proposed two new patterns and flagged a conflict with §43; the superseding design resolved that the conflict was a misreading of §43 (angle = topic), leaving one genuinely new pattern here and a calibration to §43. Prose/correspondence-scoped per Craig — email and Signal are prose, not publish artifacts. ++ ++** §48 Term-Translation Density ++ ++*** Modes ++General mode, so it runs in all three (general, prose, personal). It's a universal clarity rule in the Orwell / Plain English family, not a Craig-voice trait, and it reads to anyone editing any prose. The later number is an artifact of when it was added, not a scope signal. ++ ++*** Rule ++A sentence that forces the reader to stop and translate more than one specialized term (an acronym, a coined phrase, a product name) is too dense. One is fine; two or more in one sentence means rewrite: split the sentence, gloss one term in a parenthetical, or drop to plain language. The test is the reader's parse, not the writer's familiarity, and it is audience-relative. ++ ++*** Problem ++A writer fluent in the domain doesn't feel the translation cost of the terms, so a sentence stacking three of them reads as normal to the author and stalls the reader on every clause. Density is the metric, not any single word: two coined terms in one sentence is worse than a paragraph that introduces the same two one at a time. Audience-relative, because SAR to a defense team is shared vocabulary carrying no load, while a coined phrase only the writer holds carries full load for everyone else. ++ ++*** Basis ++Craig's edit of a customer-partner email, 2026-07-23 (work session), where one sentence stacked three terms and he flagged it. Distinct from #7 (specific AI-vocabulary words) and #30 (telegraphic fragments); this measures jargon density per sentence. ++ ++*** Before (one sentence, three terms the reader must translate) ++#+begin_example ++A ViT detector gating a VLM for enrichment is close to our own detect-then-contextualize direction. ++#+end_example ++ ++*** After (split, glossed, plain) ++#+begin_example ++Their setup is a fast detector that hands off to a heavier model for a closer read. That mirrors our own two-stage approach (find it first, then work out what it is). ++#+end_example ++ ++*** Detection ++Count the specialized terms in each sentence that a member of the intended audience would have to stop and translate. Two or more is the trigger. Acronyms, coined phrases, and product names count; shared-vocabulary terms for that audience don't. ++ ++*** History ++- 2026-07-23: proposed by Craig from the work session, drafting a customer-partner email. Placed in general mode (universal clarity rule); numbered #48, after the prose-only #47. diff --git a/working/voice-term-density/proposal-from-work.org b/working/voice-term-density/proposal-from-work.org new file mode 100644 index 0000000..d19e5ca --- /dev/null +++ b/working/voice-term-density/proposal-from-work.org @@ -0,0 +1,42 @@ +#+TITLE: New voice pattern proposed by Craig (2026-07-23, from the wo +#+SOURCE: from work +#+DATE: 2026-07-23 23:26:24 -0500 + +New voice pattern proposed by Craig (2026-07-23, from the work project): term-translation density. + +THE RULE Craig stated: if a reader has to stop and mentally translate more than one specialized term in a single sentence, the sentence should be rewritten. + +Proposed pattern text for SKILL.md (place, number, and mode-tag per your taxonomy — see notes below): + + Term-translation density. A sentence that forces the reader to stop and + translate more than one specialized term (acronym, coined phrase, product + name) is too dense. One such term is fine; two or more in one sentence means + rewrite it: split the sentence, gloss one term inline in a parenthetical, or + drop to plain language. The test is the reader's parse, not the writer's + familiarity. It is audience-relative: terms the whole audience shares (SAR to + a defense team) carry no translation load, but terms only the writer holds (a + coined 'detect-then-contextualize', or ViT and VLM stacked in one clause) do. + +Worked before/after for the profile entry (this is the live case that prompted it): + Before: A ViT detector gating a VLM for enrichment is close to our own + detect-then-contextualize direction. + After: Their setup is a fast detector that hands off to a heavier model for a + closer read. That mirrors our own two-stage approach (find it first, + then work out what it is). + +Placement notes for your call: +- Mode: I'd put it in GENERAL (applies in all three modes), because it's a + universal clarity rule in the Orwell / Plain English family, not a Craig-voice + quirk. It reads to anyone editing any prose. +- Related but distinct from existing patterns: #7 (AI-vocab words) is about + which words; #30 (jargon-fragment) is about fragments. This one is about + jargon DENSITY per sentence and the reader's translation load. Worth a + cross-reference, not a merge. +- Pairing rule: it needs the one-line Rule in SKILL.md AND a profile entry + (problem, basis, before/after, detection). The before/after above is ready. +- The skill just went to 47 patterns (recipient-priority ordering added to + prose). This would be the next number in whatever scheme you're using. + +No rush. It came up drafting a customer-partner email tonight where one sentence +stacked three terms and Craig flagged it. Applying it by hand worked; he wants it +in the regular pass so it's caught automatically. diff --git a/working/voice-term-density/skill.diff b/working/voice-term-density/skill.diff new file mode 100644 index 0000000..56f8d0d --- /dev/null +++ b/working/voice-term-density/skill.diff @@ -0,0 +1,58 @@ +--- voice/SKILL.md 2026-07-23 23:24:04.995899857 -0500 ++++ /tmp/skill.proposed 2026-07-23 23:29:48.167295513 -0500 +@@ -1,7 +1,7 @@ + --- + name: voice + description: | +- Multi-pass prose editor with three modes. General mode (default) edits arbitrary writing — research notes, essays, anyone's README prose — by walking 31 patterns: Wikipedia's Signs of AI Writing patterns plus universal good-writing rules (long-word → short-word, active-over-passive, comma splices, cliché flag, jargon-fragment-in-prose rewrite, corporate-speak nominalizations). Prose mode adds Craig's writing-voice patterns for prose he authors or sends — emails, documents, notes — on top of general (em-dash zero-tolerance, no-emphasis-formatting, contractions, semicolons → periods, sentence-split, felt-experience cut, sentence-fragment rewrite, terse-cut, single-sentence cadence, parenthetical asides, declarative-register marker). Personal mode is for publish artifacts only (commits, PR titles + bodies, PR review comments) and adds the artifact-mechanics patterns on top of prose (first-person rewrite, public-artifact scope flag, praise/correction asymmetry, finding stems, comma budget) plus per-artifact terseness budgets. Prose mode also carries one correspondence-only pattern (recipient-priority ordering) that personal mode skips. Total 47 patterns; one editorial review covers all relevant ones for the chosen mode, with an attestation receipt for the high-recurrence set. Replaces the standalone humanizer skill. Use when editing prose. Do NOT use for code, structured data, or plain bullet lists where fragments are valid. ++ Multi-pass prose editor with three modes. General mode (default) edits arbitrary writing — research notes, essays, anyone's README prose — by walking 32 patterns: Wikipedia's Signs of AI Writing patterns plus universal good-writing rules (long-word → short-word, active-over-passive, comma splices, cliché flag, jargon-fragment-in-prose rewrite, corporate-speak nominalizations, term-translation density). Prose mode adds Craig's writing-voice patterns for prose he authors or sends — emails, documents, notes — on top of general (em-dash zero-tolerance, no-emphasis-formatting, contractions, semicolons → periods, sentence-split, felt-experience cut, sentence-fragment rewrite, terse-cut, single-sentence cadence, parenthetical asides, declarative-register marker). Personal mode is for publish artifacts only (commits, PR titles + bodies, PR review comments) and adds the artifact-mechanics patterns on top of prose (first-person rewrite, public-artifact scope flag, praise/correction asymmetry, finding stems, comma budget) plus per-artifact terseness budgets. Prose mode also carries one correspondence-only pattern (recipient-priority ordering) that personal mode skips. Total 48 patterns; one editorial review covers all relevant ones for the chosen mode, with an attestation receipt for the high-recurrence set. Replaces the standalone humanizer skill. Use when editing prose. Do NOT use for code, structured data, or plain bullet lists where fragments are valid. + allowed-tools: + - Read + - Write +@@ -30,9 +30,9 @@ + + Three modes determine which patterns to walk. They nest: prose is general plus Craig's writing-voice patterns; personal is prose plus the artifact-mechanics patterns. + +-- **General** (default) — apply patterns **#1-31**. Use for writing whose author isn't Craig and that isn't a publish artifact: research notes you're editing for someone else, a quoted passage, README prose for a shared project, any third-party text. Output is well-edited human-sounding prose, but does not impose Craig's voice (first-person, contractions, em-dash elimination) — those conflict with academic, literary, or formal registers that aren't his. ++- **General** (default) — apply patterns **#1-31** and **#48** (term-translation density, a universal clarity rule that happens to carry a later number). Use for writing whose author isn't Craig and that isn't a publish artifact: research notes you're editing for someone else, a quoted passage, README prose for a shared project, any third-party text. Output is well-edited human-sounding prose, but does not impose Craig's voice (first-person, contractions, em-dash elimination) — those conflict with academic, literary, or formal registers that aren't his. + - **Prose** — apply **#1-31** plus the patterns tagged **(prose + personal)**: em-dash zero-tolerance (#13), contractions (#34), semicolons → periods (#33), sentence-split (#35), felt-experience cut (#36), sentence-fragment rewrite (#37), terse-cut (#38), no-emphasis-formatting (#41), single-sentence cadence (#43), parenthetical asides (#44), and the declarative-register marker (#45) — plus **#47 (recipient-priority ordering)** when the piece is correspondence (email, Signal, a letter). Use for prose Craig authors or sends in his own voice that isn't a publish artifact: emails, documents he writes or hands to someone, working notes, journal entries. This is the mode that finally applies his actual writing voice to the documents he most wants it on. It skips the artifact-mechanics patterns (#32, #39, #40, #42) — those assume a commit or PR and misfire on free prose (a document is legitimately third-person; a journal has no public-scope concern; praise/correction asymmetry and finding stems are PR-review rules). +-- **Personal** — apply **#1-46**. Use only for publish artifacts: commits, PR titles + bodies, and PR review comments. Adds the artifact-mechanics patterns (#32 first-person rewrite, #39 public-artifact scope flag, #40 praise/correction asymmetry, #42 finding stems, #46 comma budget) on top of everything prose mode walks. #47 is the lone exception to the nesting — it is correspondence-only, and a publish artifact is never a reply to someone's news, so personal mode stops at #46. ++- **Personal** — apply **#1-46** and **#48**. Use only for publish artifacts: commits, PR titles + bodies, and PR review comments. Adds the artifact-mechanics patterns (#32 first-person rewrite, #39 public-artifact scope flag, #40 praise/correction asymmetry, #42 finding stems, #46 comma budget) on top of everything prose mode walks. #47 is the lone exception to the nesting — it is correspondence-only, and a publish artifact is never a reply to someone's news, so personal mode stops at #46. + + If invoked without a mode argument, default to general. Prose mode is invoked explicitly with `/voice prose` (emails, authored documents). Personal-context callers (`commits.md` publish flow, `respond-to-cj-comments.md`) invoke `/voice personal`. + +@@ -53,7 +53,7 @@ + + When given text to edit: + +-1. **Identify which patterns apply** — Scan for the patterns numbered below. General mode walks #1-31 only. Prose mode adds the patterns tagged **(prose + personal)**, plus #47 when the piece is correspondence. Personal mode adds the **(personal only)** ones instead — patterns #1-46, never #47. ++1. **Identify which patterns apply** — Scan for the patterns numbered below. General mode walks #1-31 and #48. Prose mode adds the patterns tagged **(prose + personal)**, plus #47 when the piece is correspondence. Personal mode adds the **(personal only)** ones instead — patterns #1-46 and #48, never #47. + 2. **Rewrite problematic sections** — Replace each detected pattern with its rewrite. + 3. **Preserve meaning** — Keep the core message intact. + 4. **Maintain voice** — Match the intended tone (formal, casual, technical, academic, literary). +@@ -394,10 +394,16 @@ + + See `voice/references/voice-profile.org` §47 for problem, basis, examples, and history. + ++### 48. Term-Translation Density [general] ++ ++**Rule.** A sentence that forces the reader to stop and translate more than one specialized term (an acronym, a coined phrase, a product name) is too dense. One such term is fine; two or more in one sentence means rewrite it: split the sentence, gloss one term inline in a parenthetical (#44), or drop to plain language. The test is the reader's parse, not the writer's familiarity, and it is audience-relative — a term the whole audience shares (SAR to a defense team) carries no translation load, while a term only the writer holds (a coined "detect-then-contextualize", or ViT and VLM stacked in one clause) does. Distinct from #7 (which flags specific AI-vocabulary words) and #30 (which rewrites telegraphic fragments): this one measures jargon density per sentence against the reader's translation load. ++ ++See `voice/references/voice-profile.org` §48 for problem, basis, examples, and history. ++ + ## Process + + 1. Read the input text carefully. Confirm the mode (general, prose, or personal) — invocation argument or context. If a file path was given, that file is the deliverable: the final text gets written back to it in step 7. +-2. Walk patterns 1-31 in general mode; add the (prose + personal) patterns in prose mode, plus #47 when the piece is correspondence; walk patterns 1-46 in personal mode. ++2. Walk patterns 1-31 and 48 in general mode; add the (prose + personal) patterns in prose mode, plus #47 when the piece is correspondence; walk patterns 1-46 and 48 in personal mode. + 3. For each pattern, scan the text. If a match is found, rewrite it according to the pattern's rule. Patterns #39 and #45 emit flags without rewriting. + 4. After walking all patterns, ensure the revised text: + - Sounds natural when read aloud +@@ -495,6 +501,7 @@ + - Craig's voice rules from `claude-rules/commits.md` (Voice and Focus section) — patterns #32-42, split across prose mode (his authored prose and email) and personal mode (publish artifacts). + - Craig's directive, 2026-07-20 (archsetup session) — pattern #46 (comma budget), personal mode. + - Craig's edit of a Signal reply, 2026-07-23 (home session) — pattern #47 (recipient-priority ordering) and the #43 topic-vs-angle calibration, prose/correspondence. ++- Craig's edit of a customer-partner email, 2026-07-23 (work session) — pattern #48 (term-translation density), general. + - Corpus measurement (2026-05-29 phases 1-2, documented in the profile) — patterns #43-45 and the calibration notes on #7, #13, #33. + + Key insight (Wikipedia, paraphrased): LLMs use statistical algorithms to guess what should come next. The result tends toward the most statistically likely text that applies to the widest variety of cases. Patterns #1-25 detect that signature. diff --git a/working/voice-term-density/voice-profile.org.proposed b/working/voice-term-density/voice-profile.org.proposed new file mode 100644 index 0000000..7f26c03 --- /dev/null +++ b/working/voice-term-density/voice-profile.org.proposed @@ -0,0 +1,1643 @@ +#+TITLE: Voice Profile: canonical source-of-truth for the voice skill +#+DATE: 2026-05-29 +#+SOURCE: rulesets session 2026-05-29 + +* How this combines with SKILL.md (pairing rule) + +This file is the canonical source-of-truth for the voice skill's rationale, evidence, examples, and history. =voice/SKILL.md= holds the thin rule-set: one-line directives per pattern, mode applicability, and a pointer back here. Everything else (Problem, Basis, Before/After, Detection guidance, History) lives in the per-pattern sections below. + +Pairing rule. Every change to =voice/SKILL.md= MUST land alongside the corresponding update in this file. The two are normatively paired. A SKILL.md edit without a profile update is incomplete. A profile update without a SKILL.md edit is fine (rationale or evidence can deepen without changing the rule). + +Pattern numbering in both files matches: SKILL.md's =### N. <Name>= maps to this file's =* §N <Name>= section. Mode tags use the same vocabulary: =general=, =prose=, =personal=. + +When the agent runs =/voice=, it reads SKILL.md for the rules and consults this file for the examples and basis it needs to apply each pattern correctly. + +* Corpus + +** Phase 1 (2026-05-29): git commit bodies + +Git commit bodies authored by Craig Jennings across all repos under =~/code/= and =~/projects/=. After cleanup (subject lines, trailers, URL-only lines, AI-attribution lines, blank-run collapse): + +- 5355 raw commits, 1895 with non-trivial bodies +- 128608 words, 912400 characters +- 33 repos contributing; top sources: archsetup (703), rulesets (621), work (565), archangel (455), home (395) + +One register (deliberate technical prose). The view is useful but narrow on its own. + +** Phase 2 (2026-05-29): email + GitHub PR bodies + PR review comments + +Four sub-corpora added so the rules can be tested across registers. + +- *Personal email* (gmail + cmail, sent-only, body ≥50 words after cleanup): 1139 messages, 283,092 words. +- *Work email* (dmail, same filter): 22 messages, 3910 words. Small sample. +- *PR descriptions* (github.com, author cjennings, body ≥100 chars after cleanup): 9 PRs, 1613 words. Small sample. +- *PR review comments* (github.com, author cjennings, ≥20 words): 3 comments, 256 words. Tiny sample. Public GHE work isn't in this index. + +Signatures, quoted replies, and forwarded blocks stripped before analysis. Stats streamed; no corpus files written to disk. + +** Cross-register findings (the key result of Phase 2) + +The most important Phase 2 result is that *register splits matter*. Phase 1's signal from commit prose does not generalize cleanly to conversational prose. + +| Metric (per 1000 words) | Commits | Personal email | Work email | PR bodies | PR comments | +|--------------------------+---------+----------------+------------+-----------+-------------| +| Em-dash | 3.49 | 0.28 | 2.05 | 0.62 | 0.00 | +| Semicolon | 3.16 | 0.64 | 0.26 | 0.62 | 0.00 | +| Contractions | 3.57 | 38.52 | 28.13 | 17.36 | 50.78 | +| Standalone "I" | 3.85 | 36.91 | 23.79 | 8.68 | 42.97 | +| "we" | 0.22 | 8.18 | 14.83 | 1.24 | 0.00 | +| "I'm" | 0.07 | 6.04 | 3.58 | 1.24 | 7.81 | + +Three observations: + +1. Em-dashes and semicolons are concentrated in commit prose, not conversational prose. The personal-mode rules on those (§13 and §33) hold up under Phase 2, but the basis shifts: the rules mostly enforce what is already true for email and PR comments. Commit prose is the outlier register that needs the rule, not the universal pattern. +2. Contractions invert. Commits suppress contractions; email and PR-review prose use them heavily (38 to 50 per 1000). The Phase 1 contraction rule (§34) is strongly confirmed in the registers where contractions are most expected. +3. The Phase 1 curiosity (I'm/I'll surprisingly rare relative to standalone "I") was a register effect, not a personal preference. In personal email, "I'm" runs 6.04 per 1000 vs standalone I at 36.91 — ratio close to natural English. Commit prose is the outlier where "I am" beats "I'm". + +AI-writing tells stay near zero across all five corpora. "leverage" surfaces 18 times in personal email (0.064 per 1000) — small but the only non-zero hit on the watch-list outside commits. All other watch-words clock 0 to 4 per corpus. + +* Findings against the 41 SKILL.md patterns + +** Strongly confirmed by the corpus + +*Pattern 17 (no emojis).* Zero emojis in corpus. Confirmed. + +*Pattern 7 (AI vocabulary).* "delve" 0. "embark" 0. "navigate the" 0. "in the realm of" 0. "seamless" 0. "moreover" 0. "furthermore" 0. "in conclusion" 0. "additionally" 1. "robust" 1. "leverage" 1. Rule confirmed for 11 of 12 watch-words. (One exception below.) + +*Pattern 22 (filler).* "moreover" / "furthermore" / "additionally" / "in conclusion": all zero or one occurrence. Filler-phrase avoidance confirmed. + +*Pattern 32 (first-person rewrite).* Standalone "I" at 3.85 per 1000 words. Craig writes first-person heavily. This is real, not aspirational. + +*Pattern 34 (contractions).* 459 contractions total (3.57 per 1000). Top hits: =doesn't= (92), =don't= (59), =isn't= (46), =it's= (43), =can't= (40), =that's= (34). Rule confirmed. + +*Pattern 38 (terse cut).* 41.1% of paragraphs are single-sentence. Craig writes terse. Paragraph breaks land after one complete thought even when short. Confirmed indirectly via paragraph structure. + +** Aspirational (corpus contradicts, but the rule is intentional self-discipline) + +*Pattern 13 (em-dash zero-tolerance, personal mode).* Corpus rate: 3.49 em-dashes per 1000 words. Comparable to AI-generated prose. Craig USES em-dashes regularly in commit bodies. The rule overrides his habit, it doesn't reflect it. Suggested rewording: drop the "LLMs use em dashes more than humans" framing; keep the zero-tolerance directive but rationale becomes "Craig's published voice (commit messages going forward, PR bodies, emails) drops em-dashes by choice because it reads cleaner and avoids a common AI tell, regardless of his pre-rule habit." Honest about the source. + +*Pattern 33 (semicolons → period/comma).* Corpus rate: 3.16 semicolons per 1000. Craig uses semicolons regularly. Same shape as #13: rule is self-discipline, not habit-reflection. Suggested rewording: acknowledge the rule overrides habit rather than implying it codifies one. + +These two rules are still valuable. Em-dashes and semicolons both read cleaner when absent from short imperative-leaning prose. But the SKILL.md should say "this is a rule I've decided to follow," not "this is how I already write." + +** Worth challenging + +*Pattern 7 watch-word "comprehensive".* 42 occurrences in corpus (~0.33 per 1000). All other AI-tell watch-words clock near zero. "comprehensive" appears to be genuine vocabulary for Craig in technical contexts ("comprehensive test coverage", "comprehensive audit"). Suggested change: pull "comprehensive" out of the watch-list, or carve out a "watch in clusters, not solo" note that flags only when "comprehensive" co-occurs with other AI-tell words. + +** Worth adding (corpus surfaces traits the rules don't capture) + +*Single-sentence paragraph cadence.* 41.1% of paragraphs are exactly one sentence. This is distinctive. Most prose-style guides advise multi-sentence paragraphs. Suggested addition (prose + personal): a positive pattern noting "a one-sentence paragraph is a finished thought, not a fragment. Break paragraphs after one complete thought when the next thought shifts angle, even if both are short." Anti-rule against "merge short paragraphs into multi-sentence ones." + +*Parenthetical density.* 23.07 opening parens per 1000 words. Heavy parenthetical use covers asides, clarifications, and scope-narrowing in parens. Currently no rule addresses this either way. Could add a positive pattern: "parentheses for asides are part of the voice. Don't strip them in a 'clean prose' pass." + +*Question-mark rarity.* 0.33 per 1000. Craig's prose is declarative. He states things, rarely asks them. Worth noting as a register marker (when /voice personal output has questions, double-check whether they're contextual or AI rhetoric). + +** Out of corpus (commits don't test these, Phase 2 needed) + +- *Pattern 13 in long-form prose.* Commit bodies are short. Email and PR bodies may show different em-dash rates. +- *Pattern 14 (boldface).* Org-mode bold uses =*word*=, not detectable by simple grep. Markdown bold rare in commits. +- *Pattern 16 (title case in headings).* Commits don't carry headings. +- *Pattern 19 (collaborative artifacts).* Not present in commit bodies. +- *Pattern 35 (sentence split on conjunctions).* Average sentence is 18.81 words, median 14, with 28% of sentences 21+ words. Long-sentence rate is moderate. Need to inspect actual sentences to know if they're conjunction-stitched. Defer. +- *Pattern 36 (felt-experience cut).* Commit bodies wouldn't carry felt-experience prose. Email + journal corpus needed. +- *Pattern 37 (sentence fragments).* 9.7% of sentences are 1-5 words. Some are legitimate ("All eight pass."), some may be fragments. Can't tell from word-count alone. Defer to a pass that does syntactic detection. +- *Pattern 39 (public-artifact scope).* The corpus IS the public artifacts. The check is circular. Defer. +- *Pattern 40 (praise vs correction asymmetry).* Not detectable in commit bodies. Email or PR-review corpus needed. + +** Curiosities (resolved by Phase 2) + +- *=I'm=* (9 occurrences) and *=I'll=* (2 occurrences) were surprisingly rare in Phase 1 relative to standalone =I= (495 occurrences). Phase 2 resolved this. Personal email shows I'm at 1710 occurrences (6.04 per 1000), I'll at 865 (3.06), I've at 458 (1.62), I'd at 384 (1.36). The Phase 1 rarity was a register effect, not a personal preference. Commit prose uniquely suppresses contractions; conversational prose runs them at near-natural English rate. + +* Suggested deltas + +*All six deltas landed 2026-06-10* via the voice-skill revision from the work-project session: 1 and 2 are in the §13/§33 rule lines and entries, 3 is the §7 soft-flag, 4-6 became patterns §43-§45. The list is kept as the record of what was proposed on 2026-05-29. + +Six concrete edits to =voice/SKILL.md=, all of which can land independently: + +1. *#13 (Em-Dash).* Drop the "LLMs use em dashes more than humans" framing in the personal-mode section. Restate the zero-tolerance rule as self-discipline ("Craig's published voice drops em-dashes by choice"), not habit-reflection. Cite: corpus rate 3.49/1000, AI-comparable. + +2. *#33 (Semicolons).* Same shape. Restate as self-discipline. Cite: corpus rate 3.16/1000. + +3. *#7 (AI Vocabulary).* Remove "comprehensive" from the watch-list, OR add a note that "comprehensive" alone is acceptable; flag only when it co-occurs with =delve= / =leverage= / =robust= / =seamless= / =moreover= etc. Cite: 42 occurrences, all other watch-words at 0 or 1. + +4. *NEW pattern (prose + personal): "Single-sentence paragraph cadence is a feature."* 41.1% of corpus paragraphs are exactly one sentence. A one-sentence paragraph is a finished thought, not a fragment. The voice pass should not merge short paragraphs into multi-sentence ones. + +5. *NEW pattern (prose + personal): "Parentheses for asides are part of the voice."* 23 opening parens per 1000 words. Heavy parenthetical use is distinctive. Don't strip parenthetical asides in a "clean prose" pass. + +6. *Register marker (advisory, not a rewrite rule): "Declarative is the default."* 0.33 question marks per 1000. Voice personal output that contains rhetorical questions should be checked. They're often AI rhetoric, not Craig's register. + +* What Phase 2 would add + +- Email corpus (gmail + cmail, sent-only, long-form): different register, especially long-form prose flow. +- PR bodies and review comments: longer prose, deliberate register, includes the praise/correction asymmetry test ground. +- Slack messages: casual register, contraction rate, sentence-fragment rate. +- Syntactic detection: distinguish fragments from terse complete sentences for pattern #37. +- Long-form documents (résumé, proposals if any): single register but high prose density. + +* Per-pattern entries + +All patterns are entered below per the pairing rule above, §1 through §45, matching SKILL.md's numbering. + +** §1 Undue Emphasis on Significance, Legacy, and Broader Trends + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Strip statements that puff up importance by claiming an arbitrary aspect represents or contributes to a broader trend. Watch for phrases like "stands as", "serves as", "testament to", "vital role", "pivotal moment", "evolving landscape", "marks a shift", "reflects broader", "setting the stage for", "indelible mark", "deeply rooted". Replace with a concrete fact or cut the sentence entirely. + +*** Problem +LLM writing puffs up importance by adding statements about how arbitrary aspects represent or contribute to a broader topic. Watch-list words: stands/serves as, is a testament/reminder, a vital/significant/crucial/pivotal/key role/moment, underscores/highlights its importance/significance, reflects broader, symbolizing its ongoing/enduring/lasting, contributing to the, setting the stage for, marking/shaping the, represents/marks a shift, key turning point, evolving landscape, focal point, indelible mark, deeply rooted. + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +The Statistical Institute of Catalonia was officially established in 1989, marking a pivotal moment in the evolution of regional statistics in Spain. This initiative was part of a broader movement across Spain to decentralize administrative functions and enhance regional governance. +#+end_example + +*** After +#+begin_example +The Statistical Institute of Catalonia was established in 1989 to collect and publish regional statistics independently from Spain's national statistics office. +#+end_example + +*** History +- Original SKILL.md entry: significance and broader-trend puffery, with watch-list phrases drawn from Wikipedia's AI-writing guide. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §2 Undue Emphasis on Notability and Media Coverage + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Cut notability claims that list sources without giving the substance. Replace "cited in The New York Times, BBC, Financial Times" with the actual argument made in one of them. + +*** Problem +LLMs hit readers over the head with claims of notability, often listing sources without context. Watch-list words: independent coverage, local/regional/national media outlets, written by a leading expert, active social media presence. + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +Her views have been cited in The New York Times, BBC, Financial Times, and The Hindu. She maintains an active social media presence with over 500,000 followers. +#+end_example + +*** After +#+begin_example +In a 2024 New York Times interview, she argued that AI regulation should focus on outcomes rather than methods. +#+end_example + +*** History +- Original SKILL.md entry: notability inflation through bare source lists. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §3 Superficial Analyses with -ing Endings + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Cut tacked-on present-participle phrases (highlighting, underscoring, emphasizing, ensuring, reflecting, symbolizing, contributing to, cultivating, fostering, encompassing, showcasing) that add fake depth without new information. + +*** Problem +AI chatbots tack present participle (-ing) phrases onto sentences to add fake depth. Watch-list words: highlighting, underscoring, emphasizing, ensuring, reflecting, symbolizing, contributing to, cultivating, fostering, encompassing, showcasing. + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +The temple's color palette of blue, green, and gold resonates with the region's natural beauty, symbolizing Texas bluebonnets, the Gulf of Mexico, and the diverse Texan landscapes, reflecting the community's deep connection to the land. +#+end_example + +*** After +#+begin_example +The temple uses blue, green, and gold colors. The architect said these were chosen to reference local bluebonnets and the Gulf coast. +#+end_example + +*** History +- Original SKILL.md entry: -ing phrase tacking for fake analytical depth. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §4 Promotional and Advertisement-like Language + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Remove travel-brochure adjectives (vibrant, breathtaking, nestled, stunning, renowned, must-visit, profound, rich figurative use) and replace promotional framing with concrete facts about the subject. + +*** Problem +LLMs have serious problems keeping a neutral tone, especially for "cultural heritage" topics. Watch-list words: boasts a, vibrant, rich (figurative), profound, enhancing its, showcasing, exemplifies, commitment to, natural beauty, nestled, in the heart of, groundbreaking (figurative), renowned, breathtaking, must-visit, stunning. + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +Nestled within the breathtaking region of Gonder in Ethiopia, Alamata Raya Kobo stands as a vibrant town with a rich cultural heritage and stunning natural beauty. +#+end_example + +*** After +#+begin_example +Alamata Raya Kobo is a town in the Gonder region of Ethiopia, known for its weekly market and 18th-century church. +#+end_example + +*** History +- Original SKILL.md entry: travel-brochure adjective patterns. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §5 Vague Attributions and Weasel Words + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Replace vague attributions (experts say, observers have cited, industry reports, some critics argue, several sources) with a named source plus the specific claim made. + +*** Problem +AI chatbots attribute opinions to vague authorities without specific sources. Watch-list words: Industry reports, Observers have cited, Experts argue, Some critics argue, several sources or publications (when few cited). + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +Due to its unique characteristics, the Haolai River is of interest to researchers and conservationists. Experts believe it plays a crucial role in the regional ecosystem. +#+end_example + +*** After +#+begin_example +The Haolai River supports several endemic fish species, according to a 2019 survey by the Chinese Academy of Sciences. +#+end_example + +*** History +- Original SKILL.md entry: vague-authority attribution patterns. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §6 Outline-like "Challenges and Future Prospects" Sections + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Delete formulaic "Despite its prosperity, X faces challenges" wrap-ups and "Future Outlook" boilerplate. Replace with the actual events that happened or omit the section. + +*** Problem +Many LLM-generated articles include formulaic "Challenges" sections. Watch-list words: "Despite its... faces several challenges...", "Despite these challenges", "Challenges and Legacy", "Future Outlook". + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +Despite its industrial prosperity, Korattur faces challenges typical of urban areas, including traffic congestion and water scarcity. Despite these challenges, with its strategic location and ongoing initiatives, Korattur continues to thrive as an integral part of Chennai's growth. +#+end_example + +*** After +#+begin_example +Traffic congestion increased after 2015 when three new IT parks opened. The municipal corporation began a stormwater drainage project in 2022 to address recurring floods. +#+end_example + +*** History +- Original SKILL.md entry: outline-template "Challenges and Future Prospects" sections. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §7 Overused "AI Vocabulary" Words + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Flag and rewrite around the high-frequency AI vocabulary list. Watch-list words: Additionally, align with, comprehensive, crucial, delve, emphasizing, enduring, enhance, fostering, garner, highlight (verb), interplay, intricate or intricacies, key (adjective), landscape (abstract noun), pivotal, showcase, tapestry (abstract noun), testament, underscore (verb), valuable, vibrant. "comprehensive" is a soft flag because the corpus shows it as genuine Craig vocabulary he chooses to use sparingly. Suggest an alternative ("full", "complete", "thorough", or rewording to drop the adjective) and let Craig decide per instance. + +*** Problem +These words appear far more frequently in post-2023 text. They often co-occur. + +*** Basis +Corpus-measured across registers (2026-05-29). Phase 1 git commits: "comprehensive" 42 occurrences, every other watch-word 0 or 1. Phase 2 conversational and PR corpora: "comprehensive" 1 in personal email, 0 in work email, PR descriptions, and PR review comments. "leverage" 18 in personal email, 0 to 1 elsewhere. Every other watch-word stays at 0 to 4 across all five corpora. + +Two takeaways. First, "comprehensive" is concentrated in commit prose (technical-doc register: "comprehensive test coverage", "comprehensive audit") and almost absent from conversational prose. Craig has chosen to keep it on the watch-list because he is consciously trying to use it sparingly. Second, "leverage" earns a soft watch in personal email even though the rest of the list stays clean. The two together suggest the rule should flag-and-suggest individual hits in technical prose without treating any single watch-word as automatic disqualification. + +*** Before +#+begin_example +Additionally, a distinctive feature of Somali cuisine is the incorporation of camel meat. An enduring testament to Italian colonial influence is the widespread adoption of pasta in the local culinary landscape, showcasing how these dishes have integrated into the traditional diet. +#+end_example + +*** After +#+begin_example +Somali cuisine also includes camel meat, which is considered a delicacy. Pasta dishes, introduced during Italian colonization, remain common, especially in the south. +#+end_example + +*** History +- Original SKILL.md entry: high-frequency post-2023 AI vocabulary list. +- 2026-05-29 (commit =c3cf9a5=): note on "comprehensive" added with corpus measurement and soft-flag guidance. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §8 Avoidance of "is"/"are" (Copula Avoidance) + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Replace elaborate copula substitutes (serves as, stands as, marks, represents, boasts, features, offers) with plain "is" or "has". + +*** Problem +LLMs substitute elaborate constructions for simple copulas. Watch-list words: serves as, stands as, marks, represents (a), boasts, features, offers (a). + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +Gallery 825 serves as LAAA's exhibition space for contemporary art. The gallery features four separate spaces and boasts over 3,000 square feet. +#+end_example + +*** After +#+begin_example +Gallery 825 is LAAA's exhibition space for contemporary art. The gallery has four rooms totaling 3,000 square feet. +#+end_example + +*** History +- Original SKILL.md entry: copula avoidance patterns. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §9 Negative Parallelisms + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Rewrite "not only X but Y" and "it's not just about X, it's Y" constructions as a single direct claim. + +*** Problem +Constructions like "Not only...but..." or "It's not just about..., it's..." are overused as a way to claim depth. + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +It's not just about the beat riding under the vocals; it's part of the aggression and atmosphere. It's not merely a song, it's a statement. +#+end_example + +*** After +#+begin_example +The heavy beat adds to the aggressive tone. +#+end_example + +*** History +- Original SKILL.md entry: negative-parallelism stock phrasing. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §10 Rule of Three Overuse + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Break the reflexive three-item list pattern when the third item is filler. Collapse to one or two specific items. + +*** Problem +LLMs force ideas into groups of three to appear comprehensive. The third item is usually filler chosen to fit the cadence, not because it adds substance. + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +The event features keynote sessions, panel discussions, and networking opportunities. Attendees can expect innovation, inspiration, and industry insights. +#+end_example + +*** After +#+begin_example +The event includes talks and panels. There's also time for informal networking between sessions. +#+end_example + +*** History +- Original SKILL.md entry: rule-of-three cadence overuse. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §11 Elegant Variation (Synonym Cycling) + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Stop cycling synonyms for the same referent across consecutive sentences. Repeat the noun, or merge the sentences. + +*** Problem +AI has repetition-penalty code causing excessive synonym substitution. The protagonist becomes the main character becomes the central figure becomes the hero, all referring to the same person. + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +The protagonist faces many challenges. The main character must overcome obstacles. The central figure eventually triumphs. The hero returns home. +#+end_example + +*** After +#+begin_example +The protagonist faces many challenges but eventually triumphs and returns home. +#+end_example + +*** History +- Original SKILL.md entry: elegant-variation synonym cycling driven by repetition penalty. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §12 False Ranges + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Rewrite "from X to Y" constructions where X and Y are not on the same scale. List the items plainly instead. + +*** Problem +LLMs use "from X to Y" constructions where X and Y are not on a meaningful scale ("from the Big Bang to dark matter") to imply comprehensive sweep. + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +Our journey through the universe has taken us from the singularity of the Big Bang to the grand cosmic web, from the birth and death of stars to the enigmatic dance of dark matter. +#+end_example + +*** After +#+begin_example +The book covers the Big Bang, star formation, and current theories about dark matter. +#+end_example + +*** History +- Original SKILL.md entry: false-range "from X to Y" constructions. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §13 Em Dash Overuse + +*** Modes +General mode: overuse-reduction. +Prose + personal modes: zero-tolerance. + +*** Rule +Replace em-dashes (=—=) with a comma, period, colon, or parentheses, whichever fits. Zero-tolerance in prose and personal modes holds *everywhere in the text*, including inside example blocks, code-fence prose, and quoted material. An em-dash in a quoted line still gets replaced. + +*** Problem +Craig's published voice drops em-dashes by choice: they read cleaner absent from short imperative-leaning prose and their overuse is a common AI tell (LLMs use em dashes more than the median human writer, mimicking "punchy" sales writing). The rule is chosen self-discipline, not a reflection of his pre-rule habit — the corpus shows he used them regularly in commit bodies. + +*** Basis +Phase 1 corpus (git commits, 128k words): 3.49 em-dashes per 1000 words. Comparable to AI-generated prose. Phase 2 corpus reveals a sharp register split: personal email 0.28 per 1000, work email 2.05, PR descriptions 0.62, PR review comments 0.00. Em-dashes are concentrated in commit prose, almost absent from email and PR review prose. The zero-tolerance rule in prose and personal modes mostly enforces what is already true for non-commit registers. The rule still earns its place because commit prose is the high-volume register where the AI-tell em-dash habit shows up. Self-discipline, not habit-reflection, for the commit register specifically. + +*** Before +#+begin_example +The term is primarily promoted by Dutch institutions—not by the people themselves. You don't say "Netherlands, Europe" as an address—yet this mislabeling continues—even in official documents. +#+end_example + +*** After +#+begin_example +The term is primarily promoted by Dutch institutions, not by the people themselves. You don't say "Netherlands, Europe" as an address, yet this mislabeling continues in official documents. +#+end_example + +*** History +- Original SKILL.md entry: rule scoped to general overuse-reduction. +- 2026-05-26 (commit =4fac2a0=): prose mode added, rule strengthened to zero-tolerance in prose and personal. +- 2026-05-29 (commit =c3cf9a5=): Note on basis added with corpus measurement. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. +- 2026-06-10: the self-discipline reframing (a "Suggested deltas" item from 2026-05-29, never applied) moved from the findings section into the entry proper and into the SKILL.md rule line. Craig's call, from the work-project session. + +** §14 Overuse of Boldface + +*** Modes +General mode only. Prose and personal inherit it. Pattern §41 is the related Craig-voice rule covering emphasis-by-formatting in his authored prose. + +*** Rule +Strip mechanical boldface used to call out terms, acronyms, or phrases in running prose. Bold survives only for structural emphasis the document genuinely needs. + +*** Problem +AI chatbots emphasize phrases in boldface mechanically. Acronyms, names, and key terms get wrapped in bold even when the surrounding sentence already gives them stress. + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +It blends **OKRs (Objectives and Key Results)**, **KPIs (Key Performance Indicators)**, and visual strategy tools such as the **Business Model Canvas (BMC)** and **Balanced Scorecard (BSC)**. +#+end_example + +*** After +#+begin_example +It blends OKRs, KPIs, and visual strategy tools like the Business Model Canvas and Balanced Scorecard. +#+end_example + +*** History +- Original SKILL.md entry: mechanical boldface around terms and acronyms. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §15 Inline-Header Vertical Lists + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Collapse bullet lists whose items start with a bold header plus colon into running prose, unless the list structure is genuinely the right shape. + +*** Problem +AI outputs lists where items start with bolded headers followed by colons, often when a paragraph would carry the same content more naturally. + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +- **User Experience:** The user experience has been significantly improved with a new interface. +- **Performance:** Performance has been enhanced through optimized algorithms. +- **Security:** Security has been strengthened with end-to-end encryption. +#+end_example + +*** After +#+begin_example +The update improves the interface, speeds up load times through optimized algorithms, and adds end-to-end encryption. +#+end_example + +*** History +- Original SKILL.md entry: inline-header vertical list pattern. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §16 Title Case in Headings + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Lowercase headings that are reflexively title-cased. Sentence case is the default unless the project's house style is title case. + +*** Problem +AI chatbots capitalize all main words in headings even when the surrounding document uses sentence case. + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +## Strategic Negotiations And Global Partnerships +#+end_example + +*** After +#+begin_example +## Strategic negotiations and global partnerships +#+end_example + +*** History +- Original SKILL.md entry: reflexive title-case in headings. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §17 Emojis + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Remove decorative emojis from headings, bullets, and prose unless the document is a register where emoji is genuinely intended. + +*** Problem +AI chatbots often decorate headings or bullet points with emojis to add visual structure that the prose itself does not need. + +*** Basis +Corpus-measured: 2026-05-29 commit corpus shows zero emojis. The rule reflects established practice. + +*** Before +#+begin_example +🚀 **Launch Phase:** The product launches in Q3 +💡 **Key Insight:** Users prefer simplicity +✅ **Next Steps:** Schedule follow-up meeting +#+end_example + +*** After +#+begin_example +The product launches in Q3. User research showed a preference for simplicity. Next step: schedule a follow-up meeting. +#+end_example + +*** History +- Original SKILL.md entry: decorative emoji in headings and bullets. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §18 Curly Quotation Marks + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Convert curly quotation marks to straight ASCII quotes. + +*** Problem +ChatGPT uses curly quotes instead of straight quotes, which is a recognizable tell in technical and source-controlled writing. + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +He said “the project is on track” but others disagreed. +#+end_example + +*** After +#+begin_example +He said "the project is on track" but others disagreed. +#+end_example + +*** History +- Original SKILL.md entry: curly-quote substitution. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §19 Collaborative Communication Artifacts + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Strip chatbot correspondence framing ("I hope this helps", "Let me know if...", "Here is an overview of...", "Certainly!", "Of course!", "Would you like...") that leaked into the body. + +*** Problem +Text meant as chatbot correspondence gets pasted as content, carrying the assistant's framing into a document that should stand alone. Watch-list words: I hope this helps, Of course!, Certainly!, You're absolutely right!, Would you like..., let me know, here is a... + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +Here is an overview of the French Revolution. I hope this helps! Let me know if you'd like me to expand on any section. +#+end_example + +*** After +#+begin_example +The French Revolution began in 1789 when financial crisis and food shortages led to widespread unrest. +#+end_example + +*** History +- Original SKILL.md entry: collaborative-communication artifacts pasted as content. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §20 Knowledge-Cutoff Disclaimers + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Remove training-cutoff hedges ("as of my last update", "while specific details are scarce", "based on available information") and either commit to a fact or omit the claim. + +*** Problem +AI disclaimers about incomplete information get left in text, signaling the model's uncertainty rather than the author's. Watch-list words: as of [date], Up to my last training update, While specific details are limited or scarce, based on available information. + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +While specific details about the company's founding are not extensively documented in readily available sources, it appears to have been established sometime in the 1990s. +#+end_example + +*** After +#+begin_example +The company was founded in 1994, according to its registration documents. +#+end_example + +*** History +- Original SKILL.md entry: knowledge-cutoff hedging. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §21 Sycophantic/Servile Tone + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Cut servile opener phrases ("Great question!", "You're absolutely right", "That's an excellent point") and proceed straight to the substance. + +*** Problem +Overly positive, people-pleasing language reads as performance rather than communication and signals AI assistant register. + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +Great question! You're absolutely right that this is a complex topic. That's an excellent point about the economic factors. +#+end_example + +*** After +#+begin_example +The economic factors you mentioned are relevant here. +#+end_example + +*** History +- Original SKILL.md entry: sycophantic and servile opener patterns. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §22 Filler Phrases + +*** Modes +General mode only. Prose and personal inherit it. Pattern §38 is the stricter cousin for Craig's authored prose. + +*** Rule +Compress wordy filler to plain equivalents: "in order to" to "to", "due to the fact that" to "because", "at this point in time" to "now", "in the event that" to "if", "has the ability to" to "can", "it is important to note that" to nothing, "for the purpose of" to "to", "in spite of the fact that" to "although", "a great deal of" to "much", "at this juncture" to "now". + +*** Problem +Wordy filler stretches a sentence without adding precision. Cutting it shortens the prose and sharpens the claim. + +*** Basis +Corpus-measured: 2026-05-29 commit corpus shows "moreover", "furthermore", "additionally", "in conclusion" all at zero or one occurrence. Filler-phrase avoidance confirmed at the watch-list level. + +*** Before +#+begin_example +In order to achieve this goal, we need to allocate resources due to the fact that the team has the ability to deliver. At this point in time, it is important to note that the data shows progress. +#+end_example + +*** After +#+begin_example +To achieve this, we need to allocate resources because the team can deliver. The data shows progress. +#+end_example + +*** Detection +The original SKILL.md entry uses a Before to After substitution table: +- "In order to achieve this goal" to "To achieve this" +- "Due to the fact that it was raining" to "Because it was raining" +- "At this point in time" to "Now" +- "In the event that you need help" to "If you need help" +- "The system has the ability to process" to "The system can process" +- "It is important to note that the data shows" to "The data shows" +- "For the purpose of" to "To" +- "In spite of the fact that" to "Although" +- "A great deal of" to "Much" +- "At this juncture" to "Now" + +*** History +- Original SKILL.md entry: wordy filler phrase substitutions. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §23 Excessive Hedging + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Strip stacked hedges ("could potentially possibly", "might have some effect") down to a single appropriate qualifier. + +*** Problem +Over-qualifying statements weakens them without adding accuracy. One hedge does the job that three do. + +*** Basis +Observation-derived (Strunk and White, Garner). + +*** Before +#+begin_example +It could potentially possibly be argued that the policy might have some effect on outcomes. +#+end_example + +*** After +#+begin_example +The policy may affect outcomes. +#+end_example + +*** History +- Original SKILL.md entry: stacked hedge reduction. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §24 Generic Positive Conclusions + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Replace vague upbeat endings ("the future looks bright", "exciting times lie ahead", "a step in the right direction") with a concrete fact or cut the closer entirely. + +*** Problem +Vague upbeat endings give the document the shape of a press release without making any specific claim. + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +The future looks bright for the company. Exciting times lie ahead as they continue their journey toward excellence. This represents a major step in the right direction. +#+end_example + +*** After +#+begin_example +The company plans to open two more locations next year. +#+end_example + +*** History +- Original SKILL.md entry: generic positive conclusion boilerplate. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §25 Hyphenated Word Pair Overuse + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Drop reflexive hyphens from common modifier pairs (third-party, cross-functional, client-facing, data-driven, decision-making, well-known, high-quality, real-time, long-term, end-to-end) where humans hyphenate inconsistently. Less common or genuinely technical compound modifiers can keep their hyphens. + +*** Problem +AI hyphenates common word pairs with perfect consistency. Humans rarely hyphenate these uniformly, and when they do, it is inconsistent. The uniformity itself is the tell. + +*** Basis +Observation-derived (Wikipedia "Signs of AI Writing"). + +*** Before +#+begin_example +The cross-functional team delivered a high-quality, data-driven report on our client-facing tools. Their decision-making process was well-known for being thorough and detail-oriented. +#+end_example + +*** After +#+begin_example +The cross functional team delivered a high quality, data driven report on our client facing tools. Their decision making process was known for being thorough and detail oriented. +#+end_example + +*** History +- Original SKILL.md entry: hyphenated common-modifier overuse. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §26 Long Word → Short Word + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Swap long Latinate words for their short Anglo-Saxon equivalents per the Plain English wordlist: utilize to use, commence to start or begin, terminate to end, facilitate to help, demonstrate to show, sufficient to enough, prior to to before, subsequent to to after, approximately to about, endeavor to try, ascertain to find out, assistance to help, obtain to get, modification to change, implement to carry out, optimal to best, regarding to about, methodology to method, "in the event of" to "if". + +*** Problem +Long Latinate words signal effortful writing without adding precision. Anglo-Saxon roots are shorter and clearer. + +*** Basis +Observation-derived (Strunk and White, Orwell, Plain English Campaign, Garner). + +*** Before +#+begin_example +The system will utilize advanced algorithms to facilitate optimal performance. Prior to deployment, we must ascertain that the methodology is sufficient. +#+end_example + +*** After +#+begin_example +The system uses algorithms to get the best performance. Before deployment, we must check that the method works. +#+end_example + +*** History +- Original SKILL.md entry: Plain English wordlist substitutions. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §27 Active Over Passive Voice + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Rewrite passive constructions to active when the actor is recoverable from context. Flag rather than auto-rewrite when the actor genuinely does not matter, because passive is sometimes the right choice in technical contexts. + +*** Problem +Passive voice hides who did what. Active voice is shorter and clearer in most cases. Skip when the actor genuinely does not matter (technical writing about an inanimate process: "the table was created in 2024" can stay passive). + +*** Basis +Observation-derived (Strunk and White, Orwell). + +*** Before +#+begin_example +The migration was run by the deployment script. The bug was introduced in commit abc123. The fix was applied by the team. +#+end_example + +*** After +#+begin_example +The deployment script ran the migration. Commit abc123 introduced the bug. The team applied the fix. +#+end_example + +*** Detection +"to be" plus past-participle patterns where the actor is recoverable from context. + +*** History +- Original SKILL.md entry: active-over-passive with suggestion-only treatment in v1. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §28 Comma Splices + +*** Modes +General mode only. Prose and personal inherit it. The semicolon escape route is blocked in personal mode by §33. + +*** Rule +Split two independent clauses joined only by a comma into two sentences or join them with a conjunction. In general mode a semicolon is an acceptable repair. In personal mode the semicolon is itself a target (§33), so prefer the period. + +*** Problem +Comma splices read as run-ons. Either split into two sentences, join with a conjunction, or use a semicolon (in personal mode this becomes a period). + +*** Basis +Observation-derived (Strunk and White). + +*** Before +#+begin_example +The build failed, the test suite reported three errors. +#+end_example + +*** After +#+begin_example +The build failed. The test suite reported three errors. +#+end_example + +*** Detection +Two independent clauses joined only by a comma. + +*** History +- Original SKILL.md entry: comma-splice repair. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §29 Cliché Flag + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Replace business and conversational clichés with the plain meaning, including in casual register where "it's fine, it's casual" is the tell. Watch-list phrases: at the end of the day, moving forward, going forward, at this juncture, circle back, low-hanging fruit, deep dive, leverage (as verb), synergy, take it offline, ducks in a row, boil the ocean, pivot (corporate sense), keep it loose, keep it casual, touch base, circle up, hit the ground running, move the needle, on the same page, no-brainer, win-win. + +*** Problem +Clichés signal effortful prose without saying anything specific. Replace with the actual meaning. A casual, friendly, or conversational register is not a license to keep a cliché. Cut it there too. If you catch yourself justifying one as "it's fine, it's casual," that is the tell. Craig flagged this on 2026-05-22 when "keep it loose" slipped through as "acceptable casual." That is exactly the miss this note prevents. + +*** Basis +Observation-derived (Orwell, Garner). + +*** Before +#+begin_example +At the end of the day, we need to leverage our core competencies and circle back on the low-hanging fruit. +#+end_example + +*** After +#+begin_example +We need to use what we already do well and start with the easiest improvements first. +#+end_example + +*** History +- Original SKILL.md entry: business and conversational cliché list. +- 2026-05-22: Craig added "keep it loose" / "keep it casual" after a miss in earlier output. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §30 Jargon-Fragment → Complete Sentence + +*** Modes +General mode only. Prose and personal inherit it. Pattern §37 is the stricter cousin for Craig's authored prose. + +*** Rule +Rewrite telegraphic sentence fragments inside prose paragraphs as complete sentences with subject and verb. Headings and bullet items are exempt because fragments are valid there. + +*** Problem +Telegraphic fragments in prose paragraphs read as bullet-style notes leaking into running text. They lose the connective tissue a complete sentence carries. + +*** Basis +Observation-derived (Strunk and White). + +*** Before +#+begin_example +The new function handles edge cases. Empty input throws. Whitespace gets trimmed. Returns null on no match. +#+end_example + +*** After +#+begin_example +The new function handles edge cases. It throws on empty input, trims whitespace, and returns null when no match is found. +#+end_example + +*** Detection +Sentence-like fragments inside prose paragraphs that read as bullet-list shorthand. Headings and bullet items are exempt because fragments are valid there. + +*** History +- Original SKILL.md entry: jargon-fragment rewrite for prose paragraphs. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §31 Noun-ified Verbs + +*** Modes +General mode only. Prose and personal inherit it. + +*** Rule +Replace corporate-speak noun-ifications with the real noun: "the ask" to "the request", "a learn" to "the lesson", "the spend" to "the budget", "a build" to "the system" or "the prototype", "the reveal" to "the announcement", "the lift" to "the effort", "the get" to "the result". Philosophical nominalizations ("the becoming", "the unfolding") are not targets. + +*** Problem +Corporate-speak nominalization reads as performance. The real nouns are shorter and clearer. Watch-list: the ask, a learn, the spend, a build, the reveal, a do, the lift, the get, the say. + +*** Basis +Observation-derived (Garner; Craig's voice rules in claude-rules/commits.md). + +*** Before +#+begin_example +The ask was for a quick build. After the reveal, we'll do a learn. +#+end_example + +*** After +#+begin_example +The request was for a quick prototype. After the announcement, we'll review what worked. +#+end_example + +*** History +- Original SKILL.md entry: corporate-speak nominalization. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §32 First-Person Voice Rewrite + +*** Modes +Personal mode only. General and prose skip it because a research note, a document, or anyone else's text is legitimately third-person. + +*** Rule +Rewrite impersonal third-person publish-artifact bodies into first person ("I added X", "I missed Y", "I kept Z because..."). The commit subject line stays imperative per Conventional Commits ("feat: add support for X"). The body shifts to first person. Skip the rewrite for mechanical changes (a chore version bump, a typo fix) where the subject alone carries the message. + +*** Problem +Impersonal third-person ("Add support for X", "The change adds Y") reads as press-release voice in a commit body or PR description. First-person ("I added X", "I kept Y because...") sounds like one engineer talking to another. + +*** Basis +Corpus-measured across registers (2026-05-29): standalone "I" runs 3.85 per 1000 words in git commits, 36.91 in personal email, 23.79 in work email, 8.68 in PR descriptions, 42.97 in PR review comments. First-person density is roughly 10x higher in conversational registers than in commits. Craig writes first-person heavily across the board, but commit prose under-uses "I" relative to natural English. The rule strengthens the under-using register without overreaching: it asks the publish-artifact body to write the way the email body already does. + +*** Before +#+begin_example +Adds the new validation step before saving. The previous flow allowed empty values to leak into the database. This change blocks them at the API boundary. +#+end_example + +*** After +#+begin_example +I added a validation step before saving. The previous flow let empty values leak into the database. I'm blocking them at the API boundary now. +#+end_example + +*** Detection +Impersonal third-person construction in a publish-artifact body where first-person fits naturally. + +*** History +- Original SKILL.md entry: first-person rewrite for publish artifacts. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §33 Semicolon → Period or Comma + +*** Modes +Prose and personal modes. General mode keeps semicolons because academic and literary registers use them legitimately. + +*** Rule +Replace semicolons with a period (split into two sentences) or a comma (when the clauses are tightly coupled) in Craig's authored prose: emails, documents, working notes, commit-message bodies, PR descriptions, PR review comments. A formal long-form document can keep the semicolon, but the default is to split. Chosen self-discipline, not habit-reflection. + +*** Problem +Craig's published voice drops semicolons by choice. They make the writing feel unnecessarily literary, the period-split usually reads better, and dropping them removes one common AI tell. The rule overrides his pre-rule habit rather than codifying one — the corpus shows he used semicolons regularly in commit prose. + +*** Basis +Corpus-measured across registers (2026-05-29): semicolons run 3.16 per 1000 words in git commits, 0.64 in personal email, 0.26 in work email, 0.62 in PR descriptions, 0.00 in PR review comments. Same register split as em-dashes (§13). Semicolons are concentrated in commit prose; conversational prose almost never uses them. The rule mostly enforces what is already true for non-commit registers. It earns its place because commit prose is the register where Craig's habit and the AI-tell pattern overlap. + +*** Before +#+begin_example +I added the validation; the previous flow allowed empty values to leak through. +#+end_example + +*** After +#+begin_example +I added the validation. The previous flow allowed empty values to leak through. +#+end_example + +*** Detection +Semicolons in prose Craig authors: emails, documents, working notes, commit-message bodies, PR descriptions, PR review comments. + +*** History +- Original SKILL.md entry: semicolon to period or comma in Craig's authored prose. +- 2026-05-29 (commit =c3cf9a5=): basis note added with corpus measurement reframing the rule as self-discipline. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. +- 2026-06-10: the self-discipline reframing (a "Suggested deltas" item from 2026-05-29, never applied) moved from the findings section into the entry proper and into the SKILL.md rule line. Craig's call, from the work-project session. + +** §34 Contractions + +*** Modes +Prose and personal modes. General mode skips because academic, literary, and formal registers often prefer uncontracted forms. + +*** Rule +Prefer contractions in Craig's prose (it's, that's, don't, we're, I'd, won't) unless a negation or emphasis genuinely needs the uncontracted weight. + +*** Problem +Uncontracted English reads stiff in a short prose body unless a negation or emphasis needs the weight. Prefer contractions in his prose: emails, documents, commit and PR bodies. + +*** Basis +Corpus-measured across registers (2026-05-29). Contraction rate per 1000 words: git commits 3.57, personal email 38.52, work email 28.13, PR descriptions 17.36, PR review comments 50.78. Commit prose is the outlier register that suppresses contractions; conversational and PR-review prose use them heavily, near the natural-English rate. The Phase 1 curiosity (I'm 9 occurrences vs standalone I at 495 in commits) was a register effect, not a personal preference. Personal email runs I'm at 6.04 per 1000 vs standalone I at 36.91, ratio close to natural English. Top contractions in personal email: i'm 1710, it's 928, i'll 865, don't 632, you're 567, i've 458, that's 433, i'd 384, we're 307, didn't 299. The rule confirms across the board, with the strongest evidence from the conversational registers where contractions are most expected. + +*** Before +#+begin_example +It is worth noting that the change does not break the existing flow. We are confident that this is the right approach. +#+end_example + +*** After +#+begin_example +It's worth noting the change doesn't break the existing flow. We're confident this is the right approach. +#+end_example + +*** Detection +Uncontracted forms in publish-artifact prose where the contraction reads more naturally. Note pattern §38 catches "worth noting" as rhetorical padding. The example above shows isolated transformation. In practice both passes apply. + +*** History +- Original SKILL.md entry: contractions preferred in Craig's authored prose. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §35 Sentence Split on Conjunctions + +*** Modes +Prose and personal modes. General mode skips because academic and literary registers use long compound sentences deliberately. + +*** Rule +Split sentences that stack three or four clauses joined by "so", "and", "but" into two or three shorter sentences when the split does not lose meaning. + +*** Problem +Long compound sentences read easier as two or three shorter ones in a prose or publish-artifact body. Skip in academic or literary prose where deliberate long sentences are the register. + +*** Basis +Observation-derived (Craig's voice rules in claude-rules/commits.md). Corpus context: average sentence is 18.81 words, median 14, with 28% of sentences at 21+ words. Long-sentence rate is moderate. Inspection of actual sentences for conjunction-stitching is deferred to Phase 2. + +*** Before +#+begin_example +I added the validation step before saving so empty values get blocked at the API boundary, and I also added a regression test that exercises the empty-string case, but I did not change the upstream caller because that's a separate concern. +#+end_example + +*** After +#+begin_example +I added the validation step before saving so empty values get blocked at the API boundary. I added a regression test that exercises the empty-string case. I didn't change the upstream caller because that's a separate concern. +#+end_example + +*** Detection +Sentences that stack three or four clauses with commas and conjunctions ("so", "and", "but") where splitting on a conjunction would not lose meaning. + +*** History +- Original SKILL.md entry: sentence split on conjunctions for Craig's authored prose. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §36 Felt-Experience Narration + +*** Modes +Prose and personal modes. General mode skips because third-party prose is legitimately allowed to describe how something feels. + +*** Rule +Cut phrases that tell the reader how the change will feel or how often the writer will use it ("I'll feel this every time I commit", "this will be a relief", "I'm excited about", "this is going to be huge"). State what changed and let the reader decide what to do with it. + +*** Problem +Felt-experience phrases read as performance, not communication. They tell the reader how the writer wants them to receive the change rather than describing the change. + +*** Basis +Observation-derived (Craig's voice rules in claude-rules/commits.md). Commit-body corpus would not carry felt-experience prose; email and journal corpus deferred to Phase 2. + +*** Before +#+begin_example +I'm so excited about this — I'll feel the speedup every time I run the build. This is going to be a huge relief. +#+end_example + +*** After +#+begin_example +The build now finishes in roughly half the time it used to take. +#+end_example + +*** Detection +Phrases that tell the reader how the change will feel or how often the writer will use it. + +*** History +- Original SKILL.md entry: felt-experience narration cut for Craig's authored prose. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §37 Sentence Fragments → Complete + +*** Modes +Prose and personal modes. General mode keeps the softer §30, which exempts more, because the strong "every sentence" rule is Craig's voice and should not be imposed on third-party text. + +*** Rule +Rewrite every sentence fragment inside a prose paragraph in Craig's authored text as a complete sentence with subject and verb. Bullets and headings can stay fragments. Exemption: verdict formulas in PR review summaries ("Approving.", "Requesting changes.", "Approved.") are house style and stay — rewriting them imposes the rule where Craig's calibrated voice already decided otherwise. + +*** Problem +Bullet shorthand leaking into running prose ("Two changes." "Fix incoming." "Body as decision log.") reads as bullet-list notes pasted into a paragraph. Every prose sentence needs a subject and a verb in prose and personal modes. + +*** Basis +Observation-derived (Craig's voice rules in claude-rules/commits.md). Corpus context: 9.7% of sentences are 1-5 words. Some are legitimate single-word claims ("All eight pass."), some may be fragments. Word count alone cannot distinguish. Syntactic detection deferred to Phase 2. + +*** Before +#+begin_example +Big change to the validator. Three new patterns. Test coverage up. Old behavior preserved. +#+end_example + +*** After +#+begin_example +I made a big change to the validator. There are three new patterns and the test coverage is up. The old behavior is preserved. +#+end_example + +*** Detection +Sentence fragments inside prose paragraphs in any text Craig authors: an email, a document, a working note, a commit or PR body. Bullets and headings remain fair game for fragments. + +*** History +- Original SKILL.md entry: sentence-fragment rewrite for Craig's authored prose. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. +- 2026-06-10: verdict-formula exemption added. The skill survived in practice by being selectively ignored on "Approving." / "Requesting changes." / "Approved.", and selective ignoring is the same muscle that skips real patterns. Documenting the exception removes one standing occasion for judgment-override. Craig's call, from the work-project session. + +** §38 Terse Cut — Omit Needless Words + +*** Modes +Prose and personal modes. General mode keeps the softer §22 because academic registers retain "worth noting" and "it's important to understand" as legitimate transition markers. + +*** Rule +Two cuts. First strip the soft rhetorical padding ("worth noting", "it's important to understand", "as you can see", "needless to say", "obviously", "of course", "in essence", "fundamentally"). Then run the general omit-needless-words sweep the padding list only samples: read each sentence and cut or collapse every word and clause that can go without losing meaning, not only the named phrases. Forcing test, per sentence: try to delete half of it and keep only what changes meaning. + +*** Execution position (prose + personal) +§38 is not just one pattern in the walk — it is the mandatory *last* pass before any draft is presented. The SKILL.md Process makes it an explicit standalone final step, run after every other pattern. The reason is empirical: a draft that cleared the other 40 patterns still routinely runs a third too long, because ordinary verbosity matches no named trigger and the categorical detectors come back clean while the text is still bloated. Folded into the general walk, §38 gets glossed as a wordlist match. As a separate final step it gets the real per-sentence "delete half of it" sweep. A public draft shown without this pass is a defect in the same class as skipping the skill entirely. + +*** Problem +Tier 1 omit-needless-words (§26) catches rigid offenders ("the fact that", "in order to"). The original §38 added a named padding list ("worth noting", "obviously"). But a draft can clear both and still run a third too long, because ordinary verbosity matches no named trigger: "that already merged via" for "landed on", "with it still in the PR, the same fix lands" for "keeping it re-lands the fix", restated subjects, throat-clearing lead-ins, clauses the reader already has. Those slip the categorical detectors silently — the walk comes back clean while the text is still bloated. So §38 is a real walk step, not a wordlist match: after the named padding, read each sentence and try to delete half of it. Academic registers keep the transition markers, so the aggressive cut stays prose and personal only. + +*** Basis +Corpus-measured across registers (2026-05-29). Single-sentence-paragraph rate: git commits 41.1%, personal email 57.4%, work email 44.5%, PR descriptions 74.4%, PR review comments 50.0%. The terse-paragraph cadence is even more pronounced in conversational and PR-description prose than in commits. Craig writes terse across registers, with the highest density in deliberate PR descriptions where each paragraph carries one focused thought. Confirmed indirectly via paragraph structure across all five corpora. + +*** Before +#+begin_example +It's worth noting that the change doesn't break the existing flow. Needless to say, the test suite is green. Obviously, this means we can ship. +#+end_example + +*** After +#+begin_example +The change doesn't break the existing flow. The test suite is green. We can ship. +#+end_example + +*** Before (generic verbosity, no named padding) +#+begin_example +This try/except is the same isolation change that already merged via #203. With it still in the PR, the same production fix lands under a second ticket, which is what the test: label means. +#+end_example + +*** After +#+begin_example +This try/except already landed on development via #203. Keeping it re-lands a merged fix under a second ticket, like the test: label says. +#+end_example + +This second pair carries no padding phrase from the named list. Every cut is ordinary verbosity: "is the same isolation change that already merged via" collapses to "already landed on", "With it still in the PR, the same production fix lands" to "Keeping it re-lands a merged fix", "which is what ... means" to "like ... says". A wordlist match finds nothing here; the per-sentence "delete half of it" test finds all of it. + +*** Detection +Two passes. (1) Named padding phrases: "worth noting", "it's important to understand", "as you can see", "needless to say", "obviously", "of course", "in essence", "fundamentally". (2) Ordinary verbosity beyond the list: verbose verb phrases ("already merged via" → "landed on"), restated subjects, throat-clearing lead-ins, and any clause whose content the reader already has. The forcing test for pass 2 is per sentence: try to delete half of it and keep only what changes meaning. + +*** History +- Original SKILL.md entry: rhetorical-padding cut for Craig's authored prose. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. +- 2026-06-02: generalized from a named-padding-list detector to a real omit-needless-words walk step. A PR-review comment cleared the §22/§23/§26/§38-padding patterns yet still ran a third too long on ordinary verbosity; the wordlist matched none of it. Renamed "Rhetorical Padding" to "Omit Needless Words", added the per-sentence "delete half of it" forcing test and the generic-verbosity example pair above. Craig's call. +- 2026-06-05: elevated to the mandatory final pass in the SKILL.md Process (new step 7). The pattern existed and was being walked, but got glossed as one of 41; a commit message went out needing two manual Orwell-walk requests before it read terse. Made it an explicit standalone last step that runs before any draft is shown, so the terse cut happens before Craig sees the draft rather than after he asks for it. Added the "Execution position" subsection above. Craig's call. + +** §39 Public-Artifact Scope Check + +*** Modes +Personal mode only. General and prose skip because a private journal or a third-party document has no public-scope concern. Flag only; no auto-rewrite. + +*** Rule +Flag (do not auto-rewrite) local absolute paths, private repo names, and personal-tooling references in publish artifacts. Surface each match as a WARN line so the author resolves manually. Output format: +#+begin_example +WARN: line 12: "/home/cjennings/code/rulesets" — local absolute path in commit body +WARN: line 18: "claude-rules/commits.md" — personal-tooling reference; state the underlying reason instead +#+end_example + +*** Problem +Commit messages, PR descriptions, PR comments, and Linear ticket bodies are visible to teammates and anyone with read access. References to the writer's personal layout are noise to a reader who cannot reproduce it. Auto-masking risks silently editing meaningful content because a legitimate file path mention may be load-bearing, and only the author can tell. + +*** Basis +Observation-derived (Craig's voice rules in claude-rules/commits.md, Content scope section). Corpus is the public artifacts themselves, so confirmation is circular. Deferred to Phase 2. + +*** Detection +Local absolute paths (=/home/<user>/...=, =/Users/<user>/...=), private repo names (any repo not in this project's known public set), personal-tooling references (humanizer, voice, commits.md, anything under =claude-rules/=, anything under =.ai/= or =.claude/=). + +*** History +- Original SKILL.md entry: public-artifact scope flag for personal mode. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §40 Praise vs Correction Asymmetry + +*** Modes +Personal mode only. General and prose skip because the rule assumes a PR review context. + +*** Rule +Praise on a PR review is short and unjustified (the author knows why their good change is good), and it survives only as an inline pin on the line it refers to. Correction always explains the why, gently and briefly, the way a mentor would, never as a verdict from on high. Keep it brief either way. + +On an approve summary: no praise at all, not even a bare positive ("Clean.", "Solid fix."). Lead with the substantive pointer — the design note pinned inline — and close with the verdict; an approve with nothing to flag is just "Approving." "Clean fix on the stacking bug, the tri-state is the right level to solve it at, and the tests cover the edges. Approving." becomes "One design note inline, not a blocker. Approving." (or just "Approving." with nothing to flag). Cut any clause that describes, justifies, or compliments the change — if a clause references what the code does, why it works, or how good it is, delete it. + +On a finding or change-request: always give the why, gently and briefly. Not "Move this to a helper." but "I'd pull this into one helper — three copies of the same rule means the next change has to touch all three, and missing one brings the bug back." + +Verification narration is the same defect as justified praise. "I traced X and it's safe because..." pads the compliment with the reviewer's homework. Tracing the code is the reviewer's job, not content for the comment — if verification found a problem, the problem gets the words; if it found nothing, it gets zero words. + +*** Problem +Praise and correction call for opposite treatment. The author already knows why their good change is good, so justifying praise reads as flattery. Correction is the reverse. Behavior only changes when the reason lands, so a finding, change-request, or inline coaching note must always explain the why. And the why is delivered gently, the way a kind coach or mentor would. + +*** Basis +Observation-derived (Craig's voice rules in claude-rules/commits.md, Voice and Focus section). PR-review corpus needed for empirical measurement. Deferred to Phase 2. + +*** Before +#+begin_example +Nice clean migration, the provider mocks and the Normal/Boundary/Error cases are all covered which is exactly what I'd want here. Approving. Also rename `x`. +#+end_example + +*** After +#+begin_example +One naming note inline, not a blocker. Approving. +#+end_example + +The rename rationale (`x` reads as a generic placeholder; the next person won't know it's the resolved provider without tracing it) lives in the inline pin, not the summary — the summary points, the pin teaches. + +*** Before (verification narration) +#+begin_example +All three fixes look right. I traced useMapActions and the unmount cleanup is safe because the hook returns a memoized object, and the provider wraps the whole app so neither call site lands on the no-op path. +#+end_example + +*** After +#+begin_example +Approving. +#+end_example + +Nothing to flag, so the summary is the bare verdict. The old "All three fixes are clean and well-aimed" is itself praise, and praise is now cut from the approve summary entirely. + +*** Detection +In a PR review summary or comment: any praise on an approve summary (including a bare positive), a praise clause that explains why the good thing is good, a praise clause followed by the verification work that supports it, or a finding or change-request that states what to fix without saying why. + +*** History +- Original SKILL.md entry: praise-versus-correction asymmetry for PR review. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. +- 2026-06-10: verification-narration variant added after the third recurrence — a review draft praised a fix and then narrated the verification supporting the praise (the #236 draft). Added to the SKILL.md rule line and the high-recurrence attestation set. Craig's call, from the work-project session. +- 2026-07-11: bare-positive carve-out removed. An approve summary now carries no praise at all, not even "Clean." / "Solid fix." — lead with the substantive pointer, close with the verdict. Craig's ruling from a DeepSat review session (approved "One design note inline, not a blocker. Approving."). Same change applied to review-code's Posted Summary Voice and commits.md Shape 1. + +** §41 No Emphasis Formatting + +*** Modes +Prose and personal modes. General mode keeps the related but mechanical §14 (boldface strip). §41 carries Craig's own principle and covers italics and underscores too. + +*** Rule +Remove emphasis markup (bold, italics, underscore-wrapped words) used to stress a phrase in Craig's prose, and rephrase so the stress lives in word choice and sentence shape. Structural markup stays: headings, defined terms on first use where the convention is house style, code spans for literal identifiers. + +*** Problem +Craig makes his points with words, not formatting. Emphasis markup is a crutch. When a sentence leans on bold or italics to land, the wording is not doing the work. The fix is not to delete the markup and leave a flat sentence. It is to rephrase so the stress lives in the word choice and sentence shape. This is the same principle behind his terminal-rendering rule in chat, but here it is about the writing itself, not the display. + +*** Basis +Observation-derived (Craig's voice rules in claude-rules/interaction.md, No Reverse-Video Highlighting rule). Org-mode bold uses =*word*= rather than Markdown =**word**= so corpus grep for Markdown emphasis is not directly applicable. Corpus measurement deferred to Phase 2. + +*** Before +#+begin_example +This is **really** important: you must run the migration *before* deploying, or the app will crash. +#+end_example + +*** After +#+begin_example +Run the migration before deploying. Skip that step and the app crashes on the first request. +#+end_example + +*** Detection +Bold (=**...**=), italic (=*...*= or =_..._=), or underscore-wrapped words used to emphasize a phrase in Craig's prose. + +*** History +- Original SKILL.md entry: no emphasis formatting for Craig's authored prose. +- 2026-05-29: migrated to this file as the canonical home per the pairing rule. + +** §42 Finding Stems — One Claim Per Sentence + +*** Modes +Personal mode only. General and prose skip because the rule assumes a PR review finding. + +*** Rule +A PR review finding is built from clean stems, each a straightforward sentence carrying one claim: (1) where the bug is, (2) the way(s) to fix it, (3) why that's better. Cut context sentences that don't change what the author does next (ticket history, design archaeology). Rewrite the anti-pattern shapes: hedged gerund chains ("the real bug looks like the model emitting a partial set"), compressed trade-off clauses ("I'd rather X, or Y, than lose Z"), multi-claim sentences chained through so-clauses or "and", and fixes buried after a mid-sentence colon. + +*** Problem +Craig named detangling overly complex or overly wordy Claude-drafted PR review text as THE key issue he fights in PR reviews, and the reason he gates every review draft. The tangles passed all 41 then-existing patterns — §38 shortens but doesn't untangle; a sentence can be terse and still carry three claims. §40 governs praise; this governs how finding text is constructed. + +*** Basis +Observation-derived from PR #233 (2026-06-10): a review comment shipped with hedged gerund chains and compressed trade-off clauses that cleared the full walk. The three Before/After pairs below are Craig-approved rewrites from that PR. + +*** Before (multi-claim opener + context sentence) +#+begin_example +POST fixes the wipe but it's additive: it no-ops on an empty list and never removes, so "cancel all partners" and any de-selection silently stop working. PUT came from SE-195 so the confirm could reconcile the full set. The real bug is upstream: on a new tasking the confirm emits only the new provider, not the full set. Fix that, or merge with the mission's current providers before the PUT. Either way removal keeps working. +#+end_example + +*** After +#+begin_example +POST is additive: it no-ops on an empty list and never removes. That breaks "cancel all partners" and any de-selection. The real bug is upstream: on a new tasking the confirm emits only the new provider, not the full set. Fix that, or merge with the mission's current providers before the PUT, and removal keeps working. +#+end_example + +The SE-195 context sentence is cut because it doesn't change the author's next action. + +*** Before (hedged gerund chain + compressed trade-off — the calibration case Craig pulled up) +#+begin_example +The real bug looks like the model emitting a partial set on a new tasking. I'd rather fix what the confirm emits, or merge client-side before the PUT, than lose removal. +#+end_example + +*** After (where / fix / payoff) +#+begin_example +The real bug is upstream: on a new tasking the confirm emits only the new provider, not the full set. Fix that, or merge with the mission's current providers before the PUT. Either way removal keeps working. +#+end_example + +*** Before (claims joined with "and"; fix buried after a mid-sentence colon) +#+begin_example +The prefix check catches any message starting with "confirm ", and the options block exists so the LLM can resolve "number 2" style references. A typed "confirm number 2" loses the list it needs. The card click already sends a self-describing "confirm <id>": pass an explicit parameter through sendAgentMessage and strip only on that path. +#+end_example + +*** After +#+begin_example +The prefix check strips the options block from any typed message starting with "confirm ", so "confirm number 2" loses the list the LLM needs to resolve it. Strip on the card-click path instead, with an explicit parameter passed through sendAgentMessage. The click already sends a self-describing "confirm <id>", so stripping is safe there. +#+end_example + +*** Detection +In a PR review finding: a sentence carrying more than one claim (chained through so-clauses, "and", or a mid-sentence colon hiding the fix), a hedged gerund chain where a direct claim belongs, a compressed trade-off clause, or a context sentence that doesn't change the author's next action. + +*** History +- 2026-06-10: created from the PR #233 calibration session. Proposed in the work project's stems handoff, landed via the consolidated voice-skill revision. Included in the high-recurrence attestation set from day one. Craig's call. + +** §43 Single-Sentence Paragraph Cadence Is a Feature + +*** Modes +Prose and personal modes. General mode skips because third-party registers legitimately prefer multi-sentence paragraphs. + +*** Rule +A one-sentence paragraph is a finished thought, not a fragment. "Shifts angle" means shifts *topic*: break paragraphs at a topic boundary, even when both sides are one sentence. Within a single topic, consolidate its sentences into one paragraph even when each is complete, up to a ceiling of about five or six sentences, past which find a natural break. The never-merge instruction protects the break *between* topics; it never licenses fragmenting one topic across several paragraphs. + +*** Problem +Most prose-style guides advise multi-sentence paragraphs, so a generic cleanup pass merges Craig's short paragraphs and erases a distinctive feature of his voice. This is a protective pattern: it guards an existing trait rather than correcting a defect. + +The 2026-07-23 boundary refinement addresses the opposite failure, discovered the same day: reading "angle" at *sentence* granularity, so that three sentences all about one topic (a baking run, a mixer, tortillas) got split into three paragraphs as if each were a new angle. That fragments one topic and reads as a checklist rather than a person talking. Angle means topic. #43 governs the break between topics; consolidation fills in what happens within one, which the original rule never specified. The two are one rule seen from both sides, not a rule in tension with #47. + +*** Basis +Corpus-measured (2026-05-29). Single-sentence-paragraph rate: git commits 41.1%, personal email 57.4%, work email 44.5%, PR descriptions 74.4%, PR review comments 50.0%. Between 41% and 74% of Craig's paragraphs are exactly one sentence, depending on register. + +*** Before (a cleanup pass merging short paragraphs) +#+begin_example +The build now finishes in half the time, and the cache no longer invalidates on every run, which means the CI queue clears faster too. +#+end_example + +*** After +#+begin_example +The build now finishes in half the time. + +The cache no longer invalidates on every run, so the CI queue clears faster too. +#+end_example + +*** Detection +An edit pass that merged short paragraphs, or a draft whose paragraphs each stack multiple shifted angles that would read better broken apart. + +*** History +- 2026-05-29: surfaced by the Phase 1-2 corpus measurement as a "worth adding" trait; filed as suggested delta 4. +- 2026-06-10: promoted from the suggested-deltas list into a numbered pattern. Craig's call, from the work-project session. +- 2026-07-23: boundary refined (angle means topic; within-topic consolidation to a ~5-6 sentence ceiling; never-merge reframed as across-topic protection). From a home session drafting a Signal reply, where the original rule was misread at sentence granularity. Paired with the new §47. + +** §44 Parenthetical Asides Are Part of the Voice + +*** Modes +Prose and personal modes. General mode skips because third-party text owns its own aside conventions. + +*** Rule +Parentheses for asides, clarifications, and scope-narrowing are Craig's voice. Don't strip them in a cleanup pass. They're also the preferred landing spot for em-dash replacements under §13. + +*** Problem +Generic style passes treat parentheticals as clutter and strip them. For Craig they carry asides, clarifications, and scope-narrowing, and removing them flattens the voice. Protective pattern, like §43. + +*** Basis +Corpus-measured (2026-05-29): 23.07 opening parens per 1000 words across the commit corpus. Heavy parenthetical use is distinctive and consistent. + +*** Before (a cleanup pass stripping the aside) +#+begin_example +The sync runs on every startup. It skips lockfiles. It also skips build output. +#+end_example + +*** After +#+begin_example +The sync runs on every startup (skipping lockfiles and build output). +#+end_example + +*** Detection +An edit pass that removed parenthetical asides present in the source text, or an em-dash replacement under §13 where parentheses fit better than a comma or period. + +*** History +- 2026-05-29: surfaced by the Phase 1-2 corpus measurement as a "worth adding" trait; filed as suggested delta 5. +- 2026-06-10: promoted from the suggested-deltas list into a numbered pattern, with the §13 landing-spot note. Craig's call, from the work-project session. + +** §45 Declarative Register Marker + +*** Modes +Prose and personal modes, advisory. Flag only; no auto-rewrite. General mode skips. + +*** Rule +Craig's prose is declarative. When a draft contains a rhetorical question, flag it for a second look — it's usually AI rhetoric, not his register. Genuine questions to the reader (a review asking the author's intent, an email asking for a decision) stay. + +*** Problem +AI drafts reach for rhetorical questions ("So what does this mean for the build?") as a transition device. Craig states things; he rarely asks them. A rhetorical question in his voice is a tell, but a genuine question is legitimate content, so the pattern flags rather than rewrites. + +*** Basis +Corpus-measured (2026-05-29): 0.33 question marks per 1000 words across the commit corpus. His prose register is declarative. + +*** Before (rhetorical transition flagged) +#+begin_example +So what does this change for the deploy flow? The staging gate now runs before the canary, which means a bad build never reaches it. +#+end_example + +*** After +#+begin_example +The staging gate now runs before the canary, so a bad build never reaches it. +#+end_example + +*** Detection +A question mark in a draft in Craig's voice. Flag it; keep genuine questions to the reader, rewrite rhetorical ones as declarative claims. + +*** History +- 2026-05-29: surfaced by the Phase 1-2 corpus measurement as a register marker; filed as suggested delta 6. +- 2026-06-10: promoted from the suggested-deltas list into a numbered advisory pattern. Craig's call, from the work-project session. + +** §46 Comma Budget — Max Two Per Sentence + +*** Modes +Personal mode only. Prose and general modes skip. + +*** Rule +No sentence carries more than two commas. Rewrite the third comma away: split the sentence, move a clause into a parenthetical (§44) or behind a colon, or break an inline serial list into bullets or its own sentence. Count prose commas only — commas inside code spans, quoted log lines, and literal strings (version numbers, paths) don't count toward the budget. + +*** Problem +Three or more commas in one sentence almost always mark stacked clauses or an inline list doing a paragraph's work. The sentence reads fine to its author and lands as a pileup on the reader. The comma count is a mechanical proxy the walk can enforce, where "don't stack clauses" is prose advice that gets skipped. + +*** Basis +Craig's directive, 2026-07-20 (archsetup session, while gating a Hyprland issue draft): "no more than two commas per sentence. we should add that to the /voice personal pass." + +*** Before (spec-sheet line with three commas, from the draft that prompted the rule) +#+begin_example +System: Arch Linux, kernel 6.18.25-lts, AMD Strix Halo (Radeon 8060S), no plugins loaded. +#+end_example + +*** After +#+begin_example +System: Arch Linux, kernel 6.18.25-lts. GPU: AMD Strix Halo (Radeon 8060S). No plugins loaded. +#+end_example + +*** Detection +Count commas per sentence on the final text. A sentence at three or more gets restructured, not trimmed to exactly the budget — the third comma is the symptom, the stacked structure is the target. + +*** History +- 2026-07-20: added at Craig's direction from the archsetup session. Scoped to personal mode; broaden to prose only if he asks. Added to the attestation high-recurrence set at birth — a mechanical count is cheap to receipt, and new discipline fails silently without one. + +** §47 Recipient-Priority Ordering + +*** Modes +Prose mode, and only when the piece is correspondence (email, Signal, a letter). It needs a recipient, so it has no referent in a journal, a working note, or any document addressed to nobody, and it does not carry into personal mode — a commit or PR review is not a reply to someone's news. General mode skips it with the rest of Craig's voice patterns. This is the one pattern narrower than a whole mode, and the only prose pattern personal mode does not also walk. + +*** Rule +In a reply, lead with what matters most to the recipient, not with what's easiest to answer or the order they wrote it. Their news outranks your logistics. A direct question they asked can sort below personal news they shared, because the news is what they care about. + +*** Problem +The easy draft answers the explicit question first and orders the rest as it arrived. That reads as transactional — logistics before the person. Ordering by what the recipient cares about is what makes a reply read as one person talking to another rather than a ticket being closed. Nothing else in the skill governs the *order* of a reply's contents; the other patterns act within a paragraph or a sentence. + +*** Basis +Craig's edit of a Signal reply to his sister, 2026-07-23 (home session). His framing: "start with what would be the most important things to her." + +*** Before (first draft — opens with the only explicit question, cooking split across three paragraphs) +#+begin_example +Yes, I do subscribe to MasterClass — happy to share what I've watched. + +That's amazing about the sourdough. English muffins from scratch is no joke. + +The home-roasted deli meat sounds incredible. + +A stand mixer would make the bread a lot easier — worth it if you're baking this much. + +And 30 pounds — that's huge. So happy for you. +#+end_example + +*** After (Craig's order — weight first, one cooking paragraph, then the question, then the close) +#+begin_example +Thirty-plus pounds — that is huge, and I'm so happy for you. That's real work. + +And the cooking. Sourdough, English muffins, tortillas, home-roasted deli meat from scratch — that's a whole kitchen you've built, and a stand mixer would make the bread much easier if you're baking at this volume, so I say go for it. I want to hear how the tortillas come out. + +Yes, I subscribe to MasterClass — I'll send you what I've been watching. + +I miss you and I love you. Send me a few times that work for a call. +#+end_example + +The cooking paragraph runs seven sentences, a hair over the §43 ceiling. Craig called it an exception rather than re-cut a message that had already gone out. The guard is the rule; this paragraph is one sentence over it; both facts stay in the record, because a real example at the boundary teaches it better than a clean one. + +*** Detection +A reply whose opening answers a logistical or yes/no question while the recipient's substantive news sits lower. Reorder so the news they'd most want acknowledged leads. + +*** History +- 2026-07-23: added from the home session drafting a Signal reply. The first handoff proposed two new patterns and flagged a conflict with §43; the superseding design resolved that the conflict was a misreading of §43 (angle = topic), leaving one genuinely new pattern here and a calibration to §43. Prose/correspondence-scoped per Craig — email and Signal are prose, not publish artifacts. + +** §48 Term-Translation Density + +*** Modes +General mode, so it runs in all three (general, prose, personal). It's a universal clarity rule in the Orwell / Plain English family, not a Craig-voice trait, and it reads to anyone editing any prose. The later number is an artifact of when it was added, not a scope signal. + +*** Rule +A sentence that forces the reader to stop and translate more than one specialized term (an acronym, a coined phrase, a product name) is too dense. One is fine; two or more in one sentence means rewrite: split the sentence, gloss one term in a parenthetical, or drop to plain language. The test is the reader's parse, not the writer's familiarity, and it is audience-relative. + +*** Problem +A writer fluent in the domain doesn't feel the translation cost of the terms, so a sentence stacking three of them reads as normal to the author and stalls the reader on every clause. Density is the metric, not any single word: two coined terms in one sentence is worse than a paragraph that introduces the same two one at a time. Audience-relative, because SAR to a defense team is shared vocabulary carrying no load, while a coined phrase only the writer holds carries full load for everyone else. + +*** Basis +Craig's edit of a customer-partner email, 2026-07-23 (work session), where one sentence stacked three terms and he flagged it. Distinct from #7 (specific AI-vocabulary words) and #30 (telegraphic fragments); this measures jargon density per sentence. + +*** Before (one sentence, three terms the reader must translate) +#+begin_example +A ViT detector gating a VLM for enrichment is close to our own detect-then-contextualize direction. +#+end_example + +*** After (split, glossed, plain) +#+begin_example +Their setup is a fast detector that hands off to a heavier model for a closer read. That mirrors our own two-stage approach (find it first, then work out what it is). +#+end_example + +*** Detection +Count the specialized terms in each sentence that a member of the intended audience would have to stop and translate. Two or more is the trigger. Acronyms, coined phrases, and product names count; shared-vocabulary terms for that audience don't. + +*** History +- 2026-07-23: proposed by Craig from the work session, drafting a customer-partner email. Placed in general mode (universal clarity rule); numbered #48, after the prose-only #47. diff --git a/working/working-dir-orphan-check/proposal-from-work.org b/working/working-dir-orphan-check/proposal-from-work.org new file mode 100644 index 0000000..d7d82b2 --- /dev/null +++ b/working/working-dir-orphan-check/proposal-from-work.org @@ -0,0 +1,12 @@ +#+TITLE: Convention proposal from Craig's roam inbox (2026-07-23 inbo +#+SOURCE: from work +#+DATE: 2026-07-23 15:28:00 -0500 + +Convention proposal from Craig's roam inbox (2026-07-23 inbox-zero from the work project): + +Every project should have a working/ and a temp/ directory. +- working/ holds files currently being worked on before archiving, in a subdirectory named for the project/task. NOT gitignored — pushed to remote. +- temp/ holds throwaway files (discarded prototypes, etc). Gitignored, not pushed, deleted as part of the wrap-up sequence. +- When a task completes, any files it left in working/ are always (via a soft hook if possible) archived or moved to temp/. No files related to a DONE task/project should remain in working/. + +Note for the skeptical review: this convention already largely exists in working-files.md (working/ tracked from creation; temp/ gitignored for throwaway). The genuinely new asks here are (a) the automatic soft-hook that empties working/ on task DONE, and (b) temp/ deletion wired into wrap-up. Worth checking what's already implemented vs what this adds before treating it as net-new. diff --git a/working/working-dir-orphan-check/proposed.diff b/working/working-dir-orphan-check/proposed.diff new file mode 100644 index 0000000..03452dc --- /dev/null +++ b/working/working-dir-orphan-check/proposed.diff @@ -0,0 +1,24 @@ +--- claude-templates/.ai/workflows/wrap-it-up.org 2026-07-23 08:39:07.467575608 -0500 ++++ /tmp/wu2.org 2026-07-23 20:49:55.500740517 -0500 +@@ -189,6 +189,21 @@ + emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done --check todo.org + #+end_src + ++*** Flag orphaned working/ directories ++ ++#+begin_src bash ++for d in working/*/; do ++ [ -d "$d" ] || continue ++ s=$(basename "$d") ++ grep -q "working/$s/" todo.org 2>/dev/null || { echo "ORPHAN: $d (no todo.org reference)"; continue; } ++ awk -v s="$s" '/^\*\* /{h=$0} $0 ~ "working/" s "/" {print (h ~ /^\*\* (DONE|CANCELLED)/ ? "CLOSED-TASK: working/" s "/ — " h : "") ; exit}' todo.org ++done ++#+end_src ++ ++Report only — never move or delete anything here. A =working/<slug>/= whose backing task is closed (or which no task references at all) is a filing candidate per =working-files.md=: rename each file individually and move it flat into its permanent home, then remove the empty directory. ++ ++Filing is deliberately a judgment step and stays manual. Deciding each artifact's permanent home and giving it a meaningful name is what makes =assets/= searchable later, and it's also the moment you notice which artifacts aren't worth keeping. An automatic sweep would skip exactly that review, and sweeping to =temp/= would destroy the artifacts outright, since =temp/= is cleared just below. ++ + *** Clear temp/ + + #+begin_src bash diff --git a/working/working-dir-orphan-check/wrap-it-up.org.proposed b/working/working-dir-orphan-check/wrap-it-up.org.proposed new file mode 100644 index 0000000..ff50a3f --- /dev/null +++ b/working/working-dir-orphan-check/wrap-it-up.org.proposed @@ -0,0 +1,651 @@ +#+TITLE: Session Wrap-Up Workflow +#+AUTHOR: Craig Jennings +#+DATE: 2026-04-20 + +* Overview + +This workflow defines the process for ending a Claude Code session cleanly. It finalizes the session record, commits + pushes all work, and provides a warm handoff. A bare wrap also tears the session down (kills the ai-term buffer + tmux session, restoring geometry); a qualified wrap keeps the buffer, and a shutdown wrap powers the machine off. The teardown variants are set by the trigger phrase (see Teardown mode below) and act only at the very end, in Step 6. + +Triggered by Craig saying "wrap it up," "that's a wrap," "let's call it a wrap," or similar. + +* The Session Record + +Throughout the session, =.ai/session-context.org= has been maintained with: +- =* Summary= — structured distillation (empty or draft during session) +- =* Session Log= — chronological narrative of what happened, written as you go + +At wrap-up, this file becomes the permanent session record by being renamed to =.ai/sessions/YYYY-MM-DD-HH-MM-description.org=. No transcription elsewhere. The file IS the record. + +* Exit Criteria + +The wrap-up is complete when: + +1. *Summary is written.* The =* Summary= section of =.ai/session-context.org= is populated by reading the =* Session Log= — Active Goal, Decisions, Data Collected / Findings, Files Modified, Next Steps. +2. *File is archived.* =.ai/session-context.org= has been renamed to =.ai/sessions/YYYY-MM-DD-HH-MM-description.org=. The old path no longer exists. +3. *todo.org is clean.* Cleanup script ran. Any auto-fixes are staged for the wrap-up commit. Orphan planning lines surfaced for manual fix if there are any. +4. *Linear board is honest* (skip if project doesn't use Linear). Any Dev-Review ticket whose PR has merged was moved to Done or PM Acceptance per the classification rule. +5. *Git state is clean.* All changes committed + pushed to all remotes. Working tree clean. +6. *Valediction delivered.* Brief, warm closing with key accomplishments and reminders, ending with =session wrapped.= on its own line as the signoff marker. + +The absence of =.ai/session-context.org= is the signal that the last session wrapped up cleanly. Its presence at session start means the previous session was interrupted. + +* Teardown mode (set from the trigger phrase) + +The wrap itself — Steps 1 through 5 — is identical in every mode. The trigger phrase only decides what Step 6 does once commit + push and the valediction are done. Resolve the mode from the phrase before starting: + +- *Teardown* (the default) — bare "wrap it up", "that's a wrap", "let's call it a wrap". The full wrap, then Step 6 kills the ai-term buffer + the =aiv-<project>= tmux session (which takes =claude= with it) and restores the saved window geometry. This is Craig's typical end-of-day case. +- *No-teardown* — "wrap it up with summary" or "wrap it up and summarize". The full wrap, but Step 6 leaves the buffer and session intact so the summary stays readable. The explicit qualifier is what opts out of teardown. +- *Shutdown* — "wrap it up and shutdown". The full wrap, then Step 6 gates on this being the only live ai-term session and powers the machine off. Shutdown supersedes teardown (killing the buffer is moot if the box is going down). + +Why teardown waits for Step 6 and runs through a hook, never inline: teardown kills the very tmux session =claude= runs in, so an inline kill would cut the valediction off before it renders. Step 6 instead drops a sentinel after everything else is verified, and the =Stop= hook (=ai-wrap-teardown.sh=) does the actual teardown when this response ends — by which point the valediction has already been delivered. + +This depends on three functions in =.emacs.d/modules/ai-term.el= (=cj/ai-term-quit=, =cj/ai-term-live-count=, =cj/ai-term-shutdown-countdown=) and on the =Stop= hook being wired in =settings.json= (=hooks/settings-snippet.json=). If =emacsclient= or the daemon is unreachable, the sentinel is cleared and the session simply stays up — teardown degrades to a no-op, never a wedge. + +* The Workflow + +** Step 0: Refuse if sentry is live + +Before anything else, check whether sentry is running in this project. Sentry holds the working tree on its =sentry/<date>-<host>= branch and commits unattended; wrapping underneath it would archive the session anchor and tear down the buffer while the loop is still firing into it. If sentry's single-runner lock is held, stop and point at the shutdown path: + +#+begin_src bash +proj="$(basename "$(git rev-parse --show-toplevel 2>/dev/null || pwd)")" +if [ -x .ai/scripts/agent-lock ] && .ai/scripts/agent-lock status "sentry-$proj" | grep -q '^held'; then + echo "sentry is active — say 'stop sentry' first" + exit 1 +fi +#+end_src + +The stop-sentry operation (defined in =sentry.org=) owns the shutdown: it cancels the loop, disposes of the branch, and walks the approval queue. Wrap-up carries only this one guard; a =stale= lock (a crashed fire) doesn't block — only a live =held= lock does. + +** Step 1: Finalize the Summary + +*** Work the Before-Close Queue (before the Summary) + +If the session anchor (=.ai/session-context.org=) carries a =* Before-Close Queue= heading with items, work them now, oldest-first, before writing the Summary, so any resulting edits ride this wrap's commit and get described in it. The queue is the "put X on the list" shorthand (see =protocols.org=, Colloquialisms and Expansions): session-scoped work Craig deferred to wrap time. + +Per item: do it if it's clear and bounded, or promote it to a =todo.org= task if it turns out to need its own session. Never drop an item silently. Remove each line as it's handled; if one can't be finished, surface it in the valediction (Step 5) and either leave a follow-up task or state why it's dropped. + +If there's no =* Before-Close Queue= heading, or it's empty, this step is a silent no-op. + +*** Early KB reflection (capture while fresh, before the Summary) + +Before distilling the Summary, while the session is still fresh, ask: what did this session learn worth remembering, for yourself or a future agent? Reflect and stage any candidate durable facts — a decision and its why, an environment gotcha, a reference pointer, a transferable lesson. Self-answer silently; this adds no interactive turn (Craig already authorized the wrap). The candidates flow straight into the KB promotion check below, which does the actual writing and the receipt — this is the capture half, that is the commit half, one pipeline, one receipt. Reflecting here rather than reconstructing learnings after the Summary is the point: the early ask is what keeps the receipt from defaulting to "promoted 0" out of fatigue. + +Read through the =* Session Log= in =.ai/session-context.org=. Populate (or refine) the =* Summary= section: + +- *Active Goal* — one or two sentences describing the session's focus +- *Decisions* — key choices made, with enough context to recall the /why/ +- *Data Collected / Findings* — anything concrete (measurements, root causes, paths, discoveries) +- *Files Modified* — what was changed, with one-line rationale per significant file +- *Next Steps* — what should happen in the next session + +Don't repeat everything from the Log in the Summary. The Summary is distillation — pull out what's load-bearing. The Log stays in the file and is available if a future reader wants detail. + +*** KB promotion check (and the one-line instrumentation receipt) + +Before closing the Summary, ask: did this session learn anything worth promoting to the agent knowledge base? The bar is =knowledge-base.md='s inclusion criteria — durable facts with cross-project or cross-machine value (decisions and their why, environment gotchas, reference pointers, transferable lessons). Promote each qualifying fact as one =agents/= node per the rule's schema (work-classified projects skip the write per the boundary; the check still runs so the receipt below is honest). + +Then add one line at the end of the Summary, always, even when nothing moved: + +#+begin_example +KB: promoted 2 / consulted yes +#+end_example + +"promoted N" counts nodes written this session (0 most sessions); "consulted yes-no" records whether any KB query informed the session's work. The line is the input to the spec's 30-day success-metrics checkpoint — grepping session archives for =KB:= answers "are agents actually using this?" without any other instrumentation. A session that skips the line breaks the metric, so it's part of the Summary contract, not optional. + +** Step 2: Pick a description + rename + +Read the Summary's Active Goal and the prominent entries in the Session Log. Pick a 4-6 word description that would make sense as a git-commit-message-series summary for the whole session. + +Good descriptions are concrete nouns/verbs: +- =docs-ai-migration-and-ai-launcher= +- =mybitch-usb-disconnect-diagnosis= +- =ratio-system-health-check= +- =orchestration-dashboard-bug-triage= + +Avoid vague ones: +- =session-work= (useless) +- =various-improvements= (useless) +- =updates= (useless) + +Get current time and rename: + +#+begin_src bash +mkdir -p .ai/sessions +now=$(date +%Y-%m-%d-%H-%M) +# Resolve the AI_AGENT_ID-aware source path (see protocols.org "Agent-scoped +# path"); fall back to the singleton if the helper isn't present. +sc=$(.ai/scripts/session-context-path 2>/dev/null || echo .ai/session-context.org) +# Under multi-agent, fold the agent id into the archive name so two agents +# wrapping in the same minute don't collide. Single-agent: no segment. +idseg="${AI_AGENT_ID:+${AI_AGENT_ID}-}" +mv "$sc" ".ai/sessions/${now}-${idseg}DESCRIPTION.org" +#+end_src + +Replace =DESCRIPTION= with your picked slug. (=AI_AGENT_ID= should be filename-safe and unique per run; the recommended =host.project.runtime.<epoch>= shape is both. The epoch on the tail keeps a re-run of the same logical agent from resolving to a prior run's leftover anchor. See protocols.org "Agent-scoped path".) + +** Step 3: todo.org cleanup (hygiene + archive completed work) + +If the project has a =todo.org= at its root, run the cleanup script before committing. Two passes, both fast and idempotent: a hygiene pass and an archive pass. + +*** Roam inbox sweep (inbox roam mode) + +Before the cleanup scripts, sweep the roam global inbox (=~/org/roam/inbox.org=) for items that belong to this project, so any imported tasks get linted and ride the wrap commit. Delegate to [[file:inbox.org][inbox.org]] roam mode for the claimed set. + +#+begin_src bash +[ -f "$HOME/org/roam/inbox.org" ] && grep -cE '^\*\* ' "$HOME/org/roam/inbox.org" || true +#+end_src + +Skip-fast when nothing matches: if the roam clone isn't on this machine, or no item is prefixed for this project, this is a silent no-op. When claimed items exist, run roam mode's Phase B–D (file each into =todo.org=, then remove them from the shared inbox and let =roam-sync= commit + push the edit). Report the total count and how many appeared related to this project, per roam mode's scan-summary rule. + +*** Hygiene pass + +It catches a recurring pattern: org sometimes leaves noise lines like =- State "X" from "X" [date]= when a state-change log lands outside a =:LOGBOOK:= drawer and the state didn't actually change. These lines carry no information and they break org's planning-line parser by wedging between the heading and =DEADLINE:=/=SCHEDULED:=, which kicks the entry out of agenda views. + +#+begin_src bash +[ -f todo.org ] && emacs --batch -q -l .ai/scripts/todo-cleanup.el todo.org +#+end_src + +The script is fast (under half a second on a 4000-line file) and idempotent — if there's nothing to fix, it reports zero changes and exits clean. + +What it does: + +1. *Auto-deletes* bogus state-log lines (matched on identical from/to states). Any deletions show up in the wrap-up commit's diff, so they get reviewed before push. +2. *Reports* "orphan planning lines" — entries whose body has =DEADLINE:= or =SCHEDULED:= but =org-entry-get= can't read it (some other malformation kept it out of canonical position). The script doesn't auto-rewrite these because the right fix depends on whether real state-log history needs preserving — surface them and fix manually if they matter for the agenda. + +Run the report-only variant first if you want to see what would change without writing: + +#+begin_src bash +emacs --batch -q -l .ai/scripts/todo-cleanup.el --check todo.org +#+end_src + +*** Convert done sub-tasks to dated entries + +#+begin_src bash +[ -f todo.org ] && emacs --batch -q -l .ai/scripts/todo-cleanup.el --convert-subtasks todo.org +#+end_src + +=--convert-subtasks= rewrites every heading at level 3 or deeper whose TODO state is DONE/CANCELLED/FAILED into a dated event-log entry (=<stars> YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <text>=), dropping the keyword, priority cookie, and tags, and removing the now-redundant =CLOSED:= line. This enforces the =todo-format.md= depth rule that a completed *sub-task* (a heading under a parent task) becomes dated history, not a lingering DONE keyword — a shape an interactive org close (=org-log-done= → DONE + CLOSED) never applies and =--archive-done= (level-2 only) never reaches. The timestamp comes from each entry's own =CLOSED= cookie; a date-only close yields =00:00:00=. Heading text is kept verbatim. Idempotent (an already-dated heading has no keyword to match), and a done sub-task with no parseable =CLOSED= is flagged and left alone rather than stamped with a fabricated date. + +Run this *before* =--archive-done= so that when a completed level-2 parent is archived, its sub-tasks already carry their dated form. Any rewrites show up in the wrap-up commit's diff for review before push. + +Preview without writing: + +#+begin_src bash +emacs --batch -q -l .ai/scripts/todo-cleanup.el --convert-subtasks --check todo.org +#+end_src + +*** Archive completed work + +#+begin_src bash +[ -f todo.org ] && emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done todo.org +#+end_src + +=--archive-done= moves every level-2 subtree whose TODO state is DONE or CANCELLED out of the project's "Open Work" section and into its "Resolved" section, subtree intact. The two sections are matched by a unique level-1 heading containing "Open Work" (case-insensitive) and one containing "Resolved" — if either is missing or ambiguous, the file is skipped with a message, no crash. Only direct level-2 children move; a DONE entry nested under an open parent stays put. Idempotent; any moves show up in the wrap-up commit's diff for review before push. + +Preview the moves without writing: + +#+begin_src bash +emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done --check todo.org +#+end_src + +*** Flag orphaned working/ directories + +#+begin_src bash +for d in working/*/; do + [ -d "$d" ] || continue + s=$(basename "$d") + grep -q "working/$s/" todo.org 2>/dev/null || { echo "ORPHAN: $d (no todo.org reference)"; continue; } + awk -v s="$s" '/^\*\* /{h=$0} $0 ~ "working/" s "/" {print (h ~ /^\*\* (DONE|CANCELLED)/ ? "CLOSED-TASK: working/" s "/ — " h : "") ; exit}' todo.org +done +#+end_src + +Report only — never move or delete anything here. A =working/<slug>/= whose backing task is closed (or which no task references at all) is a filing candidate per =working-files.md=: rename each file individually and move it flat into its permanent home, then remove the empty directory. + +Filing is deliberately a judgment step and stays manual. Deciding each artifact's permanent home and giving it a meaningful name is what makes =assets/= searchable later, and it's also the moment you notice which artifacts aren't worth keeping. An automatic sweep would skip exactly that review, and sweeping to =temp/= would destroy the artifacts outright, since =temp/= is cleared just below. + +*** Clear temp/ + +#+begin_src bash +[ -d temp ] && find temp -mindepth 1 -delete && echo "temp/ cleared" +#+end_src + +=temp/= holds throwaway artifacts — discarded prototypes, scratch output, intermediate data (see =working-files.md=). It's gitignored in every project, so nothing here rides a commit and nothing is recoverable from git once deleted. Clearing it at wrap is what keeps ephemeral work from silting up across sessions, and it's the counterpart to =working/=, which is tracked and *never* cleared here. + +Two guards. Confirm before deleting if =temp/= holds anything a reasonable reader would call in-progress rather than throwaway — misfiled work belongs in =working/=, so move it there instead of deleting it. And skip the step entirely in a project where =temp/= is not gitignored, since that means the project is using the directory for something else. + +*** Sync child priorities + +#+begin_src bash +[ -f todo.org ] && emacs --batch -q -l .ai/scripts/todo-cleanup.el --sync-child-priority todo.org +#+end_src + +=--sync-child-priority= walks every heading with a priority cookie =[#A]=–=[#D]= and, for each of its direct child headings whose own priority cookie is /lower/ (later in the alphabet — D is below A), bumps the child to match the parent. Down-only: parents are never bumped up to match a higher-priority child. Children without a priority cookie are left alone, as are parents without one. The walk visits parents before descendants, so a multi-level chain (=[#A]= → =[#B]= → =[#D]=) collapses to the top priority in a single pass. Idempotent. + +Opt-out for deliberately-lower children: tag the heading =:no-sync:= (the literal six-character tag, including the hyphen). The script matches the tag literally on the heading line, so it works whether or not the surrounding emacs config has extended =org-tag-re= to allow hyphens. + +#+begin_example +*** TODO [#D] Follow-up: VAD :no-sync: +#+end_example + +Use this for =Follow-up:=, =Spike:=, =Stretch:= sub-tasks that are deliberately deprioritized below their parent — without the tag, the wrap-up would silently bump them back up. + +Preview the bumps without writing: + +#+begin_src bash +emacs --batch -q -l .ai/scripts/todo-cleanup.el --check-child-priority todo.org +#+end_src + +(=--check-child-priority= is the report-only alias for =--sync-child-priority --check=.) + +*** Lint org files (mechanical sweep, judgments deferred) + +#+begin_src bash +if [ -n "$LINT_ORG_FOLLOWUPS" ]; then + followups="$LINT_ORG_FOLLOWUPS" +elif [ -d "./inbox" ]; then + followups="./inbox/lint-followups.org" +else + followups=".ai/lint-followups.org" +fi +[ -f todo.org ] && emacs --batch -q -l .ai/scripts/lint-org.el \ + --fix --followups-file="$followups" todo.org +#+end_src + +The =--fix= flag is required for the writes: lint-org's CLI default is +report-only (a linter reports, it doesn't write), and this wrap-up pass is +the deliberate exception that applies fixes — its diff rides the wrap-up +commit for review. + +=lint-org= runs =org-lint= over =todo.org=, auto-applies four mechanical +categories (=item-number= counters, bare =#+begin_src= → =#+begin_example=, +multi-line planning-info merged onto one line, =**X.**= → =*X.*=), and +appends every remaining judgment item (broken file links, invalid fuzzy +links, verbatim-asterisk inside body prose, suspicious src-block languages) +to the follow-ups file as a dated org section. Mechanical fixes show up in +the wrap-up commit's diff for review before push. + +The follow-up path defaults to =./inbox/lint-followups.org= in the current +project (where the next morning's daily-prep merges it in). If the project +doesn't have an =inbox/= directory, the script falls back to +=.ai/lint-followups.org= inside the current project. Override with +=LINT_ORG_FOLLOWUPS=<path>= in the environment if needed — useful for +routing all wrap-up output to a single shared inbox across projects. + +Each project's own =inbox/= is the right default because daily-prep reads +that project's inbox at startup. Hardcoding a single project's path +(formerly =~/projects/work/inbox/=) routed every project's wrap-up findings +into the wrong inbox. + +Preview without writing — same flags as =--check= on the other scripts: + +#+begin_src bash +[ -f todo.org ] && emacs --batch -q -l .ai/scripts/lint-org.el --check todo.org +#+end_src + +The wrap-up never blocks on judgment items — they're deferred by design. +For an interactive walk of the judgments mid-day, run =/lint-org todo.org=. + +*** Inbox sanity check (surface unprocessed handoffs) + +If the project has an =inbox/= directory, verify it holds nothing but =.gitkeep=, =lint-followups.org= (the lint-org pipeline file the next morning's daily-prep consumes), and any explicitly-deferred =PROCESSED-*= files before the wrap completes. An inbox that arrived at session start with handoffs from other projects, or that received handoffs mid-session, needs =inbox.org= process mode to run and apply its value-gate dispositions. Wrapping with a dirty inbox silently defers the work to next session and accumulates handoff debt that the sender can't see. + +#+begin_src bash +unprocessed=$(find inbox -maxdepth 1 -type f \ + ! -name '.gitkeep' \ + ! -name 'lint-followups.org' \ + ! -name 'PROCESSED-*' \ + 2>/dev/null | wc -l) +if [ "$unprocessed" -gt 0 ]; then + echo "wrap-up: inbox/ has $unprocessed unprocessed item(s). Run inbox.org process mode before wrapping, or explicitly defer each item with a one-line reason in the valediction." + find inbox -maxdepth 1 -type f \ + ! -name '.gitkeep' \ + ! -name 'lint-followups.org' \ + ! -name 'PROCESSED-*' \ + -printf ' %f\n' +fi +#+end_src + +If the count is zero or the project has no =inbox/= directory, the check is a silent no-op. If non-zero, the wrap is incomplete by default. The user resolves each item (process now, defer with reason in the valediction, or delete with rationale) before the validation checklist passes. + +The check exempts =lint-followups.org= explicitly because lint-org runs earlier in the same wrap-up workflow and writes its judgment items to that file in =inbox/= by design. The file is a pipeline artifact for the next morning's =daily-prep=, not a handoff that needs the value gate. + +This integrates with =inbox.org= process mode, which stamps =:LAST_INBOX_PROCESS:= in =notes.org='s *Workflow State* section on completion. Wrap-up doesn't double-stamp. It only ensures the inbox carries nothing but the expected pipeline artifacts at session end. + +*** Cross-project router (optional — route filed keepers to their home projects) + +Runs directly after the inbox sanity check. The split between the two: the sanity check *gates* the wrap (a dirty inbox blocks until resolved); the router is *optional* (skipping it never blocks anything — the candidates just stay local until a future wrap). Spec: =docs/specs/wrapup-routing-spec.org= (D7/D8/D9). + +The candidate set is exactly the local tasks carrying a =:ROUTE_CANDIDATE:= property — keepers that inbox process mode filed this session whose inferred home is another project. Never scan the standing backlog. + +#+begin_src bash +.ai/scripts/route-batch --list +#+end_src + +*Empty set = zero interaction.* =--list= prints nothing when there are no candidates; continue the wrap silently — no prompt, no "0 items" line. + +When candidates exist, surface the batch as one line per task — the task heading, the destination project, the delivery mode (=inbox-send= file handoff), and the engine's confidence — then offer exactly two options: *go* (route the whole batch) or *skip* (leave everything local). Derive each confidence label by running the engine on the task's heading + body (=python3 .ai/scripts/route_recommend.py --item "..." --exclude "$(basename "$PWD")"=); label weak matches visibly ("weak — verify the destination") so a low-confidence route gets a human glance before the keystroke. + +On *go*: + +#+begin_src bash +.ai/scripts/route-batch --go +#+end_src + +Per candidate, the helper writes the task's subtree (children ride along; =:ROUTE_CANDIDATE:= stripped, headings promoted to top level) to a one-task handoff, delivers it via =inbox-send <destination> --file= (so the =from-<this-project>= provenance is stamped and the destination's inbox process mode dispositions it as a single item), and only after a successful send removes the subtree from the local =todo.org= — a single-file local edit the wrap is already committing. A failed send leaves that task in place and exits non-zero; report it and continue the wrap. Never write the destination's =todo.org= directly; its own inbox processing files the task per its conventions. + +On *skip*, leave every candidate in place, marker included — they resurface next wrap. + +Mis-routes are recoverable: the receiving project rejects via inbox process mode's reject-from-another-project flow, which returns the item to this project's inbox with the rationale. That reject path is why removing the local source on send is safe. + +*** Review-habit health check (surface a slipped daily task-review) + +The daily task-review habit walks the open top-level tasks on a rotating cycle, stamping =:LAST_REVIEWED:= as it goes (see =task-review.org=). This check is the watchdog for that habit. When tasks have gone too long unreviewed, the habit has slipped, and the wrap-up says so in one line — it does not re-list the tasks. + +=task-review-staleness.sh= counts top-level =[#A]= / =[#B]= / =[#C]= tasks (TODO/DOING/VERIFY) whose =:LAST_REVIEWED:= is missing or older than the threshold. Threshold 30 days is about 2.5 review cycles of slack at the default batch size — one missed week is fine, three weeks signals a problem. + +#+begin_src bash +if [ -n "$LINT_ORG_FOLLOWUPS" ]; then + followups="$LINT_ORG_FOLLOWUPS" +elif [ -d "./inbox" ]; then + followups="./inbox/lint-followups.org" +else + followups=".ai/lint-followups.org" +fi +if [ -f todo.org ]; then + stale=$(.ai/scripts/task-review-staleness.sh todo.org 30 2>/dev/null || echo 0) + if [ "$stale" -gt 0 ]; then + printf "\n* %s — Task-review health: %s top-level [#A]/[#B]/[#C] tasks unreviewed for >30 days (daily review may have slipped)\n" \ + "$(date '+%Y-%m-%d %a')" "$stale" >> "$followups" + fi +fi +#+end_src + +A non-zero count writes one summary line and nothing else — the per-task walk is the review habit's job, not the wrap-up's. This supersedes the old date-coverage scan, which flagged every dateless =[#A]= / =[#B]= task on the wrong assumption that high-priority work needs a date. No-date is a valid resting state for research and watch-list tasks; staleness, not datelessness, is the real signal. + +** Step 3.5: Linear ticket-state hygiene (skip if project doesn't use Linear) + +If the project uses Linear and has any tickets currently in *Dev Review* assigned to Craig, sweep them before the wrap-up commit. The check is fast and keeps the board honest — tickets stuck in Dev Review after their PR merges hide actual work-in-progress. + +#+begin_src +mcp__linear__list_issues assignee="me" state="Dev Review" limit=50 +#+end_src + +For each result, look up the linked PR (the =gitBranchName= field on the issue maps to a =headRefName= on the project's GitHub remote — use =gh pr list --author <github-login> --state all --json number,state,headRefName,mergedAt,title=). + +*Assumption:* the =gh= lookup expects a GitHub-family host. It holds today because the only Linear-using project (DeepSat) lives on =deepsat.ghe.com=, where =gh= talks to the GHE API. A future Linear-using project on a non-GitHub host (GitLab, Gitea, Bitbucket) would need a provider-agnostic PR lookup here — update this step when that happens. + +If a Dev-Review ticket's PR is *merged*, propose a move: + +- *Done* — chores, refactors, test-coverage backfills, dead-code removal, e2e-flake fixes, anything with no PM-visible behavior change. PR titles prefixed =chore:=, =test:=, =refactor:=, =docs:= almost always belong here. +- *PM Acceptance* — real behavior fixes or new features a PM (or end user) could verify by clicking through the app. PR titles prefixed =fix:=, =feat:= usually belong here unless the change is invisible to users. + +When in doubt, ask Craig per ticket. Don't auto-pick. After Craig confirms, move via =mcp__linear__save_issue= with =state="Done"= or =state="PM Acceptance"=. Several can run in parallel. + +Skip the step entirely if the project doesn't use Linear (e.g. personal projects, the rulesets repo). + +** Step 4: Git commit + push + +*** Step 4.0: Commit template-sync churn first (consuming projects) + +The startup workflow's Phase A rsyncs template updates from rulesets into this project's =.ai/= (=protocols.org=, =workflows/=, =scripts/=) every session that rulesets has advanced. Nothing commits that churn, so without this step it accumulates across sessions and eventually blocks Phase A.0's auto-fast-forward (git refuses to ff a dirty tree). Commit it here, as its own =chore:= commit, before the session-work commit — so the sync stays separate from what the session actually shipped and the tree ends clean. + +The guard is conservative: only auto-commit a dirty synced path when it matches the rulesets canonical byte-for-byte (a modified/new file equals canonical, or a deletion pairs with a file retired upstream). If any synced path is dirty but /doesn't/ match canonical — a local hand-edit to a file that's supposed to be sync-managed — surface it and don't auto-commit. Anything outside the three synced paths is untouched here; the normal Step 4 commit and the worktree-leftover step handle it. + +#+begin_src bash +# Skip in the rulesets repo itself: there .ai/ is a committed mirror of +# claude-templates/.ai/, kept in sync by the pre-commit hook and committed +# alongside template edits — not downstream sync churn. The presence of +# claude-templates/.ai/ in this repo is the tell. +if [ ! -d claude-templates/.ai ] && [ -d "$HOME/code/rulesets/claude-templates/.ai" ]; then + canon="$HOME/code/rulesets/claude-templates/.ai" + safe=1 + commitlist=() + while IFS= read -r line; do + f="${line:3}" # strip the 2-char status + space + rel="${f#.ai/}" + if [ -e "$f" ] && [ -e "$canon/$rel" ] && diff -q "$f" "$canon/$rel" >/dev/null 2>&1; then + commitlist+=("$f") # modified/new here, matches canonical + elif [ ! -e "$f" ] && [ ! -e "$canon/$rel" ]; then + commitlist+=("$f") # deleted here AND retired upstream + else + safe=0 # synced path dirty but != canonical + fi + done < <(git status --porcelain -- .ai/protocols.org .ai/workflows/ .ai/scripts/) + + if [ "$safe" -eq 1 ] && [ "${#commitlist[@]}" -gt 0 ]; then + git add -- "${commitlist[@]}" + git commit -q -m "chore: sync .ai tooling from templates" + echo "wrap-up: committed ${#commitlist[@]} synced .ai file(s) as a template-sync chore." + elif [ "$safe" -eq 0 ]; then + echo "wrap-up: synced .ai paths are dirty but not all match rulesets canonical — NOT auto-committing. Resolve manually:" + git status --porcelain -- .ai/protocols.org .ai/workflows/ .ai/scripts/ | sed 's/^/ /' + fi +fi +#+end_src + +The commit isn't pushed here — the push step below pushes the current branch, which carries both this chore commit and the session-work commit. A crashed session that never reaches wrap-up leaves the churn for the next startup, which surfaces it (see startup.org Phase C) so it never silently accumulates. + +*** Review changes + +#+begin_src bash +git status +git diff --stat +#+end_src + +Decide the scope of the wrap-up commit. Usually everything that changed during the session goes into one commit. If anything is intentionally not part of this session's work (pre-existing WIP, unrelated files), leave it out. + +*** Stage + +Add the renamed session file and all other session changes: + +#+begin_src bash +git add .ai/sessions/ [other modified paths] +#+end_src + +Do NOT blindly =git add .= — review what's being staged so unrelated dirty state isn't dragged in. + +*** Commit + +Commit message rules (also see protocols.org "Git Commit Requirements"): + +- Subject line: concise, describes what /shipped/. Use conventional prefixes (=docs:=, =refactor:=, =fix:=, =feat:=, =chore:=) — NEVER =session:=. +- Body: 1-3 terse sentences describing what was accomplished. +- NO Claude Code attribution. NO =Co-Authored-By=. NO references to =notes.org=, =session-context.org=, =.ai/sessions/=, "session wrap-up", or session timestamps. + +*Wrap-up commits skip the inline-approval gate.* The =commits.md= rule that requires writing the message to =/tmp/commit-<slug>.md=, printing inline, and waiting for an approve / request-changes / open-in-editor response does *not* apply to wrap-up commits. The wrap-up flow is meant to be quick — Craig has already authorized the wrap by triggering the workflow ("wrap it up"), and stopping again to approve a commit message disrupts the cadence. + +Still apply =/voice personal= silently before committing so the message reads cleanly. Just don't print and ask. Commit directly with the cleaned message. + +If a wrap-up commit needs Craig's eyes for a content reason (sensitive change, unusual scope, something he flagged earlier), surface it explicitly. Otherwise commit and move on. + +Example: +#+begin_example +docs: restructure docs/ to .ai/ and unify aix+hey into ai launcher + +Hidden .ai/ now holds Claude tooling; project-level docs/ reserved +for user-facing docs. Single 'ai' launcher (fzf multi + smart tmux ++ git-aware fetch/pull) replaces the aix script and hey alias. +#+end_example + +Use heredoc for multi-line: +#+begin_src bash +git commit -m "$(cat <<'EOF' +subject line here + +body sentences here. +EOF +)" +#+end_src + +*** Push to all remotes + +#+begin_src bash +git remote -v +#+end_src + +Push the current branch to every remote (some repos have multiple remotes — a primary host plus one or more mirrors, or different remotes for different audiences — and the loop keeps all of them current): + +#+begin_src bash +current=$(git symbolic-ref --short HEAD) +for r in $(git remote); do git push "$r" "$current"; done +#+end_src + +Then push every other local branch with a tracking upstream to its tracking remote. This catches feature branches that advanced during the session but aren't the one being wrapped up — without it, work-in-progress branches stay local-only and are at risk if the machine dies before the next wrap-up. + +#+begin_src bash +git for-each-ref --format='%(refname:short) %(upstream:remotename)' refs/heads/ | \ +while read branch remote; do + [ "$branch" = "$current" ] && continue + if [ -z "$remote" ]; then + echo " $branch: no tracking upstream — skipped (push manually with 'git push -u')" + else + git push "$remote" "$branch" + fi +done +#+end_src + +Behavior: +- *Tracked branches* → pushed to their upstream remote. +- *Untracked branches* (no upstream set) → surfaced, not pushed. Craig sets the upstream manually with =git push -u <remote> <branch>= when he's ready. Auto-creating an upstream would commit to a remote choice the workflow can't make safely. +- *Diverged or rejected pushes* → surface and stop. Don't force-push from this workflow; resolve manually. + +*** Resolve every worktree leftover + +#+begin_src bash +git status --short +#+end_src + +*Default policy: end every session with an empty =git status=.* The wrap is incomplete while anything remains dirty. There is no "leave it alone" default — every leftover gets an active resolution. The only way for a file to stay dirty across the wrap is the user explicitly saying "defer this one, leave it dirty." Surface each leftover with a concrete recommendation; the user has to actively opt out for the dirt to persist. + +This inverts the older "intentional carryover" default, which let pre-existing dirty state accumulate across sessions silently. Carryover that lives for days or weeks is almost always one of: a forgotten commit from a prior wrap, a stale change that should be discarded, or genuine in-flight work that needs an explicit stash/branch home. None of those should default to "leave it dirty." + +**** Three kinds of leftover + +| Pattern | What it is | Recommended action (apply unless user defers) | +|---+---+---| +| Generated, runtime, or lock files that no human edits — e.g., =.claude/scheduled_tasks.lock=, =.pytest_cache/=, build outputs, IDE state, editor swap files | *Runtime artifact* — created by tooling or the harness, not by the user, and shouldn't be tracked | Add the matching pattern to =.gitignore= (project-level, not =~/.gitignore_global=). For tracked files, =git rm --cached <path>=. Stage =.gitignore= and any =rm --cached= changes in *one* follow-up commit (=chore: gitignore X=), push. Re-run =git status= to confirm clean. | +| Modified or created during the session but not staged into the wrap-up commit | *Forgotten change* — real session work that should have been in the wrap commit but missed it | Stage and create a follow-up commit. Don't =--amend= the wrap-up commit once pushed (diverging history without a clear win). Push the follow-up to all remotes. | +| Was dirty at session start and still dirty at session end — work this session deliberately didn't touch | *Pre-existing dirt that needs a decision* — could be a missed commit from a prior wrap, stale abandoned work, or real in-flight work without a home | Investigate (show diff + check the originating session). Recommend one of: (a) commit now if the work is complete, (b) stash with a descriptive message if it's genuine WIP, (c) =git checkout -- <path>= / =git clean -f <path>= if stale and unwanted, (d) move to a feature branch if it's longer-running, (e) user explicitly defers and accepts the dirt. Do not silently leave dirty. | + +**** Per-file flow + +For each leftover line in =git status --short=: + +1. Identify which of the three kinds above it matches. +2. State what the file is (one line) and the recommended action. +3. Apply the action unless the user explicitly defers. +4. Re-run =git status --short= after each follow-up commit until empty (or until every remaining line is an explicit user-deferred entry). + +The pre-existing-dirt case (third row) is the one this rule most cares about. Treat each pre-existing-dirty file as a question that must get an answer this session, not as "carryover that's fine to inherit." A file that was dirty for a week before this session probably isn't going to get cleaner by waiting another week. Look at the diff, check the originating session's notes, and recommend a real resolution. + +**** When the user defers + +If the user does say "leave this one dirty for now" after seeing the recommendation, that is fine — log the deferral in the valediction so the next session knows it was an explicit choice, not a miss. Format: "Deferred (per Craig's decision today): =path/to/file= — <one-line reason>". Without that note, the next session can't distinguish "we agreed to defer" from "we forgot again." + +** Step 5: Valediction + +Brief, warm closing. 3-4 sentences max. + +Include: +- What was accomplished (specific, not generic) +- What's ready for next session +- Any critical reminders or deadlines + +Tone: warm but professional. No emoji unless Craig has explicitly requested. Acknowledge effort when session was long or difficult. + +End on a clear signoff: the *last* line of the valediction is always =session wrapped.= on its own line (lowercase, with the period, nothing after it). It's the unmistakable end-of-session marker, so don't trail it with another sentence. This is the last user-facing output — Step 6's teardown is silent. + +Example: +#+begin_example +That's a wrap. Today we restructured the entire claude-templates +ecosystem: docs/ → .ai/ across all 23 projects, unified aix + hey +into a single 'ai' launcher with git-aware fetch/pull, and cleaned +up 4 code projects on velox. Both machines fully in sync. + +Two things to pick up next: the chime README WIP (your inline notes +from earlier) and archsetup's layout-navigate tests. Both are +ratio-local uncommitted state. + +Good session. Talk tomorrow. + +session wrapped. +#+end_example + +** Step 6: Session teardown (mode-dependent) + +The last action of the wrap, and only after Step 4's commit + push is verified and the Step 5 valediction is composed. The teardown itself happens when this response ends (via the =Stop= hook), so the valediction always renders first. Act by the mode resolved up front: + +*** No-teardown mode + +Do nothing. The buffer, the =aiv-<project>= tmux session, and =claude= all stay up so the summary stays readable. The wrap is complete. + +*** Teardown mode (default) + +Confirm commit + push succeeded (Exit Criteria 5 — never tear down over unpushed work), then drop the sentinel: + +#+begin_src bash +touch "/tmp/ai-wrap-teardown-$(basename "$PWD")" +#+end_src + +That is the whole step. Don't run any =tmux kill-session=, =emacsclient=, or buffer kill inline — the =Stop= hook reads the sentinel when this response ends and runs =cj/ai-term-quit=, which kills the =aiv-<project>= session (taking =claude= with it), kills the vterm buffer, and restores geometry. The basename of =$PWD= is the key the hook matches, so the sentinel names the session it tears down. + +*** Shutdown mode + +Confirm commit + push succeeded, then evaluate the safety gate *before* committing to the shutdown — never power the box off out from under another live session: + +#+begin_src bash +emacsclient -e '(cj/ai-term-live-count)' +#+end_src + +- *Count > 1* — another ai-term session is alive. ABORT the shutdown. List the other live =aiv-*= sessions, drop *no* sentinel, and tell Craig in the valediction that it fell back to a normal wrap (no poweroff, no teardown). This gate is the load-bearing safety of the whole feature. +- *Count = 1* — this session is the only one. Drop the shutdown sentinel: + + #+begin_src bash + touch "/tmp/ai-wrap-shutdown-$(basename "$PWD")" + #+end_src + + The =Stop= hook fires =cj/ai-term-shutdown-countdown= when this response ends: it re-checks the gate, runs an abort-able 10→1 countdown in the Emacs echo area (=C-g= cancels), then =sudo shutdown now=. Shutdown supersedes teardown — do *not* also drop the teardown sentinel. + +If =emacsclient= isn't resolvable or the daemon is down, the gate can't run — abort the shutdown, fall back to a normal wrap, and say so. Don't power off on an unverifiable gate. + +* Common Mistakes to Avoid + +1. *Skipping Step 1 (Summary)* — the file becomes the record; an empty Summary makes it hard to scan at catch-up +2. *Vague description in filename* — =2026-04-20-updates.org= is useless next to =2026-04-20-13-45-docs-ai-migration.org= +3. *=git add .= without review* — drags in unrelated dirty state +4. *=session:= prefix in commit message* — explicitly forbidden; use real change categories +5. *Claude-tooling references in commit message* — describes tooling, not what shipped +6. *Forgetting to push to all remotes* — check =git remote -v=, push to each +7. *Leaving =.ai/session-context.org= in place* — its presence means "interrupted session", confuses next startup +8. *Long preachy valediction* — brief beats thorough +9. *Leaving runtime/generated files dirty without gitignoring them* — pollutes every future =git status= and erodes trust in "working tree clean" as a signal. Fix =.gitignore= during the wrap, not later. +10. *Treating "was dirty at session start, still dirty now" as fine by default* — that's how a forgotten commit from two sessions ago turns into "carryover" for two weeks. Every pre-existing dirty file needs an active resolution recommendation this session. Deferral is allowed only with an explicit user choice, logged in the valediction. + +* Validation Checklist + +Before considering wrap-up complete: + +- [ ] =.ai/session-context.org= =* Summary= section populated +- [ ] The Summary ends with the =KB: promoted N / consulted yes-no= line (promotion check ran) +- [ ] File renamed to =.ai/sessions/YYYY-MM-DD-HH-MM-description.org= +- [ ] =.ai/session-context.org= no longer exists +- [ ] =todo-cleanup.el= ran — hygiene pass + =--convert-subtasks= + =--archive-done= + =--sync-child-priority= (if =todo.org= exists at project root) +- [ ] =lint-org.el= ran on =todo.org= — mechanical fixes applied, judgments appended to follow-ups file (if =todo.org= exists) +- [ ] Any orphan-planning-line warnings reviewed (fix or accept) +- [ ] Inbox carries nothing but expected pipeline artifacts (=.gitkeep=, =lint-followups.org=, =PROCESSED-*= prefixes), OR each remaining handoff has an explicit deferral logged in the valediction +- [ ] Linear Dev-Review sweep ran; any merged-PR tickets moved to Done or PM Acceptance (skip if project doesn't use Linear) +- [ ] Template-sync churn committed as its own =chore: sync .ai tooling from templates= (consuming projects only; skipped in rulesets), or surfaced if a synced path didn't match canonical +- [ ] After wrap-up commit + push, =git status --short= is empty OR every remaining line has an explicit user-deferred decision logged in the valediction +- [ ] Each leftover was investigated and the user saw a concrete resolution recommendation +- [ ] Runtime artifacts added to =.gitignore=, follow-up commit pushed, =git status= re-verified +- [ ] Forgotten changes committed in a follow-up and pushed +- [ ] Pre-existing dirty files resolved (committed / stashed / discarded / moved to a feature branch) or explicitly deferred with a one-line reason in the valediction +- [ ] Current branch pushed to ALL remotes (verified with =git remote -v=) +- [ ] All other local branches with a tracking upstream pushed to their remote +- [ ] Any untracked-upstream branches surfaced for manual =git push -u= +- [ ] Step 6 teardown matches the trigger phrase: no-teardown leaves the buffer; teardown drops only =/tmp/ai-wrap-teardown-<project>=; shutdown gates on =cj/ai-term-live-count= = 1 and drops only =/tmp/ai-wrap-shutdown-<project>= +- [ ] No teardown/shutdown sentinel was dropped before commit + push was verified +- [ ] Shutdown aborted (fell back to normal wrap, logged in the valediction) when another =aiv-*= session was live or the gate couldn't run +- [ ] Commit message follows format (no =session:=, no Claude attribution) +- [ ] Valediction delivered (brief, specific, warm) |
