diff options
Diffstat (limited to '.ai')
45 files changed, 4889 insertions, 255 deletions
diff --git a/.ai/metrics/work-the-backlog.jsonl b/.ai/metrics/work-the-backlog.jsonl index 1067b3a..88c6764 100644 --- a/.ai/metrics/work-the-backlog.jsonl +++ b/.ai/metrics/work-the-backlog.jsonl @@ -3,3 +3,13 @@ {"ts":"2026-07-02T05:22:11-04:00","run_id":"c726f526-2e35-4513-b25c-18ef61061333","project":"rulesets","caller":"speedrun","task":"template-sync-gitignored-only-changes","outcome":"implemented-committed","defer_reason":"","upfront_decision":true,"wall_clock_s":188,"commit_sha":"ed75d3c","review_findings":0} {"ts":"2026-07-02T05:58:16-04:00","run_id":"a48f2977-4493-48a3-9238-9b2f5ff5383b","project":"rulesets","caller":"loop","task":"inbox-send-filename-collision-fix","outcome":"implemented-committed","defer_reason":"","upfront_decision":false,"wall_clock_s":300,"commit_sha":"8099377","review_findings":0} {"ts":"2026-07-02T05:58:16-04:00","run_id":"a48f2977-4493-48a3-9238-9b2f5ff5383b","project":"rulesets","caller":"loop","task":"page-me-notify-info-level","outcome":"implemented-committed","defer_reason":"","upfront_decision":false,"wall_clock_s":120,"commit_sha":"a6b534f","review_findings":0} +{"ts":"2026-07-23T23:55:16-05:00","run_id":"711400a5-c6bc-4684-b928-6b818d0d33f2","project":"rulesets","caller":"speedrun","task":"inbox-send phantom empty handoff","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"0f91a8e","review_findings":1} +{"ts":"2026-07-23T23:57:59-05:00","run_id":"711400a5-c6bc-4684-b928-6b818d0d33f2","project":"rulesets","caller":"speedrun","task":"inbox-send two smaller defects","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"a053e9d","review_findings":0} +{"ts":"2026-07-24T00:02:14-05:00","run_id":"711400a5-c6bc-4684-b928-6b818d0d33f2","project":"rulesets","caller":"speedrun","task":"lint-org todo-format checkers fire on specs","outcome":"implemented-committed","upfront_decision":true,"commit_sha":"c38bab9","review_findings":0} +{"ts":"2026-07-24T00:02:49-05:00","run_id":"711400a5-c6bc-4684-b928-6b818d0d33f2","project":"rulesets","caller":"speedrun","task":"notes.org template four lint flags","outcome":"already-satisfied","defer_reason":"already-satisfied","upfront_decision":false,"commit_sha":"","review_findings":0} +{"ts":"2026-07-24T01:41:14-05:00","run_id":"sentry-fire2-1784875274","project":"rulesets","caller":"loop","task":"cj-remove-block over-deletion + unsafe write","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"17f5d48","review_findings":0} +{"ts":"2026-07-24T01:43:52-05:00","run_id":"sentry-fire2","project":"rulesets","caller":"loop","task":"route_recommend duplicate-name tier downgrade","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"1b0f284","review_findings":0} +{"ts":"2026-07-24T02:38:48-05:00","run_id":"sentry-fire3","project":"rulesets","caller":"loop","task":"audit.bats flaky teardown","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"7f45d4b","review_findings":0} +{"ts":"2026-07-24T03:39:16-05:00","run_id":"sentry-fire4","project":"rulesets","caller":"loop","task":"todo-cleanup missing backup","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"0686784","review_findings":0} +{"ts":"2026-07-24T04:36:09-05:00","run_id":"sentry-fire5","project":"rulesets","caller":"loop","task":"attachment filename sanitization","outcome":"deferred-verify","defer_reason":"needs-deliberation","upfront_decision":false,"commit_sha":"","review_findings":0} +{"ts":"2026-07-24T07:38:47-05:00","run_id":"sentry-fire8","project":"rulesets","caller":"loop","task":"bin/ lint coverage gap","outcome":"implemented-committed","upfront_decision":false,"commit_sha":"f91feef","review_findings":1} diff --git a/.ai/notes.org b/.ai/notes.org index 71d23dc..828fde3 100644 --- a/.ai/notes.org +++ b/.ai/notes.org @@ -61,6 +61,8 @@ This section tracks decisions that need Craig's input before work can proceed. ** Current Reminders +- =[2026-07-27]= Finish the context-engineering rightsizing — Craig's explicit ask at wrap. Surface is 57,800 → 28,949 tokens; the remaining work needs *his decisions*, not execution: =verification.md= (C1 — its honesty core vs the Opus 5 over-verification warning), =interaction.md= (3,828 tok, largest remaining), the TDD rationalization table (cut or keep), and D3 the gate separation (which approval gates are preference vs guardrail). Task: "Finish context-engineering rightsizing" in todo.org. Docs in =working/context-engineering-rightsizing/= are one commit behind — reconcile them first. + - =[2026-07-14]= Review the sentry spec (docs/specs/2026-07-14-sentry-workflow-spec.org) — Craig's explicit ask at wrap: strongly suggest he reviews it before ending the next session. All 12 review findings and 10 decisions are resolved and folded in; the spec is open in his Emacs; the READY flip and the [#B] build task both wait on his deep read. ** Instructions for This Section @@ -78,10 +80,11 @@ Format: :COMMIT_AUTONOMY: yes :LOOP_MAY_COMMIT: yes +:SENTRY_MAY_IMPLEMENT: yes :LAST_SPEC_SORT: 2026-07-02 Markers maintained by workflows to record when they last ran. Read by other workflows that gate their behavior on freshness. -:LAST_AUDIT: 2026-07-04 -:LAST_INBOX_PROCESS: 2026-07-18 (10 handoffs: triage-intake redesign + birthdays feature applied; todo-cleanup seal, planning-line strip, colloquialisms convention filed; knowledge-base roam URL fixed; 2 website FYIs + 1 archsetup FYI acknowledged) +:LAST_AUDIT: 2026-07-20 (open set current — this session's shipped work (working/temp, triage-source-activation, silent-until-signal, suspend detach) closed as it went; sentry cluster consolidated (merged the /schedule tasks, added cross-host-coordination); nothing shipped-but-open per git reconcile. Live finding: the Polyglot + Subprojects scouting tasks are SCHEDULED 2026-07-20 and due.) +:LAST_INBOX_PROCESS: 2026-07-25 (consolidated home + work Claude-to-Codex MCP registry proposals into one [#B] parked spec decision; memory auditor split from the registry work) Format: one =:MARKER: YYYY-MM-DD= line per workflow. Workflows overwrite their own marker on completion. diff --git a/.ai/protocols.org b/.ai/protocols.org index 5cd69d4..b291d9e 100644 --- a/.ai/protocols.org +++ b/.ai/protocols.org @@ -84,7 +84,7 @@ Do NOT estimate, guess, or rely on memory. Just run the command. It takes one se Every session pulls rulesets first, then the local project repo. Rulesets carries the canonical behavioral rules and =.ai/= templates (the old =claude-templates= repo is folded in as a subtree at =rulesets/claude-templates/=); the project pull lands commits pushed from other machines or teammates since the last session. -Resolve any dirty-tree or merge issue at each step before moving on. Both pulls run as =git pull --ff-only= (or =git merge --ff-only= against the fetched upstream), so anything non-trivial — non-fast-forward history, dirty working tree, diverged branches — aborts. Surface the state and stop. Never auto-stash, auto-merge, or auto-rebase; the user resolves the conflict before further work. +Resolve any sync-blocking tree or merge issue at each step before moving on. The shared =git-worktree-gate sync-safe= policy permits untracked deliveries beneath =inbox/= so receiving a handoff never prevents another project from refreshing rulesets; every staged or tracked change, dirty submodule, Git operation in progress, or untracked path outside =inbox/= blocks. Both pulls run as =git pull --ff-only= (or =git merge --ff-only= against the fetched upstream), so non-fast-forward history and diverged branches also abort. Surface the state and stop. Never auto-stash, auto-merge, or auto-rebase; the user resolves the conflict before further work. Mechanics live in =startup.org= Phase A.0. The rule lives here because it governs the very first action of every session: load the freshest behavioral rules and templates before anything else runs. @@ -106,7 +106,7 @@ The epoch is baked into the id by the spawner, never minted inside =session-cont Resolve the path with =.ai/scripts/session-context-path= rather than hardcoding =.ai/session-context.org=; it prints the right path for the current =AI_AGENT_ID=. Fall back to =.ai/session-context.org= if the script isn't present (older checkouts mid-sync). Everything below — the record/recovery purpose, the update triggers, the startup existence check, the wrap-up rename — operates on that resolved path. The prose says "session-context.org" as the default name; read it as "the resolved active path" when =AI_AGENT_ID= is set. -A helper instance (a second agent running in this project while a primary session is live) follows a different contract: it skips the pulls and rsync, makes only scoped single-heading edits to shared files, leaves all git mutation to the primary, and wraps up by archiving its own context file without committing. The full rules — read/write tiers, data-integrity, light startup, helper wrap-up — live in [[file:workflows/helper-mode.org][workflows/helper-mode.org]]. A session is a helper only when something routes it there (the =ai --helper= launcher, startup's roster check, or an explicit "you are a helper" instruction); the routing itself ships behind the helper-instance feature gate and isn't live yet. +A helper instance (a second agent running in this project while a primary session is live) follows a different contract: it skips the pulls and rsync, makes only scoped single-heading edits to shared files, leaves all git mutation to the primary, and wraps up by archiving its own context file without committing. The full rules — read/write tiers, data-integrity, light startup, helper wrap-up — live in [[file:workflows/helper-mode.org][workflows/helper-mode.org]]. A session is a helper only when something routes it there: the =ai --helper= launcher (live — it checks the roster, assigns the id, and opens the helper in its own tmux window) or an explicit "you are a helper" instruction. Startup's roster check is *not* built, so a bare =claude= launched into a project that already has a live session will run full primary startup regardless. Launch helpers with =ai --helper=. This file serves two purposes with one mechanism: 1. *Crash recovery* — if the session dies mid-work, the live file is all that's left. On 2026-01-22 a session crashed during a 20-minute design discussion and all context was lost because this file wasn't being updated. @@ -187,6 +187,8 @@ Canonical rule: =~/code/rulesets/claude-rules/cross-project.md=. Every in-progress task that produces files (drafts, source documents, diagrams, scripts, sub-deliverables) gets a dedicated subdirectory under =<project-root>/working/=, named after the task. All artifacts for that task live in that subdirectory until the task is marked done. +=working/= is version-controlled from creation — it's the tracked home of in-progress work, never gitignored. Filing on completion *reorganizes* durable artifacts into permanent homes; it doesn't mark when they became durable (they were durable, and tracked, from the start). Genuinely disposable artifacts go in a gitignored =temp/= (or =/tmp=), never =working/=; the install tooling ignores =temp/= in both track and gitignore modes. + When the task ships, files are **renamed individually** (standard form: =YYYY-MM-DD-<task-slug>-<descriptor>.<ext>=) and **moved flat** into the appropriate permanent home (typically =assets/= or an area-specific =<area>/assets/=). The working subdirectory is then empty and gets deleted. ***Never rename the directory itself as a substitute for filing.*** The point is to keep =assets/= flat-searchable — a nested =assets/old-tech-deck-2026/slide.png= is harder to find than =assets/2026-05-18-tech-deck-vol2-slide-04-diagram.png=. @@ -205,6 +207,8 @@ Check =inbox/= at every task boundary (after finishing a unit of work, before re Exit 1 means handoffs are pending — process them per =inbox.org= process mode. For each accepted handoff, the act-vs-file rule: *act now* when it's clear, bounded, low-risk, in-scope, and cheaper than deferring — just do it, no asking; *file* otherwise — ask first, with filing as option 1 and "do it now" as option 2; *ask* if unsure. Exception: a proposal to change a shared asset (template workflow, rule, skill, synced script) or a substantive convention never silently acts now — it goes through the inbox engine's skeptical review and its approval (or park) step. Always reply to a handoff's sender (confirm on accept, the why on reject). Full process, the reply discipline, and the opt-in background-monitor =/loop= recipe live in =inbox.org= monitor mode. +A machine-global =Stop= hook (=inbox-boundary-check.sh=) backs this rule so it isn't prose-only. When handoffs are pending it blocks the turn once and injects the count, so a task boundary can't pass with items unseen. It soft-nudges rather than hard-blocks: on the harness re-entry it steps aside, so a mid-task pause to ask "what's next" is never hijacked into inbox processing. The rule above still governs what to do with the items; the hook only makes sure you look. + ** Recursive Reads — Honor =.aiignore= Before a naive recursive read or glob of a project tree (file inventories, "what's in this repo", broad greps), skip the noise: dependency trees (=node_modules/=, =.venv/=), build output (=dist/=, =build/=, =coverage/=), language caches (=__pycache__/=, =.pytest_cache/=, =*.pyc=), editor/OS cruft, and generated token/OAuth artifacts. These waste tokens and skew project summaries even when gitignored — a recursive read sees the disk, not git. @@ -246,13 +250,29 @@ Execute the wrap-up workflow (details in Session Protocols section below): Execute the suspend workflow ([[file:workflows/suspend.org][suspend.org]]): a capture-only mid-session pause for an abrupt departure. It appends a resume-weighted =SUSPENDED= entry to the Session Log, notes uncommitted work, and LEAVES =.ai/session-context.org= in place so the next startup resumes from it — no archive, no teardown, no valediction. The capture-only counterpart to "wrap it up" (which ends + archives + tears down) and to =/flush= (which prompts =/clear= and resumes the same session). "I need to go" is broad — if it reads as a conversational aside, confirm before suspending. +* Colloquialisms and Expansions + +Shorthand phrases Craig uses that expand to a defined action the agent applies without asking. The set is extensible: a project may add its own entries, and new shared shorthands land here. + +** "the list": the Before-Close Queue + +"Put X on the list" or "add X to the list" appends X to the Before-Close Queue, a FIFO queue of tasks and actions to finish before the session closes. Work it oldest-first at wrap-up, before teardown (=wrap-it-up.org= Step 1 works it before finalizing the Summary), and surface anything unfinished in the valediction rather than dropping it. + +The queue lives in the session anchor (=.ai/session-context.org=) under a =* Before-Close Queue= heading. Create the heading on the first "put it on the list" if it's absent, then append one line per item. It's session-scoped: it resets when the anchor is archived at wrap. Anything that must outlive the session is a =todo.org= task instead, not a list item. + +** "tell <project> <message>": cross-project handoff + +"Tell <project> <message>" drops the message in that project's =inbox/= via =inbox-send= (=python3 .ai/scripts/inbox-send.py <project> --text "<message>"=), the sanctioned cross-project handoff. Never write another project's =todo.org= or =inbox/= directly. Resolve =<project>= the way =inbox-send= does (basename match, dots stripped); if it's ambiguous, ask which project rather than guessing. + * User Information ** Calendar Management Three ways to access Craig's calendars: Google Calendar MCP (preferred, both personal + work accounts), gcalcli (fallback, personal only), Emacs org files (read-only viewer). -For tool recipes, authentication details, and credentials, see [[file:references/calendar-reference.org][calendar-reference.org]]. +For tool recipes and account details, read the calendar workflows in =.ai/workflows/=: =add-calendar-event.org=, =edit-calendar-event.org=, =delete-calendar-event.org=, =read-calendar-events.org=. They carry the MCP tool names, both account ids, the gcalcli fallback, and the conflict-check discipline. + +Credentials are needed only for a re-auth Craig performs himself. The MCP bundle's =mcp/README.org= in the rulesets repo is the authority: =gcp-oauth.keys.json= is gitignored and regenerated at install from a base64 var in the bundle, never committed. Named in prose rather than linked, because that path isn't synced into consuming projects. ** GPG Keys @@ -356,9 +376,19 @@ Craig runs a pure Wayland setup (Hyprland) and avoids XWayland/Xorg apps. - Clipboard: Use =wl-copy= and =wl-paste= (NOT =xclip= or =xsel=) - Window management: Use Hyprland commands (NOT =xkill=, =xdotool=, etc.) - Prefer Wayland-native tools over X11 equivalents -- Open URLs in browser: Use =google-chrome-stable "URL" &>/dev/null &= - - The =&>/dev/null &= is required to detach the process and suppress output - - Without it, the command may appear to hang or produce no result +- Open URLs in browser: invoke Chrome directly — never =xdg-open=, which returned success in a home session on 2026-07-26 while no tab appeared. + + Chrome is normally already running, and in that case it hands the URL to the live session and exits immediately (rc 0), printing =Opening in existing browser session.= on *stdout*. So run it in the foreground and read that line as the confirmation the tab actually opened: + + #+begin_src bash + google-chrome-stable --new-tab "URL" + #+end_src + + Don't redirect stdout away while checking for that line — verified 2026-07-27 on ratio: with =2>/dev/null= the message still appears (it isn't stderr), and with =>/dev/null= it vanishes. + + Several URLs in one invocation open as separate tabs (=google-chrome-stable --new-tab "URL1" "URL2"=). Pass them as separate words or an array — the Bash tool runs zsh, which does not word-split an unquoted =$urls= variable, so a space-joined string arrives as one malformed argument (see the zsh note below). + + *Cold start.* If Chrome is *not* already running, the command becomes the browser process and blocks. Detach that case with =&>/dev/null &=, accepting that the confirmation line is discarded — there is no session to confirm into. Don't apply the detach form unconditionally: it suppresses the very output the warm path is verified by. *** Shell aliases (=ls= → =exa=) Craig's shell aliases =ls= to =exa=, which prints nothing to non-TTY pipes (e.g. when capturing =ls= output in a Bash tool call). The result looks like the directory is empty when it isn't. @@ -412,27 +442,29 @@ Full usage: =notify --help= or see =~/.local/bin/notify= - =atq= - list all scheduled alarms - =atrm [number]= - remove an alarm by its queue number -** Paging Craig — the agent pager +** Reaching Craig — the notification vocabulary -"Page me" has two channels; pick by where Craig is. Both work from any agent runtime — nothing here is Claude-specific. +Two channels, two trigger words. "page me" is the desktop, "text me" is the phone, "text and page me" is both. Pick by where Craig is, and default to both when a run can't tell. Both work from any agent runtime (nothing here is Claude-specific). The words are what Craig says; a run deciding on its own maps the same way (away run texts, at-desk run pages, unsure does both). -- *At his laptop/desktop* — desktop =notify ... --persist= (above). It reaches him on the machine and stays up until dismissed. +- *"page me" — at his laptop/desktop.* A desktop =notify ... --persist= that reaches him on the machine and stays up until dismissed. #+begin_src bash notify info "Title" "Message" --persist #+end_src -- *Away from his laptop/desktop* — page his phone over Signal with the *agent pager*: +- *"text me" — away from his machine.* A Signal push to his phone via =agent-text=: #+begin_src bash - agent-page "Message for Craig's phone" + agent-text "Message for Craig's phone" #+end_src - =agent-page= (in =~/.local/bin= via the rulesets install) sends from the dedicated pager identity (+15045173983, registered in velox's signal-cli) to Craig's Signal account UUID, firing a normal mobile push. On velox it sends directly; on any other tailnet machine it ssh-relays the send to velox. Verified end to end 2026-07-13. Never page Craig's phone *number* — it reads as unregistered in Signal's directory; the script already targets the UUID. + =agent-text= (in =~/.local/bin= via the rulesets install) sends from the dedicated Signal identity (+15045173983) to Craig's Signal account UUID, firing a normal mobile push. The account is registered on velox (primary) and ratio (linked device), so either sends directly; a machine without it ssh-relays to velox. Verified end to end 2026-07-13 (velox) and 2026-07-20 (ratio). Never target Craig's phone *number* (it reads as unregistered in Signal's directory); the script targets the UUID. + + Caveats: a relay from a non-linked machine needs velox up on the tailnet, and each device holding the account wants a periodic =receive= (the signal-receive timer handles that). The full runbook lives in rulesets =docs/design/=. - Caveats: velox must be up and on the tailnet (the script says so and names the desktop fallback when the relay fails), and the signal-cli account wants a periodic =receive= — both tracked on the rulesets Signal-pager task, which owns the full runbook. +- *"text and page me" — both.* Fire =agent-text= and =notify= together. The phone reaches him now, the desktop note waits for his return. This is the default when a run can't tell whether he's away. -On velox, Claude sessions may also have the *signal-mcp* tool (=send_message_to_user=, same pager identity) — fine to use there, but it exists only in velox's local MCP config, so =agent-page= is the portable habit. Do *not* use the old =page-signal= shell script (removed 2026-06-12). +On velox, Claude sessions may also have the *signal-mcp* tool (=send_message_to_user=, same identity), fine to use there, but it exists only in velox's local MCP config, so =agent-text= is the portable habit. The tool was named =agent-page= before 2026-07-20; a deprecated =agent-page= shim still delegates to =agent-text=. Do *not* use the old =page-signal= shell script (removed 2026-06-12). * Session Protocols @@ -547,12 +579,13 @@ When monitoring a long-running process (rsync, large downloads, builds, VM tests ** "Wrap it up" / "That's a wrap" / "Let's call it a wrap" -When Craig says any of these phrases (or variations), execute the wrap-up workflow: [[file:workflows/wrap-it-up.org][wrap-it-up.org]]. Four steps: +When Craig says any of these phrases (or variations), execute the wrap-up workflow: [[file:workflows/wrap-it-up.org][wrap-it-up.org]]. Five load-bearing steps: 1. *Finalize the Summary* in =.ai/session-context.org= (populate the 5 subsections from the Session Log) 2. *Rename* =.ai/session-context.org= → =.ai/sessions/YYYY-MM-DD-HH-MM-description.org= 3. *Git commit + push* to all remotes (see Git Commit Requirements) -4. *Valediction* — brief, warm, specific closing +4. *Certify the clean tree* with =git-worktree-gate certify=. Any remaining staged, unstaged, untracked, submodule, or in-progress-operation state blocks wrap entirely; report each path and the exact decision needed. There is no dirty-file deferral. +5. *Valediction* — brief, warm, specific closing, reachable only after certification The absence of =.ai/session-context.org= after wrap-up is the signal that the session ended cleanly. If the file is still there at the next session start, the previous session was interrupted. diff --git a/.ai/references/calendar-reference.org b/.ai/references/calendar-reference.org deleted file mode 100644 index 5791b08..0000000 --- a/.ai/references/calendar-reference.org +++ /dev/null @@ -1,66 +0,0 @@ -#+TITLE: Calendar Reference -#+AUTHOR: Craig Jennings - -Tool recipes, authentication, and credentials for Craig's calendar -setup. Three access methods, in order of preference. - -* Google Calendar MCP Server (preferred for all calendar operations) - -Craig has the =@cocal/google-calendar-mcp= MCP server configured at user scope (=~/.claude.json=). It provides full read/write access to Google Calendar via MCP tools. - -Two accounts are authenticated: -- *personal* — craigmartinjennings@gmail.com (primary: "Craig Google") -- *work* — craig.jennings@deepsat.com (primary: "Craig Deepsat") - -MCP tools available: -- =list-events=, =search-events=, =get-event= — read events -- =create-event=, =create-events= — add events -- =update-event= — modify events -- =delete-event= — remove events -- =list-calendars=, =list-colors= — calendar metadata -- =get-freebusy= — check availability -- =manage-accounts= — add/remove/list authenticated accounts -- =respond-to-event= — accept/decline invitations -- =get-current-time= — current time in any timezone - -Use =account_id: "personal"= or =account_id: "work"= to specify which account. - -Default calendar for adding events: "Craig Google" (personal account). - -Calendar workflows are available alongside this reference: add-calendar-event, edit-calendar-event, delete-calendar-event, read-calendar-events. - -If re-authentication is needed: -- Use the =manage-accounts= MCP tool with =action: "add"= and the account nickname -- OAuth credentials: =~/projects/homelab/assets/gcp-oauth.keys.json= -- Google Cloud app is in production mode (tokens don't expire after 7 days) -- See =~/projects/homelab/.ai/gcalcli-setup.org= for Google Cloud project details - -* gcalcli (fallback for personal account only) - -Craig has =gcalcli= installed via pipx, authenticated to his personal Google account only. - -#+begin_src bash -gcalcli agenda # upcoming events -gcalcli calw # weekly view -gcalcli add --title "..." --when "..." --duration "60" # add event -gcalcli search "..." # search events -gcalcli delete "..." # delete event -#+end_src - -Use =--calendar "Craig Google"= when adding events. - -gcalcli does NOT have access to the work (DeepSat) calendar. Use the MCP server for work calendar operations. - -If gcalcli needs re-authentication, credentials are stored in the homelab project: =~/projects/homelab/assets/gcalcli-client-secret.json.gpg= (GPG encrypted). - -* Emacs org files (read-only, for viewing schedules) - -Craig's calendars are at: =~/.emacs.d/data/*cal.org= (gcal.org, dcal.org, pcal.org) - -These files are **READ-ONLY** — NEVER add anything to them. - -Use this to: -- Check meeting times and schedules -- Verify when events occurred -- See what's upcoming -- Note: only updated periodically when Emacs is running — may be stale diff --git a/.ai/scripts/agent-lock b/.ai/scripts/agent-lock new file mode 100755 index 0000000..634412c --- /dev/null +++ b/.ai/scripts/agent-lock @@ -0,0 +1,248 @@ +#!/usr/bin/env bash +# agent-lock — a mkdir-atomic advisory lock for agent workflows. +# +# Why not flock: every Bash call an agent makes is its own short-lived shell, +# so an flock taken in one /loop turn is gone by the next. This helper persists +# the lock on disk between calls (an atomic mkdir is the acquire), and a crashed +# holder's lock self-clears via age-based staleness reclaim instead of wedging +# every later acquire. +# +# Serves both of sentry's locks (the single-runner lock and the roam-write +# lock); callers pass a name, never a path — the helper owns the path scheme. +# +# Usage: +# agent-lock acquire <name> [--ttl=SECONDS] [--wait[=SECONDS]] +# Atomic acquire. exit 0 on win (fresh, or reclaimed from a stale holder); +# exit 1 when a live lock already holds <name> (deferred — a note names the +# holder on stderr). --wait polls up to SECONDS (default 30) before +# deferring; without it, acquire is single-shot win-or-lose. --ttl records +# the staleness horizon in the lock's metadata (default below). +# agent-lock refresh <name> +# Heartbeat: re-touch a held lock's mtime so it stays young. A runner +# refreshes its own lock between passes, so a live run's lock is never older +# than one pass and the TTL sizes to the longest single pass. exit 1 if the +# lock is absent (nothing to refresh). +# agent-lock release <name> +# Remove the lock. Idempotent: exit 0 even if already free. +# agent-lock status <name> +# Print "free" | "held ..." | "stale ..." plus metadata. exit 0 (a query +# never fails on lock state). +# agent-lock path <name> +# Print the resolved lock-directory path without creating it. +# +# Lock home (the helper owns this; callers pass names only): +# $AGENT_LOCK_DIR/<name>/ when AGENT_LOCK_DIR is set (tests / advanced) +# $XDG_RUNTIME_DIR/agent-locks/<name>/ the tmpfs runtime dir /run/user/<uid> +# (host-local, out of every repo, +# cleared on reboot). XDG_RUNTIME_DIR is +# the standard handle for it and is set +# in sentry's interactive launch. +# ${XDG_CACHE_HOME:-~/.cache}/agent-locks/<name>/ fallback where no runtime +# dir exists (XDG_RUNTIME_DIR unset or +# unwritable — a headless/container box) +# +# tmpfs residence is deliberate: a lock under ~/org/roam would ride roam-sync's +# `git add -A` to the other machine as a phantom hold. Host-locality is by +# construction, and reboot clears any lock a crash left behind for free. +# +# Staleness is age-based on the metadata file's mtime versus the lock's own +# recorded TTL. Heartbeat re-touches the mtime; a reclaim is always surfaced, +# never silent. + +set -euo pipefail + +DEFAULT_TTL=600 # 10 min: sized to the longest single sentry pass, since a + # live runner heartbeats between passes and stays young. +DEFAULT_WAIT=30 # bounded-wait budget for --wait (capture-guard's shape). +WAIT_INTERVAL=3 # poll cadence while waiting on a busy lock. + +usage() { + echo "usage: agent-lock {acquire|refresh|release|status|path} <name> [--ttl=N] [--wait[=N]]" >&2 + exit 2 +} + +# Resolve the base directory that holds all lock dirs, per the home scheme above. +lock_base() { + if [ -n "${AGENT_LOCK_DIR:-}" ]; then + printf '%s\n' "$AGENT_LOCK_DIR" + elif [ -n "${XDG_RUNTIME_DIR:-}" ] && [ -d "$XDG_RUNTIME_DIR" ] && [ -w "$XDG_RUNTIME_DIR" ]; then + printf '%s/agent-locks\n' "$XDG_RUNTIME_DIR" + else + printf '%s/agent-locks\n' "${XDG_CACHE_HOME:-$HOME/.cache}" + fi +} + +# Validate a lock name: non-empty, no path separators (so a name can never +# escape the base dir). +valid_name() { + case "$1" in + ''|*/*|.|..) return 1 ;; + *) return 0 ;; + esac +} + +lock_dir() { printf '%s/%s\n' "$(lock_base)" "$1"; } +meta_path() { printf '%s/meta\n' "$(lock_dir "$1")"; } + +# Read a key from a lock's metadata file; empty if absent. +meta_get() { + local key="$1" file="$2" + [ -f "$file" ] || return 0 + sed -n "s/^${key}=//p" "$file" | head -n1 +} + +# Age of a lock in whole seconds, from the metadata mtime. +lock_age() { + local file="$1" mtime now + mtime=$(stat -c %Y "$file" 2>/dev/null) || return 1 + now=$(date +%s) + printf '%s\n' "$((now - mtime))" +} + +# True when a lock dir exists but its age exceeds its recorded TTL. +is_stale() { + local name="$1" file age ttl + file="$(meta_path "$name")" + [ -f "$file" ] || return 1 + age="$(lock_age "$file")" || return 1 + ttl="$(meta_get ttl "$file")" + [ -n "$ttl" ] || ttl="$DEFAULT_TTL" + [ "$age" -gt "$ttl" ] +} + +# Write the metadata file for a freshly-taken lock. +write_meta() { + local name="$1" ttl="$2" file + file="$(meta_path "$name")" + { + printf 'pid=%s\n' "$$" + printf 'host=%s\n' "$(uname -n)" + printf 'acquired=%s\n' "$(date +%Y-%m-%dT%H:%M:%S%z)" + printf 'ttl=%s\n' "$ttl" + } > "$file" +} + +# One-line holder description for surfaced notes. +holder_desc() { + local file="$1" + printf "pid=%s host=%s age=%ss ttl=%ss" \ + "$(meta_get pid "$file")" "$(meta_get host "$file")" \ + "$(lock_age "$file" 2>/dev/null || echo '?')" "$(meta_get ttl "$file")" +} + +# Attempt a single atomic acquire. exit 0 win, 1 busy (live holder). +try_acquire() { + local name="$1" ttl="$2" dir file + dir="$(lock_dir "$name")" + file="$(meta_path "$name")" + mkdir -p "$(lock_base)" + + if mkdir "$dir" 2>/dev/null; then + write_meta "$name" "$ttl" + return 0 + fi + + # Directory exists. Reclaim it if the holder is stale; otherwise it's busy. + if is_stale "$name"; then + # Claim the stale dir atomically before removing it. `mv` of a directory is + # atomic, so when two acquirers both see the lock stale, only one's rename + # of $dir succeeds — the other's fails because $dir is already gone, and it + # falls through to busy. Never `rm -rf $dir` directly: a plain remove lets + # the loser delete the winner's freshly-created lock and double-acquire. + local claimed="$dir.stale.$$" + if mv "$dir" "$claimed" 2>/dev/null; then + echo "agent-lock: reclaimed stale lock '$name' ($(holder_desc "$claimed/meta"))" >&2 + rm -rf "$claimed" + # mkdir stays the sole grant: a concurrent fresh acquirer may win here, + # in which case our mkdir fails and we correctly defer to it. + if mkdir "$dir" 2>/dev/null; then + write_meta "$name" "$ttl" + return 0 + fi + fi + fi + return 1 +} + +cmd_acquire() { + local name="$1"; shift + local ttl="$DEFAULT_TTL" wait_total=0 + while [ $# -gt 0 ]; do + case "$1" in + --ttl=*) ttl="${1#--ttl=}" ;; + --ttl) shift; ttl="${1:-}" ;; + --wait) wait_total="$DEFAULT_WAIT" ;; + --wait=*) wait_total="${1#--wait=}" ;; + *) usage ;; + esac + shift + done + case "$ttl" in ''|*[!0-9]*) usage ;; esac + case "$wait_total" in *[!0-9]*) usage ;; esac + + local elapsed=0 + while :; do + if try_acquire "$name" "$ttl"; then + exit 0 + fi + if [ "$elapsed" -ge "$wait_total" ]; then + echo "agent-lock: '$name' busy ($(holder_desc "$(meta_path "$name")")); deferring" >&2 + exit 1 + fi + local remaining=$((wait_total - elapsed)) step + step=$(( remaining < WAIT_INTERVAL ? remaining : WAIT_INTERVAL )) + sleep "$step" + elapsed=$((elapsed + step)) + done +} + +cmd_refresh() { + local name="$1" file + file="$(meta_path "$name")" + [ -f "$file" ] || exit 1 + # Re-stamp acquired and bump mtime so the age clock restarts. + local ttl; ttl="$(meta_get ttl "$file")"; [ -n "$ttl" ] || ttl="$DEFAULT_TTL" + write_meta "$name" "$ttl" + exit 0 +} + +cmd_release() { + local name="$1" dir + dir="$(lock_dir "$name")" + rm -rf "$dir" + exit 0 +} + +cmd_status() { + local name="$1" dir file + dir="$(lock_dir "$name")" + file="$(meta_path "$name")" + if [ ! -d "$dir" ]; then + echo "free $name" + exit 0 + fi + local state="held" + is_stale "$name" && state="stale" + echo "$state $name pid=$(meta_get pid "$file") host=$(meta_get host "$file") acquired=$(meta_get acquired "$file") ttl=$(meta_get ttl "$file") age=$(lock_age "$file" 2>/dev/null || echo '?')s" + exit 0 +} + +cmd_path() { + lock_dir "$1" + exit 0 +} + +[ $# -ge 1 ] || usage +subcmd="$1"; shift +[ $# -ge 1 ] || usage +name="$1"; shift +valid_name "$name" || usage + +case "$subcmd" in + acquire) cmd_acquire "$name" "$@" ;; + refresh) cmd_refresh "$name" ;; + release) cmd_release "$name" ;; + status) cmd_status "$name" ;; + path) cmd_path "$name" ;; + *) usage ;; +esac diff --git a/.ai/scripts/apkg-to-orgdrill.py b/.ai/scripts/apkg-to-orgdrill.py new file mode 100755 index 0000000..79e24a4 --- /dev/null +++ b/.ai/scripts/apkg-to-orgdrill.py @@ -0,0 +1,251 @@ +#!/usr/bin/env -S uv run --script +# /// script +# requires-python = ">=3.11" +# dependencies = [] +# /// +"""Convert an Anki .apkg deck into an org-drill file (inverse of flashcard-to-anki.py). + +The flashcard pipeline is otherwise one-directional (org-drill -> apkg). +Decks curated on the phone, and orphaned apkgs whose .org source was never +saved, can't get back into the org source of truth. This recovers them. + +Reading needs no third-party library: an apkg is a zip holding +collection.anki2 / .anki21 (an Anki sqlite db) plus a media blob, so stdlib +zipfile + sqlite3 suffice. genanki is only needed to write apkgs, not read +them. + +Mapping (mirrors flashcard-to-anki.py's parse/build, inverted): + - Deck name (from the apkg) -> #+TITLE: + - Note Front -> ** <Front> :drill: + - Note Back (HTML) -> entry body (<br> -> newlines, + &/</> unescaped, + <hr id="answer"> stripped) + - Note tag -> * <tag> section grouping + (best-effort: the tag is a slug, + so it won't round-trip to the exact + original section title — a human + retitles) + - A fresh :ID: UUID per card -> so the output is org-drill-valid + +GUIDs in flashcard-to-anki.py are derived from the Front text, not the +:ID:, so a deck regenerated from recovered org still matches existing phone +cards by Front. Only Front/Back (Basic) note types convert; other models +(cloze, etc.) are skipped with a warning rather than silently dropped. + +Usage: + apkg-to-orgdrill.py <input.apkg> # one <deck-slug>.org per deck in cwd + apkg-to-orgdrill.py <input.apkg> --output-dir DIR + apkg-to-orgdrill.py <input.apkg> --deck "Name" --output deck.org +""" +from __future__ import annotations + +import argparse +import json +import re +import sqlite3 +import sys +import tempfile +import uuid +import zipfile +from collections import OrderedDict +from dataclasses import dataclass +from pathlib import Path + +# Collection member names Anki uses, newest schema first. +COLLECTION_NAMES = ("collection.anki21", "collection.anki2") + +_BR_RE = re.compile(r"<br\s*/?>", re.IGNORECASE) +_ANSWER_HR_RE = re.compile(r'<hr id="answer">', re.IGNORECASE) +_MEDIA_RE = re.compile(r"<img\b|\[sound:|<audio\b|<video\b", re.IGNORECASE) + + +@dataclass +class Note: + deck: str + front: str + back_html: str + tag: str + + +def html_to_org_body(back_html: str) -> list[str]: + """Invert flashcard-to-anki.py's back-of-card HTML into org body lines. + + <br> (all spellings) and a stray answer <hr> become line breaks; the + entity unescape undoes escape_html, which escaped ``&`` first — so ``&`` + is unescaped last here, or a literally-escaped ``<`` in the source + would wrongly collapse to ``<``. + """ + if not back_html: + return [] + s = _ANSWER_HR_RE.sub("\n", back_html) + s = _BR_RE.sub("\n", s) + s = s.replace("<", "<").replace(">", ">").replace("&", "&") + return s.split("\n") + + +def _slug(title: str) -> str: + return re.sub(r"[^a-z0-9]+", "-", title.lower()).strip("-") + + +def _read_collection(db_path: Path) -> list[Note]: + con = sqlite3.connect(db_path) + try: + row = con.execute("SELECT decks, models FROM col LIMIT 1").fetchone() + if row is None: + raise ValueError("collection has no col row") + decks_json, models_json = row + decks = {int(k): v["name"] for k, v in json.loads(decks_json).items()} + models = { + int(k): [f["name"] for f in v["flds"]] + for k, v in json.loads(models_json).items() + } + + # A note's deck comes from its card; the Default deck (id 1) carries + # no cards from this pipeline, so it never shows up here. + nid_to_did: dict[int, int] = {} + for nid, did in con.execute("SELECT nid, did FROM cards"): + nid_to_did.setdefault(nid, did) + + notes: list[Note] = [] + for nid, mid, flds, tags in con.execute( + "SELECT id, mid, flds, tags FROM notes" + ): + field_names = models.get(mid) + if not field_names or "Front" not in field_names or "Back" not in field_names: + print( + f"apkg-to-orgdrill: skip note {nid} — model is not a Front/Back " + f"type (fields={field_names})", + file=sys.stderr, + ) + continue + fields = flds.split("\x1f") + fi, bi = field_names.index("Front"), field_names.index("Back") + front = fields[fi] if fi < len(fields) else "" + back_html = fields[bi] if bi < len(fields) else "" + + did = nid_to_did.get(nid) + if did is None: + continue # note with no card — orphan + deck = decks.get(did) + if deck is None: + continue + + tag_list = tags.split() + tag = tag_list[0] if tag_list else "drill" + + if _MEDIA_RE.search(back_html): + print( + f"apkg-to-orgdrill: note {nid} references media; org has no " + f"media path (left inline for a human to resolve)", + file=sys.stderr, + ) + notes.append(Note(deck=deck, front=front, back_html=back_html, tag=tag)) + return notes + finally: + con.close() + + +def read_apkg(path: Path) -> list[Note]: + """Read an .apkg and return its Front/Back notes. Raises on a malformed file.""" + with zipfile.ZipFile(path) as z: # BadZipFile if it isn't a zip + names = set(z.namelist()) + col_name = next((n for n in COLLECTION_NAMES if n in names), None) + if col_name is None: + raise ValueError(f"{path}: no collection.anki2/.anki21 inside the apkg") + with tempfile.TemporaryDirectory() as td: + db_path = Path(td) / col_name + db_path.write_bytes(z.read(col_name)) + return _read_collection(db_path) + + +def notes_to_org(notes: list[Note], deck_name: str, *, new_id=None) -> str: + """Render one deck's notes as an org-drill file in the house shape.""" + if new_id is None: + new_id = lambda: str(uuid.uuid4()) # noqa: E731 + groups: "OrderedDict[str, list[Note]]" = OrderedDict() + for n in notes: + groups.setdefault(n.tag, []).append(n) + + lines: list[str] = [f"#+TITLE: {deck_name}", ""] + for tag, group in groups.items(): + lines.append(f"* {tag}") + for n in group: + lines.append(f"** {n.front} :drill:") + lines.append(":PROPERTIES:") + lines.append(f":ID: {new_id()}") + lines.append(":END:") + lines.extend(html_to_org_body(n.back_html)) + lines.append("") + return "\n".join(lines).rstrip("\n") + "\n" + + +def convert(apkg_path: Path, *, new_id=None) -> "OrderedDict[str, str]": + """apkg -> {deck_name: org_text}, one entry per deck that has Front/Back cards.""" + by_deck: "OrderedDict[str, list[Note]]" = OrderedDict() + for n in read_apkg(apkg_path): + by_deck.setdefault(n.deck, []).append(n) + out: "OrderedDict[str, str]" = OrderedDict() + for deck, deck_notes in by_deck.items(): + out[deck] = notes_to_org(deck_notes, deck, new_id=new_id) + return out + + +def main() -> int: + parser = argparse.ArgumentParser( + description="Convert an Anki .apkg deck into an org-drill file.", + ) + parser.add_argument("input", type=Path, help="Path to the .apkg file.") + parser.add_argument("--deck", help="Only convert the deck with this exact name.") + parser.add_argument( + "--output", + type=Path, + help="Output .org path. Requires a single deck (use --deck to pick one).", + ) + parser.add_argument( + "--output-dir", + type=Path, + help="Directory for per-deck .org files (default: current directory).", + ) + args = parser.parse_args() + + input_path = args.input.expanduser().resolve() + if not input_path.is_file(): + print(f"error: {input_path} not found", file=sys.stderr) + return 1 + + by_deck = convert(input_path) + if args.deck: + by_deck = OrderedDict((k, v) for k, v in by_deck.items() if k == args.deck) + if not by_deck: + print(f"error: no deck named {args.deck!r} in {input_path}", file=sys.stderr) + return 1 + if not by_deck: + print(f"error: no Front/Back cards found in {input_path}", file=sys.stderr) + return 1 + + if args.output: + if len(by_deck) != 1: + print( + f"error: --output needs a single deck; {input_path} has " + f"{len(by_deck)} ({', '.join(by_deck)}). Use --deck or --output-dir.", + file=sys.stderr, + ) + return 1 + out = args.output.expanduser().resolve() + out.parent.mkdir(parents=True, exist_ok=True) + deck, org = next(iter(by_deck.items())) + out.write_text(org, encoding="utf-8") + print(f"wrote {out} ({org.count(':drill:')} cards, deck {deck!r})") + return 0 + + out_dir = (args.output_dir or Path.cwd()).expanduser().resolve() + out_dir.mkdir(parents=True, exist_ok=True) + for deck, org in by_deck.items(): + out = out_dir / f"{_slug(deck) or 'deck'}.org" + out.write_text(org, encoding="utf-8") + print(f"wrote {out} ({org.count(':drill:')} cards, deck {deck!r})") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/.ai/scripts/cj-remove-block.py b/.ai/scripts/cj-remove-block.py index 71c7b3d..d5137a3 100755 --- a/.ai/scripts/cj-remove-block.py +++ b/.ai/scripts/cj-remove-block.py @@ -16,8 +16,12 @@ Companion to the /respond-to-cj-comments skill and to cj-scan.py. from __future__ import annotations import argparse +import os import re +import shutil import sys +import tempfile +from datetime import datetime from pathlib import Path SRC_OPEN_RE = re.compile(r"^\s*#\+begin_src\s+cj:", re.IGNORECASE) @@ -57,12 +61,83 @@ def looks_like_cj_range(lines: list[str], start: int, end: int) -> tuple[bool, s f"Line {end} does not look like a #+end_src closing fence " f"(got: {last[:60]!r})" ) + + # The range must hold exactly ONE block. Checking only the first and last + # lines let a drifted range run from one block's opener to a *later* block's + # closer: validation passed and the removal silently deleted everything + # between, prose and headings included. Drift is the case this check exists + # for, so it has to look inside the range, not just at its ends. + for offset, line in enumerate(lines[start:end - 1], start=start + 1): + if SRC_CLOSE_RE.match(line): + return False, ( + f"Range {start}..{end} covers more than one cj block — " + f"a #+end_src appears at line {offset}, before the range ends. " + f"Re-scan for current line numbers; removing this range would " + f"delete everything between the two blocks." + ) + if SRC_OPEN_RE.match(line): + return False, ( + f"Range {start}..{end} covers more than one cj block — " + f"a second #+begin_src cj: appears at line {offset}. " + f"Re-scan for current line numbers." + ) return True, "" +def _backup(path: Path) -> Path: + """Copy path to /tmp before mutating it, mirroring lint-org.el's convention. + + These are Craig's org files. lint-org.el, the other tool that rewrites them, + leaves a /tmp copy before touching anything; this matches it so a bad edit is + always recoverable without reaching for git (which only reaches the last + commit, losing intra-session work). + """ + stamp = datetime.now().strftime("%Y%m%d-%H%M%S") + base = Path(tempfile.gettempdir()) / f"{path.name}.before-cj-remove.{stamp}" + # Never overwrite an earlier backup. The skill removes several annotations + # in quick succession, so a second-resolution stamp collides and the later + # copy would replace the earlier one with already-mutated content — losing + # the pre-session original the backup exists to preserve. + dest = base + n = 2 + while dest.exists(): + dest = base.with_name(f"{base.name}-{n}") + n += 1 + shutil.copy2(path, dest) + return dest + + +def _atomic_write(path: Path, text: str) -> None: + """Write text to path via a temp sibling and os.replace. + + A bare write_text truncates the target on open, so a mid-write failure left + the org file truncated with no complete copy on disk. Writing a temp sibling + and renaming means the file is either its old content or its new content, + never a partial. + """ + # Follow a symlink to the file it names. os.replace would otherwise swap the + # symlink itself for a regular file, leaving the real target holding the old + # content — the edit silently goes nowhere. Resolving also puts the temp + # sibling on the same filesystem as the real file, which os.replace needs. + path = path.resolve() + fd, tmp = tempfile.mkstemp(dir=path.parent, prefix=f".{path.name}.", suffix=".tmp") + os.close(fd) + tmp_path = Path(tmp) + # Carry the original's permissions across. mkstemp creates 0600, and + # defaulting to the umask instead widened a deliberately-restricted file + # (a 0600 org file came back 0644). + shutil.copymode(path, tmp_path) + try: + tmp_path.write_text(text, encoding="utf-8") + os.replace(tmp_path, path) + except BaseException: + tmp_path.unlink(missing_ok=True) + raise + + def remove_range(path: Path, start: int, end: int) -> None: """Read path, validate range looks like cj content, remove the range, write back.""" - text = path.read_text() + text = path.read_text(encoding="utf-8") had_trailing_newline = text.endswith("\n") lines = text.splitlines(keepends=False) @@ -77,7 +152,9 @@ def remove_range(path: Path, start: int, end: int) -> None: new_text += "\n" elif not new_lines and had_trailing_newline: new_text = "" - path.write_text(new_text) + + _backup(path) + _atomic_write(path, new_text) def main() -> int: diff --git a/.ai/scripts/flashcard-stats.py b/.ai/scripts/flashcard-stats.py index 1fa5afb..cb580ac 100755 --- a/.ai/scripts/flashcard-stats.py +++ b/.ai/scripts/flashcard-stats.py @@ -35,7 +35,12 @@ import re import sys from pathlib import Path -CARD_RE = re.compile(r"^\*\*\s+(.+?)\s+:drill:\s*$") +# A card is a level-2 heading whose trailing org tag block includes `drill`. +# Group 1 is the front, group 2 the tag block — so a curated card multi-tagged +# :fundamental:drill: still counts (it would silently drop under a :drill:$ +# anchor, undercounting the deck). HEADING_RE bounds a card's body. +CARD_RE = re.compile(r"^\*\*\s+(.+?)\s+(:[A-Za-z0-9_@#%:]+:)\s*$") +HEADING_RE = re.compile(r"^\*{1,2}\s") ANSWER_RE = re.compile(r"^\*\*\*\s+Answer\b") PROP_START_RE = re.compile(r"^\s*:PROPERTIES:\s*$") PROP_END_RE = re.compile(r"^\s*:END:\s*$") @@ -177,7 +182,8 @@ def parse_cards(lines: list[str]) -> tuple[list[dict], int]: n = len(lines) while i < n: m = CARD_RE.match(lines[i]) - if not m: + tags = [t for t in m.group(2).split(":") if t] if m else [] + if not (m and "drill" in tags): i += 1 continue heading = m.group(1).strip() @@ -188,7 +194,7 @@ def parse_cards(lines: list[str]) -> tuple[list[dict], int]: body_lines: list[str] = [] while i < n: line = lines[i] - if line.startswith("* ") or CARD_RE.match(line): + if HEADING_RE.match(line): break if PROP_START_RE.match(line): prop_count += 1 diff --git a/.ai/scripts/flashcard-to-anki.py b/.ai/scripts/flashcard-to-anki.py index ca4c70b..e369fd8 100755 --- a/.ai/scripts/flashcard-to-anki.py +++ b/.ai/scripts/flashcard-to-anki.py @@ -10,8 +10,15 @@ Parses org-drill structure: - Top-level "* Section" headings become tags on every card under them. - Each "** Card name :drill:" entry becomes a card. Front = heading - text (sans :drill: tag). Back = entry body with newlines converted + text (sans the tag block). Back = entry body with newlines converted to <br>. + - A card may carry a second org tag ("** Card :fundamental:drill:"). + Any heading whose tag block includes `drill` is a card; the other + tags ride along as Anki tags next to the section tag, so a curated + subset stays grep-able in the source. --tag-filter <tag> emits only + the cards carrying that tag, and a subset deck built that way should + pass --guid-salt so its notes get their own GUID space (Anki dedupes + on GUID, so without it the subset imports empty against the full deck). Deck name defaults to the org #+TITLE: (so the phone deck reads as the curated title), falling back to the input basename when the source has @@ -27,6 +34,8 @@ Usage: flashcard-to-anki.py <input.org> flashcard-to-anki.py <input.org> --deck "My Deck Name" flashcard-to-anki.py <input.org> --output /path/to/deck.apkg + flashcard-to-anki.py <input.org> --tag-filter fundamental \ + --deck "DeepSat Fundamentals" --guid-salt fundamentals Requires genanki, which uv resolves automatically via the PEP 723 script metadata above. No venv or system install needed. @@ -47,6 +56,15 @@ import genanki ID_BASE = 1_500_000_000 ID_RANGE = 500_000_000 +# A card is any level-2 heading whose trailing org tag block includes `drill`. +# Group 1 is the front text, group 2 the colon-delimited tag block (e.g. +# ":fundamental:drill:") — so a curated subset can carry a second org tag +# (:fundamental:) and stay grep-able in the source without dropping from the +# full deck. HEADING_RE bounds a card's body at the next L1/L2 heading. +CARD_RE = re.compile(r"^\*\*\s+(.+?)\s+(:[A-Za-z0-9_@#%:]+:)\s*$") +HEADING_RE = re.compile(r"^\*{1,2}\s") +SECTION_RE = re.compile(r"^\*\s+(.+?)\s*$") + def stable_id(name: str, salt: str) -> int: """Derive a deterministic 32-bit id from `name` and a `salt`. @@ -120,33 +138,40 @@ def strip_org_metadata(body_lines: list[str]) -> list[str]: return cleaned -def parse(org_text: str) -> list[tuple[str, str, str]]: - """Return [(front, back_html, tag), ...] for every :drill: card.""" - cards: list[tuple[str, str, str]] = [] - current_section: str | None = None +def parse( + org_text: str, tag_filter: str | None = None +) -> list[tuple[str, str, list[str]]]: + """Return [(front, back_html, anki_tags), ...] for every :drill: card. - section_re = re.compile(r"^\*\s+(.+?)\s*$") - card_re = re.compile(r"^\*\*\s+(.+?)\s+:drill:\s*$") + A card is any level-2 heading whose trailing org tag block includes + `drill`. Non-drill org tags on the heading (e.g. :fundamental:) ride + along as Anki tags next to the section tag, so a curated subset stays + grep-able in the source. When `tag_filter` is set, only cards carrying + that org tag are returned (the subset-deck path). + """ + cards: list[tuple[str, str, list[str]]] = [] + current_section: str | None = None lines = org_text.splitlines() i = 0 while i < len(lines): line = lines[i] - sec = section_re.match(line) + sec = SECTION_RE.match(line) if sec: current_section = sec.group(1).strip() i += 1 continue - card = card_re.match(line) - if card: - front = card.group(1).strip() + m = CARD_RE.match(line) + tags = [t for t in m.group(2).split(":") if t] if m else [] + if m and "drill" in tags: + front = m.group(1).strip() body_lines: list[str] = [] i += 1 while i < len(lines): nxt = lines[i] - if nxt.startswith("* ") or card_re.match(nxt): + if HEADING_RE.match(nxt): break body_lines.append(nxt) i += 1 @@ -156,8 +181,17 @@ def parse(org_text: str) -> list[tuple[str, str, str]]: while body_lines and not body_lines[-1].strip(): body_lines.pop() back_html = "<br>".join(escape_html(ln) for ln in body_lines) - tag = section_to_tag(current_section) if current_section else "drill" - cards.append((front, back_html, tag)) + + org_tags = [t for t in tags if t != "drill"] + if tag_filter and tag_filter not in org_tags: + continue + anki_tags: list[str] = [] + if current_section: + anki_tags.append(section_to_tag(current_section)) + anki_tags.extend(org_tags) + if not anki_tags: + anki_tags = ["drill"] + cards.append((front, back_html, anki_tags)) continue i += 1 @@ -165,15 +199,28 @@ def parse(org_text: str) -> list[tuple[str, str, str]]: return cards -def build(cards: list[tuple[str, str, str]], deck_name: str) -> genanki.Deck: +def card_guid(front: str, guid_salt: str | None) -> str: + """GUID for a card's front. A salt gives a derived subset deck its own + GUID space so its notes don't collide with the full deck's (Anki dedupes + on GUID, which would otherwise import the subset empty). No salt is the + original behavior, so an unsalted deck's GUIDs and SRS state are untouched. + """ + return genanki.guid_for(guid_salt, front) if guid_salt else genanki.guid_for(front) + + +def build( + cards: list[tuple[str, str, list[str]]], + deck_name: str, + guid_salt: str | None = None, +) -> genanki.Deck: deck = genanki.Deck(stable_id(deck_name, "deck"), deck_name) model = make_model(deck_name) - for front, back, tag in cards: + for front, back, tags in cards: note = genanki.Note( model=model, fields=[front, back], - tags=[tag], - guid=genanki.guid_for(front), + tags=tags, + guid=card_guid(front, guid_salt), ) deck.add_note(note) return deck @@ -219,6 +266,16 @@ def main() -> int: help="Output .apkg path. Defaults to " "~/sync/phone/anki/<input-basename>.apkg.", ) + parser.add_argument( + "--tag-filter", + help="Emit only cards carrying this org tag (e.g. --tag-filter " + "fundamental for a curated subset deck).", + ) + parser.add_argument( + "--guid-salt", + help="Salt note GUIDs so a subset deck gets its own GUID space and " + "imports non-empty without disturbing the full deck's SRS state.", + ) args = parser.parse_args() input_path: Path = args.input.expanduser().resolve() @@ -231,12 +288,18 @@ def main() -> int: output_path: Path = (args.output or default_output_path(input_path)).expanduser().resolve() output_path.parent.mkdir(parents=True, exist_ok=True) - cards = parse(org_text) + cards = parse(org_text, tag_filter=args.tag_filter) if not cards: - print(f"error: no :drill: cards found in {input_path}", file=sys.stderr) + if args.tag_filter: + print( + f"error: no :drill: cards tagged :{args.tag_filter}: in {input_path}", + file=sys.stderr, + ) + else: + print(f"error: no :drill: cards found in {input_path}", file=sys.stderr) return 1 - deck = build(cards, deck_name) + deck = build(cards, deck_name, guid_salt=args.guid_salt) genanki.Package(deck).write_to_file(str(output_path)) print(f"wrote {output_path} ({len(cards)} cards, deck '{deck_name}')") return 0 diff --git a/.ai/scripts/inbox-send.py b/.ai/scripts/inbox-send.py index 1ebb636..663efcb 100755 --- a/.ai/scripts/inbox-send.py +++ b/.ai/scripts/inbox-send.py @@ -31,6 +31,7 @@ import os import re import shutil import sys +import tempfile from datetime import datetime from pathlib import Path @@ -48,7 +49,7 @@ def resolve_roots() -> list[Path]: config = Path.home() / ".claude" / "inbox-roots.txt" if config.is_file(): paths: list[Path] = [] - for line in config.read_text().splitlines(): + for line in config.read_text(encoding="utf-8").splitlines(): line = line.strip() if line and not line.startswith("#"): paths.append(Path(line).expanduser()) @@ -69,17 +70,28 @@ def discover_projects(roots: list[Path]) -> list[Path]: a specific project root (included directly if it qualifies). """ projects: list[Path] = [] + seen: set[Path] = set() + + def _add(p: Path) -> None: + # Dedupe on the resolved path: a roots config naming both a parent and + # one of its children would otherwise list the child project twice, at + # two different indices. + key = p.resolve() + if key not in seen: + seen.add(key) + projects.append(p) + for root in roots: if not root.is_dir(): continue if _is_project(root): - projects.append(root) + _add(root) continue for child in sorted(root.iterdir()): if not child.is_dir(): continue if _is_project(child): - projects.append(child) + _add(child) return projects @@ -194,6 +206,39 @@ def uniquify(dest: Path) -> Path: n += 1 +def _atomic_write(dest: Path, writer) -> None: + """Write to a temp file in dest's directory, then rename it into place. + + dest is another project's inbox/, and a direct write truncates the target + on open, so any mid-write failure (a full disk, an encoding error, an + interrupted process) leaves a zero-byte .org there. inbox-status counts + that phantom as a pending handoff and blocks a turn in the receiving + project over a file with no content and no sender (2026-07-23). Writing to + a temp sibling and os.replace-ing means the inbox only ever sees a complete + file. os.replace is atomic within one filesystem, and the temp sits in the + same directory as dest, so it is. + + `writer` receives the open temp path and fills it. On any failure the temp + is removed and the error re-raised, so a caught error never leaves debris. + """ + fd, tmp = tempfile.mkstemp( + dir=dest.parent, prefix=".inbox-send-", suffix=dest.suffix + ) + os.close(fd) + tmp_path = Path(tmp) + # mkstemp creates the temp 0600; give the delivered file the umask-default + # mode the old direct write produced, so inbox files stay readable as before. + umask = os.umask(0) + os.umask(umask) + os.chmod(tmp_path, 0o666 & ~umask) + try: + writer(tmp_path) + os.replace(tmp_path, dest) + except BaseException: + tmp_path.unlink(missing_ok=True) + raise + + def send_text( target_inbox: Path, message: str, @@ -209,7 +254,8 @@ def send_text( raise ValueError(f"could not derive a slug from text: {message!r}") filename = f"{now.strftime(TS_FILENAME_FMT)}-from-{source_name}-{slug}.org" dest = uniquify(target_inbox / filename) - dest.write_text(build_text_org(message, source_name, now.strftime(TS_DOC_FMT))) + body = build_text_org(message, source_name, now.strftime(TS_DOC_FMT)) + _atomic_write(dest, lambda p: p.write_text(body, encoding="utf-8")) return dest @@ -229,7 +275,7 @@ def send_file( ext = src_path.suffix filename = f"{now.strftime(TS_FILENAME_FMT)}-from-{source_name}-{slug}{ext}" dest = uniquify(target_inbox / filename) - shutil.copy2(src_path, dest) + _atomic_write(dest, lambda p: shutil.copyfile(src_path, p)) return dest @@ -310,7 +356,10 @@ def main() -> int: else: assert args.file is not None dest = send_file(target_inbox, args.file, source_name, args.name, now) - except (ValueError, FileNotFoundError) as exc: + except (ValueError, OSError) as exc: + # OSError covers FileNotFoundError (missing source), PermissionError + # (unreadable source), and any atomic-write failure — all should + # surface as the clean "inbox-send: <message>" error, never a traceback. print(f"inbox-send: {exc}", file=sys.stderr) return 1 diff --git a/.ai/scripts/inbox-status b/.ai/scripts/inbox-status index b917144..17031af 100755 --- a/.ai/scripts/inbox-status +++ b/.ai/scripts/inbox-status @@ -35,6 +35,7 @@ mapfile -t pending < <(find inbox -maxdepth 1 -type f \ ! -name '.gitkeep' \ ! -name 'lint-followups.org' \ ! -name 'PROCESSED-*' \ + ! -name '.inbox-send-*' \ -printf '%f\n' 2>/dev/null | sort) n=${#pending[@]} diff --git a/.ai/scripts/lint-org.el b/.ai/scripts/lint-org.el index 55727ef..33dc52f 100644 --- a/.ai/scripts/lint-org.el +++ b/.ai/scripts/lint-org.el @@ -38,7 +38,9 @@ ;; empty-heading bare stars with no title ;; malformed-priority-cookie [#x]-shaped token org rejected ;; level2-done-without-closed completed level-2 task with no CLOSED +;; task-missing-last-reviewed open level-2 task with no :LAST_REVIEWED: ;; subtask-done-not-dated level-3+ done sub-task still a DONE keyword +;; dated-log-heading-active-timestamp dated-log heading with a live SCHEDULED/DEADLINE ;; (anything else) surfaced as judgment with checker name ;; ;; Output format on stdout: @@ -74,6 +76,18 @@ The CLI defaults this to t (a linter reports, it doesn't write); `--fix' is what enables writes on a command-line run.") (defvar lo-current-file nil "Path of the file currently being processed.") + +(defun lo--spec-file-p () + "Non-nil when the current file lives under a docs/specs/ directory. +The four todo-format-family checkers encode todo.org completion conventions +and misfire on a spec: a spec's Decisions section legitimately carries a +level-2 DONE with no CLOSED cookie, and its review-history section carries +level-2 dated headings. docs/specs/ is the canonical spec home per the +docs-lifecycle rule, so a path segment match is the scope test. Link, +table, and structural checks still run on specs — only the todo-format +family is scoped out." + (and lo-current-file + (string-match-p "/docs/specs/" (expand-file-name lo-current-file)))) (defvar lo-followups-file nil "When non-nil, after a non-check run any judgment items are appended to this path as an org section dated today. The file is created if missing.") @@ -292,6 +306,52 @@ Craig-specific annotation marker rather than Babel src-block syntax." (lo--goto-line line) (looking-at-p "^[ \t]*#\\+begin_src[ \t]+cj:"))) +(defvar-local lo--matched-blocks-cache nil + "Cons of (TICK . REGIONS) memoizing `lo--matched-block-regions'. +TICK is the `buffer-chars-modified-tick' the regions were computed at, so a +fix applied mid-pass invalidates them.") + +(defun lo--matched-block-regions () + "Return ((BEGIN-LINE . END-LINE) ...) for every correctly paired block. +Scans lines directly rather than asking org, because org's own parser is what +mis-reads these blocks: a heading-shaped line inside a verbatim body reads as a +structural break and loses the open block. The scan applies org's real rule — +once a block is open, only its own `#+end_TYPE' closes it, so a nested +`#+begin_' or a foreign `#+end_' in the body is just text." + (let ((tick (buffer-chars-modified-tick))) + (if (eql (car lo--matched-blocks-cache) tick) + (cdr lo--matched-blocks-cache) + (let ((case-fold-search t) + (regions nil) (open-type nil) (open-line nil) (line 0)) + (save-excursion + (goto-char (point-min)) + (while (not (eobp)) + (setq line (1+ line)) + (let ((text (buffer-substring-no-properties + (line-beginning-position) (line-end-position)))) + (cond + (open-type + (when (string-match + (format "\\`[ \t]*#\\+end_%s[ \t]*\\'" + (regexp-quote open-type)) + text) + (push (cons open-line line) regions) + (setq open-type nil open-line nil))) + ((string-match "\\`[ \t]*#\\+begin_\\([^ \t\n]+\\)" text) + (setq open-type (match-string 1 text) + open-line line)))) + (forward-line 1))) + (setq lo--matched-blocks-cache (cons tick (nreverse regions))) + (cdr lo--matched-blocks-cache))))) + +(defun lo--in-matched-block-p (line) + "Non-nil when LINE sits within a correctly paired block, delimiters included. +org-lint reports `invalid-block' at the delimiter lines themselves, so the +range has to be inclusive for the suppression to reach them." + (cl-some (lambda (region) + (and (>= line (car region)) (<= line (cdr region)))) + (lo--matched-block-regions))) + (defun lo--handle-item (item) (let ((name (lo--checker-name item)) (line (lo--line item)) @@ -304,6 +364,13 @@ Craig-specific annotation marker rather than Babel src-block syntax." wrong-header-argument)) (lo--cj-comment-block-opener-p line)) nil) + ;; `invalid-block' on a block that is in fact correctly paired — the + ;; checker is org-lint's own, so this filters its output rather than + ;; fixing a local checker. A genuinely unterminated block isn't in any + ;; matched region, so it still reports. + ((and (eq name 'invalid-block) + (lo--in-matched-block-p line)) + nil) ((eq name 'item-number) (lo--apply-or-preview name line msg #'lo-fix-item-number)) ((eq name 'missing-language-in-src-block) @@ -525,6 +592,42 @@ the live file on the next `task-sorted'." "level-2 DONE/CANCELLED has no CLOSED date — add CLOSED: [YYYY-MM-DD Day]; task-sorted's aging step archives an undated completed task immediately")))))))) ;;; --------------------------------------------------------------------------- +;;; task-missing-last-reviewed check (claude-rules/todo-format.md) +;; +;; A task is stamped `:LAST_REVIEWED:' when it is *created*, not a review cycle +;; later. An agent filing a task has just written its body and graded its +;; priority, which is a review by any honest reading — so a fresh task that +;; carries no stamp reads as "never reviewed" and lands at the top of the next +;; staleness batch, where re-reviewing it is pure ceremony. Every task filed +;; during the 2026-07-23 sweep hit exactly that, which is what prompted the rule. +;; +;; Judgment-only, deliberately. The stamp's whole value is that its date is +;; true, and nothing here can know when an unstamped task was actually last +;; looked at. Auto-stamping today's date would convert a "nobody has reviewed +;; this" signal into a false "reviewed today" one — worse than the gap it +;; closes. Flag it; a human or the filing workflow supplies the honest date. +;; +;; Scope matches `task-review-staleness.sh' exactly (level-2, open keyword, +;; priority cookie), so the checker and the staleness count never disagree +;; about which headings are in the review pool. + +(defun lo--check-task-missing-last-reviewed () + "Flag an open level-2 task with a priority cookie and no `:LAST_REVIEWED:'." + (save-excursion + (goto-char (point-min)) + (let ((case-fold-search nil)) + (while (re-search-forward "^\\*\\* \\(TODO\\|DOING\\|VERIFY\\) \\[#[A-D]\\]" nil t) + (let ((hline (line-number-at-pos)) + (entry-end (save-excursion (outline-next-heading) (point)))) + (save-excursion + (forward-line 1) + (unless (re-search-forward "^[ \t]*:LAST_REVIEWED:[ \t]*[[0-9]" + entry-end t) + (lo--emit-judgment + 'task-missing-last-reviewed hline + "task has no :LAST_REVIEWED: — stamp it at creation with today's date (a task you just wrote and graded is reviewed); otherwise it enters the next staleness batch as never-reviewed")))))))) + +;;; --------------------------------------------------------------------------- ;;; level-3+ dated-header check (claude-rules/todo-format.md) ;; ;; The inverse of the level-2 check above. A completed sub-task — a heading at @@ -551,6 +654,41 @@ Emits one judgment item per offending heading (checker "level-3+ done sub-task should be a dated event-log entry (todo-format.md): run todo-cleanup.el --convert-subtasks to rewrite it"))))) ;;; --------------------------------------------------------------------------- +;;; dated-log heading with a stale active planning timestamp (todo-format.md) +;; +;; The mechanical backstop for the planning-line-strip rule. A dated event-log +;; heading (`<stars> YYYY-MM-DD Day @ ...', no TODO keyword) records completed +;; work — its date lives in the heading. An active `<...>' SCHEDULED or DEADLINE +;; left on it pins the entry to the agenda forever: org renders any headline with +;; an active planning timestamp, keyword or not, so a stale SCHEDULED shows as +;; weeks-overdue long after the work is done. Invisible to a keyword scan (no +;; TODO) and it survives --archive-done, so nothing else catches it. The +;; completion rewrite and todo-cleanup --convert-subtasks now strip the planning +;; line; this flags any that slipped through before that landed, the same way +;; subtask-done-not-dated backstops the depth rule. Judgment-only. + +(defun lo--check-dated-log-active-timestamp () + "Flag a dated event-log heading that still carries an active SCHEDULED/DEADLINE. +The heading matches `<stars> YYYY-MM-DD Day @ ...' with no TODO keyword; an +active `<...>' planning timestamp in its entry is the defect. An inactive +`[...]' timestamp is ignored (org doesn't render it on the agenda). Emits one +judgment item per offending heading (checker `dated-log-heading-active-timestamp')." + (save-excursion + (goto-char (point-min)) + (let ((case-fold-search nil)) + (while (re-search-forward + "^\\*+ [0-9]\\{4\\}-[0-9]\\{2\\}-[0-9]\\{2\\} [A-Za-z]+ @ " nil t) + (let ((hline (line-number-at-pos)) + (entry-end (save-excursion (outline-next-heading) (point)))) + (save-excursion + (forward-line 1) + (when (re-search-forward + "^[ \t]*\\(?:SCHEDULED\\|DEADLINE\\):[ \t]*<" entry-end t) + (lo--emit-judgment + 'dated-log-heading-active-timestamp hline + "dated-log heading carries an active SCHEDULED/DEADLINE — org renders any active planning timestamp (keyword or not), so it stays on the agenda as weeks-overdue; delete the planning line (todo-format.md)")))))))) + +;;; --------------------------------------------------------------------------- ;;; File processing (defun lo--backup (file) @@ -584,14 +722,22 @@ left unmodified and mechanical entries are recorded with :preview t." ;; After org-lint items: the custom table-standard scan. Runs on the ;; post-fix buffer; judgment-only, so order doesn't perturb fixes. (lo--check-tables) - ;; Same shape: flag level-2 dated headers (completion defects). - (lo--check-level2-dated-headers) - ;; Structural heading defects org-lint doesn't cover. + ;; Structural heading defects org-lint doesn't cover. These run on + ;; every org file, specs included. (lo--check-indented-headings) (lo--check-empty-headings) (lo--check-malformed-priority-cookies) - (lo--check-level2-done-without-closed) - (lo--check-subtask-done-not-dated) + ;; The todo-format family encodes todo.org completion conventions and + ;; misfires on a spec (a Decisions section's undated DONE, a + ;; review-history dated heading, a phases task with no LAST_REVIEWED). + ;; Scope them out of docs/specs/; link, table, and structural checks + ;; above still run there. + (unless (lo--spec-file-p) + (lo--check-level2-dated-headers) + (lo--check-level2-done-without-closed) + (lo--check-task-missing-last-reviewed) + (lo--check-subtask-done-not-dated) + (lo--check-dated-log-active-timestamp)) (when (and (not lo-check-only) (buffer-modified-p)) (save-buffer))) (with-current-buffer buf (set-buffer-modified-p nil)) diff --git a/.ai/scripts/route_recommend.py b/.ai/scripts/route_recommend.py index 7b36405..12ab132 100644 --- a/.ai/scripts/route_recommend.py +++ b/.ai/scripts/route_recommend.py @@ -71,6 +71,15 @@ def recommend(item: str, projects: list[str]) -> tuple[str | None, str]: if not projects: return (None, "none") + # Collapse identical names first. Projects are addressed by bare basename, so + # two projects sharing one across roots (~/code/notes, ~/projects/notes) arrive + # twice; both literal-match, and the tie test below then read that as ambiguity + # and downgraded a correct strong match to weak. Deduping here rather than in + # discover_destination_names protects every caller of the pure core, not just + # the CLI path. Order-preserving, and it collapses only identical names — two + # *different* projects matching is real ambiguity and still downgrades. + projects = list(dict.fromkeys(projects)) + item_lower = item.lower() item_tokens = _tokens(item) diff --git a/.ai/scripts/tests/agent-lock.bats b/.ai/scripts/tests/agent-lock.bats new file mode 100644 index 0000000..dbcffe1 --- /dev/null +++ b/.ai/scripts/tests/agent-lock.bats @@ -0,0 +1,214 @@ +#!/usr/bin/env bats +# +# Tests for claude-templates/.ai/scripts/agent-lock — a mkdir-atomic advisory +# lock helper for agent workflows (sentry's single-runner and roam-write +# locks). flock can't span an agent's tool calls: every Bash call is its own +# short-lived shell, so a flock dies with the call that took it. This helper +# persists the lock on disk between calls and self-clears after a crash via +# age-based staleness reclaim. +# +# Contract under test: +# agent-lock acquire <name> [--ttl=SECONDS] [--wait[=SECONDS]] +# exit 0 → acquired (fresh, or reclaimed from a stale prior holder). +# exit 1 → busy: a live lock holds <name>; deferred (note on stderr). +# exit 2 → usage error (bad/absent name, unknown subcommand). +# agent-lock refresh <name> → re-touch a held lock (heartbeat); exit 1 if absent. +# agent-lock release <name> → remove the lock; idempotent (exit 0 if already free). +# agent-lock status <name> → print free|held|stale + metadata; exit 0 (query). +# agent-lock path <name> → print the resolved lock dir path; does not create it. +# +# Staleness is age-based on the metadata file's mtime versus the lock's own +# recorded TTL, so a crashed holder's lock expires instead of wedging every +# later acquire. Heartbeat (refresh) re-touches the mtime, keeping a live +# holder's lock young. Every reclaim surfaces a note (never silent). +# +# Lock home: /run/user/<uid>/agent-locks/<name>/ (tmpfs: host-local, out of +# every repo, cleared on reboot), with ~/.cache/agent-locks/ as the fallback +# where no runtime dir exists. AGENT_LOCK_DIR overrides the base for tests and +# advanced callers; the helper otherwise owns the path scheme and callers pass +# only names. +# +# Strategy: AGENT_LOCK_DIR points every lock at a temp base, so tests never +# touch a real runtime dir. Staleness is exercised by aging the metadata +# file's mtime with `touch` rather than sleeping. + +SCRIPT="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/agent-lock" +BASH_BIN="$(command -v bash)" + +setup() { + TEST_DIR="$(mktemp -d -t agent-lock-bats.XXXXXX)" + LOCK_BASE="$TEST_DIR/locks" +} + +teardown() { + rm -rf "$TEST_DIR" +} + +lock() { + run env AGENT_LOCK_DIR="$LOCK_BASE" "$BASH_BIN" "$SCRIPT" "$@" +} + +# meta-file path for a lock name, for direct inspection / aging. +meta_of() { + printf '%s/%s/meta\n' "$LOCK_BASE" "$1" +} + +# ---- acquire: fresh win + metadata -------------------------------------- + +@test "acquire: fresh name wins (exit 0) and writes pid/host/timestamp/ttl" { + lock acquire job + [ "$status" -eq 0 ] + local meta; meta="$(meta_of job)" + [ -f "$meta" ] + grep -q "^pid=$$\|^pid=[0-9][0-9]*$" "$meta" + grep -q "^host=$(uname -n)$" "$meta" + grep -qE "^acquired=[0-9]{4}-[0-9]{2}-[0-9]{2}T" "$meta" + grep -qE "^ttl=[0-9]+$" "$meta" +} + +@test "acquire: honors an explicit --ttl in the metadata" { + lock acquire job --ttl=45 + [ "$status" -eq 0 ] + grep -q "^ttl=45$" "$(meta_of job)" +} + +# ---- acquire: contention (one winner) ----------------------------------- + +@test "acquire: a second acquire of a live lock defers (exit 1, note)" { + lock acquire job + [ "$status" -eq 0 ] + lock acquire job + [ "$status" -eq 1 ] + [[ "$output" == *job* ]] +} + +@test "acquire: two racing acquires yield exactly one winner" { + # Fire both without releasing; exactly one mkdir wins. + env AGENT_LOCK_DIR="$LOCK_BASE" "$BASH_BIN" "$SCRIPT" acquire race & p1=$! + env AGENT_LOCK_DIR="$LOCK_BASE" "$BASH_BIN" "$SCRIPT" acquire race & p2=$! + local r1=0 r2=0 + wait $p1 || r1=$? + wait $p2 || r2=$? + # One exits 0 (won), one exits 1 (deferred). + [ "$((r1 + r2))" -eq 1 ] +} + +# ---- release: frees the lock -------------------------------------------- + +@test "release: frees a held lock so the next acquire wins" { + lock acquire job + [ "$status" -eq 0 ] + lock release job + [ "$status" -eq 0 ] + [ ! -d "$LOCK_BASE/job" ] + lock acquire job + [ "$status" -eq 0 ] +} + +@test "release: is idempotent on an already-free lock (exit 0)" { + lock release never-held + [ "$status" -eq 0 ] +} + +# ---- staleness reclaim (surfaced, never silent) ------------------------- + +@test "acquire: reclaims a stale lock and surfaces the reclaim note" { + lock acquire job --ttl=1 + [ "$status" -eq 0 ] + # Age the metadata mtime well past the 1s TTL. + touch -d '1 hour ago' "$(meta_of job)" + lock acquire job --ttl=1 + [ "$status" -eq 0 ] + [[ "$output" == *reclaim* ]] + [[ "$output" == *job* ]] + # The reclaim installed fresh metadata (young again), not the aged holder's. + lock status job + [[ "$output" == *held* ]] + [[ "$output" != *stale* ]] +} + +@test "acquire: a lock inside its TTL is not stale (stays deferred)" { + lock acquire job --ttl=3600 + [ "$status" -eq 0 ] + lock acquire job --ttl=3600 + [ "$status" -eq 1 ] +} + +# ---- heartbeat (refresh keeps a live lock young) ------------------------ + +@test "refresh: re-touches a held lock so it is no longer stale" { + lock acquire job --ttl=1 + [ "$status" -eq 0 ] + touch -d '1 hour ago' "$(meta_of job)" + lock status job + [[ "$output" == *stale* ]] + lock refresh job + [ "$status" -eq 0 ] + lock status job + [[ "$output" == *held* ]] + [[ "$output" != *stale* ]] +} + +@test "refresh: an absent lock cannot be refreshed (exit 1)" { + lock refresh nothing + [ "$status" -eq 1 ] +} + +# ---- status query ------------------------------------------------------- + +@test "status: reports free for an unheld lock (exit 0)" { + lock status job + [ "$status" -eq 0 ] + [[ "$output" == *free* ]] +} + +@test "status: reports held with metadata for a live lock" { + lock acquire job --ttl=3600 + lock status job + [ "$status" -eq 0 ] + [[ "$output" == *held* ]] + [[ "$output" == *"host=$(uname -n)"* ]] +} + +# ---- path resolution: runtime dir home with cache fallback -------------- + +@test "path: resolves under AGENT_LOCK_DIR when set" { + lock path job + [ "$status" -eq 0 ] + [ "$output" = "$LOCK_BASE/job" ] + [ ! -d "$LOCK_BASE/job" ] # path does not create the lock +} + +@test "path: prefers the runtime dir home when no override is set" { + local rt="$TEST_DIR/run" + mkdir -p "$rt" + run env -u AGENT_LOCK_DIR XDG_RUNTIME_DIR="$rt" "$BASH_BIN" "$SCRIPT" path job + [ "$status" -eq 0 ] + [ "$output" = "$rt/agent-locks/job" ] +} + +@test "path: falls back to the cache home when no runtime dir exists" { + local home="$TEST_DIR/home" + mkdir -p "$home" + run env -u AGENT_LOCK_DIR -u XDG_RUNTIME_DIR -u XDG_CACHE_HOME \ + HOME="$home" "$BASH_BIN" "$SCRIPT" path job + [ "$status" -eq 0 ] + [ "$output" = "$home/.cache/agent-locks/job" ] +} + +# ---- usage errors ------------------------------------------------------- + +@test "usage: a missing name is a usage error (exit 2)" { + lock acquire + [ "$status" -eq 2 ] +} + +@test "usage: a name with a slash is rejected (exit 2)" { + lock acquire bad/name + [ "$status" -eq 2 ] +} + +@test "usage: an unknown subcommand is a usage error (exit 2)" { + lock frobnicate job + [ "$status" -eq 2 ] +} diff --git a/.ai/scripts/tests/flashcard-sync.bats b/.ai/scripts/tests/flashcard-sync.bats index 608a280..e6ffc21 100644 --- a/.ai/scripts/tests/flashcard-sync.bats +++ b/.ai/scripts/tests/flashcard-sync.bats @@ -6,6 +6,7 @@ setup() { SCRIPT_DIR="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)" SYNC="$SCRIPT_DIR/flashcard-sync" + STATS="$SCRIPT_DIR/flashcard-stats.py" TMP="$(mktemp -d)" } @@ -36,3 +37,27 @@ EOF [ "$status" -eq 1 ] [ ! -f "$HOME/sync/phone/anki/dirty.apkg" ] } + +@test "flashcard-stats: a multi-tagged :fundamental:drill: card still counts" { + # Regression guard: a curated card carrying a second org tag must not drop + # from the count. A :drill:$ anchor would have counted only one card here. + cat > "$TMP/multitag.org" <<'EOF' +#+TITLE: Multitag Test + +* Orbital Regimes +** What is LEO? :fundamental:drill: +:PROPERTIES: +:ID: c1 +:END: +Low Earth Orbit is the region below about 2000 kilometers. +** What is GEO? :drill: +:PROPERTIES: +:ID: c2 +:END: +Geostationary orbit sits at roughly 35786 kilometers of altitude. +EOF + run python3 "$STATS" "$TMP/multitag.org" + [ "$status" -eq 0 ] + [[ "$output" == *"Cards: 2"* ]] + [[ "$output" == *clean* ]] +} diff --git a/.ai/scripts/tests/inbox-status.bats b/.ai/scripts/tests/inbox-status.bats index bc8a734..27a497e 100644 --- a/.ai/scripts/tests/inbox-status.bats +++ b/.ai/scripts/tests/inbox-status.bats @@ -45,6 +45,18 @@ teardown() { [[ "$output" == *"0 pending"* ]] } +@test "inbox-status: ignores an in-flight .inbox-send-* temp file" { + mkdir "$TMP/inbox" + # inbox-send writes to a .inbox-send-* temp then renames it into place; + # during that window the temp must not read as a pending handoff, or a + # concurrent boundary check blocks on a file that's about to become real. + touch "$TMP/inbox/.inbox-send-abc123.org" + cd "$TMP" + run "$SCRIPT" + [ "$status" -eq 0 ] + [[ "$output" == *"0 pending"* ]] +} + @test "inbox-status: -q suppresses the per-item lines" { mkdir "$TMP/inbox" echo body > "$TMP/inbox/handoff.org" diff --git a/.ai/scripts/tests/test-lint-org.el b/.ai/scripts/tests/test-lint-org.el index 8e3e190..ceee209 100644 --- a/.ai/scripts/tests/test-lint-org.el +++ b/.ai/scripts/tests/test-lint-org.el @@ -193,6 +193,65 @@ real suspicious-language warning here #+end_src ") +;; invalid-block, false-positive case — a correctly paired example block whose +;; body holds a heading-shaped line. org's parser reads the `** ' inside the +;; verbatim body as a structural break, loses the open block, and flags BOTH +;; delimiters as "Possible incomplete block". +(defconst lo-test--verbatim-heading-block "\ +* Heading + +#+begin_example +** Feature Name or Topic +Body line. +#+end_example + +Trailing prose. +") + +;; invalid-block, literal-delimiter case — a paired src block whose body holds +;; a literal `#+end_example' plus a heading-shaped line. Only `#+end_src' +;; closes a src block, so all three findings here are false. +(defconst lo-test--literal-end-in-src "\ +* Heading + +#+begin_src text +#+end_example +** heading shaped +#+end_src +") + +;; invalid-block, uppercase-delimiter case — org accepts #+BEGIN_/#+END_ in +;; either case, and the pre-fix script flagged both delimiters here too. +(defconst lo-test--uppercase-verbatim-block "\ +* Heading + +#+BEGIN_EXAMPLE +** heading shaped +#+END_EXAMPLE +") + +;; invalid-block, genuine case — a block that really is never closed. The +;; suppression must not reach this one. +(defconst lo-test--unterminated-block "\ +* Heading + +#+begin_example +truly unterminated block body +") + +;; A genuinely unterminated block *after* a correctly paired one — verifies the +;; suppression is scoped per block rather than per file. +(defconst lo-test--paired-then-unterminated "\ +* Heading + +#+begin_example +** heading shaped +#+end_example + +#+begin_example +never closed +") + ;; Mixed fixture — each category once. (defconst lo-test--mixed "\ * Mixed @@ -392,6 +451,55 @@ suspicious-language judgment." (should (= 1 suspicious)))) ;;; --------------------------------------------------------------------------- +;;; invalid-block — false positives on correctly paired verbatim blocks + +(ert-deftest lo-verbatim-heading-block-emits-no-invalid-block () + "Normal: a paired example block containing a heading-shaped body line emits +no invalid-block judgment. Both delimiters are flagged by org-lint because the +parser treats the `** ' inside the verbatim body as a structural break." + (let* ((out (lo-test--run lo-test--verbatim-heading-block)) + (res (plist-get out :result)) + (judgments (lo-test--judgments (plist-get out :issues)))) + ;; File untouched, no fixes applied — suppression only, never a rewrite. + (should (equal lo-test--verbatim-heading-block res)) + (should (= 0 (plist-get out :fixes))) + (should-not (member 'invalid-block (lo-test--checkers judgments))))) + +(ert-deftest lo-literal-end-delimiter-in-src-emits-no-invalid-block () + "Boundary: a paired src block whose body holds a literal `#+end_example' and +a heading-shaped line emits no invalid-block judgment. Only `#+end_src' closes +a src block, so the interior delimiter is body text." + (let* ((out (lo-test--run lo-test--literal-end-in-src)) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should-not (member 'invalid-block (lo-test--checkers judgments))))) + +(ert-deftest lo-uppercase-verbatim-block-emits-no-invalid-block () + "Boundary: block delimiters are case-insensitive in org, so an uppercase +`#+BEGIN_EXAMPLE' pair is suppressed the same as a lowercase one." + (let* ((out (lo-test--run lo-test--uppercase-verbatim-block)) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should-not (member 'invalid-block (lo-test--checkers judgments))))) + +(ert-deftest lo-unterminated-block-still-emits-invalid-block () + "Error: a block that is never closed still emits its invalid-block judgment. +This is the finding the checker exists for — the suppression must not mask it." + (let* ((out (lo-test--run lo-test--unterminated-block)) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should (member 'invalid-block (lo-test--checkers judgments))))) + +(ert-deftest lo-invalid-block-suppression-is-scoped-per-block () + "Boundary: a paired block and an unterminated block in the same file — the +paired one is suppressed and the unterminated one still reports. Exactly one +invalid-block judgment, and it points at the unterminated opener (line 7)." + (let* ((out (lo-test--run lo-test--paired-then-unterminated)) + (judgments (lo-test--judgments (plist-get out :issues))) + (invalid (cl-remove-if-not + (lambda (i) (eq (plist-get i :checker) 'invalid-block)) + judgments))) + (should (= 1 (length invalid))) + (should (= 7 (plist-get (car invalid) :line))))) + +;;; --------------------------------------------------------------------------- ;;; --check mode (ert-deftest lo-check-mode-does-not-modify-file () @@ -739,6 +847,48 @@ missing-rules violation." (judgments (lo-test--judgments (plist-get out :issues)))) (should-not (member 'subtask-done-not-dated (lo-test--checkers judgments))))) +;;; dated-log-heading-active-timestamp check (stale SCHEDULED/DEADLINE on a +;;; completed dated-log entry — the home 2026-07-17 agenda-pollution bug) + +(ert-deftest lo-dated-log-active-scheduled-is-flagged () + "A dated-log entry still carrying an active SCHEDULED is flagged: org renders +it on the agenda forever despite the missing keyword." + (let* ((out (lo-test--run + "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 trip booked\nSCHEDULED: <2026-06-18 Thu>\nBody.\n")) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should (= 0 (plist-get out :fixes))) ; judgment-only, never auto-fixed + (should (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments))))) + +(ert-deftest lo-dated-log-active-deadline-is-flagged () + "An active DEADLINE on a dated-log entry is flagged too." + (let* ((out (lo-test--run + "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 shipped\nDEADLINE: <2026-06-25 Thu>\n")) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments))))) + +(ert-deftest lo-dated-log-clean-entry-not-flagged () + "A dated-log entry with no active planning timestamp is correct — not flagged." + (let* ((out (lo-test--run + "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 done cleanly\nBody only.\n")) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should-not (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments))))) + +(ert-deftest lo-dated-log-inactive-timestamp-not-flagged () + "An inactive [..] timestamp doesn't render on the agenda, so it isn't flagged — +only active <..> planning timestamps are the defect." + (let* ((out (lo-test--run + "* Open Work\n\n** TODO [#B] Parent\n*** 2026-06-20 Sat @ 10:00:00 -0500 recorded\nSCHEDULED: [2026-06-18 Thu]\n")) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should-not (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments))))) + +(ert-deftest lo-dated-log-active-scheduled-on-live-todo-not-flagged () + "A live TODO (keyword present) that legitimately carries an active SCHEDULED is +not a dated-log heading, so this checker leaves it alone." + (let* ((out (lo-test--run + "* Open Work\n\n** TODO [#B] Parent\n*** TODO [#C] real upcoming task\nSCHEDULED: <2026-06-18 Thu>\n")) + (judgments (lo-test--judgments (plist-get out :issues)))) + (should-not (member 'dated-log-heading-active-timestamp (lo-test--checkers judgments))))) + ;;; --------------------------------------------------------------------------- ;;; structural heading checks (org-lint gaps) @@ -817,3 +967,134 @@ heading, so it is not flagged — only two-or-more indented stars are." (provide 'test-lint-org) ;;; test-lint-org.el ends here + +;;; --------------------------------------------------------------------------- +;;; task-missing-last-reviewed (claude-rules/todo-format.md) + +(ert-deftest lo-task-without-last-reviewed-is-judgment () + "An open level-2 task with no :LAST_REVIEWED: is flagged." + (let* ((out (lo-test--run "* Open Work\n** TODO [#B] A task :feature:\nBody.\n")) + (js (lo-test--judgments (plist-get out :issues)))) + (should (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +(ert-deftest lo-task-with-last-reviewed-is-clean () + "A task carrying the property is not flagged." + (let* ((out (lo-test--run (concat "* Open Work\n** TODO [#B] A task :feature:\n" + ":PROPERTIES:\n:LAST_REVIEWED: 2026-07-23\n:END:\n" + "Body.\n"))) + (js (lo-test--judgments (plist-get out :issues)))) + (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +(ert-deftest lo-task-last-reviewed-accepts-org-timestamp () + "The org-native [YYYY-MM-DD Day] form counts, matching the staleness script." + (let* ((out (lo-test--run (concat "* Open Work\n** TODO [#B] A task :feature:\n" + ":PROPERTIES:\n:LAST_REVIEWED: [2026-07-23 Thu]\n:END:\n"))) + (js (lo-test--judgments (plist-get out :issues)))) + (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +(ert-deftest lo-done-task-without-last-reviewed-is-clean () + "Completed tasks leave the review pool, so they are never flagged." + (let* ((out (lo-test--run (concat "* Open Work\n** DONE [#B] A task :feature:\n" + "CLOSED: [2026-07-23 Thu]\n"))) + (js (lo-test--judgments (plist-get out :issues)))) + (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +(ert-deftest lo-subtask-without-last-reviewed-is-clean () + "Only level-2 tasks are in the review pool; deeper headings are not." + (let* ((out (lo-test--run (concat "* Open Work\n** TODO [#B] Parent :feature:\n" + ":PROPERTIES:\n:LAST_REVIEWED: 2026-07-23\n:END:\n" + "*** TODO A sub-task\n"))) + (js (lo-test--judgments (plist-get out :issues)))) + (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +(ert-deftest lo-cookieless-task-without-last-reviewed-is-clean () + "The staleness script selects on a priority cookie, so match that scope." + (let* ((out (lo-test--run "* Open Work\n** TODO Manual testing and validation\n")) + (js (lo-test--judgments (plist-get out :issues)))) + (should-not (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +(ert-deftest lo-verify-task-without-last-reviewed-is-judgment () + "VERIFY is in the review pool too." + (let* ((out (lo-test--run "* Open Work\n** VERIFY [#B] Waiting on Craig\n")) + (js (lo-test--judgments (plist-get out :issues)))) + (should (memq 'task-missing-last-reviewed (lo-test--checkers js))))) + +;;; --------------------------------------------------------------------------- +;;; todo-format checkers skip docs/specs/ files (claude-rules/todo-format.md) +;; +;; The four todo-format-family checkers encode todo.org completion conventions. +;; A spec legitimately uses ** DONE <decision> with no CLOSED cookie and +;; ** <dated> — <who> review-history headings, so those checkers misfire on +;; every spec. They must skip any file under a docs/specs/ path segment. + +(defun lo-test--run-at (relpath content) + "Write CONTENT to <tmpdir>/RELPATH, run lint on it, return :issues. +RELPATH is a relative path (may contain slashes) so a docs/specs/ segment +can be exercised — the checkers key on the file's path, not just its name." + (let* ((root (make-temp-file "lo-test-root-" t)) + (file (expand-file-name relpath root))) + (make-directory (file-name-directory file) t) + (unwind-protect + (progn + (with-temp-file file (insert content)) + (lo-test--reset) + (lo-process-file file) + (prog1 (list :issues lo-issues) + (lo-test--drop-buffer file))) + (delete-directory root t)))) + +(defconst lo-test--spec-decisions + "* Decisions [1/1]\n** DONE Some decision\n- Context: x\n" + "A spec Decisions section: a level-2 DONE with no CLOSED cookie.") + +(defconst lo-test--spec-history + "* Review history\n** 2026-07-14 Tue @ 02:03:28 -0500 — Claude — responder\n- What: x\n" + "A spec review-history section: a level-2 dated header.") + +(ert-deftest lo-todo-checkers-fire-on-a-normal-org-file () + "Baseline: the checkers DO fire on a non-spec path (the bug is scope, not silence)." + (let* ((out (lo-test--run-at "todo.org" lo-test--spec-decisions)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should (memq 'level2-done-without-closed cs)))) + +(ert-deftest lo-level2-done-without-closed-skips-specs () + (let* ((out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" lo-test--spec-decisions)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should-not (memq 'level2-done-without-closed cs)))) + +(ert-deftest lo-level2-dated-header-skips-specs () + (let* ((out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" lo-test--spec-history)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should-not (memq 'level-2-dated-header cs)))) + +(ert-deftest lo-dated-log-active-timestamp-skips-specs () + (let* ((c "* History\n** 2026-07-14 Tue @ 02:03:28 -0500 — did a thing\nSCHEDULED: <2026-07-20 Mon>\n") + (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should-not (memq 'dated-log-heading-active-timestamp cs)))) + +(ert-deftest lo-subtask-done-not-dated-skips-specs () + (let* ((c "* Work\n** TODO Parent\n*** DONE A sub-decision\n") + (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should-not (memq 'subtask-done-not-dated cs)))) + +(ert-deftest lo-link-checks-still-fire-on-specs () + "Only the todo-format family is scoped out; a broken link in a spec still flags." + (let* ((c "* X\n[[file:does-not-exist-xyz.org][link]]\n") + (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should (memq 'link-to-local-file cs)))) + +(ert-deftest lo-task-missing-last-reviewed-skips-specs () + "The fifth todo-format checker (added 2026-07-23) skips specs too — a spec's +phases section may carry ** TODO [#x] items that aren't backlog tasks." + (let* ((c "* Implementation phases\n** TODO [#B] Phase one\nBody.\n") + (out (lo-test--run-at "docs/specs/2026-07-14-x-spec.org" c)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should-not (memq 'task-missing-last-reviewed cs))) + ;; And still fires on a normal file. + (let* ((c "* Work\n** TODO [#B] Real backlog task\nBody.\n") + (out (lo-test--run-at "todo.org" c)) + (cs (lo-test--checkers (lo-test--judgments (plist-get out :issues))))) + (should (memq 'task-missing-last-reviewed cs)))) diff --git a/.ai/scripts/tests/test-todo-cleanup.el b/.ai/scripts/tests/test-todo-cleanup.el index ffbf2fb..1e964b3 100644 --- a/.ai/scripts/tests/test-todo-cleanup.el +++ b/.ai/scripts/tests/test-todo-cleanup.el @@ -31,6 +31,7 @@ (defun tc-test--reset (&optional check) (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-archived-to-file 0 tc-issues nil + tc-sealed 0 tc-seal nil tc-convert-subtasks nil tc-check-only (and check t) tc-archive-done t tc-sync-child-priority nil tc-current-file nil @@ -40,6 +41,7 @@ (defun tc-test--reset-sync (&optional check) (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-archived-to-file 0 tc-issues nil + tc-sealed 0 tc-seal nil tc-check-only (and check t) tc-archive-done nil tc-sync-child-priority t tc-current-file nil @@ -514,6 +516,12 @@ gitignore todo.org, then run `--archive-done' aging with the DEFAULT archive pat .gitignore contents or nil), :archive-ignored (whether git ignores the archive), :archive-exists." (let* ((root (make-temp-file "tc-git-" t)) + ;; Private backup dir: this helper writes a file literally named + ;; todo.org and runs a real (non-check) pass, so without this its + ;; backup lands in the shared temp dir under the exact production + ;; name and is indistinguishable from a real one. + (temporary-file-directory + (file-name-as-directory (make-temp-file "tc-git-bk-" t))) (todo (expand-file-name "todo.org" root)) (archive (expand-file-name "archive/task-archive.org" root)) (gi (expand-file-name ".gitignore" root))) @@ -534,7 +542,8 @@ gitignore todo.org, then run `--archive-done' aging with the DEFAULT archive pat :archive-ignored (eq 0 (call-process "git" nil nil nil "check-ignore" "-q" archive)) :archive-exists (file-readable-p archive))) - (delete-directory root t)))) + (delete-directory root t) + (delete-directory temporary-file-directory t)))) (ert-deftest tc-age-self-protect-gitignores-archive-when-todo-ignored () "When the todo file is gitignored, the aged-out archive is added to .gitignore @@ -578,6 +587,95 @@ entry is added for it." (should (> (plist-get out :archived) 0))))) ;;; --------------------------------------------------------------------------- +;;; --archive-done retention default + +(ert-deftest tc-archive-retain-default-is-one-month () + "The shipped retention default is one month (31 days), not the legacy 7. +The defvar initializes from this defconst; the live var itself is mutated by +other tests, so the immutable defconst is the stable contract to pin." + (should (= 31 tc-archive-retain-days-default))) + +;;; --------------------------------------------------------------------------- +;;; --seal: rename the working archive to resolved-YYYY-MM-DD.org + +(defun tc-test--seal (&optional opts) + "Run `--seal' against a temp todo file with a temp archive dir. +OPTS is a plist: :archive-content (seed task-archive.org with this; nil = no +working archive), :ref (YEAR MONTH DAY seal date; default (2026 7 18)), +:check, :presealed (also create resolved-<ref>.org first, to test collision). +Returns a plist: :sealed count, :issues, :working-exists, :sealed-exists, +:sealed-name, :report." + (let* ((ref (or (plist-get opts :ref) '(2026 7 18))) + (check (plist-get opts :check)) + (archive-content (plist-get opts :archive-content)) + (todo (make-temp-file "tc-seal-todo-" nil ".org")) + (adir (make-temp-file "tc-seal-arch-" t)) + (afile (expand-file-name "task-archive.org" adir)) + (sealed-name (format "resolved-%04d-%02d-%02d.org" + (nth 0 ref) (nth 1 ref) (nth 2 ref))) + (sealed (expand-file-name sealed-name adir))) + (unwind-protect + (progn + (with-temp-file todo (insert "* Open Work\n** TODO [#A] live\n")) + (when archive-content (with-temp-file afile (insert archive-content))) + (when (plist-get opts :presealed) + (with-temp-file sealed (insert "pre-existing seal\n"))) + (tc-test--reset check) + ;; Set every mode flag explicitly: tc-test--reset leaves + ;; tc-convert-subtasks untouched, so a convert test running earlier in + ;; the suite would otherwise still own the dispatch and run convert. + (setq tc-archive-done nil tc-sync-child-priority nil + tc-convert-subtasks nil tc-seal t tc-sealed 0 + tc-archive-reference-date ref + tc-archive-file afile) + (let ((report (with-output-to-string (tc-process-file todo) (tc-emit-report)))) + (tc-test--drop-buffer todo) + (list :sealed tc-sealed + :issues tc-issues + :working-exists (file-readable-p afile) + :sealed-exists (file-readable-p sealed) + :sealed-name sealed-name + :report report))) + (tc-test--drop-buffer todo) + (delete-file todo) + (delete-directory adir t)))) + +(ert-deftest tc-seal-renames-working-archive-to-dated-file () + "Normal: --seal renames task-archive.org to resolved-<seal-date>.org." + (let ((out (tc-test--seal '(:archive-content "* Resolved (archived)\n** DONE old\n" + :ref (2026 7 18))))) + (should (= 1 (plist-get out :sealed))) + (should-not (plist-get out :working-exists)) + (should (plist-get out :sealed-exists)) + (should (equal "resolved-2026-07-18.org" (plist-get out :sealed-name))) + (should (tc-test--has (plist-get out :report) "sealed task-archive.org → resolved-2026-07-18.org")))) + +(ert-deftest tc-seal-nothing-to-seal-is-a-reported-noop () + "Boundary: no working archive present — reported no-op, nothing created." + (let ((out (tc-test--seal '(:ref (2026 7 18))))) + (should (= 0 (plist-get out :sealed))) + (should-not (plist-get out :sealed-exists)) + (should (tc-test--has (plist-get out :report) "no working archive to seal")))) + +(ert-deftest tc-seal-check-mode-previews-without-renaming () + "Boundary: --check reports the seal but leaves the working archive in place." + (let ((out (tc-test--seal '(:archive-content "* Resolved (archived)\n" + :ref (2026 7 18) :check t)))) + (should (= 1 (plist-get out :sealed))) + (should (plist-get out :working-exists)) + (should-not (plist-get out :sealed-exists)) + (should (tc-test--has (plist-get out :report) "would seal")))) + +(ert-deftest tc-seal-refuses-to-clobber-existing-sealed-file () + "Error: resolved-<today>.org already exists — refuse, leave both files intact." + (let ((out (tc-test--seal '(:archive-content "* Resolved (archived)\n" + :ref (2026 7 18) :presealed t)))) + (should (= 0 (plist-get out :sealed))) + (should (plist-get out :working-exists)) + (should (plist-get out :sealed-exists)) + (should (tc-test--has (plist-get out :report) "already exists")))) + +;;; --------------------------------------------------------------------------- ;;; Sync-child-priority harness + fixtures (defun tc-test--sync (content &optional runs check) @@ -773,7 +871,7 @@ in ISSUES, in document order." (defun tc-test--reset-convert (&optional check) (setq tc-fixes 0 tc-archived 0 tc-bumped 0 tc-converted 0 tc-archived-to-file 0 - tc-issues nil + tc-issues nil tc-sealed 0 tc-seal nil tc-check-only (and check t) tc-archive-done nil tc-sync-child-priority nil tc-convert-subtasks t tc-current-file nil @@ -927,8 +1025,9 @@ CLOSED: [2026-06-27 Sat 12:50] DEADLINE: <2026-06-30 Tue> Body line. ") -(ert-deftest tc-convert-preserves-deadline-on-shared-planning-line-boundary () - "Boundary: removing the CLOSED cookie keeps a DEADLINE sharing its planning line." +(ert-deftest tc-convert-strips-deadline-sharing-the-planning-line-boundary () + "Boundary: a DEADLINE sharing the CLOSED planning line goes too — a dated-log +entry carries no active planning timestamp (todo-format.md). Body survives." (let* ((out (tc-test--convert tc-test--convert-closed-with-deadline)) (res (plist-get out :result))) (should (= 1 (plist-get out :converted))) @@ -936,8 +1035,142 @@ Body line. "^\\*\\*\\* 2026-06-27 Sat @ 12:50:00 [-+][0-9]\\{4\\} Ship the panel$" res)) (should-not (string-match-p "CLOSED:" res)) - (should (string-match-p "^DEADLINE: <2026-06-30 Tue>$" res)) + (should-not (string-match-p "DEADLINE:" res)) + (should (string-match-p "^Body line\\.$" res)))) + +(defconst tc-test--convert-closed-and-scheduled-separate-lines + "* Project Open Work +** TODO [#B] Parent task +*** DONE [#C] Book the venue :feature: +CLOSED: [2026-06-27 Sat 12:50] +SCHEDULED: <2026-06-20 Sat> +Body line. +") + +(ert-deftest tc-convert-strips-scheduled-on-its-own-line () + "Normal (the home bug): a SCHEDULED planning line on its own — the completion +rewrite dropped keyword/priority/tags but left the SCHEDULED, pinning the dated +entry to the agenda as weeks-overdue. Both planning lines go; body survives." + (let* ((out (tc-test--convert tc-test--convert-closed-and-scheduled-separate-lines)) + (res (plist-get out :result))) + (should (= 1 (plist-get out :converted))) + (should (string-match-p + "^\\*\\*\\* 2026-06-27 Sat @ 12:50:00 [-+][0-9]\\{4\\} Book the venue$" + res)) + (should-not (string-match-p "CLOSED:" res)) + (should-not (string-match-p "SCHEDULED:" res)) (should (string-match-p "^Body line\\.$" res)))) +(defconst tc-test--convert-scheduled-in-body-prose + "* Project Open Work +** TODO [#B] Parent task +*** DONE [#C] Note the mechanism :feature: +CLOSED: [2026-06-27 Sat 12:50] +An active SCHEDULED: <2026-06-20 Sat> in prose must survive. +") + +(ert-deftest tc-convert-leaves-planning-shaped-body-prose-alone () + "Boundary: a planning-shaped token inside body prose (not a canonical planning +line) is left untouched — the strip stops at the first non-planning line." + (let* ((out (tc-test--convert tc-test--convert-scheduled-in-body-prose)) + (res (plist-get out :result))) + (should (= 1 (plist-get out :converted))) + (should-not (string-match-p "CLOSED:" res)) + (should (string-match-p "An active SCHEDULED: <2026-06-20 Sat> in prose must survive\\." res)))) + (provide 'test-todo-cleanup) ;;; test-todo-cleanup.el ends here + +;;; --------------------------------------------------------------------------- +;;; Backup before mutating (parity with lint-org.el / wrap-org-table.el) +;; +;; todo-cleanup rewrites todo.org in place and left no copy behind, while both +;; sibling org-mutators back up to /tmp first. It is also the one that runs most +;; often (every wrap, every sentry cycle). Emacs's own backup does not fire under +;; --batch -q, so there was genuinely no undo short of git. + +(ert-deftest tc-backup-written-before-a-real-mutation () + "A real (non-check) run leaves a copy holding the pre-edit content. + +`temporary-file-directory' is rebound to a private dir for the duration: the +backup name derives from the *file's* basename, and the real todo.org shares +that basename, so a live sentry run writing /tmp/todo.org.before-todo-cleanup.* +would otherwise be indistinguishable from this test's own artifact. The first +version of this test globbed the shared /tmp and passed only until a real run +created one (2026-07-24)." + (let* ((dir (make-temp-file "tc-backup-" t)) + (bdir (file-name-as-directory (make-temp-file "tc-bk-" t))) + (file (expand-file-name "todo.org" dir)) + (before "* P Open Work\n** TODO [#B] parent\n*** DONE a subtask\nCLOSED: [2026-07-01 Tue]\n")) + (unwind-protect + (progn + (with-temp-file file (insert before)) + (let ((tc-check-only nil) + (tc-convert-subtasks t) + (temporary-file-directory bdir)) + (tc-process-file file)) + (let ((backups (file-expand-wildcards + (concat bdir "todo.org.before-todo-cleanup.*")))) + (should backups) + (should (string-match-p + "a subtask" + (with-temp-buffer (insert-file-contents (car backups)) + (buffer-string)))))) + (delete-directory dir t) + (delete-directory bdir t)))) + +(ert-deftest tc-no-backup-in-check-mode () + "--check writes nothing, so it must not leave a backup either. +Uses a private `temporary-file-directory' for the same isolation reason." + (let* ((dir (make-temp-file "tc-backup-" t)) + (bdir (file-name-as-directory (make-temp-file "tc-bk-" t))) + (file (expand-file-name "todo.org" dir))) + (unwind-protect + (progn + (with-temp-file file + (insert "* P Open Work\n** TODO [#B] parent\n*** DONE sub\nCLOSED: [2026-07-01 Tue]\n")) + (let ((tc-check-only t) + (tc-convert-subtasks t) + (temporary-file-directory bdir)) + (tc-process-file file)) + (should-not (file-expand-wildcards + (concat bdir "todo.org.before-todo-cleanup.*")))) + (delete-directory dir t) + (delete-directory bdir t)))) + +(ert-deftest tc-backup-never-overwrites-an-earlier-one () + "Two invocations in the same second must not collapse to one backup. + +open-tasks.org runs --convert-subtasks then --archive-done back to back, each +a sub-second batch run. With a second-resolution stamp and copy-file's +OK-IF-ALREADY-EXISTS, the second invocation overwrote the first's backup with +already-mutated content, so the true pre-session original was unrecoverable — +the exact state the backup exists to preserve (found 2026-07-24 in review)." + (let* ((dir (make-temp-file "tc-collide-" t)) + (bdir (file-name-as-directory (make-temp-file "tc-cbk-" t))) + (file (expand-file-name "todo.org" dir)) + (original (concat "* P Open Work\n** TODO [#B] parent\n*** DONE sub\n" + "CLOSED: [2026-07-01 Tue]\n" + "* P Resolved\n** DONE [#C] old\nCLOSED: [2025-01-01 Wed]\n"))) + (unwind-protect + (progn + (with-temp-file file (insert original)) + ;; Two back-to-back invocations, as the shipped workflow does. + (let ((temporary-file-directory bdir)) + (let ((tc-check-only nil) (tc-convert-subtasks t)) + (tc-process-file file)) + (let ((tc-check-only nil) (tc-convert-subtasks nil) (tc-archive-done t) + (tc-archive-retain-days nil)) + (tc-process-file file))) + (let ((backups (file-expand-wildcards + (concat bdir "todo.org.before-todo-cleanup.*")))) + ;; Both invocations kept their own backup. + (should (= (length backups) 2)) + ;; And one of them still holds the true original. + (should (cl-some (lambda (b) + (string= original + (with-temp-buffer (insert-file-contents b) + (buffer-string)))) + backups)))) + (delete-directory dir t) + (delete-directory bdir t)))) diff --git a/.ai/scripts/tests/test_apkg_to_orgdrill.py b/.ai/scripts/tests/test_apkg_to_orgdrill.py new file mode 100644 index 0000000..6a95ea4 --- /dev/null +++ b/.ai/scripts/tests/test_apkg_to_orgdrill.py @@ -0,0 +1,301 @@ +"""Tests for apkg-to-orgdrill.py — the inverse of flashcard-to-anki.py. + +The converter reads an Anki .apkg (a zip holding collection.anki2 / .anki21 +sqlite) and emits an org-drill .org in the house canonical shape. It is +stdlib-only (zipfile + sqlite3), so it imports directly — no genanki stub. + +The apkg schema these tests build by hand mirrors what genanki actually +writes, confirmed against a real apkg generated from flashcard-to-anki.py: + - col.decks : JSON {did: {"name": ...}}, always including id-1 "Default" + - col.models : JSON {mid: {"name": ..., "flds": [{"name": "Front"}, ...]}} + - notes.flds : fields joined by \x1f; tags space-padded (" tag ") + - cards : nid -> did (the Default deck carries no cards) + +The round-trip test closes the loop through flashcard-to-anki.py's own +parse(): original org -> forward parse tuples -> apkg fixture -> converter +-> recovered org -> forward parse -> assert the (front, back, tag) tuples +match. Only the apkg materialization is hand-built (the genanki boundary); +everything else is the real code on both sides. +""" +from __future__ import annotations + +import importlib.util +import json +import sqlite3 +import sys +import types +import zipfile +from pathlib import Path + +import pytest + +SCRIPTS = Path(__file__).resolve().parents[1] +CONVERTER = SCRIPTS / "apkg-to-orgdrill.py" +FORWARD = SCRIPTS / "flashcard-to-anki.py" + + +def _load(path: Path, name: str, stub_genanki: bool = False): + if stub_genanki: + sys.modules.setdefault("genanki", types.ModuleType("genanki")) + spec = importlib.util.spec_from_file_location(name, path) + assert spec and spec.loader + module = importlib.util.module_from_spec(spec) + # Register before exec: @dataclass resolves cls.__module__ via sys.modules + # (Python 3.14), which is None for an unregistered importlib module. + sys.modules[name] = module + spec.loader.exec_module(module) + return module + + +@pytest.fixture(scope="module") +def conv(): + return _load(CONVERTER, "apkg_to_orgdrill") + + +@pytest.fixture(scope="module") +def forward(): + return _load(FORWARD, "flashcard_to_anki", stub_genanki=True) + + +# --- fixture builder: write a genanki-shaped apkg by hand ------------------ + +def _make_apkg( + path: Path, + decks: dict[int, str], + models: dict[int, list[str]], + notes: list[tuple[int, int, list[str], str]], # (nid, mid, fields, tag) + cards: list[tuple[int, int]], # (nid, did) + *, + media: str = "{}", +) -> None: + """Materialize a minimal apkg matching genanki's collection.anki2 shape.""" + col_dir = path.parent / f"{path.stem}-build" + col_dir.mkdir(parents=True, exist_ok=True) + db = col_dir / "collection.anki2" + if db.exists(): + db.unlink() + con = sqlite3.connect(db) + con.execute("CREATE TABLE col (id INTEGER, decks TEXT, models TEXT)") + decks_json = {"1": {"name": "Default"}} + decks_json.update({str(did): {"name": name} for did, name in decks.items()}) + models_json = { + str(mid): {"name": f"{decks.get(list(decks)[0], 'M')} model", + "flds": [{"name": n, "ord": i} for i, n in enumerate(flds)]} + for mid, flds in models.items() + } + con.execute("INSERT INTO col (id, decks, models) VALUES (1, ?, ?)", + (json.dumps(decks_json), json.dumps(models_json))) + con.execute("CREATE TABLE notes (id INTEGER, mid INTEGER, flds TEXT, tags TEXT)") + for nid, mid, fields, tag in notes: + con.execute("INSERT INTO notes (id, mid, flds, tags) VALUES (?, ?, ?, ?)", + (nid, mid, "\x1f".join(fields), f" {tag} " if tag else " ")) + con.execute("CREATE TABLE cards (id INTEGER, nid INTEGER, did INTEGER)") + for i, (nid, did) in enumerate(cards): + con.execute("INSERT INTO cards (id, nid, did) VALUES (?, ?, ?)", (1000 + i, nid, did)) + con.commit() + con.close() + with zipfile.ZipFile(path, "w") as z: + z.write(db, "collection.anki2") + z.writestr("media", media) + + +# --- html_to_org_body ------------------------------------------------------ + +def test_html_to_org_splits_br_into_lines(conv): + assert conv.html_to_org_body("one<br>two<br>three") == ["one", "two", "three"] + + +def test_html_to_org_handles_br_variants(conv): + assert conv.html_to_org_body("a<br/>b<br />c<BR>d") == ["a", "b", "c", "d"] + + +def test_html_to_org_unescapes_entities_amp_last(conv): + # Inverts escape_html (which escapes & first): < > & -> < > &. + assert conv.html_to_org_body("x <tag> & y") == ["x <tag> & y"] + + +def test_html_to_org_preserves_a_literal_escaped_entity(conv): + # Forward-escaping the literal "<" yields "&lt;"; the inverse must + # recover "<", not "<". + assert conv.html_to_org_body("&lt;") == ["<"] + + +def test_html_to_org_strips_answer_hr(conv): + assert conv.html_to_org_body('front<hr id="answer">back') == ["front", "back"] + + +def test_html_to_org_empty_back_is_empty(conv): + assert conv.html_to_org_body("") == [] + + +# --- read_apkg ------------------------------------------------------------- + +def test_read_apkg_single_deck_recovers_front_back_tag_deck(conv, tmp_path): + apkg = tmp_path / "d.apkg" + _make_apkg( + apkg, + decks={20: "My Deck"}, + models={9: ["Front", "Back"]}, + notes=[(100, 9, ["Q1?", "A1.<br>line2"], "sec-one")], + cards=[(100, 20)], + ) + recovered = conv.read_apkg(apkg) + assert len(recovered) == 1 + note = recovered[0] + assert note.deck == "My Deck" + assert note.front == "Q1?" + assert note.back_html == "A1.<br>line2" + assert note.tag == "sec-one" + + +def test_read_apkg_multiple_decks_grouped(conv, tmp_path): + apkg = tmp_path / "multi.apkg" + _make_apkg( + apkg, + decks={20: "Deck A", 21: "Deck B"}, + models={9: ["Front", "Back"]}, + notes=[(100, 9, ["QA?", "AA"], "ta"), (101, 9, ["QB?", "AB"], "tb")], + cards=[(100, 20), (101, 21)], + ) + decks = {n.deck for n in conv.read_apkg(apkg)} + assert decks == {"Deck A", "Deck B"} + + +def test_read_apkg_skips_default_deck_without_cards(conv, tmp_path): + apkg = tmp_path / "def.apkg" + _make_apkg( + apkg, + decks={20: "Real Deck"}, + models={9: ["Front", "Back"]}, + notes=[(100, 9, ["Q?", "A"], "t")], + cards=[(100, 20)], + ) + assert {n.deck for n in conv.read_apkg(apkg)} == {"Real Deck"} + + +def test_read_apkg_warns_and_skips_non_basic_model(conv, tmp_path, capsys): + apkg = tmp_path / "cloze.apkg" + _make_apkg( + apkg, + decks={20: "Cloze Deck"}, + models={9: ["Text", "Extra"]}, # not Front/Back + notes=[(100, 9, ["some {{c1::text}}", "extra"], "t")], + cards=[(100, 20)], + ) + recovered = conv.read_apkg(apkg) + assert recovered == [] + assert "skip" in capsys.readouterr().err.lower() + + +def test_read_apkg_reads_anki21_collection_name(conv, tmp_path): + # A .anki21 collection filename must be read the same as .anki2. + apkg = tmp_path / "new.apkg" + _make_apkg( + apkg, + decks={20: "Deck"}, + models={9: ["Front", "Back"]}, + notes=[(100, 9, ["Q?", "A"], "t")], + cards=[(100, 20)], + ) + # Rewrite the zip renaming the collection member to .anki21. + with zipfile.ZipFile(apkg) as z: + data = z.read("collection.anki2") + media = z.read("media") + with zipfile.ZipFile(apkg, "w") as z: + z.writestr("collection.anki21", data) + z.writestr("media", media) + assert conv.read_apkg(apkg)[0].front == "Q?" + + +def test_read_apkg_flags_media_reference(conv, tmp_path, capsys): + apkg = tmp_path / "media.apkg" + _make_apkg( + apkg, + decks={20: "Deck"}, + models={9: ["Front", "Back"]}, + notes=[(100, 9, ["Q?", 'see <img src="x.png">'], "t")], + cards=[(100, 20)], + ) + conv.read_apkg(apkg) + assert "media" in capsys.readouterr().err.lower() + + +# --- notes_to_org ---------------------------------------------------------- + +def test_notes_to_org_emits_canonical_shape(conv): + Note = conv.Note + notes = [ + Note(deck="My Deck", front="Q1?", back_html="A1.", tag="alpha"), + Note(deck="My Deck", front="Q2?", back_html="A2.", tag="alpha"), + ] + ids = iter(["id-1", "id-2"]) + org = conv.notes_to_org(notes, "My Deck", new_id=lambda: next(ids)) + assert "#+TITLE: My Deck" in org + assert "* alpha" in org + assert "** Q1? :drill:" in org + assert ":ID: id-1" in org + assert ":ID: id-2" in org + assert org.count("* alpha") == 1 # both cards share one section + + +def test_notes_to_org_distinct_tags_get_distinct_sections(conv): + Note = conv.Note + notes = [ + Note(deck="D", front="Qa?", back_html="a", tag="alpha"), + Note(deck="D", front="Qb?", back_html="b", tag="beta"), + ] + org = conv.notes_to_org(notes, "D", new_id=lambda: "x") + assert "* alpha" in org and "* beta" in org + + +# --- round-trip through the real forward parse() --------------------------- + +def test_round_trip_matches_forward_parse_tuples(conv, forward, tmp_path): + original = ( + "#+TITLE: RT Deck\n" + "\n" + "* First Section\n" + "** What is 2+2? :drill:\n" + ":PROPERTIES:\n:ID: aaaa\n:END:\n" + "Four.\n" + "Second line with <angle> & amp.\n" + "\n" + "* Second Section\n" + "** Capital of France? :drill:\n" + "Paris.\n" + ) + tuples = forward.parse(original) # [(front, back_html, anki_tags), ...] + assert len(tuples) == 2 + + apkg = tmp_path / "rt.apkg" + _make_apkg( + apkg, + decks={20: "RT Deck"}, + models={9: ["Front", "Back"]}, + # anki_tags is a list; the apkg tags field is space-joined. + notes=[(100 + i, 9, [f, b], " ".join(tags)) + for i, (f, b, tags) in enumerate(tuples)], + cards=[(100 + i, 20) for i in range(len(tuples))], + ) + + by_deck = conv.convert(apkg) + assert set(by_deck) == {"RT Deck"} + recovered_tuples = forward.parse(by_deck["RT Deck"]) + assert recovered_tuples == tuples + + +# --- errors ---------------------------------------------------------------- + +def test_read_apkg_missing_collection_errors(conv, tmp_path): + bad = tmp_path / "bad.apkg" + with zipfile.ZipFile(bad, "w") as z: + z.writestr("media", "{}") + with pytest.raises(Exception): + conv.read_apkg(bad) + + +def test_read_apkg_not_a_zip_errors(conv, tmp_path): + notzip = tmp_path / "plain.apkg" + notzip.write_text("not a zip") + with pytest.raises(Exception): + conv.read_apkg(notzip) diff --git a/.ai/scripts/tests/test_cj_remove_block.py b/.ai/scripts/tests/test_cj_remove_block.py index 2c8dade..3cdee46 100644 --- a/.ai/scripts/tests/test_cj_remove_block.py +++ b/.ai/scripts/tests/test_cj_remove_block.py @@ -14,6 +14,34 @@ import pytest SCRIPT = Path(__file__).parent.parent / "cj-remove-block.py" +@pytest.fixture(autouse=True) +def isolated_tmpdir(tmp_path, monkeypatch): + """Give every test in this module a private TMPDIR. + + The script backs up to the system temp dir under a name derived from the + edited file's BASENAME. The real todo.org shares that basename, so any test + operating on a fixture named todo.org writes something indistinguishable + from a production backup — and an earlier version of this file globbed the + shared /tmp and unlinked every match, so a routine `make test` destroyed + Craig's real backups (found in review, 2026-07-24). + + Isolating at module scope rather than per-test is deliberate: the same bug + was fixed once in the elisp sibling and left here, so relying on each new + test to remember is exactly how it recurred. Autouse makes it structural. + """ + d = tmp_path / "_tmpdir" + d.mkdir() + # TMPDIR covers subprocess invocations of the script. + monkeypatch.setenv("TMPDIR", str(d)) + # tempfile.gettempdir() caches its answer on first call, so a test that + # loads the module in-process would keep writing to the real /tmp no matter + # what TMPDIR says. Override the cache too — this is the gap that made the + # env-var-only version still leak one backup per suite run. + import tempfile as _tempfile + monkeypatch.setattr(_tempfile, "tempdir", str(d)) + return d + + @pytest.fixture def run_remove(tmp_path): """Write content to a temp org file, run cj-remove-block, return new contents.""" @@ -155,3 +183,142 @@ class TestCjRemoveBlockSafety: err, post_content = run_remove_expecting_failure(original, start=4, end=2) assert err.returncode != 0 assert post_content == original + + +class TestMultiBlockRangeRefused: + """The validation exists to catch a drifted range, but it only checked the + first and last lines of that range. A span from one block's opening fence to + a LATER block's closing fence passed, and the removal silently deleted every + line between — real prose, headings, whole tasks — with a zero exit. Drift is + the skill's normal operating mode (respond-to-cj-comments edits the file as it + processes, and a file under cj review usually holds several blocks), so this + is the exact scenario the check was written for. Reproduced 2026-07-24.""" + + TWO_BLOCKS = ( + "* Alpha\n" + "#+begin_src cj:\n" + "note A\n" + "#+end_src\n" + "KEEP THIS LINE\n" + "* Beta\n" + "#+begin_src cj:\n" + "note B\n" + "#+end_src\n" + ) + + def test_range_spanning_two_blocks_is_refused(self, run_remove_expecting_failure): + # Lines 2..9: block one's opener through block two's closer. + err, content = run_remove_expecting_failure(self.TWO_BLOCKS, 2, 9) + assert err.returncode == 1 + assert "KEEP THIS LINE" in content, "content between the blocks was destroyed" + assert "* Beta" in content, "a heading between the blocks was destroyed" + + def test_refusal_names_the_reason(self, run_remove_expecting_failure): + err, _ = run_remove_expecting_failure(self.TWO_BLOCKS, 2, 9) + assert "more than one" in err.stderr.decode().lower() + + def test_a_correct_single_block_range_still_removes(self, run_remove): + # The fix must not over-tighten: the legitimate range still works. + out = run_remove(self.TWO_BLOCKS, 2, 4) + assert "note A" not in out + assert "KEEP THIS LINE" in out + assert "note B" in out, "the second block must be untouched" + + def test_a_nested_end_src_inside_the_range_is_refused(self, run_remove_expecting_failure): + # Any #+end_src before the final line means the range covers >1 block. + content = ( + "#+begin_src cj:\n" + "a\n" + "#+end_src\n" + "middle\n" + "#+begin_src cj:\n" + "b\n" + "#+end_src\n" + ) + err, after = run_remove_expecting_failure(content, 1, 7) + assert err.returncode == 1 + assert "middle" in after + + +class TestSafeMutation: + """The script rewrites Craig's org files (todo.org, notes.org). It wrote with + a bare write_text, which truncates the target on open, and took no backup — + so a mid-write failure left the file truncated with no copy to recover from. + lint-org.el, the other tool that mutates these files, backs up to a temp dir + first. Match that, and make the write atomic. + + Every test here redirects TMPDIR to a private directory. The backup name + derives from the file's basename, and the real todo.org shares it, so a test + globbing the shared temp dir cannot tell its own artifact from a genuine + backup — and an earlier version of this class globbed /tmp and unlinked every + match, so a routine `make test` destroyed real backups (found in review, + 2026-07-24). Never glob or delete across the shared temp dir.""" + + ONE_BLOCK = "* T\n#+begin_src cj:\nnote\n#+end_src\nkeep\n" + + def test_a_backup_is_written_before_mutating(self, tmp_path): + import subprocess, glob, os + bdir = tmp_path / "bk" + bdir.mkdir() + f = tmp_path / "todo.org" + f.write_text(self.ONE_BLOCK) + subprocess.run( + ["python3", str(SCRIPT), "--file", str(f), "--start", "2", "--end", "4"], + check=True, capture_output=True, + env={**os.environ, "TMPDIR": str(bdir)}, + ) + backups = glob.glob(str(bdir / "todo.org.before-cj-remove.*")) + assert backups, "no backup was written before mutating the org file" + assert "note" in Path(max(backups)).read_text() + + def test_no_partial_file_when_the_write_fails(self, tmp_path, monkeypatch): + import importlib.util + spec = importlib.util.spec_from_file_location("crb", SCRIPT) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + bdir = tmp_path / "bk" + bdir.mkdir() + monkeypatch.setenv("TMPDIR", str(bdir)) + f = tmp_path / "todo.org" + f.write_text(self.ONE_BLOCK) + def boom(*a, **k): + raise OSError("disk full") + monkeypatch.setattr(mod.os, "replace", boom) + with pytest.raises(OSError): + mod.remove_range(f, 2, 4) + # The original survives intact — no truncation, no partial. + assert f.read_text() == self.ONE_BLOCK + + +class TestBackupNeverOverwrites: + """Same defect class as todo-cleanup's, and more reachable here: the + respond-to-cj-comments skill removes several annotations in quick + succession, so a second-resolution stamp collides and the later backup + overwrote the earlier one with already-mutated content.""" + + TWO_BLOCKS = ( + "* A\n#+begin_src cj:\nfirst\n#+end_src\n" + "* B\n#+begin_src cj:\nsecond\n#+end_src\n" + ) + + def test_consecutive_removals_each_keep_a_backup(self, tmp_path, monkeypatch): + import subprocess, glob + bdir = tmp_path / "bk" + bdir.mkdir() + monkeypatch.setenv("TMPDIR", str(bdir)) + f = tmp_path / "todo.org" + f.write_text(self.TWO_BLOCKS) + original = f.read_text() + # Remove the second block, then the first — back to back, same second. + subprocess.run(["python3", str(SCRIPT), "--file", str(f), + "--start", "6", "--end", "8"], + check=True, capture_output=True, + env={**__import__("os").environ, "TMPDIR": str(bdir)}) + subprocess.run(["python3", str(SCRIPT), "--file", str(f), + "--start", "2", "--end", "4"], + check=True, capture_output=True, + env={**__import__("os").environ, "TMPDIR": str(bdir)}) + backups = glob.glob(str(bdir / "todo.org.before-cj-remove.*")) + assert len(backups) == 2, f"expected 2 backups, got {len(backups)}" + contents = [Path(b).read_text() for b in backups] + assert original in contents, "no backup holds the true original" diff --git a/.ai/scripts/tests/test_flashcard_stats.py b/.ai/scripts/tests/test_flashcard_stats.py index 606f7c1..46deccc 100644 --- a/.ai/scripts/tests/test_flashcard_stats.py +++ b/.ai/scripts/tests/test_flashcard_stats.py @@ -217,6 +217,31 @@ def test_parse_cards_captures_body_without_drawer_planning_or_answer_header(stat assert c["body"] == "the real answer" +def test_parse_cards_counts_a_multitag_heading_as_a_card(stats): + """A card multi-tagged :fundamental:drill: still counts; the front is clean.""" + text = "* Sec\n** Q multi? :fundamental:drill:\nthe answer\n" + cards, _ = stats.parse_cards(text.splitlines()) + assert len(cards) == 1 + assert cards[0]["heading"] == "Q multi?" + assert cards[0]["body"] == "the answer" + + +def test_parse_cards_ignores_a_tagged_heading_without_drill(stats): + """A tagged heading missing :drill: is not a drill card.""" + text = "* Sec\n** Just a note :note:\nbody\n" + cards, _ = stats.parse_cards(text.splitlines()) + assert cards == [] + + +def test_parse_cards_body_stops_at_next_multitag_card(stats): + """The body scan ends at the next L2 card even when it is multi-tagged.""" + text = "** Q1? :a:drill:\nbody1\n** Q2? :drill:\nbody2\n" + cards, _ = stats.parse_cards(text.splitlines()) + assert len(cards) == 2 + assert cards[0]["body"] == "body1" + assert cards[1]["body"] == "body2" + + def test_find_duplicate_fronts_matches_normalized_headings(stats): cards = [ {"heading": "What is LEO?"}, diff --git a/.ai/scripts/tests/test_flashcard_to_anki.py b/.ai/scripts/tests/test_flashcard_to_anki.py index 87008a8..fa38b64 100644 --- a/.ai/scripts/tests/test_flashcard_to_anki.py +++ b/.ai/scripts/tests/test_flashcard_to_anki.py @@ -158,17 +158,18 @@ Geostationary Earth Orbit. def test_parse_returns_front_back_tag_per_card(drill): cards = drill.parse(SECTIONED) assert len(cards) == 2 - assert cards[0] == ("What is LEO?", "Low Earth Orbit.", "orbital-regimes") + # The section becomes the sole Anki tag (as a one-element list). + assert cards[0] == ("What is LEO?", "Low Earth Orbit.", ["orbital-regimes"]) assert cards[1][0] == "What is GEO?" def test_parse_card_without_a_section_gets_the_drill_tag(drill): - assert drill.parse("** Lone card? :drill:\nbody\n") == [("Lone card?", "body", "drill")] + assert drill.parse("** Lone card? :drill:\nbody\n") == [("Lone card?", "body", ["drill"])] def test_parse_strips_properties_drawer_from_back(drill): text = "** Q? :drill:\n:PROPERTIES:\n:ID: abc\n:END:\nThe answer.\n" - assert drill.parse(text) == [("Q?", "The answer.", "drill")] + assert drill.parse(text) == [("Q?", "The answer.", ["drill"])] def test_parse_trims_leading_and_trailing_blank_body_lines(drill): @@ -178,7 +179,59 @@ def test_parse_trims_leading_and_trailing_blank_body_lines(drill): def test_parse_card_with_only_a_drawer_has_empty_back(drill): text = "** Q? :drill:\n:PROPERTIES:\n:ID: x\n:END:\n" - assert drill.parse(text) == [("Q?", "", "drill")] + assert drill.parse(text) == [("Q?", "", ["drill"])] + + +# --- multi-tag headings, --tag-filter, --guid-salt ------------------------- + +MULTITAG = """* Fundamentals +** What is LEO? :fundamental:drill: +Low Earth Orbit. +** What is GEO? :drill: +Geostationary Earth Orbit. +""" + + +def test_parse_multitag_heading_is_a_card_when_drill_is_present(drill): + """A heading with a second org tag still parses when drill is among them.""" + cards = drill.parse(MULTITAG) + assert len(cards) == 2 + assert cards[0][0] == "What is LEO?" + + +def test_parse_multitag_tags_ride_along_next_to_the_section_tag(drill): + """Non-drill org tags become Anki tags alongside the section tag.""" + cards = drill.parse(MULTITAG) + assert cards[0][2] == ["fundamentals", "fundamental"] # section slug + org tag + assert cards[1][2] == ["fundamentals"] # drill-only -> section only + + +def test_parse_heading_without_drill_tag_is_not_a_card(drill): + """A tagged heading missing :drill: is not a card (e.g. :note:).""" + assert drill.parse("* S\n** Just a note :note:\nbody\n") == [] + + +def test_parse_tag_filter_returns_only_cards_with_that_org_tag(drill): + """--tag-filter narrows to cards carrying the given org tag.""" + cards = drill.parse(MULTITAG, tag_filter="fundamental") + assert len(cards) == 1 + assert cards[0][0] == "What is LEO?" + + +def test_parse_body_bounded_by_any_l1_or_l2_heading(drill): + """A card body stops at the next L1/L2 heading, multi-tagged or not.""" + text = "** Q1? :a:drill:\nbody1\n** Q2? :drill:\nbody2\n" + cards = drill.parse(text) + assert cards[0][1] == "body1" + assert cards[1][1] == "body2" + + +def test_card_guid_salt_changes_the_guid(drill, monkeypatch): + """--guid-salt gives a subset deck its own GUID space; no salt is unchanged.""" + monkeypatch.setattr(drill.genanki, "guid_for", lambda *a: ":".join(a), raising=False) + assert drill.card_guid("front", None) == "front" + assert drill.card_guid("front", "fundamentals") == "fundamentals:front" + assert drill.card_guid("front", None) != drill.card_guid("front", "fundamentals") def test_parse_joins_multiline_body_with_br(drill): diff --git a/.ai/scripts/tests/test_inbox_send.py b/.ai/scripts/tests/test_inbox_send.py index f75d7a1..9b0a8c6 100644 --- a/.ai/scripts/tests/test_inbox_send.py +++ b/.ai/scripts/tests/test_inbox_send.py @@ -476,3 +476,117 @@ class TestFilenameCollisions: assert len(files) == 2 bodies = "".join(f.read_text() for f in files) assert "message one" in bodies and "message two" in bodies + + +class TestAtomicWrite: + """A send wrote straight to the destination path in another project's + inbox/, and write_text truncates on open, so any mid-write failure left a + zero-byte .org there. inbox-status counts that phantom as a pending + handoff, blocking a turn in the receiving project over a file with no + content (2026-07-23). The write must be atomic: the inbox sees a complete + file or nothing.""" + + def test_send_text_writes_utf8(self, tmp_path): + from datetime import datetime + mod = _load_module() + inbox = tmp_path / "inbox" + inbox.mkdir() + now = datetime(2026, 7, 23, 4, 36, 0) + # An em dash and an accented char — both non-ASCII. + dest = mod.send_text(inbox, "accent café and dash — here", "src", None, now) + # Reading as utf-8 must round-trip; a locale-encoded write would raise + # under a C locale, and reading back proves the bytes are utf-8. + assert "—" in dest.read_text(encoding="utf-8") + + def test_send_text_no_partial_on_write_failure(self, tmp_path, monkeypatch): + from datetime import datetime + mod = _load_module() + inbox = tmp_path / "inbox" + inbox.mkdir() + now = datetime(2026, 7, 23, 4, 36, 0) + # Force the atomic finalize to fail after the temp file is written. + def boom(*a, **k): + raise OSError("disk full") + monkeypatch.setattr(mod.os, "replace", boom) + with pytest.raises(OSError): + mod.send_text(inbox, "a message that should never half-land", "src", None, now) + # No phantom, no leftover temp: the inbox is empty. + assert list(inbox.iterdir()) == [] + + def test_send_text_leaves_no_temp_on_success(self, tmp_path): + from datetime import datetime + mod = _load_module() + inbox = tmp_path / "inbox" + inbox.mkdir() + now = datetime(2026, 7, 23, 4, 36, 0) + dest = mod.send_text(inbox, "clean send", "src", None, now) + assert list(inbox.iterdir()) == [dest] + + def test_send_file_no_partial_on_write_failure(self, tmp_path, monkeypatch): + from datetime import datetime + mod = _load_module() + inbox = tmp_path / "inbox" + inbox.mkdir() + src = tmp_path / "note.org" + src.write_text("body") + now = datetime(2026, 7, 23, 4, 36, 0) + def boom(*a, **k): + raise OSError("disk full") + monkeypatch.setattr(mod.os, "replace", boom) + with pytest.raises(OSError): + mod.send_file(inbox, src, "src", None, now) + assert list(inbox.iterdir()) == [] + + def test_send_file_leaves_no_temp_on_success(self, tmp_path): + from datetime import datetime + mod = _load_module() + inbox = tmp_path / "inbox" + inbox.mkdir() + src = tmp_path / "note.org" + src.write_text("payload") + now = datetime(2026, 7, 23, 4, 36, 0) + dest = mod.send_file(inbox, src, "src", None, now) + assert list(inbox.iterdir()) == [dest] + assert dest.read_text() == "payload" + + +class TestSmallerDefects: + """Two low-severity defects found reading inbox-send during the 2026-07-23 + sweep: an unreadable source raised an uncaught traceback instead of the + clean error every other failure path produces, and a roots config naming + both a parent and one of its children listed the same project twice.""" + + def test_unreadable_source_gives_clean_error_not_traceback( + self, project_root, run_script, tmp_path + ): + project_root("sender") + project_root("receiver") + roots = [tmp_path / "projects"] + src = tmp_path / "secret.bin" + src.write_text("x") + src.chmod(0o000) + try: + result = run_script( + ["receiver", "--file", str(src)], + cwd=tmp_path / "projects" / "sender", + roots=roots, + expect_failure=True, + ) + finally: + src.chmod(0o644) + assert result.returncode == 1 + # The clean "inbox-send: <message>" shape, not a Python traceback. + assert result.stderr.startswith("inbox-send:") + assert "Traceback" not in result.stderr + + def test_discover_projects_dedupes_parent_and_child_root(self, tmp_path): + mod = _load_module() + # A project directory, reachable both as a child of its parent root and + # as a root in its own right. + parent = tmp_path / "projects" + proj = parent / "app" + (proj / ".ai").mkdir(parents=True) + (proj / "inbox").mkdir() + found = mod.discover_projects([parent, proj]) + resolved = [p.resolve() for p in found] + assert resolved.count(proj.resolve()) == 1 diff --git a/.ai/scripts/tests/test_route_recommend.py b/.ai/scripts/tests/test_route_recommend.py index acc4755..2ec900a 100644 --- a/.ai/scripts/tests/test_route_recommend.py +++ b/.ai/scripts/tests/test_route_recommend.py @@ -122,3 +122,31 @@ def test_cli_exclude_drops_current_project(tmp_path): r = _run(["--exclude", "foo"], roots=[tmp_path / "projects"], item="fix the foo widget") assert r.returncode == 0 assert r.stdout.strip() == "none" + + +# ---------------------------------------------------------------------- +# Duplicate candidate names +# +# Projects are collapsed to bare basenames, so two projects sharing a basename +# across roots (~/code/notes and ~/projects/notes) appear twice in the candidate +# list. Both literal-match, recommend read len(strong) > 1 as an ambiguous tie, +# and a correct strong match was downgraded to weak. Latent when discovered +# 2026-07-24 (27 projects, 27 distinct basenames) but real. +# ---------------------------------------------------------------------- + +def test_duplicate_candidate_name_keeps_strong_confidence(): + assert rr.recommend("fix the notes thing", ["notes", "other"]) == ("notes", "strong") + # The same name twice must not read as a tie. + assert rr.recommend("fix the notes thing", ["notes", "notes", "other"]) == ("notes", "strong") + + +def test_genuine_ambiguity_still_downgrades(): + # Two DIFFERENT projects both matching is a real tie and stays weak — the + # dedupe must collapse identical names only, never real ambiguity. + dest, conf = rr.recommend("notes and other both", ["notes", "other"]) + assert conf == "weak" + + +def test_duplicates_do_not_change_the_chosen_destination(): + dest, _ = rr.recommend("fix the notes thing", ["notes", "notes"]) + assert dest == "notes" diff --git a/.ai/scripts/todo-cleanup.el b/.ai/scripts/todo-cleanup.el index bd8166d..516e9b1 100644 --- a/.ai/scripts/todo-cleanup.el +++ b/.ai/scripts/todo-cleanup.el @@ -5,6 +5,8 @@ ;; emacs --batch -q -l todo-cleanup.el --check todo.org # hygiene report only ;; emacs --batch -q -l todo-cleanup.el --archive-done todo.org # archive completed subtrees ;; emacs --batch -q -l todo-cleanup.el --archive-done --check todo.org # preview the archive +;; emacs --batch -q -l todo-cleanup.el --seal todo.org # seal the working archive to resolved-YYYY-MM-DD.org +;; emacs --batch -q -l todo-cleanup.el --seal --check todo.org # preview the seal ;; emacs --batch -q -l todo-cleanup.el --convert-subtasks todo.org # dated-rewrite done level-3+ sub-tasks ;; emacs --batch -q -l todo-cleanup.el --convert-subtasks --check todo.org # preview the conversion ;; emacs --batch -q -l todo-cleanup.el --sync-child-priority todo.org # bump children whose priority drifted below the parent's @@ -37,23 +39,37 @@ ;; a message. Only direct level-2 children move — a DONE entry nested under ;; an open parent stays put. ;; -;; 2. Ages the "Resolved" section: a level-2 DONE/CANCELLED subtree whose -;; CLOSED date is older than `tc-archive-retain-days' (default 7) is moved +;; 2. Ages the "Resolved" section: a level-2 DONE/CANCELLED subtree is moved ;; out to `tc-archive-file' (default `archive/task-archive.org' beside the -;; todo file), keeping only the last week of closed tasks in the file -;; itself. Only subtrees closed within the window stay; older ones, and -;; those with no parseable CLOSED date, are moved out. Set -;; `tc-archive-retain-days' to nil to disable this step (legacy in-file-only -;; behavior). The aging date is `tc-archive-reference-date' when set -;; (tests), otherwise the real current date. The archive inherits the todo -;; file's gitignore status: when the todo file is gitignored, the archive -;; path is added to .gitignore before the first write, so private task -;; history never lands in a tracked path (see +;; todo file) when its CLOSED date is older than `tc-archive-retain-days' +;; (default 31 — one month) OR its CLOSED date can't be parsed. The last +;; month of closed tasks stays browsable in the file itself; older ones age +;; out. The unparseable-CLOSED case archives too, deliberately: a +;; keyword-complete task with no readable close date is cruft, not live +;; work. Set `tc-archive-retain-days' to nil to disable this step (legacy +;; in-file-only behavior). The aging date is `tc-archive-reference-date' +;; when set (tests), otherwise the real current date. The archive inherits +;; the todo file's gitignore status: when the todo file is gitignored, the +;; archive path is added to .gitignore before the first write, so private +;; task history never lands in a tracked path (see ;; `tc--ensure-archive-gitignored'). ;; ;; Archiving is consequential, so it's never run by default; it does *not* ;; also run the hygiene passes. ;; +;; * --seal (opt-in). Renames the working archive file (`tc-archive-file', +;; default `archive/task-archive.org') to `resolved-YYYY-MM-DD.org' beside it, +;; dated by the seal run, and leaves the next `--archive-done' to recreate a +;; fresh working file. The dated file means "everything sealed as of that +;; date" — not a calendar quarter — so a task closed late in a quarter and +;; archived after the boundary is never mislabeled; cadence (e.g. quarterly) +;; becomes independent of correctness and any slip is harmless. The sealed +;; file inherits the todo file's gitignore status the same way the working +;; archive does. A no-op (reported) when there's no working archive to seal; +;; refuses to clobber an existing `resolved-<today>.org'. Honors `--check'. +;; The seal date is `tc-archive-reference-date' when set (tests), otherwise the +;; real current date. +;; ;; * --convert-subtasks (opt-in). Rewrites every level-3-and-deeper heading whose ;; TODO state is DONE/CANCELLED/FAILED into a dated event-log entry ;; (`<stars> YYYY-MM-DD Day @ HH:MM:SS -ZZZZ <text>'), dropping the keyword, @@ -84,6 +100,12 @@ ;; --check-child-priority is the report-only alias for --sync-child-priority ;; --check. +;; Before any modification a backup is copied to +;; /tmp/<basename>.before-todo-cleanup.<YYYYMMDD-HHMMSS> +;; matching lint-org.el and wrap-org-table.el. Skipped under --check, which +;; writes nothing. +;; + (require 'org) (require 'cl-lib) (require 'calendar) @@ -102,6 +124,17 @@ sub-task is terminal too and belongs in the parent's dated history.") (defconst tc--priority-cookie-regexp "\\[#\\([A-Z]\\)\\]" "Regexp matching an org priority cookie. Match group 1 is the letter.") +(defconst tc--planning-cookie-regexp + "\\(?:CLOSED\\|DEADLINE\\|SCHEDULED\\):[ \t]*[[<][^]>\n]*[]>]" + "One org planning cookie: a CLOSED/DEADLINE/SCHEDULED keyword followed by a +bracketed (inactive) or angled (active) timestamp.") + +(defconst tc--planning-line-regexp + (concat "\\`[ \t]*\\(?:" tc--planning-cookie-regexp "[ \t]*\\)+\\'") + "A whole org planning line: nothing but planning cookies and whitespace. +Anchored to a single line's contents so a line mixing a cookie with real body +text is never matched.") + (defconst tc-no-sync-tag "no-sync" "Org tag that opts a heading and all its descendants out of `--sync-child-priority'. Inherits down: a tag on an ancestor counts for @@ -112,20 +145,30 @@ every heading below it.") (defvar tc-bumped 0) (defvar tc-converted 0) (defvar tc-issues nil) +(defvar tc-sealed 0) (defvar tc-check-only nil) (defvar tc-archive-done nil) (defvar tc-sync-child-priority nil) (defvar tc-convert-subtasks nil) +(defvar tc-seal nil) (defvar tc-current-file nil) (defvar tc-current-dir nil) (defvar tc-archived-to-file 0) -(defvar tc-archive-retain-days 7 +(defconst tc-archive-retain-days-default 31 + "Default retention window (days) for the `--archive-done' file-aging step — +one month. A closed Resolved subtree stays in-file for this long before it ages +out to `tc-archive-file'; the last month of resolved work stays browsable in the +todo file itself. Named so the \"one month\" contract is explicit and testable.") + +(defvar tc-archive-retain-days tc-archive-retain-days-default "Retention window for the `--archive-done' file-aging step. A closed Resolved subtree whose CLOSED date is within this many days of the reference date stays in the in-file Resolved section; an older one is moved out to `tc-archive-file'. -A subtree with no parseable CLOSED date stays. nil disables the aging step -entirely, leaving the legacy in-file-only behavior.") +A subtree with no parseable CLOSED date is aged out too (a keyword-complete task +with no readable close date is cruft, not live work). nil disables the aging +step entirely, leaving the legacy in-file-only behavior. Defaults to +`tc-archive-retain-days-default' (one month).") (defvar tc-archive-reference-date nil "(YEAR MONTH DAY) treated as \"today\" when aging Resolved subtrees out to a @@ -479,6 +522,51 @@ step. Honors `tc-check-only' (report only)." tc-issues)))))))))) ;;; --------------------------------------------------------------------------- +;;; --seal mode: rename the working archive to a dated resolved-YYYY-MM-DD.org + +(defun tc--seal-date-string () + "YYYY-MM-DD for the seal — `tc-archive-reference-date' when set (tests), +otherwise the real current date." + (if tc-archive-reference-date + (pcase-let ((`(,y ,m ,d) tc-archive-reference-date)) + (format "%04d-%02d-%02d" y m d)) + (format-time-string "%Y-%m-%d"))) + +(defun tc-seal-archive-file () + "Rename the working archive file to `resolved-YYYY-MM-DD.org' beside it. +The next `--archive-done' run recreates a fresh working file. No-op (reported) +when there is no working archive to seal; refuses to clobber an existing +`resolved-<today>.org'. Ensures the sealed file inherits the todo file's +gitignore status. Honors `tc-check-only'." + (let ((path (tc--archive-file-path))) + (cond + ((or (null path) (not (file-readable-p path))) + (push (list :kind 'seal-nothing :file tc-current-file) tc-issues)) + (t + (let* ((dir (file-name-directory path)) + (sealed (expand-file-name + (format "resolved-%s.org" (tc--seal-date-string)) dir))) + (cond + ((file-exists-p sealed) + (push (list :kind 'seal-collision :file tc-current-file + :detail (file-name-nondirectory sealed)) + tc-issues)) + (tc-check-only + (cl-incf tc-sealed) + (push (list :kind 'seal-would :file tc-current-file + :detail (file-name-nondirectory sealed)) + tc-issues)) + (t + ;; Ignore the sealed name before the rename so its history stays as + ;; private as the working archive it derives from. + (tc--ensure-archive-gitignored sealed) + (rename-file path sealed) + (cl-incf tc-sealed) + (push (list :kind 'seal-done :file tc-current-file + :detail (file-name-nondirectory sealed)) + tc-issues)))))))) + +;;; --------------------------------------------------------------------------- ;;; --sync-child-priority mode (defun tc--heading-priority-letter () @@ -617,6 +705,14 @@ before their descendants — a [#A] → [#B] → [#D] chain collapses in one pas ;; as written). Idempotent: an already-dated heading has no done keyword, so it ;; is skipped. A done sub-task with no parseable CLOSED cookie can't be dated, so ;; it is flagged and left alone rather than stamped with a fabricated date. +;; +;; The planning line goes entirely. A dated-log entry carries its date in the +;; heading, so CLOSED is redundant and an active DEADLINE/SCHEDULED is wrong: org +;; renders any headline with an active planning timestamp — keyword or not — so a +;; SCHEDULED left on a dated-log heading pins it to the agenda as weeks-overdue +;; long after the work is done. The conversion deletes the whole planning line, +;; not just the CLOSED cookie (todo-format.md; lint checker +;; `dated-log-heading-active-timestamp' backstops any that slip through). (defun tc--closed-parts-in-entry () "Return a plist (:year :month :day :dow :hour :minute) from the CLOSED cookie @@ -676,6 +772,27 @@ in-progress `org-map-entries' walk; markers track their headings across edits." nil 'file) (nreverse targets))) +(defun tc--strip-planning-lines-in-entry () + "Delete the canonical planning line(s) directly under the heading at point. +A planning line is one composed solely of CLOSED/DEADLINE/SCHEDULED cookies and +whitespace. Walks the lines immediately after the heading and stops at the first +non-planning line, so a planning-shaped line deeper in the body (e.g. in a code +block) is never touched. Returns the count of lines removed." + (save-excursion + (org-back-to-heading t) + (forward-line 1) + (let ((removed 0) (continue t)) + (while (and continue (not (eobp))) + (let ((line (buffer-substring-no-properties + (line-beginning-position) (line-end-position)))) + (if (string-match-p tc--planning-line-regexp line) + (progn + (delete-region (line-beginning-position) + (min (1+ (line-end-position)) (point-max))) + (cl-incf removed)) + (setq continue nil)))) + removed))) + (defun tc--convert-one-subtask (marker) "Convert the done sub-task heading at MARKER to a dated event-log entry. Under `tc-check-only' the conversion is reported but not performed." @@ -698,27 +815,13 @@ Under `tc-check-only' the conversion is reported but not performed." (push (list :kind 'convert-would :file tc-current-file :line line :heading title :new new) tc-issues) - ;; Replace the heading line, then drop the now-redundant CLOSED - ;; cookie from the entry (its date now lives in the header). Only - ;; the cookie goes: a planning line can also carry DEADLINE: or - ;; SCHEDULED: beside it, and those survive on their line. A line - ;; left blank by the removal is deleted whole. + ;; Replace the heading line, then drop the whole planning line. The + ;; date now lives in the header, so CLOSED is redundant and an active + ;; DEADLINE/SCHEDULED would wrongly pin this completed entry to the + ;; agenda (todo-format.md). Both go, not just the CLOSED cookie. (delete-region (line-beginning-position) (line-end-position)) (insert new) - (let ((end (save-excursion - (or (outline-next-heading) (goto-char (point-max))) - (point)))) - (save-excursion - (when (re-search-forward "CLOSED:[ \t]*\\[[^]]*\\][ \t]*" end t) - (replace-match "") - (let ((bol (line-beginning-position)) - (eol (line-end-position))) - (if (string-match-p "\\`[ \t]*\\'" - (buffer-substring bol eol)) - (delete-region bol (min (1+ eol) (point-max))) - (goto-char bol) - (when (looking-at "[ \t]+") - (replace-match ""))))))) + (tc--strip-planning-lines-in-entry) (push (list :kind 'convert-done :file tc-current-file :line line :heading title :new new) tc-issues))))))) @@ -735,9 +838,37 @@ event-log entry, pulling the timestamp from its CLOSED cookie. Honors ;;; --------------------------------------------------------------------------- ;;; Driver + reporting +(defun tc--backup (file) + "Copy FILE to /tmp before any modification. Skipped in --check mode. + +Matches `lint-org.el' and `wrap-org-table.el', the other tools that rewrite +these org files. todo-cleanup runs the most often of the three (every wrap, +every sentry cycle), and Emacs's own backup does not fire under --batch -q, so +without this a mechanical rewrite has no undo short of git — which recovers +only to the last commit and loses intra-session work." + (let* ((base (format "%s%s.before-todo-cleanup.%s" + temporary-file-directory + (file-name-nondirectory file) + (format-time-string "%Y%m%d-%H%M%S"))) + (backup base) + (n 2)) + ;; Never overwrite an earlier backup. A second-resolution stamp collides + ;; when two invocations run back to back, which the shipped workflow does + ;; (open-tasks.org runs --convert-subtasks then --archive-done, each a + ;; sub-second batch run). Overwriting there replaces the true pre-session + ;; original with already-mutated content — losing exactly what the backup + ;; exists to preserve. Suffix instead, so every invocation keeps its own. + (while (file-exists-p backup) + (setq backup (format "%s-%d" base n)) + (setq n (1+ n))) + (copy-file file backup nil) + backup)) + (defun tc-process-file (file) (setq tc-current-file (file-name-nondirectory file)) (setq tc-current-dir (file-name-directory (expand-file-name file))) + (unless tc-check-only + (tc--backup file)) (with-current-buffer (find-file-noselect file) (org-mode) (cond @@ -747,6 +878,8 @@ event-log entry, pulling the timestamp from its CLOSED cookie. Honors (tc-sync-child-priority-in-file)) (tc-convert-subtasks (tc-convert-subtasks-in-file)) + (tc-seal + (tc-seal-archive-file)) (t ;; Pass 1: auto-fix bogus state logs (or report under --check). (org-map-entries #'tc-fix-bogus-state-log-in-entry nil 'file) @@ -865,10 +998,26 @@ event-log entry, pulling the timestamp from its CLOSED cookie. Honors (plist-get i :file) (plist-get i :line) (plist-get i :heading) (plist-get i :detail))))))))) +(defun tc--emit-seal-report () + (dolist (i (reverse tc-issues)) + (pcase (plist-get i :kind) + ('seal-done + (princ (format "todo-cleanup --seal: sealed task-archive.org → %s\n" + (plist-get i :detail)))) + ('seal-would + (princ (format "todo-cleanup --seal: would seal task-archive.org → %s — CHECK MODE (no writes)\n" + (plist-get i :detail)))) + ('seal-collision + (princ (format "todo-cleanup --seal: %s already exists — not sealing (already sealed today?)\n" + (plist-get i :detail)))) + ('seal-nothing + (princ "todo-cleanup --seal: no working archive to seal\n"))))) + (defun tc-emit-report () (cond (tc-archive-done (tc--emit-archive-report)) (tc-sync-child-priority (tc--emit-sync-report)) (tc-convert-subtasks (tc--emit-convert-report)) + (tc-seal (tc--emit-seal-report)) (t (tc--emit-hygiene-report)))) (defun tc-main () @@ -886,6 +1035,9 @@ event-log entry, pulling the timestamp from its CLOSED cookie. Honors (when (member "--convert-subtasks" command-line-args-left) (setq tc-convert-subtasks t) (setq command-line-args-left (delete "--convert-subtasks" command-line-args-left))) + (when (member "--seal" command-line-args-left) + (setq tc-seal t) + (setq command-line-args-left (delete "--seal" command-line-args-left))) ;; --check-child-priority is the report-only alias for ;; `--sync-child-priority --check'. (when (member "--check-child-priority" command-line-args-left) @@ -893,7 +1045,7 @@ event-log entry, pulling the timestamp from its CLOSED cookie. Honors (setq command-line-args-left (delete "--check-child-priority" command-line-args-left))) (if (null command-line-args-left) (progn - (princ "Usage: emacs --batch -q -l todo-cleanup.el [--check] [--archive-done | --convert-subtasks | --sync-child-priority | --check-child-priority] FILE...\n") + (princ "Usage: emacs --batch -q -l todo-cleanup.el [--check] [--archive-done | --seal | --convert-subtasks | --sync-child-priority | --check-child-priority] FILE...\n") (kill-emacs 1)) (let ((files command-line-args-left)) (setq command-line-args-left nil) @@ -912,6 +1064,7 @@ ert-run-tests-batch-and-exit'." (cl-every (lambda (a) (cond ((member a '("--check" "--archive-done" + "--seal" "--convert-subtasks" "--sync-child-priority" "--check-child-priority")) diff --git a/.ai/sessions/2026-07-19-21-15-launcher-hardening-inbox-hook-colloquialisms.org b/.ai/sessions/2026-07-19-21-15-launcher-hardening-inbox-hook-colloquialisms.org new file mode 100644 index 0000000..8f5207c --- /dev/null +++ b/.ai/sessions/2026-07-19-21-15-launcher-hardening-inbox-hook-colloquialisms.org @@ -0,0 +1,238 @@ +#+TITLE: Session Context +#+AUTHOR: Craig Jennings +#+DATE: 2026-07-18 + +* Summary + +** Active Goal + +Three tasks shipped this session, all closed, pushed, velox synced (HEAD f6a2701). No active goal remaining; next is a fresh backlog pick. + +1. ai-launcher-hardening [#C] (113e8d8, 2b619f1, closed 33c6d7b): hardened =claude-templates/bin/ai= per its measurable acceptance criteria. Launcher tests 9 → 42, four pure decision cores extracted (=_git_prep_action=, =_order_windows=, =_match_window_id=, =_git_is_dirty=) each N/B/E, footgun audit + /refactor pass fully dispositioned, shellcheck clean + shfmt -i2 -ci + suite green before/after + live smoke correct. +2. inbox-boundary-check hook [#B] (94e54f6, closed f6a2701): soft-nudge Stop hook enforcing the task-boundary inbox check. 6 bats, wired in settings.json + snippet, protocols note. Takes effect next session. +3. colloquialisms / "the list" convention [#B] (8fd9e39, closed f6a2701): protocols.org Colloquialisms section + wrap-it-up Step 1 Before-Close Queue sub-step. 4 bats. Live now. + +KB: promoted 0 / consulted no + +** Decisions + +- The task's acceptance criteria are FIXED and live in the task body (todo.org, the 4 moves per todo-format.md's "Making an open-ended task measurable"). Follow them; don't re-derive scope. +- Interactive runtime picker is OUT of this :solo: scope (design call) — file separately if wanted. +- Characterization discipline per testing.md (refined this session, commit 179c495): record-not-spec, full Normal/Boundary/Error set per unit, negative/boundary cases are the bug-finders, extract-pure-core-when-IO-blocks IS the hardening. + +** Data Collected / Findings + +- Surface (22 fns). COVERED (via scripts/tests/ai-launcher-runtime.bats, 9 tests, runtime path): resolve_agent_cmd, build_runtime_choices, pick_runtime, build_instructions, print modes. UNCOVERED (17 to net): usage, check_deps, attach_session, create_window, maybe_add_candidate, build_candidates, fetch_candidates, git_status_indicator, annotate_candidates, auto_pull_if_clean, read_selections, sort_windows, find_window_id, prep_git_single, attach_mode, single_mode, multi_mode, print_launch_mode. +- Pure/near-pure (characterize directly, N/B/E): git_status_indicator, maybe_add_candidate (dedup), annotate_candidates (format), read_selections (parse), usage. tmux/git-coupled (extract pure core + thin wrapper): sort_windows (ordering), create_window, attach_session, find_window_id, prep_git_single, auto_pull_if_clean. ~3 functional tests over single_mode/multi_mode/attach_mode against a throwaway tmux session. +- Canonical bin/ai is claude-templates/bin/ai; installed via make install's bin loop (symlink to ~/.local/bin/ai). Tests live in scripts/tests/ (not .ai mirror). shellcheck + shfmt present; kcov NOT installed. + +** Files Modified + +This session (all pushed to origin/main, velox synced to d49be09): sentry build a8b6cf4/ccc9c26/c6383e9/8c0a56b, trial fix beb7f0b, gui-open b3195e9, flashcard apkg converter a143679 + multi-tag a14e43b, task closes a760d8e, knowledge-arch landing 179c495/2e19048/94df71e, flake fix 94015e6, launcher scoping d49be09. Nothing in flight — tree clean at the flush. + +** Next Steps + +Launcher hardening is done and pushed. Pick the next backlog task. Strong candidates surfaced this session: the two [#B] :feature: shared-asset proposals (todo-cleanup dated-seal already shipped ddbd47f; remaining backlog includes the inbox-boundary-check hook, the colloquialisms/the-list convention, and the build-to-prototype ui-prototyping extension). Manual: the sentry overnight live-trial on ratio (4-part task) stays Craig's to run. + +STANDING (still in force this session): run the suite as its OWN step and read it green before each commit; sync velox (git pull + make install, ssh 100.127.238.103) after each push; /review-code + /voice personal per commit; keep velox current. + +* Session Log + +** 2026-07-18 Sat 17:52 CDT — Startup + inbox inventory + +Ran startup.org. Phase A.0: rulesets pull skipped (dirty tree — =.claude/settings.json= carries a harness-written model flip opus → claude-fable-5[1m] against the committed opus pin bd76d98; needs Craig's call). =make install= nothing new. Project fetch clean. Phase A: no crash anchor (clean prior wrap), =.ai/= synced from templates, staleness 3 tasks >7 days, roam inbox 18 items (all foreign — archsetup/takuzu/home/clock-panel, none rulesets-claimed), KB 97 nodes / no relevant titles / best-practices path unresolved, spec-sort + host-identity probes silent. + +Inbox: 7 pending. Acted on the two trivial ones immediately: +- website priority-scheme FYI — deleted (pure FYI, loop already closed by their "done"). +- website roam-KB-hosting-moved — applied the factual origin-URL fix in =claude-rules/knowledge-base.md= (git@cjennings.net:roam.git → cjennings@cjennings.net:git/roam.git), verified ratio's =~/org/roam= remote already points at the new URL. Deleted the inbox file. Commit + reply to website pending. Surfaced to Craig: rulesets.git itself is still publicly browsable on cgit (his open decision, per website's note). + +** 2026-07-18 Sat 18:02 CDT — Triage-intake redesign applied (item 1 of 3) + +Craig approved option 1 (apply as sent). Copied both sent canonicals over =claude-templates/.ai/workflows/{triage-intake,daily-prep}.org=, synced the mirror via =sync-check.sh --fix=, full suite green (374 + 67 pytest, all ERT expected, all bats ok, make exit 0). Review ran (/review-code --staged, inline): verdict Approve — the two remaining "suggested-actions" mentions are dated history entries (correct to keep), ORDER still governs the on-request long form, daily-prep line 58 keeps old "Action items" wording but is behaviorally accurate (Minor, not fixed). Commit 6e48714 (voice-passed, gate skipped — .ai/ tracked). Deleted the 3 inbox files, replied to work (delivered to their inbox 18:02). + +Three substantive shared-asset proposals queued for skeptical-review surfacing: +1. work 07-17: todo-cleanup.el --archive-done dated-seal model (retain 7→31, unparseable-CLOSED archiving, --seal rename). Craig ratified the design at work. Verified: todo-cleanup.el still has retain default 7, no seal — proposal premise current. +2. home 07-17: strip SCHEDULED/DEADLINE on dated-rewrite completion (todo-format.md + todo-cleanup --convert-subtasks) + new lint-org checker dated-log-heading-active-timestamp. +3. work 07-18: triage-intake Phase C/D redesign (three-section digest TASKS/FYI/MISC, close-by-default, reroute modifier) + daily-prep 3b one-liner — Craig's 2026-07-18 ruling, edited canonicals attached. Verified: diffs coherent, author lines clean (no "& Claude" regression), no other template references the retired format, plugins untouched by design. + +** 2026-07-18 Sat 18:11 CDT — Items 2 & 3 filed; 2 new home handoffs arrived + +Craig said "proceed" → filed both items 2 and 3 as [#B] :feature:solo: in todo.org, each with the design preserved to docs/design/ and cross-linked (both touch todo-cleanup.el, build as one batch). Replied to work (filed) for item 2. Route checks: both none (local keepers). settings.json resolved itself — Craig's /model opus set it back to match the committed opus pin (bd76d98), tree now clean on that file. + +Task-boundary inbox check caught 5 NEW home handoffs (17:53 + 18:04), two distinct proposals: +- Item 4: upcoming-birthdays feature promotion (cover + upcoming_birthdays.py + test + full daily-prep copy). Reviewed the code — clean, stdlib-only, 19 pytest cases (Normal/Boundary/Error, leap-day, placeholder-year 1900→None, window boundaries, CLI). Isolated the daily-prep reconcile: home's copy PREDATES my triage change, so only two birthday hunks apply (Heads-Up item 2 + Phase A source 8) — NOT a wholesale overwrite, which would revert my 3b triage edit. Home flagged this correctly. +- Item 5: "Colloquialisms and Expansions" + "the list" before-close-queue convention. Touches protocols.org + wrap-it-up.org (both synced) — a norm-adoption design call. + +** 2026-07-18 Sat 18:22 CDT — Item 4 (birthdays) applied + +Craig approved. Copied upcoming_birthdays.py + test into claude-templates/.ai/scripts/, applied ONLY the two birthday hunks to canonical daily-prep (verified my 3b triage edit survived — both copies still carry "three-section digest"), synced mirror. Suite green pytest 374→393 (19 new, all pass). Commit 80ebb74. Deleted 4 inbox files, replied to home (delivered 18:22). Also answered Craig's side question: all 27 projects carry the task-boundary inbox-check instruction (protocols.org synced), but it's a prose behavioral rule, not an enforced hook. + +Remaining: item 5 (colloquialisms/the-list convention). Still uncommitted from earlier: knowledge-base.md roam-URL fix + todo.org filings (items 2,3) + 2 docs/design proposals — batch at close-out. Craig also asked to explore a hook design for the inbox check after item 4. + +** 2026-07-18 Sat 19:56 CDT — Hook exploration, item 5 filed, inbox closed, all committed + +Explored the inbox-check hook with Craig. Key framing that emerged: two rails split by "must fire at a wall-clock instant, or just needs to be seen soon?" — boundary rail (Stop hook / UserPromptSubmit, fires when the agent yields, never interrupts mid-task) vs cron/at (must-fire-now: meeting alarms, meds, deadlines). The Stop event maps onto the rule's own boundary definition ("before reporting back"). Recommended soft-nudge Stop hook (stop_hook_active guard, once per turn) + UserPromptSubmit visibility injector. Filed as [#B] :feature: with full design. Craig extended the idea to a general boundary-check reminder rail (session-save nudge, uncommitted-drift nudge, soft reminders coming due) — offered a design note, he declined (not restarting, so no cold-start concern). + +Craig aborted the wrap (not restarting after all) and said complete the tasks. Actions: +- Filed item 5 (colloquialisms / "the list" convention) as [#B] :feature: — a cross-project convention adoption touching protocols.org + wrap-it-up.org, so filed not applied; needs Craig's adoption decision + a short design pass. Replied to home, design preserved to docs/design/. +- New archsetup FYI arrived mid-work (package-drift audit accepted on their side) — pure loop-close, deleted, no reply owed. +- Committed everything: 6523ed5 (knowledge-base roam URL fix), bc1d81d (4 backlog task filings + 3 docs/design proposals). Plus earlier 6e48714 (triage redesign), 80ebb74 (birthdays). + +STATE: inbox at zero, tree clean, suite green. 4 commits ahead of origin/main (6e48714, 80ebb74, 6523ed5, bc1d81d), 0 behind — UNPUSHED, awaiting Craig's push call. :LAST_INBOX_PROCESS: stamped 2026-07-18. + +Backlog filed this session (all [#B]): todo-cleanup dated-seal, dated-log planning-line strip + lint checker (batches with the seal), inbox-boundary-check hook, colloquialisms/the-list convention. + +** 2026-07-18 Sat 20:06 CDT — Two [#C] :quick:solo: closeouts (power-through) + +Craig picked the quick wins first. Both DONE + CLOSED, task-shaped (top-level): +- coverage-summary.el local-only doc (2cb7c1b, pushed after this): stated local-only status in the .el commentary header + elisp-testing.md "Measuring it" section (the gitignored .claude/scripts/ install is by-design, not a CI gap — Craig's 2026-06-28 decision). Sent emacs-wttrin a handoff to revert its contradicting header claim. +- install-ai on PATH (d2b1bef): new claude-templates/bin/install-ai thin launcher, resolves its own symlink chain and execs scripts/install-ai.sh. make install's existing bin loop links it to ~/.local/bin/install-ai (same as ai/agent-page) — resolves the task's open question (no dedicated sync, no dotfiles copy; the symlink is the canonical). 3 launcher bats incl. symlink-invocation. Verified live: install-ai --help runs from PATH. Suite green pytest 393, all bats/ERT pass. + +Remaining top solo work: the two batched todo-cleanup [#B] :feature:solo: tasks (dated-seal + planning-line strip) — the strongest next power-through target. + +** 2026-07-18 Sat 21:04 CDT — Batched todo-cleanup build (both [#B] :feature:solo: DONE) + +Craig said yes to the batch. Built both TDD in one commit ddbd47f (10 files, +746/-79), both DONE+CLOSED. + +Task A (dated-seal): retain default 7→31 via new defconst tc-archive-retain-days-default; unparseable-CLOSED archiving made explicit in the contract/commentary; new --seal mode (tc-seal-archive-file) renames task-archive.org → resolved-YYYY-MM-DD.org beside it, dated by seal run (not quarter — avoids late-quarter mislabeling), next --archive-done recreates fresh working file; sealed file inherits gitignore status; --seal flag + dispatch + report + CLI recognizer. 5 ERT tests. + +Task B (planning strip + lint): --convert-subtasks now strips the WHOLE planning line (CLOSED+SCHEDULED+DEADLINE) via tc--strip-planning-lines-in-entry, reversing the old CLOSED-only behavior (the enshrining test tc-convert-preserves-deadline... was rewritten to assert stripping). Stops at first non-planning line so body prose survives. New lint-org checker dated-log-heading-active-timestamp (flags active <..> SCHEDULED/DEADLINE on a keyword-less dated heading, ignores inactive [..]). todo-format.md sub-task rule step 5 + VERIFY path. 5 convert + 5 lint tests. + +Debugging note: seal tests failed only in the FULL suite — root cause was a latent bug, tc-test--reset never cleared tc-convert-subtasks, so a convert test's mode flag leaked and won the tc-process-file cond over seal. Fixed the reset + made the seal harness set all mode flags explicitly. Full suite green (todo-cleanup 52, lint-org 57, pytest 393, all bats). Replied to work + home (delivered). + +STATE at 21:04: ddbd47f committed, NOT yet pushed (2 quick-win commits 2cb7c1b/d2b1bef already pushed earlier). Reconcile clean before commit (0/0). + +** 2026-07-19 Sun 04:35 CDT — SENTRY BUILD STARTED (no-approvals + auto-flush) + +Craig approved the sentry spec after his deep read and told me to build it in no-approvals + auto-flush mode. Big pivot from the launcher-hardening [#C] (that was read-only, no edits to bin/ai — left as-is, still TODO). + +Spec flipped DRAFT → READY → DOING (docs/specs/2026-07-14-sentry-workflow-spec.org, ID f6c51f27-d7a2-4b63-9ff9-5ba005a66dfb). Metadata Status → doing. Build task decomposed into 4 phase sub-tasks under "** DOING [#B] Sentry workflow" with :SPEC_ID: binding. + +BUILD PLAN (resume anchor — re-read the spec's Design + Decisions + Implementation phases if context was cleared): +- Phase 1: =agent-lock= helper (canonical claude-templates/.ai/scripts/) — mkdir-atomic acquire, PID/host/ISO metadata, age-staleness reclaim (surfaced), bounded-wait contention (~30s), heartbeat refresh, acquire/release/status subcommands. Home: /run/user/<uid>/agent-locks/<name>/ with ~/.cache/agent-locks/ fallback. bats-tested. TDD. NOTHING calls it yet. +- Phase 2: =sentry.org= engine (.ai/workflows/, mirror synced) — entry ticket :COMMIT_AUTONOMY:, interactive entry gates, ff-only reconcile, sentry/YYYY-MM-DD-<host> branch, 10-pass runner (probe→work→session-context→commit), digest + approval queue, skip-and-note, spine-exclusion + fire-end digest commit, stall-notify after 2 skips, stop-sentry op. INDEX.org. +- Phase 3: companion reconciliations — knowledge-base.md, inbox.org core §5, roam-sync.sh header, triage-intake.org, wrap-it-up.org active-sentry guard. +- Phase 4: make test green; file live-trial follow-ups (Craig runs the trial); flip spec → IMPLEMENTED, close parent. + +Mode: no-approvals (commit+push per logical unit, Step 0 reconcile + pre-push check still run), auto-flush armed. COMMIT_AUTONOMY: yes + LOOP_MAY_COMMIT: yes in notes.org. + +Starting Phase 1 (agent-lock, TDD). + +** 2026-07-19 Sun @ 04:38:35 -0500 — flushed +Auto-flush before the sentry build. Clean boundary: spec flipped DRAFT→READY→DOING and decomposed (495e36b, pushed), nothing half-edited. Resuming into Phase 1 (agent-lock helper, TDD) per Summary Next Steps. + +** 2026-07-19 Sun @ 04:52 CDT — PHASE 1 DONE (agent-lock), pushed + +Built the agent-lock helper TDD (canonical claude-templates/.ai/scripts/agent-lock + mirror). Subcommands acquire/refresh/release/status/path. mkdir-atomic acquire, PID/host/ISO/TTL metadata, mtime-based staleness reclaim (surfaced note), heartbeat refresh, XDG_RUNTIME_DIR/agent-locks home with ~/.cache fallback. 18 bats tests. + +Self-review (review-code --staged, inline) caught one real concurrency bug pre-commit: the stale-reclaim path was rm -rf + mkdir in two steps, letting two acquirers who both see a lock stale double-acquire (loser deletes winner's fresh dir). Fixed with atomic-rename claim (mv wins-or-fails, then mkdir stays sole grant); strengthened test 7 to assert fresh metadata after reclaim. Verdict cleared to Approve. + +Full suite green (make test exit 0, 0 not-ok; agent-lock 18/18). Commit a8b6cf4, pushed to origin/main (0 behind, pre-push reconcile clean). /voice personal ran on the message. + +NEXT: Phase 2 — sentry.org engine (.ai/workflows/, mirror synced). Entry ticket :COMMIT_AUTONOMY:, interactive entry gates (dirty-tree / red-suite), ff-only reconcile, sentry/YYYY-MM-DD-<host> branch, pass runner (probe→work→session-context→commit contract), digest + morning-approval queue, skip-and-note semantics, spine-exclusion + fire-end digest commit, stall-notify after 2 unmerged-branch skips, stop-sentry op, INDEX.org entry. Re-read spec Design paragraphs (branch mechanics, locks, roam writes, unattended safety, pass list, digest) + the 10 decisions. + +** 2026-07-19 Sun @ 05:06 CDT — SENTRY BUILD COMPLETE (all 4 phases, pushed) + +All four sentry phases shipped, committed, pushed to origin/main, suite green throughout: +- Phase 1 a8b6cf4 — agent-lock helper (18 bats). Pre-commit review caught + fixed a reclaim double-acquire race (atomic-rename claim). +- Phase 2 ccc9c26 — sentry.org engine + INDEX entry. All 10 decisions / 12 findings reflected. +- Phase 3 c6383e9 — roam writers (knowledge-base.md, inbox.org §5) acquire roam-write lock + edit-plus-trigger (roam-sync sole committer); roam-sync.sh header; triage-intake note; wrap-it-up Step 0 active-sentry guard. Lock-name derivation pinned identically in sentry.org + wrap-it-up (sentry-<repo-basename>). +- Phase 4 8c0a56b — make test green at HEAD; spec DOING→IMPLEMENTED (dated history + Metadata mirror); build task + phase sub-tasks closed (sub-tasks → dated event-log entries); live trial filed as structured "Manual testing and validation" task. + +FINAL STATE: tree clean (only untracked spine), 0/0 vs origin/main, spec IMPLEMENTED, sync-check clean. Live agent-lock smoke test on the real runtime dir passed (acquire→held on ratio→release). Each commit ran /review-code (inline for docs) + /voice personal. + +REMAINING (Craig's, not agent-buildable): the overnight live-trial night on ratio — arm sentry, exercise the entry gates, observe one fire end to end, run the morning branch review. Filed as the 4-part manual-testing task in todo.org. Its findings become follow-up tasks. The launcher-hardening [#C] (read-only bin/ai) is still TODO, untouched (was pre-empted by the sentry build). + +** 2026-07-19 Sun @ 15:38 CDT — Live trial feedback #1 processed (roam-denylist mis-park) + +Craig ran a full sentry session from the WORK project. First trial finding: the inbox-zero pass (P2), running from ~/projects/work, parked the whole 19-item roam inbox as a cross-project boundary crossing and refused to tidy it — over-reading knowledge-base.md's work-denylist as "don't touch roam from work." Craig ruled: roam is a shared resource; the denylist only ever gated durable agents/ KB-node writes (a confidentiality guard). Reading roam + tidying the roam inbox are allowed from any project, work included. + +Fix (commit beb7f0b, pushed): applied work's prepared knowledge-base.md "Scope of the denylist — durable KB-node writes only" paragraph (naming the 2026-07-19 mis-park), plus one-line companion notes in sentry.org P2 and inbox.org roam mode pointing at the rule. Suite green. Replied to work (inbox-send, delivered 15:35), deleted the two work inbox items. + +Commits so far: a8b6cf4, ccc9c26, c6383e9, 8c0a56b (build), beb7f0b (trial fix #1). All pushed, tree clean. + +PENDING: 2 untracked archsetup items in inbox/ (11:47, gui-open proposal — unrelated to sentry, separate pass). Awaiting any further trial findings from Craig. + +** 2026-07-19 Sun @ 15:44 CDT — Inbox cleared: archsetup gui-open proposal applied + +Processed the two archsetup items (gui-open proposal). gui-open is a dotfiles-shipped launcher (a34d479, on PATH via ~/.dotfiles) that shows a file from a short-lived agent shell reliably — detaches through systemd-run --user (no shell reap), resolves the Hyprland instance after a restart, verifies a visible client. Cleared the value gate (fixes documented fragility both rules already flagged). + +Fix (commit b3195e9, pushed): interaction.md "Showing Craig Visuals" + desktop-capture.md "Showing the user something" now launch via gui-open instead of google-chrome-stable ... & / hyprctl dispatch exec imv. Guidance-only (tool is dotfiles-owned, not a rulesets script); added an "if not on PATH, needs a dotfiles pull" note. Replied to archsetup (delivered 15:42), deleted both inbox files. Inbox now clear. + +DAILY-DRIVER NOTE: gui-open ships via dotfiles a34d479 — confirmed on ratio (symlinked ~/.local/bin/gui-open → ~/.dotfiles). velox needs a dotfiles pull to have it, or the new rule guidance references a missing tool there. Flag to Craig. + +Session commits total: a8b6cf4, ccc9c26, c6383e9, 8c0a56b (sentry build) + beb7f0b (trial fix #1) + b3195e9 (gui-open guidance). All pushed, tree clean, inbox zero. + +** 2026-07-19 Sun @ 15:52 CDT — velox brought current (standing: keep it synced this session) + +Craig: do the dotfiles pull on velox now, and keep velox up to date with our changes for the rest of the session. Reached velox over tailscale (100.127.238.103, cjennings@). +- dotfiles: git pull → c3ef604 (was 9914a25), make restow hyprland (clean, no conflicts). gui-open symlinked ~/.local/bin/gui-open → dotfiles common tier; verified: direct exec prints usage, ~/.local/bin on interactive-login PATH (positions 1-2), interactive shell resolves it. Earlier "MISSING" was only the non-interactive SSH PATH. +- rulesets: git fetch + merge --ff-only → b3195e9 (was 3ae71be; also caught up on upcoming_birthdays, install-ai, docs/design proposals). make install relinked install-ai + agent-page, rest skipped. + +STANDING INSTRUCTION (rest of session): after each push, also update velox — git pull + make install on ~/code/rulesets, and a dotfiles pull + restow if dotfiles changed. ssh via 100.127.238.103. zsh gotcha: don't word-split an unquoted $VAR for the ssh command; run ssh inline. Non-interactive SSH PATH lacks ~/.local/bin — test tools via resolved path or zsh -lic. + +** 2026-07-19 Sun @ 18:42 CDT — Speedrun complete (2 flashcard tasks shipped) + +No-approvals speedrun over the 2 flashcard :solo: tasks, both TDD + review + voice, each its own commit, pushed, velox synced after each. +- apkg-to-orgdrill.py (a143679): inverse converter, stdlib zipfile+sqlite3, 17 tests + real-genanki round-trip. Grounded the apkg schema by generating/inspecting a real one first. +- flashcard multi-tag reconcile (a14e43b): broadened CARD_RE in to-anki + stats for :fundamental:drill:, added --tag-filter + --guid-salt + drill-membership guard; re-derived against current canonical (kept the #+TITLE fix); parse() 3rd element now anki-tag list. End-to-end verified (2→1 with --tag-filter). 465/100 real-deck check needs the work deck (not runnable here). +- Closed both as dated event-log entries (a760d8e). + +Suite green throughout (419 pytest). Session commits now: sentry build (4) + trial fix + gui-open + flashcard (2) + closes. All pushed, velox current at a760d8e. + +NEXT (Craig asked): discuss the ai launcher hardening [#C] task (todo.org line ~208). Remind him what it is. + +** 2026-07-19 Sun @ 19:04 CDT — Landed the testing/acceptance knowledge-architecture change + +Craig's directive: build the characterization-test + measurable-acceptance discipline into the workflows, and decide the KB-vs-rule boundary. Answer settled in discussion: apply-every-time discipline → rules (single source, auto-loaded, review-gated contributions via inbox); pull-when-relevant cross-project facts → KB. Almost none of this is KB material (it has a rule home); the KB is the capture/holding-pen, promotion moves it OUT into rules. + +Two commits: +- 179c495 (characterization discipline): testing.md — defined a characterization test (record-not-spec, Feathers recipe), required the full Normal/Boundary/Error set per unit, and the key lens that negative/boundary cases are the bug-finders; + the "extracting the pure core IS the hardening" framing in the refactor-for-testability section. code-quality.org gained the precondition that behavior-preserving rests on a characterization net. +- 2e19048 (measurable acceptance): todo-format.md new subsection "Making an open-ended task measurable (so it can be :solo:)" — bound surface / characterization net / disposition findings / objective floor; qualifying answer = dispositioned report. work-the-backlog's keystone defer item now recognizes absence-phrased open-ended goals and routes them to get criteria. + +Scope note: start-work and /refactor are NOT rulesets files (not installed at ~/.claude/skills/; registered elsewhere). Their pointers are a follow-up in whatever repo defines them — flagged to Craig. add-tests + review-code (rulesets skills) already well-wired to testing.md. + +Suite green, lint 0/0, velox synced to 2e19048. + +** 2026-07-19 Sun @ 19:14 CDT — Flaky-test fix + a discipline note + +While landing the start-work pointer (94df71e), chained make test into the commit command and the commit went through on a RED run — the bundling verification.md warns against. The red turned out flaky (rename-ai-artifact.bats teardown race, unrelated to the md edit; suite green on re-run, change sound), but the gate failed open. Correction going forward: run the suite as its own step, read it, then commit. + +Root-caused + fixed the flake (94015e6): rename-ai-artifact.bats teardown hit "rm: cannot remove .git: Directory not empty" because git background auto-maintenance (maintenance run --auto after commit) wrote into .git after the body, racing rm -rf. Fix: gc.auto 0 + maintenance.auto false in the throwaway repo before any git command arms them. Diagnose-not-mask (no rm retry). Verified 30 file runs + 2 full-suite runs green. + +Also corrected an earlier wrong claim: start-work + refactor ARE rulesets files (.claude/commands/, symlinked to ~/.claude/commands), already wired to testing.md for characterization. The acceptance-pattern landing is fully complete — no out-of-repo follow-up. + +** 2026-07-19 Sun @ 19:20 CDT — Scoped the ai-launcher-hardening task (dogfooded the pattern) + +Applied the measurable-acceptance pattern to the launcher task (d49be09). Rewrote the open-ended body with the 4 moves: bounded surface (22 fns, 5 covered via ai-launcher-runtime.bats / 17 uncovered, all named), characterization-net plan (pure-core extraction called out for the tmux/git-coupled fns), dispositioned audit (per-fn footgun matrix + /refactor pass), objective floor (shellcheck/shfmt/green suite/per-fn coverage + ~3 functional tests over single/multi/attach_mode). Split the interactive runtime picker OUT of :solo: (design call). Task is now genuinely :solo:-ready. :LAST_REVIEWED: bumped to 2026-07-19. + +The actual hardening work (bring 17 fns under characterization tests, extract pure cores, run the audit + refactor) is a substantial next-session speedrun candidate — now bounded. + +** 2026-07-19 Sun @ 20:19:15 -0500 — flushed +Interactive flush at a clean boundary (launcher task scoped + committed d49be09, tree clean, nothing half-edited). Resuming into option 2: work the ai-launcher-hardening characterization sweep per the refreshed Summary Next Steps. + +** 2026-07-19 Sun @ ~21:30 CDT — LAUNCHER HARDENING COMPLETE (2 commits + close, pushed, velox synced) + +Worked the ai-launcher-hardening [#C] :refactor:solo: end to end per its measurable acceptance criteria. Craig present, chose push+sync+close. + +Sequence: +- Grounded in bin/ai (540 lines), confirmed green baseline (make test exit 0), pinned the gate targets: shellcheck-clean (the bash hook enforces shellcheck only; shfmt is deliberately NOT hook-gated), shfmt house style = -i 2 -ci (agent-page conforms, bin/ai did not). +- Commit 113e8d8 (net + source-testable): wrapped dispatch in main() behind a run-vs-sourced guard so the file is sourceable for unit tests; cleared 4 pre-existing shellcheck warnings on touched paths (@{u} quoted ×3 = SC1083, literal display tilde disabled = SC2088). New scripts/tests/ai-launcher-characterization.bats, 20 tests: pure/near-pure N/B/E (git_status_indicator, maybe_add_candidate, annotate_candidates, read_selections, build_candidates, usage) + 4 functional (create_window, find_window_id, sort_windows, attach_mode) against a PRIVATE tmux socket (TMUX_TMPDIR + unset TMUX) so nothing touches Craig's live ai session; throwaway repos disable gc.auto/maintenance.auto (the rename-ai flake). +- Commit 2b619f1 (refactor): extracted four pure cores — _git_prep_action (git-prep none/pull/report classifier, shared by prep_git_single + auto_pull_if_clean), _order_windows (sort_windows ordering), _match_window_id (find_window_id), _git_is_dirty (DRY the dirty triad ×3). Added 13 unit tests for the cores incl. the safety case (dirty repo never auto-pulled). shfmt -i2 -ci applied. 42 launcher tests green, live black-box smoke correct for claude+codex. +- Commit 33c6d7b: closed the task DONE+CLOSED (top-level ** → task-shaped) with the dispositioned resolution note. + +Footgun audit fully dispositioned (report delivered inline to Craig): unquoted-expansion/word-split — clean; errexit-off — intentional/documented, kept; subshell state loss — none (globals set in main shell); exit-code propagation — intentional; sort_windows race — none (two-pass park-then-reassign is correct); create_window sleep 0.1 — declined (intentional); git-prep error paths (dirty/detached/no-upstream) — all correct, dirty-safety now a named test. /refactor: 4 extractions + shfmt applied; build_candidates exit-status + create_window sleep declined. + +Honest coverage limit stated: attach_session and full end-to-end of single/multi/fetch stay partially covered (terminal step attaches/blocks on fzf, can't run headless); their decision logic was extracted into the netted cores. + +STATE: 33c6d7b pushed, 0/0 vs origin/main, velox synced to 33c6d7b (make install: ai linked), tree clean (only untracked session anchor), suite 367 ok / 0 not-ok. Interactive runtime picker stayed OUT of scope (design call, on the generic-agent-runtime parent). No memory promotion needed (nothing durable/cross-project beyond the repo record). + +** 2026-07-19 Sun @ ~22:00 CDT — No-approvals speedrun: inbox hook + colloquialisms (both [#B] DONE) + +Craig said "no approvals speedrun time" after the launcher task. Built two [#B] backlog tasks, each TDD + inline review + /voice personal, committed + pushed + velox-synced. + +Task 1 — inbox-boundary-check hook (94e54f6): soft-nudge Stop hook backing the "check inbox/ at every task boundary" rule (was prose-only). Blocks the yield once + injects the pending count when =inbox-status -q= exits 1, steps aside on =stop_hook_active= re-entry (soft, not hard-block, so a mid-task pause never wedges), self-skips on no-inbox/no-inbox-status/clean. Prefers project-local =.ai/scripts/inbox-status=, PATH fallback. Wired ahead of =ai-wrap-teardown= in =.claude/settings.json= (which the live =~/.claude/settings.json= symlinks to → live global on both boxes) + =settings-snippet.json= + README row; glob-installed by =make install-hooks=. protocols.org Inbox Monitoring Cadence notes the enforcement. 6 bats. Live-verified on ratio (clean no-op, pending fixture blocks). Loads next session (hooks read at session start), so it didn't interfere with this wrap. + +Task 2 — colloquialisms / "the list" (8fd9e39): new =* Colloquialisms and Expansions= section in protocols.org. "put X on the list" → session-scoped Before-Close Queue (=* Before-Close Queue= heading in the session anchor, resets on archive, todo.org for must-outlive); "tell <project> <msg>" → =inbox-send=. wrap-it-up Step 1 gained a "Work the Before-Close Queue (before the Summary)" sub-step so queued work rides the wrap commit. Design calls: reference in protocols.org not per-project notes.org (synced = shared norm); queue in the session anchor as home did; wrap step at front of Step 1 not a half-step (keeps "Steps 1-5" framing). 4 documentation-integrity bats. Canonical+mirror synced. + +Both closes in f6a2701. Full suite green throughout (0 not-ok). velox synced to f6a2701 (make install ok). Then Craig said wrap it up. diff --git a/.ai/sessions/2026-07-20-23-36-signal-pager-and-triage-phone-push.org b/.ai/sessions/2026-07-20-23-36-signal-pager-and-triage-phone-push.org new file mode 100644 index 0000000..fe43f61 --- /dev/null +++ b/.ai/sessions/2026-07-20-23-36-signal-pager-and-triage-phone-push.org @@ -0,0 +1,185 @@ +#+TITLE: Session Context — Sentry live trial (ratio, 2026-07-19) +#+AUTHOR: Craig Jennings + +* Summary + +** Active Goal + +Post-flush session. Two items shipped: (1) the archsetup voice-#46 (comma-budget) handoff, reviewed and committed rulesets-side (2ea5d9a); (2) the triage-intake auto-mode phone-push, built onto agent-text, signal-only per Craig's ruling (e27aea2). Items 2 (reply-correlation spec) and 3 (token-rotation discussion) from the prior next-steps list remain untouched, carried forward. + +** Decisions (this session, all shipped) + +- Voice pattern #46 (comma budget, max two per sentence): personal mode only, in the attestation high-recurrence set. Craig's 2026-07-20 direction via archsetup; committed rulesets-side 2ea5d9a. +- Triage auto-mode phone-push is signal-only (Craig's option 1): a quiet sweep's "nothing" heartbeat never pushes to the phone — silent-until-signal governs the phone channel too. Send half shipped (e27aea2); reply-polling deferred to the reply-correlation spec. +- Sentry live trial: ran on ratio, 8 hourly fires, stopped on "sentry off", branch fast-forwarded to main and deleted. +- working/ is tracked-from-creation + gitignored temp/ in both modes (Shape A). Committed. +- Triage source activation: general (synced) plugins gate on a per-project =:TRIAGE_SOURCES:= declaration; project-specific plugins active by presence; gate in triage-intake Phase 0 (interactive + unattended). Spec IMPLEMENTED. Migrations done (home + work declared their sources). +- Silent-until-signal: a POLICY not a mechanism — an in-session monitor fire heartbeats =<workflow> at HH:MM: nothing= on an empty check; detection stays in-session (MCP-safe). Applied to sentry, auto triage-intake, auto inbox-zero. Spec IMPLEMENTED. +- Suspend detach change applied to canonical. Sentry cluster consolidated (merged /schedule tasks, added cross-host-coordination task). Task audit stamped 2026-07-20. +- Polyglot: case-by-case, no option-2 machinery (bundles already compose; only coverage-makefile.txt collides, and it's a manual paste). Subprojects: don't promote (N=1). Both closed. + +** Data Collected / Findings — Signal pager (what the resume needs) + +RECONCILED (2026-07-13, in the task's dated log): there is ONE pager identity, =+15045173983=, registered in *velox's* signal-cli (account file 465310). =signal-mcp= is a velox-local MCP server (invisible from ratio). ratio's signal-cli holds only Craig's personal number (note-to-self, no push). =agent-page= already shipped (=claude-templates/bin/agent-page=): runs signal-cli directly on velox, ssh-relays from anywhere else, desktop-fallback on failure, 4 bats, live-verified. protocols.org "Paging Craig" was already rewritten around the two channels (notify desktop + agent-page phone). Craig's Signal UUID: =b1b5601e-6126-47f8-afaa-0a59f5188fde=. Reliability finding: both signal-cli accounts throw receive-staleness warnings (velox ~40 days, ratio ~26) — the Signal protocol wants regular receives, and a systemd receive timer on velox is the roam-sync-shaped fix. + +REMAINING deliverables (this is the work to do): (1) the runbook proper — send + read-replies + receive-timer + signal-cli account/setup notes, the Signal equivalent of the retired ntfy runbook, canonical home in rulesets docs/; (2) a systemd receive timer on velox (roam-sync-shaped) so receives don't go stale; (3) the ssh-over-tailnet-only vs register-ratio-as-a-linked-device decision (a design call for Craig); (4) confirm protocols.org "Paging Craig" is accurate (agent-page did most of it — verify, don't redo). Task body has full history. Source: home handoff 2026-07-04. + +** Files Modified (this session — all committed + pushed to main) + +Post-flush commits: 2ea5d9a (voice #46 comma budget — SKILL.md + voice-profile.org), 9721a49 (chore: mark archsetup handoff PROCESSED), e27aea2 (feat: triage phone-push via agent-text — canonical + mirror triage-intake.org, todo.org task closed). Left unstaged: .claude/settings.json model change (opus→fable, not this session's work — deferred). + +Earlier same-session (pre-flush): f625cf5 (working/temp feat), b02eade + 4d87f35 (triage source activation spec + build), 986d6ca + 0767af8 + 0e9958a (silent-until-signal spec + Phases), af565ba (suspend detach), 70fbe01 (link fixes), 93a2e6d (working-dir filing), c82b625 (sentry cluster), up to fecdf8c, plus the pager work (302b062, 6145489) and overnight sentry commits. + +** Next Steps + +Item 1 (triage phone-push) shipped this session. Two remain from the prior list: + +1. *Spec the reply-correlation follow-up.* The two-way gap: with the account linked on velox + ratio, a Signal reply reaches BOTH devices (one account, Signal fans out per-device, independent queues) and neither knows which page/session it answers. Only bites when two sessions page-and-wait at once (fire-and-forget is fine). Options laid out to Craig: (1) correlation tag stamped on the page, echoed in reply; (2) quote-reply matching (needs verifying signal-cli surfaces quotes); (3) single reply-owner (velox-only). Overlaps the helper-instance work (same shared-channel problem). This spec also owns the deferred triage-intake reply-polling (phone-recv) half. Also fix the runbook "Reading replies" section, which oversells it. Write as a spec in docs/specs/ (spec-create spine). + +2. *Discuss the token-rotation helper* (todo.org =[#C] Token-rotation helper for @a-bonus/google-docs-mcp OAuth refresh=, :feature:quick:) — a discussion first, not a build. Read the task body. + +Publish flow (review + voice + suite) for any commits. + +KB: promoted 0 / consulted no + +* Session Log + +** 2026-07-19 23:52 CDT — Sentry armed (entry) + +Startup ran clean: rulesets pull no-op, make install nothing new, project repo up to date on main (0/0). Session-context was absent (prior session 2026-07-19-21-15 wrapped cleanly). + +Sentry entry gates, all passed with Craig present: +- Autonomy ticket: =:COMMIT_AUTONOMY: yes= in notes.org Workflow State. +- Host: ratio (intended live-trial machine). +- Dirty-tree gate: tracked tree clean; only two untracked inbox files (=2026-07-19-2141-from-.emacs.d-version-working-directories.org=, =2026-07-19-2350-from-home-craig-approved-the-colloquialisms-the.org=) — do not block; sentry's inbox-zero pass (pass 2) handles them. +- Green-suite gate: =make test= exit 0, 377 bats ok, all pytest + ERT green. +- Prior sentry branch: none. +- Reconcile: main 0 behind / 0 ahead of upstream. +- =agent-lock= helper present. + +Created daily branch =sentry/2026-07-19-ratio= from HEAD (main @ f76bf40). Working tree now sits on this branch overnight. Arming the hourly loop next. + +Two inbox items pending at entry, both to be handled by pass 2 (inbox zero): +- .emacs.d working/ version-control ruling — a shared-asset change (working-files convention + install .gitignore behavior across projects). Parks for morning approval; does not fire unattended. +- home colloquialisms wiring approval — already shipped on the rulesets side in the prior session (protocols.org Colloquialisms section + wrap-it-up Before-Close Queue step, commits 8fd9e39/f6a2701). Inbox zero replies-and-files, confirming it's already wired. + +** 2026-07-20 00:06 CDT — Fire 1 digest (manual first fire, Craig present) + +Ran end to end on branch =sentry/2026-07-19-ratio=. Single-runner lock =sentry-rulesets= acquired at fire start, on the sentry branch, tree clean. Pass-by-pass: + +- P1 roam pull — SKIPPED: =~/org/roam= working tree dirty; roam-sync owns that case (pass is read-only ff-only otherwise). No write. +- P2 inbox zero — RAN. Two project handoffs processed. (a) home's colloquialisms + "the list" before-close-queue: already wired canonically before it arrived — replied to home confirming, marked PROCESSED. (b) .emacs.d's working/ tracked-from-creation ruling: shared-asset + convention change with a temp/-placement design decision → PARKED (see approval queue). Staged proposal at =working/sentry-2026-07-19-working-files-ruling/proposed.org=, filed a [#B] VERIFY, replied to .emacs.d with two findings. Committed 727a900. +- P3 triage intake — SKIPPED (probe-too-loose; queued as a finding). The template-synced general plugins (personal Gmail/calendar/cmail/Telegram/GitHub-PRs) are present in *every* project, so the "plugins present" probe self-activates triage everywhere — including rulesets, which is not a triage target. Running it would file Craig's personal action items into rulesets' todo.org and run trash/mark-read/star hygiene on his real accounts unattended. Skipped for scope + safety. No write. +- P4 todo cleanup — RAN: hygiene 0 fixes, --convert-subtasks 0, --archive-done 0. todo.org already clean. No write. +- P5 task audit — RAN (mechanical subset). No unambiguous autonomous staleness fixes: all open tasks carry a priority cookie + type tag except the intentional "Manual testing and validation" container; spec-lifecycle clean (P7); no dead file: links in todo.org. The full reconciliation (per-task fact-check, consolidation, parent-retirement, judgment flags) is interactive → queued. Did NOT bump :LAST_AUDIT: (only the mechanical subset ran). No write. +- P6 working-files hygiene — RAN: flagged =working/inbox-zero-phase-e/= (tracked, backs the now-IMPLEMENTED autonomous-batch-execution spec) as a filing candidate. Filing is a judgment move (3 inbound spec links need updating) → queued. No write. +- P7 spec status board — RAN: clean. All specs READY or IMPLEMENTED; the sentry spec is IMPLEMENTED; no DOING spec with a closed/missing bound parent. No write. +- P8 link integrity — RAN (report-only): ~4 candidate broken file: links in live docs — the autonomous-batch spec's =../../.ai/workflows/inbox-zero.org= (renamed to inbox.org), two folded-in spec-review siblings (wrapup-routing, pattern-catalog), and =subprojects-log.org=. Some may be intentional (review docs folded in and deleted). No unattended rewrites. No write. +- P9 git health — RAN: clean. main = origin/main (0/0), no unpushed commits on other branches, no stale merged branches, sentry branch is the only local extra. No write. +- P10 prep + symlink freshness — SKIPPED: no prep dir (work/home only). No write. + +Fire-end: only org/spine files touched this fire (todo.org, inbox renames, working/ proposal — all org) → conditional suite skipped per contract. Session-context spine is untracked in rulesets, so no digest commit needed; tracked tree is clean. Single-runner lock released. + +Branch state after fire: one commit ahead of main (727a900). Next scheduled fire at :17. + +** 2026-07-20 00:21 CDT — Fire 2 digest (scheduled, :17 cron) + +Lock =sentry-rulesets= acquired, on the sentry branch, tree clean. Little changed in the 14 min since fire 1 — one new inbox handoff, everything else steady. + +- P1 roam pull — RAN: =~/org/roam= clean this fire (was dirty in fire 1), ff-only pull → already up to date. No write. +- P2 inbox zero — RAN: one new handoff. .emacs.d routed three sentry-workflow design considerations from its own hand-run trial (codebase-gated bug/enhancement logging pass; system-health pass; sibling-machine freshness pass). Filed as =** TODO [#C] Sentry vNext passes — from live-trial design input= rather than applied (design input for the Living Document; two hinge on an unresolved cross-driver coordination question). Replied to .emacs.d, marked PROCESSED. Committed 067ed55. +- P3 triage intake — SKIPPED: same probe-too-loose reason as fire 1; the finding is already in the approval queue, not re-queued. +- P4 todo cleanup — RAN: 0 fixes. No write. +- P5 task audit — no change since fire 1; the full-audit-due item stays in the queue, not re-added. +- P6 working-files hygiene — RAN: =working/inbox-zero-phase-e/= still a filing candidate (already queued fire 1, not re-queued). =working/sentry-2026-07-19-working-files-ruling/= is an *active* working dir backing the open VERIFY task — correctly not a candidate. No new queue item. +- P7 spec status board — RAN: clean, no DOING specs. No write. +- P8 link integrity — not re-scanned: no doc changes since fire 1's scan except the new backlog task, whose one internal =file:todo.org::*...= link resolves (KB lesson-detection heading exists). Fire 1's report stands. No write. +- P9 git health — RAN: clean, main = origin/main (0/0), branch now 2 ahead. No write. +- P10 prep freshness — SKIPPED: no prep dir. No write. + +Fire-end: only org files touched → conditional suite skipped. Spine untracked → no digest commit; tracked tree clean. Lock released. Branch 2 commits ahead of main (727a900, 067ed55). No new approval-queue items this fire. + +** 2026-07-20 01:20 CDT — Fire 3 digest (scheduled, :17 cron) + +Quiet read-only fire — nothing changed in the hour since fire 2. All passes no-op or skip; no writes, no commit, no new approval-queue items. + +- P1 roam pull — RAN: clean, ff-only → already up to date. +- P2 inbox zero — RAN: 0 new items. +- P3 triage intake — SKIPPED: same probe-too-loose reason (finding already queued fire 1). +- P4 todo cleanup — RAN: 0 fixes. +- P5 task audit — no change; full-audit-due item stays queued. +- P6 working-files hygiene — RAN: same two dirs (inbox-zero-phase-e already queued; sentry-ruling active for the open VERIFY). No new item. +- P7 spec status board — RAN: clean, no DOING specs. +- P8 link integrity — not re-scanned: no doc changes since fire 1. +- P9 git health — RAN: clean, main = origin/main, branch 2 ahead. +- P10 prep freshness — SKIPPED: no prep dir. + +Branch unchanged at 2 commits ahead. Lock released. + +** 2026-07-20 02:20 CDT — Fire 4 digest (scheduled, :17 cron) + +Quiet read-only fire, same as fire 3 — nothing changed in the hour. All passes no-op or skip; no writes, no commit, no new queue items. P1 roam up to date; P2 inbox 0 new; P3 triage skipped (finding already queued); P4 0 fixes; P5 no change (full audit still queued); P6 same two working dirs (inbox-zero-phase-e queued, sentry-ruling active); P7 clean, no DOING specs; P8 not re-scanned (no doc changes); P9 clean, branch 2 ahead; P10 skipped (no prep dir). Lock released. + +** 2026-07-20 03:20 CDT — Fire 5 digest (scheduled, :17 cron) + +Quiet read-only fire, same as fires 3-4. No changes in the hour; all passes no-op or skip; no writes, no commit, no new queue items. P1 roam up to date; P2 inbox 0 new; P3 triage skipped (queued); P4 0 fixes; P5 no change; P6 same two working dirs; P7 clean; P8 not re-scanned; P9 clean, branch 2 ahead; P10 skipped. Lock released. + +** 2026-07-20 04:20 CDT — Fire 6 digest (scheduled, :17 cron) + +Quiet read-only fire, same as fires 3-5. No changes in the hour; all passes no-op or skip; no writes, no commit, no new queue items. P1 roam up to date; P2 inbox 0 new; P3 triage skipped (queued); P4 0 fixes; P5 no change; P6 same two working dirs; P7 clean; P8 not re-scanned; P9 clean, branch 2 ahead; P10 skipped. Lock released. + +** 2026-07-20 05:20 CDT — Fire 7 digest (scheduled, :17 cron) + +Quiet read-only fire, same as fires 3-6. No changes in the hour; all passes no-op or skip; no writes, no commit, no new queue items. P1 roam up to date; P2 inbox 0 new; P3 triage skipped (queued); P4 0 fixes; P5 no change; P6 same two working dirs; P7 clean; P8 not re-scanned; P9 clean, branch 2 ahead; P10 skipped. Lock released. + +** 2026-07-20 06:20 CDT — Fire 8 digest (scheduled, :17 cron) + +Quiet read-only fire, same as fires 3-7. No changes in the hour; all passes no-op or skip; no writes, no commit, no new queue items. P1 roam up to date; P2 inbox 0 new; P3 triage skipped (queued); P4 0 fixes; P5 no change; P6 same two working dirs; P7 clean; P8 not re-scanned; P9 clean, branch 2 ahead; P10 skipped. Lock released. + +* Sentry approval queue (2026-07-19) + +Judgment/parked items from the overnight fires. Review top to bottom; run or discard each. + +** [Fire 1 · P2] Apply the .emacs.d working/ tracked-from-creation ruling +- *What:* update the canonical working-files convention + add a gitignored temp/ pattern across projects. +- *Why:* Craig's ruling relayed from .emacs.d (2026-07-19). Parked because it's a shared-asset + convention change and carries a design decision (temp/ must be ignored in both track and gitignore modes — it's orthogonal to the personal-tooling set). +- *Prepared:* =working/sentry-2026-07-19-working-files-ruling/proposed.org= (exact edits + the two findings). Filed as the =** VERIFY [#B] working/ tracked-from-creation + gitignored temp/= task in todo.org. +- *Fires on approval:* edit =claude-rules/working-files.md= (tracked-from-creation + temp/ subsections), =scripts/install-ai.sh= + =scripts/sweep-gitignore-tooling.sh= (emit temp/ ignore in both modes — decide shape (a) vs (b) in the proposal), =.ai/protocols.org= Working-Files Convention (one-line mirror); then =scripts/sync-check.sh --fix= for the .ai mirror, run =make test=, commit. + +** [Fire 1 · P3] Tighten the sentry triage-intake pass probe — SPECCED during morning review +- *What:* the narrow "fix the probe" framing grew, on Craig's flexibility question, into a per-project source-activation model for triage-intake. +- *Resolution:* written up as a spec for review — =docs/specs/2026-07-20-triage-source-activation-spec.org= (DRAFT), linked from the =** TODO [#B] Triage source activation= task. General plugins gate on a per-project =:TRIAGE_SOURCES:= declaration; project-specific plugins stay active by presence; the activation layer lives in triage-intake Phase 0 so it fixes the interactive over-pull too; sentry's pass-3 probe reads the same signal. +- *Next:* Craig's deep read → DRAFT → READY → spec-response decomposes the 5 phases. Two decisions still open (declaration format, whether interactive adopts the same gate). No code landed — this item is now tracked by the spec + task, not the queue. + +** [Fire 1 · P5] A full interactive task-audit is due +- *What:* run task-audit.org interactively (its consolidation / parent-retirement / judgment-flag phases need Craig). +- *Why:* :LAST_AUDIT: is 2026-07-04 (~16 days); task-review-staleness flags 2 top-level tasks unreviewed >7 days. Fire 1's mechanical subset found nothing unambiguous to auto-fix, so the marker was deliberately not bumped. +- *Fires on approval:* "let's do a task audit" (or task-review for the lighter pass). + +** [Fire 1 · P6] File working/inbox-zero-phase-e/ to a permanent home +- *What:* file the three artifacts under =working/inbox-zero-phase-e/= per working-files.md and update inbound links. +- *Why:* the backing work (autonomous-batch-execution spec) is IMPLEMENTED, so the working dir is a filing candidate. Filing is a judgment move — 3 inbound =file:= links in =docs/specs/2026-06-16-autonomous-batch-execution-spec.org= point at it and need updating in the same move. +- *Fires on approval:* rename+move the 3 files flat into their permanent home, update the spec's 3 links, delete the empty working subdir. + +** [Fire 1 · P8] Resolve ~4 candidate broken file: links in live docs (report-only) +- *What:* fix or confirm-intentional the broken links P8 flagged. +- *Why:* report-only pass; no unattended rewrites. The clearest real one: =docs/specs/2026-06-16-autonomous-batch-execution-spec.org= links =../../.ai/workflows/inbox-zero.org= (renamed to inbox.org). Others (=wrapup-routing-spec-review.org=, a pattern-catalog inbox source, =subprojects-log.org=) may be folded-in/deleted review docs — Craig's call. +- *Fires on approval:* update the inbox-zero.org→inbox.org link; decide the rest. + +** 2026-07-20 Mon @ 15:28:41 -0500 — flushed (auto) +Auto-flush checkpoint mid-session. Session's shipped work is all committed + pushed (main == origin == velox). In flight: resuming into finishing the Signal pager task ([#B] in todo.org) — the runbook, velox receive timer, and the ssh-vs-linked-device decision. See Summary → Next Steps and Data Collected for the reconciled facts needed to resume blind. + +** 2026-07-20 Mon @ 16:48 CDT — Notification vocabulary split (page/text) + agent-page → agent-text rename +Craig's follow-on from the reply-ambiguity discussion: reserve trigger words by channel. "page me" = desktop notify, "text me" = Signal, "text and page me" = both. Renamed the tool agent-page → agent-text to match (deprecated agent-page shim delegates to it, removable later). Rewrote protocols.org "Paging Craig" → "Reaching Craig", page-me.org, work-the-backlog's away-run logic, INDEX, and the runbook; updated install-ai doc comments; bats renamed agent-page.bats → agent-text.bats (5 pass incl. shim delegation). /review-code + /voice ran; make test exit 0, shellcheck clean, no new lint. Commit 6145489, pushed; make install re-run on both ratio + velox, agent-text + shim resolve on both. Vocabulary decision arc (this session): first "page=Signal, notify=desktop" (rejected, inverted current default), then "page=desktop, message=Signal" (Craig's), then settled on "text me" for Signal (agent-text). NOTE excluded from staging: .claude/settings.json shows model opus→fable (not mine — left unstaged). + +** 2026-07-20 Mon @ 23:36 CDT — Post-flush resume: voice #46 handoff + triage phone-push (item 1) +Resumed from the auto-flush anchor. Two things landed. + +Inbox handoff (archsetup, voice pattern #46 comma budget): archsetup edited voice/SKILL.md + voice-profile.org through the shared ~/.claude/skills/voice symlink (points into rulesets) at Craig's 2026-07-20 direction — "no more than two commas per sentence" in the /voice personal pass. Reviewed both files (count 45→46 everywhere, personal-only set four→five, #46 in the attestation high-recurrence set, before/after from the Hyprland draft). /review-code approved, /voice personal on the message, make test green. Committed 2ea5d9a; replied to archsetup; marked PROCESSED (chore 9721a49). The .claude/settings.json model change (opus→fable) was left unstaged — not this session's work. + +Item 1 — triage-intake auto-mode phone-push, built. Folded a "Phone delivery" subsection into canonical triage-intake.org: a full-three-section sweep pushes to Craig's phone over Signal via agent-text, with a pointer from End-of-sweep output and a Living Document note. Craig's design ruling (option 1, signal-only): a quiet sweep's "nothing" heartbeat never reaches the phone — silent-until-signal governs the phone channel too — while the in-session heartbeat stays as proof-of-life. Send half only; reply-polling (old phone-recv) deferred to the reply-correlation follow-up (item 2), which owns the recv wiring. sync-check --fix synced the mirror; make test green; /review-code approved; /voice personal. Committed e27aea2. todo.org task closed DONE (send half shipped, recv deferred). + +Next: items 2 (reply-correlation spec) and 3 (token-rotation discussion) remain untouched — carried to the next session. + +** 2026-07-20 Mon @ 16:14 CDT — Signal pager DONE (all four deliverables + topology decision) +Finished the task end to end. (1) Runbook: docs/design/2026-07-20-signal-pager-runbook.org — send, read-replies, receive timer, signal-cli account/setup notes, the resolved decision. (2) Receive timer: scripts/signal-receive.sh + scripts/systemd/signal-receive.{service,timer} (roam-sync-shaped, 15-min, no-ops where the account isn't local, 4 bats); stowed via dotfiles common; enabled + verified on both ratio and velox; a manual drain cleared the 47-day staleness live. (3) Topology decision (Craig, option 2 — linked device over ssh-relay-only): ratio linked as Device 2 "ratio-pager", direct send verified; agent-page generalized from a velox-only check to "any machine holding the account sends directly, else relay" (bats updated). (4) protocols.org "Paging Craig": verified accurate, no edit. Publish flow: /review-code caught a runbook/script contradiction (runbook claimed no-op-clean; script errored) → added the account-presence guard + a bats case. make test green, shellcheck clean. Commits: rulesets 302b062 (feat(pager)), dotfiles d8b0462 (feat(systemd)); both pushed; velox synced. todo.org task closed DONE. Craig's follow-up worry addressed: the pager only fires on an explicit page-me or an away-run's end-of-set page; the receive timer is silent (receive-only, no push). diff --git a/.ai/sessions/2026-07-24-18-05-sentry-implement-pass-review-and-hook-fixes.org b/.ai/sessions/2026-07-24-18-05-sentry-implement-pass-review-and-hook-fixes.org new file mode 100644 index 0000000..6c60a8f --- /dev/null +++ b/.ai/sessions/2026-07-24-18-05-sentry-implement-pass-review-and-hook-fixes.org @@ -0,0 +1,386 @@ +#+TITLE: Session Context — 2026-07-23 +#+AUTHOR: Craig Jennings +#+DATE: 2026-07-23 + +* Summary + +** Active Goal + +A long session (2026-07-23 into 2026-07-24) spanning several arcs: applied two Craig-ordered sentry amendments, ran a no-approvals speedrun over four solo tasks, shipped the sentry implement-pass feature itself, ran eight overnight sentry fires that found and fixed real bugs, then subjected the whole night's output to an adversarial review that found defects in the fixes, repaired those, absorbed a repo-wide fail-open security fix from .emacs.d, and processed the inbox to zero. Ended with the branch merged to main (unpushed by Craig's choice) and the session wrapped. + +** Decisions + +- Sentry gains an opt-in solo-implementation pass (pass 12, gated on =:SENTRY_MAY_IMPLEMENT:=, separate from =:COMMIT_AUTONOMY:=) plus refactor-finding in pass 11. Craig's direction, after the discussion that the branch already contains blast radius and a skeptical premise-first review is the fact-checker that makes fixing-on-a-branch safe. Shipped to main. +- Reviews must fact-check the *premise* (reproduce the bug) before judging the diff, not just check the diff is clean. Craig's correction; saved as harness memory =feedback-reviews-verify-premise=. Across the night the premise check killed roughly one wrong hypothesis per real bug. +- Craig chose NOT to push main at wrap. The hook fail-open fix therefore stays undelivered to consuming projects until he pushes. Flagged and reaffirmed. + +** Data Collected / Findings + +- The dominant defect class across the session, five instances over two projects: a quality gate that enumerates its inputs instead of discovering them, so a new input is silently skipped and the green check reads as covered. Promoted to a KB node this wrap. +- The secret-scan pre-commit hook failed open on any git error, in ALL FIVE language bundles (not just the elisp one .emacs.d reported). Two of the five were hooks I wrote the day before by copying bash — I propagated the defect. Fixed across all eleven sites; graded [#A]. +- The adversarial review round found four real defects in my overnight fixes (mode-widening and symlink-clobbering in cj-remove-block's atomic write, a same-second backup collision in two tools, a test that deleted real backups from shared /tmp) plus one I'd left: the cj-block range check still can't prove it's deleting the block that was scanned. All repaired except the last, which needs a CLI-contract decision and is filed [#B]. +- I repeated my own worst mistake pattern three times: shipping a change whose correctness depended on shared /tmp state (the backup tests), and twice concluding causation from a single-sample measurement (the audit flake A/B, the /tmp-copy comparison). The re-run/isolation habit caught each. + +** Files Modified + +- Merged to main (267d1de): cj-remove-block range guard + atomic write, todo-cleanup backup, route_recommend dedupe, audit.bats flake fix, lint.sh bin/ coverage, plus the review-round repairs. +- Hook fail-open fix (f0c1bc4): all five bundles' pre-commit, the elisp validate-el cap removal, the cross-bundle test now discovering variants, two adopted .emacs.d bats suites. +- Earlier: sentry.org pass 11/12 + marker (pushed), the speedrun's four fixes, the two .dotfiles amendments, voice #47, four approved parked proposals. +- KB: =agents/20260724180443-enumerate-vs-discover-gate-failure.org=. + +** Next Steps + +- Push main (5+ commits ahead, all local). The hook security fix is the load-bearing one. +- The cj-block wrong-block design question: content assertion vs re-scan vs bottom-up removal. Filed [#B] at the top of todo.org. +- Seven parked VERIFYs await Craig, [#A] account-binding guard from home first, then the telegram down-is-launch fix and its engine sibling. +- Optional: ~1600 backup files accumulated in /tmp from the night's runs (harmless, cleared on reboot). + +KB: promoted 1 / consulted no + +* Session Log + +** 02:34 — Startup + +Ran startup.org. Rulesets already current; =make install= had nothing new to link; project repo clean and up to date. =.ai/= synced from templates. No prior =session-context.org= — last session (2026-07-20 23:36, signal-pager + triage phone-push) wrapped cleanly. + +Findings surfaced: 13 top-level tasks unreviewed >7 days; roam inbox holds 9 items; 8 unprocessed project inbox handoffs plus =inbox/lint-followups.org=; KB has 101 =:agent:= nodes but no best-practices node resolved and nothing matching "rulesets". Active Reminder from 2026-07-14 (Craig's deep read of the sentry spec) reads stale — sentry has since shipped and dogfooded live in takuzu and archangel. + +Craig's instruction on arrival: finish startup, then begin sentry. + +** 02:35 — Craig-ordered amendments applied before launching sentry + +Two of the eight inbox handoffs are Craig's own orders relayed from the dotfiles session, and both gate a correct sentry run, so they land before the loop starts. The rest of the inbox stays for sentry's own inbox pass. + +Amendment 1 — sentry pass list (=.dotfiles=, 2026-07-21). Sentry never checks email or messengers, and gains a bug-finding pass. The handoff shipped dotfiles' whole edited =sentry.org=, but that copy forked from a pre-silent-until-signal canonical: rsyncing it would have reverted the heartbeat/digest split committed 2026-07-20. Applied their three intended changes onto current canonical by hand instead, and kept the =:TRIAGE_SOURCES:= activation-probe language their Pass 3 rewrite had dropped. + +Reviewed the staged diff before committing and found three defects in my own edit. Pass 11 as Craig worded it ran "the project's linters and suite," which would have violated sentry's own anti-pattern 5 (no per-pass suite run) eleven times a night; it now runs linters and static analysis only and reads the entry baseline's suite result. The Living Document line still said "ten mechanical passes." And the KB-deferral parenthetical called KB promotion "the proposal's eleventh pass," which collided with the real pass 11 — reworded. Suite green (exit 0) before and after. Committed 33949c5, unpushed pending Craig's call. + +** 02:43 — Sentry armed, fire 1 (working) + +Entry gates all passed: =:COMMIT_AUTONOMY: yes=, tracked tree clean, suite green (exit 0, run twice on this content), no prior =sentry/*= branch, zero behind upstream. Branch =sentry/2026-07-23-ratio= created from HEAD. Loop armed hourly at :07 (job ef8eb9d4, session-only, auto-expires in 7 days). Craig's repo working tree belongs to sentry until the morning merge — reclaiming it early means saying "stop sentry". If he has rulesets files open in Emacs, buffers want reverting after the merge. + +Fire 1 digest: + +- P1 roam pull — ran, already up to date. +- P2 inbox zero — ran. Project inbox: 1 of 5 executed (org-drill's memory-sweep completion note, informational, marked PROCESSED + reply sent). 4 park (below). Roam inbox: 9 items, 8 belong to other projects and route at wrap-up, not here; the 1 rulesets item ("every project should have a working and a temp directory") is shipped except for one clause, queued below. +- P3 triage intake — skipped: no project-specific plugin and no =:TRIAGE_SOURCES:= declaration, so no active source. Correct behavior under the activation gate. +- P4 todo cleanup — ran, no-op. +- P5 task audit — deferred to a later fire; takuzu's dogfood note says the full audit is too heavy hourly, and fire 1 already carries the bug-finding load. +- P6 working-files hygiene — skipped: no =working/= directory. +- P7 spec status board — ran. No =DOING= spec with a closed build parent. +- P8 link integrity — ran, and the checker itself turned out to be broken (see P11). +- P9 git health — ran. =main= is 1 ahead of =origin/main= (the amendments commit, unpushed pending Craig's call). Sentry branch correctly has no upstream. No stale branches. +- P10 prep freshness — skipped: no =daily-prep/=. +- P11 bug finding — ran. Rotating area this fire: =lint-org.el= and the synced =.ai/= templates. Three defects verified and filed as graded tasks (commit dc7791d): link resolution against cwd rather than the linted file's directory [#B], todo-format checkers firing on spec files [#C], and the notes.org template tripping four flags in every project [#C]. Two of the three were independently reported by takuzu and smoke tonight, which is what prompted checking them. + +The three bug tasks record their severity x frequency arithmetic in the body, including where the two inputs disagreed, so the grades can be argued rather than just overridden. + +** 02:47 — Finished the park path the fire skipped + +The inbox-boundary Stop hook caught a real gap. Fire 1 wrote the four proposals into the approval queue and stopped there, but =inbox.org='s park path is three things, not one: move the proposal into =working/<slug>/= beside a prepared diff, file a =[#B]= VERIFY carrying the decision package, and reply to the sender. I'd done the queue entry and none of the rest, so the senders were waiting on nothing and the decisions lived only in a session anchor that gets archived. + +Completed all four. Each =working/= dir now holds the original proposal, a =proposed.diff=, and the full proposed file. Both org-file diffs were verified by linting the proposed version: the notes.org template goes from four flags to zero, and the fix turned out to need one thing smoke didn't identify — the =invalid-block= pair isn't the example content confusing the parser generally, it's the literal =** Feature Name or Topic= line inside the block, which org reads as a heading. A comma-escape on that one line clears both. Also found two more column-0 bold lines beyond the two reported. + +Replies sent to takuzu, archangel, and smoke. dotfiles' ack needed nothing back and is marked PROCESSED. Inbox is at zero. + +** 03:35 — Fire 2 (working) + +Lock acquired, branch state verified. Digest: + +- P1 roam pull — ran, already up to date. +- P2 inbox zero — ran, no-op. Project inbox at zero. Roam inbox unchanged at 9; 8 belong to other projects and route at wrap-up, the 1 rulesets item is already queued. +- P3 triage intake — skipped: no active source surviving the mail/messenger exclusion. +- P4 todo cleanup — ran in check mode across all three passes, zero fixes, zero conversions, zero archives. +- P5 task audit — mechanical subset only, per the parked guidance and takuzu's note that the full audit is too heavy hourly. Staleness is 20, up from 13 at startup, which is just tonight's 7 new tasks arriving never-reviewed. Not a defect. +- P6 working-files hygiene — ran, and this fire is the first where its probe fires, since fire 1's park work created =working/=. All four dirs have open backing VERIFY tasks, so nothing to flag. The pass works. +- P7 spec status board — ran. Seven IMPLEMENTED, two READY (inbox-workflow-consolidation, encourage-kb-contribution). No DOING spec with a closed build parent. +- P8 link integrity — ran across todo.org, notes.org, the anchor, and all nine specs. Two findings, both in the docs-lifecycle spec, both prose containing a bare =file:= that org parses as a bracketless link (=file:→id:= in a sentence about converting link types, and =keep-file:-links-through-pilot= in a hyphenated phrase). Org behaving as documented rather than a defect, so a digest line and no task. +- P9 git health — main still 1 ahead of origin, unpushed, awaiting the approval-queue item. Sentry branch correctly has no upstream. No stale branches. +- P10 prep freshness — skipped: no =daily-prep/=. +- P11 bug finding — rotating area this fire: the =.ai/scripts/= shell helpers. Shellcheck clean on all seven of the sentry-critical ones (agent-lock, agent-roster, capture-guard, flashcard-sync, inbox-status, self-inject.sh, session-context-path). One SC2034 pair in task-review-staleness.sh, verified as a false positive — the two names are positional fields in a =read -r= that exist to put =value= in the right slot. Digest line, not a task. + +*** Retraction: the [#B] I filed in fire 1 was wrong + +The main result of this fire is negative. Fire 1 filed a =[#B]= claiming lint-org resolves =file:= links against the process cwd rather than the linted file's directory. It doesn't, and the task is CANCELLED. + +The claimed evidence was that linting the notes.org template from the repo root reports two siblings missing while linting from its own directory doesn't. The first half was never run. Those findings came from the =/tmp= copy I made while preparing smoke's diff, and in =/tmp= the siblings genuinely are missing, so org-lint was right. I compared two different files and read the difference as a bug, then wrote a fix direction for a mechanism I hadn't checked. + +Caught it here only because P8 surfaced =link-to-local-file= again and I went looking for the checker in lint-org.el to fix it. It isn't there — it's org-lint's own, which contradicted my stated fix and forced the retest. Three runs plus a direct =default-directory= probe settled it. + +Correction sent to takuzu, who had been told about it. Their own finding is unaffected and still filed. The useful residue: both =link-to-local-file= and the todo-format checkers are upstream org-lint checkers, so scoping them means filtering org-lint's output, not editing a local checker — which changes the fix direction on the [#C] task too. + +** 04:35 — Fire 3 (working) + +Lock acquired, branch state clean. Digest: + +- P1 roam pull — ran, already up to date. +- P2 inbox zero — ran, no-op. Both inboxes unchanged. +- P3 triage intake — skipped: no active source surviving the mail/messenger exclusion. +- P4 todo cleanup — ran. Hygiene and convert-subtasks no-op; =--archive-done= moved one subtree, the =CANCELLED= lint-org task fire 2 retracted. Committed 962f3c0. +- P5 task audit — mechanical subset. Staleness 19, down one from the archive. +- P6 working-files hygiene — ran. All four =working/= dirs still have open backing VERIFY tasks. Nothing to flag. +- P7 spec status board — ran. Two READY specs, no DOING with a closed parent. +- P8 link integrity — ran across todo.org, notes.org, and all nine specs. Same two known prose false positives in the docs-lifecycle spec, unchanged. No new findings. +- P9 git health — main 1 ahead of origin, unpushed, still queued. No stale branches. +- P10 prep freshness — skipped: no =daily-prep/=, no broken symlinks. +- P11 bug finding — rotating area this fire: the Python scripts under =.ai/scripts/=. All 15 compile; no Python linter on this box, so the pass became a targeted read of =inbox-send.py=, the script this session has exercised hardest. Three defects verified, filed as two tasks (8995016). + +*** The inbox-send finding + +The one that matters: =inbox-send= writes straight to the destination path, and =write_text= truncates on open, so any mid-write failure leaves a zero-byte =.org= in *another project's* inbox. That phantom isn't inert — =inbox-status= counts it, so it trips the receiving project's boundary hook and blocks a turn there over a file with no content and no sender context. Meanwhile the sender saw an error and retries, so the target collects a second one. + +Reproduced end to end with a non-ASCII message under a C locale with UTF-8 mode disabled: send fails, zero-byte file lands in the destination, =inbox-status= there reports it pending. The encoding case is just the trigger I could force; the defect is the non-atomic write, which a full disk or an interrupted process reaches the same way. Graded [#B], fix direction is temp-file-plus-=os.replace= with =encoding="utf-8"= pinned. + +Two smaller ones grouped as [#C]: =send_file= raises an uncaught =PermissionError= traceback because =main= catches only =ValueError= and =FileNotFoundError=, and =discover_projects= doesn't dedupe, so a roots config naming both a parent and its child lists the same project at two indices. + +Notable that this is the first fire where the rotating area produced findings in code the night's own work depended on. Fire 1 read the linter, fire 2 the shell helpers, fire 3 the script that carried every reply I sent. + +** 05:35 — Fire 4 (working) + +Passes 1 through 10 all quiet: roam already current, both inboxes unchanged, todo cleanup no-op across all three passes, all four =working/= dirs still backed by open VERIFY tasks, two READY specs and nothing stuck, the same two known prose false positives in the docs-lifecycle spec, main still 1 ahead and unpushed, no prep dir and no broken symlinks. Staleness 21, up two from fire 3's new tasks. + +P11 rotating area: =claude-templates/bin/= — the launcher and paging scripts. Shellcheck clean on all four. The findings came from reading and from a config-sanity check, and one of them is a live condition on this machine rather than a latent code defect. + +*** ratio's second Signal account has been cold for 8 days + +=signal-cli listAccounts= on ratio warns that messages were last received 8 days ago. The natural read is that the pager channel has gone stale, which would matter — it's the "text me" path. It hasn't. Established by elimination: ratio holds two accounts, =signal-receive.sh= hardcodes the pager (=+15045173983=) as its only target, and the timer's own journal shows it draining that account cleanly every 15 minutes, the last run 2 minutes before the warning printed. So the cold account is Craig's personal number, and nothing on this machine keeps it warm. + +What I can't establish is whether that matters. The session history describes that registration as note-to-self with no push, and Signal's tolerance for a quiet linked device isn't something I verified. Filed as [#C] with the uncertainty stated rather than graded up on a guess. If it does matter, the fix is a second timer instance rather than a code change, since =signal-receive.sh= already takes an account argument — which makes it a dotfiles handoff, that repo owning the unit. Not sent: the content is speculative and a 05:35 handoff asserting a problem I haven't confirmed would land as noise. + +*** agent-text's direct send is unbounded + +The relay path passes =ConnectTimeout=10= to ssh; the local-account path calls =signal-cli send= with no bound. Since agent-text is invoked by agents, a stall blocks the calling turn with no output and no way to tell a hang from a slow send. Not hypothetical contention — the receive timer holds the same account for ~16 seconds every cadence. Filed in the same [#C]. + +** 06:35 — Fire 5 (working, but the bug hunt came up empty) + +Passes 1 through 10 all no-op: roam current, both inboxes unchanged, todo cleanup clean across all three passes, four =working/= dirs still backed, two READY specs, the same two known prose false positives, main 1 ahead unpushed, no broken symlinks. Staleness 22, up one from fire 4's new task. + +P11 rotating area: =hooks/= and =scripts/= — the machine-wide hooks and the repo's own install and maintenance scripts, the last major uncovered surface. *Zero real bugs.* All five shell hooks shellcheck clean, all Python hooks compile, =hooks/tests= present, =hooks/__pycache__= correctly gitignored with nothing tracked. Six shellcheck findings across =scripts/=, every one dispositioned as a non-bug: + +- =SC2094= (read and write the same file in one pipeline) in =install-ai.sh= and =sweep-gitignore-tooling.sh= — false positive both times. The pattern is a =>>= append with a =[ -s "$gi" ]= stat inside the group. Appending doesn't truncate, and a stat isn't a content read, so the leading-blank-line logic is correct in both the file-exists and file-absent cases. +- =SC2088= (tilde doesn't expand in quotes) in =doctor.sh= and =audit.sh= — both are display strings printed to the user, not paths used for I/O. =audit.sh= says so in a comment on the line above. +- =SC2295= (unquoted expansion inside =${..}=) in =audit.sh= and =diff-lang.sh= — technically correct, no live trigger; =$HOME= carries no glob characters. +- =SC2164= (=cd= without =|| exit=) in =lint.sh= and =status.sh= — =status.sh= already guards its path upstream. =lint.sh= is genuinely unguarded and, since it sets =-u= but not =-e=, a failed =cd= would let it lint the invocation directory instead of the repo root. Confirmed =cd ""= fails rather than silently succeeding, so the hazard is real in shape but needs the running script's own parent to be unreachable. Not reachable in practice; a one-line =|| exit= would close it if anyone touches the file. + +This is the first fire whose hunt found nothing, which is the expected shape — takuzu's dogfood report predicted the rotating hunt goes quiet after the first few fires clear the standing defects. Recording the area covered so the next fire doesn't re-read it. + +*** Two measurement errors this fire, both caught before they became findings + +Worth flagging as a pattern, since fire 1's retraction was the same class. First, a probe loop passed =--convert-subtasks --check= through an unquoted variable; zsh doesn't word-split, so it arrived as one argument and =todo-cleanup= printed a bare =normal-top-level()=. That looked like a tool crash and was my loop. Protocols warns about exactly this. + +Second, chasing that, I read =exit=0= from a bad-flag run and nearly filed a silent-pass defect — a cleanup tool that exits 0 on a bad flag would let =wrap-it-up= believe a pass ran when it didn't. The =0= was =tail='s exit code, not emacs'. Measured without the pipeline, both =todo-cleanup= and =lint-org= exit 255 on an unknown flag and on a missing file, and 0 on success. No defect. + +Three times tonight a conclusion came from a bad measurement. Twice it was caught in the same fire; once (fire 1) it reached a filed task and a sent handoff before the retest caught it. + +** 07:35 — Fire 6 (working) — the night's most consequential finding + +Passes 1 through 10 all no-op again: roam current, inboxes unchanged, todo cleanup clean, working dirs backed, two READY specs, the same two prose false positives, main 1 ahead unpushed, no broken symlinks. Staleness 22, flat. + +P11 rotating area: =languages/= and the bundle install machinery — the last major uncovered surface. Not quiet. + +*** The python and typescript bundles ship no secret-scan hook + +=bash=, =elisp=, and =go= each carry =githooks/pre-commit=, =claude/hooks/validate-*.sh=, =claude/settings.json=, and a seed =CLAUDE.md=. =python= and =typescript= carry none of the four. The pre-commit hook is the credential scanner, so installing either of those bundles gives a project no secret scan on commit — while README's Bundle structure section documents all four as what every bundle follows. + +Verified live rather than inferred, by scanning every project with =.claude/rules/=: =work= (python) and =clock-panel= (python + typescript) both have no =githooks/= and no settings. All four elisp projects have both. =work= is the one that matters — a work repo is where a leaked credential costs most and is likeliest to reach a company remote. + +The history settles intent. Both bundles were added 2026-05-31; =go='s githooks landed 2026-06-02 and =bash='s 2026-06-23. The rollout swept the bundles added *after* these two and skipped these two. And =install-lang.sh= guards its copy with =[ -d "$SRC/githooks" ]=, so the install succeeds and reports nothing missing, which is how this stayed invisible for almost two months. + +Filed [#A], SCHEDULED today — the only [#A] of the night. The fix is find-not-fix per the pass contract, so nothing was changed. The task carries a second half worth as much as the port itself: make =install-lang= warn when a bundle lacks a component the README documents, so the next partial bundle announces itself rather than installing quietly. + +*** Same drift shape, smaller + +=bash='s pre-commit has =cd "$REPO_ROOT" || exit 1=; the =go= and =elisp= copies of that line dropped the guard. Impact is genuinely low — git chdirs to the working-tree root before running a hook, so the checks run against the right tree regardless, and neither script sets =-e=. Filed [#C] for the two-character fix, mostly because the shape is the same as the [#A]: a fix landed in one bundle copy and stopped there. + +Two fires' worth of evidence now says bundle-to-bundle propagation is where this repo leaks changes. + +** 00:51 — Fire 1 of the 2026-07-24 run (working) — first fire with pass 12 live + +Lock acquired, branch =sentry/2026-07-24-ratio= verified, tree clean outside the spine. Digest: + +- P1 roam pull — ran, already up to date. +- P2 inbox zero — ran, no-op. Inbox at zero (the question-capture proposal was parked during entry). +- P3 triage intake — skipped: no active source surviving the mail/messenger exclusion. +- P4 todo cleanup — ran with real work. =--archive-done= moved three completed speedrun tasks out of Open Work into Resolved; =--convert-subtasks= normalized the tree. Committed. +- P5 task audit — mechanical subset. Staleness 13, back to the pre-speedrun baseline now that the four solo tasks closed. +- P6 working-files hygiene — ran. All =working/= dirs still have open backing VERIFY tasks (the parked proposals). No orphans. +- P7 spec status board — ran. Two READY specs, nothing stuck. +- P8 link integrity — ran. The docs-lifecycle spec still shows its two known prose false positives (=file:→id:= and =keep-file:-links-through-pilot=, bare =file:= tokens org parses as bracketless links). Correctly unaffected by tonight's spec-scoping, since =link-to-local-file= is org-lint's own checker, not a todo-format one. No new findings. +- P9 git health — main level with origin, sentry branch correctly has no upstream, no stale branches. +- P10 prep freshness — skipped: no =daily-prep/=. +- P11 bug and refactor finding — ran, first fire under the widened pass. Rotating area: the cross-project routing scripts (=route_recommend.py=, =broadcast.py=), untouched by the previous six areas. One verified latent bug filed [#D] (below). Refactor note not worth a task: =recommend='s two weak-tier branches collapse to a single =_tiebreak= call, since =_tiebreak= on a one-element list returns that element. Two lines, no behavior change, so it's a digest line rather than backlog noise. +- P12 solo-task implementation — *active this run* (=:SENTRY_MAY_IMPLEMENT: yes=) but a correct no-op: zero eligible tasks. All four open =:solo:= TODOs were completed in tonight's speedrun, so the ready bucket is empty. Nothing to implement, nothing deferred. + +*** The finding + +=route_recommend='s =discover_destination_names= collapses projects to bare basenames, so two projects sharing a basename across roots would both literal-match and read as an ambiguous tie, downgrading a correct strong match to weak. Reproduced by direct probe. Latent rather than live: 27 projects, 27 distinct basenames today. The destination stays right, only the tier is wrong, so the cost is an extra routing prompt. Filed [#D] with the order-preserving dedupe as the fix. + +** 01:33 — Fire 2 (working) — pass 12's first real implementations + +Digest: + +- P1 roam pull — ran, already up to date. +- P2 inbox zero — ran. Project inbox at zero. The roam inbox dropped 9 → 5 (another session filed some); all 5 remaining belong to archsetup or work, none rulesets-claimed, so roam mode is correctly a no-op here. +- P3 triage intake — skipped: no active source. +- P4 todo cleanup — ran, all three checks clean (fire 1 did the archiving). +- P5 task audit — mechanical subset. Staleness 13, flat. +- P6 working-files hygiene — ran. All =working/= dirs still backed by open VERIFYs. +- P7 spec status board — ran. Two READY specs, nothing stuck. +- P8 link integrity — ran. Only the docs-lifecycle spec's two known prose false positives. +- P9 git health — main level with origin, sentry branch correctly upstream-less. +- P10 prep freshness — skipped: no =daily-prep/=. +- P11 bug and refactor finding — rotating area: the cj-comment tooling (=cj-scan.py=, =cj-remove-block.py=). Two findings filed, one of them serious. +- P12 solo-task implementation — *two tasks implemented and committed to the branch* (17f5d48, 1b0f284). Both premise-checked before a line was written. + +*** The serious find: cj-remove-block destroys content + +=looks_like_cj_range= validated only the first and last lines of a range. A span from one cj block's opener to a *later* block's closer passed, and the removal then deleted everything between — prose, headings, whole tasks — silently, exit 0. That is exactly the failure the check exists to prevent, and drift is its normal case, since =respond-to-cj-comments= edits the file while processing and a file under cj review usually holds several blocks. Reproduced on a two-block fixture that lost a heading and two content lines. Filed [#B], then implemented in pass 12 after an independent re-verification on a different fixture shape. + +Same file, second defect fixed alongside: =remove_range= rewrote the org file with a bare =write_text= (truncates on open) and took no backup, so a mid-write failure would leave =todo.org= truncated with nothing to recover from. =lint-org.el= already backs these files up to =/tmp= before mutating; cj-remove-block now matches that and writes atomically. + +*** The measurement lesson, third time tonight + +A full-suite run went red on =audit.bats= test 4. I stashed my changes, saw it pass clean, and had a one-sample A/B pointing straight at my own diff. Re-ran three times with the changes restored and it passed every time. The failure is an intermittent teardown flake (=rm -rf= racing something still writing into a fixture =.git/objects=), not my change, and I nearly filed the wrong cause off a single sample. Filed [#C] with the git-background-gc theory explicitly labelled a lead rather than a verified cause. + +Committed only on a genuinely green re-run, not on the red with a hand-wave. + +** 02:33 — Fire 3 (working) — the flake's cause traced, and my own lead disproved + +Digest: + +- P1 roam pull — ran, already up to date. +- P2 inbox zero — ran, no-op. Project inbox at zero; roam holds 5, none rulesets-claimed. +- P3 triage intake — skipped: no active source. +- P4 todo cleanup — ran. =--archive-done= moved fire 2's two completed tasks to Resolved. Committed. +- P5 task audit — mechanical subset. Staleness 13, flat. +- P6 working-files hygiene — ran, all dirs backed. +- P7 spec status board — ran, two READY, nothing stuck. +- P8 link integrity — ran, only the known docs-lifecycle prose false positives. +- P9 git health — main level with origin, branch upstream-less, nothing stale. +- P10 prep freshness — skipped. +- P11 bug and refactor finding — no new findings. The fire's whole investigative budget went to confirming fire 2's filed flake, which is the honest place for it; a hunt that finds nothing new is a result. +- P12 solo-task implementation — one task implemented and committed (7f45d4b). + +*** Confirming the cause before fixing, and disproving my own lead + +Fire 2 filed the =audit.bats= flaky teardown with a stated theory: =gc.auto='s loose-object threshold. Pass 12's premise check went after that theory rather than the fix, and killed it — a fixture holds five objects against a default threshold of 6700, so that mechanism cannot fire. + +The real cause came from a =GIT_TRACE= run: =git commit= spawns =git maintenance run --auto --quiet --detach= on git 2.55. The commit returns while the detached process is still writing a pack, and teardown's =rm -rf= races it. Every failed run had left a =tmp_pack_*= behind, which is the thread that led there. + +Fix: =maintenance.auto false= plus =gc.auto 0= in the fixture, killing the background writer rather than retrying the delete (a retry loop hides a live process instead of removing it). Validated over 20 consecutive clean runs against a ~1-in-8 baseline, and recorded honestly in the task that 20 clean runs alone would be ~7% likely by luck — the trace is the evidence, the runs confirm. + +This is the second night running where the premise check changed the outcome. Fire 2 it stopped a wrong causation call; here it stopped me implementing a fix for a mechanism that was never operating. Both times the cost of checking was minutes and the cost of not checking would have been a plausible, wrong, committed change. + +** 03:33 — Fire 4 (working) — two hypotheses killed, one real gap closed + +Digest: + +- P1 roam pull — ran, up to date. P2 inbox zero — no-op, both surfaces clean. P3 triage — skipped, no active source. +- P4 todo cleanup — ran. Archived fire 3's completed task to Resolved. Committed. +- P5 task audit — mechanical subset. Staleness 13, flat all night. +- P6 working-files — all dirs backed. P7 spec board — two READY, nothing stuck. P8 link integrity — only the known prose false positives. P9 git health — clean. P10 — skipped, no prep dir. +- P11 bug and refactor finding — rotating area: the org-file mutators, chosen because =cj-remove-block= yielded a serious find in that same class last fire. One gap filed. +- P12 solo-task implementation — one task implemented and committed (0686784). + +*** What the area review actually found, and what it disproved + +The lead was that =todo-cleanup.el= might lose data. It rewrites =todo.org=, creates archive files, and moves subtrees *between* files, which is the shape that bit =cj-remove-block=. Two specific hypotheses, both tested, both wrong: + +- *"A mid-move failure loses a subtree from both files."* No. The order is delete-from-buffer, write-archive, save-todo.org-last, so an archive-write failure aborts before the save. Verified by making the archive directory unwritable: exit 255, =todo.org= byte-identical, content intact. +- *"Errors are swallowed on the mutation path."* No. The only =ignore-errors= in the file wrap =call-process "git"=, never a write. + +What survived was narrower and real: todo-cleanup mutates with *no backup*, while both sibling mutators (=lint-org.el=, =wrap-org-table.el=) copy to =/tmp= first, and =cj-remove-block= joined them last fire. It is also the one that runs most often. Confirmed empirically that Emacs's own backup does not fire under =--batch -q=, so there was genuinely no undo short of git. Filed [#C], then implemented in P12. + +Also checked my own change for regression rather than assuming: a missing input file exits 255 and creates nothing, identical to the pre-change version tested from git. + +*** Running tally on the premise habit + +Three fires, three times it changed the outcome. Fire 2 it stopped a wrong causation call. Fire 3 it disproved my own filed =gc.auto= theory before I could implement against it. Here it killed two data-loss hypotheses before they became tasks, leaving only the gap that was actually there. The pattern is consistent: the cheap check keeps a plausible story from becoming a committed change. + +** 04:33 — Fire 5 (working) — the first real defer, and last fire's fix proving itself + +Digest: + +- P1 roam pull — ran, up to date. P2 inbox zero — no-op, both surfaces clean. P3 triage — skipped, no active source. +- P4 todo cleanup — ran, archived fire 4's completed task. *Confirmed last fire's backup fix working live*: the real =--archive-done= run left =/tmp/todo.org.before-todo-cleanup.20260724-043331=. Dogfooded within an hour of shipping. +- P5 task audit — staleness 13, flat all night. P6 working-files — all backed. P7 spec board — two READY. P8 link integrity — only the known prose false positives. P9 git health — clean. P10 — skipped. +- P11 bug and refactor finding — rotating area: the attachment/email handlers, chosen because they parse genuinely untrusted input, unlike every internal-tooling area covered so far. One finding filed. +- P12 solo-task implementation — *deferred*, and correctly. First defer of the run. + +*** The find: attachment filenames are partly sanitized, in two different ways + +Both writers derive on-disk names from the =filename= an email declares, and both sanitize incompletely, covering *different* gaps. =eml-view= cleans the name but interpolates the extension raw. =gmail-fetch='s =safe_filename= handles path separators and leading =..= and nothing else. Probed both with the same adversarial set: a =; rm -rf ~= extension survives in both, a literal newline survives in both, a 300-character extension produces a 314-character filename in both. + +Two things I checked so this doesn't get over-graded later. Not RCE — files are written through Python =open=, never a shell. Not path traversal — =splitext= only returns an extension when the last dot follows the last separator, so =ext= can never hold a slash, and the traversal case is neutralized in both scripts. The genuine harms are narrower: a newline in a filename breaks downstream tooling that reads the directory as a line-delimited list, and an unbounded extension blows the 255-byte limit so a crafted attachment aborts extraction. Graded [#C] on that honest read rather than the scarier one. + +I also corrected my own framing mid-investigation. I first read this as the familiar "one sibling hardened, the other not" pattern from the last two fires. It isn't — =safe_filename= is narrower than it looks, and the two scripts are *differently* incomplete. Neither handles newlines or length. + +*** Why pass 12 deferred instead of implementing + +The fix itself is clear, but where the shared sanitizer lives is a design call: a shared helper module (clean, but a new synced template file plus =importlib= gymnastics for kebab-named scripts), duplicate it in both (self-contained, but drift — the exact defect class that produced three separate findings tonight), or patch each in place (smallest diff, permanent divergence). That is deliberation, not a quick factual question, so checklist item 4 fires and the unattended loop defers. Filed the VERIFY with the three options and my lean, and left the task *un-=:solo:=-tagged* so a later run doesn't pick it up and guess. + +This is the checklist discriminating rather than rubber-stamping. Four fires implemented; this one correctly didn't. + +sentry at 05:33: nothing (bug hunt swept =scripts/*.py=; two candidate gaps both disproved — =workflow-integrity.py= *is* gated, its bats runs the real checker against the real canonical tree under =make test=, and =update-skills.py= is an on-demand maintenance command rather than a gate. Recorded so neither gets re-investigated.) + +sentry at 06:33: nothing (bug hunt swept =wrap-org-table.el=, the third org-file mutator and the one with prior history — its load-time dispatch caused the 2026-07-09 corruption. Clean on every probe: the entry-script guard correctly refuses to dispatch when lint-org merely =require='s it, it backs up before writing like its siblings, and it is block-aware — a table inside =#+begin_example= stayed verbatim while a real over-budget table wrapped onto continuation rows with rules. Second consecutive quiet hunt.) + +** 07:33 — Fire 8 (working) — a time bomb I planted four fires ago went off + +Digest: P1-P10 all no-op (roam current, both inboxes clean, todo cleanup clean, staleness 13, working dirs backed, two READY specs, only the known prose link false positives, git clean). P11 found a real coverage gap. P12 implemented it, and the suite caught a regression of my own making. + +*** The find: the only ungated shell in the repo + +=scripts/lint.sh= sweeps =scripts/*.sh=, the language hooks, and the language githooks. It never touched =claude-templates/bin/= — zero references. Those four scripts (=ai=, =agent-text=, =agent-page=, =install-ai=) are the ones =make install= symlinks onto PATH, which makes them the *most* exposed shell in the repo and left them the only shell with no gate over it. All four are clean today, so the guard is a no-op by design; it exists so a future regression can't pass silently. The =ai= launcher was hardened to 42 tests recently and nothing enforced that going forward. + +The test pins *coverage*, not cleanliness: it plants a broken file in each swept location and asserts lint complains, so a location that stops being swept fails the suite rather than passing quietly. + +Surfaced a bigger question I did *not* answer overnight: rulesets ships shellcheck enforcement to consuming projects (the bash bundle's pre-commit, =validate-bash.sh=) and runs none on itself. Filed as a VERIFY, because turning it on would surface the false positives dispositioned earlier this session and choosing between fixing them or adding disable-directives is a preference, not a fact. + +*** The regression: my own test, detonating on schedule + +The full suite went red on the two backup tests I added in fire 4. Not a flake — a time bomb. They globbed =/tmp/todo.org.before-todo-cleanup.*=, but the backup name derives from the file's *basename*, and the real =todo.org= shares it. So a live sentry run's genuine backup was indistinguishable from the test's own artifact, and the check-mode test (which asserts *no* backup exists) failed the moment fire 5's real archive pass created one. + +They passed when written only because no real backup existed yet. Three now sit in =/tmp=. Both tests rebind =temporary-file-directory= to a private dir, and I verified they pass with the real backups present rather than by clearing them. + +Worth naming plainly: I shipped a test whose correctness depended on the state of a shared directory that the code under test writes to in production. That is the "no shared mutable state" rule in =testing.md=, and I broke it while fixing a different durability bug. The suite caught it two fires later, which is the argument for running the full suite every fire rather than only the touched file. + +* Sentry approval queue (2026-07-23) + +Five items. Each names what, why, and the exact edit. Items 2 through 5 are also filed as =VERIFY [#B]= tasks in =todo.org= with prepared diffs under =working/=, so they survive this anchor being archived — say "approve the parked <topic>" for any of them. + +** 1. Push main to origin + +What: =git switch main && git push origin main= (then switch back, or leave main checked out if sentry is done). + +Why: commit 33949c5 (the two .dotfiles amendments) is on main and unpushed. The sentry spec called this out — an unpushed commit on main diverges across ratio and velox and breaks the next startup fast-forward. It's ahead-only, so the push is clean. Held because =commits.md= requires explicit confirmation before any push. + +** 2. interaction.md — remove the fenced-code-block carve-out (org-drill) + +What: in =claude-rules/interaction.md=, the "No Reverse-Video Highlighting in Chat Output" rule currently says fenced code blocks "are acceptable when the user explicitly wants a block to copy". Replace that sentence with a plain-text-always statement covering fences as well. + +Why: org-drill relayed Craig's 2026-05-30 direction ("always always list it out without markup"), saved there as a project memory. The rule as written contradicts it, and the directive belongs in the shared rule rather than one project's memory. It's a convention change to an always-on rule, so it parks rather than lands. + +Note: this session violated the tightened form of the rule several times already (fenced blocks in chat), which is evidence for the proposal rather than against it. + +** 3. sentry.org Living Document — fold in the two dogfood runs (takuzu, archangel) + +What: append to sentry.org's pass list and notes: (a) make the rotating-angle bug hunt an official pass — already done tonight as pass 11, so this reduces to noting takuzu's corroboration; (b) add randomized property sweeps as a sanctioned quiet-fire activity; (c) note that =todo-cleanup --archive-done= touches =.gitignore= on its first archive, so an "org-only" pass can produce a real commit and trigger the fire-end suite; (d) note that in a project gitignoring =.ai/=, quiet fires produce zero commits, so =git log main..sentry/*= understates the night and the anchor's heartbeat list is the only record; (e) replace the per-fire full task audit with a mechanical subset hourly plus judgment items queued once nightly. + +Why: two independent first-live-run reports, no engine defects in either. Item (e) matches what fire 1 did by instinct (P5 deferred). sentry.org is a shared synced asset, so the edit parks. + +** 4. notes.org template — clear the four lint flags (smoke) + +What: in =claude-templates/.ai/notes.org=, rephrase the two column-0 =**bold**= lines so neither starts with =**=, and comma-escape the =#+begin_example= block's own markers. Then =scripts/sync-check.sh --fix=. + +Why: filed as a [#C] bug task tonight with the reproduction. The edit itself parks because it changes a template every project inherits. Related finding worth Craig's attention: those two flags are mechanical, not judgment, so =lint-org --fix= run anywhere would rewrite the template and create drift against canonical. + +** 5. wrap-it-up.org — delete temp/ at wrap (from Craig's own roam capture) + +What: add a cleanup sub-step to =.ai/workflows/wrap-it-up.org= that removes the project's =temp/= contents during teardown. + +Why: Craig's roam inbox item asks for exactly this ("temp is ... deleted as a part of the wrap up sequence"). The rest of that capture shipped on 2026-07-20 (=working/= tracked from creation, =temp/= gitignored, the sweep backfilling it), but =wrap-it-up.org= has no mention of =temp/= at all, so this clause was missed. Parks as both a shared-asset edit and a destructive one. + +** 02:35 — Craig-ordered amendments applied before launching sentry (earlier) + +Amendment 2 — KB personal roots (=.dotfiles=, 2026-07-21). =~/.dotfiles= classified Unknown under =knowledge-base.md='s personal-roots list, blocking KB writes from that project twice (the sentry pass-list rule tonight, the xdg-desktop-portal gotcha on 2026-07-04). Grepped for other copies of the enumeration: =knowledge-base.md:25= is the only place that lists roots for *classification*. The other three hits (=triggers.md=, =broadcast.org=, =session-harvest.org=, =work-the-backlog.org=) enumerate roots for project *discovery*, a different question, and =~/.dotfiles= reaches those through the machine-local =~/.claude/inbox-roots.txt=. Edited the one line. diff --git a/.ai/sessions/2026-07-25-09-24-invalid-block-filter-push-task-review.org b/.ai/sessions/2026-07-25-09-24-invalid-block-filter-push-task-review.org new file mode 100644 index 0000000..d0814f9 --- /dev/null +++ b/.ai/sessions/2026-07-25-09-24-invalid-block-filter-push-task-review.org @@ -0,0 +1,83 @@ +#+TITLE: Session Context — 2026-07-24 +#+AUTHOR: Craig Jennings + +* Summary + +** Active Goal + +An evening of backlog clearing in the order Craig picked: push the delivery-blocking commits, fix the lint-org =invalid-block= false positive that home had just unblocked, then run a task-review cycle. All three finished. + +** Decisions + +- The =invalid-block= fix is a filter on org-lint's output, not a local checker edit. Home settled the open question and I re-verified both halves before acting: the string appears nowhere in =lint-org.el=, and =org-lint--checkers= enumerates it in batch Emacs. Same resolution as the earlier =link-to-local-file= episode. +- Left the =,**= comma-escape in =claude-templates/.ai/notes.org= in place. The task said the fix "lets that escape be reverted," but the escape is correct org for a literal =**= inside a verbatim block, so reverting trades correctness for nothing once the finding is suppressed. Reasoning recorded in the closed task rather than acting on the permission silently. +- Task review: all seven in the batch kept as-is, no new =:quick:= or =:solo:= tags, confirmed by Craig in one pass. +- Work's sentry loop was left alone. Craig said "sentry stop" here, but the running loop belongs to the work project (branch =sentry/2026-07-25-ratio=, firing hourly into =aiv-work:1.1=), and the stop procedure's later steps — lock release, branch disposition, approval queue — need that session's context. Surfaced rather than half-executed. + +** Data Collected / Findings + +- =invalid-block= is org-lint's own checker. Reproduced the false positive three ways: a paired example block with a heading-shaped body line flags both delimiters, a paired src block holding a literal =#+end_example= flags three lines, and a genuinely unterminated block flags once and must keep doing so. +- Verified the fix against home's live fixture in both directions: its =.ai/notes.org= produced exactly the two reported findings (lines 386 and 398) under the pre-change script and zero under the new one, file untouched. +- The uppercase-delimiter path (=#+BEGIN_EXAMPLE=) was handled but untested. Confirmed it was a real trigger — 2 findings before, 0 after — before adding the boundary test. +- Part of the task-review staleness count measures work that can't be delegated rather than work nobody read. This batch was the deliberation-heavy tail, where every task needs a decision from Craig mid-stream, which is why the speedruns kept stepping past them. +- The KB orphan task cites a 2026-07-01 snapshot of 53 agent nodes. Startup counted 104, so the KB has doubled and the snapshot is worth even less than the task body assumed. +- Five =[#D]= tasks carry no =:LAST_REVIEWED:= at all. The staleness script excludes =[#D]= but =lint-org= flags them, so the two tools disagree permanently. Left alone; needs a decision about which is right. + +** Files Modified + +- =claude-templates/.ai/scripts/lint-org.el= (+ mirror) — =lo--matched-block-regions= pairs blocks by line scan under org's real rule, memoized on the buffer modification tick; =lo--handle-item= drops an =invalid-block= finding inside a paired region. +- =claude-templates/.ai/scripts/tests/test-lint-org.el= (+ mirror) — five tests: heading-in-example, literal =#+end_example= in src, uppercase delimiters, unterminated block still reports, and one file with both proving per-block scoping. +- =todo.org= — closed the =invalid-block= task with its verification record, stamped seven review dates, inserted a missing properties drawer. +- KB: =agents/20260725093500-parser-cannot-verify-its-own-misreading.org=. + +** Next Steps + +- Work's sentry is still running on ratio and untouched. Stopping it properly means saying "sentry stop" in the work session (pane =aiv-work:1.1=), which handles the branch disposition and the overnight approval queue. +- Home is waiting on this push to re-run its =invalid-block= fixture. +- Next review batch starts with the agent-source improvements and flashcard tooling tasks. +- Seven parked VERIFYs still await Craig, the =[#A]= account-binding guard from home first. +- The sentry spec still wants Craig's deep read before the READY flip. + +KB: promoted 1 / consulted no + +* Session Log + +** Startup + +Startup ran clean: rulesets already current, =make install= had nothing new to link, project repo up to date with origin (5 commits ahead, unpushed — carried over from the prior session by Craig's choice). =.ai/= synced from templates with no churn. No crash anchor. Task staleness reported 13 top-level tasks unreviewed for >7 days. + +** Inbox — home's answer on invalid-block + +One pending handoff: home answering the open question left in the =[#C] lint-org invalid-block false-positives= task — is =invalid-block= lint-org.el's own checker or org-lint's? Home says org-lint's, and I re-verified both halves rather than taking it: =grep invalid-block= over =claude-templates/.ai/scripts/lint-org.el= returns nothing, and =org-lint--checkers= enumerates =invalid-block= in batch Emacs alongside =link-to-local-file=. So the fix is a filter on org-lint's output, matching the =link-to-local-file= episode. + +Folded into the existing task as a dated sub-entry rather than filing anything new — the task was already filed and only needed its open question closed. Also recorded home's regression fixture: its own =.ai/notes.org= PENDING DECISIONS block (lines 386-398) is left unescaped on purpose and trips both delimiters, so the filter should take those two findings to zero without touching the file. Replied to home confirming, and asked them to keep the fixture unescaped pending a ping when the filter lands. Inbox back to zero. + +** Pushed main + +Craig picked the evening's order: push, then the invalid-block fix, then a task review. Pre-push reconcile showed ahead-only by 5, no divergence, so =git push origin main= went out and verified at 0/0. The secret-scan fail-open fix is now delivered to consuming projects. + +** invalid-block filter + +Built test-first. Reproduced the failure three ways before writing anything: a paired example block with a heading-shaped body line flags both delimiters; a paired src block holding a literal =#+end_example= flags three lines; a genuinely unterminated block flags once and must keep doing so. Wrote four ERT tests covering those plus a mixed file, watched three fail for the right reason, then implemented. + +The fix is =lo--matched-block-regions=: a line scan that pairs blocks under org's real rule (once open, only the block's own =#+end_TYPE= closes it), memoized on the buffer's modification tick. =lo--handle-item= then drops an =invalid-block= finding falling inside a matched region, delimiters included, since org-lint reports at the delimiters. Line-scanning instead of asking org is the whole point — org's parser is the thing that's confused. + +One test failed after the implementation on an off-by-one in my own expectation (the unterminated opener is line 7, not 8); the code was right and I corrected the test. Verified against home's live fixture both ways: two findings under the pre-change mirror copy, zero under the new canonical, home's file untouched. Synced canonical → mirror, full =make test= green at exit 0. + +Review caught two things I fixed rather than filed: the uppercase-delimiter path was handled but untested (confirmed it was a real trigger — 2 findings before, 0 after — then added the boundary test), and the cache-tick comparison used =eq= where =eql= is strictly correct. Verdict Approve, committed 8822b0d after the voice pass. Pinged home that the filter landed and the fixture can be re-run. + +Left the =,**= comma-escape in the notes template alone. The task said the fix "lets that escape be reverted," but the escape is correct org for a literal =**= in a verbatim block, so reverting buys nothing now that the finding is suppressed. Recorded the reasoning in the closed task rather than acting on the permission silently. + +** Task review + +Batch of 7 from the staleness script, oldest first. Every one came back Keep with no new =:quick:= or =:solo:= tag, and Craig confirmed the batch in one pass. Backed todo.org up to /tmp first (matching the mutator convention) and stamped by exact line number rather than a global replace, since ten other tasks carried the same 2026-07-13 date and only six were in the batch. The Sentry vNext task had no properties drawer at all, so it got one. + +Two observations worth keeping. The uniform Keep-with-no-tags result isn't the review going soft: this batch is the deliberation-heavy tail, where every task needs a decision from Craig somewhere in the middle, which is exactly why the speedruns kept stepping past them and why they aged. So part of the staleness count is measuring work that can't be handed off rather than work nobody read. And the KB orphan task cites a 2026-07-01 snapshot of 53 agent nodes; tonight's startup counted 104, so the KB has doubled and the snapshot is worth even less than the task body already assumed. + +** Inbox — home's acknowledgment + +Home replied confirming both messages landed, the fixture stays unescaped, and they'll pick the filter up after the rulesets push and their next clean startup sync. A pure FYI asking nothing, so it skipped the skeptical review and got no reply (acking an ack loops). Deleted; inbox back to zero. It does corroborate that the unpushed commit is the only thing standing between home and the fix. + +** Task review (cont.) + +Staleness went 13 → 6. Lint flags five more tasks missing =:LAST_REVIEWED:= entirely (lines 327, 336, 434, 593, 602) — all =[#D]=, which the staleness script excludes but the checker doesn't. Pre-existing, not touched. diff --git a/.ai/sessions/2026-07-25-15-34-clean-wrap-and-inbox-safe-sync.org b/.ai/sessions/2026-07-25-15-34-clean-wrap-and-inbox-safe-sync.org new file mode 100644 index 0000000..737b1c5 --- /dev/null +++ b/.ai/sessions/2026-07-25-15-34-clean-wrap-and-inbox-safe-sync.org @@ -0,0 +1,85 @@ +#+TITLE: Session Context — 2026-07-25 +#+AUTHOR: Craig Jennings + +* Summary + +** Active Goal + +Guarantee that rulesets can only report a successful wrap with a completely clean Git worktree, while allowing other projects to refresh rulesets when its only residue is untracked inbox deliveries. Implemented, reviewed, fully tested, and prepared for a strict self-hosted wrap. + +** Decisions + +- One executable, =git-worktree-gate=, owns both repository-state policies. =strict= means no staged, unstaged, untracked, dirty-submodule, or in-progress-operation state; =sync-safe= permits only untracked paths beneath =inbox/=. +- Cleanup failure is not a degraded wrap. It leaves the session open and must report each path, its Git state, why it cannot be resolved safely, and the decision Craig needs. Dirty-file deferrals, valediction, and teardown are forbidden in that state. +- Final wrap verification is a HEAD-bound certificate stored in the Git directory and freshly rechecked inside the existing teardown hook. Integrating the check avoids the concurrent-hook race documented by Codex. +- Another project's structured Edit/Write call may not resolve through an installed symlink into rulesets. The runtime hook denies it and points the sender to =inbox-send rulesets=. +- The two MCP-registry handoffs were consolidated into one parked =[#B]= specification decision. The memory auditor remains separate; no machine-owned MCP configuration was promoted into canonical rulesets. + +** Data Collected / Findings + +- The prior rulesets archive completed at 09:24; the inherited three tracked changes were written at 10:27. The repository was dirtied after wrap through a later write path, which is why wrap certification alone needed the symlink-aware boundary guard. +- The previous wrap prose contradicted itself: clean Git state was an exit criterion, but the leftover and inbox sections allowed explicit deferral. The teardown hook checked only the sentinel. +- Adversarial review found a fail-open process-substitution edge: a low-level =git status= failure could appear as an empty stream. The gate now captures status and its exit code in Git-directory temporary files and blocks on failure. +- Current Codex hooks support Stop blocking with =continue=false= and run matching commands concurrently. A user-level =hooks.json= is installed; a new Codex session must complete the normal hook review/trust step. +- Full =make test= passed twice on the implementation. The final run includes 436 core Python tests, 72 hook tests, language suites, ERT suites, and all Bats suites. Focused additions cover inbox-only pulls, staged/unstaged/untracked/submodule/operation states, status failure, certificate/HEAD drift, Claude and Codex Stop outputs, installation, and realpath-based write denial. +- The wrap roam sweep found Craig's 120-column table question. It was already enforced by =org-tables.md=, =lint-org='s =org-table-standard= judgment, and =wrap-org-table.el=; the live lint pass flagged the existing over-wide table. Removed the duplicate capture and synced roam. + +** Files Modified + +- =claude-templates/bin/git-worktree-gate= — shared strict/sync-safe classifier plus certificate/verify modes. +- Startup protocol/workflow mirrors and =claude-templates/bin/ai= — inbox-only state remains visible but no longer blocks fast-forward refresh. +- Wrap protocol/workflow mirrors and =hooks/ai-wrap-teardown.sh= — no deferral escape, actionable hard blocker, final certificate, fresh teardown verification. +- =hooks/rulesets-write-boundary.py=, Claude/Codex hook configuration, cross-project rule, Makefile, and hook documentation — prevent structured writes through installed symlinks and install the enforcement on both runtimes. +- Bats and pytest suites — repository-state, launcher, teardown, installer, and cross-project boundary regressions. +- =todo.org= and workflow state — parked the consolidated MCP registry spec decision and recorded inbox processing. +- Three inherited backlog files — retain the post-09:24 definition that speedrunnable means =:solo:=. + +** Next Steps + +- Start a new Codex session and review/trust the new user-level hooks when =/hooks= prompts; Claude already reads the linked hook configuration. +- Say "spec the MCP registry sync" when ready to design the separate host-level registry reconciler. +- The existing =inbox/lint-followups.org= pipeline retains its 14 current judgment items, including the over-wide table and older missing review stamps/links; they do not represent uncommitted work after this wrap. + +KB: promoted 0 / consulted no + +* Session Log + +** Startup and clean-worktree investigation + +Ran the required startup workflow. The canonical rulesets pull and template sync were blocked by three tracked modifications; two untracked inbox handoffs were also pending. Read the project and global behavioral rules, recent session archive, wrap workflow, relevant hook and launcher code, and the current Codex MCP and hook documentation needed to evaluate the inbox proposals. + +Investigated Craig's requirement that rulesets finish with an absolutely clean worktree while still allowing downstream projects to sync when rulesets has received inbox deliveries. The current wrap workflow states a clean exit criterion but later permits explicitly deferred dirty files, and its teardown hook checks only the wrap sentinel rather than Git state. The startup shell's tracked-change check already ignores untracked inbox files, but the general launcher dirty check does not distinguish inbox deliveries from other untracked residue. + +The three tracked files now dirty were written at 10:27, after the latest archived rulesets session wrapped at 09:24. That establishes a post-wrap write path: a strict wrap gate can guarantee the state at completion, but preventing later contamination also needs a realpath-aware cross-project write guard because globally installed rules and workflows are symlinks into this repository. + +The proposed design is one shared repository-state classifier with two policies. Strict wrap requires no staged, unstaged, or untracked entries and no dirty submodules. Inbox-safe sync permits an otherwise clean tracked/index state with untracked entries only below =inbox/=; those deliveries do not block pull or template sync. The strict check should run after the final push and again inside the ordered teardown hook, tied to the checked HEAD, while the inbox-safe policy should drive both startup and the =ai= launcher. Tests should cover every Git state, unusual path names, inbox-only sync, hook behavior, and the canonical/template mirrors. + +Two pending inbox handoffs both propose a shared Claude-to-Codex MCP registry mirror. They pass the value gate but overlap. The recommendation is to consolidate them into one =[#B]= specification with Codex-only entries preserved, atomic and redacted updates, dependency and health checks, both-machine verification, and the Claude-memory audit split into a separate task. No inbox disposition or project implementation has been applied. + +Completed the investigation plan without changing product code. The worktree proposal is ready for Craig's approval; implementation, tests, and inbox filing remain deliberately pending. + +** Clean-wrap invariant clarified + +Craig confirmed that cleanup failure must prevent wrap-up entirely. The agent must keep the session open and report exactly what remains in the Git worktree, why it could not resolve each item safely, and the specific action or decision Craig needs to supply. A warning, deferred-file exception, valediction, archived-as-complete status, or teardown is not an acceptable substitute for a clean tree. + +** Clean-wrap enforcement implemented + +Craig approved implementation and asked for a full wrap when it is done. Added =git-worktree-gate= as the single policy executable: strict mode rejects every staged, unstaged, untracked, dirty-submodule, and in-progress-operation state; sync-safe mode permits only untracked =inbox/= deliveries. Certificate and verify modes bind the final clean check to HEAD inside the Git directory. + +Wired sync-safe behavior into both startup workflow copies and the =ai= launcher. The picker labels inbox-only state distinctly and now fast-forwards a behind repository with inbox deliveries present while refusing other untracked residue. + +Made wrap cleanup fail closed: removed every dirty-file and inbox deferral escape, added the post-push clean certificate as a hard prerequisite to valediction, and required an exact path/state/needed-decision report when cleanup cannot finish. The existing teardown hook now freshly verifies the certificate and HEAD before consuming either sentinel, emits the runtime-appropriate Claude or Codex Stop blocker, and leaves the sentinel/session intact on failure. Added global Codex hook configuration and installed its symlink; Codex will require its normal hook review/trust on a new session. + +Added =rulesets-write-boundary.py= and configured Claude and Codex Edit/Write hooks. It resolves targets through symlinks and denies another project's write when the real path lands in rulesets, directing the proposal through =inbox-send rulesets=. The cross-project rule now states the installed-symlink case explicitly. + +Focused verification is green: 10 state-gate Bats cases, 13 teardown-hook cases including dirty/changed-HEAD/missing-certificate and both runtime outputs, 36 launcher cases including inbox-only pull, 5 installer cases, and 5 Python write-boundary cases. + +** Inbox — MCP registry proposals consolidated + +Processed the two pending 2026-07-25 handoffs from work and home. They were duplicate evidence for a host-level Claude-to-Codex MCP registry reconciler, not project-level implementation requests. Filed one =[#B]= parked specification decision in =todo.org= with preservation, redaction, atomicity, transport, health-check, two-machine, malformed-input, and token-rotation gates; split the memory auditor into separate future work. Deleted both inbound handoffs, stamped =:LAST_INBOX_PROCESS:=, sent acknowledgements to both source projects, and verified zero pending project handoffs. + +** Review, full verification, and wrap cleanup + +The first full =make test= run passed. Adversarial self-review then found the Git-status process-substitution fail-open and a path-resolution fail-open in the cross-project hook; fixed both, added status-error and sequencer regressions, aligned =protocols.org= with the new policies, and reran the complete suite successfully. + +Executed wrap cleanup: no sentry lock, todo hygiene/convert/archive/priority passes made no changes, lint-org applied zero mechanical changes and refreshed the 14-item judgment pipeline, project inbox remained at zero, route-batch had no candidates, and 30-day staleness was zero. The shared roam inbox held one rulesets capture about 120-column tables; verified the enforcement already exists in three layers, removed the duplicate, and pushed the roam update with the repository's sync helper. diff --git a/.ai/sessions/2026-07-27-17-02-context-engineering-rightsizing.org b/.ai/sessions/2026-07-27-17-02-context-engineering-rightsizing.org new file mode 100644 index 0000000..54020e4 --- /dev/null +++ b/.ai/sessions/2026-07-27-17-02-context-engineering-rightsizing.org @@ -0,0 +1,176 @@ +* Summary + +** Active Goal + +Review three Anthropic posts on context engineering against what this repo ships downstream, then act on the findings. Ended with the always-loaded rules surface cut from ~57,800 tokens to ~28,949 (plus 13,461 path-scoped), two mechanisms proven in a live session, and two of my own bugs found and fixed — one of which had killed Craig's work session. + +** Decisions + +- *Split rules by blast radius, not by size.* What must hold whether or not you're publishing, and where a violation is permanent and reaches other people, stays always-loaded. Everything recoverable can ride a trigger. That's what let =commits.md= and =testing.md= ship without waiting on any pilot. +- *Goal is output quality first, tokens second* (Craig's correction). Anthropic's 80% was a finding, not a target. P4/Phase 8 (effort reduction) dropped outright for trading quality for cost; P5 (positive framing over prohibition) promoted as the lever that actually targets guardrails working against output. +- *Don't apply the posts additively.* The harness system prompt already carries most of what the Opus 5 guide recommends adding, near-verbatim. Adding it to =claude-rules/= would worsen the duplicate-and-conflict problem the first post opens with. The posts' value here is subtractive. +- *Everything authored in or about the repo is first person* (Craig's instruction), with one carve-out: a comment describing what the code does stays third person, since there the code is the actor. +- *No unmerging home.* Its domains are already separated by tag (24 finances, 18 kit, 17 jrestate). The real problem is a tag namespace flattening four orthogonal axes, and cross-domain priority is a judgment no scheme can make — splitting projects hides the question rather than answering it. + +** Data Collected / Findings + +- *Path-scoping works at user level*, confirmed by =/context= in a live work session: 17 generic rules listed, the three path-scoped ones absent. Deterministic glob match, so no trial needed for that tier. +- *User-level and project-level rules both load, project wins.* work and =.emacs.d= carried 19 byte-identical duplicates, and a stale project copy silently overrode the fresh global rule. +- *My token estimates were 45% low.* Real ratio 2.28 tok/word. =commits.md= was 12,800 tokens, not the ~7,000 I claimed. =/context= reported the true per-file numbers the whole time and I used a word-count estimate because it was easier to compute from inside the repo. +- *Two loading paths, not one.* Memory files arrive via the harness; =protocols.org= and the workflows are read by startup and land in Messages. They shrink by editing the workflow, not by scoping a rule. +- *41 execution/hygiene workflows against 6 discovery/design.* The system is heavily built on the half of the problem that got easier. +- *The instructions don't practice what they demand:* =commits.md= argued terseness at 5,561 words, =interaction.md= bans bold while the rules carry 591 bold markers, =testing.md= argues TDD across eight more rows of rationalizations. +- *Two of my own mechanical guards failed the same day, both certifying success while doing damage.* =wrap-org-table.el= reflowed a table into a worse shape and =lint-org= then passed it; the wrap-teardown hook consumed a two-hour-old sentinel and killed Craig's live work session. + +** Files Modified + +Seven commits, all pushed, velox synced throughout. =6c1ea8b= peer-reasoning rule + Chrome convention + KB probe fix. =0adcb1a= =paths:= frontmatter on the three file-type rules + the lint checker that catches prose/frontmatter mismatch. =7ea1d7b= generic rules no longer ship per project, sweep + gitignored session anchor. =79ed3b0= the three rightsizing docs. =2c664cb= =hooks/session-start-disarm.sh= for the sentinel bug. =d74d98d= docs corrected against live measurements. =931f364= =commits.md= → invariant core + =publish= skill. =2f45b6e= =testing.md= → directive core + =testing-standards= skill, approval-gate signal fixed, first-person directive. + +** Next Steps + +Everything remaining needs Craig's decisions rather than execution — see the =[#B] Finish context-engineering rightsizing= task and the =[2026-07-27]= reminder. In order: reconcile the three working docs (one commit behind), then C1 (=verification.md='s honesty core vs the over-verification warning), =interaction.md=, the TDD rationalization table, and D3 (which approval gates are preference vs guardrail). + +Also open: the work sentry triage split and the recurring-loop proposal, both filed =[#B]= with their reviews. The sentry spec review is still waiting, now two weeks old. + +KB: promoted 2 / consulted no + +* Session Log + +** Startup — 2026-07-27 10:25 CDT + +Ran startup. Rulesets already current; =make install= had nothing new to link; project repo clean at f2609d9 with no upstream drift. =.ai/= synced from templates (no churn — the sync is a no-op mirror refresh in this repo). Previous session wrapped cleanly (no session-context anchor present). + +Startup signals: 6 top-level tasks unreviewed for >7 days; roam inbox empty; KB at 106 =:agent:= nodes but the best-practices node path resolved empty (=rg -l 'agent-kb-best-practices'= found nothing — worth checking whether that node exists); no spec-sort or host-identity flags; language-bundle sync silent. + +Five new inbox handoffs arrived since the last wrap. Read all five and ran the skeptical review on each before surfacing dispositions. + +Disposed of one without asking: home's 07-26 10:21 file was a pure FYI acknowledging that the parked MCP-registry spec decision and the separate memory-auditor track matched its handoff. It asked for nothing, so it needed no reply and no approval — deleted it. Four remain, all shared-asset or convention changes, all waiting on Craig's approval per the inbox engine's core §2. + +Skeptical-review conclusions on the four pending: + +- *archsetup publish-lock* — design sound, acceptance checks testable. Three gaps: TTL sizing has to survive an indefinite human approval wait (refresh-on-re-entry covers the agent, not a Craig who steps away); the blocked second session's behavior (wait / defer / stop) is unspecified; and it's a real build, not a quick edit. Recommend filing =[#B] :feature:=. +- *home Chrome convention* — protocols.org already prefers google-chrome-stable over xdg-open, so the new parts are =--new-tab=, multi-URL, and the confirmation line. The confirmation half contradicts the existing =&>/dev/null &= form, which discards exactly the message to be verified. Recommend applying with a foreground-when-running / background-on-cold-start reconciliation. +- *work sentry triage correction* — Craig's 07-27 correction supersedes his 07-21 ruling; today's work fire missed a Hayk DM and a Kostya PR-review request. The gap is that the current rule excludes by category (mail / messenger) and the new split is work-vs-personal, which category can't express. Shipping plugins: cmail, personal-gmail, personal-calendar, telegram, github-prs — no general work-mail plugin, so work's source is project-specific. A denylist of personal plugin names fails open on the next personal source added; a per-plugin eligibility declaration is the durable shape, and that's a design call. Recommend filing =[#B] :bug:= (Major × most-users-frequently = P2). +- *work peer-reasoning rule* — approved exact text, well-formed. Two notes: it's a reasoning contract in a file scoped to communication style (the framing line should widen), and "process serves the outcome" sits one reading away from licensing deviation from the mandatory gates. Its own wording says surface-before-proceeding, so no edit needed, but that's the line to watch. Recommend installing as written. + +** Inbox pass applied — commit 6c1ea8b + +Craig approved all four dispositions plus the probe fix. Two corrections from him along the way: I had inverted the render-merge guard (numerals belong to the options list, dashes to every other enumeration in the same message — I did the reverse), and processed items shouldn't be left sitting in =inbox/=. + +Shipped: the peer-reasoning section at the top of =claude-rules/interaction.md= with the file's framing line widened; the Chrome convention rewritten in canonical =protocols.org=; the KB best-practices probe switched from a content grep to a filename =find=. Filed two =[#B]= tasks (sentry triage split, repository publish-lock), both stamped =:LAST_REVIEWED: 2026-07-27=. Swept the 40-file =PROCESSED-*= backlog out of =inbox/= along with the four handoffs; =inbox-status= now reports 0. Replies sent to archsetup, home, and work. + +*The review caught my own error.* I had written that Chrome's confirmation line prints to stderr and told every project to capture it with =2>&1=. It prints to *stdout*; stderr is empty. Verified both directions on ratio before correcting. An agent following the original text would have captured stderr, seen nothing, and concluded the tab failed to open — the exact silent-failure shape as the KB probe it shipped alongside. I asserted a stream rather than checking it, inside the same change that told others to verify. Side effect: four =about:blank= tabs opened in Craig's live browser during the check. + +Deliberate departure recorded: =route_recommend= returned =work strong= for the sentry task, but the work happens in rulesets' canonical =sentry.org=, so it's a local keeper and I left it unstamped rather than letting the wrap router offer to ship it away. + +Possible KB promotion at wrap: an identifier that lives only in a filename is invisible to a content search, and the lookup fails *silently* rather than erroring. Adjacent to the existing enumerate-vs-discover node but a distinct failure surface. Two instances in one commit (the KB probe, the Chrome stream) argue it generalizes. + +Startup-workflow bug found while checking the KB nudge: Phase A resolves the best-practices node with =rg -l 'agent-kb-best-practices' "$ra"=, which greps file *content*. The node's slug lives in its filename, so the probe returns empty and the contribute nudge points at nothing — in every project, every session. The node exists at =~/org/roam/agents/20260620232112-agent-kb-best-practices.org=. Synced-workflow change, so it waits on approval too. + +** Pushed and synced velox + +Pushed 6c1ea8b to origin/main (ahead-only, reconciled immediately before). On ratio, so velox needed the pull: it fast-forwarded and =make install= linked three things it had been missing since 2026-07-25 — the Codex =hooks.json=, =rulesets-write-boundary.py=, and =git-worktree-gate=. That drift is exactly the one-time-setup case =daily-drivers.md= names: the files traveled with the pull, but nothing re-runs the installer, so the symlinks only land where someone runs it. A new Codex session on velox will now hit the hook review/trust prompt, which was already on the 2026-07-25 next-steps list. + +** Context-engineering rightsizing — analysis and rollout plan + +Craig supplied three Anthropic posts (the 2026-07-24 Claude 5 context-engineering post, the Opus 5 prompting guide, the 2026-07-06 Fable field guide) and asked for a review, proposals, and a consistency audit of what this repo ships downstream. Then he reframed twice, and both reframes were better than the question I'd been answering. + +*First reframe:* consider the files as *his prompts*, not my context. That changed the finding. My first pass measured the always-loaded surface (32,123 words — =claude-rules/= 25,386 + =protocols.org= 6,620 + CLAUDE.md 117, roughly 40k tokens before the user's first word) and proposed shrinking it. Read as a map he hands every project, the finding is different: 41 execution/hygiene workflows against 6 discovery/design, seven to one. That ratio was right when the risk was the model doing things wrong. The field guide's claim is the bottleneck moved to the human's ability to clarify unknowns, so the system is heavily built on the half that got easier. + +*Second reframe:* metrics per claim, not one go/no-go. Turns the rollout into a set of separable testable claims rather than one bet. + +Three checkable "doesn't practice what it demands" findings: =commits.md= argues terseness at 5,561 words (longest file in the set); =interaction.md= bans bold in chat while the rules carry 591 bold markers; =testing.md= mandates TDD then argues eight more rows against rationalizations. + +*The finding that changed the plan:* the harness system prompt already carries most of what the Opus 5 guide recommends adding — its task-scope block, correction-narration block, and subagent cap are present nearly verbatim, and post 1's replacement comment guidance is present as the post's own new wording. So applying the posts additively would make the duplicate-and-conflict problem worse. The posts' value here is subtractive. It also exposes a third dedup axis nobody has audited: =claude-rules/= against the harness prompt, invisible from inside the repo. + +*Pilot selection rule* (the part that matters more than the list): the six pilot files were chosen because a silent miss is *detectable*, not because they're small. Four have a mechanical checker (=lint-org= =org-table-standard=, spec-board grep, =spec-review=), two produce an error Craig sees in seconds. =daily-drivers.md= and =emacs.md= were considered and held back — low risk, but a miss surfaces too slowly to learn from inside the trial window. + +Artifacts in =working/context-engineering-rightsizing/=: =proposals.org= (P1-P6, conflicts C1-C2, the from-your-side-of-the-desk section), =rollout.org= (Phases 0-8, decisions D1-D7, target trajectory), =metrics.org= (claim-by-claim testability, pilot go/no-go with the denominator rule, turn-back vs abandon triggers). + +Two honesty notes carried into the docs: I have a stake in arguing my own instructions should be shorter, so the plan weights mechanical detectors over my self-report; and about half the posts' claims aren't testable here without an eval harness, so those are labelled judgment rather than measurement so a future session doesn't mistake an adopted opinion for a tested result. + +Not started. Awaiting D1 (confirm pilot set) and D2 (skill index in the core). + +** Path-scoping shipped (0adcb1a) and work pre-synced + +The session's biggest finding: Claude Code scopes a rule by a =paths:= field in YAML frontmatter, and none of the 20 rules had one — even though three already declared a file-type scope in their =Applies to:= prose line. So =todo-format.md= (4,494), =org-tables.md= (464), and =emacs.md= (923) loaded into every session in every project, contradicting their own first line. 5,896 words. Fixed by adding the frontmatter, plus a =lint.sh= checker that warns when prose names a concrete extension without matching frontmatter (flags exactly those three, nothing else), plus teaching the heading check to skip a frontmatter block. Always-loaded rules surface: 25,386 → 19,505. + +Also confirmed from the docs: user-level and project-level rules *both* load, and project rules take priority. So work and =.emacs.d= carry 19 byte-identical duplicate copies, and a stale project copy overrides a fresh global one — which is exactly what was happening to work's =interaction.md= between this morning's commit and its next startup. + +Pre-synced work via =scripts/sync-language-bundle.sh ~/projects/work= (rulesets' own installer, run early rather than waiting for work's startup) so Craig's next work session is a valid test rather than one running the set it loaded before the sync. Verified: all four files now match canonical, frontmatter present, and work's =.claude/= is gitignored there so nothing was dirtied. + +Open question the next session answers: does =paths:= frontmatter apply to *user-level* rules or project-level only? The docs don't draw the distinction. =/context= in a fresh session settles it — if =todo-format.md= is absent from Memory files until an org file is opened, it works. If it's listed, the frontmatter is inert (no harm) and semantic skills are the only route. + +Not done: the double-load fix. Removing the 19 duplicates means changing what =install-lang= pushes into projects, and there may be a teammate-facing reason for them. Surfaced as Craig's call, not urgent — wasteful, not harmful. + +** De-duplicated the rules layer, unblocked sync (7ea1d7b, 79ed3b0) + +Craig confirmed no teammates depend on the per-project rule copies, so I removed them. =install-lang.sh= no longer copies the generic rules; =sync-language-bundle.sh= sweeps the ones earlier installs left, guarded on the global rule existing so a machine mid-bootstrap isn't stranded with none. Swept 20 files each from work and =.emacs.d=, leaving only their language rules plus work's =publishing.md= overlay. Three existing tests encoded the old contract and were rewritten; the generic-drift test now asserts sweep-not-repair, which is the stronger fix since the drifted copy outranked the global rule while it existed. Four new tests cover the sweep, the two keep-cases, and the no-global-rule guard. + +Also gitignored =.ai/session-context.org= and =.ai/session-context.d/=. This repo tracks =.ai/=, so the live anchor read as untracked all session and =git-worktree-gate= reported rulesets sync-blocked — meaning every other project skipped its rulesets pull until wrap, every session. Craig spotted the blocked state and inferred it was why I pre-synced work; it wasn't (rules load at launch, before the startup sync runs, which was the real reason), but chasing his inference found the anchor problem, which was the better bug. + +Corrections from Craig this stretch: the goal is output quality first, token reduction second — my docs led with the wrong number and P4 (effort reduction) should be demoted or dropped since it trades quality for cost. And all authored prose goes first person; I amended the first commit rather than leaving it. Code comments stay third-person by agreement, since they describe what the code does for the next reader. + +Docs not yet updated for either the goal reordering or the last two hours of findings (path-scoping, the double-load, the harness overlap). That's the next task. + +** Inbox: archsetup ack + +archsetup acknowledged the publish-lock acceptance and the three implementation gaps, confirming the decision stays closed on its side. Pure FYI, nothing asked, no reply owed. Deleted it. Inbox back to zero. + +** Killed Craig's work session with my own hook, then fixed it (2c664cb) + +Craig's 13:20 work session was blocked repeatedly and then had its terminal closed under it. The cause was mine, from Saturday's clean-wrap work. + +=wrap-it-up= drops =/tmp/ai-wrap-teardown-<project>= so the =Stop= hook tears down once the wrap certifies clean. I deliberately made a failed certification *preserve* the sentinel, so a wrap blocked by a dirty tree could retry on a later stop. I never bounded that retry to the session. work's 11:37 wrap left an uncertified sentinel armed; the 13:20 session's stops were all blocked by it failing certification; then startup's two commits (task filing, template sync) made the tree clean, the next stop certified, and =cj/ai-term-quit= killed the tmux session mid-work. + +Two others were armed and dangerous at the same moment: archsetup's since Saturday 15:02 on a live attached terminal, and home's from 13:21 on a live session. Disarmed all three by hand (backed up to =/tmp/disarmed-sentinels=) before writing any fix, since both were minutes from the same fate. + +Fix: =hooks/session-start-disarm.sh= clears the project's sentinels at =SessionStart= — a new session means the wrap that armed one is gone. Within-session retry is untouched (the hook only runs at session start) and a test pins that so the deliberate behavior isn't lost to the fix. Four tests on the disarm including project-scoping, one on the retry. Wired into =.claude/settings.json=, installed on both machines, =wrap-it-up.org= documents the session-scoping with the worked failure. + +Diagnostic note worth keeping: I found it by reading work's own crashed session anchor, which showed startup completing normally and then stopping dead, plus its git log showing two commits at 13:21 — the exact moment the tree went clean. The anchor being left behind by the interrupted session is what made the timeline reconstructable. That's the crash-recovery purpose earning itself. + +** /context settled both open questions; docs corrected (d74d98d) + +Craig ran =/context= in work. Memory files lists 17 generic rules; =todo-format.md=, =org-tables.md=, and =emacs.md= are absent, and only =python-testing.md= and =publishing.md= come from the project's own rules dir. So *path-scoping works at user level* and *the de-duplication holds*. Both were open. + +Three corrections the live numbers forced: + +1. *My token figures were low by ~45%.* Real ratio is 2.28 tok/word, not the ~1.3 I assumed. =commits.md= is 12,800 tokens (I said ~7,000); =claude-rules/= was ~57,800/session before today, now 44,410, with 13,390 path-scoped out. Worth naming the actual error: =/context= reports per-file token counts and I used a word-count estimate instead because it was easier to compute from inside the repo. The instrument existed the whole time. +2. *Two loading paths, not one.* Memory files arrive via the harness at session start. =protocols.org= and the workflows are *read by startup*, so they land in Messages and never appear under Memory files. They shrink by editing the workflow, not by scoping a rule. My "always-loaded surface" number conflated them. +3. *The harness's own suggestion* names =commits.md=, =testing.md=, =MEMORY.md= as the top three to prune — independently the same Phase 4 list I'd proposed. + +Because a glob match is deterministic, the remaining work splits: path-scopable rules ship with no trial (=docs-lifecycle.md= on =docs/**= is next), and only semantic-condition rules need the skills route and the stop conditions. =commits.md= is the real test there — largest single item, and almost all publish machinery that only applies when a commit is in play. + +Recorded a caution the confirmation doesn't cover: path-scoping fires on a *read* of a matching file, so creating a new org file from scratch never triggers =todo-format.md=. Edits are safe (Edit requires a prior read). + +Also folded in Craig's goal correction (quality first, tokens second): P4/Phase 8 dropped outright since lowering effort trades quality for cost, P5 promoted since positive-framing-over-prohibition is what targets guardrails working against output. + +** Split commits.md: 12,800 tokens → 2,342 always-loaded (931f364) + +Craig picked the commits.md split over docs-lifecycle after I checked the latter and found I'd overstated it — =docs-lifecycle.md= scopes to "any project carrying a docs/ tree," a *project-level* condition a glob can't express, and 2 of its 6 trigger points are creation cases a read-triggered path rule misses. Only three rules ever named a concrete extension and all three are already converted, so there is no other clean path-scope candidate. + +The split line is *blast radius*, not size. Stayed always-loaded (1,027 words / ~2,342 tokens): author identity, the no-AI-attribution ban, the generated-document byline rule, the public-artifact content-scope rules, and "If You Catch Yourself." Moved to =publish/SKILL.md= (4,871 words): message format, Voice and Focus, PR description structure, Review and Publish Steps 0-2, the three review shapes, hook authorization, merge strategy, the pre-commit checklist. + +Why that line: if the skill fails to trigger I don't know the flow and have to be told — visible and recoverable. I don't silently commit with AI attribution, because that guard never moved. Only the recoverable half rides the skill-triggering bet, which is what let this ship without waiting on the pilot. + +*Verified by using it.* The skill registered mid-session and I invoked =/publish= to publish its own commit; it loaded with the full flow present. Content conserved and checked rather than assumed: 5,561 words in, 5,898 across both files (delta = frontmatter + the pointer added to the core). Repointed five cross-references in =voice=, =review-code=, =inbox.org=, and =no-approvals.org= that named moved sections. + +Always-loaded rules surface: 44,410 → ~33,950 tokens. Started the day at ~57,800. + +Noted and deliberately not done: =publish/SKILL.md= is a single 4,871-word blob, and both posts argue a long skill should use progressive disclosure internally. It loads on demand now, which is the win worth taking; splitting it further is its own change. + +Also surfaced: the Step 2 =.ai=-tracking heuristic misfires here. It reads tracked =.ai/= as "shared team repo → skip the approval gate," but rulesets tracks =.ai/= as a committed mirror while being a private single-user repo. I kept asking rather than skipping, and flagged it to Craig. + +** testing.md split, gate fixed, first-person directive added (2f45b6e) + +Three changes. *testing.md split* the same way as commits.md — by what has to be resident, not by size. Core keeps TDD-is-default and the three-category requirement (347 words), because those fire *before* any code is written, which is exactly when no skill has been summoned. Everything else → =testing-standards= skill (2,903 words): characterization recipes, per-category detail, property/mutation testing, pyramid, integration rules, naming, test-quality and mocking rules, coverage targets, spike exception, anti-patterns. + +*Approval-gate fix.* The publish flow decided whether to ask by checking whether =.ai/= is tracked, as a proxy for "team repo." Wrong in the direction that matters: rulesets, home, and work all track =.ai/= and all three are private single-user repos, so the rule skipped the gate on Craig's three most-used projects. Now checks whether any remote is on a host other than cjennings.net. Verified both directions including a synthetic GitHub remote. Every current project → gate applies, which matches how the flow has actually been run all session. + +*First-person directive* added to the always-loaded core, at Craig's instruction. One existed for commit bodies/PR prose but it moved into the publish skill, and it never covered code comments at all. Now: everything authored in or about the repo is first person, with one carve-out — a comment describing what the code *does* stays third person, since there the code is the actor. + +Also split =publish/SKILL.md= internally: PR descriptions + the three review shapes → =references/pull-requests.md=, since a plain commit never needs them. SKILL.md 5,012 → 3,888 words. + +*Surface: ~57,800 tokens this morning → ~28,949 always-loaded now* (plus 13,461 path-scoped). Largest remaining: =interaction.md= 3,828, =verification.md= 3,388, =commits.md= core 2,804, =subagents.md= 2,373, =cross-project.md= 2,305. + +Risk recorded rather than buried: testing.md's margin is thinner than commits.md's. If =testing-standards= fails to trigger mid-test-writing I lose the mocking-boundary rules — a quality regression, visible in review, but a real bet where commits.md's moved half was purely procedural. Also moved the TDD rationalization table rather than cutting it; the posts say that kind of over-argument is counterproductive now, but deleting Craig's defense against me skipping TDD is his call. diff --git a/.ai/sessions/2026-07-29-06-36-adversarial-review-flow-and-telegram-fixes.org b/.ai/sessions/2026-07-29-06-36-adversarial-review-flow-and-telegram-fixes.org new file mode 100644 index 0000000..12cbe3d --- /dev/null +++ b/.ai/sessions/2026-07-29-06-36-adversarial-review-flow-and-telegram-fixes.org @@ -0,0 +1,514 @@ +#+TITLE: Session Context — 2026-07-28 +#+AUTHOR: Craig Jennings +#+DATE: 2026-07-28 + +* Summary + +** Active Goal + +Started as inbox triage on a telegram-plugin bug report and became two things: shipping the cross-project fixes that arrived overnight, then designing and dogfooding a mandatory isolated adversarial review before every commit — which immediately found real defects in its own design and in everything filed afterward. + +** Decisions + +- *Merge colliding fixes rather than sequence them.* The parked down-is-launch diff still carried the bad =loadChats= call and cited the segfault gotcha the other fix rewrites. Applying either alone would have shipped a file arguing against itself. +- *"Adversarial", not "hostile"* (Craig). An agent told to attack manufactures findings, so the stance carries a substantiation floor: a finding not substantiated against the diff is dropped. +- *Re-review until the reviewer approves* (Craig's addition, the thing I had missed). Same reviewer continued, not a fresh one — a fresh reviewer can't tell an addressed finding from one that never existed. Bounded at three rounds or first recurrence. +- *Dispatch on every commit*, with the reviewer's own Phase 0 ruling triviality. A floor written as "small" or "mechanical" puts the judgment back with the author, whose judgment is the thing being checked. +- *Pass the requirement source, withhold the rationale.* A ticket is not the author's model; it was written first and by someone else, so it's the only input that can contradict the author's claim. +- *=cycle=, not =pass=, for one sentry loop* (Craig). home proposed =pass=; =sentry.org= already uses it as a numbered noun for the eleven hygiene passes, so =Pass 11= would have collided with =pass 12=. +- *Drop the =references/= link rather than sync the directory.* The four calendar workflows already travel and already carry the recipes. +- *Strip the wrap-org-table task to Verified / Open questions* after its review loop bounded out. The measurements were never what failed. + +** Data Collected / Findings + +- *The telegram bug.* =(telega--loadChats 'main)= sends a bare symbol on the wire; =tdat_plist_value= (=telega-dat.c=) accepts only =(=, =[=, ="=, =-=, digit, =t=, =:=, =n= and calls =assert(false)= on =m=. Verified against telega's source at four points rather than trusting the handoff. Exposure was manual triage only — sentry excludes messengers, so home's eleven overnight cycles never loaded the plugin. +- *=wrap-org-table.el= splits logical rows*, and =lint-org= doesn't merely miss it — it *causes* it. =lint-org.el:424= calls the same broken predicate, so it reports the tool's own correct output as "missing rule between rows — wrap-org-table.el reflows it" when nothing is missing, then reports the corrupted result clean. Idempotence is broken: the tool corrupts its own output on a second run. +- *The isolated reviewer earned its keep on its first four uses*, finding: that withholding the ticket made my claim self-certifying; that my =subagents.md= override reaffirmed the Prompt Contract field that would destroy the isolation; a fourth verdict (=Needs Discussion=) I'd asserted didn't exist; and four successive wrong root-cause analyses on the table bug. +- *rulesets is itself exposed* to the table bug: =todo.org='s four-row attachment-sanitization table. Don't reflow until fixed. +- *Seven =../../= link sites* across four synced workflows resolve only in rulesets. =scripts/lint.sh='s =check_md_links= was built for that class and misses them because it matches markdown syntax only. + +** Files Modified + +Seven commits, all pushed, velox synced after each. =bff0138= merged telegram fixes. =43a4cf7= post-load liveness check. =614e3b1= removed finished working dirs. =ca508a1= filed three handoff findings. =f3f5bfd= the =fire= → =cycle= rename (72 sites plus three that had leaked outside =sentry.org=). =3a933a2= the =references/= analysis. =5999f88= dropped the dead link and deleted the stale file behind it. =ecd5d7b= filed the wrap-org-table bug. + +Rules changed: =publish/SKILL.md= Step 1 rewritten (dispatch contract, four defined verdicts, the loop, bounds), =review-code/SKILL.md= (two levels of dispatch, adversarial contract, re-review mode), =claude-rules/subagents.md= (Isolation Override; Prompt Contract field 2 inverts), and the three unattended callers taught to park. + +** Next Steps + +- *Three of the four items Craig queued are untouched*: the winvm =[#C]= lint defects, the context-engineering rightsizing (needs his four decisions), and the sentry spec deep read (two weeks old). +- *Rule gap found by using the rule*: =Needs Discussion= exits to the user, but nothing says what happens after the user answers — whether the round counter resets. I treated it as a fresh review; that judgment isn't written down. +- *Item 2 is now qualified*: its fix says "run =wrap-org-table.el=", and that tool has a live corruption bug. The specific table is safe, but verify the output rather than trust it. +- Four =[#B]= bugs filed tonight and unstarted: the table splitter, the =../../= links, plus the two carried in. + +KB: promoted 1 / consulted no + +* Session Log + +** 11:55 — Startup + +Ran startup. Rulesets already current, project repo clean and current, =make +install= had nothing new to link, =.ai/= synced from templates. No crash anchor +— previous session (context-engineering rightsizing, 2026-07-27 17:02) wrapped +cleanly. + +Findings: 6 tasks unreviewed >7 days; roam inbox holds 4 items; KB at 108 +=:agent:= nodes with nothing matching this project. Spec-sort and host-identity +probes silent. Language-bundle check silent. + +** 12:05 — Inbox: the telegram segfault root cause + +Four new inbox files from =.emacs.d=, two pairs: a 06:15 intro note + plugin +file, then a 07:21 correction + superseding plugin file. The correction retracts +one secondary claim from the 06:15 write-up (that the "19 of ~50 chats" reading +was truncation caused by the bug — it wasn't; 19 is the real account size, +measured by work at the wire level). Root cause and fix unchanged. + +The proposal: =triage-intake.telegram.org= Step 1 calls =(telega--loadChats +'main)=, and that bare symbol kills =telega-server= outright. + +I verified the whole chain against telega's own source rather than taking the +handoff's word for it (=elpa/telega-20260706.2147/=): + +- =telega--loadChats= (telega-tdlib.el:2190) drops its argument straight into + the request as =:chat_list= with no conversion. Confirmed. +- The C parser =tdat_plist_value= (server/telega-dat.c:466) accepts only =(=, + =[=, ="=, =-=, a digit, =t=, =:=, or =n= to start a value; anything else + prints "Unexpected char '%c' in plist value" and calls =assert(false)=. + =main= starts with =m=. Confirmed, and the accepted-char list in the handoff + is exactly right. +- telega's own callers all pass the object: telega.el:290, telega.el:295, + telega-tdlib-events.el:516. Confirmed. +- The symbol shorthand lives in a different layer — telega-filter.el:472 and + telega-folders.el:99 convert =(eq cl-fspec 'main)= to ='(:@type + "chatListMain")=. Confirmed. + +Every claim holds. Grepped all workflows for other bare-symbol arguments to +=telega--*= wrappers: only the two sites this diff already fixes. + +*The complication I found.* There is already a parked =[#B]= VERIFY (todo.org:136, +from =.emacs.d= 2026-07-24) proposing a different fix to this same file — the +down-is-launch fix, with a prepared diff in +=working/triage-telegram-down-launch/=. Its proposed file still carries the bad +=loadChats= call at its lines 52 and 122, and its third hunk adds prose citing +"tdlib segfaults in native mode (SEGFAULT gotcha below)" — pointing at the very +section this new handoff rewrites to say the crashes were our bug, not tdlib's. +Applying either one alone leaves the file half-right. They need merging, not +sequencing. + +Also: home and work both declare =telegram= in =:TRIAGE_SOURCES:=, so both have +been killing telega-server on every sweep. Work already patched locally and +verified at the wire level. Home has not been told. + +** 11:58 — Staged the batch, closed the loops + +The Stop hook counted the four files still sitting in =inbox/=. The triage was +done; what remained was Craig's merge decision, not mine. So I staged all four +into =working/triage-telegram-segfault-rootcause/= (the 07:21 pair as +=.proposed= + correction note, the 06:15 pair kept as superseded context — its +blast-radius and follow-up sections aren't repeated in the correction). Inbox +back to clean, =inbox-status= rc=0. + +Left =working/triage-telegram-down-launch/= untouched: the parked VERIFY's body +links its =proposed.diff= by path, and moving it would break the link before +Craig has decided anything. If he takes the merge, that consolidates then. + +Replied to =.emacs.d= — confirmed the verification, and told them the one thing +they couldn't see: their own 2026-07-24 parked diff collides with this, both in +the stale =loadChats= call it still carries and in the native-mode prose it adds +citing the gotcha this rewrites. Also declined their suggested bare-symbol lint: +one consumer file, two call sites both fixed, so the stated rule is cheaper than +a checker with a single input. + +Notified home directly with the mechanism and the two-site fix, told it to patch +locally rather than wait on the canonical, and carried the correction forward +explicitly so it doesn't inherit the retracted truncation story. + +Nothing applied to the canonical. That still waits on Craig. + +** 12:10 — Craig picked the merge. Applied and committed. + +Merged both fixes into one version rather than sequencing them. Base was the +segfault-corrected file, then the three parked hunks on top: the down-is-launch +directive, the SCAN-FAILED-only-after-launch-attempted rewording, and =(setq +telega-use-docker t)= restored to the Step 1 code block. + +The reconcile that made merging worth doing. The parked hunk's new comment said +"tdlib segfaults in native mode (SEGFAULT gotcha below)", pointing at the +section the other fix rewrites to say those deaths were our own bad argument. +Left alone the file would have argued against itself. I changed the Step 1 +comment to state plainly that the two are separate concerns (the deaths happened +*in* docker mode, so docker mode is neither a defense against the loadChats bug +nor evidence for itself), and reworded the Quick Reference line from "tdlib +segfaults outside docker mode" to "crashed in native mode (2026-06-09)" with the +same disambiguation. + +That reword also removed a host-identity violation I hadn't gone looking for. +The original asserted "Craig's daemon currently has telega-use-docker nil" — a +mutable machine fact stated as fixed in a synced doc. I checked the actual +default (=telega-customize.el:514=, =defcustom telega-use-docker nil=) and wrote +the durable claim instead. + +Verified: both live call sites use the TL object, the two remaining ='main= +occurrences are inside the gotcha prose describing the bug, lint-org clean on the +changed file, mirror synced, =make test= green before (exit 0) and after (exit +0). + +todo.org: closed the parked =**= VERIFY as =DONE= + =CLOSED:= per todo-format.md +with the merge rationale in the body. Promoted its =***= engine child (SCAN +FAILED must not advance the sentinel) to top-level =**= VERIFY so it doesn't get +buried under a DONE parent. Kept its =:LAST_REVIEWED: 2026-07-24= rather than +stamping today — I moved it and judged it separate, but nobody re-derived its +content, so the older date keeps it honest. + +Review: Approve, no Critical or Important. Two Minor, both surfaced rather than +fixed. The gotcha now advises a post-load =(process-live-p ...)= check that the +Step 1 recipe doesn't actually do, and adding it would extend the recipe past the +two fixes Craig approved. + +Committed =bff0138=. Not pushed — that's a separate confirmation. + +** 12:20 — home replied, and corrected my urgency read + +home accepted, patched both call sites locally, and re-verified the diagnosis +independently rather than trusting it. Useful correction back: home declares +telegram in =:TRIAGE_SOURCES:=, but sentry excludes messengers from triage +intake, so the plugin never loaded on a sentry fire. Eleven overnight fires ran +clean against the broken file. I had assumed the sweeps were affected; the real +exposure is manual triage intake only. + +Told home its stopgap won't be reverted into a broken state — the next rsync +replaces it with canonical content carrying the same fix. On this machine that +lands as soon as its next startup runs, since the rsync reads the local rulesets +working tree. velox needs the push. + +Inbox back to zero. + +** 14:10 — Pushed, and closed the velox gap + +Pushed =1675613..bff0138= to origin after the pre-push reconcile (still 1 ahead, +0 behind). + +=.emacs.d= replied while the push was in flight. It had already verified bff0138 +here, re-run its own workflows rsync, and confirmed the corrected form landed. +It withdrew the bare-symbol lint suggestion, conceded the home omission, and +flagged one real gap: bff0138 was committed but not pushed, so velox stayed +exposed. That was true when written and stale by the time it arrived. + +Checked velox over tailscale (this host is ratio, per =uname -n=). It was 1 +behind / 0 ahead and sync-safe, so I fast-forwarded it through the same +=git-worktree-gate sync-safe= check startup uses. Velox is now at bff0138 and its +workflow file carries the corrected call, with the only bare ='main= occurrences +inside the gotcha prose. Both daily drivers covered. + +Corrected read carried into both replies: the exposure was manual triage intake +only, not the automated sweeps, because sentry excludes messengers. + +** Open follow-ups (surfaced to Craig, not acted on) + +1. The gotcha tells callers to check =(process-live-p (telega-server--proc))= + after a load, but the Step 1 recipe doesn't do it. Now that the corrected call + shouldn't kill the server, that check is what would catch a regression. Left + out deliberately as scope creep past the two approved fixes. +2. Both =working/triage-telegram-*= dirs are completed-task artifacts and want + filing per working-files.md. Revised read after checking: delete both + outright. Every file is tracked (=b19d420= and =bff0138=), so git holds them + permanently and a copy in =assets/= would only duplicate history. Nothing + links to them. + +** 15:00 — Second inbox round: .emacs.d self-correction + winvm lint findings + +=.emacs.d= wrote back to say it had overcorrected on home: it accepted "home was +in the blast radius" and then recorded that home "had been killing telega-server +on every sweep too", which home's own sentry data refutes. It fixed its task +record rather than leaving it. I told it the pattern wasn't one-sided — I made +the same move this morning, estimating home's blast radius instead of measuring +it, and home's data is what corrected me. It also offered the emacs-side half of +a completed-vs-truncated signal, which it has filed as =[#C]=, if the +=process-live-p= recipe change lands. + +=winvm= sent a link-integrity pass with three findings, all reproduced against +the rulesets source rather than only its local copy. I verified all three: + +1. =protocols.org:273= links =references/calendar-reference.org=, but the rsync + set is only =protocols.org=, =workflows/=, =scripts/=. Dead link in every + consuming project. home and =.emacs.d= have no =.ai/references/= at all. +2. =retrospectives/PRINCIPLES.org:38= violates the org-table standard. + =lint-org= confirms, checker =org-table-standard=. +3. =protocols.org= lints 8 mechanical + 19 judgment =misplaced-heading=, all + from Markdown =**bold**= in an org file. 48 bold spans, 14 line-initial. + +The check winvm couldn't run: adding =references/= to the sync brings =--delete= +with it, which would destroy project-owned files there. work is the only project +with a =.ai/references/= and its copy is byte-identical to the source, so nothing +would be lost. That clears the path to sync-it over drop-the-link. + +One refinement back to winvm on finding 3: the 8 mechanical hits are would-fix, +and the 14 line-initial bold spans are exactly what =lint-org --fix= would +rewrite without knowing they're emphasis rather than headings. So that half is +not safe to run mechanically even though the count suggests it is. + +Filed as =[#B] :bug:= (the dead link, Minor x every-project-every-sync = P2) and +=[#C] :chore:= (the two cosmetic defects, P3), both stamped =:LAST_REVIEWED:= +today since I graded them today. Verified todo.org's lint counts are unchanged +from the pre-edit baseline (mechanical=2, judgment=12, both pre-existing and +already in =inbox/lint-followups.org=). + +Inbox back to zero. todo.org is left uncommitted deliberately — Craig has a +pending decision that will likely touch it again, so batching beats a churn +commit. + +** 18:45 — Craig took both follow-ups. Liveness check staged. + +Added the liveness check to the Step 1 load call. It now returns ='server-died= +instead of an unconditional ='loaded=, and new prose routes that into the +existing SCAN FAILED definition rather than inventing a category: a server that +dies during load leaves a thin hash, which reads exactly like a quiet account, +which is the same false all-clear the down/not-loaded rule already prevents. + +Self-review caught a real defect in my own first draft. I called +=(telega-server-live-p)= bare, but Step 0 guards the same predicate with +=fboundp= because telega may not be loaded. A launch that failed outright would +have signalled void-function instead of returning the clean contract. Added the +guard, matching Step 0's idiom. Verified the predicate is exactly the +=process-live-p= expression the gotcha names (=telega-server.el:221=). + +Verified: parens balance at depth 0, lint-org 0/0, mirror identical, =make test= +green (exit 0) on the final state. + +Not committed — waiting on the approval gate. + +** 18:48 — Third inbox round + +=.emacs.d= sent a closing FYI marked no-action, agreeing my framing of the shared +failure (blast radius estimated rather than measured) named the trigger rather +than the failure. Deleted without reply, since replying to "nothing owed back" +is noise. + +home proposed renaming sentry.org's noun-sense "fire" to "pass", after Craig read +its "nine fires" as nine emergencies: "I assume you mean nine crises, not nine +loop cycles and I begin to get scared." The problem is real, well-evidenced, and +reaches Craig directly through digest headings. + +But the proposed term is wrong, and the reason home gave for it is the +disqualifier. =sentry.org= already uses "pass" as a precise numbered noun — the +pass list, the Pass Runner, "eleven finding/hygiene passes", "pass 12". With +exactly eleven hygiene passes, home's proposed =** Pass 11= heading collides with +an existing referent. That trades a term Craig misreads as urgent for one that is +genuinely ambiguous. + +Counter-proposed *cycle*: zero occurrences in the file, and Craig's own word in +the quote home cited. Checked and rejected "sweep" (3 uses) and "run" (used as a +noun). Filed =[#C] :chore:= with the grading and the collision analysis; replied +to home with the counter-proposal. + +** 18:50 — Three commits, and the collision confirmed from live evidence + +Craig approved both follow-ups and the =cycle= term. Three commits: + +- =43a4cf7= the liveness check. +- =614e3b1= removed both telegram working dirs. Filing by deletion, since every + file was already in git via =b19d420= and =bff0138= and an =assets/= copy would + only duplicate history. +- =ca508a1= filed the three handoff findings with their gradings. + +home wrote back confirming the "pass" collision was real, and that it had already +walked into it: its anchor now carries =** Pass 11= meaning the eleventh cycle, +three lines from =pass 12 (solo-task implementation)= meaning the twelfth item in +the pass list. Same file, two referents, introduced by its own normalization an +hour earlier. It found that in live evidence faster than reading the file would +have caught it. + +It was blocked on Craig's confirmation and had written a memory saying "pass", so +I sent the confirmation immediately. The memory was the urgent half — a stale one +teaches every future home session the ambiguity, where the anchor is one file. + +Recording Craig's approval flipped the sentry task to =:solo:=. The term was the +only judgment it carried, and the completion check is objective, so it can ride a +backlog run rather than waiting for someone to touch =sentry.org=. + +Pushed =bff0138..ca508a1=, velox fast-forwarded to match. + +** 19:45 — The cycle rename, and the leak home's scope missed + +home did its side first: renormalized its anchor by restoring the +pre-normalization backup and re-running fire→cycle from clean, rather than +reverse-mapping pass→cycle. That was the right call — reverse-mapping would have +needed a judgment on every instance to separate its own conversions from genuine +pass-list references, where re-running from clean makes it structural. It also +corrected its memory and handed the canonical back. + +Did the canonical rename with a script rather than by eye: protect the verb sites +by explicit pattern, assert zero unclassified =-ed/-ing= forms survive, then +substitute. 72 noun instances converted, 4 verb sites untouched (=/loop= fires +again, two "fires on approval", "record of what fired"). + +*The leak home's scope missed.* The term wasn't confined to =sentry.org=. +=wrap-it-up.org= said "a crashed fire", and =todo-cleanup.el= and its test both +said "every sentry fire" — all three naming a sentry cycle. Renaming only +=sentry.org= would have split the vocabulary across files. Found by grepping +every file that mentions sentry, then re-grepping without a context window after +the first pass truncated short-line matches and hid them. + +Verified: exactly 3 verb instances left in sentry.org, no placeholder leaked, the +digest commit template now reads =<date> <time> cycle=, capitalized plurals +handled, lint 0/0 on sentry.org, suite green (exit 0 — load-bearing here, since +=todo-cleanup.el= and its test are under test). The two =wrap-it-up.org= lint +findings are pre-existing and identical at HEAD. + +Committed =f3f5bfd=, pushed, velox fast-forwarded and verified. Notified home +(with the leak it hadn't seen) and =.emacs.d=. Closed the task =DONE=. + +Four commits this session, all pushed, both daily drivers current, inbox at zero. + +** 20:00 — Roam inbox zero, then the adversarial-review design + +Roam scan: 3 items, 1 claimed (=rulesets:= prefix), 2 unowned gear links left +for Craig. Filed the claimed one, removed it from roam under capture-guard + +roam-write lock, triggered =roam-sync=. A local =.emacs.d= FYI also cleared. + +The claimed item: "code reviews must occur before every commit an agent does, +and they should be hostile reviews from a subagent without the agent's context." + +Craig's decisions: *adversarial* rather than hostile (he took my push-back that +an agent told to attack manufactures findings), plus a requirement I had missed — +a re-review loop that runs until the reviewer approves — and that the rules must +not contradict afterward. Unopposed recommendations I proceeded on: dispatch on +every commit with the reviewer's own gate deciding triviality, and the flow lands +in the =publish= skill. + +Wrote it into three files: =publish/SKILL.md= Step 1 (dispatch contract, +adversarial-with-substantiation, the loop, bounds), =review-code/SKILL.md= (the +two levels of dispatch, the adversarial contract, re-review mode), and +=claude-rules/subagents.md= (a new Isolation Override section, since three +separate size rules there said don't dispatch small work). + +** 20:30 — Dogfooded it, and the reviewer found nine things + +Ran the new flow on its own diff: dispatched an isolated adversarial reviewer +with the diff plus a one-line claim, withholding everything else. It returned +REQUEST CHANGES with seven Important and two Minor, every one substantiated. I +checked each against the files rather than accepting them, and all nine were +real: + +1. =Skipped= from Phase 0 is not =Approve=, so trivial diffs dead-ended at a gate + with no defined pass. +2. =no-approvals.org= line 73 (the actual execution step) still described the old + inline unbounded flow; I had only updated the preamble at line 49. +3. =work-the-backlog.org= and =sentry.org= — the unattended callers — had no + receiver for "stop and surface to the user". The speedrun routes to + work-the-backlog, not the file I updated. +4. The Step 2 exception still said the review runs "when it applies", which my + rewrite had made false. +5. *The sharpest one.* Withholding the ticket/plan makes the author's claim + self-certifying and strands =review-code='s Intent-vs-Delivery criterion — the + one aimed at exactly the inherited-scope error this gate exists to catch. A + ticket is not the author's model; it is the independent record of what was + asked. I had not considered this. +6. The loop turned on the verdict token, so a single Minor could burn all three + rounds and escalate. +7. My override said it "doesn't relax the Prompt Contract" while field 2 of that + contract says paste your context verbatim — which would destroy the isolation + the whole change is built on. +8. todo.org carried an unrelated =references/= rewrite, and the new task body + listed as open the questions the same commit answered. +9. =subagents.md= still says subagent output is a claim to verify, unreconciled + with "approval is the reviewer's to give". + +All nine fixed. The =references/= hunk is split out as =3a933a2=. Round 2 sent +back to the same reviewer, which is the loop working as designed. + +** 21:30 — The loop closed at three rounds + +Round 2 (three findings): =review-code='s adversarial contract still said two +inputs, so the round-1 fix landed in =publish= but not in the text the reviewer +reads — the two files disagreed at the one seam the change was about. A +count/list regression my own fix introduced ("exactly two things" over a +three-item list). And a fourth verdict I had missed entirely: =review-code= +emits =Needs Discussion=, which is exactly what an adversarial reviewer reaches +for on an architectural objection, and the flow had no handling for it. + +Fixing =Needs Discussion= reproduced round 1's finding 3 one level down, which I +caught myself: all four unattended parking clauses keyed on "the loop hits its +bound", and a first-round =Needs Discussion= is not a bound hit. Rewrote all four +to trigger on "the review can't reach approval" with the three causes named, and +asked the reviewer to verify that independently rather than take it from me. + +Round 3: APPROVE, with one Minor — a recurrence of the count/list mismatch, since +my fourth-verdict fix left the lead-in saying "three outcomes" above four +bullets. Fixed the numeral. The committed diff therefore differs from the +approved one by exactly that word, which the flow's own Minor-only rule permits +rather than spending a fourth round. + +The reviewer also verified things I had not asked about and would not have +checked: that =failed= is already a legal outcome slug in work-the-backlog's +metrics table, that the three workflow mirrors carry identical blob hashes rather +than merely similar text, and that no fifth verdict token exists anywhere in +=review-code=. It filed one follow-up correctly rather than fixing it in-diff — +=start-work.md= Phase 7 still summarizes the publish flow instead of pointing at +it, and was stale before today. Filed =[#C] :chore:solo:=. + +Convergence shape across the three rounds: nine findings, three, one Minor. The +residue in later rounds was integration error from the previous round's fixes +rather than new design problems, which is the shape the bound is calibrated for. + +Committed =8062460=, pushed, velox fast-forwarded and verified. + +Six commits this session, all pushed, both daily drivers current, inbox at zero, +tree clean. + +** 2026-07-29 — Walking the remaining items, and a long lesson + +*Item 1, the =references/= dead link* (=5999f88=). Craig picked drop-the-link. +The adversarial review returned =Needs Discussion= — the fourth verdict, on its +first real use — and widened the fix twice, both correctly. My replacement prose +said credentials "live in the rulesets repo" without naming a file, which would +have sent readers to =calendar-reference.org=, whose three paths had been dead +since May. Now names =mcp/README.org=. And the file itself was orphaned by the +link removal, so both copies and the empty =references/= dirs are gone. Round 2 +approved with three Low findings, all against the task record: my count was wrong +(seven sites, not five) and =scripts/lint.sh='s =check_md_links= already exists +for that class, missing them only because it matches markdown syntax. + +*Rule gap found by using the rule.* =Needs Discussion= exits to the user, but I +never wrote what happens after the user answers. Treated it as a fresh review on +the reasoning that Craig adjudicated and the scope changed. Flagged to Craig; not +yet written into the skill. + +*The wrap-org-table bug* (=ecd5d7b=). work reported that the tool splits a logical +row and lint passes the result. Everything I *measured* held. Everything I +*inferred* on top was refuted, four times: + +1. Root cause "absence of rules in the input" — refuted by the double-run repro + (the tool corrupts its own correct, rule-delimited output). +2. "Tested and killed the empty-cell hypothesis" — the fixture was confounded. + With no hlines the code short-circuits at =:184= before the predicate is + reached, so I varied the empty cell while the path that reads it was switched + off, got a negative, and wrote it down as settled. +3. "The reporter's row-below observation discriminates between the paths" — it + doesn't; both produce that signature. Claimed twice. +4. "A static scan can't see primary-path exposure, needs simulation" — wrong, and + worse, I sent it to work, who built on it. Then my *corrected* advice (run the + predicate over rule-delimited groups) was also wrong: work implemented it and + showed it can't discriminate at any threshold, with worked examples where the + same structural signature has opposite correct verdicts. + +Three corrections sent to work, plus a fourth acknowledging their disproof. The +review loop *bounded out* at three rounds — the first bound-out under the new +rule, working as designed. Craig adjudicated by stripping the task to Verified / +Two-fixes-that-work / Open-questions, which is the right shape: the measurements +were never the problem. + +The fix that survived is work's read, not mine: check idempotence (reflow twice, +diff) rather than build a detector. It's true by construction and needs nobody to +decide what a group means. + +*The lesson, stated plainly.* Every refutation across three rounds landed on +inference, never on a measurement. I ran experiments and then over-read them, +repeatedly, at full confidence. work — after I'd sent them three wrong analyses — +labelled their own uncertain number as a floor on a population they couldn't +cleanly define, unprompted. That discipline is the thing to copy. + +I also found the 2026-07-27 note where I'd already observed this exact failure +and left it in a session summary instead of filing it. work's framing: a correct +observation recorded and then read as fine is the same failure as a green check +on a corrupted table. diff --git a/.ai/workflows/INDEX.org b/.ai/workflows/INDEX.org index b031dbe..f18d953 100644 --- a/.ai/workflows/INDEX.org +++ b/.ai/workflows/INDEX.org @@ -58,6 +58,9 @@ This index must list every =.org= file in =.ai/workflows/= except this one and e - Speedrun triggers: "speedrun", "no approvals speedrun", "speedrun these: <task set>" — any phrase containing "speedrun" routes here (the preset), never to =no-approvals.org= - Manual triggers: "work the backlog", "work the backlog with <task set>" (file-only defaults) - Synthesis trigger: "synthesize backlog metrics" — read the per-project metrics logs, compute trends + the corrections signal, write one =:agent:metrics:= KB node (personal projects only) +- =sentry.org= — the overnight hygiene supervisor: an interval loop (default hourly) that walks a fixed pass list (roam pull, inbox zero, triage, todo cleanup, task audit, working-files hygiene, spec board, link integrity, git health, prep freshness), commits each writing pass to a throwaway =sentry/<date>-<host>= branch (never pushed), and parks every judgment call and destructive action in a morning-approval queue. Gated on =:COMMIT_AUTONOMY: yes= plus interactive entry gates (clean tree, green suite) with Craig present. Locks via =agent-lock=; morning teardown (review, squash-merge, delete) is Craig's, never automated. + - Triggers: "start sentry", "run sentry", "arm sentry", "sentry mode", "start sentry every <interval>" + - Stop trigger: "stop sentry", "stand down sentry", "sentry off" ** Calendar @@ -110,7 +113,7 @@ This index must list every =.org= file in =.ai/workflows/= except this one and e - Situational triggers: "broadcast the <event> to all projects", "broadcast that <situation>", "let every project know I'll be away ..." - =flashcard-review.org= — review an org-drill flashcard file, restructure cards to question-form headings (no answer hints), audit content accuracy against project source-of-truth via subagent, rewrite source preserving SRS state, regenerate the Anki =.apkg= to =~/sync/phone/anki/=. Person cards use "Who is X? Tell me about their Y."; talking-points cards stay as-is. Script behavior: =flashcard-to-anki.py= strips =:PROPERTIES:= drawers + =SCHEDULED:= / =DEADLINE:= planning lines from Anki output. - Triggers: "review the flashcards", "update the flashcards", "review the drill deck", "update the drill deck", "refresh the Anki cards", "let's run the flashcard-review workflow" -- =page-me.org= — set a timed notification (desktop =notify=; phone via =agent-page= when Craig is away). +- =page-me.org= — set a timed notification. "page me" desktop =notify=, "text me" phone via =agent-text=, "text and page me" both. - Triggers: anything containing the word "page" used as a verb ("page me", "page me in 10 minutes", "page me at 3pm", "page my phone") - =status-check.org= — proactive long-running-job updates. - Triggers: "keep me posted on this", "provide status checks on this job", "let me know when it's done", "monitor this for me". Auto: any job estimated 10+ min. diff --git a/.ai/workflows/code-quality.org b/.ai/workflows/code-quality.org index 3ac3e9d..3c4ed8f 100644 --- a/.ai/workflows/code-quality.org +++ b/.ai/workflows/code-quality.org @@ -9,6 +9,13 @@ One trigger that runs every behavior-preserving quality pass over a scope of orchestrator — each pass keeps its own discipline and its own confirm gate; this workflow only sequences them and collects the residue. +*Behavior-preserving rests on a test net.* The passes below claim to preserve +behavior, but a refactor on untested code is a guess, not a preservation. Where +the scope has no tests, bring it under a characterization net first +(Normal/Boundary/Error per unit, per the =testing-standards= skill's "Adding Tests to Existing +Untested Code") — that net is what turns "behavior-preserving" from an assertion +into something the green suite actually verifies across each pass. + The passes it chains: 1. =/refactor= — structural and logic cleanup on measurable metrics (complexity, diff --git a/.ai/workflows/helper-mode.org b/.ai/workflows/helper-mode.org index a6acfa7..b32d574 100644 --- a/.ai/workflows/helper-mode.org +++ b/.ai/workflows/helper-mode.org @@ -12,13 +12,14 @@ The governing fact behind every rule below: the session-context split isolates e * When to Use This Workflow -No operator trigger phrase. A helper reaches this contract one of three ways: +No operator trigger phrase. A helper reaches this contract one of two ways: - The =ai --helper= launcher routes here after the roster confirms a live agent (the deterministic path). -- Startup's roster check finds the session is not alone and routes here instead of running normal startup (the safety net for a raw =claude= launch). - An explicit "you are a helper, follow helper-mode.org" instruction (the manual fallback). -If none of those applies — the roster shows the session is alone — this is a primary session. Run normal [[file:startup.org][startup.org]], not this. +There is deliberately no third way, and the gap matters: *startup does not check the roster*. A bare =claude= launched into a project that already has a live session runs full primary startup — pulls, rsync, inbox processing — without ever reaching this file. That safety net is designed (see Status below) but unbuilt, so nothing catches a raw launch. Use =ai --helper=. + +If neither route applies, this is a primary session. Run normal [[file:startup.org][startup.org]], not this. * Identity @@ -92,10 +93,28 @@ A helper does not run normal startup. It runs a light version: When the helper's work is done: -1. Re-run the roster (=.ai/scripts/agent-roster=) to learn whether a primary is still live. +1. Re-run the roster to learn whether a primary is still live. Pass the project root explicitly — =agent-roster= defaults to =$PWD= and keeps only agents at or inside that root, so calling it from a subdirectory hides a primary sitting at the root and reports "alone": + + #+begin_src bash + root="$(git rev-parse --show-toplevel 2>/dev/null || pwd)" + if [ -x "$root/.ai/scripts/agent-roster" ]; then + "$root/.ai/scripts/agent-roster" "$root"; rc=$? + else + rc=2 + fi + echo "roster rc=$rc" + #+end_src + + Read rc as =wrap-it-up.org= Step 0 does: 1 means a primary is still live, 0 means this helper is orphaned, and 2 (or an absent script) means unavailable — which takes the same archive-only path as 1, because leaving work uncommitted is recoverable and committing under a live primary is not. 2. *Primary still live (the normal case):* finalize the Summary in the helper's own =.ai/session-context.d/<id>.org=, archive it to =.ai/sessions/YYYY-MM-DD-HH-MM-<id>-<description>.org=, and stop. Do NOT commit, push, or run hygiene — the primary's next commit picks up the archived file and any scoped edits the helper left in the tree. 3. *Orphaned helper (roster shows the helper is now alone):* the primary already exited, so the helper assumes full closing duties — the git ban lifts because the concurrency that justified it is gone. Commit and push the tree (including the helper's own edits, which would otherwise strand as a dirty tree), per the normal wrap-up flow in [[file:wrap-it-up.org][wrap-it-up.org]]. * Status -Phase 1.5 of the generic-agent-runtime spec. This contract is the canonical home; the spawn paths (=ai --helper=, startup's roster branch) and the [[file:wrap-it-up.org][wrap-it-up.org]] helper branch route here. Those wiring pieces ship behind the spec's bats-then-drills-then-pilot gate and are not yet live; until then, the manual "you are a helper" instruction is how a session adopts this contract. +Phase 1.5 of the generic-agent-runtime spec. This contract is the canonical home; the spawn paths and the [[file:wrap-it-up.org][wrap-it-up.org]] helper branch route here. + +Live now: =ai --helper <project>= (roster check, id assignment, helper opener, its own tmux window), the explicit "you are a helper" instruction, and the wrap-it-up.org Step 0 helper branch. + +Not built yet, and worth knowing because it is the gap you can fall into: *startup has no roster check*. A second session launched as a bare =claude= in a project that already has one runs full primary startup — pulls, rsync, inbox processing — with no idea another agent is live. Until that safety net exists, =ai --helper= is not merely the preferred path, it is the only one that makes a helper without being told. + +Also unbuilt: the live-helper gate that pauses a primary's file-wide hygiene passes (=todo-cleanup.el=, =lint-org.el=, =wrap-org-table.el=) while a helper is mid-edit. Data-integrity rule 1 above describes the intended behavior; nothing enforces it yet, so a primary running hygiene can still clobber a helper's just-written scoped edit. diff --git a/.ai/workflows/inbox.org b/.ai/workflows/inbox.org index 3bd9335..6faa20f 100644 --- a/.ai/workflows/inbox.org +++ b/.ai/workflows/inbox.org @@ -163,6 +163,20 @@ An org capture is usually only a few seconds of mid-finalize state, so =--wait= - *Auto inbox zero (=/loop=) cycle* → don't surface or wait further; defer the roam reconcile to the next cycle, which is itself the retry at loop cadence. The items were already filed in Phase C, so the next cycle's Phase C status-check drops the duplicates and its Phase D removes them. Note one line: "roam reconcile deferred — a capture is still open; next cycle catches it." - *Wrap-up sub-step* → don't block the wrap. Skip the roam reconcile for this run and surface one line: "Skipped roam-inbox reconcile — a live org-capture is open against it; claimed items stay and get caught next run." The items were already filed into =todo.org= in roam mode Phase C, so the next roam run's Phase C status-check drops the duplicates and its Phase D removes them — the skip self-heals. +*The roam-write lock (around the Phase D edit).* Capture-guard protects against a live *human* capture; the roam-write lock protects against a concurrent *agent* writer (a sentry inbox pass, a KB promotion) editing =~/org/roam= at the same time. Acquire it after the capture-guard clears and release it after the edit-and-trigger, so the two guards nest — capture-guard underneath, the agent lock around the write: + +#+begin_src bash +if [ -x .ai/scripts/agent-lock ]; then + .ai/scripts/agent-lock acquire roam-write --wait || { echo "roam-write busy; deferring roam reconcile" >&2; exit 1; } +fi +# capture-guard (above), then the Phase D read-modify-write of ~/org/roam/inbox.org, +# then trigger the sync — roam-sync stays the only committer: +systemctl --user start roam-sync.service +[ -x .ai/scripts/agent-lock ] && .ai/scripts/agent-lock release roam-write +#+end_src + +Degrade gracefully when =agent-lock= isn't installed (an older checkout mid-sync): the write proceeds unlocked, today's behavior. A *present* helper reporting the lock busy after its bounded wait defers the roam reconcile (the auto-loop and wrap-up paths already defer-and-retry per the fallback list above); an *absent* helper never blocks it. + * Core §6 — Priority-scheme check This gates filing whenever there are accept-and-file items. Check whether =todo.org= has a top-of-file priority scheme (an explicit legend defining =[#A]= through =[#D]= semantics and mandatory/optional tag conventions — a =* <Project> Priority Scheme= section or similar). @@ -332,7 +346,7 @@ When Craig has put the session in no-approvals mode, an accepted item may be imp 2. *Quick* — the whole implementation, including verification, is under ~15 minutes. 3. *Solo* — you can carry it end to end without a decision from Craig. Manual verification you perform yourself is fine; needing Craig to choose an option, approve a design, or resolve an ambiguity is not. -All three → implement it, verify, then commit and push at the end of that item (the Step 0 reconcile and pre-push check from =commits.md= still run). Miss any one and it doesn't self-apply: a shared-asset or convention change needs Craig's decision, so it fails *solo* and routes to the defer-and-stage park (core §2 / core §3); an oversized item fails *quick* and gets filed. +All three → implement it, verify, then commit and push at the end of that item (the Step 0 reconcile and pre-push check from the =publish= skill still run). Miss any one and it doesn't self-apply: a shared-asset or convention change needs Craig's decision, so it fails *solo* and routes to the defer-and-stage park (core §2 / core §3); an oversized item fails *quick* and gets filed. ** Replying to handoffs @@ -355,6 +369,8 @@ If either can't be satisfied — a half-done item, a failure introduced during t Reads the *global roam inbox* (=~/org/roam/inbox.org=), Craig's cross-project GTD capture: one shared file every project can see. This mode routes each roam item to the project that owns it. The current session claims only the items belonging to THIS project, files them into the project's =todo.org=, and removes them from the shared inbox. Everything it doesn't own stays. +*Allowed from any project, work included.* Tidying the shared roam inbox is housekeeping on a shared resource, not a cross-project boundary crossing and not a durable KB-node write, so the =knowledge-base.md= work-denylist doesn't gate it (a sentry inbox-zero pass mis-parked the whole inbox as a boundary crossing from the work project on 2026-07-19 — the error this note closes). Reading roam and tidying its inbox are fine from work; only promoting a durable =agents/= node stays work-denylisted. + The aspiration is inbox zero: after this mode runs, the current project's local handoff inbox has been processed (Phase A delegates to process mode) and the shared roam inbox no longer contains items explicitly owned by this project. This is distinct from the wrap-up inbox/transcript routing feature (which moves session-filed keepers between projects). This routes the shared roam capture file by ownership prefix. @@ -466,14 +482,14 @@ A recurring, *interactive* roam check. Trigger phrase: "auto inbox zero" (match ** Per cycle 1. Run roam mode's scan (Phase A local check + Phase B roam scan), read-only — no =git pull=. The capture-guard still gates any write: use =capture-guard --wait= (core §5) so a transient capture clears itself; if it's still open after the wait, *defer this cycle's roam reconcile to the next cycle* rather than surfacing — the loop cadence is the retry, and the filed items get swept next time. The rare write hands its git to =roam-sync= (roam Phase D). -2. *Nothing found* → no inbox summary. One acknowledgement line: =ran at HH:MM, nothing found=. Nothing else. The acknowledge-only-on-empty rule keeps a quiet inbox quiet. +2. *Nothing found* → no inbox summary. One heartbeat line: =inbox zero at HH:MM: nothing= (HH:MM local, from =date=) — the silent-until-signal policy, see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=. Nothing else. Keeping a quiet inbox quiet is the whole point. 3. *Items found* → summarize the found items, file them as tasks (roam Phase C), and *append them to a displayed queue* — the harness task list, via =TaskCreate= — so the queue accumulates across cycles. Then ask: "run this batch next?" - *Yes* → chain into =work-the-backlog.org= as an explicit second step after routing completes: pass it the eligibility query over the queued items (status =TODO= + =:solo:= per the scheme header, priority-ordered), =file-only= mode, paging off, cap 1. The highest-priority eligible candidate runs; the rest wait for the next tick or a later yes. - *No* → they stay queued for a later go. This mode never implements anything itself — routing ends here, and the execution loop lives in =work-the-backlog.org=, its one home. 4. *Cross-cycle dedup.* Subsequent cycles add only *newly-found* items to the same displayed queue, never re-surfacing what's already there. Dedup against the queue (the =TaskCreate= list), not against what's already been implemented — a find that was queued-but-not-yet-run must not reappear, and one already filed into =todo.org= is dropped by roam Phase C's status check. -A find is always surfaced and filed; execution happens only through the =work-the-backlog.org= chain and waits for Craig's yes. A quiet inbox produces only the timestamped acknowledgement. =auto inbox zero= is inherently in-session because its chain step waits for that yes. +A find is always surfaced and filed; execution happens only through the =work-the-backlog.org= chain and waits for Craig's yes. A quiet inbox produces only the =inbox zero at HH:MM: nothing= heartbeat. =auto inbox zero= is inherently in-session because its chain step waits for that yes. ** Fully-unattended pass (=/schedule=) — vNext, not v1 diff --git a/.ai/workflows/no-approvals.org b/.ai/workflows/no-approvals.org index 5f54b96..6b5c7fa 100644 --- a/.ai/workflows/no-approvals.org +++ b/.ai/workflows/no-approvals.org @@ -35,7 +35,7 @@ Mode resets when: The interaction gates that step the workflow back to Craig for an "OK to proceed?" check: -- The commit-message gate in =commits.md= Step 2. Print the final commit message inline, then commit immediately. No "approve / request changes / open in editor" prompt. +- The commit-message gate in the =publish= skill, Step 2. Print the final commit message inline, then commit immediately. No "approve / request changes / open in editor" prompt. - The PR-description gate. Print the final body, then create the PR. - The PR-review-reply gate. Print the final reply, then post. - "Ready to start?" / "Plan looks like X, proceed?" gates before implementation work begins. @@ -46,10 +46,10 @@ The interaction gates that step the workflow back to Craig for an "OK to proceed The engineering-discipline gates protect quality, not Craig's interaction time. They remain in force: -- =/review-code= against the staged diff before every commit. Critical and Important findings still block. Minor findings still surface. No "proceed anyway" override unless Craig has given it explicitly for this batch. +- =/review-code= against the staged diff before every commit, dispatched as an isolated adversarial reviewer per the =publish= skill's Step 1 — no-approvals removes *interaction* gates, never the isolation. Critical and Important findings still block, and the re-review loop still runs to approval. Minor findings still surface. No "proceed anyway" override unless Craig has given it explicitly for this batch. If the review can't reach approval — three rounds, a recurring finding, or a =Needs Discussion= verdict — that is a genuine question: park the item per step 4 and move to the next one rather than committing past a standing finding. - =/voice personal= on every publish artifact (commit messages, PR titles + bodies, PR review comments). The full pattern walk happens. The printed result just doesn't wait for approval. - The full test suite + lint + compile before commit (per =verification.md=). -- Fetch-and-reconcile in =commits.md= Step 0. +- Fetch-and-reconcile in the =publish= skill, Step 0. - Session Log updates per =protocols.org=. Every state-mutating turn writes to =.ai/session-context.org= before the closing message. The log is the crash-recovery anchor while Craig is away. Missing entries lose work. - Subagent review-gate cadence (=subagents.md=). Review each subagent's output before the next dispatch. - Destructive or irreversible operations per =CLAUDE.md='s "Executing actions with care": force-push, =rm -rf=, dropping a column, dropping a branch, package removal. These need explicit consent regardless of mode. No-approvals is for *interaction* gates, not destructive-action consent. @@ -70,7 +70,7 @@ For each item: - Do the work. - Update the Session Log per the rules in =protocols.org=. -- Before any commit: run =/review-code= against the staged diff. Surface Critical and Important findings inline; fix them and re-review until clean. Minor findings show but don't block. +- Before any commit: dispatch the isolated adversarial reviewer per the =publish= skill's Step 1 — never review your own staged diff inline. Surface Critical and Important findings; fix them and send the updated diff back to the *same* reviewer until it approves. Minor findings show but don't block and never earn another round. If the review can't reach approval — three rounds, a finding that recurs after being reported fixed, or a =Needs Discussion= verdict — park the item per step 4 with the standing findings and move on; don't commit past a blocking finding. - Draft the commit message. Run =/voice personal= (the skill, or walk the patterns inline if unavailable). Print the final message inline before committing so the log shows it. - Commit and push. - One-line status between items ("Task X done, on to Y.") so Craig knows what's happening when he checks back in. diff --git a/.ai/workflows/page-me.org b/.ai/workflows/page-me.org index bfa92c6..7a3b792 100644 --- a/.ai/workflows/page-me.org +++ b/.ai/workflows/page-me.org @@ -13,9 +13,17 @@ Uses the =notify= command (info type) for consistent notifications across all AI Craig says *"page me"* (or variations like "page me in 10 minutes", "page me at 3pm"). -The word "page" is the trigger for this workflow. It means: set a timed notification. +The word "page" is the trigger for this workflow. It means: set a timed notification on the *desktop* channel (=notify=). -Previously called "set-alarm" -- renamed to "page-me" for a distinctive, short trigger phrase that won't collide with common words like "remind" or "alert." +Two sibling triggers pick a different channel; the timed =at= machinery below is identical for all three, only the fired command changes: + +- *"page me"* — desktop =notify= (this workflow's default). +- *"text me"* — a Signal push to Craig's phone via =agent-text= (the away channel). +- *"text and page me"* — both, for when he might be either place. + +Scope the triggers to the reflexive "me": "page me" and "text me", not a bare "page" or "text" in prose. The full channel vocabulary lives in protocols.org "Reaching Craig". + +"page" was chosen (renamed from the old "set-alarm") for a distinctive, short trigger that won't collide with common words like "remind" or "alert". * Problem We're Solving @@ -113,18 +121,18 @@ notify info "Page" "Your message here" --persist The =--persist= flag keeps the notification on screen until manually dismissed. All page-me notifications should use =--persist= by default. -** Paging Craig's phone (away from the machine) +** Texting Craig's phone (the "text me" channel) -The timed =notify= alarm above is the desktop channel. When Craig is away from the machine (or asks to be paged "on my phone"), use the agent pager instead — a Signal push to his phone from any machine or agent runtime: +The timed =notify= alarm above is the desktop channel. When Craig says "text me" (or a run expects him away from the machine), use =agent-text= instead, a Signal push to his phone from any machine or agent runtime: #+begin_src bash -agent-page "Build finished — ready for your eyes" +agent-text "Build finished, ready for your eyes" -# Timed phone page: same at-daemon pattern, different channel -echo "agent-page 'Meeting starts in 5'" | at 3:25pm +# Timed phone message: same at-daemon pattern, different channel +echo "agent-text 'Meeting starts in 5'" | at 3:25pm #+end_src -Channel selection and the pager's mechanics live in protocols.org "Paging Craig — the agent pager". When in doubt, fire both: the desktop notification persists for whenever he returns, the phone push reaches him now. +Channel selection and the mechanics live in protocols.org "Reaching Craig". On "text and page me", fire both: the desktop notification persists for whenever he returns, the phone push reaches him now. ** Managing Alarms diff --git a/.ai/workflows/sentry.org b/.ai/workflows/sentry.org new file mode 100644 index 0000000..b25fc14 --- /dev/null +++ b/.ai/workflows/sentry.org @@ -0,0 +1,227 @@ +#+TITLE: Sentry — Overnight Hygiene Supervisor +#+AUTHOR: Craig Jennings +#+DATE: 2026-07-19 + +* Overview + +Sentry is an interval loop that keeps a project's hygiene current while Craig is away. Each cycle walks a fixed list of passes — roam pull, inbox zero, triage (no mail or messengers), todo cleanup, task audit, working-files hygiene, spec board, link integrity, git health, prep freshness, bug and refactor finding, and (opt-in) solo-task implementation — and commits each pass's writing to a throwaway daily branch. Nothing pushes. In the morning Craig reviews the branch, squash-merges what he wants, and deletes it. + +The design goal is a project that greets the morning already tidy, with every judgment call and every destructive action parked in an approval queue rather than executed unattended. Sentry does the mechanical sweeping; Craig does the deciding. + +This file is the engine. It owns the entry gates, the branch mechanics, the lock model, the per-cycle pass runner, the digest and approval queue, the skip semantics, and the stop-sentry shutdown. The =agent-lock= helper (=.ai/scripts/agent-lock=) provides the locks. The passes reuse existing workflows (=inbox.org=, =triage-intake.org=, =clean-todo.org=, =task-audit.org=) under sentry's unattended contract. + +* When to Use This Workflow + +Craig arms sentry at the end of a session, with the machine left running, to have overnight hygiene done by morning. + +Triggers: + +- "start sentry", "run sentry", "arm sentry", "sentry mode" +- "let sentry watch this overnight", "keep this tidy overnight" +- "start sentry hourly", "start sentry every <interval>" (sets the loop interval) + +Stop trigger (see Stop Sentry below): + +- "stop sentry", "stand down sentry", "sentry off" + +Sentry is deliberately *not* auto-armed. Running it in a project is a per-project grant (the =:COMMIT_AUTONOMY:= marker) plus a deliberate launch with Craig at the terminal for the entry gates. + +* Prerequisite — the autonomy ticket + +Sentry commits unattended. =commits.md= gates commits on Craig's approval, so sentry needs standing, per-project authorization to run at all. Before anything else, read the project's =.ai/notes.org= Workflow State block for: + +: :COMMIT_AUTONOMY: yes + +If the marker is absent or not =yes=, decline to start and name the marker: + +: Sentry needs ":COMMIT_AUTONOMY: yes" in .ai/notes.org Workflow State to run — it commits unattended. Add it to grant, or run the hygiene passes by hand. + +No half-running mode: a project without the grant doesn't run sentry's read-only passes either. The grant is one line away, so this is a deliberate opt-in, not a barrier. + +A second, *independent* marker gates the solo-task implementation pass (pass 12): + +: :SENTRY_MAY_IMPLEMENT: yes + +=:COMMIT_AUTONOMY:= lets sentry commit its hygiene sweeps to the branch; =:SENTRY_MAY_IMPLEMENT:= additionally lets it implement solo, decision-free backlog tasks on the branch. The split exists because the two carry different morning costs: hygiene is a two-minute merge, implemented code is a review session. A project can run hygiene-only sentry without the implement pass, and most should until sentry has quiet weeks behind it. Absent =:SENTRY_MAY_IMPLEMENT:=, pass 12 skips; sentry still runs every other pass. Requires =:COMMIT_AUTONOMY:= alongside it — implementing implies committing. + +* Entry — interactive, with Craig present + +Craig types the sentry trigger, so the first moves run with him at the terminal. Do them in order; each gate that fails stops entry until Craig answers. + +1. *Autonomy ticket* — the prerequisite above. Absent → decline and stop. + +2. *Dirty-tree gate.* =git diff --quiet HEAD= (tracked modifications only; untracked and gitignored files never block — an inbox drop or scratch file is not in-progress work). If the tracked tree is dirty, describe what's dirty and offer, inline-numbered per =interaction.md=: + + 1. Finish the job — commit the in-progress work first (recommended if it's a coherent unit) + 2. Stash it — =git stash= and start sentry on a clean tree + 3. Roll back named changes — discard specific files (names them) + + Wait for an answer. Sentry can't start unattended from a dirty state; that's the point. + +3. *Green-suite gate.* Run the project's full suite (=make test=, or the project's equivalent — detect it). Read the output. If anything is red, describe the failures and offer to investigate before arming. The loop starts only on a green baseline, because every unattended cycle measures itself against "did I break this?" and a pre-existing red poisons that check. + +4. *Prior sentry branch.* =git branch --list 'sentry/*'=. An unmerged =sentry/*= branch from a previous night means the morning review didn't happen. Surface it and offer to squash-merge or delete it now (Craig is present); don't stack a second sentry branch on the first. + +5. *Reconcile the project branch.* Fetch and fast-forward-only against upstream — the same reconcile =startup= runs: + + : git fetch --all --prune + : git rev-list --left-right --count @{u}...HEAD + + Zero-behind → continue. Behind-only and clean → =git merge --ff-only @{u}=. Diverged → surface to Craig (he's present); don't auto-resolve. + +6. *Create the daily branch.* From HEAD: + + : git switch -c "sentry/$(date +%F)-$(uname -n)" + + The host suffix (=uname -n=) stops a same-date collision between the two daily drivers. The working tree now sits on this branch overnight — the launch hands the repo to sentry until the morning merge. Reclaiming it mid-night means stopping sentry first (see Stop Sentry). Note the Emacs buffer-revert caveat to Craig if he has the repo open: files change on disk under him overnight, so buffers want reverting after the morning merge (see =emacs.md=). + +7. *Arm the loop.* Start =/loop= at the interval (default hourly; Craig's "every <interval>" phrase overrides) with the per-cycle body being one sentry cycle (the Pass Runner below). Confirm the arming in one line: interval, branch name, project. + +* The lock model + +Two locks, both served by =.ai/scripts/agent-lock= (names only; the helper owns the paths, which live on tmpfs under =$XDG_RUNTIME_DIR/agent-locks/=, host-local and cleared on reboot). + +*Single-runner lock* (=sentry-<project>=, where =<project>= is the repo-root basename: =basename "$(git rev-parse --show-toplevel)"= — the same derivation =wrap-it-up.org='s guard uses, so the two agree on the lock name). Each cycle acquires it at cycle start and releases it at cycle end, and refreshes it between passes (the heartbeat, so a live cycle's lock never ages past one pass). If =/loop= fires again while a previous cycle still holds it, the new cycle's acquire fails and the cycle skips with one digest line — no two cycles run at once. The bounded wait is short (a few seconds); a live cycle means defer, not queue. + +*Roam-write lock* (=roam-write=). A pass that edits a file under =~/org/roam= acquires it, runs =capture-guard --wait= (the human-capture layer stays underneath), edits the working tree, triggers =systemctl --user start roam-sync.service=, and releases. The lock spans only edit-plus-trigger. Sentry never runs =git= against =~/org/roam= — roam-sync stays the repo's only committer (the 2026-06-24 one-git-owner rule). Pass 1's =pull --ff-only= is the sole, read-only exception. + +Every reclaim of a stale lock surfaces in the digest — the helper prints the reclaim note, and the cycle records it. A reclaim during a genuinely slow pass is possible, so it's never silent. + +* The Pass Runner — one contract per pass + +Each cycle, after acquiring the single-runner lock and verifying branch state (below), walks the pass list in order. Every pass follows the same four-step contract: + +1. *Probe* — a cheap existence check for the pass's target (named per pass below). Absent → the pass is one skip line in the digest and nothing more. This is what makes the pass list portable: passes self-activate where their target exists and stay silent elsewhere, with zero per-project configuration. + +2. *Work* — run the pass under the unattended contract. Quick, solo, already-agreed mechanical actions execute. Anything destructive or requiring judgment does *not* execute — it appends to the morning-approval queue (what, why, the exact command or edit that fires on approval). A pass runs fully or not at all; there is no reduced-form pass. + +3. *Session-context entry* — a pass that does or queues work appends its digest line to the =session-context.org= Session Log (path resolved via =.ai/scripts/session-context-path=) before its commit, so a crash between them still leaves the trail. Per-pass lines for an all-quiet cycle (every pass probe-skipped or no-op) are not written one by one — the cycle collapses to a single heartbeat at cycle-end (below), so an idle cycle doesn't spray one skip line per pass. + +4. *Commit* — if the pass wrote to disk, commit it: =chore(sentry): <pass> — <what changed>=. One commit per writing pass. A probe-skip or a no-op pass writes nothing and commits nothing. + +Between passes, refresh the single-runner lock (=agent-lock refresh sentry-<project>=) — the heartbeat. + +** Branch-state verification (cycle start, before the passes) + +After acquiring the lock, confirm the cycle is safe to run: + +- *On the right branch* — HEAD is =sentry/<today>-<host>=. If the loop was armed on a prior day and crossed midnight, the branch keeps the arming date; that's fine, morning teardown handles it. If HEAD is somehow *not* a sentry branch (an interrupted stop, a manual checkout), skip the whole cycle with a digest line rather than committing onto main. +- *Clean of foreign changes* — =git diff --quiet HEAD= excluding the spine set (=session-context.org= / =session-context.d/=, resolved via =session-context-path=). Sentry's own spine writes must not trip this; a genuinely unexpected dirty tree (something outside the spine changed and wasn't committed by a prior pass) poisons the cycle — skip it with a digest line, the next cycle retries. + +* Unattended safety — skip, never degrade + +With no one at the terminal, any unsafe state makes the affected scope skip with one digest line, and the next cycle retries. Unsafe states and their scope: + +- *Unexpected dirty tree* (non-spine) → skip the whole cycle. +- *Lost or un-acquirable single-runner lock* → skip the cycle (another cycle holds it, or the helper is missing). +- *A pass's own precondition unmet* (its probe fails, or a dependency is dirty) → skip that pass only. +- *Red suite at cycle-end* (see below) → the commits stay on the branch, flagged in the digest for morning review; the cycle doesn't roll back. + +Skips are never silent and never partial. Inside a *working* cycle, a pass line means the pass fully ran and a skip line names why it didn't. An *all-quiet* cycle is not a silent skip either: its single =sentry at HH:MM: nothing= heartbeat is the explicit record that every pass found nothing, standing in for a wall of identical skip lines. The anti-silence rule targets a pass that hides work it should have surfaced; a quiet cycle has surfaced that there was none. + +** Multi-day stall notification + +An unmerged prior =sentry/*= branch at cycle start (the morning review never happened) skips the cycle. After the *second consecutive* cycle skipped for this reason, send one persistent desktop notification naming the project and branch: + +: sentry stalled: <branch> unmerged — merge or delete to resume + +Then repeat at most daily. Persistent notify matches the paging convention — it stays on screen until dismissed. A multi-day stall never stays silent. + +* The pass list (v1) + +In order. Each names its detection probe. A pass whose probe fails is one skip line. + +1. *Roam pull* — =git -C ~/org/roam pull --ff-only=. Probe: =~/org/roam= is a git clone. Skipped when the roam tree is dirty (roam-sync owns that case) or the clone is absent. Read-only and ff-only — the one narrow exception to "don't touch roam git," so later passes read a fresh tree. + +2. *Inbox zero* — run =inbox.org= roam mode under the no-approvals contract: quick+solo+agreed items execute, shared-asset and convention proposals park (prepared diff, =VERIFY= task, sender reply) in the approval queue. Edits to =~/org/roam/inbox.org= take the roam-write lock + =capture-guard=. Probe: the roam clone or a project =inbox/= exists. Tidying the shared roam inbox is allowed from *any* project session, work included — it's housekeeping on a shared resource, not a durable KB-node write, so the work-denylist doesn't gate it (=knowledge-base.md=). Never park it as a cross-project boundary crossing. + +3. *Triage intake — mail and messenger sources excluded.* Run =triage-intake.org=, loading only its non-mail, non-messenger source plugins (calendar, PR/ticketing). The mail and messenger plugins — cmail, any Gmail variant, Telegram, Signal, chat DMs — are never loaded by a sentry cycle: Craig ruled 2026-07-21 that sentry doesn't check email or messengers. A manual "triage intake" still scans everything. Probe: the project has at least one *active* triage source that survives that exclusion — a project-specific plugin (=.ai/project-workflows/triage-intake.*.org=), or a non-empty =:TRIAGE_SOURCES:= declaration naming general plugins that exist. Mere presence of the template-synced general plugins does *not* activate the pass; a project that declares no sources, or whose only declared sources are mail or messengers, probe-skips (see =docs/specs/2026-07-20-triage-source-activation-spec.org=). Destructive actions (deleting, archiving, sending) queue; they never cycle unattended. + +4. *Todo cleanup* — the =clean-todo.org= mechanics (hygiene pass + =--archive-done= + =--convert-subtasks=). Probe: a root =todo.org=. Note that =--archive-done= is not purely an org-file pass on its first run in a project: it creates =archive/task-archive.org= and appends a =.gitignore= entry, so it produces a real tracked-file commit and correctly trips the cycle-end conditional suite. (archangel, first live run 2026-07-21.) + +5. *Task audit* — the *mechanical subset* of =task-audit.org= hourly (staleness counts, structural checks, cookie recomputation); the judgment half (priority regrades, consolidations, merge candidates) runs *once per night* and queues its findings rather than repeating them every cycle. Probe: a root =todo.org=. A full audit every hour is too heavy and re-surfaces the same judgment calls all night. (takuzu, first live run 2026-07-21.) Factual staleness fixes that are unambiguous still execute. + +6. *Working-files hygiene* — flag =working/<slug>/= directories whose backing task is closed (a filing candidate per =working-files.md=). Probe: a =working/= directory exists. The filing itself queues (it's a judgment move). + +7. *Spec status board* — the =docs-lifecycle= grep for spec keywords, surfacing any =DOING= spec whose bound build parent is closed. Probe: =docs/specs/= exists. + +8. *Link integrity* — broken =file:= links in the project's org files, via =lint-org.el=. Probe: =lint-org.el= present. Report-only into the digest; no unattended rewrites. + +9. *Git health* — uncommitted drift, unpushed commits on other branches, stale branches, main-behind-origin. Probe: =.git=. Report into the digest. + +10. *Prep + symlink freshness* — stale daily-prep docs, broken symlinks. Probe: the prep dir / symlinks exist (work and home only, in practice). + +11. *Bug and refactor finding* — hunt for real bugs and worthwhile refactoring opportunities in the project's codebase: static analysis (=shellcheck= for shell, the project's own linters for its languages), config sanity checks, plus one targeted code-reading area per cycle. Rotate the area across cycles and name it in the digest, so coverage accumulates over a night instead of re-reading the same corner. Randomized property sweeps (generate-and-verify against an engine's own invariants) are good quiet-cycle work here, reaching past a frozen test corpus. Expect the pass to go honestly quiet after the first few cycles find the standing defects; a quiet hunt is a result, not a failure. (takuzu, first live run 2026-07-21: three real fixes in the first four cycles, then quiet.) This pass does *not* run the test suite — the entry baseline already ran it, and re-running it hourly is anti-pattern 5; read the entry result instead. Probe: the project carries a codebase — source under version control beyond its org and tooling files. File each verified bug as a graded task in =todo.org= per the severity × frequency matrix (=todo-format.md=), and each refactoring opportunity as a =:refactor:= task, deduped against existing tasks; an unverifiable suspicion is a digest line, not a task. *Find, never fix in this pass* — the finding files a task and stops. A fix happens only in the opt-in implementation pass below, and only after the finding is a filed task that pass then re-verifies from scratch (see the premise rule there). A freshly-found "bug" can be a misread — one was filed and retracted two cycles apart on 2026-07-23 — so the file-then-verify-then-fix pipeline is deliberate: the task is the checkpoint, not a same-breath fix. (Added at Craig's order 2026-07-21, first dogfooded in dotfiles; refactor-finding added 2026-07-24.) + +12. *Solo-task implementation (opt-in — =:SENTRY_MAY_IMPLEMENT:=)* — work the backlog's solo, decision-free tasks on the branch. Probe: =.ai/notes.org= Workflow State carries =:SENTRY_MAY_IMPLEMENT: yes= *and* the project holds =:COMMIT_AUTONOMY:= (the implement pass commits). Absent the marker, skip — this pass is off by default, because it turns the morning from a two-minute merge into a code review, and that's the project owner's call. When on: invoke =work-the-backlog.org= under its unattended-loop contract (no pre-flight Q&A — there's no Craig overnight), eligibility =TODO= + =:solo:=, with the defer checklist deciding act-vs-file. The overnight-only tightening: only the *ready* bucket implements (clears every checklist item with zero open decisions); a task needing even one quick decision defers to a =VERIFY= rather than guessing, exactly as the loop caller already does. Commit each logical change to the sentry branch; *never push* — the morning review and merge is the gate, same as every other pass. The full quality bar holds (TDD, suite green before each commit, the isolated adversarial review per =publish= Step 1 with its re-review loop, =/voice=), and the review here runs the *premise check first*: reproduce the bug or confirm the problem is real before judging the diff. The review is the fact-checker that a filed claim never got, and it is what makes fixing-on-a-branch safe (Craig, 2026-07-24). A task that fails its premise check is not implemented — the finding was wrong, and that outcome is a digest line, not a commit. A task whose review never reaches approval — three rounds, a recurring finding, or a =Needs Discussion= verdict — is the same shape: no commit, and a digest line naming the standing findings, so the morning review sees what the reviewer would not pass rather than finding the task silently absent. (Added at Craig's direction 2026-07-24: overnight implement-on-branch, gated and never-pushed.) + +(KB lesson promotion — the pass the original proposal listed eleventh — is deferred to vNext. An unattended judgment pass writing to the shared knowledge base waits until sentry has quiet weeks behind it and a designed detection heuristic. See the filed lesson-detection-heuristic task.) + +* Cycle-end — conditional suite, then the digest commit + +After the passes: + +1. *Conditional suite run.* If any pass this cycle modified files *outside* the org/spine set (a code-touching pass, rare but possible via fixtures), run the full suite once. A green run confirms the cycle's commits are safe; a red run flags the digest for morning review — the commits stay on the branch (nothing is pushed, so the morning gate catches it). No per-pass suite runs: the entry run is the green baseline, and hourly per-commit runs would turn a seconds-long cycle into minutes all night. Cycles that only touched org/spine files skip this. + +2. *Heartbeat or digest, then commit.* Decide quiet vs working. A *quiet* cycle — every pass probe-skipped or no-op, nothing added to the approval queue — writes a single heartbeat line to the Session Log, =sentry at HH:MM: nothing= (HH:MM local, from =date=), and no per-pass digest block. A *working* cycle — any pass ran, wrote, or queued — writes its full per-pass digest block. Then commit any accumulated spine writes in one sweep: =chore(sentry): digest — <date> <time> cycle= for a working cycle, =chore(sentry): heartbeat — <date> <time>= for a quiet one, so even a quiet cycle leaves a clean tree for the next branch-state check (where the spine is untracked, the mirror-only case, there is nothing to commit and the heartbeat line stays in the working-tree anchor). This is the silent-until-signal policy (see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=): an all-quiet night collapses from a wall of no-op digests to a list of one-line heartbeats, while a cycle that actually did or queued something still writes the full record. + +3. *Release the single-runner lock.* + +* The digest and the approval queue + +*Digest.* A *working* cycle appends its block to the =session-context.org= Session Log (the spine the cycle already writes), so it survives a crash, rides the session archive, and is on screen in the running session. One block per working cycle: the timestamp, then one line per pass (ran + what, or skipped + why), plus any lock reclaim notes. A *quiet* cycle (nothing done or queued) writes no block — just the one heartbeat line =sentry at HH:MM: nothing= (the silent-until-signal policy). The per-pass block is a working-cycle artifact; it still carries one line per pass so a real skip inside a working cycle is never hidden. + +*Approval queue.* Destructive and judgment actions accumulate under one heading in the same file — =* Sentry approval queue (<date>)= — newest last. Each item carries three things: *what* (the action), *why* (what triggered it), and the *exact command or edit* that fires on approval. The morning review is Craig reading this heading top to bottom and running or discarding each item. + +* Morning teardown — Craig's, documented not automated + +Sentry never merges its own branch. In the morning Craig: + +1. Reviews the digest and the approval queue in =session-context.org=. +2. Runs or discards each approval-queue item. +3. Reviews the branch: =git log main..sentry/<date>-<host>= and the diff. +4. Squash-merges what he wants (=git switch main && git merge --squash sentry/<date>-<host>=, then one clean commit) or cherry-picks selectively. +5. Deletes the branch: =git branch -D sentry/<date>-<host>=. +6. Reverts any Emacs buffers still showing the pre-merge on-disk state (=emacs.md= buffer-revert caveat). + +A bad night is discarded by deleting one branch — nothing reached main, nothing was pushed. + +In a project that gitignores =.ai/=, the whole spine is untracked, so quiet cycles produce no commits at all and =git log main..sentry/<date>-<host>= understates the night's activity. There the anchor's heartbeat list is the only record of what fired. Read the anchor, not just the log. (archangel, first live run 2026-07-21.) + +* Stop Sentry + +Trigger: "stop sentry" (and synonyms above). Sentry owns its own shutdown: + +1. *Cancel the loop* — stop the =/loop= (=ScheduleWakeup= stop / the loop's stop path). No further cycles. +2. *Release the single-runner lock* if this context holds it. +3. *Branch disposition* — offer, inline-numbered: + 1. Squash-merge the day's branch into main now (walk the morning teardown steps 3-5 interactively) + 2. Leave it named for later review (=sentry/<date>-<host>= stays; review at leisure) +4. *Approval queue* — offer to walk the queued items now, or carry them (they stay under the heading for whenever Craig reviews). + +Stopping sentry is the only way to reclaim the working tree mid-night. The entry gate fronts the handoff; stop-sentry ends it. + +* Wrap-up interaction + +=wrap-it-up.org= refuses while sentry is live: it detects the single-runner lock (=agent-lock status sentry-<project>= → held) and stops with "sentry is active — say 'stop sentry' first." The shutdown logic lives here, not in wrap-up; wrap-up carries only the one guard. + +* Common Mistakes + +1. *Running without the =:COMMIT_AUTONOMY:= grant* — sentry commits unattended; the marker is the entry ticket, and its absence is a hard stop, not a degrade. +2. *Starting from a dirty or red tree* — the entry gates exist because an unattended cycle can't tell Craig's in-progress work from a regression. Answer the gate; don't bypass it. +3. *Committing onto main* — every writing pass commits to the daily =sentry/*= branch. A cycle that finds HEAD off the sentry branch skips rather than commits. +4. *Running a =git= write against =~/org/roam=* — roam-sync is the only committer. Sentry edits the tree under the roam-write lock and triggers the sync; it never commits or pushes roam. +5. *A per-pass suite run* — the suite runs at entry (baseline) and conditionally at cycle-end (only when a pass touched non-org files). Hourly per-commit runs all night is the anti-pattern the suite policy exists to prevent. +6. *Executing a judgment or destructive action unattended* — those queue for the morning with their exact command. The pass did its detection; Craig makes the call. The one sanctioned exception is pass 12's solo-task implementation, and only because it inherits work-the-backlog's full defer checklist (data-loss and irreversible actions defer, never execute) plus a premise-verifying review, and it commits to the branch rather than acting on anything live. +7. *A silent skip* — inside a working cycle, every skip writes a digest line naming why; a missing pass with no line reads as "ran clean" when it didn't. The one exception is not a violation: an all-quiet cycle collapses to a single =sentry at HH:MM: nothing= heartbeat instead of one skip line per pass — the heartbeat is the explicit "nothing to do" record, per the silent-until-signal policy. +8. *Degrading a pass to a reduced form* — a pass runs fully or skips. No half-passes. +9. *Letting an unmerged branch stall silently* — after two consecutive unmerged-branch skips, the persistent desktop notify cycles. Don't suppress it. +10. *Merging sentry's branch automatically* — the morning teardown is Craig's. Sentry creates and commits; it never merges or deletes its own branch. + +* Living Document + +Sentry ships with eleven finding/hygiene passes, one opt-in implementation pass, and a deferred KB pass. The pass list, the interval default, the =:SENTRY_MAY_IMPLEMENT:= default, and the queue-vs-execute line for each pass are the knobs most likely to move with dogfooding. The implement pass especially is new (2026-07-24) and unproven at scale — watch the corrections signal (work-the-backlog's metric for autonomous commits later reverted or hand-fixed) before widening it past the projects that opt in. Fold in what the live trial surfaces — a pass that queues too eagerly, a probe that misfires, a digest line that wants more detail. Refine as the signal arrives. + +* History + +Built 2026-07-19 from the sentry spec (=docs/specs/2026-07-14-sentry-workflow-spec.org=, ID f6c51f27-d7a2-4b63-9ff9-5ba005a66dfb) — 10 decisions and 12 review findings resolved before the build. Phase 1 shipped the =agent-lock= helper (commit =a8b6cf4=); this file is Phase 2, the engine. Phase 3 reconciles the roam writers (=inbox.org=, =knowledge-base.md=) to acquire the roam-write lock and adds the =wrap-it-up.org= guard. diff --git a/.ai/workflows/startup.org b/.ai/workflows/startup.org index 929d482..2262eea 100644 --- a/.ai/workflows/startup.org +++ b/.ai/workflows/startup.org @@ -29,10 +29,16 @@ Inside a rulesets session, the project-repo refresh below covers this — the ru #+begin_src bash rs="$HOME/code/rulesets" if [ -d "$rs/.git" ]; then - if (cd "$rs" && git diff --quiet --ignore-submodules HEAD -- 2>/dev/null); then + gate="$rs/claude-templates/bin/git-worktree-gate" + if [ -x "$gate" ] && "$gate" sync-safe "$rs"; then + (cd "$rs" && git pull --ff-only origin main 2>&1) | tail -3 + elif [ ! -x "$gate" ] \ + && (cd "$rs" && git diff --quiet --ignore-submodules HEAD -- 2>/dev/null); then + # Bootstrap fallback for a checkout old enough not to have the shared + # gate yet. The pull that follows installs it for subsequent starts. (cd "$rs" && git pull --ff-only origin main 2>&1) | tail -3 else - echo "rulesets: dirty working tree — using as-is, skipping pull" + echo "rulesets: changes beyond untracked inbox deliveries — using as-is, skipping pull" fi else echo "rulesets: not a git checkout — skipping" @@ -40,11 +46,11 @@ fi #+end_src Behavior: -- *Clean working tree* → fast-forward pull. =git pull --ff-only= refuses any merge or rebase, so the operation is either a no-op (already current) or a clean advance. -- *Dirty working tree* → skip the pull. Don't auto-stash and don't auto-merge — those would either lose work or invite conflicts at the worst possible moment (session start). +- *Clean working tree, or untracked deliveries only beneath =inbox/=* → fast-forward pull. =git pull --ff-only= refuses any merge or rebase, so the operation is either a no-op (already current) or a clean advance. Inbox files are queue input, not source-tree work, and do not block other projects from receiving rulesets updates. +- *Any staged or tracked change, dirty submodule, Git operation in progress, or untracked file outside =inbox/=* → skip the pull. Don't auto-stash and don't auto-merge — those would either lose work or invite conflicts at the worst possible moment (session start). - *Non-fast-forward history* → =--ff-only= aborts with an error. Surface that to the user; the rsync still proceeds against the working tree as-is. -*Template-freshness policy (applies to every dirty-check in the synced workflows).* "Dirty" means *tracked modifications only*. Untracked and gitignored files — an inbox drop, a file left in the tree to read, scratch output — never block a template pull, a fast-forward, or a monitoring gate. Projects were falling behind on templates because somebody sent them a task; that's the failure this policy closes. The checks here already comply (=git diff --quiet HEAD= sees only tracked changes; the ff gate uses =--untracked-files=no=), and any dirty-check added to a synced workflow follows the same rule. One deliberate exception: the rsync WIP-guard below counts untracked files *within rulesets' own synced source paths*, because an untracked half-written template is exactly the WIP it exists to hold back — that guard is about rulesets' outbound content, not the consuming project's local state. +*Template-freshness policy (applies to every dirty-check in the synced workflows).* The shared =git-worktree-gate sync-safe= policy is the source of truth: untracked files beneath =inbox/= and gitignored files do not block a pull, fast-forward, or monitoring gate; every other staged, tracked, untracked, submodule, or in-progress-operation state does. Projects must not fall behind merely because somebody sent them a task, but an arbitrary scratch file is not silently treated as safe. One deliberate exception remains: the rsync WIP-guard below is narrower than the repository gate and counts untracked files within rulesets' own synced source paths, because an untracked half-written template is exactly the WIP it exists to hold back. *** Install rulesets symlinks into ~/.claude (idempotent) @@ -74,8 +80,11 @@ if [ -d .git ]; then current=$(git symbolic-ref --short HEAD 2>/dev/null) dirty=0 - if ! git diff --quiet --ignore-submodules HEAD -- 2>/dev/null \ - || [ -n "$(git status --porcelain --untracked-files=no)" ]; then + gate="$HOME/code/rulesets/claude-templates/bin/git-worktree-gate" + if [ -x "$gate" ]; then + "$gate" sync-safe "$PWD" >/dev/null 2>&1 || dirty=1 + elif ! git diff --quiet --ignore-submodules HEAD -- 2>/dev/null \ + || [ -n "$(git status --porcelain --untracked-files=no)" ]; then dirty=1 fi @@ -107,8 +116,8 @@ fi #+end_src Behavior, per branch: -- *Behind only, current branch, clean tree* → =git merge --ff-only= advances HEAD. -- *Behind only, current branch, dirty tree* → fetched but not advanced. Surface so Craig can ff manually after dealing with the dirty state. +- *Behind only, current branch, sync-safe tree* → =git merge --ff-only= advances HEAD. An untracked =inbox/= delivery is sync-safe. +- *Behind only, current branch, sync-blocking tree* → fetched but not advanced. Surface so Craig can ff manually after dealing with the reported state. - *Behind only, non-checkout branch* → =git fetch . upstream:branch= advances the ref without touching the working tree. - *Diverged* (ahead and behind) → leave alone. Surface for Craig to resolve. Don't auto-rebase or auto-merge. - *Ahead only* or *up to date* → silent no-op. @@ -170,12 +179,14 @@ These calls have no dependencies on each other. Issue them all together in one m 10. =[ -f "$HOME/org/roam/inbox.org" ] && grep -cE '^\*\* ' "$HOME/org/roam/inbox.org" || true= — count items in the roam global inbox (=~/org/roam/inbox.org=), the roam-mode startup nudge. Silent if the roam clone isn't on this machine. Phase C reads the file when the count is non-zero, splits total vs items related to this project, and surfaces the offer (see =inbox.org= roam mode). Read-only; never files at startup. 11. KB surface prep (the read + contribute startup nudges; see =docs/specs/2026-06-16-encourage-kb-contribution-spec.org=). Gated on the agent KB clone. Counts =:agent:= nodes, lists up to 5 whose content matches the current project basename (titles only; a few most-recent nodes as a fallback when nothing matches), and resolves the best-practices node path. Read-only; silent when the clone is absent. Phase C surfaces the relevant titles (consult) and the best-practices link (contribute). + The best-practices lookup matches the node's *filename*, not its content. A roam node's slug lives only in its filename, so the earlier content-grep (=rg -l 'agent-kb-best-practices'=) matched nothing and the contribute nudge silently pointed at an empty path in every project, every session, for as long as it shipped. =find= rather than a glob keeps the probe identical under bash and zsh (zsh aborts on an unmatched glob) — the same reason the spec-sort probe below uses =find=. + #+begin_src bash ra="$HOME/org/roam/agents" if [ -d "$ra" ]; then proj=$(basename "$PWD") echo "kb-total: $(rg -l '#\+filetags:.*:agent:' "$ra" 2>/dev/null | wc -l)" - echo "kb-bestpractices: $(rg -l 'agent-kb-best-practices' "$ra" 2>/dev/null | head -1)" + echo "kb-bestpractices: $(find "$ra" -maxdepth 1 -name '*agent-kb-best-practices*.org' -print -quit 2>/dev/null)" matches=$(rg -il "$proj" "$ra" 2>/dev/null | head -5) [ -z "$matches" ] && matches=$(\ls -t "$ra"/*.org 2>/dev/null | head -3) echo "kb-relevant-titles:" diff --git a/.ai/workflows/suspend.org b/.ai/workflows/suspend.org index 3691f60..166f9c9 100644 --- a/.ai/workflows/suspend.org +++ b/.ai/workflows/suspend.org @@ -23,8 +23,10 @@ straight: Refreshes the anchor in place, prompts Craig to type =/clear=, and a hook resumes the *same* logical session in a fresh context. Craig is still here. - *suspend* (this workflow) — *leave.* Captures richly into the anchor, leaves - the file in place, and Craig walks away. The next session is a cold startup - that detects the present anchor and resumes from it. + the file in place, detaches the tmux client so the session parks in the + re-attachable set, and Craig walks away. The next session is a cold startup + that detects the present anchor and resumes from it — or Craig re-attaches the + still-live session directly. - =wrap-it-up= ([[file:wrap-it-up.org][wrap-it-up.org]]) — *end.* Writes the Summary, archives the anchor into =.ai/sessions/=, commits + pushes, and runs the phrase-dependent teardown. @@ -91,7 +93,32 @@ when the Summary body is from an earlier thread. that set — but the default shared behavior is to leave the tree alone.) 4. *Leave =.ai/session-context.org= in place.* Do not archive it. 5. *Brief handoff* — one or two lines: what was captured, where the resume - pointer is, the most-active thread. End and let Craig go. + pointer is, the most-active thread. This is the last thing Craig sees before + the view detaches (Step 6), so deliver it complete. +6. *Detach the tmux client.* As the final action, detach the client viewing the + =aiv-<project>= session so it drops out of Craig's active view while staying + alive in the background. This is a DETACH, not a teardown: the session and the + agent process keep running, nothing is killed, no context is lost. + + #+begin_src bash + sess=$(tmux display-message -p '#S' 2>/dev/null) + [ -n "$sess" ] && tmux detach-client -s "$sess" + #+end_src + + Run it as the very last tool call, after the handoff text has rendered — tmux + preserves the pane, so Craig sees the full handoff when he re-attaches. Unlike + wrap-up's teardown (which must defer to a =Stop= hook because it kills the + session the agent runs in, which would cut off the valediction), detach runs + inline: it disconnects the view but leaves the agent's session alive, so + nothing is cut off. Degrade gracefully — if not inside tmux (=$TMUX= unset, no + session), skip silently and the session simply stays attached. + + Why detach on every suspend: Craig cycles his live agent sessions in Emacs + with alt-space, and rotates through everything — including re-attaching + detached ai-term sessions — with shift+alt+space. A suspended session left + attached clutters the active rotation; detaching parks it in the + re-attachable set, which is what makes suspend-and-walk-away work. Re-attach + is one keystroke (shift+alt+space) or =tmux attach -t aiv-<project>=. * What suspend does NOT do @@ -103,8 +130,12 @@ does beyond capture: - No KB / memory promotion sweep. - No Linear / board reconciliation. - No session-record archive (the file stays live). -- No teardown (the ai-term buffer + tmux session stay up). It drops no - =Stop=-hook teardown sentinel, so the wrap-teardown hook stays dormant. +- No teardown. Suspend DETACHES the tmux client (Step 6) but never kills the + session: the =aiv-<project>= session and the agent process stay alive in the + background, only the view disconnects. It drops no =Stop=-hook teardown + sentinel, so the wrap-teardown hook stays dormant. Teardown — killing the + session — is wrap-it-up's job, not suspend's; detach is the lighter move that + parks a still-live session. - No blind commit of working files (step 3). - No valediction. A suspend is a pause, not a goodbye. diff --git a/.ai/workflows/triage-intake.org b/.ai/workflows/triage-intake.org index 9e08142..55cc939 100644 --- a/.ai/workflows/triage-intake.org +++ b/.ai/workflows/triage-intake.org @@ -11,6 +11,8 @@ Think of it as the ER intake queue: every new message, invite, and PR notificati *This file is the engine.* It carries no sources of its own. Every source it scans comes from a *source plugin* — a =triage-intake.<source>.org= file the engine loads at Phase 0. The engine is source-agnostic and project-agnostic; the project- and account-specific knowledge lives entirely in the plugins. To add a source, drop a plugin file. To change one, edit its plugin. Never wire a source into this file. +*Which sources a project pulls is a per-project choice.* A *project-specific* plugin (=.ai/project-workflows/triage-intake.*.org=, never synced) is active by presence — dropping it is the declaration. A *general* plugin (=.ai/workflows/triage-intake.*.org=, template-synced into every project — personal Gmail, cmail, calendar, Telegram, GitHub PRs) is active only when the project names its basename in a =:TRIAGE_SOURCES:= line in =.ai/notes.org= Workflow State (space-separated basenames, e.g. =:TRIAGE_SOURCES: personal-gmail cmail=). A project that declares nothing and owns no project plugin pulls nothing. This is the Phase 0 activation gate — presence is capability, the declaration is activation (see =docs/specs/2026-07-20-triage-source-activation-spec.org=). + Distinct from =daily-prep.org=: - *daily-prep* — heavier, once daily, builds the day's plan + standup brief + meeting prep + time blocks. - *triage-intake* — fast, repeatable, just answers "what's new since last check?" @@ -37,6 +39,8 @@ Typical timing: Do *not* use when running daily-prep — daily-prep already does this as Phase 3. +Also runs unattended as sentry's triage pass (=sentry.org=, pass 3): sentry invokes this engine under its no-approvals contract, where destructive actions (deleting, archiving, sending) queue for the morning-approval review instead of firing. The trigger phrases above are unchanged — a manual "triage intake" always routes here directly. + * Execution @@ -56,16 +60,18 @@ ls .ai/workflows/triage-intake.*.org .ai/project-workflows/triage-intake.*.org 2 The glob exclude is automatic: =triage-intake.*.org= matches the plugins but not this engine file (=triage-intake.org= has no second dot-segment), so the engine never loads itself. After globbing, for each plugin file: -1. Read it. -2. Evaluate its =ENABLED= precondition. If false, *announce the skip with its reason* ("skipping linear — mcp__linear not present") and move on. -3. The surviving set is the source list for Phases A-D. +1. *Activation gate.* A *general* plugin (from =.ai/workflows/=, template-synced into every project) is active only if its basename appears in the project's =:TRIAGE_SOURCES:= declaration (=.ai/notes.org= Workflow State — a space-separated list of source basenames). If it isn't declared, it is *inactive*: announce it ("inactive: personal-gmail — not in :TRIAGE_SOURCES:") and skip it. A *project-specific* plugin (from =.ai/project-workflows/=, never synced) is always active — dropping it there is itself the per-project declaration. This is what stops the synced general plugins from self-activating in projects that aren't triage targets: presence is capability, the declaration is activation (see =docs/specs/2026-07-20-triage-source-activation-spec.org=). An absent or empty =:TRIAGE_SOURCES:= means no general sources are active; a project with no declaration and no project plugin has no active sources, so triage no-ops there. +2. Read it. +3. Evaluate its =ENABLED= precondition. If false, *announce the skip with its reason* ("skipping linear — mcp__linear not present") and move on. +4. The surviving set — active and enabled — is the source list for Phases A-D. -*Announce the loaded set before scanning* so the omission can't hide: +*Announce the loaded set before scanning* so the omission can't hide — inactive (undeclared) plugins are named too, so a general plugin left out of =:TRIAGE_SOURCES:= is a visible choice, not a silent drop: #+begin_example -Loaded 5 source plugins: - general: personal-gmail, personal-calendar, cmail, github-prs +Loaded 2 source plugins (:TRIAGE_SOURCES: personal-gmail cmail): + general: personal-gmail, cmail project: deepsat-gmail + inactive (undeclared): personal-calendar, github-prs, telegram skipped: linear (mcp__linear not present) #+end_example @@ -203,6 +209,22 @@ Auto mode runs as a =/loop= in the *live session*, not a detached cron job: Running in the live session means MCP auth (Slack, Gmail, Linear) is inherited from the session — the headless-auth wall that blocks a detached cron run does not apply. A durable cross-session schedule is out of scope here; that belongs to the morning-ops orchestrator, which can later invoke auto mode's accumulate behavior as its triage limb. The close/stop commands below require a live session by design. +*** Phone delivery — push each signal sweep via =agent-text= + +Auto mode exists for when Craig is away from the desk, so a sweep that surfaces something worth seeing is delivered to his phone, not just printed into a session he isn't watching. After a sweep that renders the full three sections — one with real deltas or an unacked-list change (see "End-of-sweep output" below) — send that same output to his phone over Signal with =agent-text=: + +#+begin_src bash +agent-text "$SWEEP_SUMMARY" +#+end_src + +The pushed text is the *fuller* three-section shape, not a terse one-liner: the per-source deltas, the responses-awaiting-acknowledgment list, and the timestamp, led by a ⚠ SCAN FAILED banner if any source failed. + +*Signal-only — never on a quiet sweep.* An empty sweep (the =triage intake at HH:MM: nothing= heartbeat) does *not* push to the phone. Silent-until-signal (see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=) governs the phone channel too, so the phone stays silent until a sweep has real signal. The in-session heartbeat still prints as proof the loop ran; the phone is reserved for something that actually needs Craig. (Craig's ruling, 2026-07-20: the higher-cost channel doesn't buzz with "nothing.") + +If =agent-text= isn't on =PATH=, fall back to inline delivery and say so once. + +*Reply polling is deferred.* The send half ships here; polling the phone for Craig's replies (the =phone-recv= half of the retired ntfy design) waits on the reply-correlation follow-up. With the Signal account linked on more than one device, a reply fans out to every device and neither knows which page it answers — that has to be resolved before auto mode reads replies back. Until then auto mode pushes but does not poll, and Craig acts on a pushed summary from wherever he picks it up. + ** Preconditions and Close-out Auto mode borrows the inbox monitor-mode gates (=inbox.org= monitor mode): do not start on a dirty worktree or a red test suite — a close's batch commit would otherwise sweep up unrelated changes — and leave the tree clean and green when the loop stops. Surface a blocker with inline numbered options per =interaction.md= and wait. @@ -218,11 +240,13 @@ Each sweep runs Phase 0 (load *both* plugin dirs — the loud requirement still - DOES update an active daily-prep in Update mode and re-open it on change (per =daily-prep.org=). - DOES report, deltas-only, with loud scan-failure banners (Phase C rules unchanged). -** End-of-sweep output — three sections +** End-of-sweep output — three sections, or one heartbeat + +*Silent-until-signal (see =docs/specs/2026-07-20-silent-until-signal-monitors-spec.org=).* An *empty sweep* — no deltas since the previous sweep and no change to the awaiting-acknowledgment list — collapses to a single heartbeat line and nothing else: =triage intake at HH:MM: nothing= (HH:MM local, from =date=). Detection still runs in full (Phase 0 plus the A-D scan, against the session's inherited MCP auth); only the output collapses, so a long unattended run stops filling the session with identical "no changes" blocks. A sweep with real deltas or an unacked-list change prints the full three sections below, and — when away — pushes them to Craig's phone via =agent-text= (see "Phone delivery" above). The empty-sweep heartbeat is never pushed. -1. *Deltas* — what changed since the *previous sweep* (the standard Phase C summary scoped to the inter-sweep delta; one line if nothing: "HH:MM sweep: no changes"). +1. *Deltas* — what changed since the *previous sweep* (the standard Phase C summary scoped to the inter-sweep delta). 2. *Responses awaiting your acknowledgment* — every Slack reply, email, or message directed at Craig that he hasn't acknowledged or had the agent answer. A *running list carried forward across sweeps* until Craig acks each item or closes the triage. An away user's first need is "who's waiting to hear back from me," which a delta-only sweep loses the moment it scrolls past. -3. *Timestamp* — the current date, time, and timezone on the sweep's own final line, so an away reader sees how fresh the summary is without computing it. Print it on *every* sweep, including a quiet "no changes" one — on a quiet sweep the stamp is the proof the loop ran. Generate it with: +3. *Timestamp* — the current date, time, and timezone on the sweep's own final line, so an away reader sees how fresh the summary is without computing it. Print it on every sweep that prints these three sections. On an *empty* sweep there is no separate timestamp line — the heartbeat (=triage intake at HH:MM: nothing=) is itself the freshness stamp and the proof the loop ran. Generate it with: #+begin_src bash date "+%A %Y-%m-%d %H:%M:%S %Z (%z)" @@ -416,6 +440,9 @@ Update the engine as the orchestration pattern evolves; update a plugin as its s *** Updates and Learnings +**** 2026-07-20: Phone delivery for signal sweeps (=agent-text=, send half) +Auto mode now pushes a full-three-section sweep to Craig's phone over Signal via =agent-text=, the away-from-desk delivery the retired ntfy design carried before ntfy was torn down (2026-07-04). Transport is =agent-text= (the renamed Signal pager), not ntfy. Signal-only by Craig's 2026-07-20 ruling: a quiet sweep's =nothing= heartbeat never reaches the phone — silent-until-signal governs the phone channel too, so the higher-cost channel only fires when a sweep has real signal, while the in-session heartbeat stays as proof the loop ran. Falls back to inline when =agent-text= is absent. Only the send half ships; reply polling (the old =phone-recv=) waits on the reply-correlation follow-up, because a Signal reply fans out to every linked device and neither knows which page it answers. + **** 2026-07-18: Three-section digest + close-by-default (Phase C/D rewrite) Craig's ruling after a 42h-gap sweep where the long-form report (top signals + per-source breakdown + 7-option action menu) was followed by "summarize the notable items" — and the digest that answered it was the report he wanted first. Phase C now renders ==TASKS== (work items needing Craig, solo-executable first, priority order) / ==FYI== (work context, no action owed) / ==MISC== (everything outside the project, actions stated inline), then exactly two options (close-and-file / close-and-execute-solo, the latter only when solo items exist), timestamp last. Per-source blocks and the itemized action menu are gone from the default surface (long form on request). Phase D became the close: it runs as the next action no matter what Craig replies (unless he explicitly holds), includes the mail hygiene on every scanned account without itemized confirmation, files the TASKS, clears resolved unacked items, advances the sentinel, and tears down started services. Solo = mechanical + standing-approved + no prose under Craig's name; prose sends and destructive non-mail actions stay gated. The stay-open-until-confirmed exit loop is retired. Same-day addendum: the "and reroute" modifier ("1 and reroute") — MISC items are surfaced-only by default (never filed to this project's todo.org); appending the modifier delivers each outside-project item to its owner's inbox via inbox-send per the cross-project rule. diff --git a/.ai/workflows/triage-intake.telegram.org b/.ai/workflows/triage-intake.telegram.org index 5039a8b..1319da5 100644 --- a/.ai/workflows/triage-intake.telegram.org +++ b/.ai/workflows/triage-intake.telegram.org @@ -30,12 +30,27 @@ Telega does not autostart with the Emacs daemon. "Down" is its normal state unless Craig has Telegram open in Emacs. The scan therefore runs the full lifecycle every time, never skips because the server is down: +⚠ *DOWN / not-loaded is the TRIGGER to launch, never a reason to skip or fail.* +This is the exact mistake two projects (work + home, 2026-07-24) made: they +probed telega, saw =(telega-server-live-p)= nil or telega not =featurep=, and +reported =SCAN FAILED: telegram — not loaded= or a silent SKIP — a *blind* +sweep — instead of running Step 1 to start it. A down or unloaded telega is the +normal entry state; =(telega t)= both LOADS the package and STARTS the docker +server (work confirmed: down → =(telega t)= → Ready, 18 chats). So the plugin +MUST run Step 1's launch whenever telega is down/unloaded, wait for Ready, then +scan. =SCAN FAILED= is reserved for a launch that was actually ATTEMPTED and did +not reach Ready (image missing, server crash on start, daemon unreachable) — +never for the pre-launch down state itself. The =:ENABLED:= guard above tests +whether telega is INSTALLED (=fboundp=), not whether the server is up; a down +server never disables the source. + 1. Record prior state: TELEGA_WAS_RUNNING via (telega-server-live-p). 2. Launch (only if not running): emacsclient -e "(progn (setq telega-use-docker t) (telega t) 'started)" - The setq is mandatory defense: tdlib segfaults outside docker mode - (2026-06-09), and Craig's daemon currently has telega-use-docker nil. - Wait ~2s for Ready, then (telega--loadChats 'main) until telega--chats + The setq is mandatory defense: tdlib crashed in native mode when this was + set up (2026-06-09) — a separate matter from the SEGFAULT gotcha, which is + about the loadChats argument — and Craig's daemon defaults to nil. + Wait ~2s for Ready, then (telega--loadChats '(:@type "chatListMain")) until telega--chats is populated. 3. Check messages: the maphash unread scan in ** Scan Step 2 (filters the messageContactRegistered join-notice noise). @@ -48,10 +63,13 @@ lifecycle every time, never skips because the server is down: Verify: telega-server-live-p → nil, no zevlg/telega-server container in docker ps. If Craig had it running, leave it untouched. -If any lifecycle step fails (docker image missing, server crash, daemon -unreachable), the sweep reports it as SCAN FAILED at the top of the summary -per the engine's failure rule — never as a silent skip. Craig gets real -traffic here. +If any lifecycle step fails *after the launch was attempted* (docker image +missing, server crash on start, daemon unreachable, Ready never reached), the +sweep reports it as SCAN FAILED at the top of the summary per the engine's +failure rule — never as a silent skip. This does NOT cover the ordinary +pre-launch down state: a down server means "run Step 1," not "SCAN FAILED." +Craig gets real traffic here, so a blind sweep that skipped the launch is worse +than a clean failure — it hides real unread messages behind a false all-clear. ** Scan @@ -85,22 +103,58 @@ TELEGA_WAS_RUNNING=$(emacsclient -e "(and (fboundp 'telega-server-live-p) (teleg *** Step 1 — start (docker mode) if not already running, wait for Ready #+begin_src bash -# `(telega t)` starts without popping the root buffer. Docker mode (the stable -# path — see the SEGFAULT gotcha) reconnects the persisted ~/.telega session in -# ~2s. Then load the main chat list so telega--chats populates. +# `(telega t)` starts without popping the root buffer. Docker mode reconnects the +# persisted ~/.telega session in ~2s. Then load the main chat list so +# telega--chats populates. +# +# The `(setq telega-use-docker t)` is mandatory and must come BEFORE `(telega t)`: +# tdlib crashed in native mode when this was first set up (2026-06-09), and the +# daemon's default is nil unless something (e.g. an Emacs-config :custom) has +# already forced it. It was missing here while the Quick Reference required it — +# a session that started telega without it on a native-mode daemon would take the +# untested path. Match the Quick Reference exactly. +# +# Note this is a SEPARATE concern from the SEGFAULT gotcha below: that gotcha is +# about the `loadChats` argument, and the deaths it explains happened in docker +# mode. Docker mode is not a defense against it, and it is not evidence for +# docker mode. Keep both. emacsclient -e "(progn + (setq telega-use-docker t) (unless (and (fboundp 'telega-server-live-p) (telega-server-live-p)) (telega t)) 'started)" # Poll until Ready with chats synced, or a crash/timeout. Background this with an # until-loop so the wait doesn't block; exit on Ready-with-chats OR an abnormal # server exit. Then force a chat-list load if the hash is thin: -emacsclient -e "(progn (ignore-errors (telega--loadChats 'main)) (ignore-errors (telega--loadChats 'main)) 'loaded)" +# NOTE: the chat-list argument must be a TL object, not the symbol 'main. +# `telega--loadChats' puts it straight into the request as :chat_list, and a +# bare symbol kills the server outright (see the SEGFAULT gotcha below). +# +# The liveness check on the tail is the load's only failure signal. `ignore-errors' +# catches nothing here, because a bad argument kills the server process rather than +# signalling in elisp, so without this the call returns 'loaded either way. +# The `fboundp' guard matches Step 0: if the launch failed outright telega is not +# loaded, and that should read as 'server-died like any other failure rather than +# signalling void-function. +emacsclient -e "(progn (ignore-errors (telega--loadChats '(:@type \"chatListMain\"))) (ignore-errors (telega--loadChats '(:@type \"chatListMain\"))) (if (and (fboundp 'telega-server-live-p) (telega-server-live-p)) 'loaded 'server-died))" #+end_src On a persisted session telega reaches status "Ready" within ~2s; the chat list loads over a few more. If =(hash-table-count telega--chats)= is 0 or thin, re-issue =telega--loadChats= and poll until it stabilizes. +⚠ *=server-died= is SCAN FAILED, never a quiet account.* A server that dies +during the load leaves a thin =telega--chats= hash, and a thin hash reads exactly +like an account with little unread. That is the same false all-clear the +down/not-loaded rule exists to prevent, arriving one step later in the lifecycle. +It also fits the SCAN FAILED definition above: the launch was attempted and did +not hold. So on =server-died=, report SCAN FAILED rather than scanning, and never +report a low unread count from that run. + +This is the independent evidence the SEGFAULT gotcha asks for when it says to +treat a short chat list as a real short list. Without the check there is no way +to tell the two apart, which is how the =loadChats= crash stayed invisible +through two investigations. + *** Step 2 — read unread, classified by last-message type The single most important filter: =messageContactRegistered=. Telegram counts a @@ -157,24 +211,61 @@ stays non-nil). =telega-server-kill= is what actually stops the server. Call left in =docker ps=. Skipping this whole branch when =TELEGA_WAS_RUNNING= is t is the point of Step 0: never tear down a session Craig is actively using. -⚠ *SEGFAULT GOTCHA — crashes are spontaneous; treat server death as routine.* -The dockerized =telega-server= (=zevlg/telega-server:latest=, image built -2026-06-04, tdlib 1.8.64) SIGSEGVs (exit 139) *on its own*, minutes-to-hours -into a session — 11 host coredumps between 2026-06-09 and 2026-06-11, several at -times when no triage verb was running. The 2026-06-11 investigation reproduced -the crash-free verbs and the spontaneous deaths side by side: coredump -backtraces show a corrupted stack (memory corruption in the musl build), and -no newer image exists upstream. Earlier theories — "native mode is the trigger", -"toggle-read is the trigger" — were timing coincidences; the verbs are sound. +⚠ *SEGFAULT GOTCHA — this was our bug, not tdlib's. Root-caused 2026-07-28.* +=telega-server= dies with =Unexpected char 'm' in plist value= followed by +=Assertion failed: false (telega-dat.c: tdat_plist_value: 500)=. The cause was +this workflow: Step 1 called =(telega--loadChats 'main)=. + +The chain. =telega--loadChats= is a raw TL wrapper — it drops its argument into +the request as =:chat_list= with no conversion. =telega-server--send= then +=prin1='s the whole plist, and =telega--tl-pack= passes atoms through untouched, +so the symbol goes out on the wire bare as =main=. The C parser +(=server/telega-dat.c=, =tdat_plist_value=) accepts only =(=, =[=, ="=, =-=, a +digit, =t=, =:=, or =n= to start a value. It hits =m=, prints that line, and +calls =assert(false)=, which aborts the process. The =m= in the error is +literally the first character of =main=. + +The symbol shorthand is real but belongs to a different layer: +=telega-filter.el= and =telega-folders.el= convert =(eq cl-fspec 'main)= into +='(:@type "chatListMain")=. The raw TL layer never does. telega's own callers +always pass the object (=telega.el:290=, =telega-tdlib-events.el:516=). + +Proved by experiment, not inference (2026-07-28): from a live Ready server, +=(telega--loadChats 'main)= killed it within seconds and added one coredump, +with that exact assertion; a restart plus =(telega--loadChats '(:@type +"chatListMain"))= survived three consecutive calls with no new coredump and no +assertion. + +*The previous entry here was wrong and cost real time.* It recorded the deaths +as spontaneous musl memory corruption and declared "the verbs are sound", which +sent later investigations at the docker image and tdlib versions instead of at +this file. The corrupted stack in the backtraces is what an =assert= abort looks +like, not independent evidence of a memory bug. If crashes are ever seen again +with *no* triage verb running, that is a genuinely separate cause and needs its +own investigation — do not reuse the old spontaneous-crash story to explain it. + +*This crash kills a scan; it does not silently shorten one.* An earlier draft of +this section claimed the reported "19 chats of ~50" was truncation caused by the +bad call. That was wrong, and work disproved it at the wire level on 2026-07-28: +with the corrected call their count is 19 before the first load and 19 after five +(four on =chatListMain=, one on =chatListArchive=). Nineteen is the real size of +that account. The same reading here — 19 stable across three corrected loads — +was already sitting in the evidence and should have retired the claim before it +was written down. Treat a short chat list as a real short list unless something +independently shows the server died mid-sync. + +=ignore-errors= around the call never helped — the failure is the server process +dying, not an elisp signal, so there is nothing for it to catch. That is why the +death is easy to miss from inside elisp, and why a caller should check +=(process-live-p (telega-server--proc))= after a load rather than trusting a +returned value. Operationally: docker mode stays mandatory (=telega-use-docker= = t; the setq before =(telega t)= is still the right defense), and *every action batch checks the server first* — =(process-live-p (telega-server--proc))= — restarting via -=(telega t)= when dead and re-checking Ready before firing verbs. A mid-sweep -death is recoverable, not an abort: restart, confirm Ready, resume. Durable-fix -candidates if the crashing gets worse: pin a pre-2026-06 image digest, build -=telega-server= natively against tdlib, or report upstream to zevlg with the -coredumps (=coredumpctl list /usr/bin/telega-server=). +=(telega t)= when dead and re-checking Ready before firing verbs. Any argument +handed to a =telega--*= TL wrapper must be a TL object or a plain +string/number/list, never a bare symbol. Defense in depth: even if the server does die, the scan still works because it reads the cached =telega--chats= hash, not a live query. A dead server is diff --git a/.ai/workflows/work-the-backlog.org b/.ai/workflows/work-the-backlog.org index 090841d..ea3f402 100644 --- a/.ai/workflows/work-the-backlog.org +++ b/.ai/workflows/work-the-backlog.org @@ -54,7 +54,7 @@ For the task set, in order, until the run cap is hit: 1. *Eligibility gate* (below). Ineligible → record =skipped-ineligible=, next task. 2. *Scope read* of the relevant code. Cheap; just enough to run the defer checklist. 3. *Defer checklist* (below). Any hit → defer: file the =VERIFY= naming the gap and record =deferred-VERIFY= (or, under the speedrun preset, route a quick-question gap to the pre-flight Q&A), next task. -4. *Implement* under the project's commit discipline: TDD red→green→refactor, then =/review-code --staged=, fix all Critical/Important findings, then close the task per =todo-format.md='s completion rules. Decompose into as many logical commits as the change needs — size is not capped. If implementation fails partway, leave the tree working, record =failed=, surface it, and continue to the next task. +4. *Implement* under the project's commit discipline: TDD red→green→refactor, then the isolated adversarial review (=publish= Step 1) with its re-review loop, fix all Critical/Important findings, then close the task per =todo-format.md='s completion rules. Decompose into as many logical commits as the change needs — size is not capped. If implementation fails partway, leave the tree working, record =failed=, surface it, and continue to the next task. 5. *Commit autonomy branch:* - =file-only= → surface the diff, do *not* commit. Record =implemented-diff-surfaced=. - =autonomous-commit= → =/voice personal= on the message, commit individually, push per the project's flow. Record =implemented-committed=. @@ -70,6 +70,8 @@ A task is autonomous-safe when *both* hold. This layer is a lookup, not a judgme 1. *Status is =TODO=* — never =VERIFY=, =DOING=, =DONE=, or =CANCELLED=. =VERIFY= marks "awaiting Craig's input"; auto-implementing one defeats the check it represents. The do-not-implement set is safe-by-omission: anything not plainly =TODO= (plus any project-declared "hold" marker) is out. 2. *Tagged =:solo:=* — the autonomy tag, resolved against the project's priority/tag scheme header in =todo.org= (never hardcoded). =:solo:= carries the hard definition in =todo-format.md=: completable and verifiable without Craig beyond at most one or two quick decisions answerable up front, no design deliberation. A project whose scheme declares a different autonomous-safe tag set overrides the default. +Terminology: *speedrunnable means tagged =:solo:=*. It does not mean =:quick:= or require =:quick:solo:=. The =TODO= status check above is the execution-state gate over that speedrunnable set. + Priority and =:next:= drive *ordering* within the eligible set, not eligibility ([#A] before [#B] before [#C], then the author's ordering). =:quick:= is an effort hint for batching and duration estimates — never a gate. Task *size* is deliberately absent from this gate. A large but well-specified, decision-free task is in scope and gets decomposed into per-logical-commit chunks during implementation. Size never sends a task away; only *deliberation* or *risk* does (the checklist below). @@ -80,7 +82,7 @@ Task *size* is deliberately absent from this gate. A large but well-specified, d After the scope read, run each eligible candidate through the checklist. Each item is a concrete, answerable question, not an adjective. *Any* hit — or any "unsure" — defers the task. Only a task that clears every item is implemented. -1. *Test-writability (the keystone).* Can I write the failing test from the task text — plus any decisions gathered up front — without inventing a requirement? *No / unsure* → underspecified. Under the speedrun preset, if the gap is one or two quick answerable questions, route it to the pre-flight Q&A; otherwise file a =VERIFY= noting what's missing. Under the unattended loop, file the =VERIFY= (no one to ask). +1. *Test-writability (the keystone).* Can I write the failing test from the task text — plus any decisions gathered up front — without inventing a requirement? *No / unsure* → underspecified. Under the speedrun preset, if the gap is one or two quick answerable questions, route it to the pre-flight Q&A; otherwise file a =VERIFY= noting what's missing. Under the unattended loop, file the =VERIFY= (no one to ask). *Open-ended goals are a specific, recognizable failure of this item:* a task phrased as an absence ("find bugs until none remain," "refactor until nothing worthwhile is left," "clean it up") has no writable acceptance test and so isn't really =:solo:=, even when tagged. Don't guess a stopping point — defer it and note that it needs measurable acceptance criteria (bound the surface, characterization net, dispositioned findings, objective floor — see =todo-format.md='s "Making an open-ended task measurable"). Once those are in the task body, it becomes runnable. 2. *Data-loss / irreversible / external operation.* Does implementing it require any of: =rm= of non-scratch data, =git reset --hard= / force-push, =DROP= / =DELETE= / =TRUNCATE=, file truncate/overwrite of persisted content, a schema or data migration, any external or shared-state mutation, any credential touch? *Yes* → do NOT implement; file a =VERIFY= naming the risk. This is the hard safety gate; an upfront answer never overrides it without an explicit checkpoint. 3. *Already-satisfied.* Does the scope read show the desired end-state already holds? *Yes* → file a =VERIFY= noting it and move on. Don't make a no-op change. 4. *Design deliberation.* Does the task carry an unresolved design question, a "weigh these approaches" with real tradeoffs, or a TBD that isn't a quick factual answer? *Yes* → under the speedrun preset, if it collapses to one or two quick questions, route to the pre-flight Q&A; otherwise file and surface as a =/start-work= candidate. Under the loop, file. The discriminator is *quick-answerable question* vs *deliberation* — never task size. @@ -108,7 +110,8 @@ Autonomy changes who approves, not what quality means. Per task, non-negotiable: - *TDD* per =testing.md=: red first, green, refactor. The keystone checklist item already proved the failing test is writable. - *Verification* per =verification.md=: fresh evidence, full suite green before any commit. -- *=/review-code --staged=* before every commit; Critical and Important findings block until fixed. +- *Isolated adversarial review* before every commit, dispatched per the =publish= skill's Step 1 — never an inline self-review, however small the diff. Critical and Important findings block until fixed, and each fix goes back to the *same* reviewer until it approves. Minor findings never earn another round. + - *When the review can't reach approval* — three rounds without it, a finding that recurs after being reported fixed, or a =Needs Discussion= verdict — the unattended run has no one to ask. Record the task =failed= with the standing findings in its result, leave the tree working, and continue to the next task. Never commit past a blocking finding because nobody is awake to adjudicate — an unreviewed commit landing overnight is the outcome this gate exists to prevent. - *=/voice personal=* on every commit message on the =autonomous-commit= path (or the patterns walked inline if the skill is unavailable), message printed inline so the log shows what landed. - *Task closure* per =todo-format.md=: depth-based completion (keyword + =CLOSED:= at level 2, dated rewrite at level 3+). - *One logical change per commit.* A large task becomes several commits, not one omnibus. @@ -152,7 +155,7 @@ With paging on, fire one page when the set is done or the cap is hit — end-of- notify info "Page" "<project>: <N> done, <M> remaining — <one-line summary>" --persist #+end_src -=--persist= keeps it on screen until dismissed, and =info= is the page-me urgency convention (persistent but never crash-scary). The page fires when the set completes *or* the cap stops the run — either way exactly once. The message carries the project name, the completed count, and the remaining count (with skipped tasks noted in the run summary) so Craig can confirm ready and name the next project in one reply. =notify= is the desktop paging surface; a run that expects Craig to be away also fires =agent-page= with the same message (the Signal phone channel — protocols.org "Paging Craig — the agent pager"). +=--persist= keeps it on screen until dismissed, and =info= is the notification urgency convention (persistent but never crash-scary). The notification fires when the set completes *or* the cap stops the run, either way exactly once. The message carries the project name, the completed count, and the remaining count (with skipped tasks noted in the run summary) so Craig can confirm ready and name the next project in one reply. =notify= is the desktop channel (the "page me" surface); a run that expects Craig to be away also fires =agent-text= with the same message (the Signal phone channel, "text me"). See protocols.org "Reaching Craig". * Metrics diff --git a/.ai/workflows/wrap-it-up.org b/.ai/workflows/wrap-it-up.org index 5ce88a5..ecd3d22 100644 --- a/.ai/workflows/wrap-it-up.org +++ b/.ai/workflows/wrap-it-up.org @@ -24,11 +24,13 @@ The wrap-up is complete when: 2. *File is archived.* =.ai/session-context.org= has been renamed to =.ai/sessions/YYYY-MM-DD-HH-MM-description.org=. The old path no longer exists. 3. *todo.org is clean.* Cleanup script ran. Any auto-fixes are staged for the wrap-up commit. Orphan planning lines surfaced for manual fix if there are any. 4. *Linear board is honest* (skip if project doesn't use Linear). Any Dev-Review ticket whose PR has merged was moved to Done or PM Acceptance per the classification rule. -5. *Git state is clean.* All changes committed + pushed to all remotes. Working tree clean. +5. *Git state is certified clean.* All changes are committed + pushed to all remotes, =git-worktree-gate certify= succeeded at the current HEAD, and the working tree has no staged, unstaged, untracked, submodule, or in-progress-operation state. 6. *Valediction delivered.* Brief, warm closing with key accomplishments and reminders, ending with =session wrapped.= on its own line as the signoff marker. The absence of =.ai/session-context.org= is the signal that the last session wrapped up cleanly. Its presence at session start means the previous session was interrupted. +*A helper session meets a shorter list.* Criteria 1 and 2 apply to its own context file (archived under =.ai/sessions/YYYY-MM-DD-HH-MM-<id>-<description>.org=), and 6 applies. Criteria 3, 4, and 5 do not: hygiene, the Linear pass, and all git mutation belong to the primary, so a helper that satisfied criterion 5 would have violated its contract to get there. Step 0 routes this. + * Teardown mode (set from the trigger phrase) The wrap itself — Steps 1 through 5 — is identical in every mode. The trigger phrase only decides what Step 6 does once commit + push and the valediction are done. Resolve the mode from the phrase before starting: @@ -43,8 +45,58 @@ This depends on three functions in =.emacs.d/modules/ai-term.el= (=cj/ai-term-qu * The Workflow +** Step 0: Helper branch — a helper wraps only itself + +Resolve first whether this session is a helper, because a helper's wrap is a different and much shorter workflow. Everything from Step 1 down — the hygiene passes, the inbox check, the commit, the push, the clean-tree certificate — is primary-only under the role contract in [[file:helper-mode.org][helper-mode.org]], and running any of it from a helper is exactly the concurrency failure that contract exists to prevent. + +A session is a helper when =AI_HELPER=1= in its environment (=ai --helper= sets it) or when it adopted helper-mode.org this session by instruction. If neither holds, this is a primary: skip to Step 0.5 and wrap normally. + +#+begin_src bash +echo "AI_HELPER=${AI_HELPER:-unset} AI_AGENT_ID=${AI_AGENT_ID:-unset}" +#+end_src + +For a helper, re-run the roster — the answer decides which wrap applies: + +#+begin_src bash +root="$(git rev-parse --show-toplevel 2>/dev/null || pwd)" +if [ -x "$root/.ai/scripts/agent-roster" ]; then + "$root/.ai/scripts/agent-roster" "$root"; rc=$? +else + rc=2 +fi +echo "roster rc=$rc" +#+end_src + +Pass the project root explicitly. =agent-roster= defaults to =$PWD= and keeps only agents whose cwd is at or inside that root, so running it from a subdirectory hides a primary sitting at the root — and the "alone" that produces is read below as *orphaned*, which is the one branch that commits and pushes. Capture =rc= inside the branch too: =[ -x … ] && …; echo $?= reports the status of the whole list, so an absent script reads as 1 (others live) rather than 2 (unavailable). + +- *Primary still live (rc 1)* — the normal case. Finalize the =* Summary= in the helper's own context file — same contract as Step 1, KB receipt line included (resolve it with =AI_AGENT_ID=<id> .ai/scripts/session-context-path=), archive it to =.ai/sessions/YYYY-MM-DD-HH-MM-<id>-<description>.org= so it can't collide with the primary's archive name, deliver the valediction, and stop. Do NOT commit, push, or run any hygiene pass. The helper's scoped edits stay in the tree and the primary's next commit carries them along with the archived file — say so in the valediction, so Craig knows the work is real but not yet pushed. +- *Alone (rc 0) — orphaned helper* — the primary exited first, so the git ban lifts: the concurrency that justified it is gone, and stopping here would strand the helper's edits as a dirty tree nobody owns. Run the full wrap below starting at Step 0.5, exactly as a primary would. +- *Roster unavailable (rc 2, or the script absent)* — take the archive-only path, the same as primary-still-live. Leaving work for the next session to commit is recoverable; guessing "orphaned" and committing underneath a live primary is not. + +** Step 0.5: Refuse if sentry is live + +Before anything else, check whether sentry is running in this project. Sentry holds the working tree on its =sentry/<date>-<host>= branch and commits unattended; wrapping underneath it would archive the session anchor and tear down the buffer while the loop is still firing into it. If sentry's single-runner lock is held, stop and point at the shutdown path: + +#+begin_src bash +proj="$(basename "$(git rev-parse --show-toplevel 2>/dev/null || pwd)")" +if [ -x .ai/scripts/agent-lock ] && .ai/scripts/agent-lock status "sentry-$proj" | grep -q '^held'; then + echo "sentry is active — say 'stop sentry' first" + exit 1 +fi +#+end_src + +The stop-sentry operation (defined in =sentry.org=) owns the shutdown: it cancels the loop, disposes of the branch, and walks the approval queue. Wrap-up carries only this one guard; a =stale= lock (a crashed cycle) doesn't block — only a live =held= lock does. + ** Step 1: Finalize the Summary +*** Work the Before-Close Queue (before the Summary) + +If the session anchor (=.ai/session-context.org=) carries a =* Before-Close Queue= heading with items, work them now, oldest-first, before writing the Summary, so any resulting edits ride this wrap's commit and get described in it. The queue is the "put X on the list" shorthand (see =protocols.org=, Colloquialisms and Expansions): session-scoped work Craig deferred to wrap time. + +Per item: do it if it's clear and bounded, or promote it to a =todo.org= task if it turns out to need its own session. Never drop an item silently. Remove each line as it's handled; if one can't be finished, surface it in the valediction (Step 5) and either leave a follow-up task or state why it's dropped. + +If there's no =* Before-Close Queue= heading, or it's empty, this step is a silent no-op. + *** Early KB reflection (capture while fresh, before the Summary) Before distilling the Summary, while the session is still fresh, ask: what did this session learn worth remembering, for yourself or a future agent? Reflect and stage any candidate durable facts — a decision and its why, an environment gotcha, a reference pointer, a transferable lesson. Self-answer silently; this adds no interactive turn (Craig already authorized the wrap). The candidates flow straight into the KB promotion check below, which does the actual writing and the receipt — this is the capture half, that is the commit half, one pipeline, one receipt. Reflecting here rather than reconstructing learnings after the Summary is the point: the early ask is what keeps the receipt from defaulting to "promoted 0" out of fatigue. @@ -167,6 +219,16 @@ Preview the moves without writing: emacs --batch -q -l .ai/scripts/todo-cleanup.el --archive-done --check todo.org #+end_src +*** Clear temp/ + +#+begin_src bash +[ -d temp ] && find temp -mindepth 1 -delete && echo "temp/ cleared" +#+end_src + +=temp/= holds throwaway artifacts — discarded prototypes, scratch output, intermediate data (see =working-files.md=). It's gitignored in every project, so nothing here rides a commit and nothing is recoverable from git once deleted. Clearing it at wrap is what keeps ephemeral work from silting up across sessions, and it's the counterpart to =working/=, which is tracked and *never* cleared here. + +Two guards. Confirm before deleting if =temp/= holds anything a reasonable reader would call in-progress rather than throwaway — misfiled work belongs in =working/=, so move it there instead of deleting it. And skip the step entirely in a project where =temp/= is not gitignored, since that means the project is using the directory for something else. + *** Sync child priorities #+begin_src bash @@ -241,7 +303,7 @@ For an interactive walk of the judgments mid-day, run =/lint-org todo.org=. *** Inbox sanity check (surface unprocessed handoffs) -If the project has an =inbox/= directory, verify it holds nothing but =.gitkeep=, =lint-followups.org= (the lint-org pipeline file the next morning's daily-prep consumes), and any explicitly-deferred =PROCESSED-*= files before the wrap completes. An inbox that arrived at session start with handoffs from other projects, or that received handoffs mid-session, needs =inbox.org= process mode to run and apply its value-gate dispositions. Wrapping with a dirty inbox silently defers the work to next session and accumulates handoff debt that the sender can't see. +If the project has an =inbox/= directory, verify it holds nothing but =.gitkeep=, =lint-followups.org= (the lint-org pipeline file the next morning's daily-prep consumes), and =PROCESSED-*= files before the wrap completes. An inbox that arrived at session start with handoffs from other projects, or that received handoffs mid-session, needs =inbox.org= process mode to run and apply its value-gate dispositions. Wrapping with an unprocessed inbox silently defers the work to next session and accumulates handoff debt that the sender can't see. #+begin_src bash unprocessed=$(find inbox -maxdepth 1 -type f \ @@ -250,7 +312,7 @@ unprocessed=$(find inbox -maxdepth 1 -type f \ ! -name 'PROCESSED-*' \ 2>/dev/null | wc -l) if [ "$unprocessed" -gt 0 ]; then - echo "wrap-up: inbox/ has $unprocessed unprocessed item(s). Run inbox.org process mode before wrapping, or explicitly defer each item with a one-line reason in the valediction." + echo "wrap-up blocked: inbox/ has $unprocessed unprocessed item(s). Run inbox.org process mode before wrapping." find inbox -maxdepth 1 -type f \ ! -name '.gitkeep' \ ! -name 'lint-followups.org' \ @@ -259,7 +321,7 @@ if [ "$unprocessed" -gt 0 ]; then fi #+end_src -If the count is zero or the project has no =inbox/= directory, the check is a silent no-op. If non-zero, the wrap is incomplete by default. The user resolves each item (process now, defer with reason in the valediction, or delete with rationale) before the validation checklist passes. +If the count is zero or the project has no =inbox/= directory, the check is a silent no-op. If non-zero, the wrap is blocked. Process each item through its value-gate disposition, or delete it only when that workflow's rationale authorizes deletion, before continuing. The check exempts =lint-followups.org= explicitly because lint-org runs earlier in the same wrap-up workflow and writes its judgment items to that file in =inbox/= by design. The file is a pipeline artifact for the next morning's =daily-prep=, not a handoff that needs the value gate. @@ -469,17 +531,17 @@ Behavior: git status --short #+end_src -*Default policy: end every session with an empty =git status=.* The wrap is incomplete while anything remains dirty. There is no "leave it alone" default — every leftover gets an active resolution. The only way for a file to stay dirty across the wrap is the user explicitly saying "defer this one, leave it dirty." Surface each leftover with a concrete recommendation; the user has to actively opt out for the dirt to persist. +*Hard policy: end every session with an empty =git status=.* The wrap is incomplete while anything remains dirty. There is no deferral exception and no "wrapped with known changes" state: unresolved dirt means the session remains open and wrap-up does not occur. This inverts the older "intentional carryover" default, which let pre-existing dirty state accumulate across sessions silently. Carryover that lives for days or weeks is almost always one of: a forgotten commit from a prior wrap, a stale change that should be discarded, or genuine in-flight work that needs an explicit stash/branch home. None of those should default to "leave it dirty." **** Three kinds of leftover -| Pattern | What it is | Recommended action (apply unless user defers) | +| Pattern | What it is | Recommended action | |---+---+---| | Generated, runtime, or lock files that no human edits — e.g., =.claude/scheduled_tasks.lock=, =.pytest_cache/=, build outputs, IDE state, editor swap files | *Runtime artifact* — created by tooling or the harness, not by the user, and shouldn't be tracked | Add the matching pattern to =.gitignore= (project-level, not =~/.gitignore_global=). For tracked files, =git rm --cached <path>=. Stage =.gitignore= and any =rm --cached= changes in *one* follow-up commit (=chore: gitignore X=), push. Re-run =git status= to confirm clean. | | Modified or created during the session but not staged into the wrap-up commit | *Forgotten change* — real session work that should have been in the wrap commit but missed it | Stage and create a follow-up commit. Don't =--amend= the wrap-up commit once pushed (diverging history without a clear win). Push the follow-up to all remotes. | -| Was dirty at session start and still dirty at session end — work this session deliberately didn't touch | *Pre-existing dirt that needs a decision* — could be a missed commit from a prior wrap, stale abandoned work, or real in-flight work without a home | Investigate (show diff + check the originating session). Recommend one of: (a) commit now if the work is complete, (b) stash with a descriptive message if it's genuine WIP, (c) =git checkout -- <path>= / =git clean -f <path>= if stale and unwanted, (d) move to a feature branch if it's longer-running, (e) user explicitly defers and accepts the dirt. Do not silently leave dirty. | +| Was dirty at session start and still dirty at session end — work this session deliberately didn't touch | *Pre-existing dirt that needs a decision* — could be a missed commit from a prior wrap, stale abandoned work, or real in-flight work without a home | Investigate (show diff + check the originating session). Recommend one of: (a) commit now if the work is complete, (b) stash with a descriptive message if it's genuine WIP, (c) =git checkout -- <path>= / =git clean -f <path>= if stale and unwanted, or (d) move to a feature branch if it's longer-running. Do not silently leave dirty. | **** Per-file flow @@ -487,18 +549,40 @@ For each leftover line in =git status --short=: 1. Identify which of the three kinds above it matches. 2. State what the file is (one line) and the recommended action. -3. Apply the action unless the user explicitly defers. -4. Re-run =git status --short= after each follow-up commit until empty (or until every remaining line is an explicit user-deferred entry). +3. Apply the action when it is safe and authorized. +4. Re-run =git status --short= after each follow-up commit until empty. The pre-existing-dirt case (third row) is the one this rule most cares about. Treat each pre-existing-dirty file as a question that must get an answer this session, not as "carryover that's fine to inherit." A file that was dirty for a week before this session probably isn't going to get cleaner by waiting another week. Look at the diff, check the originating session's notes, and recommend a real resolution. -**** When the user defers +**** When cleanup cannot be completed -If the user does say "leave this one dirty for now" after seeing the recommendation, that is fine — log the deferral in the valediction so the next session knows it was an explicit choice, not a miss. Format: "Deferred (per Craig's decision today): =path/to/file= — <one-line reason>". Without that note, the next session can't distinguish "we agreed to defer" from "we forgot again." +Stop the wrap. Do not deliver the valediction, print =session wrapped.=, drop a teardown/shutdown sentinel, or describe the session as complete. Report: + +1. Every remaining path and its exact Git state. +2. What the file is and why the agent cannot safely resolve it alone. +3. The concrete action or decision Craig needs to provide to make the tree clean. + +An explicit decision to keep a file dirty changes the outcome from "wrapping" to "leaving the session interrupted." It never satisfies this workflow. + +*** Final clean-tree certificate — hard gate + +After all commits are pushed and every leftover appears resolved, run the shared gate: + +#+begin_src bash +gate="$(command -v git-worktree-gate 2>/dev/null || true)" +[ -n "$gate" ] || gate="$HOME/code/rulesets/claude-templates/bin/git-worktree-gate" +if [ ! -x "$gate" ]; then + echo "wrap blocked: git-worktree-gate is unavailable; install rulesets tooling and retry" + exit 1 +fi +"$gate" certify "$PWD" +#+end_src + +The certificate lives inside the Git directory, so it does not dirty the worktree. It records the exact verified HEAD. A non-zero result is a hard stop governed by "When cleanup cannot be completed" above. Step 5 is unreachable until certification succeeds. ** Step 5: Valediction -Brief, warm closing. 3-4 sentences max. +Only after the final clean-tree certificate succeeds, deliver a brief, warm closing. 3-4 sentences max. Include: - What was accomplished (specific, not generic) @@ -535,7 +619,7 @@ Do nothing. The buffer, the =aiv-<project>= tmux session, and =claude= all stay *** Teardown mode (default) -Confirm commit + push succeeded (Exit Criteria 5 — never tear down over unpushed work), then drop the sentinel: +Confirm commit + push and the final clean-tree certificate succeeded (Exit Criteria 5 — never tear down over unpushed or dirty work), then drop the sentinel: #+begin_src bash touch "/tmp/ai-wrap-teardown-$(basename "$PWD")" @@ -543,6 +627,8 @@ touch "/tmp/ai-wrap-teardown-$(basename "$PWD")" That is the whole step. Don't run any =tmux kill-session=, =emacsclient=, or buffer kill inline — the =Stop= hook reads the sentinel when this response ends and runs =cj/ai-term-quit=, which kills the =aiv-<project>= session (taking =claude= with it), kills the vterm buffer, and restores geometry. The basename of =$PWD= is the key the hook matches, so the sentinel names the session it tears down. +*The sentinel is session-scoped.* If certification fails, the =Stop= hook blocks and leaves the sentinel armed on purpose, so a wrap blocked by a dirty tree retries on a later stop without re-running this workflow. It does *not* survive the session: =session-start-disarm.sh= clears it at =SessionStart=, because a wrap that never certified is not a pending teardown once its session is gone. Before that hook existed, an uncertified sentinel sat armed indefinitely and fired in whatever session next reached a clean tree — work's 2026-07-27 11:37 wrap killed the 13:20 session mid-work, and archsetup's sat armed on a live terminal for two days. If teardown is still wanted in a new session, run this workflow again. + *** Shutdown mode Confirm commit + push succeeded, then evaluate the safety gate *before* committing to the shutdown — never power the box off out from under another live session: @@ -573,7 +659,8 @@ If =emacsclient= isn't resolvable or the daemon is down, the gate can't run — 7. *Leaving =.ai/session-context.org= in place* — its presence means "interrupted session", confuses next startup 8. *Long preachy valediction* — brief beats thorough 9. *Leaving runtime/generated files dirty without gitignoring them* — pollutes every future =git status= and erodes trust in "working tree clean" as a signal. Fix =.gitignore= during the wrap, not later. -10. *Treating "was dirty at session start, still dirty now" as fine by default* — that's how a forgotten commit from two sessions ago turns into "carryover" for two weeks. Every pre-existing dirty file needs an active resolution recommendation this session. Deferral is allowed only with an explicit user choice, logged in the valediction. +10. *Treating "was dirty at session start, still dirty now" as fine* — that's how a forgotten commit from two sessions ago turns into "carryover" for two weeks. Every pre-existing dirty file must be resolved or the wrap remains blocked. +11. *Calling a blocked cleanup a wrap* — if the strict gate fails, report the paths and needed decisions; do not valedict, certify completion, or tear down. * Validation Checklist @@ -586,19 +673,20 @@ Before considering wrap-up complete: - [ ] =todo-cleanup.el= ran — hygiene pass + =--convert-subtasks= + =--archive-done= + =--sync-child-priority= (if =todo.org= exists at project root) - [ ] =lint-org.el= ran on =todo.org= — mechanical fixes applied, judgments appended to follow-ups file (if =todo.org= exists) - [ ] Any orphan-planning-line warnings reviewed (fix or accept) -- [ ] Inbox carries nothing but expected pipeline artifacts (=.gitkeep=, =lint-followups.org=, =PROCESSED-*= prefixes), OR each remaining handoff has an explicit deferral logged in the valediction +- [ ] Inbox carries nothing but expected committed or ignored pipeline artifacts (=.gitkeep=, =lint-followups.org=, =PROCESSED-*= prefixes); any untracked inbox delivery was processed before wrap - [ ] Linear Dev-Review sweep ran; any merged-PR tickets moved to Done or PM Acceptance (skip if project doesn't use Linear) - [ ] Template-sync churn committed as its own =chore: sync .ai tooling from templates= (consuming projects only; skipped in rulesets), or surfaced if a synced path didn't match canonical -- [ ] After wrap-up commit + push, =git status --short= is empty OR every remaining line has an explicit user-deferred decision logged in the valediction +- [ ] After wrap-up commit + push, =git-worktree-gate certify "$PWD"= succeeded at the current HEAD - [ ] Each leftover was investigated and the user saw a concrete resolution recommendation - [ ] Runtime artifacts added to =.gitignore=, follow-up commit pushed, =git status= re-verified - [ ] Forgotten changes committed in a follow-up and pushed -- [ ] Pre-existing dirty files resolved (committed / stashed / discarded / moved to a feature branch) or explicitly deferred with a one-line reason in the valediction +- [ ] Pre-existing dirty files resolved (committed / stashed / discarded / moved to a feature branch); otherwise wrap stopped with an actionable blocker report - [ ] Current branch pushed to ALL remotes (verified with =git remote -v=) - [ ] All other local branches with a tracking upstream pushed to their remote - [ ] Any untracked-upstream branches surfaced for manual =git push -u= - [ ] Step 6 teardown matches the trigger phrase: no-teardown leaves the buffer; teardown drops only =/tmp/ai-wrap-teardown-<project>=; shutdown gates on =cj/ai-term-live-count= = 1 and drops only =/tmp/ai-wrap-shutdown-<project>= - [ ] No teardown/shutdown sentinel was dropped before commit + push was verified +- [ ] The teardown hook can re-verify the clean-tree certificate before consuming a sentinel - [ ] Shutdown aborted (fell back to normal wrap, logged in the valediction) when another =aiv-*= session was live or the gate couldn't run - [ ] Commit message follows format (no =session:=, no Claude attribution) - [ ] Valediction delivered (brief, specific, warm) |
