diff options
Diffstat (limited to 'scripts')
| -rwxr-xr-x | scripts/agenda-render-cache | 50 | ||||
| -rwxr-xr-x | scripts/bootstrap-packages.sh | 133 | ||||
| -rwxr-xr-x | scripts/calendar-sync-run | 63 | ||||
| -rwxr-xr-x | scripts/ratio-transcribe | 203 | ||||
| -rwxr-xr-x | scripts/remote-repository-reset.sh | 19 | ||||
| -rwxr-xr-x | scripts/setup-telega.sh | 12 |
6 files changed, 455 insertions, 25 deletions
diff --git a/scripts/agenda-render-cache b/scripts/agenda-render-cache new file mode 100755 index 00000000..b1e4d490 --- /dev/null +++ b/scripts/agenda-render-cache @@ -0,0 +1,50 @@ +#!/usr/bin/env bash +# Write the agenda render cache for an external renderer. +# +# Runs a batch Emacs rather than talking to the daemon, because the surface +# that reads the cache has to keep working while Emacs is down -- and +# emacsclient is exactly the thing that cannot. Nothing here touches a running +# Emacs, so it is safe to fire from a timer alongside an active session. +# +# The agenda file list is resolved the same way the editor resolves it, by +# loading the config's own resolver rather than restating the list here. A +# second copy of that list would drift the first time a calendar source is +# added. +# +# Usage: agenda-render-cache +# Env: EMACS_D -- config directory (default ~/.emacs.d) +# EMACS -- emacs binary (default emacs) +# AGENDA_RENDER_FILES -- colon-separated org files to read instead of +# the configured agenda list. Lets a caller ask +# about a known set, and lets the tests run +# against a fixture rather than whatever happens +# to be on the machine's real agenda today. + +set -euo pipefail + +EMACS_D="${EMACS_D:-$HOME/.emacs.d}" +EMACS="${EMACS:-emacs}" + +if [ ! -d "$EMACS_D/modules" ]; then + echo "agenda-render-cache: no modules directory at $EMACS_D/modules" >&2 + exit 1 +fi + +# -Q keeps the daemon's init out of it: this needs three modules, not a full +# editor. load-prefer-newer stops a stale .elc from answering for changed +# source, which would silently write yesterday's logic. +exec "$EMACS" --batch -Q \ + --eval '(setq load-prefer-newer t)' \ + -L "$EMACS_D/modules" \ + --eval '(progn + (require (quote user-constants)) + (require (quote org-agenda-config)) + (setq org-todo-keywords cj/org-todo-keywords) + (require (quote agenda-query)) + (setq org-agenda-files + (let ((override (getenv "AGENDA_RENDER_FILES"))) + (if (and override (not (string-empty-p override))) + (split-string override ":" t) + (cj/--org-agenda-scan-files)))) + (princ (cj/agenda-render-cache-update)) + (terpri))' diff --git a/scripts/bootstrap-packages.sh b/scripts/bootstrap-packages.sh new file mode 100755 index 00000000..9ba9fc69 --- /dev/null +++ b/scripts/bootstrap-packages.sh @@ -0,0 +1,133 @@ +#!/usr/bin/env bash +# +# Install every package this config asks for, headlessly, before first launch. +# +# On a fresh machine init.el pulls ~190 packages over the network one at a +# time. A single dead download used to abort startup outright; +# modules/package-resilience.el now records the failure and lets init finish, +# and this script is what turns that into a completed install: load init.el in +# batch, retry whatever is still missing, and repeat while progress is being +# made. Doing it here rather than in a GUI session means a fresh install never +# meets the debugger. +# +# Usage: scripts/bootstrap-packages.sh +# Exit: 0 when every package is installed, non-zero otherwise. +# +# Environment: +# EMACS emacs binary to use (default: emacs) +# BOOTSTRAP_PASSES maximum passes over the set (default: 4) +# BOOTSTRAP_TIMEOUT seconds allowed per pass (default: 1800) +# +# Note: a pass loads the whole config in batch, so every :config block runs +# headlessly. stdin is closed and each pass is bounded by a timeout so a +# prompt or a hung network fetch fails the pass instead of stalling forever. + +set -uo pipefail + +emacs_bin="${EMACS:-emacs}" +max_passes="${BOOTSTRAP_PASSES:-4}" +pass_timeout="${BOOTSTRAP_TIMEOUT:-1800}" + +emacs_dir="${BOOTSTRAP_DIR:-$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)}" +log_dir="$(mktemp -d -t emacs-bootstrap-XXXXXX)" + +cleanup() { rm -rf "$log_dir"; } +trap cleanup EXIT + +# --batch implies -q, which skips early-init.el. That file is where the package +# archives, use-package-always-ensure, and package-resilience all live, so a +# pass that loaded only init.el would install almost nothing and would not even +# have cj/package-bootstrap-batch defined. Load both, in the order a real +# startup does. user-emacs-directory is set first so the pass bootstraps the +# checkout this script lives in rather than whatever $HOME/.emacs.d happens to +# be. +load_form="$(cat <<EOF +(progn + (setq load-prefer-newer t) + (setq user-emacs-directory "${emacs_dir}/") + (setq package-user-dir (expand-file-name "elpa" user-emacs-directory)) + (load (expand-file-name "early-init.el" user-emacs-directory) nil t) + (load (expand-file-name "init.el" user-emacs-directory) nil t) + (cj/package-bootstrap-batch)) +EOF +)" + +echo "bootstrap: installing packages for $emacs_dir" +echo "bootstrap: up to $max_passes passes, ${pass_timeout}s each" + +# use-package calls its ensure function at macro-expansion time when a file is +# being byte-compiled, and emits no runtime call at all. So a pass that loads +# .elc files installs nothing and would still exit 0 -- a false pass, the same +# shape as every other gate in this repo that was green because it never ran. +# Refuse rather than warn: there is no use for a bootstrap that cannot install, +# and a warning above a success line is read as a success. A genuinely fresh +# machine has no .elc and never sees this. +# Every directory the config puts on its load-path, not just modules/, since a +# use-package form anywhere in them would be consumed the same way. +if compgen -G "$emacs_dir/modules/*.elc" >/dev/null 2>&1 \ + || compgen -G "$emacs_dir/custom/*.elc" >/dev/null 2>&1 \ + || compgen -G "$emacs_dir/assets/*.elc" >/dev/null 2>&1 \ + || compgen -G "$emacs_dir/*.elc" >/dev/null 2>&1; then + echo "bootstrap: REFUSING - byte-compiled modules are present." >&2 + echo "bootstrap: use-package consumes :ensure at compile time, so a pass over" >&2 + echo "bootstrap: .elc files installs nothing and would report success anyway." >&2 + echo "bootstrap: run 'make clean-compiled' first, then bootstrap." >&2 + exit 2 +fi + +pass=1 +passes_run=0 +status=1 +while [ "$pass" -le "$max_passes" ]; do + log="$log_dir/pass-$pass.log" + echo "bootstrap: pass $pass of $max_passes ..." + + timeout "$pass_timeout" "$emacs_bin" --batch \ + --eval "$load_form" </dev/null >"$log" 2>&1 + status=$? + passes_run=$((passes_run + 1)) + + case "$status" in + 0) + echo "bootstrap: every package is installed (pass $pass)" + break + ;; + 1) + # Exit 1 is only meaningful when the pass actually said what is + # missing. Anything else exiting 1 is a different failure, and + # retrying it four times then blaming packages would be a lie. + if grep -E '^package-bootstrap: [0-9]+ missing:' "$log"; then + : # another pass can clear them; installing one unblocks others + else + echo "bootstrap: pass $pass exited 1 without reporting missing packages" >&2 + tail -30 "$log" >&2 + break + fi + ;; + 124) + echo "bootstrap: pass $pass hit the ${pass_timeout}s timeout" >&2 + tail -20 "$log" >&2 + ;; + *) + # init itself failed for some reason other than a missing package. + echo "bootstrap: pass $pass failed to load init (exit $status)" >&2 + tail -30 "$log" >&2 + break + ;; + esac + + pass=$((pass + 1)) +done + +if [ "$status" -ne 0 ]; then + echo "bootstrap: FAILED after $passes_run pass(es)" >&2 + # No pass ran at all when the ceiling is zero, and the glob would then match + # nothing and print a tail error over the real message. + if [ "$passes_run" -gt 0 ]; then + echo "bootstrap: tail of the last pass follows" >&2 + tail -30 "$log" >&2 + fi + exit "$status" +fi + +exit 0 diff --git a/scripts/calendar-sync-run b/scripts/calendar-sync-run new file mode 100755 index 00000000..12181b8d --- /dev/null +++ b/scripts/calendar-sync-run @@ -0,0 +1,63 @@ +#!/usr/bin/env bash +# Sync every configured calendar from its .ics feed, once, and wait for it. +# +# Runs a batch Emacs rather than talking to the daemon, for the same reason +# agenda-render-cache does: the org files this writes feed the agenda, waybar +# and the projected wallpaper, and none of those should go stale because the +# editor happens to be down. Nothing here touches a running Emacs, so it is +# safe to fire from a timer alongside an active session. +# +# Why a script at all: calendar-sync used to start its hourly timer from +# `org-agenda-mode-hook', so a session where the agenda was never opened never +# synced at all. That is not a rare corner -- it is every reboot until the +# first agenda call. The timer owns the schedule now, and the editor's own +# auto-start is off. +# +# The exit code is the point. `calendar-sync-batch-run-and-report' waits for +# every calendar to leave the syncing state and returns non-zero if any did +# not land, so a failed fetch shows up in `systemctl --user status' instead of +# being swallowed. +# +# Usage: calendar-sync-run +# Env: EMACS_D -- config directory (default ~/.emacs.d) +# EMACS -- emacs binary (default emacs) +# CALENDAR_SYNC_CONFIG -- private config file holding the calendar +# list, instead of the configured default. +# CALENDAR_SYNC_STATE -- sync-state file, instead of the configured +# default. Lets a test run without writing +# the real session's persisted state. +# CALENDAR_SYNC_TIMEOUT -- seconds to wait for all calendars to settle. + +set -euo pipefail + +EMACS_D="${EMACS_D:-$HOME/.emacs.d}" +EMACS="${EMACS:-emacs}" + +if [ ! -d "$EMACS_D/modules" ]; then + echo "calendar-sync-run: no modules directory at $EMACS_D/modules" >&2 + exit 1 +fi + +# The overrides are set before the module loads on purpose: the private config +# is read at load time, and `defvar'/`defcustom' both leave an already-bound +# value alone. This is the same seam the in-editor conversion worker uses. +pre="" +if [ -n "${CALENDAR_SYNC_CONFIG:-}" ]; then + pre="$pre (setq calendar-sync-private-config-file \"$CALENDAR_SYNC_CONFIG\")" +fi +if [ -n "${CALENDAR_SYNC_STATE:-}" ]; then + pre="$pre (setq calendar-sync--state-file \"$CALENDAR_SYNC_STATE\")" +fi +if [ -n "${CALENDAR_SYNC_TIMEOUT:-}" ]; then + pre="$pre (setq calendar-sync-batch-timeout $CALENDAR_SYNC_TIMEOUT)" +fi + +# -Q keeps the daemon's init out of it: this needs the calendar-sync modules, +# not a full editor. load-prefer-newer stops a stale .elc from answering for +# changed source, which would silently sync with yesterday's logic. +exec "$EMACS" --batch -Q \ + --eval "(progn (setq load-prefer-newer t)$pre)" \ + -L "$EMACS_D/modules" \ + --eval '(progn + (require (quote calendar-sync)) + (kill-emacs (calendar-sync-batch-run-and-report)))' diff --git a/scripts/ratio-transcribe b/scripts/ratio-transcribe new file mode 100755 index 00000000..8db59b66 --- /dev/null +++ b/scripts/ratio-transcribe @@ -0,0 +1,203 @@ +#!/usr/bin/env bash +# ratio-transcribe - Transcribe audio on my own transcription host, with speaker labels +# Usage: ratio-transcribe <audio-file> [language] +# +# Same contract as assemblyai-transcribe: the transcript goes to stdout, one line +# per speaker turn ("HH:MM:SS Speaker A: text"); progress and errors go to stderr; +# any failure exits non-zero with nothing on stdout. +# +# The work happens on a host that runs the meeting-transcribe queue (whisper-cpp +# plus pyannote). This script copies the audio over ssh, drops a job into the +# queue, waits, and prints the result. The job id is a hash of the audio and its +# options, so if the connection drops or the laptop sleeps, running the same +# command again just collects the finished transcript. If the host can't be +# reached at all, the same queue and worker run on this machine instead. +# +# Optional environment: +# SPEAKERS exact number of speakers, when you know it +# MIN_SPEAKERS, MAX_SPEAKERS a range instead +# TRANSCRIBE_HOST ssh name of the host (default: ratio) +# TRANSCRIBE_TIMEOUT seconds to wait for the job (default: 3600) +# TRANSCRIBE_POLL seconds between checks (default: 10) +# TRANSCRIBE_LOCAL=1 skip the host and run here +# TRANSCRIBE_WORKER path to the local worker + +set -euo pipefail + +AUDIO="${1:-}" +LANG_CODE="${2:-en}" +HOST="${TRANSCRIBE_HOST:-ratio}" +TIMEOUT="${TRANSCRIBE_TIMEOUT:-3600}" +POLL="${TRANSCRIBE_POLL:-10}" +WORKER="${TRANSCRIBE_WORKER:-$HOME/.local/share/pyannote-diarize/src/transcribe-worker}" +STATE=".local/state/meeting-transcribe" # relative to the home directory, on either machine + +if [[ -z "$AUDIO" ]]; then + echo "Usage: ratio-transcribe <audio-file> [language]" >&2 + echo "Example: SPEAKERS=3 ratio-transcribe meeting.m4a en" >&2 + exit 1 +fi + +if [[ ! -f "$AUDIO" ]]; then + echo "Error: Audio file not found: $AUDIO" >&2 + exit 1 +fi +# scp reads "name:with:colons" as host:path; an absolute path removes the ambiguity. +AUDIO="$(realpath -- "$AUDIO")" + +# Everything below ends up in a job file and on command lines, so check it first. +if [[ ! "$LANG_CODE" =~ ^[A-Za-z]{2,8}(-[A-Za-z0-9]{1,8})*$ ]]; then + echo "Error: Invalid language code: $LANG_CODE" >&2 + exit 1 +fi + +for name in SPEAKERS MIN_SPEAKERS MAX_SPEAKERS; do + value="${!name:-}" + if [[ -n "$value" && ! "$value" =~ ^[1-9][0-9]*$ ]]; then + echo "Error: $name must be a positive whole number of speakers, got: $value" >&2 + exit 1 + fi +done +if [[ -n "${SPEAKERS:-}" && ( -n "${MIN_SPEAKERS:-}" || -n "${MAX_SPEAKERS:-}" ) ]]; then + echo "Error: give an exact SPEAKERS count or a MIN/MAX speaker range, not both" >&2 + exit 1 +fi +if [[ -n "${MIN_SPEAKERS:-}" && -n "${MAX_SPEAKERS:-}" ]] && (( MIN_SPEAKERS > MAX_SPEAKERS )); then + echo "Error: MIN_SPEAKERS cannot exceed MAX_SPEAKERS (speaker range)" >&2 + exit 1 +fi + +for tool in jq sha256sum; do + if ! command -v "$tool" &> /dev/null; then + echo "Error: $tool command not found" >&2 + exit 1 + fi +done + +EXT="${AUDIO##*.}" +[[ "$EXT" =~ ^[A-Za-z0-9]{1,5}$ ]] || EXT="bin" +EXT="${EXT,,}" + +if [[ -n "${SPEAKERS:-}" ]]; then + COUNT_TAG="s${SPEAKERS}" +elif [[ -n "${MIN_SPEAKERS:-}${MAX_SPEAKERS:-}" ]]; then + COUNT_TAG="r${MIN_SPEAKERS:-x}-${MAX_SPEAKERS:-x}" +else + COUNT_TAG="auto" +fi +JOB_ID="$(sha256sum "$AUDIO" | cut -c1-16)-${LANG_CODE,,}-${COUNT_TAG}" + +JOB_JSON=$(jq -cn \ + --arg language "$LANG_CODE" \ + --arg name "$(basename "$AUDIO")" \ + --arg speakers "${SPEAKERS:-}" --arg min "${MIN_SPEAKERS:-}" --arg max "${MAX_SPEAKERS:-}" \ + '{language: $language} + + (if $speakers != "" then {speakers: ($speakers | tonumber)} else {} end) + + (if $min != "" then {min_speakers: ($min | tonumber)} else {} end) + + (if $max != "" then {max_speakers: ($max | tonumber)} else {} end) + + {original_name: $name}') + +# ssh reads stdin unless told not to, which would swallow the input of any loop +# this script is called from. Only the job-file upload needs stdin. +remote() { ssh -n -o BatchMode=yes -o ConnectTimeout=8 "$HOST" "$@"; } +remote_with_stdin() { ssh -o BatchMode=yes -o ConnectTimeout=8 "$HOST" "$@"; } + +# One word for where the job stands on the host: done, failed, queued or new. +remote_status() { + remote "cd $STATE 2>/dev/null || { echo new; exit 0; } + if [ -e done/$JOB_ID.txt ]; then echo done + elif [ -e failed/$JOB_ID.log ]; then echo failed + elif [ -d incoming/$JOB_ID ] || [ -d work/$JOB_ID ]; then echo queued + else echo new; fi" +} + +print_transcript() { # $1 = the transcript text + if [[ -z "${1//[[:space:]]/}" ]]; then + echo "Error: the transcript came back empty" >&2 + exit 1 + fi + echo "Transcription complete! (${SECONDS}s total)" >&2 + printf '%s\n' "$1" +} + +run_remote() { + local status + status=$(remote_status) + + if [[ "$status" == "failed" ]]; then + echo "An earlier attempt at this job failed; trying again..." >&2 + remote "rm -f $STATE/failed/$JOB_ID.log" + status="new" + fi + + if [[ "$status" == "new" ]]; then + echo "Uploading audio file to $HOST..." >&2 + # Copy into uploading/, then rename into incoming/. The queue only ever sees + # a complete job. + remote "mkdir -p $STATE/incoming $STATE/uploading/$JOB_ID" + scp -q -o BatchMode=yes "$AUDIO" "$HOST:$STATE/uploading/$JOB_ID/audio.$EXT" < /dev/null + printf '%s' "$JOB_JSON" | remote_with_stdin "cat > $STATE/uploading/$JOB_ID/job.json" + remote "mv $STATE/uploading/$JOB_ID $STATE/incoming/$JOB_ID" + echo "Job $JOB_ID queued. Waiting for completion..." >&2 + elif [[ "$status" == "queued" ]]; then + echo "Job $JOB_ID is already queued on $HOST. Waiting for completion..." >&2 + fi + + while true; do + # A dropped connection is not a failed job; keep asking until the timeout. + status=$(remote_status 2> /dev/null) || status="unreachable" + case "$status" in + done) + print_transcript "$(remote "cat $STATE/done/$JOB_ID.txt")" + return 0 + ;; + failed) + echo "Error: transcription failed on $HOST" >&2 + remote "cat $STATE/failed/$JOB_ID.log" >&2 || true + exit 1 + ;; + esac + if (( SECONDS >= TIMEOUT )); then + echo "Error: no result after ${TIMEOUT}s. The job is still with $HOST;" >&2 + echo "run the same command again to collect the transcript." >&2 + exit 1 + fi + sleep "$POLL" + [[ "$status" == "unreachable" ]] || echo "Processing... (${SECONDS}s elapsed)" >&2 + done +} + +run_local() { + if [[ ! -x "$WORKER" ]]; then + echo "Error: $HOST is unreachable and there is no local worker at $WORKER" >&2 + exit 1 + fi + local state="$HOME/$STATE" + if [[ ! -s "$state/done/$JOB_ID.txt" ]]; then + echo "Running the transcription locally (this machine is slower; expect a wait)..." >&2 + rm -f "$state/failed/$JOB_ID.log" + rm -rf "$state/uploading/$JOB_ID" + mkdir -p "$state/incoming" "$state/uploading/$JOB_ID" + cp "$AUDIO" "$state/uploading/$JOB_ID/audio.$EXT" + printf '%s' "$JOB_JSON" > "$state/uploading/$JOB_ID/job.json" + [[ -d "$state/incoming/$JOB_ID" ]] || mv "$state/uploading/$JOB_ID" "$state/incoming/$JOB_ID" + HF_HUB_OFFLINE=1 "$WORKER" >&2 < /dev/null + fi + if [[ -e "$state/failed/$JOB_ID.log" ]]; then + echo "Error: local transcription failed" >&2 + cat "$state/failed/$JOB_ID.log" >&2 + exit 1 + fi + if [[ ! -e "$state/done/$JOB_ID.txt" ]]; then + echo "Error: the local worker finished without producing a transcript" >&2 + exit 1 + fi + print_transcript "$(< "$state/done/$JOB_ID.txt")" +} + +if [[ -z "${TRANSCRIBE_LOCAL:-}" ]] && remote true 2> /dev/null; then + run_remote +else + [[ -n "${TRANSCRIBE_LOCAL:-}" ]] || echo "$HOST is unreachable." >&2 + run_local +fi diff --git a/scripts/remote-repository-reset.sh b/scripts/remote-repository-reset.sh deleted file mode 100755 index e9a243a8..00000000 --- a/scripts/remote-repository-reset.sh +++ /dev/null @@ -1,19 +0,0 @@ -#!/bin/sh -# Craig Jennings -# post archsetup step to reset remote upstream repositories on emacs -# configuration and doftiles. - -cd ~/emacs.d/ -git remote remove origin -git remote add github git@github.com:cjennings/dotemacs.git -git remote add origin git@cjennings.net:dotemacs.git -git branch -M main -git push -u origin main - - -cd ~/.dotfiles/ -git remote remove origin -git remote add github git@github.com:cjennings/dotfiles.git -git remote add origin git@cjennings.net:dotfiles.git -git branch -M main -git push -u origin main diff --git a/scripts/setup-telega.sh b/scripts/setup-telega.sh index 9a8c0d02..351f98ba 100755 --- a/scripts/setup-telega.sh +++ b/scripts/setup-telega.sh @@ -7,8 +7,8 @@ # - Verifies docker is installed and the daemon is responsive. # - Verifies the user can talk to docker without sudo (group membership). # - Pulls the telega-server image if a public one is configured (env var -# `TELEGA_DOCKER_IMAGE'); otherwise prints the in-Emacs build command -# (`M-x telega-server-build') for the user to run once. +# `TELEGA_DOCKER_IMAGE'); otherwise points at `make telega-image', which +# builds the pinned image from docker/telega-server/Dockerfile. # - Installs the `telega' Emacs package via package.el if it isn't # already in package-user-dir. modules/telega-config.el uses # `:ensure nil' (a stale MELPA index can 404 and take startup down @@ -52,10 +52,10 @@ pull_or_announce_image() { if [[ -z "$TELEGA_DOCKER_IMAGE" ]]; then cat <<EOF → no public image configured (set TELEGA_DOCKER_IMAGE to override) - build the telega-server image once from inside Emacs: - M-x telega-server-build - telega.el handles the docker build under the hood when - \`telega-use-docker' is t (set in modules/telega-config.el). + build the telega-server image once from the repo's Dockerfile: + make telega-image + modules/telega-config.el pins \`cj/telega-docker-image' to the tag + that target builds (docker/telega-server/Dockerfile). EOF return 0 fi |
