Files
stack/tools/agent-watch/agent-watch.sh
T

731 lines
35 KiB
Bash
Executable File

#!/usr/bin/env bash
# agent-watch.sh — isolated condition watcher for fleet agents (v2: systemd timers).
#
# Each watch is its own transient systemd --user timer + service (cron-style):
# no loop process, no orphan risk, no host-reboot fragility (v1's nohup model
# died at reboot), cgroup-isolated, journald-logged. Agents never hand-write
# watch scripts or sleep loops — a watch is one CLI call.
#
# Usage:
# agent-watch.sh start --name <id> --session <tmux-session> \
# --when '<shell command; true = met>' --message "text to send" \
# [--class actionable|reaction|human|digest|terminal-log] \
# [--interval 30] [--timeout 3600] [--repeat] [--quiet-timeout] \
# [--socket <tmux-socket>] # e.g. mosaic-fleet for fleet seats
# agent-watch.sh list
# agent-watch.sh status [--json] # exit 0 = clean, 3 = any stale watch or dead meta-watch, 6 = user bus unreachable
# agent-watch.sh stop <name>
# agent-watch.sh log <name>
# agent-watch.sh meta-install [--interval 300] [--unit-name agent-watch-meta]
# agent-watch.sh meta-remove [--unit-name agent-watch-meta]
#
# Rules encoded (guides/FLEET-COMMS.md, WAKE-DOCTRINE.md):
# - Interval floor 10s: a watcher is a fallback cadence, never a tight poll.
# - Delivery via agent-send.sh only; rc=2 = reached the pane as draft =
# DELIVERED, never retried. Real failures retry twice, then give up loudly.
# - Conditions run CRON-STYLE: clean-ish env (HOME/PATH/MOSAIC_* pass
# through), cwd=$HOME. Do not rely on ambient credentials; use absolute
# paths and the credential helper.
# - Messages carry [watch:<name>] so the recipient can trace or stop them.
# - One-shot by default; --repeat re-arms after each delivery.
# - Timeout (default 3600s): terminal-log note unless --quiet-timeout,
# then the timer is cancelled. A watch is never forever; re-arm deliberately.
# - State: $STATE_ROOT/<name>/ (config, log). Units: agent-watch-<name>.{timer,service}
# - Expected deaths are marked: completion, timeout, and broken-condition
# paths write a `terminated` marker BEFORE stopping the timer, so those
# watches show as retired (owed nothing), never as stale.
# - Stale = config present, timer gone, no terminated marker: unexpected
# loss. list/status deliver AT MOST ONE notice per staleness episode: a
# noclobber claim on stale-noticed admits exactly one of N concurrent
# callers; a crash between claim and send can drop that episode's notice
# (staleness stays visible in status output / exit 3 regardless). The
# stale claim clears on re-arm (start) and observed recovery; the
# terminated marker clears on re-arm only — activity during a stop
# window must not erase it (T16W2 race). D62 CLOSED by the meta-watch
# (T19): a persistent, ENABLED systemd user timer runs the hidden
# `_scan` path on a fixed cadence — the SAME classification and
# claim-first notice logic list/status use — so a lost watch is noticed
# without anyone querying. Each scan stamps a heartbeat file; list and
# status report meta health (timer active + heartbeat age vs cadence),
# so a dead meta-watch is visible on the existing query surface instead
# of silently recreating D62 one level up. status is the machine-readable
# liveness answer (JSON or human; retired watches are listed separately,
# exit 3 when any watch is stale OR an installed meta-watch is dead).
# - Ambiguous unpinned socket resolution refuses with rc 4; pass --socket
# or set MOSAIC_TMUX_SOCKET to choose deliberately.
# - No user bus (XDG_RUNTIME_DIR / DBUS_SESSION_BUS_ADDRESS unset or bus
# unreachable): systemctl --user fails with EMPTY output, which a naive
# classifier reads as "timer gone". Every classifying query (list /
# status / _scan / start / meta-install) refuses loudly instead — rc 6,
# diagnostic naming the bus, zero claims, zero notices (T24).
set -euo pipefail
VERSION="2.1.2"
SEND="${MOSAIC_AGENT_SEND:-$HOME/.config/mosaic/tools/tmux/agent-send.sh}"
STATE_ROOT="${MOSAIC_WATCH_STATE:-$HOME/.cache/mosaic-agent-watch}"
FLOOR_INTERVAL=10
CLASSES="actionable|reaction|human|digest|terminal-log"
UNIT_PREFIX="agent-watch"
META_UNIT_DEFAULT="agent-watch-meta"
META_FLOOR_INTERVAL=60 # meta cadence floor: a detection net, never a poll
RC_NO_BUS=6 # T24: user bus unreachable — classify nothing, notify nothing
die() { echo "agent-watch: $*" >&2; exit 2; }
usage() { sed -n '2,/^set -euo pipefail/{/^set -euo pipefail/d;p}' "$0" | sed 's/^# \{0,1\}//'; }
# T24 (measured live 2026-08-23 03:33): a caller without the user bus gets a
# FAILING systemctl --user whose empty output the classifier read as "timer
# gone" — every live watch classified LOST, false claims written, false
# notices delivered (self-healed only via T16W2 alive-release, masking real
# losses in the window). An unreachable bus is a broken instrument, not a
# fleet of dead watches: systemctl's own failure (rc!=0, EMPTY stdout) is
# distinguishable from an empty-but-successful query (rc!=0 with TEXT like
# "inactive" on stdout). Every path that classifies probes first and
# refuses to classify at all on failure.
require_user_bus() {
local err rc=0
err="$(systemctl --user show-environment 2>&1 1>/dev/null)" || rc=$?
if [[ "$rc" -ne 0 ]]; then
{
echo "agent-watch: systemd user bus UNREACHABLE — refusing to classify watch liveness (rc $RC_NO_BUS)"
echo "agent-watch: systemctl --user show-environment failed rc=$rc: ${err}"
echo "agent-watch: no claims written, no notices sent; a failed query is NOT 'no timers'."
echo "agent-watch: fix the caller: export XDG_RUNTIME_DIR=/run/user/$(id -u) DBUS_SESSION_BUS_ADDRESS=unix:path=/run/user/$(id -u)/bus"
} >&2
exit "$RC_NO_BUS"
fi
}
state_dir() { echo "$STATE_ROOT/$1"; }
unit_for() { echo "$UNIT_PREFIX-$1"; }
say_log() { # per-watch durable log (journal also carries it, but `log` reads this)
local dir; dir="$(state_dir "$1")"; mkdir -p "$dir"
echo "[$(date -u +%FT%TZ)] ${2:-}" >> "$dir/watch.log"
}
# Terminal marker (review B1): every intentional stop path records WHY the
# timer is about to vanish, BEFORE the stop. Stale detection (config present +
# timer gone) ignores markered watches: expected deaths are owed nothing.
# Writing pre-stop closes the race: a detector can never observe timer-gone
# with the marker still absent.
mark_terminal() { # $1=name $2=reason
printf '%s\n' "$2" > "$(state_dir "$1")/terminated"
say_log "$1" "terminal: $2"
}
# ── start ─────────────────────────────────────────────────────────────────────
cmd_start() {
local name="" session="" when="" message="" cls="actionable" socket=""
local interval=30 timeout=3600 repeat=0 quiet=0
while [[ $# -gt 0 ]]; do
case "$1" in
--name) name="$2"; shift 2 ;;
--session) session="$2"; shift 2 ;;
--socket) socket="$2"; shift 2 ;;
--when) when="$2"; shift 2 ;;
--message) message="$2"; shift 2 ;;
--class) cls="$2"; shift 2 ;;
--interval) interval="$2"; shift 2 ;;
--timeout) timeout="$2"; shift 2 ;;
--repeat) repeat=1; shift ;;
--quiet-timeout) quiet=1; shift ;;
*) die "start: unknown argument: $1" ;;
esac
done
[[ "$name" =~ ^[a-z0-9][a-z0-9-]*$ ]] || die "start: --name must be lowercase-hyphens (got: '$name')"
[[ "$name" != *.* ]] || die "start: --name must not contain dots (systemd unit naming)"
[[ -n "$session" ]] || die "start: --session is required"
# Pin the target to an EXACT session name. Without the leading `=`, tmux
# PREFIX-matches and returns rc=0 on the wrong session. Measured 2026-08-24
# on this host: `jarvis` and `jarvis-enhance` are a prefix pair split across
# two servers, so on the default server `-t jarvis` resolves to
# `jarvis-enhance` at rc=0 today -- no dead seat required. Normalising here
# rather than at each use covers the guard below AND all four fire-time
# `$SEND -s "$SESSION"` sites, because send-message.sh already expands
# `=sess` to a pane-qualified target. Preserve an already-pinned session,
# and pin only its session component for compound `session:window.pane`
# targets. This keeps discovery from constructing `==session:window.pane`.
case "$session" in
=*) ;;
*:*) session="=${session%%:*}:${session#*:}" ;;
*) session="=$session" ;;
esac
[[ -n "$when" ]] || die "start: --when is required (quoted shell command; exit 0 = met)"
[[ -n "$message" ]] || die "start: --message is required"
[[ "$cls" =~ ^($CLASSES)$ ]] || die "start: --class must be one of: $CLASSES"
[[ "$interval" =~ ^[0-9]+$ ]] || die "start: --interval must be a number"
[[ "$interval" -ge "$FLOOR_INTERVAL" ]] || die "start: --interval floor is ${FLOOR_INTERVAL}s (got ${interval}s) — no tight polls"
[[ "$timeout" =~ ^[0-9]+$ ]] || die "start: --timeout must be a number (seconds)"
[[ -x "$SEND" ]] || die "sender not found/executable: $SEND"
command -v systemctl >/dev/null 2>&1 || die "systemctl not on PATH (v2 requires systemd --user)"
require_user_bus # T24: fail naming the bus, not as a cryptic systemd-run error
# B1 (2026-08-29): socket default resolution. Precedence: explicit
# --socket > MOSAIC_TMUX_SOCKET (launcher-exported) > unique socket hit
# > refusal on ambiguity. Socket discovery scans tmux's own socket dir,
# ${TMUX_TMPDIR:-/tmp}/tmux-UID (codex PR #1466: TMPDIR is wrong here).
# Validate an environment-selected socket just like an explicit socket so a
# stale launcher pin cannot create a watcher that can never deliver.
if [[ -z "$socket" && -n "${MOSAIC_TMUX_SOCKET:-}" ]]; then
socket="$MOSAIC_TMUX_SOCKET"
fi
if [[ -n "$socket" ]]; then
tmux -L "$socket" has-session -t "$session" 2>/dev/null || die "no tmux session '$session' on socket '$socket'"
else
local hits="" sname sf hit_count
local socket_dir="${TMUX_TMPDIR:-/tmp}/tmux-$(id -u)"
for sf in "$socket_dir"/*; do
[ -S "$sf" ] || continue
sname="${sf##*/}"
tmux -L "$sname" has-session -t "$session" 2>/dev/null && hits="$hits $sname"
done
hit_count=$(printf '%s' "$hits" | wc -w)
if [ "$hit_count" -gt 1 ]; then
echo "agent-watch: session '$session' exists on multiple sockets:$hits — pass --socket explicitly" >&2
exit 4
elif [ "$hit_count" -eq 1 ]; then
socket="${hits# }"
else
die "no tmux session '$session' (default socket)"
fi
fi
local unit; unit="$(unit_for "$name")"
if systemctl --user is-active "$unit.timer" >/dev/null 2>&1; then
die "a watcher named '$name' is already running (unit $unit.timer); stop it first or pick another name"
fi
local dir; dir="$(state_dir "$name")"
mkdir -p "$dir"
# printf %q, not bare quotes (D43): the config is SOURCED by each tick; a
# single quote in message/condition used to close the string early and kill
# the watcher silently after start reported success. Measured 2026-08-21.
{
printf 'NAME=%q\n' "$name"
printf 'SESSION=%q\n' "$session"
printf 'SOCKET=%q\n' "$socket"
printf 'CLASS=%q\n' "$cls"
printf 'MESSAGE=%q\n' "$message"
printf 'CONDITION=%q\n' "$when"
printf 'INTERVAL=%s\n' "$interval"
printf 'TIMEOUT=%s\n' "$timeout"
printf 'STARTED=%s\n' "$(date +%s)"
printf 'REPEAT=%s\n' "$repeat"
printf 'QUIET_TIMEOUT=%s\n' "$quiet"
} > "$dir/config"
# Fail loudly here, not in a detached tick nobody is reading.
if ! ( set -e; source "$dir/config" ) 2>/dev/null; then
rm -f "$dir/config"
die "could not write a sourceable config for '$name'; watcher NOT started"
fi
# Re-arm clears episode/terminal markers: a fresh staleness episode must be
# notifiable again, and a re-armed watch is no longer retired.
rm -f "$dir/stale-noticed" "$dir/terminated" 2>/dev/null || true
# One transient timer per watch = the isolation. Each tick is a fresh
# process in its own cgroup: a crash kills that tick only; the journal
# carries the history; reboot cancels cleanly (a watch is re-armed by
# whoever still wants it — that is deliberate, WAKE-DOCTRINE's "a watcher
# is a fallback cadence, not a steady-state mechanism").
# --on-active fires the first tick in ~1s; --on-unit-active-sec re-arms
# after every tick. systemd composes the two as OR.
: > "$dir/watch.log"
if ! systemctl --user start "$unit.timer" 2>/dev/null; then
# transient timer does not exist yet — create it
# ENV PASS-THROUGH (measured 2026-08-23, D62): a tick runs in the unit's
# environment, NOT the arming shell's. MOSAIC_WATCH_STATE or
# MOSAIC_AGENT_SEND set at arm time but not passed here made _tick resolve
# a DIFFERENT state root, miss its config, and fire the poison pill two
# seconds after start — the watch silently died while `start` had already
# reported success. Any override the arming shell used must travel with
# the unit, or start and tick disagree about which watch they serve.
local extra_env=()
[[ -n "${MOSAIC_WATCH_STATE:-}" ]] && extra_env+=(--setenv=MOSAIC_WATCH_STATE="$MOSAIC_WATCH_STATE")
[[ -n "${MOSAIC_AGENT_SEND:-}" ]] && extra_env+=(--setenv=MOSAIC_AGENT_SEND="$MOSAIC_AGENT_SEND")
if ! systemd-run --user --collect \
--unit="$unit" \
--description="agent-watch: $name (-> $session${socket:+ on $socket})" \
--on-active=1s \
--on-unit-active="${interval}s" \
--setenv=HOME="$HOME" \
--setenv=PATH="$PATH" \
--setenv=MOSAIC_BRAIN_HOME="${MOSAIC_BRAIN_HOME:-$HOME/.mosaic}" \
${extra_env[@]+"${extra_env[@]}"} \
bash "$(readlink -f "$0")" _tick "$name" >> "$dir/watch.log" 2>&1; then
rm -f "$dir/config"
die "systemd-run failed for $unit (see $dir/watch.log); watcher NOT started"
fi
fi
say_log "$name" "watcher started: every ${interval}s, timeout ${timeout}s, -> $session${socket:+ on $socket} ($cls)"
echo "started watcher '$name' (unit $unit.timer): every ${interval}s, timeout ${timeout}s, -> ${socket:+$socket/}$session ($cls)"
echo "condition: $when"
}
# ── _tick (hidden; run BY the transient service each interval) ────────────────
cmd_tick() {
local name="$1"
local dir; dir="$(state_dir "$name")"
local unit; unit="$(unit_for "$name")"
# Poison pill (v1.1.0 semantics): no config = stopped/cleaned; cancel timer.
if [[ ! -r "$dir/config" ]]; then
systemctl --user stop "$unit.timer" "$unit.service" >/dev/null 2>&1 || true
exit 0
fi
# shellcheck disable=SC1090
source "$dir/config"
# Timeout: a watch is never forever.
if (( $(date +%s) - STARTED >= TIMEOUT )); then
if [[ "$QUIET_TIMEOUT" -eq 0 ]]; then
"$SEND" ${SOCKET:+-L "$SOCKET"} -s "$SESSION" -C terminal-log \
-m "[watch:$NAME] timeout after ${TIMEOUT}s — watcher retired" >/dev/null 2>&1 || true
fi
say_log "$NAME" "timeout after ${TIMEOUT}s"
mark_terminal "$NAME" "timeout after ${TIMEOUT}s"
systemctl --user stop "$unit.timer" >/dev/null 2>&1 || true
rm -f "$dir/pid" 2>/dev/null || true
exit 0
fi
set +e
( cd "$HOME" && bash -c "$CONDITION" ) >/dev/null 2>&1
local rc=$?
set -e
if [[ "$rc" -eq 0 ]]; then
say_log "$NAME" "condition met (rc=0)"
local tries=0 drc
while :; do
set +e
"$SEND" ${SOCKET:+-L "$SOCKET"} -s "$SESSION" -C "$CLASS" -m "[watch:$NAME] $MESSAGE"
drc=$?
set -e
case "$drc" in
0) say_log "$NAME" "delivered (rc=0)"; break ;;
2) say_log "$NAME" "rc=2: reached pane as draft — delivered, NOT retried"; break ;;
*) tries=$((tries + 1))
if [[ "$tries" -ge 3 ]]; then
say_log "$NAME" "delivery failed rc=$drc after 3 attempts — giving up"
# NO terminal marker here, deliberately: the recipient got
# nothing, so the watch is still owed. It stays stale-detectable
# and the LOST notice is truthful (B1 covers only paths that
# already told the recipient something).
systemctl --user stop "$unit.timer" >/dev/null 2>&1 || true
exit 4
fi
sleep 5 ;;
esac
done
if [[ "$REPEAT" -eq 1 ]]; then
say_log "$NAME" "--repeat: re-arming"
return 0
fi
say_log "$NAME" "watcher complete"
mark_terminal "$NAME" "complete: condition met, notice delivered"
systemctl --user stop "$unit.timer" >/dev/null 2>&1 || true
exit 0
elif [[ "$rc" -ne 1 ]]; then
# 1 = ordinary false; anything else = broken condition. Say so, retire.
say_log "$NAME" "condition exited rc=$rc (not 0/1) — broken; stopping"
"$SEND" ${SOCKET:+-L "$SOCKET"} -s "$SESSION" -C terminal-log \
-m "[watch:$NAME] condition broken (rc=$rc), watcher stopped: $CONDITION" >/dev/null 2>&1 || true
mark_terminal "$NAME" "condition broken (rc=$rc)"
systemctl --user stop "$unit.timer" >/dev/null 2>&1 || true
exit 5
fi
return 0
}
# ── stale detection (D62) ─────────────────────────────────────────────────
# Config present + timer gone + no terminal marker = the watch will never
# fire and nobody was told. Claim-first (review B2): the stale-noticed marker
# is created atomically (noclobber) BEFORE any send, so exactly one of N
# concurrent callers sends and the others return immediately. Crash semantics
# are AT-MOST-ONCE per episode: a crash between claim and successful send
# drops that episode's notice (the claim survives, preventing a later
# duplicate); staleness itself stays visible in list/status output and exit
# code 3 regardless. Delivery follows fired-watch semantics: rc=0/2 counts as
# delivered and the claim is held; exhausted retries release the claim so the
# next detection tries again.
notice_stale_once() {
local name="$1"
local dir; dir="$(state_dir "$name")"
[[ -r "$dir/config" ]] || return 0
[[ -e "$dir/terminated" ]] && return 0 # expected death: owed nothing
if ! ( set -o noclobber; printf '%s\n' "$(date -u +%FT%TZ)" > "$dir/stale-noticed" ) 2>/dev/null; then
return 0 # another caller holds the claim for this episode
fi
say_log "$name" "stale claim acquired (config present, timer gone, not terminal)"
# shellcheck disable=SC1090
source "$dir/config"
# Post-claim re-check: if the timer recovered while we raced for the claim,
# the episode ended. Release the claim without sending.
# T24: the re-check itself can fail (bus dropped since entry) — an EMPTY
# answer from systemctl must not read as "still gone": release the claim
# (nothing was verified, nothing sent) and fail loud rather than fabricate
# a LOST notice on an unreadable system.
local rerc=0 reout
reout="$(systemctl --user is-active "$(unit_for "$name").timer" 2>/dev/null)" || rerc=$?
if [[ "$rerc" -ne 0 && -z "$reout" ]]; then
rm -f "$dir/stale-noticed"
say_log "$name" "user bus unreachable at post-claim re-check; claim released, NO notice sent"
require_user_bus # prints the diagnostic, exits rc $RC_NO_BUS
fi
if [[ "$rerc" -eq 0 ]]; then
rm -f "$dir/stale-noticed"
say_log "$name" "timer recovered after claim; notice suppressed, claim released"
return 0
fi
local tries=0 drc=1
while :; do
set +e
"$SEND" ${SOCKET:+-L "$SOCKET"} -s "$SESSION" -C actionable \
-m "[watch:$NAME] LOST: its timer is gone but the watch state remains; it will never fire like this. Re-arm (agent-watch.sh start) or stop it (agent-watch.sh stop $NAME)."
drc=$?
set -e
[[ "$drc" -eq 0 || "$drc" -eq 2 ]] && break
tries=$((tries + 1)); [[ "$tries" -ge 3 ]] && break
sleep 5
done
if [[ "$drc" -eq 0 || "$drc" -eq 2 ]]; then
say_log "$NAME" "stale notice delivered (rc=$drc); claim held"
else
rm -f "$dir/stale-noticed" # claim released: nothing was delivered
say_log "$NAME" "stale notice delivery FAILED rc=$drc after 3 attempts — claim released, will retry on next detection"
fi
}
# ── shared classification (T19) ──────────────────────────────────────────────
# One classification source for list/status/_scan. T16W2 semantics preserved
# exactly: an observed-alive timer releases the episode claim ONLY — a timer
# mid-stop can still report active while mark_terminal's stop is in flight, so
# activity must never clear `terminated` (only start/re-arm does).
classify_watch() { # $1=name -> echoes alive|retired|stale
local dir; dir="$(state_dir "$1")"
if systemctl --user is-active "$(unit_for "$1").timer" >/dev/null 2>&1; then
rm -f "$dir/stale-noticed" 2>/dev/null || true
echo alive
elif [[ -e "$dir/terminated" ]]; then
echo retired
else
echo stale
fi
}
# ── meta-watch (T19/D62): detection without a query ──────────────────────────
# The scan path is nothing new: it is the SAME classification + claim-first
# notices list/status run, invoked on a cadence by a persistent, ENABLED
# systemd user timer instead of a human/orchestrator query. A tick is just
# another concurrent caller of the T16 semantics, so tick-vs-query races still
# yield exactly one LOST per episode.
#
# Why this does not recreate D62 one level up:
# - watches are TRANSIENT units: reboot wipes them while their state dirs
# still promise delivery (the measured loss mode). The meta-watch is a
# persistent, enabled unit — reboot cannot strand it; it returns with
# timers.target.
# - it holds no per-obligation state. Its only artifact is a heartbeat that
# AGES: a dead meta-watch leaves an absence signal, not a silent promise.
# - list/status surface meta health (timer active + heartbeat age vs
# cadence) on the query surface that already exists, with a JSON field
# for machine consumption. Pre-T19 a dead watch was invisible even when
# queried; post-T19 a dead meta-watch is visible whenever queried.
# - a same-host meta-meta-watch would share fate with the meta (one systemd
# user instance) and add nothing. The honest terminal for liveness is
# off-host dead-man supervision (WAKE-DOCTRINE); out of scope here.
meta_conf() { echo "$STATE_ROOT/meta-watch.conf"; } # flat files on purpose:
meta_heartbeat() { echo "$STATE_ROOT/meta-watch.heartbeat"; } # scan loop reads dirs only
meta_state() { # rc 0 = installed (globals below set) · rc 2 = not installed
MU_UNIT=""; MU_INTERVAL=0; MU_TIMER="inactive"; MU_AGE="none"; MU_HEALTHY=0
local conf; conf="$(meta_conf)"
[[ -r "$conf" ]] || return 2
# shellcheck disable=SC1090
source "$conf"
MU_UNIT="${META_UNIT:-$META_UNIT_DEFAULT}"
MU_INTERVAL="${META_INTERVAL:-300}"
if systemctl --user is-active "$MU_UNIT.timer" >/dev/null 2>&1; then MU_TIMER="active"; fi
local hb; hb="$(meta_heartbeat)"
if [[ -r "$hb" ]]; then
MU_AGE=$(( $(date +%s) - $(stat -c %Y "$hb") ))
if (( MU_AGE < 0 )); then MU_AGE=0; fi
fi
local limit=$(( MU_INTERVAL * 2 + 60 ))
if [[ "$MU_TIMER" == "active" && "$MU_AGE" != "none" && "$MU_AGE" -le "$limit" ]]; then
MU_HEALTHY=1
fi
return 0
}
meta_line() { # human one-liner for list/status output
if meta_state; then
if [[ "$MU_HEALTHY" -eq 1 ]]; then
echo "meta-watch: healthy (unit $MU_UNIT timer active, heartbeat ${MU_AGE}s old, cadence ${MU_INTERVAL}s)"
else
local age="$MU_AGE"
if [[ "$MU_AGE" == "none" ]]; then age="never"; fi
echo "meta-watch: DEAD (unit $MU_UNIT timer=$MU_TIMER, heartbeat $age, cadence ${MU_INTERVAL}s) — autonomous detection DOWN; queries still work"
fi
else
echo "meta-watch: not installed (no autonomous detection; agent-watch.sh meta-install)"
fi
}
# ── list / status / stop / log ───────────────────────────────────────────────
cmd_list() {
require_user_bus # T24: an unreachable bus reads as "all timers gone" — refuse before classifying
echo "active watches (transient timers):"
systemctl --user list-timers --all --no-legend "${UNIT_PREFIX}-*.timer" 2>/dev/null || true
local stale=() retired=() d n
for d in "$STATE_ROOT"/*/; do
[[ -d "$d" ]] || continue
n="$(basename "$d")"
[[ -r "$d/config" ]] || continue
case "$(classify_watch "$n")" in
retired) retired+=("$n ($(cat "$d/terminated" 2>/dev/null || echo '?'))") ;;
stale) stale+=("$n") ;;
esac
done
# Print the collected sections first, then do notification work: a slow or
# failing sender must not bury the inventory (mirrors cmd_status, S2).
local x
if [[ ${#retired[@]} -gt 0 ]]; then
echo "retired (expected stop; see log <name>):"
for x in ${retired[@]+"${retired[@]}"}; do echo " $x"; done
fi
if [[ ${#stale[@]} -gt 0 ]]; then
echo "stale state (config present, timer gone):"
for x in ${stale[@]+"${stale[@]}"}; do echo " $x"; done
fi
echo "$(meta_line)" # T19: meta-watch health rides the same surface
for x in ${stale[@]+"${stale[@]}"}; do
notice_stale_once "$x"
done
}
# Machine-readable liveness: one call, JSON or human, exit 0 = none stale,
# exit 3 = one or more stale. Never a substitute for arming real watches; it
# answers "did my watches survive" (WAKE-DOCTRINE: check the instrument).
cmd_status() {
# S1: status takes at most one option, --json, and nothing else.
[[ $# -le 1 ]] || die "status: takes at most one option (--json), got: $*"
local json=0
case "${1:-}" in
"") ;;
--json) json=1 ;;
*) die "status: unknown argument: $1 (usage: status [--json])" ;;
esac
require_user_bus # T24: an unreachable bus reads as "all timers gone" — refuse before classifying
local alive=() stale=() retired=() d n
for d in "$STATE_ROOT"/*/; do
[[ -d "$d" ]] || continue
[[ -r "$d/config" ]] || continue
n="$(basename "$d")"
case "$(classify_watch "$n")" in
alive) alive+=("$n") ;;
retired) retired+=("$n") ;;
stale) stale+=("$n") ;;
esac
done
# Print the collected answer FIRST (S2), then do notification work: a slow
# or failing sender must not delay the JSON/human liveness answer.
# T19: meta-watch health rides the same machine answer (additive schema-1
# field; consumers ignoring unknown fields are unaffected).
local meta_installed=0
if meta_state; then meta_installed=1; fi
if [[ "$json" -eq 1 ]]; then
local ja="" js="" jr=""
[[ "${#alive[@]}" -gt 0 ]] && ja="$(printf '"%s",' "${alive[@]}" | sed 's/,$//')"
[[ "${#stale[@]}" -gt 0 ]] && js="$(printf '"%s",' "${stale[@]}" | sed 's/,$//')"
[[ "${#retired[@]}" -gt 0 ]] && jr="$(printf '"%s",' "${retired[@]}" | sed 's/,$//')"
local mj
if [[ "$meta_installed" -eq 1 ]]; then
local m_h="false" m_t="false"
if [[ "$MU_HEALTHY" -eq 1 ]]; then m_h="true"; fi
if [[ "$MU_TIMER" == "active" ]]; then m_t="true"; fi
local m_age="$MU_AGE"
if [[ "$MU_AGE" == "none" ]]; then m_age="null"; fi
mj=$(printf '"installed":true,"healthy":%s,"timer_active":%s,"heartbeat_age_s":%s,"interval_s":%s,"unit":"%s"' \
"$m_h" "$m_t" "$m_age" "$MU_INTERVAL" "$MU_UNIT")
else
mj='"installed":false'
fi
printf '{"schema":1,"total":%d,"alive":[%s],"stale":[%s],"retired":[%s],"meta":{%s}}\n' \
"$(( ${#alive[@]} + ${#stale[@]} + ${#retired[@]} ))" "$ja" "$js" "$jr" "$mj"
else
echo "alive: ${#alive[@]}${alive[@]:+ (${alive[*]})}"
echo "stale: ${#stale[@]}${stale[@]:+ (${stale[*]})}"
echo "retired: ${#retired[@]}${retired[@]:+ (${retired[*]})}"
echo "$(meta_line)"
fi
local s
for s in ${stale[@]+"${stale[@]}"}; do
notice_stale_once "$s"
done
local rc=0
if [[ "${#stale[@]}" -gt 0 ]]; then rc=3; fi
# T19: an installed-but-dead meta-watch is itself a liveness failure —
# fail loud on the instrument, not only on the watches it guards.
if [[ "$meta_installed" -eq 1 && "$MU_HEALTHY" -ne 1 ]]; then rc=3; fi
exit "$rc"
}
cmd_stop() {
local name="$1"
local unit; unit="$(unit_for "$name")"
systemctl --user stop "$unit.timer" "$unit.service" >/dev/null 2>&1 || true
rm -rf "$(state_dir "$name")" 2>/dev/null || true
echo "stopped watcher '$name' (state removed)"
}
cmd_log() {
local name="$1"
local dir; dir="$(state_dir "$name")"
[[ -r "$dir/watch.log" ]] && cat "$dir/watch.log"
echo "--- journal (unit $(unit_for "$name").service) ---"
journalctl --user -u "$(unit_for "$name").service" --no-pager -n 40 2>/dev/null | tail -n +2 || true
}
# ── _scan (hidden; run BY the meta-watch service each cadence) ───────────────
# Same classification + claim-first notices as list/status — a meta tick is
# just another concurrent caller of the T16 semantics. Heartbeat is stamped
# LAST: it proves a COMPLETED scan, not a scheduled one (a crashing scan
# leaves the heartbeat aging, which is exactly the dead-meta signal).
cmd_scan() { # $1 = meta unit name (informational, for the journal line)
require_user_bus # T24: a bus-less scan classifies everything LOST — refuse, stamp no heartbeat
local d n stale=()
for d in "$STATE_ROOT"/*/; do
[[ -d "$d" ]] || continue
[[ -r "$d/config" ]] || continue
n="$(basename "$d")"
if [[ "$(classify_watch "$n")" == "stale" ]]; then stale+=("$n"); fi
done
local s
for s in ${stale[@]+"${stale[@]}"}; do
notice_stale_once "$s"
done
printf '%s\n' "$(date -u +%FT%TZ)" > "$(meta_heartbeat)"
echo "agent-watch meta-watch (${1:-$META_UNIT_DEFAULT}) scan: ${#stale[@]} stale, notices attempted"
}
# ── meta-install / meta-remove (explicit; nothing self-installs) ────────────
cmd_meta_install() {
local interval=300 unit="$META_UNIT_DEFAULT"
while [[ $# -gt 0 ]]; do
case "$1" in
--interval) interval="$2"; shift 2 ;;
--unit-name) unit="$2"; shift 2 ;;
*) die "meta-install: unknown argument: $1" ;;
esac
done
[[ "$unit" =~ ^agent-watch-[a-z0-9][a-z0-9-]*$ ]] || die "meta-install: --unit-name must be agent-watch-<lowercase-hyphens> so it stays visible under the agent-watch-* inventory (got: '$unit')"
[[ "$interval" =~ ^[0-9]+$ ]] || die "meta-install: --interval must be a number"
[[ "$interval" -ge "$META_FLOOR_INTERVAL" ]] || die "meta-install: --interval floor is ${META_FLOOR_INTERVAL}s for the meta-watch (got ${interval}s) — a detection net, not a poll"
command -v systemctl >/dev/null 2>&1 || die "systemctl not on PATH (meta-watch requires systemd --user)"
[[ -x "$SEND" ]] || die "sender not found/executable: $SEND"
require_user_bus # T24: installing without the bus writes units the manager never loads
# Collision guard: meta units share the watch-unit namespace; never shadow
# an existing watch's units.
local bare="${unit#agent-watch-}"
[[ ! -r "$(state_dir "$bare")/config" ]] || die "a watch named '$bare' already exists; its units would collide with $unit.*"
# T19R O1: a meta installed under a DIFFERENT unit name must not be
# displaced silently — overwriting the conf strands the old timer (two
# live metas, one heartbeat). Refuse; meta-remove first. A same-name
# re-install is the idempotent repair path (unit files rewritten, timer
# re-enabled and restarted) — preserved below.
if meta_state; then
[[ "$MU_UNIT" == "$unit" ]] || \
die "meta-install: a meta-watch is already installed as '$MU_UNIT.timer' (conf: $(meta_conf)); run 'agent-watch.sh meta-remove' first — installing '$unit' would run two metas on one heartbeat"
fi
local was_active=0
if systemctl --user is-active "$unit.timer" >/dev/null 2>&1; then was_active=1; fi
local self; self="$(readlink -f "$0")"
# Env travel (D62 lesson, same as start): the scan must resolve the SAME
# state root and sender the installing shell used. systemd Environment=
# carries these verbatim, so reject values it cannot (no spaces/quotes).
local v
for v in "$HOME" "${MOSAIC_WATCH_STATE:-}" "${MOSAIC_AGENT_SEND:-}" "$self"; do
[[ -z "$v" || "$v" =~ ^[[:alnum:]_./:=+-]+$ ]] || die "meta-install: value has characters a systemd unit cannot carry verbatim: '$v'"
done
local ud="$HOME/.config/systemd/user"
mkdir -p "$ud"
{
echo "# generated by agent-watch.sh meta-install $(date -u +%FT%TZ); change = re-install, uninstall = meta-remove"
echo "[Unit]"
echo "Description=agent-watch meta-watch: autonomous stale-watch detection (${interval}s cadence)"
echo ""
echo "[Timer]"
echo "OnBootSec=1min" # post-reboot first scan, even though the
echo "OnUnitActiveSec=${interval}s" # service has never run this boot
echo "AccuracySec=5s"
echo ""
echo "[Install]"
echo "WantedBy=timers.target" # enablement survives reboot (the whole point)
} > "$ud/$unit.timer"
{
echo "# generated by agent-watch.sh meta-install $(date -u +%FT%TZ); change = re-install, uninstall = meta-remove"
echo "[Unit]"
echo "Description=agent-watch meta-watch scan (autonomous stale detection)"
echo ""
echo "[Service]"
echo "Type=oneshot"
echo "TimeoutStartSec=10min" # retries (3x5s sleeps) must not trip the default
echo "Environment=HOME=$HOME"
if [[ -n "${MOSAIC_WATCH_STATE:-}" ]]; then echo "Environment=MOSAIC_WATCH_STATE=$MOSAIC_WATCH_STATE"; fi
if [[ -n "${MOSAIC_AGENT_SEND:-}" ]]; then echo "Environment=MOSAIC_AGENT_SEND=$MOSAIC_AGENT_SEND"; fi
echo "ExecStart=$self _scan $unit"
} > "$ud/$unit.service"
mkdir -p "$STATE_ROOT"
printf 'META_UNIT=%q\nMETA_INTERVAL=%s\n' "$unit" "$interval" > "$(meta_conf)"
systemctl --user daemon-reload
systemctl --user enable --now "$unit.timer" >/dev/null
# Repair path (same-name re-install): a literal restart re-arms the timer
# on the freshly written unit files; enable --now alone would leave an
# already-active timer on its old schedule.
if [[ "$was_active" -eq 1 ]]; then
systemctl --user restart "$unit.timer" >/dev/null
fi
# Explicit first scan NOW: fail fast at install time, not one cadence later;
# it also anchors OnUnitActiveSec for steady cadence.
systemctl --user start "$unit.service"
echo "meta-watch installed and enabled: $unit.timer, every ${interval}s (persistent unit: reboot-safe)"
echo "first scan complete; heartbeat: $(meta_heartbeat)"
echo "state root: $STATE_ROOT"
}
cmd_meta_remove() {
local unit=""
while [[ $# -gt 0 ]]; do
case "$1" in
--unit-name) unit="$2"; shift 2 ;;
*) die "meta-remove: unknown argument: $1" ;;
esac
done
if [[ -z "$unit" ]]; then
if [[ -r "$(meta_conf)" ]]; then
# shellcheck disable=SC1090
unit="$( . "$(meta_conf)" && echo "${META_UNIT:-}" )"
fi
[[ -n "$unit" ]] || unit="$META_UNIT_DEFAULT"
fi
systemctl --user disable --now "$unit.timer" >/dev/null 2>&1 || true
systemctl --user reset-failed "$unit.service" "$unit.timer" >/dev/null 2>&1 || true
rm -f "$HOME/.config/systemd/user/$unit.timer" "$HOME/.config/systemd/user/$unit.service"
systemctl --user daemon-reload >/dev/null 2>&1 || true
rm -f "$(meta_conf)" "$(meta_heartbeat)"
echo "meta-watch removed: $unit.{timer,service} uninstalled, meta state cleared"
}
case "${1:-}" in
start) shift; cmd_start "$@" ;;
list) cmd_list ;;
status) shift; cmd_status "$@" ;;
stop) shift; cmd_stop "$1" ;;
log) shift; cmd_log "$1" ;;
meta-install) shift; cmd_meta_install "$@" ;;
meta-remove) shift; cmd_meta_remove "$@" ;;
_tick) shift; cmd_tick "$1" ;;
_scan) shift; cmd_scan "$@" ;;
-h|--help|*) usage ;;
esac