← back to Exo Cluster Watchdog
watchdog.sh
252 lines
#!/bin/bash
# exo-cluster-watchdog — "NEVER allow exo to just be off."
# Runs every 2 min (launchd com.steve.exo-cluster-watchdog).
# For each node: probe exo :52415. If DOWN and reachable, RESTART it.
# Local node -> kickstart its own exo-keepalive. Peers -> ssh in and heal.
# Anything unreachable/unrecoverable -> loud alert + FAIL/WARN heartbeat.
#
# Heartbeat verdict uses the fleet-health vocabulary (PASS/WARN/FAIL) so
# fleet-health-rollup reads it correctly (per TK-10546 rule).
set -uo pipefail
DIR="$HOME/Projects/exo-cluster-watchdog"
DATA="$DIR/data"
LATEST="$DATA/latest.json"
ALERTS="$DATA/alerts.log"
PORT=52415
mkdir -p "$DATA"
# Node table: name | host | ssh_target(LOCAL | user@host | NOAUTH)
# NOAUTH = do NOT ssh this node (key not authorized / avoid touching it); probe+alert only.
NODES=(
"macstudio3|127.0.0.1|LOCAL"
"mac1|stevesmcstudio2.lan|steve@stevesmcstudio2.lan"
"mac2|macstudio1.lan|NOAUTH"
)
EXPECTED=${#NODES[@]}
now() { date -u +%Y-%m-%dT%H:%M:%SZ; }
exo_up() { curl -s --max-time 4 "http://$1:$PORT/state" >/dev/null 2>&1; }
pings() { ping -c1 -t2 "$1" >/dev/null 2>&1; }
# Fabric detection (LOCAL queries — reliable even if this shell's net egress is sandboxed).
# TB only if a 10.0.0.x address lives on the Thunderbolt Bridge (bridge0) or a Thunderbolt
# hardware en* port — NOT a VPN utun / Colima / Docker bridge that also uses 10.0.0.x.
detect_fabric() {
local ifc
ifc=$(ifconfig 2>/dev/null | awk '/^[a-z0-9]+:/{i=$1; sub(":","",i)} /inet 10\.0\.0\./{print i; exit}')
[ -z "$ifc" ] && { echo LAN; return; }
case "$ifc" in
bridge0) echo TB ;; # macOS Thunderbolt Bridge
en*) networksetup -listallhardwareports 2>/dev/null \
| grep -B1 "Device: $ifc\$" | grep -qi thunderbolt && echo TB || echo LAN ;;
*) echo LAN ;; # utun/vmenet/bridge100/etc = not TB
esac
}
# Remote self-contained heal — HONEST exit codes (only claims success if it really acted):
# 0 = exo already up locally on the peer 7 = kicked its exo-keepalive launchd label
# 8 = started exo directly (keepalive absent, but ~/exo+uv exist for THIS ssh user)
# 9 = CANNOT heal: no keepalive AND this ssh user has no ~/exo (exo likely runs under a
# different account) -> needs harden-peer.sh under the exo-owning user; do NOT fake it.
REMOTE_HEAL='curl -s --max-time 3 http://127.0.0.1:52415/state >/dev/null 2>&1 && exit 0; \
launchctl kickstart -k gui/$(id -u)/com.steve.exo-keepalive 2>/dev/null && exit 7; \
if [ -x "$HOME/.local/bin/uv" ] && [ -d "$HOME/exo" ]; then \
cd "$HOME/exo" && nohup "$HOME/.local/bin/uv" run --extra mlx exo >/tmp/exo-watchdog.out 2>&1 & exit 8; \
fi; exit 9'
serving=0; down=(); healed=(); heal_candidates=(); unreachable=(); notes=()
for row in "${NODES[@]}"; do
IFS='|' read -r name host tgt <<<"$row"
if exo_up "$host"; then
serving=$((serving+1)); continue
fi
# exo is DOWN on this node -> try to heal
if [ "$tgt" = "LOCAL" ]; then
launchctl kickstart -k "gui/$(id -u)/com.steve.exo-keepalive" 2>/dev/null
heal_candidates+=("$name|$host|local-kick"); notes+=("$name exo was down -> local exo-keepalive kicked (pending re-probe)")
elif [ "$tgt" = "NOAUTH" ]; then
down+=("$name"); notes+=("$name exo down — ssh heal DISABLED (key not authorized yet); probe+alert only, NOT touching this box")
elif pings "$host"; then
# host up; try ssh heal (BatchMode so it fails fast if key not authorized)
rc=$(ssh -o BatchMode=yes -o ConnectTimeout=6 -o StrictHostKeyChecking=accept-new "$tgt" "$REMOTE_HEAL" >/dev/null 2>&1; echo $?)
if [ "$rc" = "7" ] || [ "$rc" = "8" ] || [ "$rc" = "0" ]; then
heal_candidates+=("$name|$host|ssh-rc$rc"); notes+=("$name exo down -> ssh heal rc=$rc (pending re-probe)")
elif [ "$rc" = "9" ]; then
down+=("$name"); notes+=("$name exo down + reachable, but NO keepalive and this ssh-user has no ~/exo (exo runs under another account) — run harden-peer.sh as the exo-owning user; NOT faking a heal")
else
down+=("$name"); notes+=("$name UP on LAN but exo down AND ssh heal failed (rc=$rc, no key/authz) — CANNOT self-heal, needs Steve")
fi
else
unreachable+=("$name"); down+=("$name"); notes+=("$name UNREACHABLE (machine off/asleep) — cannot power on remotely")
fi
done
# Re-probe after heal attempts (give exo a moment to bind) to recount serving
sleep 6
serving=0
for row in "${NODES[@]}"; do IFS='|' read -r name host tgt <<<"$row"; exo_up "$host" && serving=$((serving+1)); done
# Resolve heal candidates by a FRESH probe: 'healed' means VERIFIED up, not merely attempted.
# A candidate that still isn't serving becomes 'down' — so a WARN can never ship with a
# populated 'healed' for a node that's actually still down (Cody gate, cycle 1).
for cand in "${heal_candidates[@]:-}"; do
[ -z "$cand" ] && continue
IFS='|' read -r cname chost cvia <<<"$cand"
if exo_up "$chost"; then
healed+=("$cname($cvia)")
else
down+=("$cname"); notes+=("$cname heal attempted ($cvia) but still DOWN on re-probe — NOT counting as healed")
fi
done
# ---- Model-instance keepalive (TK-11903, 2026-09-18) ----
# exo "up" is not the same as "a model is resident". The ring evicts instances under memory
# pressure (two 32GB nodes sit at 1-6GB free); openclaw night jobs then fail every call.
# If the local ring reports 0 instances and data/wanted-instance.json exists, re-POST it.
# Rate-limited (one attempt per 5 min: a fresh load takes ~130s to first token) and NEVER
# counted as healed on the same pass — verdict stays WARN until a later pass sees it resident.
# Test seam: `watchdog.sh --test` reads state from $WD_TEST_STATE and only PRINTS the POST.
TEST_MODE=0; [ "${1:-}" = "--test" ] && TEST_MODE=1
WANTED="$DATA/wanted-instance.json"; RELAUNCH_STAMP="$DATA/.instance-relaunch-ts"
instances=-1
if [ "$TEST_MODE" = 1 ]; then
state_json=$(cat "${WD_TEST_STATE:-/dev/null}" 2>/dev/null)
else
state_json=$(curl -s --max-time 4 "http://127.0.0.1:$PORT/state" 2>/dev/null)
fi
if [ -n "$state_json" ]; then
instances=$(printf '%s' "$state_json" | python3 -c 'import json,sys
try: print(len(json.load(sys.stdin).get("instances") or {}))
except Exception: print(-1)' 2>/dev/null || echo -1)
fi
instance_note=""
if [ "$instances" = "0" ] && [ -s "$WANTED" ]; then
last=$(cat "$RELAUNCH_STAMP" 2>/dev/null || echo 0); nowe=$(date +%s)
if [ $((nowe - last)) -ge 300 ]; then
echo "$nowe" > "$RELAUNCH_STAMP"
wanted_model=$(python3 -c 'import json,sys
def walk(o):
if isinstance(o,dict):
if "modelId" in o: return o["modelId"]
for v in o.values():
r=walk(v)
if r: return r
if isinstance(o,list):
for v in o:
r=walk(v)
if r: return r
print(walk(json.load(open(sys.argv[1]))) or "?")' "$WANTED" 2>/dev/null)
if [ "$TEST_MODE" = 1 ]; then
instance_note="0 instances on ring -> WOULD re-POST $wanted_model (test mode, not sent)"
else
# TK-11996: memory / boot-grace / thrash guard — a 17GB load onto a swapping box hung
# the Mac into power-button resets (09-25 11:35 + 11:52). Unmeasured => BLOCK.
guard=$(bash "$DIR/instance-load-guard.sh" 2>/dev/null || echo "BLOCK: guard failed to run")
if [[ "$guard" != ALLOW* ]]; then
instance_note="0 instances on ring -> re-POST SUPPRESSED ($guard); $wanted_model not loaded"
else
body=$(python3 -c 'import json,sys; print(json.dumps({"instance": json.load(open(sys.argv[1]))}))' "$WANTED")
rc=$(curl -s --max-time 10 -o /dev/null -w '%{http_code}' -X POST "http://127.0.0.1:$PORT/instance" -H 'Content-Type: application/json' -d "$body")
instance_note="0 instances on ring -> re-POSTed $wanted_model (http $rc); pending re-probe, NOT counted as healed"
fi
fi
else
instance_note="0 instances on ring; relaunch attempted $((nowe - last))s ago, waiting for load (no re-POST)"
fi
elif [ "$instances" = "0" ]; then
instance_note="0 instances on ring and no $WANTED to relaunch — needs a human"
elif [ "$instances" = "-1" ]; then
instance_note="instance count NOT MEASURED (state unreadable)"
fi
[ -n "$instance_note" ] && notes+=("$instance_note")
# Verdict (fleet vocab)
self_up=$(exo_up 127.0.0.1 && echo 1 || echo 0)
if [ "$serving" -ge "$EXPECTED" ] && [ "$instances" -ge 1 ]; then
verdict="PASS"
elif [ "$serving" -ge "$EXPECTED" ]; then
verdict="WARN" # nodes serving but NO model instance resident (or not measured) -> never PASS
elif [ "$self_up" = "0" ]; then
verdict="FAIL" # this node's exo is down and didn't recover
else
verdict="WARN" # serving < expected but self is up (peer(s) degraded)
fi
fabric=$(detect_fabric)
downstr=$(IFS=,; echo "${down[*]:-}")
healstr=$(IFS=,; echo "${healed[*]:-}")
head="exo cluster $serving/$EXPECTED serving | instances: $instances | fabric: $fabric"
[ -n "$downstr" ] && head="$head | down: $downstr"
[ -n "$healstr" ] && head="$head | healed: $healstr"
# last_incident: keep flap history visible even when currently PASS (Cody Hole 3, cycle 2).
# non-PASS now -> stamp now; else carry the last recorded incident ts from alerts.log.
if [ "$verdict" != "PASS" ]; then
last_incident="$(now)"
else
last_incident=$(tail -1 "$ALERTS" 2>/dev/null | awk '{print $1}')
fi
# notes -> temp file (one per line) so python encodes them as a proper JSON array (no
# hand-rolled quote escaping of arbitrary note text).
NOTES_TMP="$DATA/.notes.$$"; : > "$NOTES_TMP"
for nz in "${notes[@]:-}"; do [ -n "$nz" ] && printf '%s\n' "$nz" >> "$NOTES_TMP"; done
# Write heartbeat (python3 = correct JSON escaping). verdict/status stay PASS/WARN/FAIL
# (fleet-health-rollup vocabulary); adds notes[] + last_incident for flap observability.
INSTANCES="$instances" VERDICT="$verdict" SERVING="$serving" EXPECTED="$EXPECTED" FABRIC="$fabric" \
SELF_UP="$self_up" DOWNSTR="$downstr" HEALSTR="$healstr" HEAD="$verdict: $head" \
GEN="$(now)" LAST_INCIDENT="${last_incident:-}" NOTES_FILE="$NOTES_TMP" \
python3 - "$LATEST" <<'PY'
import json, os, sys, tempfile
try:
with open(os.environ['NOTES_FILE']) as f:
notes = [l.rstrip('\n') for l in f if l.strip()]
except Exception:
notes = []
d = {
"generated_at": os.environ['GEN'],
"verdict": os.environ['VERDICT'], "status": os.environ['VERDICT'],
"serving": int(os.environ['SERVING']), "expected": int(os.environ['EXPECTED']),
"fabric": os.environ['FABRIC'], "self_up": int(os.environ['SELF_UP']),
"instances": int(os.environ.get('INSTANCES', '-1')),
"down": os.environ['DOWNSTR'], "healed": os.environ['HEALSTR'],
"notes": notes, "last_incident": os.environ.get('LAST_INCIDENT', ''),
"headline": os.environ['HEAD'],
}
# ATOMIC write (Cody cycle 2): temp in same dir -> os.replace, so consumers
# (fleet-health-rollup, dw-canary-meta-watchdog) never read a half-written file.
target = sys.argv[1]
tmp = None
try:
fd, tmp = tempfile.mkstemp(dir=os.path.dirname(target) or '.', prefix='.hb.')
with os.fdopen(fd, 'w') as f:
f.write(json.dumps(d))
os.replace(tmp, target)
except Exception as e:
sys.stderr.write("heartbeat write failed: %s\n" % e)
if tmp:
try: os.unlink(tmp)
except Exception: pass
sys.exit(1) # surface the failure instead of silently shipping garbage
PY
hb_rc=$?
rm -f "$NOTES_TMP"
# Alert on any non-PASS (append + macOS notification). De-nag: only if state changed.
PREVF="$DATA/.prev_verdict"
prev=$(cat "$PREVF" 2>/dev/null || echo "")
echo "$verdict" > "$PREVF"
if [ "$verdict" != "PASS" ]; then
echo "$(now) $verdict :: $head :: ${notes[*]:-}" >> "$ALERTS"
if [ "$verdict" != "$prev" ]; then
osascript -e "display notification \"$head\" with title \"exo watchdog: $verdict\"" >/dev/null 2>&1 || true
fi
fi
echo "$verdict :: $head"
# If the heartbeat couldn't be written, exit non-zero so launchd-job-canary flags it
# (a silent write failure would otherwise look healthy while the heartbeat goes stale).
[ "${hb_rc:-0}" = "0" ] || { echo "WARN: heartbeat write failed (rc=$hb_rc)"; exit 1; }