#!/usr/bin/env bash # MI50 runaway watchdog — fallback layer "A". # # Runs on the Proxmox HOST (10.0.0.4) via a systemd timer (every ~2 min). It is the # independent backstop to Lyra's own in-app dream-cycle budget ("C", in lyra/dream.py): # if the MI50 is busy too LONG or runs too HOT, it stops the llama.cpp backend and # pings Brian — regardless of what caused it. Trips on duration only after a full hour # of *continuous* busy, so a legitimate ~40-min manual workload runs untouched. # # The GPU lives on the host; the llama.cpp container ("lyra-brain") runs inside LXC # CT202. So temp/use come from host rocm-smi, and the stop goes via `pct exec`. # # See docs/superpowers/specs/2026-07-04-mi50-runaway-guards-design.md set -uo pipefail # --- tunables (override in the .service via Environment=) --- CTID="${CTID:-202}" # LXC holding the docker container CONTAINER="${CONTAINER:-lyra-brain}" MAX_BUSY_SEC="${MAX_BUSY_SEC:-3600}" # 1 hr continuous busy -> stop TEMP_KILL_C="${TEMP_KILL_C:-97}" # junction >= this ... TEMP_KILL_STREAK="${TEMP_KILL_STREAK:-3}" # ... for this many consecutive checks (~6 min) NTFY_URL="${NTFY_URL:-}" # e.g. https://ntfy.sh (empty => log only) NTFY_TOPIC="${NTFY_TOPIC:-}" BUSY_STATE="${BUSY_STATE:-/run/mi50-watchdog.busy_since}" HOT_STATE="${HOT_STATE:-/run/mi50-watchdog.hot_streak}" now="$(date +%s)" alert() { # $1 title, $2 message logger -t mi50-watchdog "$2" if [[ -n "$NTFY_URL" && -n "$NTFY_TOPIC" ]]; then curl -s -m 8 -H "Title: $1" -H "Priority: urgent" -H "Tags: warning" \ -d "$2" "$NTFY_URL/$NTFY_TOPIC" >/dev/null 2>&1 || true fi } stop_backend() { # $1 reason pct exec "$CTID" -- docker stop "$CONTAINER" >/dev/null 2>&1 || true rm -f "$BUSY_STATE" "$HOT_STATE" alert "MI50 watchdog stopped the card" "$1" } # Nothing to guard if the backend isn't even running. running="$(pct exec "$CTID" -- docker inspect -f '{{.State.Running}}' "$CONTAINER" 2>/dev/null || echo false)" if [[ "$running" != "true" ]]; then rm -f "$BUSY_STATE" "$HOT_STATE" exit 0 fi use="$(rocm-smi --showuse 2>/dev/null | awk -F: '/GPU use \(%\)/ {gsub(/[^0-9]/, "", $NF); print $NF; exit}')" junction="$(rocm-smi --showtemp 2>/dev/null | awk -F: '/junction/ {gsub(/[^0-9.]/, "", $NF); print $NF; exit}')" # --- duration rule: accumulate continuous busy time in a state file --- busy=0 [[ "${use:-}" =~ ^[0-9]+$ ]] && (( use > 0 )) && busy=1 if (( busy )); then [[ -f "$BUSY_STATE" ]] || echo "$now" > "$BUSY_STATE" since="$(cat "$BUSY_STATE" 2>/dev/null || echo "$now")" elapsed=$(( now - since )) if (( elapsed >= MAX_BUSY_SEC )); then stop_backend "MI50 busy ${elapsed}s continuously (>= ${MAX_BUSY_SEC}s) — stopped ${CONTAINER}." exit 0 fi else rm -f "$BUSY_STATE" # idle breaks the streak fi # --- temperature rule: independent of duration --- if [[ "${junction:-}" =~ ^[0-9.]+$ ]]; then jint="${junction%.*}" if (( jint >= TEMP_KILL_C )); then streak=$(( $(cat "$HOT_STATE" 2>/dev/null || echo 0) + 1 )) echo "$streak" > "$HOT_STATE" if (( streak >= TEMP_KILL_STREAK )); then stop_backend "MI50 junction ${jint}C >= ${TEMP_KILL_C}C for ${streak} checks — stopped ${CONTAINER}." exit 0 fi else rm -f "$HOT_STATE" # cooled off, reset the streak fi fi exit 0