c212099738
Independent Proxmox-host backstop to the in-app dream budget: a systemd timer runs every ~2 min and stops lyra-brain if the MI50 is busy >=1hr continuously OR junction >=97C for ~6 min, then pings Brian via ntfy. Trips on duration only after a full hour so a legit ~40-min manual workload runs untouched. GPU temp/use read from host rocm-smi; stop via 'pct exec 202 -- docker stop'. Parsing + duration/temp decision logic dry-run-verified locally against real rocm-smi output format (4 scenarios). NOT yet installed/live-verified — card is off and Brian's away; install + trip-test per deploy/mi50-watchdog/README.md when it's back. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_015yrEb5qpPGv2FjyxrB7LLk
83 lines
3.3 KiB
Bash
Executable File
83 lines
3.3 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# MI50 runaway watchdog — fallback layer "A".
|
|
#
|
|
# Runs on the Proxmox HOST (10.0.0.4) via a systemd timer (every ~2 min). It is the
|
|
# independent backstop to Lyra's own in-app dream-cycle budget ("C", in lyra/dream.py):
|
|
# if the MI50 is busy too LONG or runs too HOT, it stops the llama.cpp backend and
|
|
# pings Brian — regardless of what caused it. Trips on duration only after a full hour
|
|
# of *continuous* busy, so a legitimate ~40-min manual workload runs untouched.
|
|
#
|
|
# The GPU lives on the host; the llama.cpp container ("lyra-brain") runs inside LXC
|
|
# CT202. So temp/use come from host rocm-smi, and the stop goes via `pct exec`.
|
|
#
|
|
# See docs/superpowers/specs/2026-07-04-mi50-runaway-guards-design.md
|
|
set -uo pipefail
|
|
|
|
# --- tunables (override in the .service via Environment=) ---
|
|
CTID="${CTID:-202}" # LXC holding the docker container
|
|
CONTAINER="${CONTAINER:-lyra-brain}"
|
|
MAX_BUSY_SEC="${MAX_BUSY_SEC:-3600}" # 1 hr continuous busy -> stop
|
|
TEMP_KILL_C="${TEMP_KILL_C:-97}" # junction >= this ...
|
|
TEMP_KILL_STREAK="${TEMP_KILL_STREAK:-3}" # ... for this many consecutive checks (~6 min)
|
|
NTFY_URL="${NTFY_URL:-}" # e.g. https://ntfy.sh (empty => log only)
|
|
NTFY_TOPIC="${NTFY_TOPIC:-}"
|
|
BUSY_STATE="${BUSY_STATE:-/run/mi50-watchdog.busy_since}"
|
|
HOT_STATE="${HOT_STATE:-/run/mi50-watchdog.hot_streak}"
|
|
|
|
now="$(date +%s)"
|
|
|
|
alert() { # $1 title, $2 message
|
|
logger -t mi50-watchdog "$2"
|
|
if [[ -n "$NTFY_URL" && -n "$NTFY_TOPIC" ]]; then
|
|
curl -s -m 8 -H "Title: $1" -H "Priority: urgent" -H "Tags: warning" \
|
|
-d "$2" "$NTFY_URL/$NTFY_TOPIC" >/dev/null 2>&1 || true
|
|
fi
|
|
}
|
|
|
|
stop_backend() { # $1 reason
|
|
pct exec "$CTID" -- docker stop "$CONTAINER" >/dev/null 2>&1 || true
|
|
rm -f "$BUSY_STATE" "$HOT_STATE"
|
|
alert "MI50 watchdog stopped the card" "$1"
|
|
}
|
|
|
|
# Nothing to guard if the backend isn't even running.
|
|
running="$(pct exec "$CTID" -- docker inspect -f '{{.State.Running}}' "$CONTAINER" 2>/dev/null || echo false)"
|
|
if [[ "$running" != "true" ]]; then
|
|
rm -f "$BUSY_STATE" "$HOT_STATE"
|
|
exit 0
|
|
fi
|
|
|
|
use="$(rocm-smi --showuse 2>/dev/null | awk -F: '/GPU use \(%\)/ {gsub(/[^0-9]/, "", $NF); print $NF; exit}')"
|
|
junction="$(rocm-smi --showtemp 2>/dev/null | awk -F: '/junction/ {gsub(/[^0-9.]/, "", $NF); print $NF; exit}')"
|
|
|
|
# --- duration rule: accumulate continuous busy time in a state file ---
|
|
busy=0
|
|
[[ "${use:-}" =~ ^[0-9]+$ ]] && (( use > 0 )) && busy=1
|
|
if (( busy )); then
|
|
[[ -f "$BUSY_STATE" ]] || echo "$now" > "$BUSY_STATE"
|
|
since="$(cat "$BUSY_STATE" 2>/dev/null || echo "$now")"
|
|
elapsed=$(( now - since ))
|
|
if (( elapsed >= MAX_BUSY_SEC )); then
|
|
stop_backend "MI50 busy ${elapsed}s continuously (>= ${MAX_BUSY_SEC}s) — stopped ${CONTAINER}."
|
|
exit 0
|
|
fi
|
|
else
|
|
rm -f "$BUSY_STATE" # idle breaks the streak
|
|
fi
|
|
|
|
# --- temperature rule: independent of duration ---
|
|
if [[ "${junction:-}" =~ ^[0-9.]+$ ]]; then
|
|
jint="${junction%.*}"
|
|
if (( jint >= TEMP_KILL_C )); then
|
|
streak=$(( $(cat "$HOT_STATE" 2>/dev/null || echo 0) + 1 ))
|
|
echo "$streak" > "$HOT_STATE"
|
|
if (( streak >= TEMP_KILL_STREAK )); then
|
|
stop_backend "MI50 junction ${jint}C >= ${TEMP_KILL_C}C for ${streak} checks — stopped ${CONTAINER}."
|
|
exit 0
|
|
fi
|
|
else
|
|
rm -f "$HOT_STATE" # cooled off, reset the streak
|
|
fi
|
|
fi
|
|
exit 0
|