From 652e44f7b3f4e140397a6c5db7734b2b8a0a6b2a Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 7 Jul 2026 20:43:37 -0500 Subject: [PATCH] Add JARVIS health self-check (Phase 2 reliability monitoring) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit deploy/jarvis-health.sh (cron */5): detects silent failures the service watchdog cannot — stalled crons (cron.log >10min stale = 2+ missed runs), Arc jobs stuck running >30min, disk >85%, and watchdog service restarts. Writes auto-resolving rows to the alerts table (shown in admin panel) and emails myronblair@gmail.com on any NEW finding via the reactor Gmail SMTP creds. Verified live end-to-end: injected a fake stuck job -> alert raised + email sent; cleared it -> alert auto-resolved. Installed in root crontab. Co-Authored-By: Claude Fable 5 --- deploy/jarvis-health.sh | 95 +++++++++++++++++++++++++++++++++++++++++ 1 file changed, 95 insertions(+) create mode 100644 deploy/jarvis-health.sh diff --git a/deploy/jarvis-health.sh b/deploy/jarvis-health.sh new file mode 100644 index 0000000..23922e3 --- /dev/null +++ b/deploy/jarvis-health.sh @@ -0,0 +1,95 @@ +#!/bin/bash +# JARVIS Health Self-Check — runs every 5 min via root cron (separate from the +# service watchdog). Detects silent failures the watchdog can't: stalled crons, +# stuck Arc jobs, low disk, and services the watchdog had to restart. Writes +# findings to the `alerts` table (auto-resolving when clear) and emails on any +# NEW finding. Added 2026-07-07 (Phase 2 reliability). +set -u +[ -r /etc/jarvis/db.env ] && . /etc/jarvis/db.env +DB_USER="jarvis_user"; DB_NAME="jarvis_db" +MYSQL=(mysql -u "$DB_USER" -p"${JARVIS_DB_PASS:-}" "$DB_NAME" -N -B -e) +CRONLOG=/var/log/jarvis/cron.log +WDLOG=/var/log/jarvis/watchdog.log +ALERT_TO="myronblair@gmail.com" +REACTOR_ENV=/etc/jarvis-arc/reactor.env +NEW_FINDINGS="" + +sql() { "${MYSQL[@]}" "$1" 2>/dev/null; } +esc() { printf '%s' "$1" | sed "s/'/''/g"; } + +# Raise (or keep) an alert for a condition. Emails only when it is newly raised. +# $1=source_key $2=severity $3=title $4=message +raise() { + local key sev title msg exists + key=$(esc "$1"); sev=$(esc "$2"); title=$(esc "$3"); msg=$(esc "$4") + exists=$(sql "SELECT COUNT(*) FROM alerts WHERE source_key='$key' AND resolved=0") + if [ "${exists:-0}" = "0" ]; then + sql "INSERT INTO alerts (alert_type,title,message,severity,source_key,auto_resolve,created_at) + VALUES ('health','$title','$msg','$sev','$key',1,NOW())" + NEW_FINDINGS="${NEW_FINDINGS}- [$2] $3: $4"$'\n' + fi +} +# Clear a condition's alert when it's no longer true. +clear_cond() { + local key; key=$(esc "$1") + sql "UPDATE alerts SET resolved=1, resolved_at=NOW() + WHERE source_key='$key' AND resolved=0 AND auto_resolve=1" +} + +# 1) Disk usage on / > 85% +DISK=$(df / | tail -1 | awk '{print $5}' | tr -d '%') +if [ "${DISK:-0}" -gt 85 ]; then + raise "health:disk" "critical" "Disk usage high" "Root filesystem at ${DISK}% (threshold 85%)." +else clear_cond "health:disk"; fi + +# 2) Arc jobs stuck in 'running' > 30 min +STUCK=$(sql "SELECT COUNT(*) FROM arc_jobs WHERE status='running' + AND COALESCE(started_at, created_at) < NOW() - INTERVAL 30 MINUTE") +if [ "${STUCK:-0}" -gt 0 ]; then + raise "health:arc_stuck" "critical" "Arc jobs stuck" "$STUCK Arc job(s) have been 'running' for over 30 minutes." +else clear_cond "health:arc_stuck"; fi + +# 3) Cron stalled — cron.log is written by facts_collector every 3 min. If it +# hasn't changed in 10 min (>2 consecutive missed runs), cron work has stopped. +if [ -f "$CRONLOG" ]; then + AGE=$(( $(date +%s) - $(stat -c %Y "$CRONLOG") )) + if [ "$AGE" -gt 600 ]; then + raise "health:cron" "critical" "Cron jobs stalled" "No cron activity in $((AGE/60)) min (facts_collector runs every 3 min) — crons appear stopped." + else clear_cond "health:cron"; fi +fi + +# 4) Watched service restarted by the watchdog in the last 6 min +if [ -f "$WDLOG" ]; then + RECENT=$(awk -v cutoff="$(date -d '6 minutes ago' '+%Y-%m-%d %H:%M:%S')" \ + 'match($0,/^\[([0-9-]+ [0-9:]+)\]/,m){ if(m[1]>=cutoff && /restarted successfully/) print }' "$WDLOG") + if [ -n "$RECENT" ]; then + SVC=$(printf '%s' "$RECENT" | grep -oE '(nginx|php8.3-fpm|mariadb|redis-server)' | sort -u | tr '\n' ' ') + raise "health:wd_restart:$(date +%Y%m%d%H%M)" "warning" "Service auto-restarted" "Watchdog restarted: ${SVC:-a service}. Investigate why it died." + fi +fi + +# Email any NEW findings via the reactor's Gmail SMTP creds. +if [ -n "$NEW_FINDINGS" ]; then + GPASS="" + [ -r "$REACTOR_ENV" ] && GPASS=$(grep -E '^GMAIL_PASS=' "$REACTOR_ENV" | cut -d= -f2-) + if [ -n "$GPASS" ]; then + GMAIL_PASS="$GPASS" ALERT_TO="$ALERT_TO" FINDINGS="$NEW_FINDINGS" python3 - <<'PY' +import os, smtplib, ssl, socket +from email.mime.text import MIMEText +user = "myronblair@gmail.com" +msg = MIMEText("JARVIS health self-check raised new alerts on %s:\n\n%s" % (socket.gethostname(), os.environ["FINDINGS"])) +msg["Subject"] = "JARVIS health alert" +msg["From"] = user; msg["To"] = os.environ["ALERT_TO"] +try: + with smtplib.SMTP("smtp.gmail.com", 587, timeout=20) as s: + s.starttls(context=ssl.create_default_context()) + s.login(user, os.environ["GMAIL_PASS"]) + s.send_message(msg) + print("health-email: sent") +except Exception as e: + print("health-email: FAILED", e) +PY + else + echo "health-email: no GMAIL_PASS available, skipped" + fi +fi