mirror of
https://github.com/myronblair/jarvis
synced 2026-07-28 08:43:00 -05:00
Compare commits
2 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 8399048252 | |||
| 652e44f7b3 |
@@ -29,6 +29,14 @@ if mysqldump -u"$DB_USER" -p"$DB_PASS" "$DB_NAME" > "$TMPDIR/jarvis_db.sql" 2>>"
|
||||
cp -a /etc/nginx/sites-enabled/jarvis "$TMPDIR/files/etc/nginx-site-jarvis" 2>>"$LOG"
|
||||
cp -a /etc/systemd/system/jarvis-arc.service "$TMPDIR/files/etc/jarvis-arc.service" 2>>"$LOG"
|
||||
crontab -l > "$TMPDIR/files/etc/root-crontab.txt" 2>>"$LOG"
|
||||
# Phase 1/2 additions: secrets + systemd drop-ins + operational scripts
|
||||
# (these live outside /var/www/jarvis and /opt/jarvis-arc, so must be
|
||||
# captured explicitly or a restore comes back with no keys/DB pass).
|
||||
cp -a /etc/jarvis-arc "$TMPDIR/files/etc/jarvis-arc-etc" 2>>"$LOG" # reactor.env
|
||||
cp -a /etc/jarvis "$TMPDIR/files/etc/jarvis-etc" 2>>"$LOG" # db.env
|
||||
cp -a /etc/systemd/system/jarvis-arc.service.d "$TMPDIR/files/etc/jarvis-arc.service.d" 2>>"$LOG"
|
||||
mkdir -p "$TMPDIR/files/usr-local-bin"
|
||||
cp -a /usr/local/bin/jarvis-*.sh "$TMPDIR/files/usr-local-bin/" 2>>"$LOG" # health/deploy/watchdog/netscan
|
||||
|
||||
tar -czf "$OUTFILE" -C "$TMPDIR" jarvis_db.sql files
|
||||
SIZE=$(du -sh "$OUTFILE" | cut -f1)
|
||||
|
||||
@@ -0,0 +1,95 @@
|
||||
#!/bin/bash
|
||||
# JARVIS Health Self-Check — runs every 5 min via root cron (separate from the
|
||||
# service watchdog). Detects silent failures the watchdog can't: stalled crons,
|
||||
# stuck Arc jobs, low disk, and services the watchdog had to restart. Writes
|
||||
# findings to the `alerts` table (auto-resolving when clear) and emails on any
|
||||
# NEW finding. Added 2026-07-07 (Phase 2 reliability).
|
||||
set -u
|
||||
[ -r /etc/jarvis/db.env ] && . /etc/jarvis/db.env
|
||||
DB_USER="jarvis_user"; DB_NAME="jarvis_db"
|
||||
MYSQL=(mysql -u "$DB_USER" -p"${JARVIS_DB_PASS:-}" "$DB_NAME" -N -B -e)
|
||||
CRONLOG=/var/log/jarvis/cron.log
|
||||
WDLOG=/var/log/jarvis/watchdog.log
|
||||
ALERT_TO="myronblair@gmail.com"
|
||||
REACTOR_ENV=/etc/jarvis-arc/reactor.env
|
||||
NEW_FINDINGS=""
|
||||
|
||||
sql() { "${MYSQL[@]}" "$1" 2>/dev/null; }
|
||||
esc() { printf '%s' "$1" | sed "s/'/''/g"; }
|
||||
|
||||
# Raise (or keep) an alert for a condition. Emails only when it is newly raised.
|
||||
# $1=source_key $2=severity $3=title $4=message
|
||||
raise() {
|
||||
local key sev title msg exists
|
||||
key=$(esc "$1"); sev=$(esc "$2"); title=$(esc "$3"); msg=$(esc "$4")
|
||||
exists=$(sql "SELECT COUNT(*) FROM alerts WHERE source_key='$key' AND resolved=0")
|
||||
if [ "${exists:-0}" = "0" ]; then
|
||||
sql "INSERT INTO alerts (alert_type,title,message,severity,source_key,auto_resolve,created_at)
|
||||
VALUES ('health','$title','$msg','$sev','$key',1,NOW())"
|
||||
NEW_FINDINGS="${NEW_FINDINGS}- [$2] $3: $4"$'\n'
|
||||
fi
|
||||
}
|
||||
# Clear a condition's alert when it's no longer true.
|
||||
clear_cond() {
|
||||
local key; key=$(esc "$1")
|
||||
sql "UPDATE alerts SET resolved=1, resolved_at=NOW()
|
||||
WHERE source_key='$key' AND resolved=0 AND auto_resolve=1"
|
||||
}
|
||||
|
||||
# 1) Disk usage on / > 85%
|
||||
DISK=$(df / | tail -1 | awk '{print $5}' | tr -d '%')
|
||||
if [ "${DISK:-0}" -gt 85 ]; then
|
||||
raise "health:disk" "critical" "Disk usage high" "Root filesystem at ${DISK}% (threshold 85%)."
|
||||
else clear_cond "health:disk"; fi
|
||||
|
||||
# 2) Arc jobs stuck in 'running' > 30 min
|
||||
STUCK=$(sql "SELECT COUNT(*) FROM arc_jobs WHERE status='running'
|
||||
AND COALESCE(started_at, created_at) < NOW() - INTERVAL 30 MINUTE")
|
||||
if [ "${STUCK:-0}" -gt 0 ]; then
|
||||
raise "health:arc_stuck" "critical" "Arc jobs stuck" "$STUCK Arc job(s) have been 'running' for over 30 minutes."
|
||||
else clear_cond "health:arc_stuck"; fi
|
||||
|
||||
# 3) Cron stalled — cron.log is written by facts_collector every 3 min. If it
|
||||
# hasn't changed in 10 min (>2 consecutive missed runs), cron work has stopped.
|
||||
if [ -f "$CRONLOG" ]; then
|
||||
AGE=$(( $(date +%s) - $(stat -c %Y "$CRONLOG") ))
|
||||
if [ "$AGE" -gt 600 ]; then
|
||||
raise "health:cron" "critical" "Cron jobs stalled" "No cron activity in $((AGE/60)) min (facts_collector runs every 3 min) — crons appear stopped."
|
||||
else clear_cond "health:cron"; fi
|
||||
fi
|
||||
|
||||
# 4) Watched service restarted by the watchdog in the last 6 min
|
||||
if [ -f "$WDLOG" ]; then
|
||||
RECENT=$(awk -v cutoff="$(date -d '6 minutes ago' '+%Y-%m-%d %H:%M:%S')" \
|
||||
'match($0,/^\[([0-9-]+ [0-9:]+)\]/,m){ if(m[1]>=cutoff && /restarted successfully/) print }' "$WDLOG")
|
||||
if [ -n "$RECENT" ]; then
|
||||
SVC=$(printf '%s' "$RECENT" | grep -oE '(nginx|php8.3-fpm|mariadb|redis-server)' | sort -u | tr '\n' ' ')
|
||||
raise "health:wd_restart:$(date +%Y%m%d%H%M)" "warning" "Service auto-restarted" "Watchdog restarted: ${SVC:-a service}. Investigate why it died."
|
||||
fi
|
||||
fi
|
||||
|
||||
# Email any NEW findings via the reactor's Gmail SMTP creds.
|
||||
if [ -n "$NEW_FINDINGS" ]; then
|
||||
GPASS=""
|
||||
[ -r "$REACTOR_ENV" ] && GPASS=$(grep -E '^GMAIL_PASS=' "$REACTOR_ENV" | cut -d= -f2-)
|
||||
if [ -n "$GPASS" ]; then
|
||||
GMAIL_PASS="$GPASS" ALERT_TO="$ALERT_TO" FINDINGS="$NEW_FINDINGS" python3 - <<'PY'
|
||||
import os, smtplib, ssl, socket
|
||||
from email.mime.text import MIMEText
|
||||
user = "myronblair@gmail.com"
|
||||
msg = MIMEText("JARVIS health self-check raised new alerts on %s:\n\n%s" % (socket.gethostname(), os.environ["FINDINGS"]))
|
||||
msg["Subject"] = "JARVIS health alert"
|
||||
msg["From"] = user; msg["To"] = os.environ["ALERT_TO"]
|
||||
try:
|
||||
with smtplib.SMTP("smtp.gmail.com", 587, timeout=20) as s:
|
||||
s.starttls(context=ssl.create_default_context())
|
||||
s.login(user, os.environ["GMAIL_PASS"])
|
||||
s.send_message(msg)
|
||||
print("health-email: sent")
|
||||
except Exception as e:
|
||||
print("health-email: FAILED", e)
|
||||
PY
|
||||
else
|
||||
echo "health-email: no GMAIL_PASS available, skipped"
|
||||
fi
|
||||
fi
|
||||
Reference in New Issue
Block a user