#!/usr/bin/env bash
# wakelai-watchdog — detects a hung wakelai-nextjs process (event loop stalled,
# listening but not serving) and restarts it, capturing diagnostics first.
# Invoked every minute by wakelai-watchdog.timer.
set -uo pipefail

SERVICE="wakelai-nextjs.service"
URL="http://127.0.0.1:9100/"
TIMEOUT=8            # per-request seconds; a healthy app answers in <1s
RECHECK_DELAY=6     # gap between the two confirmation probes
LOGDIR="/var/log/wakelai-watchdog"
REPORTDIR="/var/log/wakelai-reports"
LOG="$LOGDIR/watchdog.log"

log() { echo "$(date '+%Y-%m-%d %H:%M:%S') $*" >>"$LOG"; }

probe() { # 0 = healthy
  local code
  code=$(curl -s -o /dev/null -w '%{http_code}' --max-time "$TIMEOUT" "$URL" 2>/dev/null)
  [[ "$code" =~ ^(200|30[0-9])$ ]]
}

# Don't act while systemd is (re)starting the service itself.
state=$(systemctl show -p ActiveState --value "$SERVICE" 2>/dev/null)
if [[ "$state" == "activating" || "$state" == "deactivating" ]]; then
  exit 0
fi

probe && exit 0                       # healthy on first try — the common case
sleep "$RECHECK_DELAY"
probe && exit 0                       # transient blip — recovered on recheck

# Confirmed unresponsive twice. Capture what it was doing, then restart.
PID=$(systemctl show -p MainPID --value "$SERVICE" 2>/dev/null)
log "UNRESPONSIVE (pid=$PID). Capturing diagnostics before restart."
{
  echo "==== $(date '+%F %T') watchdog capture, pid=$PID ===="
  ps -o pid,ppid,%cpu,%mem,etime,stat,cmd -p "$PID" 2>/dev/null
  echo "--- ss backlog on :9100 ---"; ss -ltnp 2>/dev/null | grep ':9100'
} >>"$LOGDIR/hang-$(date '+%Y%m%d-%H%M%S').txt" 2>&1

if [[ -n "$PID" && "$PID" != "0" ]]; then
  kill -USR2 "$PID" 2>/dev/null && log "sent SIGUSR2 -> diagnostic report to $REPORTDIR"
  sleep 3   # give Node a moment to flush the report if the loop yields
fi

log "restarting $SERVICE"
systemctl restart "$SERVICE"
sleep 8
if probe; then
  log "RECOVERED: $SERVICE healthy after restart"
else
  log "STILL UNHEALTHY after restart — needs manual attention"
fi
