#!/bin/bash
# Hermes instance health checker
# Checks all RUNNING instances and reports unhealthy ones
# Run via cron every 5 minutes: */5 * * * * /home/ashraffarid2010/wakelai.com/scripts/check-instance-health.sh

set -e

LOG_FILE="/home/ashraffarid2010/wakelai.com/logs/instance-health.log"
DB="postgresql://wakelai:wakelai_password@localhost:55433/wakelai"

# Ensure log directory exists
mkdir -p "$(dirname "$LOG_FILE")"

echo "[$(date '+%Y-%m-%d %H:%M:%S')] Checking instance health..." >> "$LOG_FILE"

# Get all RUNNING instances with their deploy_payload
PGPASSWORD=wakelai_password psql -h localhost -p 55433 -U wakelai -t -A -F'|' -d wakelai -c "SELECT id, slug, subdomain, deploy_payload::text FROM instances WHERE status = 'RUNNING' AND deployment_mode = 'container' AND deploy_payload IS NOT NULL" 2>/dev/null | while IFS='|' read -r id slug subdomain payload; do
    if [ -z "$slug" ]; then
        continue
    fi

    # Extract ports: try deploy_payload first, then fall back to nginx config
    dashboard_port=$(echo "$payload" | python3 -c "
import sys,json
d=json.load(sys.stdin)
dp = d.get('dashboardPort')
if dp is None and 'raw' in d:
    dp = d['raw'].get('dashboardPort')
print(dp if dp is not None else '')
" 2>/dev/null || echo "")
    gateway_port=$(echo "$payload" | python3 -c "
import sys,json
d=json.load(sys.stdin)
gp = d.get('gatewayPort')
if gp is None and 'raw' in d:
    gp = d['raw'].get('gatewayPort')
print(gp if gp is not None else '')
" 2>/dev/null || echo "")

    # Fall back to parsing the nginx config if payload doesn't have the ports
    if [ -z "$dashboard_port" ]; then
        nginx_conf="/etc/nginx/conf.d/instance-${slug}.conf"
        if [ -f "$nginx_conf" ]; then
            dashboard_port=$(grep -oP 'proxy_pass\s+http://127\.0\.0\.1:\K\d+' "$nginx_conf" | head -1)
            gateway_port=$(grep -oP 'location /v1/.*proxy_pass\s+http://127\.0\.0\.1:\K\d+' "$nginx_conf" | head -1)
        fi
    fi

    # If still unknown, use defaults
    dashboard_port=${dashboard_port:-4860}
    gateway_port=${gateway_port:-8642}

    # Check if ports are numeric
    if ! [[ "$dashboard_port" =~ ^[0-9]+$ ]] || ! [[ "$gateway_port" =~ ^[0-9]+$ ]]; then
        echo "  WARNING: Instance $slug ($subdomain) has invalid port config: dashboard=$dashboard_port gateway=$gateway_port" >> "$LOG_FILE"
        continue
    fi

    # Check dashboard
    dashboard_ok=false
    if timeout 5 bash -c "echo > /dev/tcp/127.0.0.1/$dashboard_port" 2>/dev/null; then
        dashboard_ok=true
    fi

    # Check gateway
    gateway_ok=false
    if timeout 5 bash -c "curl -s -o /dev/null -w '%{http_code}' http://127.0.0.1:$gateway_port/v1/ 2>/dev/null | grep -q '404\|200'" 2>/dev/null; then
        gateway_ok=true
    fi

    if [ "$dashboard_ok" = false ] || [ "$gateway_ok" = false ]; then
        echo "  UNHEALTHY: Instance $slug ($subdomain) - dashboard=$dashboard_ok gateway=$gateway_ok" >> "$LOG_FILE"

        # Try to restart the container
        if [ "$dashboard_ok" = false ]; then
            container=$(docker ps --filter "name=${slug}" --format "{{.Names}}" | head -1)
            if [ -n "$container" ]; then
                echo "  Restarting dashboard in $container..." >> "$LOG_FILE"
                docker exec -d "$container" hermes dashboard --host 0.0.0.0 --port "$dashboard_port" --no-open --skip-build 2>/dev/null || true
            fi
        fi

        # Update instance status to ERROR if both are down
        if [ "$dashboard_ok" = false ] && [ "$gateway_ok" = false ]; then
            PGPASSWORD=wakelai_password psql -h localhost -p 55433 -U wakelai -d wakelai -c "UPDATE instances SET status = 'ERROR', last_error = 'Both dashboard and gateway not responding as of $(date)' WHERE id = $id" 2>/dev/null || true
            echo "  Marked instance $id as ERROR" >> "$LOG_FILE"
        fi
    else
        echo "  HEALTHY: Instance $slug ($subdomain)" >> "$LOG_FILE"
    fi
done

echo "[$(date '+%Y-%m-%d %H:%M:%S')] Health check complete." >> "$LOG_FILE"
