#!/usr/bin/env bash
#
# petav3 — service health check
# ----------------------------------------------------------------------------
# Operational, re-runnable check of the long-lived services that keep petav3
# alive: php-fpm, apache2, redis, Horizon, the Laravel scheduler, and the
# Baileys bridge.
# (The database is GCP Cloud SQL — a remote managed service, not a local unit —
#  so it isn't checked here; server-verify.sh tests the app's DB connection.)
#
# Unlike server-verify.sh (a one-time "is the environment installed correctly?"
# gate), this answers "are the services healthy RIGHT NOW?" — active + enabled
# state, recent restarts/failures, and a per-service liveness probe. Safe to run
# anytime; exit code is non-zero if any CRITICAL service is down, so it works as
# a cron / monitoring probe.
#
#   sudo bash scripts/service-check.sh           # one-shot report
#   sudo bash scripts/service-check.sh --watch    # refresh every 5s
#   sudo bash scripts/service-check.sh --restart-failed   # try to bounce down units
#
# Honors PHP_VERSION, APP_DIR, BRIDGE_PORT overrides (same as the other scripts).
# ----------------------------------------------------------------------------
set -uo pipefail

PHP_VERSION="${PHP_VERSION:-8.4}"
APP_DIR="${APP_DIR:-/var/www/html/peta}"
BRIDGE_PORT="${BRIDGE_PORT:-8088}"

WATCH=false
RESTART_FAILED=false
for arg in "$@"; do
    case "$arg" in
        --watch)          WATCH=true ;;
        --restart-failed) RESTART_FAILED=true ;;
        -h|--help) sed -n '2,20p' "$0"; exit 0 ;;
    esac
done

C_RESET='\033[0m'; C_GREEN='\033[1;32m'; C_RED='\033[1;31m'; C_YELLOW='\033[1;33m'; C_BLUE='\033[1;34m'; C_DIM='\033[2m'
PHP_BIN="php${PHP_VERSION}"

# The services we manage. Field: "unit|label|critical(1/0)"
# critical=1 means a down unit fails the whole check (non-zero exit).
SERVICES=(
    "php${PHP_VERSION}-fpm|PHP-FPM ${PHP_VERSION}|1"
    "apache2|Apache (web)|1"
    "redis-server|Redis (queue/cache)|1"
    "horizon|Horizon (queue worker)|1"
    "petav3-scheduler|Laravel scheduler (cron)|1"
    "baileys-wa-bridge|Baileys WhatsApp bridge|0"
)

DOWN=0          # critical units genuinely down (failed/inactive/dead)
DEGRADED=0      # non-critical units down, transitional units, or failing probes

# Transitional systemd states. A unit mid-restart reports one of these briefly —
# the classic case is Horizon right after a deploy's `horizon:terminate`: with
# fast_termination off it drains in-flight jobs (supervisor timeouts run to
# hundreds of seconds) and systemd sits in `deactivating` running ExecStop until
# the master exits, then `activating (auto-restart)` on the way back up. That is
# a routine restart window, NOT an outage — so we never count it as critical
# DOWN (which would fail the check / a deploy's exit code); it is DEGRADED at most.
is_transitional() { [[ "$1" == "activating" || "$1" == "deactivating" || "$1" == "reloading" ]]; }

# Read is-active, giving a unit caught mid-transition a few brief chances to
# settle to its resting state before we judge it. A real blip resolves to
# `active`; a genuinely stuck unit stays transitional and is reported degraded.
read_active() {
    local unit="$1" tries="${TRANSIENT_RETRIES:-3}" state
    state="$(systemctl is-active "$unit" 2>/dev/null)"
    while is_transitional "$state" && (( tries-- > 0 )); do
        sleep 2
        state="$(systemctl is-active "$unit" 2>/dev/null)"
    done
    echo "$state"
}

# Per-service liveness probe beyond "systemd says active". Echoes a short status
# string; returns 0 healthy, 1 unhealthy.
probe() {
    case "$1" in
        redis-server)
            local r; r="$(redis-cli ping 2>/dev/null)"
            [[ "$r" == "PONG" ]] && { echo "PONG"; return 0; }
            echo "no PONG"; return 1 ;;
        "php${PHP_VERSION}-fpm")
            [[ -S "/run/php/php${PHP_VERSION}-fpm.sock" ]] && { echo "fpm.sock present"; return 0; }
            echo "socket missing"; return 1 ;;
        apache2)
            local code; code="$(curl -fsS -o /dev/null -w '%{http_code}' http://localhost/ 2>/dev/null)"
            [[ "$code" =~ ^(200|301|302)$ ]] && { echo "HTTP ${code}"; return 0; }
            echo "HTTP ${code:-no response}"; return 1 ;;
        horizon)
            # horizon:status prints "Horizon is running." / "... inactive." / "... paused."
            local s; s="$($PHP_BIN "${APP_DIR}/artisan" horizon:status 2>/dev/null | tr -d '\r')"
            if grep -qi 'running' <<<"$s"; then echo "running"; return 0
            elif grep -qi 'paused'  <<<"$s"; then echo "paused"; return 1
            else echo "${s:-not running}"; return 1; fi ;;
        petav3-scheduler)
            # systemd "active" already means the unit is up; confirm the actual
            # schedule:work process is alive (it's what fires the Kernel schedule).
            pgrep -f "artisan schedule:work" >/dev/null 2>&1 \
                && { echo "schedule:work running"; return 0; }
            echo "schedule:work process missing"; return 1 ;;
        baileys-wa-bridge)
            curl -fsS "http://localhost:${BRIDGE_PORT}/health" >/dev/null 2>&1 \
                && { echo "/health OK"; return 0; }
            echo "/health unreachable"; return 1 ;;
        *) echo "-"; return 0 ;;
    esac
}

report() {
    DOWN=0; DEGRADED=0
    echo -e "${C_BLUE}petav3 service health${C_RESET}  ${C_DIM}$(date '+%Y-%m-%d %H:%M:%S %Z') on $(hostname)${C_DIM}${C_RESET}"
    printf "${C_DIM}%-26s %-9s %-9s %-7s %-9s %s${C_RESET}\n" "SERVICE" "ACTIVE" "ENABLED" "RESTART" "UPTIME" "HEALTH"
    echo "------------------------------------------------------------------------------------------"

    for entry in "${SERVICES[@]}"; do
        IFS='|' read -r unit label critical <<<"$entry"

        if ! systemctl list-unit-files "${unit}.service" >/dev/null 2>&1 \
             && ! systemctl cat "${unit}.service" >/dev/null 2>&1; then
            printf "%-26s ${C_RED}%-9s${C_RESET} %-9s %-7s %-9s %s\n" "$label" "MISSING" "-" "-" "-" "unit not installed"
            [[ "$critical" == "1" ]] && ((DOWN++)) || ((DEGRADED++))
            continue
        fi

        local active enabled restarts since uptime health hc
        active="$(read_active "$unit")"
        enabled="$(systemctl is-enabled "$unit" 2>/dev/null || echo '-')"
        restarts="$(systemctl show "$unit" -p NRestarts --value 2>/dev/null)"; restarts="${restarts:-0}"
        since="$(systemctl show "$unit" -p ActiveEnterTimestamp --value 2>/dev/null)"
        if [[ -n "$since" && "$since" != "n/a" ]]; then
            local start_epoch now_epoch secs
            start_epoch="$(date -d "$since" +%s 2>/dev/null || echo 0)"
            now_epoch="$(date +%s)"
            secs=$(( now_epoch - start_epoch ))
            if   (( secs < 60 ));    then uptime="${secs}s"
            elif (( secs < 3600 ));  then uptime="$(( secs/60 ))m"
            elif (( secs < 86400 )); then uptime="$(( secs/3600 ))h"
            else uptime="$(( secs/86400 ))d"; fi
        else
            uptime="-"
        fi

        # Colorize active state — transitional (mid-restart) is yellow, not red.
        local active_disp
        if [[ "$active" == "active" ]]; then active_disp="${C_GREEN}active${C_RESET}   "
        elif is_transitional "$active"; then active_disp="${C_YELLOW}${active}${C_RESET}"
        else active_disp="${C_RED}${active:-dead}${C_RESET}"; fi

        # Health probe only meaningful when the unit is fully active.
        if [[ "$active" == "active" ]]; then
            health="$(probe "$unit")"; hc=$?
        elif is_transitional "$active"; then
            health="(restarting)"; hc=1
        else
            health="(not active)"; hc=1
        fi
        local health_disp
        if [[ "$hc" -eq 0 ]]; then health_disp="${C_GREEN}${health}${C_RESET}"; else health_disp="${C_YELLOW}${health}${C_RESET}"; fi

        # Flag restart churn
        local restart_disp="$restarts"
        [[ "${restarts:-0}" -gt 0 ]] && restart_disp="${C_YELLOW}${restarts}${C_RESET}"

        printf "%-26s %b %-9s %b %-9s %b\n" \
            "$label" "$active_disp" "$enabled" "$restart_disp" "$uptime" "$health_disp"

        # Tally. A critical unit only counts as DOWN when it is genuinely dead
        # (failed/inactive/dead). A transitional unit (mid-restart) — or any
        # failing probe on a still-active unit — is DEGRADED: visible in the
        # report, but it does NOT fail the check or a deploy's exit code.
        if [[ "$active" != "active" || "$hc" -ne 0 ]]; then
            if [[ "$critical" == "1" && "$active" != "active" ]] && ! is_transitional "$active"; then
                ((DOWN++))
            else
                ((DEGRADED++))
            fi
        fi

        # Optionally bounce a down unit — but never kick one that is already
        # restarting on its own (transitional), or we just interrupt its recovery.
        if [[ "$RESTART_FAILED" == "true" && "$active" != "active" ]] && ! is_transitional "$active"; then
            echo -e "    ${C_YELLOW}↻ restarting ${unit}...${C_RESET}"
            systemctl restart "$unit" 2>&1 | sed 's/^/      /' || true
        fi
    done

    echo "------------------------------------------------------------------------------------------"
    if [[ "$DOWN" -eq 0 && "$DEGRADED" -eq 0 ]]; then
        echo -e "${C_GREEN}All services healthy.${C_RESET}"
    else
        echo -e "${C_RED}Critical down: ${DOWN}${C_RESET}   ${C_YELLOW}Degraded: ${DEGRADED}${C_RESET}"
        if [[ "$DOWN" -gt 0 || "$DEGRADED" -gt 0 ]]; then
            echo -e "${C_DIM}Inspect a unit: systemctl status <unit> | journalctl -u <unit> -n 50 --no-pager${C_DIM}${C_RESET}"
            echo -e "${C_DIM}Auto-bounce down units: sudo bash $0 --restart-failed${C_RESET}"
        fi
    fi
}

if [[ "$WATCH" == "true" ]]; then
    trap 'echo; exit 0' INT
    while true; do
        clear
        report
        echo -e "${C_DIM}(--watch: refreshing every 5s; Ctrl-C to stop)${C_RESET}"
        sleep 5
    done
else
    report
    [[ "$DOWN" -gt 0 ]] && exit 1
    exit 0
fi
