#!/usr/bin/env bash
#
# petav3 — deploy / code-update script
# ----------------------------------------------------------------------------
# Run this on the server EVERY TIME you ship new code. It performs a safe,
# zero-surprise Laravel deploy:
#
#   lock → pull → (conditional) composer/npm/bridge installs → DB backup
#   → permissions → [maintenance down] migrate → cache rebuild
#   → reload php-fpm/opcache [maintenance up] → restart Horizon/scheduler/SSR
#   (+ bridge if its code changed) → health check.
#
# SHORT MAINTENANCE WINDOW (default). The site stays LIVE through the slow part
# of a deploy — pull, composer, `npm run build`, the mysqldump — and only goes
# down for migrate → cache rebuild → php-fpm reload. That is ~10-20s instead of
# the 2-5 minutes the whole deploy takes.
#
# The trade, stated exactly: from the `git reset` until the window closes, the
# site serves NEW PHP source that does not yet have its new vendor packages,
# its rebuilt assets, or its migrations. Requests to pages this deploy did not
# touch are fine — the schema hasn't moved yet (migrate runs later, inside the
# window), so their unchanged queries still hold — but a page the deploy DID
# change can 500 for those couple of minutes. (Note this is
# the opposite mismatch from an atomic-release/symlink deploy, where old code
# briefly meets the new schema; here the code moves first and the schema
# catches up, which is why migrations do NOT have to be backward-compatible.)
#
# Migrations run with the site down and the code already new, so nothing above
# constrains what a migration may do — a column drop or rename is as safe here
# as it was before. What DOES stretch the window is a migration that rewrites a
# big table: the window is only short if the migrations are.
#
# Use --full-maintenance to get the old behaviour (down for the ENTIRE deploy)
# whenever that trade is wrong:
#   • the deploy changes a hot, high-traffic page in a way that needs its new
#     column or new package immediately
#   • composer.lock moved a lot, or you are shipping something you're unsure of
#   • you simply want nobody touching a half-updated tree
#
# Built to be re-run safely and to FAIL LOUDLY and CLEANLY:
#   • single-instance lock (no overlapping deploys)
#   • records the previous commit and AUTO-ROLLS-BACK code + assets on failure
#   • always lifts maintenance mode, even if a step dies, then exits non-zero
#   • expensive steps (composer/npm) run only when their lockfile actually changed
#   • the front-end is built into a staging dir and swapped in with a rename, so
#     the live site never reads an emptied public/build (no ViteManifestNotFound),
#     and the outgoing build's hashed chunks survive one generation, so a browser
#     tab opened just before the deploy doesn't 404 on a lazy-loaded chunk
#   • full timestamped log at storage/logs/deploy.log
#
# Usage:
#   sudo bash scripts/deploy-update.sh                    # standard deploy (short window)
#   sudo bash scripts/deploy-update.sh --full-maintenance # down for the whole deploy
#   sudo bash scripts/deploy-update.sh --force-build      # rebuild deps even if unchanged
#   sudo bash scripts/deploy-update.sh --skip-pull        # deploy the current working tree
#   sudo bash scripts/deploy-update.sh --no-migrate       # skip migrations
#   sudo bash scripts/deploy-update.sh --no-maintenance
#   sudo bash scripts/deploy-update.sh --rollback         # revert to the previous deploy
#   sudo bash scripts/deploy-update.sh --branch main --yes
#
# Exit codes: 0 ok · 1 deploy failed (rolled back) · 2 bad args · 3 unhealthy after deploy.
#
# Overrides (env): PHP_VERSION, APP_DIR, APP_USER, APP_GROUP, DEPLOY_BRANCH,
#                  AUTO_ROLLBACK (true/false), BACKUP_DB (true/false).
# ----------------------------------------------------------------------------
set -Eeuo pipefail

# ============================================================================
# Configuration
# ============================================================================
PHP_VERSION="${PHP_VERSION:-8.4}"
APP_DIR="${APP_DIR:-/var/www/html/peta}"
APP_USER="${APP_USER:-ubuntu}"      # owns the code; runs git/composer/artisan/workers
APP_GROUP="${APP_GROUP:-www-data}"  # web-server group (PHP-FPM writes storage/ via group)
WEB_USER="${WEB_USER:-www-data}"    # PHP-FPM / Apache runtime user
DEPLOY_BRANCH="${DEPLOY_BRANCH:-}"            # empty = current checked-out branch
AUTO_ROLLBACK="${AUTO_ROLLBACK:-true}"        # roll code back if a step fails
BACKUP_DB="${BACKUP_DB:-true}"                # mysqldump before migrating
PHP_BIN="php${PHP_VERSION}"

LOCK_FILE="/tmp/petav3-deploy.lock"
LOG_FILE="${APP_DIR}/storage/logs/deploy.log"
PREV_SHA_FILE="${APP_DIR}/storage/app/deploy/prev-sha"
BACKUP_DIR="${APP_DIR}/storage/app/deploy/backups"

# Flags
SKIP_PULL=false
FORCE_BUILD=false
DO_MIGRATE=true
USE_MAINTENANCE=true
FULL_MAINTENANCE=false   # true = down for the whole deploy (pre-short-window behaviour)
DO_ROLLBACK=false
ASSUME_YES=false

# ============================================================================
# Output / logging
# ============================================================================
C_RESET='\033[0m'; C_BLUE='\033[1;34m'; C_GREEN='\033[1;32m'; C_YELLOW='\033[1;33m'; C_RED='\033[1;31m'; C_DIM='\033[2m'
_ts() { date '+%Y-%m-%d %H:%M:%S'; }
log()  { echo -e "${C_BLUE}==>${C_RESET} $*"; _logfile "==> $*"; }
ok()   { echo -e "${C_GREEN}  ✓${C_RESET} $*"; _logfile "  OK  $*"; }
warn() { echo -e "${C_YELLOW}  ! ${C_RESET}$*"; _logfile "  WARN $*"; }
err()  { echo -e "${C_RED}  ✗ $*${C_RESET}" >&2; _logfile "  ERR $*"; }
_logfile() { [[ -n "${LOG_READY:-}" ]] && echo "[$(_ts)] $*" >> "$LOG_FILE" 2>/dev/null || true; }

# ============================================================================
# State used by traps
# ============================================================================
IN_MAINTENANCE=false
PREV_SHA=""
NEW_SHA=""
CODE_ADVANCED=false      # true once the working tree has moved to new code
FRONTEND_BUILT=false     # true once a new client+SSR bundle was actually built this run
DEPLOY_OK=false

# ============================================================================
# Argument parsing
# ============================================================================
while [[ $# -gt 0 ]]; do
    case "$1" in
        --skip-pull)       SKIP_PULL=true ;;
        --force-build)     FORCE_BUILD=true ;;
        --no-migrate)      DO_MIGRATE=false ;;
        --no-maintenance)  USE_MAINTENANCE=false ;;
        --full-maintenance) FULL_MAINTENANCE=true ;;
        --rollback)        DO_ROLLBACK=true ;;
        --yes|-y)          ASSUME_YES=true ;;
        --branch)          DEPLOY_BRANCH="${2:-}"; shift ;;
        --branch=*)        DEPLOY_BRANCH="${1#*=}" ;;
        -h|--help)         sed -n '2,64p' "$0"; exit 0 ;;
        *) err "Unknown argument: $1"; exit 2 ;;
    esac
    shift
done

# ----------------------------------------------------------------------------
# Run mode: resolve ONCE whether we can drop privileges to the app user, then
# use a single prefix everywhere. No silent "try as user, retry as root" — that
# would double-run side-effecting commands (e.g. migrate) on a real failure.
# ----------------------------------------------------------------------------
RUN_AS=()
if [[ "$(id -un)" == "${APP_USER}" ]]; then
    RUN_AS=()
elif sudo -u "${APP_USER}" true 2>/dev/null; then
    RUN_AS=(sudo -u "${APP_USER}")
else
    RUN_AS=()
    APP_USER_NOTE="could not sudo to ${APP_USER}; git/artisan run as $(id -un)"
fi
GIT() { "${RUN_AS[@]}" git -C "${APP_DIR}" "$@"; }
ART() { "${RUN_AS[@]}" "$PHP_BIN" "${APP_DIR}/artisan" "$@"; }

# Drop the stale bootstrap/cache manifests (package discovery, services, config).
# A leftover manifest that lists a dev-only provider — e.g. laravel-debugbar's
# Fruitcake\LaravelDebugbar\ServiceProvider — crashes artisan on boot right after
# a '--no-dev' composer install removes that package, before package:discover can
# rebuild it. Must be a plain filesystem rm: artisan can't boot to clear its own
# broken cache. These files are regenerated by package:discover / config:cache.
# Delete as root (the script always runs as root — see preflight), NOT via RUN_AS:
# these files may be root-owned from an earlier deploy, and a dropped-privilege
# 'sudo -u ${APP_USER} rm' would fail on them and trip the ERR trap here.
clear_bootstrap_cache() {
    rm -f \
        "${APP_DIR}/bootstrap/cache/packages.php" \
        "${APP_DIR}/bootstrap/cache/services.php" \
        "${APP_DIR}/bootstrap/cache/config.php"
}

maintenance_down() {
    [[ "$USE_MAINTENANCE" == "true" ]] || return 0
    # errors::maintenance is the countdown splash (resources/views/errors/
    # maintenance.blade.php) — pre-rendered to static HTML so public/index.php can
    # serve it without booting the framework. The bare `down` is the fallback for
    # a server whose checkout predates that view.
    if ART down --render="errors::maintenance" --retry=15 >/dev/null 2>&1 || ART down --retry=15 >/dev/null 2>&1; then
        IN_MAINTENANCE=true
        ok "Maintenance mode ON"
    else
        warn "Could not enable maintenance mode (continuing)"
    fi
}
maintenance_up() {
    [[ "$IN_MAINTENANCE" == "true" ]] || return 0
    if ART up >/dev/null 2>&1; then IN_MAINTENANCE=false; ok "Maintenance mode OFF"
    else err "FAILED to lift maintenance mode — run: $PHP_BIN artisan up"; fi
}

# ----------------------------------------------------------------------------
# Where the maintenance window opens and closes.
#
#   short (default) — the site serves through pull/composer/npm/backup and is
#                     only down for migrate → cache rebuild → php-fpm reload.
#   full  (--full-maintenance) — down from before the pull until after the
#                     workers restart, i.e. what this script used to always do.
#
# Every call site invokes BOTH the _full and _short variant at its own point;
# exactly one of them acts, so the two schedules never overlap.
# ----------------------------------------------------------------------------
maint_down_full()  { [[ "$FULL_MAINTENANCE" == "true" ]] && maintenance_down || true; }
maint_down_short() { [[ "$FULL_MAINTENANCE" == "true" ]] || maintenance_down; }
maint_up_short()   { [[ "$FULL_MAINTENANCE" == "true" ]] || maintenance_up; }
maint_up_full()    { [[ "$FULL_MAINTENANCE" == "true" ]] && maintenance_up || true; }

# ============================================================================
# Traps — clean up no matter how we exit
# ============================================================================
on_error() {
    local exit_code=$? line=${1:-?}
    # Disarm the trap and -e so the recovery steps below can't re-enter or abort us.
    trap - ERR
    set +e
    err "Deploy failed at line ${line} (exit ${exit_code})."
    if [[ "$AUTO_ROLLBACK" == "true" && "$CODE_ADVANCED" == "true" && -n "$PREV_SHA" ]]; then
        # In the short-window mode the site is still SERVING at this point, on a
        # tree we already know is broken — and the rollback below re-runs composer
        # and a full asset build, which is minutes. Take it down for that: an
        # aborted deploy is exactly the case the maintenance page exists for.
        maintenance_down
        warn "Auto-rolling back to ${PREV_SHA:0:10}..."
        if GIT reset --hard "$PREV_SHA" >/dev/null 2>&1; then
            ok "Code reverted to ${PREV_SHA:0:10}"
            # Rebuild PHP deps, front-end assets and clear caches against the old
            # code — public/build is gitignored so 'reset' alone leaves stale assets.
            COMPOSER_ALLOW_SUPERUSER=1 composer install --no-interaction --prefer-dist \
                --no-dev --optimize-autoloader --no-scripts -d "${APP_DIR}" >/dev/null 2>&1 \
                && { clear_bootstrap_cache; ART package:discover >/dev/null 2>&1; } \
                && ok "PHP dependencies restored" || warn "Dependency restore failed — inspect manually"
            if command -v npm >/dev/null; then
                ( cd "${APP_DIR}" && npm ci >/dev/null 2>&1 && npm run build >/dev/null 2>&1 ) \
                    && ok "Front-end assets rebuilt for rolled-back code" \
                    || warn "Asset rebuild failed — front-end may be stale; rebuild manually"
            fi
            chown -R "${APP_USER}:${APP_GROUP}" "${APP_DIR}" >/dev/null 2>&1 || true
            ART optimize:clear >/dev/null 2>&1 || true
            systemctl reload "php${PHP_VERSION}-fpm" >/dev/null 2>&1 || true
            # The SSR daemon lazy-loads hashed chunks from bootstrap/ssr, which
            # the asset rebuild above just rewrote — bounce it or it keeps
            # looking for hashes that no longer exist on disk
            # (ERR_MODULE_NOT_FOUND on every not-yet-cached page).
            systemctl restart petav3-inertia-ssr >/dev/null 2>&1 || true
            warn "DB migrations are NOT auto-reverted. If a migration ran, restore a backup from ${BACKUP_DIR} or run '$PHP_BIN artisan migrate:rollback'."
        else
            err "Rollback FAILED — manual intervention needed. Previous commit: ${PREV_SHA}"
        fi
    fi
    maintenance_up
    err "See ${LOG_FILE} for the full log."
    exit "${exit_code:-1}"
}
on_exit() {
    # Safety net: never leave the site in maintenance mode on an unexpected exit.
    [[ "$DEPLOY_OK" == "true" ]] || maintenance_up
}
trap 'on_error $LINENO' ERR
trap on_exit EXIT

# ============================================================================
# Preflight
# ============================================================================
[[ "$(id -u)" -eq 0 ]] || { err "Run as root (sudo) — needed for systemctl and chown."; exit 1; }
[[ -f "${APP_DIR}/artisan" ]] || { err "No Laravel app at ${APP_DIR} (run server-setup.sh first)."; exit 1; }
command -v "$PHP_BIN" >/dev/null || { err "${PHP_BIN} not found."; exit 1; }
command -v composer >/dev/null || { err "composer not found."; exit 1; }
command -v git >/dev/null || { err "git not found."; exit 1; }

mkdir -p "$(dirname "$PREV_SHA_FILE")" "$BACKUP_DIR" "$(dirname "$LOG_FILE")" 2>/dev/null || true
touch "$LOG_FILE" 2>/dev/null && LOG_READY=1 || warn "Cannot write ${LOG_FILE} (logging to console only)"
# Mark the repo safe SYSTEM-WIDE (/etc/gitconfig) so git works regardless of
# which user runs it — including the APP_USER we sudo to. A --global write would
# only touch root's config and not help the dropped-privilege git process.
git config --system --add safe.directory "${APP_DIR}" 2>/dev/null || true

# Single-instance lock — refuse to run two deploys at once.
exec 9>"$LOCK_FILE"
if ! flock -n 9; then
    err "Another deploy is already running (lock: ${LOCK_FILE}). Aborting."
    exit 1
fi

cd "${APP_DIR}"
log "petav3 deploy starting — $(_ts) on $(hostname)"
[[ -n "${APP_USER_NOTE:-}" ]] && warn "$APP_USER_NOTE"

# If we run git/artisan as APP_USER but the repo is owned by someone else, git
# would abort with "dubious ownership" and couldn't write .git. Normalise once.
if [[ -n "${RUN_AS[*]:-}" && -d "${APP_DIR}/.git" ]]; then
    REPO_OWNER="$(stat -c '%U' "${APP_DIR}/.git" 2>/dev/null || echo '?')"
    if [[ "$REPO_OWNER" != "$APP_USER" ]]; then
        warn "Repo owned by '${REPO_OWNER}', expected '${APP_USER}' — chowning ${APP_DIR} (first-run fix)..."
        chown -R "${APP_USER}:${APP_GROUP}" "${APP_DIR}"
        ok "Ownership normalised to ${APP_USER}:${APP_GROUP}"
    fi
fi

# ============================================================================
# Manual rollback path (--rollback): revert to the previous deploy and rebuild.
# ============================================================================
if [[ "$DO_ROLLBACK" == "true" ]]; then
    [[ -f "$PREV_SHA_FILE" ]] || { err "No previous deploy recorded (${PREV_SHA_FILE} missing)."; exit 1; }
    TARGET="$(cat "$PREV_SHA_FILE")"
    [[ -n "$TARGET" ]] || { err "Previous-SHA file is empty."; exit 1; }
    log "Rolling back to ${TARGET:0:10}..."
    if [[ "$ASSUME_YES" != "true" ]]; then
        read -r -p "Reset ${APP_DIR} to ${TARGET:0:10} and rebuild? [y/N] " ans
        [[ "$ans" =~ ^[Yy]$ ]] || { warn "Aborted by user."; exit 0; }
    fi
    maintenance_down
    PREV_SHA="$(GIT rev-parse HEAD)"; CODE_ADVANCED=true
    GIT reset --hard "$TARGET"
    COMPOSER_ALLOW_SUPERUSER=1 composer install --no-interaction --prefer-dist --no-dev --optimize-autoloader --no-scripts
    clear_bootstrap_cache
    ART package:discover
    command -v npm >/dev/null && { npm ci && npm run build; }
    chown -R "${APP_USER}:${APP_GROUP}" "${APP_DIR}"
    ART optimize:clear; ART config:cache; ART view:cache
    ART route:cache || warn "route:cache failed (closure route?) — skipped"
    systemctl reload "php${PHP_VERSION}-fpm" || true
    ART horizon:terminate || true
    ART queue:restart >/dev/null 2>&1 || true
    # schedule:work and the SSR daemon are long-running processes still holding
    # the rolled-back-FROM code in memory — bounce them onto the restored tree
    # (the SSR one also re-reads bootstrap/ssr, which npm run build just rewrote).
    systemctl restart petav3-scheduler 2>/dev/null || true
    systemctl restart petav3-inertia-ssr 2>/dev/null || true
    maintenance_up
    DEPLOY_OK=true
    ok "Rollback to ${TARGET:0:10} complete."
    warn "Database migrations are NOT auto-reverted — handle manually if needed."
    exit 0
fi

# ============================================================================
# 1. Record current state (for rollback + change detection)
# ============================================================================
PREV_SHA="$(GIT rev-parse HEAD)"
CUR_BRANCH="$(GIT rev-parse --abbrev-ref HEAD)"
[[ -n "$DEPLOY_BRANCH" ]] || DEPLOY_BRANCH="$CUR_BRANCH"
ok "Current commit ${PREV_SHA:0:10} on branch '${CUR_BRANCH}' (deploying '${DEPLOY_BRANCH}')"

# Refuse to clobber uncommitted local changes (config edits, hotfixes, etc.).
if [[ "$SKIP_PULL" != "true" ]]; then
    if ! GIT diff --quiet 2>/dev/null || ! GIT diff --cached --quiet 2>/dev/null; then
        err "Working tree has uncommitted changes. Commit/stash them, or use --skip-pull to deploy as-is."
        GIT status --short | sed 's/^/    /' || true
        exit 1
    fi
fi

# ============================================================================
# 2. Pull
#
# In the default SHORT window the site is still LIVE here: swapping the PHP
# source under a running app is the one genuinely inconsistent moment of the
# deploy (OPcache's default revalidate_freq bounds it to ~2s of mixed state,
# then vendor/assets catch up over the next steps). --full-maintenance takes
# the site down first instead.
# ============================================================================
maint_down_full

if [[ "$SKIP_PULL" == "true" ]]; then
    warn "Skipping git pull (--skip-pull) — deploying current working tree"
    NEW_SHA="$PREV_SHA"
else
    log "Fetching and fast-forwarding to origin/${DEPLOY_BRANCH}..."
    GIT fetch --prune origin
    GIT checkout "$DEPLOY_BRANCH"
    CODE_ADVANCED=true     # from here a failure should roll the code back
    GIT reset --hard "origin/${DEPLOY_BRANCH}"
    NEW_SHA="$(GIT rev-parse HEAD)"
    if [[ "$NEW_SHA" == "$PREV_SHA" ]]; then
        ok "Already up to date (${NEW_SHA:0:10}) — re-running deploy steps anyway"
    else
        ok "Updated ${PREV_SHA:0:10} → ${NEW_SHA:0:10}"
        GIT --no-pager log --oneline "${PREV_SHA}..${NEW_SHA}" 2>/dev/null | sed 's/^/    /' | head -20 || true
    fi
fi

# ============================================================================
# 3. Detect what changed (to skip expensive steps when possible)
# ============================================================================
changed() {
    [[ "$FORCE_BUILD" == "true" ]] && return 0
    [[ "$SKIP_PULL" == "true" ]] && return 0
    [[ "$PREV_SHA" == "$NEW_SHA" ]] && return 1
    GIT diff --name-only "$PREV_SHA" "$NEW_SHA" -- "$1" 2>/dev/null | grep -q . && return 0 || return 1
}
migrations_changed() {
    [[ "$FORCE_BUILD" == "true" || "$SKIP_PULL" == "true" ]] && return 0
    [[ "$PREV_SHA" == "$NEW_SHA" ]] && return 1
    GIT diff --name-only "$PREV_SHA" "$NEW_SHA" -- 'database/migrations/' 2>/dev/null | grep -q . && return 0 || return 1
}

# ============================================================================
# 4. PHP dependencies (only when composer.lock changed). Runs as root; the
#    permissions step below re-chowns everything to the app user.
# ============================================================================
if changed composer.lock || changed composer.json; then
    log "composer.lock changed — installing PHP dependencies..."
    COMPOSER_ALLOW_SUPERUSER=1 composer install --no-interaction --prefer-dist --no-dev --optimize-autoloader --no-scripts
    clear_bootstrap_cache
    ART package:discover
    ok "Composer dependencies updated"
else
    ok "composer.lock unchanged — skipping composer install"
    clear_bootstrap_cache
    ART package:discover >/dev/null 2>&1 || true
fi

# ============================================================================
# 5. Frontend build (npm ci only when lockfile changed; build whenever code moved)
# ============================================================================
if ! command -v npm >/dev/null; then
    warn "npm not found — skipping frontend build"
else
    if changed package-lock.json || changed package.json; then
        log "package-lock.json changed — npm ci..."
        npm ci
        ok "Node dependencies updated"
    else
        ok "package-lock.json unchanged — skipping npm ci"
    fi
    if [[ "$PREV_SHA" != "$NEW_SHA" || "$FORCE_BUILD" == "true" || "$SKIP_PULL" == "true" ]]; then
        # Vite EMPTIES its output directory before writing the new bundle, and with
        # the short maintenance window the site is LIVE while that happens — the
        # window does not open until migrate, several minutes later. Building
        # straight into public/build therefore deletes public/build/manifest.json
        # for the whole build, and @vite() in app.blade.php has nothing to read:
        # every request during those minutes dies with
        # ViteManifestNotFoundException, maintenance splash and all. That was a
        # real production 500 on 2026-08-12, not a theoretical one.
        #
        # So build into a STAGING directory and swap it in with two renames. The
        # live site keeps reading the previous, COMPLETE build right up to the
        # instant the new one is in place, and the gap where public/build does not
        # exist is microseconds instead of minutes. A failed build now leaves the
        # running site untouched too, rather than stranding it on an empty dir.
        BUILD_DIR="${APP_DIR}/public/build"
        BUILD_NEW="${APP_DIR}/public/build.new"
        BUILD_OLD="${APP_DIR}/public/build.old"
        rm -rf "$BUILD_NEW" "$BUILD_OLD" "${APP_DIR}/public/build.prev"

        # PIN THE NODE HEAP. Node picks its own heap cap from TOTAL machine RAM —
        # roughly a quarter of it — so the same commit gets 2083 MB on this 8 GB
        # production box and 4144 MB on the 16 GB dev box. The client build peaks
        # at ~2.85 GB RSS (measured 2026-09-02), which fits under the dev cap and
        # not under this one: b13e2797f1 built fine on dev and died here with
        # "Ineffective mark-compacts near heap limit" — V8 aborting, SIGABRT,
        # exit 134. That is NOT the kernel OOM killer (which is 137 and means the
        # BOX is out of RAM); MemAvailable was 5.3 GB at the time. Inheriting the
        # cap from the machine is the bug, so set it. 4096 leaves ~2.4 GB spare
        # here, which matters because this box has no swap.
        log "Building frontend assets (npm run build — client + SSR bundles)..."
        # Read by vite.config.js, and only for the CLIENT bundle — the SSR bundle
        # still goes to bootstrap/ssr.
        VITE_BUILD_OUT_DIR="public/build.new" NODE_OPTIONS="--max-old-space-size=4096" npm run build
        ok "Frontend built (client + SSR)"

        if [[ ! -d "$BUILD_NEW" ]]; then
            err "Build wrote nothing to ${BUILD_NEW} — check that vite.config.js still honours VITE_BUILD_OUT_DIR."
            exit 1
        fi

        # Carry the outgoing build's hashed chunks into the new directory, so a tab
        # opened seconds before the deploy doesn't 404 on a chunk it lazy-loads.
        # -n never clobbers, so the new manifest.json and new assets always win;
        # -a preserves mtimes, which is what the prune below keys on — without it
        # every asset would look freshly built forever.
        if [[ -d "$BUILD_DIR" ]]; then
            cp -an "${BUILD_DIR}/." "${BUILD_NEW}/" 2>/dev/null || true
        fi

        # The swap itself: two renames on one filesystem.
        if [[ -d "$BUILD_DIR" ]]; then
            mv "$BUILD_DIR" "$BUILD_OLD"
        fi
        mv "$BUILD_NEW" "$BUILD_DIR"
        rm -rf "$BUILD_OLD"
        FRONTEND_BUILT=true

        # Bound the pile-up: carried-over chunks keep their original mtime, so
        # anything not rebuilt for a week is unreferenced by every manifest a
        # live browser could still be holding. Vite rewrites every asset the
        # CURRENT manifest points at on each build, so this can't delete a
        # file the running site needs.
        find "${BUILD_DIR}/assets" -type f -mtime +7 -delete 2>/dev/null || true
        ok "New build swapped in atomically (no manifest gap; open tabs won't 404)"
    else
        ok "No code change — skipping frontend build"
    fi
fi

# ============================================================================
# 6. wa-bridge (Baileys) deps — only if its code/deps changed
#    (restarting the bridge drops the live WhatsApp socket, so do it sparingly)
# ============================================================================
BRIDGE_CHANGED=false
if changed 'wa-bridge/'; then
    BRIDGE_CHANGED=true
    if command -v npm >/dev/null && [[ -f "${APP_DIR}/wa-bridge/package-lock.json" ]]; then
        log "wa-bridge changed — installing bridge dependencies..."
        ( cd "${APP_DIR}/wa-bridge" && npm ci )
        ok "wa-bridge dependencies updated"
    fi
fi

# ============================================================================
# 6b. zoom-bot (webinar assistant sidecar) deps + Chromium — when its code
#     changed, or on the first deploy that ships it (node_modules absent).
#     PLAYWRIGHT_BROWSERS_PATH pins the browser inside the checkout so the
#     Chromium this install puts down is the one the scheduler user finds —
#     Playwright's default cache is per-user, which fails only in prod.
#     Verify any zoom-bot change with: php artisan zoom:bot-doctor
# ============================================================================
ZOOM_BOT_DIR="${APP_DIR}/tools/zoom-bot"
if [[ -f "${ZOOM_BOT_DIR}/package-lock.json" ]] && command -v npm >/dev/null; then
    if changed 'tools/zoom-bot/' || [[ ! -d "${ZOOM_BOT_DIR}/node_modules" ]]; then
        log "zoom-bot changed — installing sidecar dependencies + Chromium..."
        ( cd "${ZOOM_BOT_DIR}" \
            && npm ci \
            && PLAYWRIGHT_BROWSERS_PATH="${ZOOM_BOT_DIR}/pw-browsers" npx playwright install --with-deps chromium ) \
            && ok "zoom-bot dependencies updated" \
            || warn "zoom-bot install FAILED — the webinar assistant cannot launch; run: php artisan zoom:bot-doctor"
    fi
fi

# ============================================================================
# 7. Pre-migration DB safety backup — deliberately OUTSIDE the maintenance
#    window. On a production-sized database the dump is one of the longest
#    steps of the whole deploy, and nothing about it needs the site to be down:
#    --single-transaction takes a consistent snapshot of a live InnoDB database.
# ============================================================================
if [[ "$DO_MIGRATE" == "true" ]]; then
    # Take an optional safety backup only when this code delta actually touches
    # migrations (or on a forced/no-pull run) — avoids dumping the DB on every
    # no-op redeploy. Note: a migration left pending by an earlier half-finished
    # deploy isn't in this diff, so it would apply below without a fresh backup.
    if migrations_changed || [[ "$SKIP_PULL" == "true" || "$FORCE_BUILD" == "true" ]]; then
        if [[ "$BACKUP_DB" == "true" ]] && command -v mysqldump >/dev/null; then
            # DB_HOST is REQUIRED here, and its absence is what made every backup
            # between 2026-07-26 and 2026-08-07 a 20-byte empty gzip: the database
            # is remote, mysqldump defaulted to localhost, found nothing, and the
            # deploy warned once and migrated anyway. On 2026-08-07 the production
            # database was dropped and the "backup" taken that morning was empty.
            # See docs/incident-2026-08-07-production-database-wipe.md.
            DB_HOST="$(grep -E '^DB_HOST=' "${APP_DIR}/.env" | head -1 | cut -d= -f2- | tr -d '"'"'"'' )"
            DB_PORT="$(grep -E '^DB_PORT=' "${APP_DIR}/.env" | head -1 | cut -d= -f2- | tr -d '"'"'"'' )"
            DB_DATABASE="$(grep -E '^DB_DATABASE=' "${APP_DIR}/.env" | head -1 | cut -d= -f2- | tr -d '"'"'"'' )"
            DB_USERNAME="$(grep -E '^DB_USERNAME=' "${APP_DIR}/.env" | head -1 | cut -d= -f2- | tr -d '"'"'"'' )"
            DB_PASSWORD="$(grep -E '^DB_PASSWORD=' "${APP_DIR}/.env" | head -1 | cut -d= -f2- | tr -d '"'"'"'' )"
            if [[ -n "$DB_DATABASE" ]]; then
                BK="${BACKUP_DIR}/db-$(date '+%Y%m%d-%H%M%S').sql.gz"
                log "Backing up '${DB_DATABASE}' at ${DB_HOST:-127.0.0.1} before migrating → ${BK}"
                # NOT --single-transaction. In MySQL 8 it issues FLUSH TABLES,
                # which needs the GLOBAL RELOAD privilege — and this account has
                # ALL PRIVILEGES on its own schema but only USAGE on *.*, so the
                # dump aborted after 367 bytes of header. --lock-tables=false
                # keeps it from locking a live production schema instead.
                #
                # The trade, stated plainly: without --single-transaction the
                # dump is not one consistent snapshot, so rows can shift under
                # it mid-run. For a pre-migration safety copy that is the right
                # trade — a slightly inconsistent 188 MB backup restores; the
                # 20-byte file this replaces does not. Grant RELOAD to the
                # backup account and put --single-transaction back if you want
                # snapshot consistency.
                #
                # --set-gtid-purged=OFF so the dump can be restored to a
                # different server without inheriting the source's GTID state.
                if MYSQL_PWD="$DB_PASSWORD" mysqldump --quick --no-tablespaces \
                        --lock-tables=false --set-gtid-purged=OFF \
                        -h "${DB_HOST:-127.0.0.1}" -P "${DB_PORT:-3306}" \
                        -u "$DB_USERNAME" "$DB_DATABASE" 2>>"$LOG_FILE" | gzip > "$BK"; then
                    # A successful exit is NOT proof of a usable dump. Verify the
                    # bytes: an empty gzip stream is 20 bytes, and 56 of those
                    # accumulated unnoticed because nothing ever looked.
                    BK_BYTES="$(stat -c%s "$BK" 2>/dev/null || echo 0)"
                    if (( BK_BYTES < 1024 )); then
                        rm -f "$BK"
                        err "DB backup is only ${BK_BYTES} bytes — that is an empty dump, not a backup."
                        err "Refusing to migrate. Fix the credentials/host in .env, or set BACKUP_DB=false to proceed deliberately."
                        exit 1
                    fi
                    ok "DB backup written ($(du -h "$BK" 2>/dev/null | cut -f1))"
                    ls -1t "${BACKUP_DIR}"/db-*.sql.gz 2>/dev/null | tail -n +11 | xargs -r rm -f
                else
                    # Previously a warn-and-continue. That is the line that let a
                    # schema change run against a database with no recoverable
                    # copy — the warning scrolled past in a long deploy log and
                    # the migration proceeded regardless.
                    rm -f "$BK"
                    err "DB backup FAILED — refusing to migrate without one (see ${LOG_FILE} for mysqldump's error)."
                    err "Set BACKUP_DB=false to deploy deliberately without a backup."
                    exit 1
                fi
            fi
        fi
    fi
fi

# ============================================================================
# 8. Permissions — re-own everything to the app user BEFORE the cache rebuild,
#    so the caches (and bootstrap/cache/*) end up app-owned, not root-owned.
#    Also outside the window: composer/npm ran as root, and a chown -R over
#    vendor/ + node_modules/ is tens of thousands of inodes. It only relabels
#    files the running site is already reading, so it is safe while live.
# ============================================================================
log "Fixing ownership and permissions..."
chown -R "${APP_USER}:${APP_GROUP}" "${APP_DIR}"
chmod -R ug+rwX "${APP_DIR}/storage" "${APP_DIR}/bootstrap/cache"
# setgid so new files in these dirs inherit APP_GROUP regardless of creator.
find "${APP_DIR}/storage" "${APP_DIR}/bootstrap/cache" -type d -exec chmod g+s {} + 2>/dev/null || true
# ACLs: keep storage writable by BOTH the deploy user and the web user, now and
# for future files (-d), so the ubuntu:www-data split can't break laravel.log etc.
if command -v setfacl >/dev/null; then
    setfacl -R  -m u:"${APP_USER}":rwX -m u:"${WEB_USER}":rwX "${APP_DIR}/storage" "${APP_DIR}/bootstrap/cache" 2>/dev/null || true
    setfacl -dR -m u:"${APP_USER}":rwX -m u:"${WEB_USER}":rwX "${APP_DIR}/storage" "${APP_DIR}/bootstrap/cache" 2>/dev/null || true
fi
ok "Permissions set"

# ############################################################################
# >>> MAINTENANCE WINDOW OPENS (short mode) <<<
#
# Everything from here to the php-fpm reload is what genuinely wants the site
# quiet: the schema changes, and `optimize:clear` briefly leaves the app with
# no compiled config/route cache. Typically 10-20 seconds — but a migration
# that rewrites a large table holds the window open for as long as it locks,
# so the window is only short if the migrations are.
# ############################################################################
maint_down_short

# ============================================================================
# 9. Database migrations
# ============================================================================
if [[ "$DO_MIGRATE" == "true" ]]; then
    # Always run migrate: 'migrate --force' is idempotent (applies only pending
    # migrations, a fast no-op when none are pending). Gating it on the git diff
    # meant migrations from a same-SHA redeploy or an earlier crashed deploy were
    # silently skipped. Left UNGUARDED on purpose — a real migration failure must
    # trip the ERR trap and roll the deploy back, not be swallowed.
    log "Running migrations..."
    ART migrate --force
    ok "Migrations applied"
else
    warn "Migrations skipped (--no-migrate)"
fi

# ============================================================================
# 10. Storage link + cache rebuild
# ============================================================================
[[ -L "${APP_DIR}/public/storage" ]] || ART storage:link >/dev/null 2>&1 || true

log "Rebuilding framework caches..."
ART optimize:clear
ART config:cache
ART route:cache || warn "route:cache failed (closure route?) — skipped"
ART view:cache
ART event:cache >/dev/null 2>&1 || true
ok "Caches rebuilt"

# ============================================================================
# 11. Reload PHP-FPM (clears OPcache so new code is actually served)
# ============================================================================
if systemctl list-unit-files "php${PHP_VERSION}-fpm.service" >/dev/null 2>&1; then
    systemctl reload "php${PHP_VERSION}-fpm" && ok "php${PHP_VERSION}-fpm reloaded (OPcache cleared)" \
        || { systemctl restart "php${PHP_VERSION}-fpm" && ok "php${PHP_VERSION}-fpm restarted"; }
fi

# ############################################################################
# >>> MAINTENANCE WINDOW CLOSES (short mode) <<<
#
# The web tier is now fully on the new code. The worker/service restarts below
# do not serve HTTP, so they run with the site back up: Horizon and the
# scheduler are queue-side, and the SSR daemon merely falls back to
# client-side rendering while it bounces.
# ############################################################################
maint_up_short

# ============================================================================
# 12. Restart workers / services so they pick up new code
# ============================================================================
log "Restarting queue workers and services..."
# Horizon: terminate gracefully; systemd (Restart=always) brings it back on new code.
if systemctl is-active --quiet horizon 2>/dev/null; then
    ART horizon:terminate >/dev/null 2>&1 || systemctl restart horizon || true
    ok "Horizon told to restart"
else
    warn "horizon.service not active — starting it"
    systemctl start horizon 2>/dev/null || true
fi
# Any plain queue workers.
ART queue:restart >/dev/null 2>&1 || true

# ----------------------------------------------------------------------------
# Laravel scheduler (petav3-scheduler.service → `php artisan schedule:work`).
# This is the cron replacement that fires EVERYTHING in app/Console/Kernel.php
# (dowayai poll, the WhatsApp broadcast/flow/reminder backstops,
# Zoom sync). Horizon only runs jobs already on the queue — without this unit
# none of the scheduled commands ever run.
#
# The deploy SELF-PROVISIONS the unit: it (re)writes it when missing or when the
# template below changed (repairs servers set up before the unit existed), then
# restarts it every deploy — schedule:work is long-running, so like Horizon it
# must be bounced to run new code. Content matches server-setup.sh byte-for-byte
# so a fresh box and a deploy converge on the same unit.
# ----------------------------------------------------------------------------
SCHEDULER_UNIT="petav3-scheduler.service"
SCHED_PHP_BIN="$(command -v "$PHP_BIN" || true)"
scheduler_unit_content() {
    cat <<UNIT
[Unit]
Description=petav3 Laravel scheduler (schedule:work)
After=network.target redis-server.service mysql.service

[Service]
Type=simple
User=${APP_USER}
Group=${APP_GROUP}
Restart=always
RestartSec=3
TimeoutStopSec=60
WorkingDirectory=${APP_DIR}
ExecStart=${SCHED_PHP_BIN} ${APP_DIR}/artisan schedule:work
# Stop the scheduler ONLY, never its cgroup: zoom:run-bot detaches long-lived
# webinar bots that must survive a deploy's scheduler restart. The default
# control-group kill would SIGKILL a bot mid-webinar, leaving a phantom
# panelist broadcasting to the live audience.
KillMode=process

[Install]
WantedBy=multi-user.target
UNIT
}
if [[ -z "$SCHED_PHP_BIN" ]]; then
    warn "Cannot resolve ${PHP_BIN} to an absolute path — scheduler unit not managed this deploy"
elif ! command -v systemctl >/dev/null 2>&1; then
    warn "systemctl not available — run the scheduler manually: $PHP_BIN artisan schedule:work"
else
    if ! cmp -s <(scheduler_unit_content) "/etc/systemd/system/${SCHEDULER_UNIT}" 2>/dev/null; then
        scheduler_unit_content > "/etc/systemd/system/${SCHEDULER_UNIT}"
        systemctl daemon-reload
        ok "${SCHEDULER_UNIT} installed/updated"
    fi
    systemctl enable "$SCHEDULER_UNIT" >/dev/null 2>&1 || true
    if systemctl restart "$SCHEDULER_UNIT" 2>/dev/null; then
        ok "Scheduler (schedule:work) restarted"
    else
        warn "Scheduler restart FAILED — scheduled tasks are not running; check: journalctl -u ${SCHEDULER_UNIT} -n 50 --no-pager"
    fi
fi

# Inertia SSR render server: bounce whenever a NEW BUNDLE was built this run —
# the Node daemon lazy-loads hashed chunks from bootstrap/ssr and keeps the old
# hashes in memory until restarted, so a rebuild without a restart leaves it
# asking the disk for files the build just deleted (ERR_MODULE_NOT_FOUND logged
# on every not-yet-cached page — bit production on 2026-08-13, when a rebuild
# reached the server without this restart firing). Keyed on FRONTEND_BUILT, not
# on the SHA diff, so it can never disagree with the build decision in step 5
# (the old SHA-diff condition missed --skip-pull, which rebuilds the bundle).
# Guarded (unit may not exist on older servers); SSR down = client-side-render
# fallback, never an outage.
if systemctl list-unit-files petav3-inertia-ssr.service >/dev/null 2>&1; then
    if [[ "$FRONTEND_BUILT" == "true" ]]; then
        systemctl restart petav3-inertia-ssr \
            && ok "petav3-inertia-ssr restarted (new SSR bundle loaded)" \
            || warn "petav3-inertia-ssr restart FAILED — the daemon still serves the OLD bundle and will log SSR errors until you run: sudo systemctl restart petav3-inertia-ssr"
    else
        ok "No new frontend build — leaving petav3-inertia-ssr running"
    fi
fi

# Baileys bridge: only bounce it when its code changed (avoids dropping the WA socket).
if systemctl list-unit-files baileys-wa-bridge.service >/dev/null 2>&1; then
    if [[ "$BRIDGE_CHANGED" == "true" || "$FORCE_BUILD" == "true" ]]; then
        systemctl restart baileys-wa-bridge && ok "baileys-wa-bridge restarted (code changed)" || warn "bridge restart failed"
    else
        ok "wa-bridge unchanged — leaving its WhatsApp session connected"
    fi
fi

# ============================================================================
# 13. Lift maintenance (full mode only — short mode lifted it above) + record
#     success
# ============================================================================
maint_up_full

# Record the pre-deploy commit so --rollback can return here — but only when we
# actually moved, so a no-op re-deploy doesn't erase the real rollback anchor.
if [[ "$NEW_SHA" != "$PREV_SHA" ]]; then
    echo "$PREV_SHA" > "$PREV_SHA_FILE" 2>/dev/null || true
    chown "${APP_USER}:${APP_GROUP}" "$PREV_SHA_FILE" 2>/dev/null || true
fi
DEPLOY_OK=true   # from here the EXIT trap won't force maintenance mode

# ============================================================================
# 14. Post-deploy health check (informational — does NOT roll back, but the
#     exit code must not lie: an unhealthy box exits non-zero for CI).
# ============================================================================
echo
# Best-effort: give Horizon a bounded moment to finish its terminate→drain→restart
# cycle so the report below shows it "active" instead of mid-restart. This is a
# courtesy, NOT a correctness gate — service-check.sh treats a still-transitional
# unit as degraded (exit 0), so a long in-flight job drain can never block or fail
# the deploy here.
if systemctl list-unit-files horizon.service >/dev/null 2>&1; then
    for _ in $(seq 1 "${HORIZON_SETTLE_TRIES:-15}"); do
        if [[ "$(systemctl is-active horizon 2>/dev/null)" == "active" ]]; then break; fi
        sleep 2
    done
fi
HEALTH_RC=0
SELF_DIR="$(cd "$(dirname "$0")" && pwd)"
if [[ -f "${SELF_DIR}/service-check.sh" ]]; then
    log "Post-deploy health check:"
    bash "${SELF_DIR}/service-check.sh" || HEALTH_RC=$?
else
    warn "service-check.sh not found — skipping health check"
fi

echo
ok "Deploy complete: ${PREV_SHA:0:10} → ${NEW_SHA:0:10}  ($(_ts))"
_logfile "DEPLOY OK ${PREV_SHA:0:10} -> ${NEW_SHA:0:10}"

if [[ "$HEALTH_RC" -ne 0 ]]; then
    warn "Code deployed, but a service is unhealthy — investigate (exit 3)."
    exit 3
fi
exit 0
