#!/bin/bash
# eco status server (namespace mode) - start/stop/status/gui wrapper.
#
# Canonical copy lives in the eco repo at scripts/, alongside eco-dev;
# /sf/bernina/bin/eco-status-server is a symlink to this file, so editing
# the checkout takes effect immediately - nothing to redeploy.
#
# Runs in the FOREGROUND by default, which is what systemd's Type=simple
# wants and what the accompanying user unit calls. Use `-b` to detach for
# interactive use.
#
# Everything is overridable by environment, so the same script serves a
# personal checkout, the shared checkout and the systemd unit unchanged:
#
#   ECO_STATUS_SERVER_CHECKOUT  eco checkout to run (prepended to PYTHONPATH)
#   ECO_STATUS_SERVER_CONFIG    server config JSON
#   ECO_STATUS_SERVER_PYTHON    interpreter (needs eco's full dependency set)
#   ECO_STATUS_SERVER_URL       what `status`/`stop` talk to
#   ECO_STATUS_SERVER_LOG       log file used by -b
#
# A file at /sf/bernina/config/eco_status_server/env (or
# $ECO_STATUS_SERVER_ENV) is sourced first if it exists, so a site can set
# those in one place without editing this script.
set -euo pipefail

ENV_FILE="${ECO_STATUS_SERVER_ENV:-/sf/bernina/config/eco_status_server/env}"
# shellcheck source=/dev/null
[ -r "$ENV_FILE" ] && . "$ENV_FILE"

CHECKOUT="${ECO_STATUS_SERVER_CHECKOUT:-/sf/bernina/code/gac-bernina/eco}"
CONFIG="${ECO_STATUS_SERVER_CONFIG:-/sf/bernina/config/eco_status_server/bernina_namespace.json}"
PYTHON="${ECO_STATUS_SERVER_PYTHON:-/sf/bernina/applications/python/.pixi/envs/bpy312/bin/python}"
URL="${ECO_STATUS_SERVER_URL:-http://$(hostname -s):8091}"
LOG="${ECO_STATUS_SERVER_LOG:-$HOME/.eco/status_server_$(hostname -s).log}"

# Matches only the running server, never this script or an ssh command line
# that mentions it (the bracket keeps the pattern from matching itself).
PGREP_PATTERN='[e]co\.status_server --mode'

usage() {
    cat <<EOF
usage: $(basename "$0") <command> [options]

  start [-b] [-- <extra args>]  run the server (foreground; -b to detach)
  stop                          stop the running server on this host
  restart                       stop, then start detached
  status                        query $URL/health
  wait [timeout]                block until the server reports ready
  logs [-f]                     show the detached-mode log
  stats                         recent /status/snapshot & /status/capture calls
  gui                           launch the Qt status/reinit GUI (detached)
  config                        print the resolved configuration

Environment (current values):
  ECO_STATUS_SERVER_CHECKOUT=$CHECKOUT
  ECO_STATUS_SERVER_CONFIG=$CONFIG
  ECO_STATUS_SERVER_PYTHON=$PYTHON
  ECO_STATUS_SERVER_URL=$URL
  ECO_STATUS_SERVER_LOG=$LOG
EOF
}

die() { echo "$(basename "$0"): $*" >&2; exit 1; }

check_prereqs() {
    [ -d "$CHECKOUT/eco" ] || die "no eco checkout at $CHECKOUT (set ECO_STATUS_SERVER_CHECKOUT)"
    # Not fatal, but almost always the explanation when half the namespace
    # fails to initialize: without the gateway list, Channel Access falls
    # back to the local broadcast domain. A systemd unit inherits nothing,
    # so this is the first thing to check there.
    if [ -z "${EPICS_CA_ADDR_LIST:-}" ]; then
        echo "$(basename "$0"): warning: EPICS_CA_ADDR_LIST is not set;" >&2
        echo "  most components will fail to connect. See the EnvironmentFile" >&2
        echo "  written by the install-user-service script alongside this one." >&2
    fi
    [ -r "$CONFIG" ] || die "no server config at $CONFIG (set ECO_STATUS_SERVER_CONFIG)"
    [ -x "$PYTHON" ] || die "no interpreter at $PYTHON (set ECO_STATUS_SERVER_PYTHON)"
}

server_pid() { pgrep -f "$PGREP_PATTERN" | head -1; }

# Is this host's server owned by the systemd user service? If so, start/stop
# have to go through systemd: killing the process directly makes systemd
# consider the unit stopped, so a manual restart afterwards leaves an
# unmanaged process running while `systemctl --user status` says inactive.
# That is not hypothetical - it is exactly the state this host was found in.
managed_by_systemd() {
    # Never true when *this* invocation is the unit's own ExecStart, or the
    # guard below would make the service refuse to start itself: systemd
    # would see exit 1, retry, and hit the start-rate limit. systemd sets
    # INVOCATION_ID in every service's environment and nowhere else.
    [ -n "${INVOCATION_ID:-}" ] && return 1
    command -v systemctl >/dev/null 2>&1 || return 1
    case "$(systemctl --user is-active eco-status-server 2>/dev/null)" in
        active|activating|reloading) return 0 ;;
        *) return 1 ;;
    esac
}

# The account running the server, when it is not this one. Two accounts
# writing the same run tree is fine (that is what the group-writable modes
# in eco.utilities.datafiles are for), but silently stopping someone else's
# server is not.
server_owner() {
    local pid="$1"
    [ -n "$pid" ] && ps -o user= -p "$pid" 2>/dev/null | tr -d ' '
}

cmd_start() {
    local background=0
    if [ "${1:-}" = "-b" ] || [ "${1:-}" = "--background" ]; then background=1; shift; fi
    [ "${1:-}" = "--" ] && shift
    if managed_by_systemd; then
        die "the systemd user service is running this host's server; use
  systemctl --user restart eco-status-server
(or 'systemctl --user stop' it first if you really want to run it by hand)"
    fi
    check_prereqs
    local pid; pid="$(server_pid || true)"
    if [ -n "$pid" ]; then
        local owner; owner="$(server_owner "$pid")"
        die "already running (pid $pid, user ${owner:-unknown}); use 'restart' or 'stop' first"
    fi

    export PYTHONPATH="$CHECKOUT${PYTHONPATH:+:$PYTHONPATH}"
    if [ "$background" -eq 1 ]; then
        mkdir -p "$(dirname "$LOG")"
        # </dev/null matters: several eco components prompt for credentials
        # on stdin during __init__, and a headless server must get EOF
        # rather than block forever waiting for an answer nobody will type.
        nohup "$PYTHON" -u -m eco.status_server --mode namespace \
            --config "$CONFIG" "$@" </dev/null >>"$LOG" 2>&1 &
        echo "started pid $! -> $LOG"
        echo "the namespace init takes minutes; watch it with: $(basename "$0") wait"
    else
        exec "$PYTHON" -u -m eco.status_server --mode namespace \
            --config "$CONFIG" "$@" </dev/null
    fi
}

cmd_stop() {
    if managed_by_systemd; then
        echo "stopping via systemd (the service owns this host's server)"
        systemctl --user stop eco-status-server
        return 0
    fi
    local pid; pid="$(server_pid || true)"
    [ -z "$pid" ] && { echo "not running"; return 0; }
    local owner; owner="$(server_owner "$pid")"
    if [ -n "$owner" ] && [ "$owner" != "$(id -un)" ]; then
        die "pid $pid belongs to '$owner', not you - stop it from that account"
    fi
    kill "$pid"
    for _ in $(seq 1 30); do
        sleep 1
        pgrep -f "$PGREP_PATTERN" >/dev/null || { echo "stopped (was pid $pid)"; return 0; }
    done
    die "pid $pid did not exit after 30 s"
}

cmd_status() {
    local pid; pid="$(server_pid || true)"
    if [ -n "$pid" ]; then
        local owner; owner="$(server_owner "$pid")"
        local managed="unmanaged"
        managed_by_systemd && managed="systemd user service"
        echo "process: running, pid $pid, user ${owner:-unknown} ($managed)"
    else
        echo "process: not running on $(hostname -s)"
    fi
    # urllib rather than curl|python: the health formatting is a real
    # program, and quoting it through a shell pipeline is how it silently
    # turns into "no answer" the first time someone edits it.
    # `|| true`: a server that is not up yet is a normal state for this
    # command, and under `set -e` a non-zero exit here would abort the
    # `wait` loop that calls it.
    "$PYTHON" - "$URL" <<'PYEOF' || true
import json, sys, urllib.request

RED_BOLD = "\033[1;31m"
YELLOW = "\033[33m"
RESET = "\033[0m"
color = sys.stdout.isatty()

def c(s, code):
    return f"{code}{s}{RESET}" if color else s

url = sys.argv[1].rstrip("/") + "/health"
try:
    with urllib.request.urlopen(url, timeout=15) as r:
        h = json.load(r)
except Exception as exc:
    print("health:  no answer from %s (%s)" % (url, exc))
    raise SystemExit(1)

print("state:   {state}  (generation {generation}, up {up:.1f} min)".format(
    state=h["state"], generation=h["generation"], up=h["uptime_s"] / 60))
print("init:    {i}/{t} components, {f} failed".format(
    i=h["n_initialized"], t=h["n_target_names"], f=h["n_failed"]))
print("serving: {d} status detectors, {m} monitorable".format(
    d=h["n_direct_read"], m=h.get("n_monitorable")))
print("process: {r:.0f} MB, {t} threads, {c:.0f} s cpu".format(
    r=h.get("rss_mb") or 0, t=h.get("n_threads"), c=h.get("cpu_seconds") or 0))
# failed_required: components that are BOTH in required_names() and in
# failed_names(), i.e. the setup is not supposed to fail these - see
# NamespaceMonitorStore.connection_report(). A merely-optional failure
# (failed_names but not required) stays in the plain line below, unhighlighted.
failed_required = h.get("failed_required") or []
other_failed = [n for n in (h.get("failed_names") or []) if n not in failed_required]
if failed_required:
    print(c(
        "!!! {n} REQUIRED component(s) failed to initialize: {names}".format(
            n=len(failed_required), names=", ".join(failed_required)),
        RED_BOLD,
    ))
if other_failed:
    print("failed:  " + ", ".join(other_failed))
for rec in h.get("recordings", []):
    if rec.get("running"):
        print("recording {id}: {u} updates, {s} stored".format(
            id=rec["recording_id"], u=rec["n_updates"], s=rec["n_stored"]))
PYEOF
}

is_ready() {
    "$PYTHON" - "$URL" <<'PYEOF' >/dev/null 2>&1
import json, sys, urllib.request
with urllib.request.urlopen(sys.argv[1].rstrip("/") + "/health", timeout=15) as r:
    raise SystemExit(0 if json.load(r).get("ready") else 1)
PYEOF
}

# One progress+ETA line, from /health alone: elapsed time in the current
# state (state_seconds) and how far init has gotten (n_initialized of
# n_target_names) linearly project how much longer it needs. Rough on
# purpose - components do not all cost the same - but far better than a
# bare spinner for something that can take several minutes.
progress_line() {
    "$PYTHON" - "$URL" <<'PYEOF' 2>/dev/null || true
import json, sys, urllib.request

with urllib.request.urlopen(sys.argv[1].rstrip("/") + "/health", timeout=15) as r:
    h = json.load(r)

state, elapsed = h["state"], h.get("state_seconds") or 0
i, t = h.get("n_initialized") or 0, h.get("n_target_names") or 0
line = f"{state}: {i}/{t} initialized ({h.get('n_failed')} failed)"
if t and i:
    frac = i / t
    eta = elapsed * (1 - frac) / frac if frac > 0 else None
    bar_width = 24
    filled = int(round(bar_width * min(frac, 1.0)))
    bar = "#" * filled + "-" * (bar_width - filled)
    line += f"  [{bar}] {frac*100:5.1f}%  elapsed {elapsed:.0f}s"
    if eta is not None:
        line += f"  eta ~{eta:.0f}s"
else:
    line += f"  elapsed {elapsed:.0f}s"
print(line)
PYEOF
}

cmd_wait() {
    # split, not `local a=.. b=$((a))`: under `set -u` bash evaluates the
    # arithmetic before the first assignment is visible, and the command
    # fails with "timeout: unbound variable".
    local timeout="${1:-1800}"
    local end=$((SECONDS + timeout))
    while [ $SECONDS -lt $end ]; do
        if is_ready; then cmd_status; return 0; fi
        progress_line
        sleep 15
    done
    die "not ready after ${timeout}s"
}

cmd_stats() {
    "$PYTHON" - "$URL" <<'PYEOF' || true
import json, sys, urllib.request

url = sys.argv[1].rstrip("/") + "/stats"
try:
    with urllib.request.urlopen(url, timeout=15) as r:
        d = json.load(r)
except Exception as exc:
    print("stats:   no answer from %s (%s)" % (url, exc))
    raise SystemExit(1)

s = d["summary"]
if not s.get("n"):
    print("no requests served yet")
    raise SystemExit(0)
print(f"{s['n']} operation(s) recorded, {s['n_errors']} error(s)")
if s.get("avg_duration_s") is not None:
    print(f"duration: avg {s['avg_duration_s']:.2f}s  "
          f"min {s['min_duration_s']:.2f}s  max {s['max_duration_s']:.2f}s")
if s.get("last_error"):
    print(f"last error: {s['last_error']}")

print()
print(f"{'when':>8s}  {'kind':<9s} {'dur(s)':>7s}  {'entries':>7s}  error")
import time
for e in d["recent"][-20:]:
    age = time.time() - e["at"]
    dur = e.get("duration_s")
    dur_s = f"{dur:.2f}" if dur is not None else "?"
    n = e.get("n_entries")
    err = e.get("error") or ""
    print(f"{age:7.0f}s  {e['kind']:<9s} {dur_s:>7s}  {str(n) if n is not None else '-':>7s}  {err}")
PYEOF
}

cmd_gui() {
    check_prereqs
    export PYTHONPATH="$CHECKOUT${PYTHONPATH:+:$PYTHONPATH}"
    nohup "$PYTHON" -m eco.status_server.gui --url "$URL" "$@"         </dev/null >/dev/null 2>&1 &
    disown
    echo "launched (pid $!)"
}

case "${1:-}" in
    start)   shift; cmd_start "$@" ;;
    stop)    cmd_stop ;;
    restart)
        if managed_by_systemd; then
            echo "restarting via systemd"
            systemctl --user restart eco-status-server
        else
            cmd_stop; shift; cmd_start -b "$@"
        fi ;;
    status)  cmd_status ;;
    wait)    shift; cmd_wait "$@" ;;
    logs)    shift; [ "${1:-}" = "-f" ] && tail -f "$LOG" || tail -n 100 "$LOG" ;;
    stats)   cmd_stats ;;
    gui)     shift; cmd_gui "$@" ;;
    config)  usage ;;
    ""|-h|--help|help) usage ;;
    *)       usage; exit 2 ;;
esac
