#!/bin/sh
# Torwarden server installer.
#
# POSIX sh, not bash: the estates this ships into include minimal images where
# /bin/bash is not installed, and an installer that cannot run is worse than one
# that is slightly more verbose.
set -eu

BASE="${TORWARDEN_RELEASE_BASE:-https://releases.torwarden.com}"
CHANNEL="${TORWARDEN_CHANNEL:-latest}"
LISTEN=":8080"
RESTART=1
DATA_DIR="/var/lib/torwarden"
CONF_DIR="/etc/torwarden"
BIN="/usr/local/bin/torwarden-server"

die() { echo "install-server: $*" >&2; exit 1; }

usage() {
    cat <<EOF
Usage: install-server.sh [options]

  --listen ADDR    address to bind (default :8080)
  --data-dir PATH  where metrics and fleet data live (default /var/lib/torwarden)
  --version REL    install a specific release (e.g. v1.4.0) instead of latest
  --no-restart     replace the binary but leave the running server alone; you
                   restart it yourself in a maintenance window. Note the new
                   binary does not take effect until you do.
  --help           this text

Set TORWARDEN_RELEASE_BASE to install from an internal mirror instead of
$BASE — required for hosts with no internet access.
EOF
}

while [ $# -gt 0 ]; do
    case "$1" in
        --listen)   LISTEN="${2:-}"; [ -n "$LISTEN" ] || die "--listen needs a value"; shift 2 ;;
        --data-dir) DATA_DIR="${2:-}"; [ -n "$DATA_DIR" ] || die "--data-dir needs a value"; shift 2 ;;
        --no-restart) RESTART=0; shift ;;
        --version)   CHANNEL="${2:-}"; [ -n "$CHANNEL" ] || die "--version needs a release, e.g. v1.4.0"; shift 2 ;;
        --help|-h)  usage; exit 0 ;;
        *)          die "unknown option $1 (try --help)" ;;
    esac
done

[ "$(id -u)" = "0" ] || die "must run as root"

case "$(uname -m)" in
    x86_64|amd64)  ARCH=amd64 ;;
    aarch64|arm64) ARCH=arm64 ;;
    *) die "unsupported architecture $(uname -m); Torwarden ships amd64 and arm64" ;;
esac

command -v curl >/dev/null 2>&1 || die "curl is required"

TMP="$(mktemp -d)"
trap 'rm -rf "$TMP"' EXIT

echo "Downloading torwarden-server ($ARCH) from $BASE/$CHANNEL ..."
curl -fsSL "$BASE/$CHANNEL/torwarden-server-linux-$ARCH" -o "$TMP/torwarden-server" \
    || die "download failed. If this host has no internet, set TORWARDEN_RELEASE_BASE to your mirror."

# Verify against the published checksums. An unverified binary that runs as a
# service is exactly the supply-chain problem the buyers of this product audit
# for, so a missing SHA256SUMS is a hard failure, not a warning.
curl -fsSL "$BASE/$CHANNEL/SHA256SUMS" -o "$TMP/SHA256SUMS" \
    || die "could not fetch SHA256SUMS from $BASE/$CHANNEL"
WANT="$(grep " torwarden-server-linux-$ARCH\$" "$TMP/SHA256SUMS" | awk '{print $1}')"
[ -n "$WANT" ] || die "SHA256SUMS has no entry for torwarden-server-linux-$ARCH"
GOT="$(sha256sum "$TMP/torwarden-server" | awk '{print $1}')"
[ "$WANT" = "$GOT" ] || die "checksum mismatch: expected $WANT, got $GOT"
echo "Checksum verified."

# Nothing to do if this is already the installed binary. Config
# management runs this on a schedule, and an installer that restarts the
# service every single run is one nobody is allowed to schedule.
#
# The config has to exist for that to be true. A current binary next to a
# missing server.conf is a half-installed machine — a rebuild that kept
# /usr/local/bin but lost /etc, most obviously — and skipping there exited 0
# having written nothing and left the service stopped.
if [ -f "$CONF_DIR/server.conf" ] && [ -x "$BIN" ] && [ "$GOT" = "$(sha256sum "$BIN" | awk '{print $1}')" ]; then
    # Matching binary is not the same as a working install. An upgrade
    # interrupted after the stop leaves this exact state — right version, wrong
    # service — and answering "nothing to do" there is what made it
    # unrecoverable by re-running.
    if systemctl is-enabled --quiet torwarden-server 2>/dev/null &&
       ! systemctl is-active --quiet torwarden-server 2>/dev/null; then
        echo "The server is already at this version, but its service is not running. Starting it ..."
        systemctl start torwarden-server
        exit $?
    fi
    echo "The server is already at this version; nothing to do."
    exit 0
fi

id torwarden >/dev/null 2>&1 || useradd --system --home-dir "$DATA_DIR" --shell /sbin/nologin torwarden

# Upgrade or fresh install? The difference matters: on an upgrade the unit is
# already running, and "systemctl enable --now" does NOT restart a running unit
# — it returns success and changes nothing. That is how an operator ends up
# believing they patched a server they did not.
UPGRADE=0
OLD_VERSION=""
WAS_ACTIVE=0
if [ -x "$BIN" ]; then
    UPGRADE=1
    OLD_VERSION="$("$BIN" --version 2>/dev/null | head -1 || echo unknown)"
    systemctl is-active --quiet torwarden-server 2>/dev/null && WAS_ACTIVE=1
    cp -p "$BIN" "$BIN.prev" 2>/dev/null || true
    echo "Upgrading in place. Current: $OLD_VERSION"
fi

# Everything that can fail over the network happens before anything on this
# machine is stopped or replaced.
#
# The unit used to be fetched after the service had been stopped and the binary
# swapped, so a download that failed there left monitoring switched off, and the
# version check above then answered every retry with "nothing to do" — because
# the binary did by then match. A transient network error took the monitoring
# server down and removed the way to bring it back.
curl -fsSL "$BASE/$CHANNEL/torwarden-server.service" -o "$TMP/unit" \
    || die "could not fetch the systemd unit; nothing on this machine has been changed"

if [ "$UPGRADE" = "1" ] && [ "$WAS_ACTIVE" = "1" ] && [ "$RESTART" = "1" ]; then
    echo "Stopping torwarden-server ..."
    systemctl stop torwarden-server
fi

install -o root -g root -m 0755 "$TMP/torwarden-server" "$BIN"
mkdir -p "$CONF_DIR" "$DATA_DIR"
chown torwarden:torwarden "$DATA_DIR"
chmod 750 "$DATA_DIR"

if [ -f "$CONF_DIR/server.conf" ]; then
    echo "Keeping existing $CONF_DIR/server.conf"
else
    cat > "$CONF_DIR/server.conf" <<EOF
listen   = $LISTEN
data_dir = $DATA_DIR

# Retention per resolution, in days.
retention_raw_days = 7
retention_1m_days  = 30
retention_5m_days  = 90
retention_1h_days  = 730
EOF
    chown root:torwarden "$CONF_DIR/server.conf"
    chmod 640 "$CONF_DIR/server.conf"
fi

UNIT=/etc/systemd/system/torwarden-server.service
# sha256sum rather than cmp: cmp lives in diffutils and is absent from a
# minimal Rocky or UBI image, where a missing command exits 127 and "! cmp"
# reads as "they differ" — so every upgrade claimed the unit had been edited and
# wrote a .new file instead of installing it. sha256sum already verifies the
# download, so it is present by definition.
if [ -f "$UNIT" ] && [ "$(sha256sum < "$TMP/unit" | awk '{print $1}')" != \
                      "$(sha256sum < "$UNIT" | awk '{print $1}')" ]; then
    # Someone has edited this unit — a different user, extra hardening, a
    # tmpfiles path. Overwriting it silently on upgrade throws that away, and
    # the failure shows up as a service that will not start.
    cp "$TMP/unit" "$UNIT.new"
    echo "NOTE: $UNIT differs from the one shipped with this release."
    echo "      Yours has been kept. The new one is at $UNIT.new — diff them if"
    echo "      you want the changes."
else
    install -o root -g root -m 0644 "$TMP/unit" "$UNIT"
fi

# SELinux systems (RHEL, Rocky, Alma, Fedora) need the binary relabelled or
# systemd refuses to execute it.
command -v restorecon >/dev/null 2>&1 && restorecon -F "$BIN" 2>/dev/null || true

systemctl daemon-reload

if [ "$RESTART" = "0" ]; then
    systemctl enable torwarden-server >/dev/null 2>&1 || true
    cat <<EOF

Binary replaced. The running server is still the old one — restart when ready:

    sudo systemctl restart torwarden-server

EOF
    exit 0
fi

# enable --now starts a stopped unit; restart is what actually picks up a new
# binary on a running one. Doing both covers fresh installs and upgrades.
systemctl enable torwarden-server >/dev/null 2>&1 || true
systemctl restart torwarden-server

# An upgrade that does not come back is the failure that matters. Check, and
# put the old binary back rather than leaving the operator down.
# Type=simple means systemd reports "active" the instant it execs the binary,
# so is-active alone would call a crash-looping upgrade a success. Wait for it
# to be active AND still be the same process a few seconds later.
i=0
while [ "$i" -lt 20 ]; do
    if systemctl is-active --quiet torwarden-server; then break; fi
    i=$((i + 1))
    sleep 1
done
RESTARTS_BEFORE="$(systemctl show -p NRestarts --value torwarden-server 2>/dev/null || echo 0)"
sleep 5
RESTARTS_AFTER="$(systemctl show -p NRestarts --value torwarden-server 2>/dev/null || echo 0)"
if [ "$RESTARTS_AFTER" != "$RESTARTS_BEFORE" ]; then
    echo "The server started and then exited (restart counter moved $RESTARTS_BEFORE -> $RESTARTS_AFTER)." >&2
    systemctl stop torwarden-server 2>/dev/null || true
fi
if ! systemctl is-active --quiet torwarden-server || [ "$RESTARTS_AFTER" != "$RESTARTS_BEFORE" ]; then
    if [ "$UPGRADE" = "1" ] && [ -x "$BIN.prev" ]; then
        echo "The new server did not start. Rolling back to the previous binary." >&2
        mv -f "$BIN.prev" "$BIN"
        command -v restorecon >/dev/null 2>&1 && restorecon -F "$BIN" 2>/dev/null || true
        systemctl start torwarden-server || true
        die "upgrade failed and was rolled back; see: journalctl -u torwarden-server -n 50"
    fi
    die "the server did not start; see: journalctl -u torwarden-server -n 50"
fi
rm -f "$BIN.prev"

NEW_VERSION="$("$BIN" --version 2>/dev/null | head -1 || echo unknown)"
if [ "$UPGRADE" = "1" ]; then
    cat <<EOF

Upgraded and restarted.
  was: $OLD_VERSION
  now: $NEW_VERSION

Schema migrations run at start and are forward-only; confirm with:

    sudo torwarden-server check
EOF
    exit 0
fi

cat <<EOF

Torwarden server installed and listening on $LISTEN.

There is no default password. The URL that creates your first administrator is
printed once, in the log:

    journalctl -u torwarden-server | grep -A2 'no accounts yet'
EOF
