#!/usr/bin/env bash
# =============================================================================
# Maree-CareFlow - Bare Metal Installer (Ubuntu 22.04 / 24.04)
# =============================================================================
# Usage: sudo bash stack-baremetal/install.sh
# =============================================================================
set -euo pipefail

# WHICH VERSION THIS INSTALLER INSTALLS
#
# It used to `git clone` the default branch and `git pull origin main`, so a
# customer received whatever happened to be on main at the moment they ran it -
# including, on eight consecutive commits in one day, a tree whose Docker stack
# did not come up at all. Nothing recorded WHICH code they got, so a support
# call could not be traced back to a build.
#
# It now checks out a TAG by default. Set MCF_REF to install a different tag,
# branch or commit (`MCF_REF=main` restores the old moving-target behaviour, and
# is the right choice only for development).
MCF_VERSION="${MCF_VERSION:-1.7.15}"
MCF_PACKAGE_URL="${MCF_PACKAGE_URL:-https://maree-careflow.com.au/download/careflow-${MCF_VERSION}.tar.gz}"
# Stamped by scripts/stamp-package-checksum.sh from the artefact that will be
# published, and gated by CI. Never hand-edit: a wrong value fails every
# install, and a plausible-looking one fails none of them for the wrong reason.
MCF_PACKAGE_SHA256="${MCF_PACKAGE_SHA256:-8c5a0c961d555ac7e13af98d6610aeb9d8837da363ec730d3bce17be9caf22d7}"

LOG=/tmp/careflow-baremetal-install.log
# The log captures everything this run prints - on a re-run that can include
# credential-bearing prompt lines - so it must never be world-readable
# (council finding: the default 644 log exposed the live DB password to any
# local user). Created 600 before anything is written to it.
install -m 600 /dev/null "$LOG" 2>/dev/null || true
chmod 600 "$LOG" 2>/dev/null || true
exec > >(tee -a "$LOG") 2>&1

# ── Colour helpers ────────────────────────────────────────────────────────────
RED='\033[0;31m'; GREEN='\033[0;32m'; YELLOW='\033[1;33m'
CYAN='\033[0;36m'; BOLD='\033[1m'; NC='\033[0m'

ok()   { echo -e "${GREEN}  ✔  $*${NC}"; }
info() { echo -e "${CYAN}  ℹ  $*${NC}"; }
warn() { echo -e "${YELLOW}  ⚠  $*${NC}"; }
err()  { echo -e "${RED}  ✘  $*${NC}" >&2; exit 1; }
step() { echo -e "\n${BOLD}${CYAN}━━━  $*  ━━━${NC}"; }

# ── Runtime mode + secret-preservation helpers ────────────────────────────────
# CRITICAL DATA-SAFETY: PHI_ENCRYPTION_KEY decrypts every patient record already
# stored on this server. Rotating it on a re-run or upgrade would make all
# existing encrypted PHI permanently unreadable - a silent, catastrophic data
# loss. So this installer NEVER regenerates an encryption key that already
# exists: it reads the current value out of backend.env and reuses it verbatim
# (the same guarantee the JWT keypair already has), minting a fresh key only
# when none is present. Re-running the installer is therefore always safe.
ENV_FILE=/etc/maree-careflow/backend.env
UPGRADE_MODE=0

# Print the value of KEY=... from the existing env file (empty if absent). Only
# the leading "KEY=" is stripped, so base64 values containing =,+,/ survive.
cf_env() { [[ -f "$ENV_FILE" ]] && sed -n "s/^$1=//p" "$ENV_FILE" | head -n1 || true; }

# Extract the DB password embedded in the existing DATABASE_URL (empty if absent).
# Host-agnostic on purpose (R2): a re-run must recover the password whether the
# database is local (@localhost) or managed (@your-project.neon.tech etc.).
cf_db_pass() {
  [[ -f "$ENV_FILE" ]] || return 0
  sed -n 's#^DATABASE_URL=postgresql+asyncpg://[^:]*:\(.*\)@[^@]*$#\1#p' "$ENV_FILE" | head -n1
}

# ── Managed-PostgreSQL mode (R2) ──────────────────────────────────────────────
# By default this installer runs a LOCAL PostgreSQL 16 on this server. To use a
# MANAGED PostgreSQL service instead (Neon, Supabase, RDS, Azure - anywhere you
# cannot `sudo -u postgres`), export these before running:
#
#   MCF_DB_HOST=your-db-host.example.com   (required - switches managed mode ON)
#   MCF_DB_PORT=5432                       (optional)
#   MCF_DB_NAME=maree_careflow             (optional - database MUST already exist)
#   MCF_DB_USER=careflow                   (optional - user MUST already exist)
#   MCF_DB_PASSWORD=...                    (required in managed mode)
#   MCF_DB_SSLMODE=verify-full             (default since v1.7.0; the server certificate
#                                           MUST be valid for the hostname - the application
#                                           refuses to boot in production with require/prefer/
#                                           disable to a remote host)
#   MCF_DB_SSLROOTCERT=/etc/ssl/certs/ca-certificates.crt  (CA bundle used to verify it)
#   MCF_DB_TLS_INTERNAL_NETWORK=1          (ONLY for a database reached over a private network
#                                           TLS would not cross; writes DATABASE_TLS_INTERNAL_NETWORK=true)
#
# In managed mode the installer does NOT install or start a local PostgreSQL,
# does NOT create the role or database (your provider's console owns those),
# verifies real connectivity up front, and attempts each extension as your
# user with a warning instead of a failure - the schema migrations themselves
# create extensions where permitted and are proven by CI to succeed as a
# NON-superuser without pgvector. On a re-run the mode is recovered from the
# existing backend.env automatically, so the flags are only needed once.
MCF_DB_HOST="${MCF_DB_HOST:-}"
MCF_DB_PORT="${MCF_DB_PORT:-5432}"
MCF_DB_NAME="${MCF_DB_NAME:-maree_careflow}"
MCF_DB_USER="${MCF_DB_USER:-careflow}"
MCF_DB_PASSWORD="${MCF_DB_PASSWORD:-}"
MCF_DB_SSLMODE="${MCF_DB_SSLMODE:-verify-full}"
MCF_DB_SSLROOTCERT="${MCF_DB_SSLROOTCERT:-/etc/ssl/certs/ca-certificates.crt}"
MCF_DB_TLS_INTERNAL_NETWORK="${MCF_DB_TLS_INTERNAL_NETWORK:-0}"
CF_TLS_INTERNAL_LINE=""   # becomes DATABASE_TLS_INTERNAL_NETWORK=true only on explicit opt-in

# Re-run recovery: if backend.env already points at a non-local host, stay in
# managed mode without requiring the flags again.
if [[ -z "$MCF_DB_HOST" && -f "$ENV_FILE" ]]; then
  _existing_host="$(sed -n 's#^DATABASE_URL=postgresql+asyncpg://[^@]*@\([^:/?]*\).*#\1#p' "$ENV_FILE" | head -n1)"
  if [[ -n "$_existing_host" && "$_existing_host" != "localhost" && "$_existing_host" != "127.0.0.1" ]]; then
    MCF_DB_HOST="$_existing_host"
    MCF_DB_PORT="$(sed -n 's#^DATABASE_URL=postgresql+asyncpg://[^@]*@[^:/?]*:\([0-9]*\).*#\1#p' "$ENV_FILE" | head -n1)"
    MCF_DB_PORT="${MCF_DB_PORT:-5432}"
    MCF_DB_NAME="$(sed -n 's#^DATABASE_URL=postgresql+asyncpg://[^@]*@[^/]*/\([^?]*\).*#\1#p' "$ENV_FILE" | head -n1)"
    MCF_DB_NAME="${MCF_DB_NAME:-maree_careflow}"
    MCF_DB_USER="$(sed -n 's#^DATABASE_URL=postgresql+asyncpg://\([^:]*\):.*#\1#p' "$ENV_FILE" | head -n1)"
    MCF_DB_USER="${MCF_DB_USER:-careflow}"
    # Recover the SSL mode too - the async URL carries ?ssl=<mode>. Without
    # this, every re-run/--upgrade of a managed install silently rewrote the
    # URLs back to sslmode=require, changing DATABASE_URL and breaking
    # installs configured with prefer/disable (council finding).
    _existing_ssl="$(sed -n 's#^DATABASE_URL=.*[?&]ssl=\([a-z-]*\).*#\1#p' "$ENV_FILE" | head -n1)"
    [[ -n "$_existing_ssl" ]] && MCF_DB_SSLMODE="$_existing_ssl"
  fi
fi

DB_MODE="local"
if [[ -n "$MCF_DB_HOST" && "$MCF_DB_HOST" != "localhost" && "$MCF_DB_HOST" != "127.0.0.1" ]]; then
  DB_MODE="managed"
fi

for arg in "$@"; do
  case "$arg" in
    --upgrade) UPGRADE_MODE=1 ;;
    -h|--help)
      echo "Usage: sudo bash $0 [--upgrade]"
      echo ""
      echo "  (no flag)   Fresh install, or a safe re-run. Existing encryption keys,"
      echo "              secrets, database and data are always preserved when present."
      echo "  --upgrade   Update an existing install in place: refresh code, Python"
      echo "              dependencies, run database migrations and restart services,"
      echo "              reusing the existing configuration and keys unchanged."
      exit 0 ;;
    *) echo "Unknown argument: $arg (try --help)" >&2; exit 1 ;;
  esac
done

# ── Banner ────────────────────────────────────────────────────────────────────
echo ""
echo -e "${BOLD}${CYAN}"
echo "  ╔════════════════════════════════════════════════════╗"
echo "  ║        Maree-CareFlow Installer                    ║"
echo "  ║        Ubuntu 22.04 / 24.04 - Bare Metal          ║"
echo "  ╚════════════════════════════════════════════════════╝"
echo -e "${NC}"

# ── 1. Check root ──────────────────────────────────────────────────────────────
if [[ $EUID -ne 0 ]]; then
  err "This script must be run as root. Try: sudo bash $0"
fi

# ── 2. apt update + base packages ─────────────────────────────────────────────
step "1/25 - Updating apt and installing base packages"
apt-get update -y

# PYTHON 3.12 IS NOT IN UBUNTU 22.04, AND WE ADVERTISE 22.04 AS SUPPORTED.
#
# 24.04 (noble) ships python3.12 in main. 22.04 (jammy) ships 3.10 and has no
# python3.12 package at all, so `apt-get install python3.12` failed with
# "E: Unable to locate package python3.12" and, under `set -euo pipefail`,
# aborted the installer at step 1 of 28 - before anything useful happened and
# with no message explaining why. download.html and the requirements table both
# list "Ubuntu 22.04 LTS or 24.04 LTS", so this was a total failure for every
# customer on the older LTS.
#
# The deadsnakes PPA is the standard route to a newer CPython on jammy. It is
# added ONLY when the distribution genuinely lacks the package, so 24.04 keeps
# using its own packages from main and gains no third-party source it does not
# need. If the PPA cannot be added we stop HERE with an actionable message,
# rather than 27 steps later with a broken virtualenv.
if ! apt-cache show python3.12 >/dev/null 2>&1; then
  _codename="$(lsb_release -cs 2>/dev/null || echo unknown)"
  info "python3.12 is not available on this release (${_codename}); adding the deadsnakes PPA"
  apt-get install -y software-properties-common gnupg ca-certificates \
    || err "Could not install software-properties-common, which is needed to add a Python 3.12 source."
  add-apt-repository -y ppa:deadsnakes/ppa \
    || err "Could not add the deadsnakes PPA, which provides Python 3.12 on Ubuntu ${_codename}. Maree-CareFlow needs Python 3.11 or newer. Either allow this PPA, or install Python 3.12 yourself and re-run."
  apt-get update -y
  apt-cache show python3.12 >/dev/null 2>&1 \
    || err "python3.12 is still unavailable after adding the deadsnakes PPA. This host cannot be provisioned automatically - install Python 3.12 manually and re-run."
  ok "Python 3.12 source available"
fi

# nginx is DELIBERATELY NOT in this list. This stack's one web server is
# Caddy (installed below); earlier versions of this script also installed
# nginx they never configured or used, and Ubuntu starts nginx on :80 the
# moment apt installs it - so Caddy, which needs :80 (ACME redirect) and
# :443 (the site), failed to bind on exactly the servers this installer had
# just provisioned, and the HTTPS front door was dead behind a warn-only
# service start. The disable step before Caddy starts (step 21) cleans up
# servers provisioned by those earlier versions.
apt-get install -y \
  curl \
  git \
  jq \
  gcc \
  build-essential \
  libssl-dev \
  libffi-dev \
  python3.12 \
  python3.12-venv \
  python3.12-dev \
  gnupg \
  lsb-release \
  ca-certificates \
  apt-transport-https \
  software-properties-common \
  rsync
ok "Base packages installed"

# ── 3. PostgreSQL 16 ──────────────────────────────────────────────────────────
step "2/25 - Installing PostgreSQL 16"
if [[ "$DB_MODE" == "managed" ]]; then
  # Managed mode (R2): the database lives at $MCF_DB_HOST - installing and
  # running a second, empty PostgreSQL here would only waste RAM and confuse
  # every future debugging session. Only the client tools are needed.
  if ! command -v psql &>/dev/null; then
    apt-get install -y postgresql-client
  fi
  ok "Managed PostgreSQL mode: using ${MCF_DB_HOST}:${MCF_DB_PORT} - no local server installed"
elif ! command -v psql &>/dev/null; then
  curl -fsSL https://www.postgresql.org/media/keys/ACCC4CF8.asc \
    | gpg --dearmor -o /usr/share/keyrings/postgresql-archive-keyring.gpg
  echo "deb [signed-by=/usr/share/keyrings/postgresql-archive-keyring.gpg] \
https://apt.postgresql.org/pub/repos/apt $(lsb_release -cs)-pgdg main" \
    > /etc/apt/sources.list.d/pgdg.list
  apt-get update -y
  # postgresql-16-pgvector supplies the "vector" extension the schema REQUIRES
  # (AI semantic search embeddings). Without it, CREATE EXTENSION vector - and
  # therefore the database migrations - fail on a fresh server.
  apt-get install -y postgresql-16 postgresql-contrib-16 postgresql-16-pgvector
  ok "PostgreSQL 16 installed (incl. contrib + pgvector)"
else
  ok "PostgreSQL already installed: $(psql --version)"
  # Belt-and-braces: a pre-existing PostgreSQL may lack the pgvector package.
  # Best-effort install for the detected major version; the per-extension
  # check further below is the real gate and prints exact fix instructions.
  PG_MAJOR=$(psql --version | grep -oE '[0-9]+' | head -1)
  apt-get install -y "postgresql-${PG_MAJOR}-pgvector" "postgresql-contrib-${PG_MAJOR}" 2>/dev/null \
    || warn "Could not auto-install postgresql-${PG_MAJOR}-pgvector - availability is verified below"
fi

# ── 4. Redis 7 ────────────────────────────────────────────────────────────────
step "3/25 - Installing Redis 7"
if ! command -v redis-server &>/dev/null; then
  curl -fsSL https://packages.redis.io/gpg \
    | gpg --dearmor -o /usr/share/keyrings/redis-archive-keyring.gpg
  echo "deb [signed-by=/usr/share/keyrings/redis-archive-keyring.gpg] \
https://packages.redis.io/deb $(lsb_release -cs) main" \
    > /etc/apt/sources.list.d/redis.list
  apt-get update -y
  apt-get install -y redis
  ok "Redis installed"
else
  ok "Redis already installed"
fi

# ── 5. Node.js 24 ─────────────────────────────────────────────────────────────
step "4/25 - Installing Node.js 24"
if ! command -v node &>/dev/null || [[ "$(node --version | cut -d. -f1 | tr -d 'v')" -lt 24 ]]; then
  curl -fsSL https://deb.nodesource.com/setup_24.x | bash -
  apt-get install -y nodejs
  ok "Node.js $(node --version) installed"
else
  ok "Node.js $(node --version) already installed"
fi

# ── 6. System users ───────────────────────────────────────────────────────────
step "5/25 - Creating system users"
# EVERY service user gets an EXPLICIT, real home directory in /var/lib.
# useradd -r without -d records /home/<user> in passwd but never creates it,
# and that broke TWO services on every fresh install (both caught by the
# bare-metal end-to-end CI job):
#   - ollama writes its keypair under $HOME/.ollama and restart-looped
#     forever on "mkdir /home/ollama: permission denied";
#   - asyncpg (the app's PostgreSQL driver) stats $HOME/.postgresql/ for
#     client TLS certs, and with the units' ProtectHome=yes a home under
#     /home answers EACCES instead of ENOENT - which asyncpg treats as
#     fatal, so the API reported db=down while migrations (run as root)
#     had succeeded over the same URL. A home in /var/lib is untouched by
#     ProtectHome and simply answers "no client certs here".
# The directories step below creates and owns each home. usermod repairs
# servers provisioned by earlier versions of this script on any re-run.
declare -A _sysuser_home=(
  [careflow]=/var/lib/maree-careflow
  [ollama]=/var/lib/ollama
)
# `minio` is NOT created any more (Wave 4 item 31d). It is also NOT stopped: the
# case below used to `systemctl stop minio`, and with nothing left to start it
# again an upgrade would have taken a working object store - and every document
# in it - offline on a machine the operator thought they were only updating.
for sysuser in careflow ollama; do
  if ! id "$sysuser" &>/dev/null; then
    useradd -r -s /sbin/nologin -d "${_sysuser_home[$sysuser]}" "$sysuser"
    ok "Created user: $sysuser (home ${_sysuser_home[$sysuser]})"
  else
    ok "User already exists: $sysuser"
  fi
  _old_home="$(getent passwd "$sysuser" | cut -d: -f6)"
  if [[ "$_old_home" != "${_sysuser_home[$sysuser]}" ]]; then
    # usermod refuses (exit 8) while the user has RUNNING processes - and on
    # exactly the older-provisioned servers this repair targets, the services
    # are running. Stop them first; the start steps later in this script
    # bring everything back. Never swallow the usermod itself: a silently
    # skipped repair recreates the asyncpg/ollama home defects.
    case "$sysuser" in
      careflow) systemctl stop careflow-api careflow-worker careflow-beat 2>/dev/null || true ;;
      ollama)   systemctl stop ollama 2>/dev/null || true ;;
    esac
    # An ollama homed elsewhere (e.g. /usr/share/ollama from the official
    # installer) may hold its keypair and downloaded models in $HOME/.ollama -
    # carry them over instead of orphaning them.
    if [[ "$sysuser" == "ollama" && -d "${_old_home}/.ollama" ]]; then
      mkdir -p /var/lib/ollama
      rsync -a "${_old_home}/.ollama" /var/lib/ollama/ || true
    fi
    usermod -d "${_sysuser_home[$sysuser]}" "$sysuser"
    if [[ -d "$_old_home" ]]; then
      ok "$sysuser home repaired to ${_sysuser_home[$sysuser]} (was ${_old_home})"
    else
      ok "$sysuser home repaired to ${_sysuser_home[$sysuser]} (was the non-existent ${_old_home})"
    fi
  fi
done

# ── 7. Directories ────────────────────────────────────────────────────────────
step "6/25 - Creating directories"
# /var/lib/maree-careflow MUST exist here, not merely at the permissions step
# (27/28): careflow-api/worker/beat all carry it in ReadWritePaths, and systemd
# FAILS mount-namespace setup (status=226/NAMESPACE) for a unit whose
# ReadWritePaths entry does not exist - so starting the services (step 24)
# before the directory existed meant NO fresh install could ever start the
# API. Caught by the bare-metal end-to-end CI job's first run.
mkdir -p \
  /opt/maree-careflow \
  /etc/maree-careflow \
  /var/lib/maree-careflow/uploads \
  /var/backups/careflow \
  /var/lib/ollama/models \
  /var/www/maree-careflow/dist
# Service users exist (step 5) - own the data directories NOW, before any
# service starts, so ollama does not thrash in a restart loop until the
# late permissions step happens to rescue them. Step 27 re-asserts these
# (belt-and-braces, idempotent).
chown -R careflow:careflow /var/lib/maree-careflow
chmod 700 /var/lib/maree-careflow /var/lib/maree-careflow/uploads
# The nightly backup unit runs as careflow and writes to /var/backups/careflow
# (BACKUP_DIR default) - /var/backups itself is root-owned, so without this
# the very first scheduled backup would die on mkdir permission.
chown careflow:careflow /var/backups/careflow
chmod 700 /var/backups/careflow
# KEPT DELIBERATELY after MinIO left this installer (item 31d): on a fresh install
# neither the user nor the directory exists and this is a guarded no-op, while on an
# upgrade it preserves the ownership of a practice's existing object-store data
# instead of leaving it for a later permissions sweep to get wrong.
chown -R minio:minio /var/lib/minio 2>/dev/null || true
chown -R ollama:ollama /var/lib/ollama 2>/dev/null || true
ok "Directories created (data dirs owned by their service users up front)"

# ── 8. Install application source ─────────────────────────────────────────────
step "7/25 - Installing application source"
if [[ -n "${CAREFLOW_SOURCE_DIR:-}" && -d "${CAREFLOW_SOURCE_DIR}" ]]; then
  info "Copying from CAREFLOW_SOURCE_DIR=${CAREFLOW_SOURCE_DIR}"
  rsync -a --exclude='.git' "${CAREFLOW_SOURCE_DIR}/" /opt/maree-careflow/
elif [[ -d "$(dirname "$(realpath "$0")")/../backend" ]]; then
  SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
  REPO_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd)"
  info "Copying from repo at: ${REPO_ROOT}"
  rsync -a --exclude='.git' --exclude='node_modules' --exclude='__pycache__' \
    --exclude='*.pyc' --exclude='.env' --exclude='venv' \
    "${REPO_ROOT}/" /opt/maree-careflow/
  ok "Files copied from local repo"
else
  # WHY THIS DOWNLOADS A PACKAGE INSTEAD OF CLONING THE REPOSITORY
  #
  # This used to `git clone` a PRIVATE repository with no token, deploy key or
  # credential helper anywhere in the script - a 404 for every customer who ever
  # ran it, behind an error message that blamed a missing ref.
  #
  # Issuing a deploy token or making the repo public would each have put
  # docs/CONFIDENTIAL/ - the financial model, funding strategy and board papers -
  # on every customer's disk, because `git clone` copies everything tracked. The
  # package is built from an allowlist and cannot carry them (guardrail 56).
  #
  # A clone verified its own contents through git's object hashes; a plain
  # download does not. The SHA-256 below is therefore checked BEFORE anything is
  # extracted, and a mismatch is fatal.
# >>> MCF_FETCH_BEGIN - extracted verbatim by scripts/test-install-from-package.sh
  command -v curl >/dev/null 2>&1 || err "curl is required to download Maree-CareFlow. Install curl and re-run."
  command -v sha256sum >/dev/null 2>&1 || err "sha256sum is required to verify the download. Install coreutils and re-run."

  _tmp="$(mktemp -d)"
  trap 'rm -rf "${_tmp}"' EXIT
  _tarball="${_tmp}/careflow.tar.gz"

  info "Downloading Maree-CareFlow ${MCF_VERSION} from ${MCF_PACKAGE_URL}"
  curl -fsSL --retry 3 --retry-delay 2 -o "${_tarball}" "${MCF_PACKAGE_URL}" \
    || err "Could not download ${MCF_PACKAGE_URL}. Check network/DNS, then confirm the version is listed on https://maree-careflow.com.au"

  _actual="$(sha256sum "${_tarball}" | cut -d' ' -f1)"
  if [[ "${_actual}" != "${MCF_PACKAGE_SHA256}" ]]; then
    echo "     expected ${MCF_PACKAGE_SHA256}" >&2
    echo "     actual   ${_actual}" >&2
    err "DOWNLOAD VERIFICATION FAILED - refusing to install. The file that arrived is not the one this installer was built to install. Usually a truncated download - try again. If it repeats, do NOT work around it: contact support@maree-careflow.com.au before installing anything."
  fi
  ok "Download verified (sha256 ${_actual})"

  # tar only writes what the archive contains, so .env, secrets and uploads -
  # none of which are in the package - survive an upgrade untouched.
  tar -xzf "${_tarball}" -C "${_tmp}" || err "Could not extract the package."
  [[ -d "${_tmp}/careflow-${MCF_VERSION}" ]] || err "The package does not contain careflow-${MCF_VERSION}/ as expected."
  mkdir -p /opt/maree-careflow
  cp -a "${_tmp}/careflow-${MCF_VERSION}/." /opt/maree-careflow/ || err "Could not write to /opt/maree-careflow."
  info "Installed Maree-CareFlow ${MCF_VERSION}"
# <<< MCF_FETCH_END

fi
ok "Application source installed"

# ── 9. Python virtualenv ──────────────────────────────────────────────────────
step "8/25 - Setting up Python virtualenv"
python3.12 -m venv /opt/maree-careflow/venv
/opt/maree-careflow/venv/bin/pip install --upgrade pip wheel setuptools --quiet
info "Installing Python packages (this takes 2-5 minutes)..."
/opt/maree-careflow/venv/bin/pip install --require-hashes -r /opt/maree-careflow/backend/requirements.txt --quiet
ok "Python packages installed"

# ── 10. Ollama ────────────────────────────────────────────────────────────────
step "9/25 - Installing Ollama"
#
# WHY THIS DOES NOT ABORT THE INSTALL (B67, caught by the bare-metal E2E job):
#   This was `curl -fsSL https://ollama.ai/install.sh | sh` with no retry,
#   under `set -e`. When ollama.ai answered HTTP 503 - a third-party outage
#   nobody here controls - the installer DIED at step 9 of 28 with a bare
#   "curl: (22)". The customer was left with system users, application source
#   and a virtualenv, but no database configuration, no services and no
#   explanation. On-premise AI is OPTIONAL (AI_ENABLED defaults to false and
#   the product is a complete CRM without it), so a vendor's bad afternoon
#   must not cost the customer their installation.
#
#   Now: retry a few times, and if it still fails, say so loudly, record that
#   AI is unavailable, and CARRY ON. OLLAMA_AVAILABLE gates the service
#   enable/start below so systemd is never asked to run a binary that is not
#   there.
OLLAMA_AVAILABLE=1
if ! command -v ollama &>/dev/null; then
  export OLLAMA_HOME=/var/lib/ollama
  _ollama_script="$(mktemp)"
  if curl -fsSL --retry 3 --retry-delay 3 --retry-all-errors --connect-timeout 15 \
       https://ollama.ai/install.sh -o "$_ollama_script" \
     && OLLAMA_HOME=/var/lib/ollama sh "$_ollama_script"; then
    ok "Ollama installed"
  else
    OLLAMA_AVAILABLE=0
    warn "Could not install Ollama from ollama.ai (the site may be down, or this
     server may have no outbound access to it). CONTINUING WITHOUT ON-PREMISE AI -
     everything else installs normally and the product is fully usable; the AI
     features stay off until you install Ollama and set AI_ENABLED=true in
     /etc/maree-careflow/backend.env. To add it later:
       curl -fsSL https://ollama.ai/install.sh | OLLAMA_HOME=/var/lib/ollama sh
       systemctl enable --now ollama"
  fi
  rm -f "$_ollama_script"
else
  ok "Ollama already installed"
fi

# ── 11. Object storage: provisioned by nothing, preserved if present ──────────
#
# MinIO USED TO BE INSTALLED HERE (Wave 4 item 31d removed it). Two downloads from
# dl.min.io, a systemd unit, a system user, an env file and a bucket step. All of it
# is gone, because dl.min.io has answered 410 Gone since 2026-09-11: on any new
# install those steps could only ever print a warning and continue.
#
# NOTHING HERE TOUCHES AN INSTALL THAT ALREADY HAS ONE. The binary, the unit, the
# data in /var/lib/minio and the credentials in backend.env are all left exactly as
# they are, and the switch below is read from the existing configuration rather than
# re-derived - so a practice upgrading with the object store on keeps serving every
# document in it. The warning tells them how to come off it while it still works.
CF_OBJECT_STORE="$(cf_env OBJECT_STORE_ENABLED)"
CF_OBJECT_STORE="${CF_OBJECT_STORE:-false}"
if [[ "$CF_OBJECT_STORE" == "true" ]]; then
  warn "This install uses the optional S3/MinIO object store, and MinIO can no longer
     be downloaded (dl.min.io returns 410). Your documents stay reachable on this
     upgrade - nothing about your object store has been changed - but MOVE THEM to
     this server's own disk while the store is still running:

       sudo -u careflow bash -c 'set -a; . /etc/maree-careflow/backend.env; set +a; \\
         cd /opt/maree-careflow/backend && exec /opt/maree-careflow/venv/bin/python \\
         scripts/migrate_object_store_to_local.py'          # dry run, writes nothing

     To copy for real, put --apply INSIDE the quotes, right after the script name:
       ... scripts/migrate_object_store_to_local.py --apply'

     THE WHOLE LINE MATTERS: a bare
     \`python backend/scripts/...\` - which is what this warning said before item 31e -
     runs the system Python, which has none of the application's packages, and with
     no environment the application refuses to start at all because
     PHI_ENCRYPTION_KEY is unset. The rescue command for a practice's clinical
     documents has to be one they can paste.

     It copies, verifies every file by hash, and NEVER deletes the source. Once you
     have watched the application serve the files, set OBJECT_STORE_ENABLED=false in
     /etc/maree-careflow/backend.env and restart careflow-api."
else
  info "Object storage: documents are stored on this server's own disk (the default)"
fi

# ── 12. Caddy 2 ───────────────────────────────────────────────────────────────
step "10/25 - Installing Caddy 2"
if ! command -v caddy &>/dev/null; then
  curl -1sLf 'https://dl.cloudsmith.io/public/caddy/stable/gpg.key' \
    | gpg --dearmor -o /usr/share/keyrings/caddy-stable-archive-keyring.gpg
  curl -1sLf 'https://dl.cloudsmith.io/public/caddy/stable/debian.deb.txt' \
    > /etc/apt/sources.list.d/caddy-stable.list
  apt-get update -y
  apt-get install -y caddy
  ok "Caddy installed"
else
  ok "Caddy already installed"
fi

# ── 13. Interactive prompts (or reuse existing config in --upgrade) ────────────
if [[ "$UPGRADE_MODE" -eq 1 ]]; then
  [[ -f "$ENV_FILE" ]] || err "--upgrade needs an existing install, but ${ENV_FILE} was not found.
     Run the installer without --upgrade to perform a fresh install first."
  step "Upgrade mode - reusing existing configuration (no prompts; nothing regenerated)"
  # Read CAREFLOW_DOMAIN first; fall back to the legacy ALLOWED_HOSTS key so an
  # install written by a pre-1.3.83 version still upgrades cleanly.
  DOMAIN="$(cf_env CAREFLOW_DOMAIN)"
  [[ -n "$DOMAIN" ]] || DOMAIN="$(cf_env ALLOWED_HOSTS | cut -d, -f1)"
  # NEVER default to the vendor's domain here: a corrupt/legacy env would make
  # Caddy attempt public ACME issuance for maree-careflow.com.au from a
  # customer server (council finding). No recoverable domain = stop and say so.
  [[ -n "$DOMAIN" ]] || err "Could not recover this install's domain from ${ENV_FILE}. Re-run WITHOUT --upgrade and enter the domain at the prompt."
  DB_PASSWORD="$(cf_db_pass)"
  [[ -n "$DB_PASSWORD" ]] || err "Could not read the existing database password from ${ENV_FILE}."
  ADMIN_EMAIL="$(cf_env ADMIN_EMAIL)"
  SMTP_HOST="$(cf_env SMTP_HOST)"
  SMTP_PORT="$(cf_env SMTP_PORT)"; SMTP_PORT="${SMTP_PORT:-587}"
  SMTP_USER="$(cf_env SMTP_USER)"; SMTP_USER="${SMTP_USER:-${ADMIN_EMAIL}}"
  SMTP_PASSWORD="$(cf_env SMTP_PASSWORD)"; SMTP_PASSWORD="${SMTP_PASSWORD:-CHANGE_ME}"
  # The env file stores SMTP_PASSWORD single-quoted (it may contain spaces or
  # shell metacharacters); unwrap one quoting layer here so the re-written
  # file does not nest quotes on every upgrade.
  if [[ "$SMTP_PASSWORD" == \'*\' ]]; then
    SMTP_PASSWORD="${SMTP_PASSWORD#\'}"; SMTP_PASSWORD="${SMTP_PASSWORD%\'}"
    SMTP_PASSWORD="${SMTP_PASSWORD//\'\\\'\'/\'}"
  fi
  INSTALL_MODELS=n
  ok "Domain: ${DOMAIN}   Admin: ${ADMIN_EMAIL}   (existing config + keys preserved)"
else
  echo ""
  step "Configuration - answer each question (press Enter for default)"
  echo ""

  # On a re-run, default the domain and DB password to the values already in
  # backend.env so pressing Enter keeps the working credential instead of
  # minting a new one that would not match the existing PostgreSQL role.
  DEFAULT_DOMAIN="$(cf_env CAREFLOW_DOMAIN)"
  [[ -n "$DEFAULT_DOMAIN" ]] || DEFAULT_DOMAIN="$(cf_env ALLOWED_HOSTS | cut -d, -f1)"
  DEFAULT_DOMAIN="${DEFAULT_DOMAIN:-maree-careflow.com.au}"
  DEFAULT_DB_PASS="$(cf_db_pass)"
  DEFAULT_DB_PASS="${DEFAULT_DB_PASS:-$(openssl rand -base64 18 | tr -dc 'a-zA-Z0-9' | head -c24)}"

  read -r -p "  Domain name [${DEFAULT_DOMAIN}]: " DOMAIN_INPUT
  DOMAIN="${DOMAIN_INPUT:-${DEFAULT_DOMAIN}}"
  ok "Domain: ${DOMAIN}"

  # The default is NEVER echoed: on a re-run it is the LIVE database
  # password, and every prompt line lands in the (0600) install log - a
  # secret in a prompt is a secret in a file.
  if [[ -n "$(cf_db_pass)" ]]; then
    _db_prompt_hint="keep existing"
  else
    _db_prompt_hint="auto-generate"
  fi
  read -r -p "  Database password [${_db_prompt_hint}]: " DB_PASSWORD_INPUT
  DB_PASSWORD="${DB_PASSWORD_INPUT:-${DEFAULT_DB_PASS}}"
  # Refuse characters that break the SQL literal, the connection URL, or
  # env-file parsing. The generated default is alphanumeric; a typed
  # password must be too (asyncpg URLs do not accept raw @ : / ' # or
  # spaces, and silently mangling them would lock the install out).
  if [[ ! "$DB_PASSWORD" =~ ^[A-Za-z0-9]+$ ]]; then
    err "The database password may contain letters and digits only (it is embedded in a connection URL). Press Enter next time to auto-generate a safe one."
  fi

  read -r -p "  Admin email: " ADMIN_EMAIL
  [[ -n "$ADMIN_EMAIL" ]] || err "Admin email cannot be empty"
  ok "Admin email: ${ADMIN_EMAIL}"

  # A plain re-run defaults every SMTP prompt to the values already in
  # backend.env (Enter keeps them) - previously pressing Enter here WIPED a
  # working SMTP configuration on every re-run (council finding). The
  # existing password is never echoed.
  DEFAULT_SMTP_HOST="$(cf_env SMTP_HOST)"
  DEFAULT_SMTP_PORT="$(cf_env SMTP_PORT)"; DEFAULT_SMTP_PORT="${DEFAULT_SMTP_PORT:-587}"
  DEFAULT_SMTP_USER="$(cf_env SMTP_USER)"; DEFAULT_SMTP_USER="${DEFAULT_SMTP_USER:-${ADMIN_EMAIL}}"
  read -r -p "  SMTP host [${DEFAULT_SMTP_HOST:-skip - configure later}]: " SMTP_HOST
  SMTP_HOST="${SMTP_HOST:-${DEFAULT_SMTP_HOST}}"

  read -r -p "  SMTP port [${DEFAULT_SMTP_PORT}]: " SMTP_PORT_INPUT
  SMTP_PORT="${SMTP_PORT_INPUT:-${DEFAULT_SMTP_PORT}}"

  read -r -p "  SMTP username [${DEFAULT_SMTP_USER}]: " SMTP_USER_INPUT
  SMTP_USER="${SMTP_USER_INPUT:-${DEFAULT_SMTP_USER}}"

  if [[ -n "$SMTP_HOST" ]]; then
    DEFAULT_SMTP_PASSWORD="$(cf_env SMTP_PASSWORD)"
    if [[ "$DEFAULT_SMTP_PASSWORD" == \'*\' ]]; then
      DEFAULT_SMTP_PASSWORD="${DEFAULT_SMTP_PASSWORD#\'}"; DEFAULT_SMTP_PASSWORD="${DEFAULT_SMTP_PASSWORD%\'}"
      DEFAULT_SMTP_PASSWORD="${DEFAULT_SMTP_PASSWORD//\'\\\'\'/\'}"
    fi
    if [[ -n "$DEFAULT_SMTP_PASSWORD" && "$DEFAULT_SMTP_PASSWORD" != "CHANGE_ME" ]]; then
      read -r -p "  SMTP password [keep existing]: " SMTP_PASSWORD
      SMTP_PASSWORD="${SMTP_PASSWORD:-${DEFAULT_SMTP_PASSWORD}}"
    else
      read -r -p "  SMTP password: " SMTP_PASSWORD
    fi
    # A single quote cannot be represented identically to BOTH consumers of
    # backend.env (the shell that sources it and systemd's EnvironmentFile
    # parser - their quoting grammars differ, council finding), so refuse it
    # rather than hand the services a silently different password.
    if [[ "$SMTP_PASSWORD" == *"'"* ]]; then
      err "The SMTP password may not contain a single quote (') - the env file is read by two parsers whose quoting rules differ. Use a password without ' (most providers let you generate an app password)."
    fi
  else
    SMTP_PASSWORD="CHANGE_ME"
  fi

  echo ""
  echo -e "  ${YELLOW}Ollama AI models enable clinical note summarisation.${NC}"
  echo -e "  ${YELLOW}Models are 4-8 GB each - download takes 5-20 minutes on a fast connection.${NC}"
  read -r -p "  Pull Ollama AI models now? [n - skip, download later]: " INSTALL_MODELS_INPUT
  INSTALL_MODELS="${INSTALL_MODELS_INPUT:-n}"
fi

echo ""
info "Proceeding with installation..."
echo ""

# ── 14. Generate JWT RSA keypair ──────────────────────────────────────────────
step "11/25 - Generating JWT RSA keypair"
if [[ ! -f /etc/maree-careflow/jwt_private.pem ]]; then
  openssl genrsa -out /etc/maree-careflow/jwt_private.pem 2048 2>/dev/null
  openssl rsa \
    -in /etc/maree-careflow/jwt_private.pem \
    -pubout \
    -out /etc/maree-careflow/jwt_public.pem 2>/dev/null
  chmod 600 /etc/maree-careflow/jwt_private.pem
  chmod 644 /etc/maree-careflow/jwt_public.pem
  ok "JWT RS256 key pair generated"
else
  ok "JWT keys already exist - reusing"
fi

# ── 15. Generate (or REUSE) secret keys ───────────────────────────────────────
# See the data-safety note at the top of this script: any encryption key that is
# already present in backend.env is REUSED verbatim, never rotated, so PHI (and
# TOTP/OAuth secrets) already stored on this server stay decryptable across every
# re-run and --upgrade. Fresh keys are minted only for values not yet present.
step "12/25 - Generating (or reusing) secret keys"
TOTP_KEY="$(cf_env TOTP_ENCRYPTION_KEY)";    TOTP_KEY="${TOTP_KEY:-$(openssl rand -base64 32)}"
PHI_KEY="$(cf_env PHI_ENCRYPTION_KEY)";       PHI_KEY="${PHI_KEY:-$(openssl rand -base64 32)}"
OAUTH_KEY="$(cf_env OAUTH_ENCRYPTION_KEY)";   OAUTH_KEY="${OAUTH_KEY:-$(openssl rand -base64 32)}"
SECRET_KEY="$(cf_env SECRET_KEY)";            SECRET_KEY="${SECRET_KEY:-$(openssl rand -base64 48)}"
LICENCE_SECRET="$(cf_env LICENCE_SECRET)";    LICENCE_SECRET="${LICENCE_SECRET:-$(openssl rand -hex 32)}"
# Object-store credentials are CARRIED FORWARD if this install has them and are
# never INVENTED if it does not (Wave 4 item 31d). Generating a real-looking
# credential for a store that was never provisioned is the trap item 31c step 2
# found: it makes an unconfigured install look configured. Empty is the truth.
MINIO_ROOT_USER="$(cf_env MINIO_ACCESS_KEY)"
MINIO_ROOT_PASSWORD="$(cf_env MINIO_SECRET_KEY)"
# Same reasoning for the endpoint: `localhost:9000` used to be hard-coded here, so a
# fresh install that never had an object store was handed a perfectly plausible
# address for one. Empty on a fresh install, unchanged on an upgrade.
MINIO_ENDPOINT_EXISTING="$(cf_env MINIO_ENDPOINT)"
if [[ -f "$ENV_FILE" ]]; then
  ok "Existing secret keys reused - PHI/TOTP/OAuth encryption keys NOT rotated (stored data stays readable)"
else
  ok "Secret keys generated"
fi

# ── 16. Configure PostgreSQL ──────────────────────────────────────────────────
step "13/25 - Configuring PostgreSQL"

if [[ "$DB_MODE" == "managed" ]]; then
  # Managed mode (R2): the provider's console owns role and database creation -
  # `sudo -u postgres` does not exist here and must never be attempted. What we
  # CAN and MUST do is prove, before writing a single config file, that the
  # credentials actually reach the actual database. Fail fast and loud.
  [[ -n "$MCF_DB_PASSWORD" || -n "$(cf_db_pass)" ]] || \
    err "Managed mode needs MCF_DB_PASSWORD (no existing password found in $ENV_FILE)."
  DB_PASSWORD="${MCF_DB_PASSWORD:-$(cf_db_pass)}"
  export PGPASSWORD="$DB_PASSWORD"
  # Error text goes to a private temp file, never a predictable /tmp name root could be tricked into truncating.
  MCF_DB_ERR="$(mktemp)"
  if ! psql "host=${MCF_DB_HOST} port=${MCF_DB_PORT} dbname=${MCF_DB_NAME} user=${MCF_DB_USER} sslmode=${MCF_DB_SSLMODE}" \
       -tc "SELECT 1" 2>"$MCF_DB_ERR" | grep -q 1; then
    echo ""
    cat "$MCF_DB_ERR" >&2 || true
    err "Cannot reach the managed database at ${MCF_DB_HOST}:${MCF_DB_PORT}/${MCF_DB_NAME} as ${MCF_DB_USER}.
   Check host, port, database name, user, password and that this server's IP is
   allowed by the provider's firewall / IP allow-list. Nothing was installed
   against the wrong database."
  fi
  ok "Managed database reachable: ${MCF_DB_HOST}:${MCF_DB_PORT}/${MCF_DB_NAME} as ${MCF_DB_USER} (sslmode=${MCF_DB_SSLMODE})"

  # v1.7.0 (zero-root design R1): a remote database must be reached with the
  # server certificate VERIFIED. `require` encrypts but accepts any certificate,
  # so a hijack of the database hostname would read every clinical record; the
  # application now refuses to boot in production without verify-full. An
  # install made before v1.7.0 carries ?ssl=require in its URLs - prove that
  # verify-full works against the same server and rewrite the URLs, or stop
  # HERE, before any file is written or any migration runs, with the fix spelt
  # out. The only exception is an explicitly declared private network.
  if [[ "$MCF_DB_SSLMODE" != "verify-full" ]]; then
    if [[ "$MCF_DB_TLS_INTERNAL_NETWORK" == "1" ]]; then
      warn "MCF_DB_TLS_INTERNAL_NETWORK=1: keeping sslmode=${MCF_DB_SSLMODE} and writing DATABASE_TLS_INTERNAL_NETWORK=true. Use this ONLY for a database on a private network TLS would not cross."
      CF_TLS_INTERNAL_LINE="DATABASE_TLS_INTERNAL_NETWORK=true"
    elif psql "host=${MCF_DB_HOST} port=${MCF_DB_PORT} dbname=${MCF_DB_NAME} user=${MCF_DB_USER} sslmode=verify-full sslrootcert=${MCF_DB_SSLROOTCERT}" \
         -tc "SELECT 1" 2>"$MCF_DB_ERR" | grep -q 1; then
      ok "The server's TLS certificate verifies for ${MCF_DB_HOST} - the connection URLs will use sslmode=verify-full (was ${MCF_DB_SSLMODE})"
      MCF_DB_SSLMODE="verify-full"
    else
      echo ""
      cat "$MCF_DB_ERR" >&2 || true
      err "The managed database at ${MCF_DB_HOST} was reached with sslmode=${MCF_DB_SSLMODE}, but its TLS
   certificate does NOT verify for that hostname (sslmode=verify-full failed). Since v1.7.0 the
   application refuses to start in production with an unverified connection to a remote database,
   so this install would come up as a 503. Nothing was written and no migration has run.
   Fix one of:
     - ask the provider for the hostname on their certificate and use that as MCF_DB_HOST;
     - point MCF_DB_SSLROOTCERT at the provider's CA bundle (currently ${MCF_DB_SSLROOTCERT});
     - ONLY if the database is on a private network TLS would not cross: re-run with
       MCF_DB_TLS_INTERNAL_NETWORK=1.
   See docs/UPGRADE_GUIDE.md, section 1.6.3 -> 1.7.0."
    fi
  fi

  # Best-effort extensions: many managed providers allow these for normal
  # users; where refused, the migrations' own guarded CREATE EXTENSION and the
  # CI-proven non-superuser path make them optional. Warn, never fail.
  for ext in "uuid-ossp" "pgcrypto" "pg_trgm" "btree_gist"; do
    psql "host=${MCF_DB_HOST} port=${MCF_DB_PORT} dbname=${MCF_DB_NAME} user=${MCF_DB_USER} sslmode=${MCF_DB_SSLMODE}" \
      -c "CREATE EXTENSION IF NOT EXISTS \"${ext}\";" >/dev/null 2>&1 \
      || warn "Could not enable extension '${ext}' on the managed database - continuing (the schema migrations handle this)."
  done
  if psql "host=${MCF_DB_HOST} port=${MCF_DB_PORT} dbname=${MCF_DB_NAME} user=${MCF_DB_USER} sslmode=${MCF_DB_SSLMODE}" \
       -c 'CREATE EXTENSION IF NOT EXISTS "vector";' >/dev/null 2>&1; then
    ok "Managed database configured (vector available - semantic document search enabled)"
  else
    ok "Managed database configured"
    warn "The pgvector (\"vector\") extension is not available on the managed database.
     THE INSTALL CONTINUES - it is not required. Everything clinical, billing,
     rostering and compliance works exactly the same. What you do not get:
     local semantic ('meaning-based') search over uploaded documents. Many
     providers (Neon, Supabase, RDS) let you enable it from their console -
     re-run this installer afterwards if you do."
  fi
  unset PGPASSWORD
else

[[ "$DB_MODE" == "local" ]] && systemctl start postgresql

# CREATE on first run; ALTER on re-runs so a password typed at the prompt is
# actually APPLIED to the role (previously a re-run wrote the new password
# into backend.env while PostgreSQL kept the old one - auth failed and the
# install died at the readiness gate with a misleading symptom). The
# password is charset-restricted at the prompt, so the literal is safe.
if sudo -u postgres psql -tc "SELECT 1 FROM pg_roles WHERE rolname='careflow'" | grep -q 1; then
  sudo -u postgres psql -c "ALTER ROLE careflow WITH PASSWORD '${DB_PASSWORD}';"
else
  sudo -u postgres psql -c "CREATE USER careflow WITH PASSWORD '${DB_PASSWORD}';"
fi

sudo -u postgres psql -tc "SELECT 1 FROM pg_database WHERE datname='maree_careflow'" | grep -q 1 || \
  sudo -u postgres psql -c "CREATE DATABASE maree_careflow OWNER careflow;"

sudo -u postgres psql -d maree_careflow -c "GRANT ALL PRIVILEGES ON DATABASE maree_careflow TO careflow;"

# ── The RUNTIME role: what the application actually connects as (B70) ────────
#
# `careflow` OWNS the schema and runs the migrations. An owner is bound by the
# row-level security policies for QUERIES but not for DDL, so an application
# connecting as the owner could - at any moment, and silently - run
# "DROP POLICY tenant_isolation ON clients" and delete its own tenant wall.
# Anything that reaches SQL execution as the application inherits that.
#
# So the application gets a SECOND role that owns nothing. It can read and
# write every row it is entitled to and nothing about day-to-day use changes;
# it simply cannot remove the wall it lives behind. Migration 0109 grants it
# exactly the application privilege set, from the same source file the policies
# come from.
#
# The runtime password is generated, never prompted: the operator has no reason
# to know it, and one fewer secret to mistype is one fewer failed install.
#
# READ BACK FIRST on a re-run (B72). Minting a fresh one unconditionally - as
# the first version did - ALTERs the live application's credential at step
# 14/28 while the running services still hold the old one in memory, and they
# are not restarted until step 24/28. Every new pooled connection in that
# window fails authentication, and if the run aborts in between (the
# fail-closed pre-upgrade backup gate is inside it) the operator is told
# nothing was changed while the application is left unable to reconnect. The
# owner password beside this one is read back for exactly this reason
# (cf_db_pass, and the comment above it); this now matches.
# The [[ -f ]] guard matters: this runs under `set -euo pipefail`, and on a
# FRESH install backend.env does not exist yet. A bare `sed -n ... "$ENV_FILE"`
# fails, the pipeline's status is that failure, the assignment inherits it and
# set -e kills the installer at step 14/28 - which is exactly what happened on
# the first landing of this change (the bare-metal E2E job caught it, with the
# database provisioned and nothing else). cf_db_pass() above guards the same
# way, for the same reason.
RUNTIME_DB_PASSWORD=""
if [[ -f "$ENV_FILE" ]]; then
  RUNTIME_DB_PASSWORD="$(sed -n 's#^RUNTIME_DATABASE_URL=postgresql+asyncpg://careflow_runtime:\([^@]*\)@.*#\1#p' "$ENV_FILE" | head -n1)"
fi
if [[ -z "$RUNTIME_DB_PASSWORD" ]]; then
  RUNTIME_DB_PASSWORD="$(head -c 32 /dev/urandom | od -An -tx1 | tr -d ' \n')"
fi
if sudo -u postgres psql -tc "SELECT 1 FROM pg_roles WHERE rolname='careflow_runtime'" | grep -q 1; then
  sudo -u postgres psql -c "ALTER ROLE careflow_runtime WITH PASSWORD '${RUNTIME_DB_PASSWORD}';"
else
  sudo -u postgres psql -c "CREATE ROLE careflow_runtime LOGIN NOSUPERUSER NOBYPASSRLS NOCREATEDB NOCREATEROLE PASSWORD '${RUNTIME_DB_PASSWORD}';"
fi

# The SYSTEM identity (F4-3c part 2): the login the app's CONTEXTLESS surfaces
# use (login, public token routes, cross-tenant background tasks) once the PHI
# tables fail close for the bound runtime pool. Never a marker member - being
# unbound IS its function. Password read back on re-runs, exactly as the
# runtime password is (the B72 mid-upgrade credential-loss lesson).
SYSTEM_DB_PASSWORD=""
if [[ -f "$ENV_FILE" ]]; then
  SYSTEM_DB_PASSWORD="$(sed -n 's#^SYSTEM_DATABASE_URL=postgresql+asyncpg://careflow_system:\([^@]*\)@.*#\1#p' "$ENV_FILE" | head -n1)"
fi
if [[ -z "$SYSTEM_DB_PASSWORD" ]]; then
  SYSTEM_DB_PASSWORD="$(head -c 32 /dev/urandom | od -An -tx1 | tr -d ' \n')"
fi
if sudo -u postgres psql -tc "SELECT 1 FROM pg_roles WHERE rolname='careflow_system'" | grep -q 1; then
  sudo -u postgres psql -c "ALTER ROLE careflow_system WITH PASSWORD '${SYSTEM_DB_PASSWORD}';"
else
  sudo -u postgres psql -c "CREATE ROLE careflow_system LOGIN NOSUPERUSER NOBYPASSRLS NOCREATEDB NOCREATEROLE PASSWORD '${SYSTEM_DB_PASSWORD}';"
fi
sudo -u postgres psql -d maree_careflow -c "GRANT CONNECT ON DATABASE maree_careflow TO careflow_system;"
sudo -u postgres psql -d maree_careflow \
  -c "REVOKE TEMPORARY ON DATABASE maree_careflow FROM careflow_system;" \
  -c "REVOKE CREATE ON DATABASE maree_careflow FROM careflow_system;"
# CONNECT only - deliberately NOT TEMP, and TEMP is revoked below (B74). The
# Docker path (02-app-role.sh) was corrected to CONNECT-only in v1.3.95 but
# this line was missed: it granted the runtime role TEMP, which is the
# table-shadowing door, and relied on grant_runtime_role.py ten steps later to
# take it back - a warn-and-continue step, so any failure there shipped an
# install with the door explicitly propped open. This psql runs as the postgres
# superuser, which owns the local database, so the REVOKE genuinely takes
# effect here (unlike migration 0109's, which runs as the non-owner app role).
sudo -u postgres psql -d maree_careflow -c "GRANT CONNECT ON DATABASE maree_careflow TO careflow_runtime;"
sudo -u postgres psql -d maree_careflow \
  -c "REVOKE TEMPORARY ON DATABASE maree_careflow FROM PUBLIC;" \
  -c "REVOKE TEMPORARY ON DATABASE maree_careflow FROM careflow_runtime;" \
  -c "REVOKE CREATE ON DATABASE maree_careflow FROM PUBLIC;" \
  -c "REVOKE CREATE ON DATABASE maree_careflow FROM careflow_runtime;"

# F4-3c: the fail-closed marker role. It is CLUSTER-GLOBAL, NOLOGIN, owns
# nothing, and exists only to be tested by RLS policy expressions - a session
# whose login role INHERITS it, and which sets no tenant context, sees zero
# rows on the measured beachhead instead of every row.
#
# Migration 0111 also creates it, but the migration runs as the app OWNER
# (careflow), which we deliberately create WITHOUT CREATEROLE - so on this
# separated bare-metal shape the migration's CREATE ROLE is skipped
# (permission-tolerant by design) and the wall would never arm. Creating it
# HERE, as the postgres superuser, is what arms role-separated installs, and
# re-creating it here on every run is also what re-arms after a fresh-cluster
# disaster-recovery restore (roles are cluster-global and pg_restore does not
# carry them).
#
# Two grants, and the INHERIT distinction is load-bearing (proven on PG16):
#   * the runtime role is granted the marker INHERITING (the default), so
#     pg_has_role(runtime, marker, 'USAGE') is TRUE and its contextless
#     sessions are gated - this is the arming.
#   * the owner is granted the marker WITH ADMIN but NON-INHERITING, so it can
#     re-grant the marker (grant_runtime_role.py, DR) yet is itself NOT gated
#     (pg_has_role(owner, marker, 'USAGE') stays FALSE) - the owner must keep
#     running migrations contextlessly.
#
# PostgreSQL 16+ ONLY. The per-membership WITH INHERIT/SET options this depends
# on are PG16 syntax; on PG<=15 the GRANTs are a hard syntax error. This
# installer installs PG16 by default, but when pointed at a PRE-EXISTING or
# managed PostgreSQL (which could be 15) the marker step must fail LEGIBLY, not
# abort the whole install with a raw psql syntax error. Unarmed is safe
# (phase-2 equivalent), so on PG<16 we skip the arming with a clear note rather
# than dying. server_version_num is an integer like 160004 for 16.4.
PG_VERNUM="$(sudo -u postgres psql -d maree_careflow -tAc 'SHOW server_version_num' 2>/dev/null | tr -d '[:space:]')"
if [[ "${PG_VERNUM:-0}" -ge 160000 ]]; then
  sudo -u postgres psql -d maree_careflow -v ON_ERROR_STOP=1 <<'SQL'
DO $$ BEGIN
  IF to_regrole('careflow_bound') IS NULL THEN
    CREATE ROLE "careflow_bound" NOLOGIN;
  END IF;
END $$;
GRANT "careflow_bound" TO careflow_runtime WITH INHERIT TRUE;
GRANT "careflow_bound" TO careflow WITH ADMIN OPTION, INHERIT FALSE, SET FALSE;
-- The PHI marker (F4-3c part 2). Created here, owner holds ADMIN
-- (non-inheriting), but NOT granted to the runtime role: ARMING belongs
-- exclusively to grant_runtime_role.py, which first proves a REAL
-- authenticated connection on the app's SYSTEM_DATABASE_URL - a catalog
-- step here cannot see a password mismatch, and arming past one would turn
-- every login on the install into a zero-rows 401.
DO $$ BEGIN
  IF to_regrole('careflow_bound_phi') IS NULL THEN
    CREATE ROLE "careflow_bound_phi" NOLOGIN;
  END IF;
END $$;
GRANT "careflow_bound_phi" TO careflow WITH ADMIN OPTION, INHERIT FALSE, SET FALSE;
SQL
else
  warn "PostgreSQL ${PG_VERNUM:-unknown} is older than 16, which the fail-closed marker role requires (PG16 per-membership INHERIT/SET options). Skipping the fail-closed arming: the tenant wall stays permissive-when-unset (phase-2 - working, but not armed on the measured beachhead). Upgrade this database to PostgreSQL 16+ and re-run the installer to arm it."
fi

# Create each required extension in its OWN psql call. A single multi-statement
# psql -c runs as ONE transaction, so one unavailable extension would silently
# roll back ALL of them and the install would die later, mid-migration, with a
# confusing error. Per-statement creation isolates failures, and the explicit
# "vector" gate below fails fast with the exact fix if pgvector is missing.
# Enabling an extension applies ONLY to this database - it cannot affect any
# other database, cPanel account, or hosted domain on the server.
for ext in "uuid-ossp" "pgcrypto" "pg_trgm" "btree_gist"; do
  sudo -u postgres psql -d maree_careflow -c "CREATE EXTENSION IF NOT EXISTS \"${ext}\";"
done
# pgvector is OPTIONAL, and this used to refuse the install without it.
#
# The hard failure was wrong on the facts. CI runs a dedicated job - "Verify
# NON-superuser install without pgvector (managed PG / cPanel path)" - that
# migrates the whole schema from empty as a NOSUPERUSER role in a database with
# no `vector` extension, and asserts the core schema is intact. So the installer
# was refusing a configuration the project proves works on every push, and it
# refused it for the two cases most likely to hit it: a VPS pointed at managed
# PostgreSQL (Neon, Supabase, RDS - where you cannot create an untrusted
# extension), and any distro without the pgvector package.
#
# What is actually lost without it, stated plainly rather than implied: the
# `local_rag_embeddings.embedding` column falls back to text, so LOCAL SEMANTIC
# SEARCH over uploaded documents is unavailable. Every clinical, billing,
# rostering and compliance feature is unaffected.
if sudo -u postgres psql -d maree_careflow -c 'CREATE EXTENSION IF NOT EXISTS "vector";' >/dev/null 2>&1; then
  ok "PostgreSQL configured (uuid-ossp, pgcrypto, pg_trgm, vector enabled on maree_careflow only)"
else
  ok "PostgreSQL configured (uuid-ossp, pgcrypto, pg_trgm enabled on maree_careflow only)"
  warn "The pgvector (\"vector\") extension is not available on this server.
     THE INSTALL CONTINUES - it is not required. Everything clinical, billing,
     rostering and compliance works exactly the same.
     What you do not get: local semantic ('meaning-based') search over uploaded
     documents. Ordinary keyword search is unaffected.
     To add it later (as root), matching your PostgreSQL major version:
         apt-get install -y postgresql-16-pgvector
         sudo -u postgres psql -d maree_careflow -c 'CREATE EXTENSION IF NOT EXISTS \"vector\";'
     Then re-run this installer - it is safe to re-run."
fi
fi  # end local/managed DB_MODE branch

# ── 17. Write backend.env ─────────────────────────────────────────────────────
step "14/25 - Writing /etc/maree-careflow/backend.env"
# The database URLs are the ONE place the local/managed split reaches the app.
# Both drivers' SSL parameters are proven against the real drivers:
# psycopg2 takes ?sslmode=..., asyncpg takes ?ssl=... (same accepted values).
if [[ "$DB_MODE" == "managed" ]]; then
  MCF_URL_HOSTPART="${MCF_DB_USER}:${DB_PASSWORD}@${MCF_DB_HOST}:${MCF_DB_PORT}/${MCF_DB_NAME}"
  # verify-full needs the CA bundle libpq/asyncpg should trust; both drivers
  # take it as sslrootcert (percent-encoded path).
  MCF_ROOTCERT_Q=""
  if [[ "$MCF_DB_SSLMODE" == "verify-full" ]]; then
    MCF_ROOTCERT_Q="&sslrootcert=$(printf '%s' "$MCF_DB_SSLROOTCERT" | sed 's#/#%2F#g')"
  fi
  MCF_ASYNC_URL="postgresql+asyncpg://${MCF_URL_HOSTPART}?ssl=${MCF_DB_SSLMODE}${MCF_ROOTCERT_Q}"
  MCF_SYNC_URL="postgresql+psycopg2://${MCF_URL_HOSTPART}?sslmode=${MCF_DB_SSLMODE}${MCF_ROOTCERT_Q}"
else
  MCF_ASYNC_URL="postgresql+asyncpg://careflow:${DB_PASSWORD}@localhost:5432/maree_careflow"
  MCF_SYNC_URL="postgresql+psycopg2://careflow:${DB_PASSWORD}@localhost:5432/maree_careflow"
fi
# The runtime (non-owner) role exists only where THIS installer created it.
# On a MANAGED database the operator owns the role model and we do not invent
# roles in someone else's cluster - the value is left empty, the application
# connects as it always has, and migration 0109 prints which mode the install
# is in rather than pretending. Documented in docs/UPGRADE_GUIDE.md so a
# managed-database operator can create the role themselves and fill it in.
if [[ "$DB_MODE" == "managed" ]]; then
  # Read back what the operator set by hand, if anything. backend.env is
  # regenerated on every run, so without this an operator who followed the
  # managed-database steps in docs/UPGRADE_GUIDE.md would have the separation
  # silently reverted by their next upgrade, with no message (B72). The role
  # NAME is read back too: writing "careflow_runtime" unconditionally made
  # migration 0109 report "the installer creates it" on the one avenue where
  # the installer deliberately does not.
  MCF_RUNTIME_ASYNC_URL="$(cf_env RUNTIME_DATABASE_URL)"
  MCF_RUNTIME_DB_ROLE="$(cf_env CAREFLOW_RUNTIME_DB_ROLE)"
  MCF_SYSTEM_ASYNC_URL="$(cf_env SYSTEM_DATABASE_URL)"
  MCF_SYSTEM_DB_ROLE="$(cf_env CAREFLOW_SYSTEM_DB_ROLE)"
  # v1.7.0: the application's TLS gate checks these two URLs as well. When the
  # probe above proved verify-full (or the operator declared a private network),
  # an older `?ssl=require` on the read-back URLs is rewritten the same way the
  # main URL is, so a B70 role-separated managed install upgrades cleanly.
  if [[ "$MCF_DB_SSLMODE" == "verify-full" ]]; then
    MCF_RUNTIME_ASYNC_URL="$(printf '%s' "$MCF_RUNTIME_ASYNC_URL" | sed -E "s#([?&])ssl=(require|prefer|allow|verify-ca)(&sslrootcert=[^&]*)?#\1ssl=verify-full${MCF_ROOTCERT_Q}#")"
    MCF_SYSTEM_ASYNC_URL="$(printf '%s' "$MCF_SYSTEM_ASYNC_URL" | sed -E "s#([?&])ssl=(require|prefer|allow|verify-ca)(&sslrootcert=[^&]*)?#\1ssl=verify-full${MCF_ROOTCERT_Q}#")"
  fi
else
  MCF_RUNTIME_ASYNC_URL="postgresql+asyncpg://careflow_runtime:${RUNTIME_DB_PASSWORD}@localhost:5432/maree_careflow"
  MCF_RUNTIME_DB_ROLE="careflow_runtime"
  MCF_SYSTEM_ASYNC_URL="postgresql+asyncpg://careflow_system:${SYSTEM_DB_PASSWORD}@localhost:5432/maree_careflow"
  MCF_SYSTEM_DB_ROLE="careflow_system"
fi
# Re-runs and --upgrade regenerate this file, so every OPERATOR-ADJUSTABLE
# value must be read back from the existing file first, with the template's
# value only as the fresh-install default - otherwise an upgrade silently
# reverts settings the customer changed (council finding: AI_ENABLED,
# SMTP_TLS, the from-address, the Ollama models, LOG_LEVEL, ALLOWED_ORIGINS,
# storage settings were all being reset on every re-run).
CF_AI_ENABLED="$(cf_env AI_ENABLED)";               CF_AI_ENABLED="${CF_AI_ENABLED:-false}"
CF_SMTP_TLS="$(cf_env SMTP_TLS)";                   CF_SMTP_TLS="${CF_SMTP_TLS:-true}"
CF_SMTP_FROM="$(cf_env SMTP_FROM_ADDRESS)";         CF_SMTP_FROM="${CF_SMTP_FROM:-${ADMIN_EMAIL}}"
CF_LOG_LEVEL="$(cf_env LOG_LEVEL)";                 CF_LOG_LEVEL="${CF_LOG_LEVEL:-info}"
CF_OLLAMA_MODEL="$(cf_env OLLAMA_DEFAULT_MODEL)";   CF_OLLAMA_MODEL="${CF_OLLAMA_MODEL:-qwen2.5:7b}"
CF_OLLAMA_EMBED="$(cf_env OLLAMA_EMBEDDING_MODEL)"; CF_OLLAMA_EMBED="${CF_OLLAMA_EMBED:-nomic-embed-text}"
CF_ALLOWED_ORIGINS="$(cf_env ALLOWED_ORIGINS)";     CF_ALLOWED_ORIGINS="${CF_ALLOWED_ORIGINS:-https://${DOMAIN}}"
CF_STORAGE_PROVIDER="$(cf_env STORAGE_PROVIDER)";   CF_STORAGE_PROVIDER="${CF_STORAGE_PROVIDER:-local}"
CF_UPLOAD_DIR="$(cf_env UPLOAD_DIR)";               CF_UPLOAD_DIR="${CF_UPLOAD_DIR:-/var/lib/maree-careflow/uploads}"
# SMTP_PASSWORD is the one free-text credential a human types: it may hold
# spaces or shell metacharacters, and this file is both sourced as shell
# (the migration step) and parsed by systemd's EnvironmentFile - so it is
# written single-quoted, with any embedded single quote escaped. A quoted
# value that arrives via cf_env on a re-run is unwrapped by the shell when
# sourced, so preservation still round-trips verbatim.
SMTP_PASSWORD_ESC="${SMTP_PASSWORD//\'/\'\\\'\'}"
cat > /etc/maree-careflow/backend.env <<ENVFILE
# Maree-CareFlow Backend Environment
# Generated by install.sh on $(date -u +"%Y-%m-%d %H:%M:%S UTC")
# WARNING: Keep this file secret - do not commit to version control.

DATABASE_URL=${MCF_ASYNC_URL}
DATABASE_SYNC_URL=${MCF_SYNC_URL}
${CF_TLS_INTERNAL_LINE}
# The role the RUNNING APPLICATION connects as (B70). It owns nothing, so it
# cannot DROP POLICY or DISABLE ROW LEVEL SECURITY on its own tables - the
# tenant wall stops being advisory. DATABASE_URL above stays the owner and is
# what migrations, backups and every operator runbook use; leave both in place.
RUNTIME_DATABASE_URL=${MCF_RUNTIME_ASYNC_URL}
CAREFLOW_RUNTIME_DB_ROLE=${MCF_RUNTIME_DB_ROLE}
# The SYSTEM identity (F4-3c part 2): the login the contextless surfaces use
# (login itself, public token routes, cross-tenant background tasks) once the
# PHI tables fail close for the bound runtime role. Never a marker member.
# The PHI wall only ARMS after grant_runtime_role.py proves a real
# authenticated connection on this URL.
SYSTEM_DATABASE_URL=${MCF_SYSTEM_ASYNC_URL}
CAREFLOW_SYSTEM_DB_ROLE=${MCF_SYSTEM_DB_ROLE}
REDIS_URL=redis://localhost:6379/0

SECRET_KEY=${SECRET_KEY}
LICENCE_SECRET=${LICENCE_SECRET}
JWT_PRIVATE_KEY_FILE=/etc/maree-careflow/jwt_private.pem
JWT_PUBLIC_KEY_FILE=/etc/maree-careflow/jwt_public.pem
JWT_ALGORITHM=RS256
JWT_ACCESS_TOKEN_EXPIRE_MINUTES=15
JWT_REFRESH_TOKEN_EXPIRE_DAYS=30

TOTP_ENCRYPTION_KEY=${TOTP_KEY}
PHI_ENCRYPTION_KEY=${PHI_KEY}
OAUTH_ENCRYPTION_KEY=${OAUTH_KEY}

ADMIN_EMAIL=${ADMIN_EMAIL}

SMTP_HOST=${SMTP_HOST}
SMTP_PORT=${SMTP_PORT}
SMTP_USER=${SMTP_USER}
SMTP_PASSWORD='${SMTP_PASSWORD_ESC}'
SMTP_FROM_ADDRESS=${CF_SMTP_FROM}
SMTP_TLS=${CF_SMTP_TLS}

MINIO_ENDPOINT=${MINIO_ENDPOINT_EXISTING}
MINIO_ACCESS_KEY=${MINIO_ROOT_USER}
MINIO_SECRET_KEY=${MINIO_ROOT_PASSWORD}
MINIO_BUCKET_DOCUMENTS=careflow-documents
MINIO_USE_SSL=false
OBJECT_STORE_ENABLED=${CF_OBJECT_STORE}

OLLAMA_URL=http://localhost:11434
OLLAMA_DEFAULT_MODEL=${CF_OLLAMA_MODEL}
OLLAMA_EMBEDDING_MODEL=${CF_OLLAMA_EMBED}

# File storage. NOT inside /opt/maree-careflow: that directory is the git
# clone, and scripts/update.sh runs 'git pull' in it - clinical documents
# must never live in a working tree where 'git clean' could remove them or
# 'git add -A' could commit them. /var/lib is the FHS location for
# application state and is added to the unit's ReadWritePaths.
# (These git commands are quoted with PLAIN QUOTES on purpose: this text
# sits inside an UNQUOTED heredoc, where backticks EXECUTE. The backticked
# originals silently ran git pull / git add -A in the operator's cwd on
# every install, and when a git pull actually fast-forwarded, its
# multi-line output landed IN this file and broke the env parse - caught
# by the bare-metal E2E job.)
# (The old code default, /data/uploads, was outside ReadWritePaths
# entirely, so ProtectSystem=strict made it read-only even for root.)
STORAGE_PROVIDER=${CF_STORAGE_PROVIDER}
UPLOAD_DIR=${CF_UPLOAD_DIR}
ENVIRONMENT=production
DEBUG=false
# The code tree is root-owned and read-only to the service (hardening), so
# Python must not try to write __pycache__ next to the modules.
PYTHONDONTWRITEBYTECODE=1
LOG_LEVEL=${CF_LOG_LEVEL}
ALLOWED_ORIGINS=${CF_ALLOWED_ORIGINS}
CAREFLOW_DOMAIN=${DOMAIN}

AI_ENABLED=${CF_AI_ENABLED}

CELERY_BROKER_URL=redis://localhost:6379/1
CELERY_RESULT_BACKEND=redis://localhost:6379/2
ENVFILE

# Ownership model for /etc/maree-careflow (council-hardened): ROOT owns the
# secrets, the careflow GROUP may read them, nobody may write them at
# runtime. The service user must be able to READ its secrets BEFORE the
# services start (step 24) - the old careflow-owns-everything model fixed
# that but also let a compromised worker rewrite its own LICENCE_SECRET,
# JWT keypair or DATABASE_URL and persist across restarts. root:careflow
# with 750/640 gives the read without the write (the units' ProtectSystem
# =strict + the removal of /etc/maree-careflow from ReadWritePaths enforce
# it inside the service namespace as well).
chown -R root:careflow /etc/maree-careflow
chmod 750 /etc/maree-careflow
chmod 640 /etc/maree-careflow/backend.env /etc/maree-careflow/jwt_private.pem
chmod 644 /etc/maree-careflow/jwt_public.pem 2>/dev/null || true
ok "backend.env written (root:careflow, mode 640 - readable, not writable, by the service)"

# ── Backup encryption settings (backup.env) ───────────────────────────────────
# The nightly timer is enabled below, but the backup script REFUSES to write
# unencrypted PHI - so without a passphrase the schedule fires and every run
# refuses, which is "backups enabled" in name only (council finding). Mirror
# the cPanel installer: generate a strong passphrase on first install, never
# overwrite an existing one, and tell the operator to store it OFF this box.
if [[ ! -f /etc/maree-careflow/backup.env ]]; then
  cat > /etc/maree-careflow/backup.env <<BACKUPENV
# Maree-CareFlow backup encryption. Generated by install.sh.
# STORE A COPY OF THIS PASSPHRASE SOMEWHERE SAFE OFF THIS SERVER:
# a backup you cannot decrypt is not a backup.
CAREFLOW_BACKUP_PASSPHRASE=$(openssl rand -hex 32)
BACKUPENV
  chown root:careflow /etc/maree-careflow/backup.env
  chmod 640 /etc/maree-careflow/backup.env
  ok "backup.env written with a generated passphrase - copy it somewhere safe OFF this server"
else
  chown root:careflow /etc/maree-careflow/backup.env
  chmod 640 /etc/maree-careflow/backup.env
  ok "backup.env already exists - passphrase preserved"
fi

# ── 19. Write Caddyfile ───────────────────────────────────────────────────────
step "15/25 - Writing Caddyfile"
CADDYFILE_TPL=/opt/maree-careflow/stack-baremetal/Caddyfile.template
if [[ -f "$CADDYFILE_TPL" ]]; then
  sed "s/__DOMAIN__/${DOMAIN}/g" "$CADDYFILE_TPL" > /etc/caddy/Caddyfile
  ok "Caddyfile written from template"
else
  cat > /etc/caddy/Caddyfile <<CADDYFILE
${DOMAIN} {
  encode gzip

  header {
    Strict-Transport-Security "max-age=31536000; includeSubDomains; preload"
    X-Content-Type-Options nosniff
    X-Frame-Options DENY
    X-XSS-Protection "1; mode=block"
    Referrer-Policy strict-origin-when-cross-origin
    Permissions-Policy "camera=(), microphone=(), geolocation=()"
    -Server
  }

  handle /api/* {
    reverse_proxy localhost:8000
  }

  handle /ws/* {
    reverse_proxy localhost:8000 {
      header_up Upgrade {>Upgrade}
      header_up Connection {>Connection}
    }
  }

  handle {
    root * /var/www/maree-careflow/dist
    try_files {path} /index.html
    file_server
  }
}
CADDYFILE
  ok "Caddyfile written"
fi

# ── 20. Systemd unit files ────────────────────────────────────────────────────
step "16/25 - Installing systemd unit files"
SYSTEMD_SRC=/opt/maree-careflow/stack-baremetal/systemd
if [[ -d "$SYSTEMD_SRC" ]]; then
  cp "$SYSTEMD_SRC"/*.service /etc/systemd/system/
  # Timers too - the nightly backup is a .timer, and copying only *.service
  # would install the job with nothing to trigger it.
  cp "$SYSTEMD_SRC"/*.timer /etc/systemd/system/ 2>/dev/null || true
  ok "Systemd unit files installed"
else
  warn "Systemd unit files not found at $SYSTEMD_SRC - skipping"
fi

# ── 21. systemctl daemon-reload ───────────────────────────────────────────────
step "17/25 - Reloading systemd"
systemctl daemon-reload
ok "systemd reloaded"

# ── 22. Enable services ───────────────────────────────────────────────────────
step "18/25 - Enabling services"
# postgresql is enabled only when this server actually runs it (local mode).
[[ "$DB_MODE" == "local" ]] && systemctl enable postgresql 2>/dev/null || true
systemctl enable redis-server \
  careflow-api careflow-worker careflow-beat caddy 2>/dev/null || true
# Only enable the optional components whose binaries actually installed -
# systemd must never be told to run something that is not on disk (B67).
[[ "${OLLAMA_AVAILABLE:-1}" == "1" ]] && systemctl enable ollama 2>/dev/null || true
# Nightly encrypted backup. Until now NO installer scheduled one, so a clinical
# records system ran with zero backups until an operator happened to read the
# script's header. Enabling the timer is the whole point of shipping the script.
systemctl enable careflow-backup.timer 2>/dev/null || true
systemctl start careflow-backup.timer 2>/dev/null || true
if systemctl is-enabled careflow-backup.timer >/dev/null 2>&1; then
  ok "Nightly backup timer enabled (02:15 local; set CAREFLOW_BACKUP_PASSPHRASE or CAREFLOW_BACKUP_AGE_RECIPIENT in /etc/maree-careflow/backup.env - the backup REFUSES to write unencrypted PHI)"
else
  warn "Could not enable careflow-backup.timer - this install has NO scheduled backup. Enable it manually before going live."
fi
ok "Services enabled"

step "19/25 - Starting base services"
[[ "$DB_MODE" == "local" ]] && systemctl start postgresql
systemctl start redis-server
if [[ "${OLLAMA_AVAILABLE:-1}" == "1" ]]; then
  systemctl start ollama || warn "ollama failed to start - check: journalctl -u ollama -n 20"
else
  info "Skipping Ollama (not installed) - on-premise AI features stay off"
fi
# Caddy is this stack's ONE web server and it must own :80 (ACME redirect)
# and :443 (the site). Earlier versions of this installer also apt-installed
# nginx they never configured - and Ubuntu starts nginx on :80 the moment it
# is installed - so Caddy failed to bind on exactly the servers this script
# had just provisioned, behind a warn-only start. This stack installs onto a
# dedicated server (it provisions PostgreSQL, Redis, Ollama and the
# app), so an nginx found here is either our own earlier version's leftover
# or a distro default - disable it so the proxy the stack actually uses can
# bind its ports. Stated loudly, never silently.
if systemctl is-active --quiet nginx 2>/dev/null || systemctl is-enabled --quiet nginx 2>/dev/null; then
  # If nginx carries anything beyond the distro default site, this server is
  # serving OTHER sites - disabling nginx would take them down persistently.
  # Refuse and explain, unless the operator explicitly claims port 80 with
  # MCF_TAKE_PORT_80=1.
  # `|| true` is LOAD-BEARING: find exits non-zero when either directory is
  # absent (nginx.org packages ship conf.d but no sites-enabled - exactly the
  # already-running-nginx server this guard exists for) and `head -1` can
  # SIGPIPE find under pipefail - either would kill the installer here with
  # no message at all (council finding, empirically reproduced). An empty
  # result already means "no custom config".
  _nginx_custom="$(find /etc/nginx/sites-enabled /etc/nginx/conf.d -mindepth 1 \
    ! -name default ! -name '*.dpkg-*' 2>/dev/null | head -1 || true)"
  if [[ -n "$_nginx_custom" && "${MCF_TAKE_PORT_80:-0}" != "1" ]]; then
    err "nginx is running on this server WITH ITS OWN SITE CONFIG (${_nginx_custom}).
   Maree-CareFlow's bare-metal stack uses Caddy as its web server and needs
   ports 80/443, but disabling nginx would take this server's other sites
   down. This installer is for a DEDICATED server. If you really want this
   box (and its ports 80/443) for Maree-CareFlow, re-run with
   MCF_TAKE_PORT_80=1 - nginx will be disabled (re-enable later with:
   systemctl enable --now nginx)."
  fi
  systemctl disable --now nginx 2>/dev/null || true
  ok "nginx disabled - it held port 80 and this stack's web server is Caddy (re-enable: systemctl enable --now nginx)"
fi
# reload-or-restart, NOT start: apt starts Caddy at install time with the
# DISTRO DEFAULT Caddyfile, so a plain `start` here is a no-op and Caddy
# keeps serving the default page on :80 with nothing on :443 - the HTTPS
# front door was dead on every fresh install (caught by the bare-metal E2E
# job's front-door check). reload-or-restart loads the Caddyfile this
# script just wrote, and equals `start` when Caddy is not yet running.
systemctl reload-or-restart caddy || warn "caddy failed to start - check: journalctl -u caddy -n 20"
ok "Base services started"

# ── 23. Run database migrations ───────────────────────────────────────────────
step "20/25 - Running database migrations"
sleep 3
# Fail-closed pre-upgrade backup (B64): install.sh --upgrade is the upgrade
# command the download page tells customers to run, and until now only
# scripts/update.sh had this gate - so the documented path migrated live data
# with NO snapshot behind it. A failed migration with no backup is the top
# data-loss risk; a verified backup MUST succeed before we migrate, or we
# abort. Fresh installs have no data yet and skip it.
if [[ "$UPGRADE_MODE" -eq 1 && "${CAREFLOW_SKIP_PREUPGRADE_BACKUP:-0}" != "1" ]]; then
  if [[ -x /opt/maree-careflow/deploy/scripts/careflow-backup.sh || -f /opt/maree-careflow/deploy/scripts/careflow-backup.sh ]]; then
    echo "  -> Taking a fail-closed pre-upgrade backup first..."
    if ! bash /opt/maree-careflow/deploy/scripts/careflow-backup.sh --preupgrade; then
      err "Pre-upgrade backup FAILED - aborting the upgrade (NO migration run).
     Fix the backup, or set CAREFLOW_SKIP_PREUPGRADE_BACKUP=1 to override at your own risk."
    fi
  else
    warn "careflow-backup.sh not found in the existing install - upgrading WITHOUT a pre-upgrade backup (pre-1.3.8x install)."
  fi
fi
cd /opt/maree-careflow/backend
# backend.env lives in /etc/maree-careflow (the systemd EnvironmentFile), so it is
# NOT auto-loaded here. Export it so Alembic's env.py sees the real DATABASE_URL
# (otherwise it would fall back to the config default / raise). Run via
# `python -m alembic` so the current directory is importable - env.py imports the
# `app` package, which the bare `alembic` console script could not resolve
# (ModuleNotFoundError: No module named 'app').
set -a
# shellcheck disable=SC1091
. /etc/maree-careflow/backend.env
set +a
/opt/maree-careflow/venv/bin/python -m alembic upgrade head
ok "Migrations complete"

# Apply the runtime role's privileges, idempotently, as the OWNER (B72).
# Migration 0109 grants them ONCE and is then stamped forever, so any install
# whose runtime role appears later - a managed database, cPanel, or an operator
# following the manual steps - would otherwise never receive them and the
# application would answer "permission denied" on every query with no way back
# short of hand-written SQL. This step is a no-op when there is no runtime role.
if ! /opt/maree-careflow/venv/bin/python \
     /opt/maree-careflow/backend/scripts/grant_runtime_role.py; then
  warn "Could not apply the runtime role's privileges - the application may be"
  warn "unable to read its own data. The reason is printed above. Re-run:"
  # The env file MUST be sourced. `sudo -u careflow` starts a clean
  # environment, so the previously-printed command reached the script with no
  # DATABASE_URL at all - it printed "no DATABASE_URL - nothing to do" and
  # exited 0, and the operator, having run the documented repair and seen a
  # clean exit, believed the install was fixed while every query still
  # answered "permission denied" (B73, Council X X-7). Verified by running it.
  warn "  sudo bash -c 'set -a; . /etc/maree-careflow/backend.env; set +a; exec /opt/maree-careflow/venv/bin/python /opt/maree-careflow/backend/scripts/grant_runtime_role.py'"
fi

# ── 24. Deploy frontend ───────────────────────────────────────────────────────
# Sovereign: deploy the COMMITTED, CI-verified frontend/dist that ships with the
# source. It is byte-identical to a from-source build (the CI dist-freshness gate
# guarantees this), so the server runs NO package manager - no npm, no pnpm, no
# build - eliminating the install-time supply-chain surface entirely.
step "21/25 - Deploying frontend (committed CI-verified build)"
if [[ ! -f /opt/maree-careflow/frontend/dist/index.html ]]; then
  err "frontend/dist/index.html is missing - the source checkout is incomplete."
fi
rsync -a /opt/maree-careflow/frontend/dist/ /var/www/maree-careflow/dist/
ok "Frontend deployed (no build step)"

# ── 25. Start application services ───────────────────────────────────────────
step "22/25 - Starting application services"
# restart, NOT start: on an --upgrade or re-run the services are already
# active, and `start` is a no-op - the old processes would keep serving the
# OLD code from memory while the database had just been migrated to the new
# head, so /health/ready answered 503 "schema stale" forever (or worse, a
# migration-less upgrade silently kept serving the previous version). The
# help text always promised "restart services"; now it is true. restart
# equals start for a unit that is not running, so fresh installs behave
# identically.
systemctl restart careflow-api
sleep 5
systemctl restart careflow-worker careflow-beat
ok "Application services (re)started on the just-installed code"

# ── 26. Wait for backend ──────────────────────────────────────────────────────
step "23/25 - Waiting for backend to be ready"
# WHY /health/ready AND NOT /api/v1/health:
#   There is no /api/v1/health route. main.py registers /health and
#   /health/ready; the only /api/v1 health route in the codebase is
#   /api/v1/ai/health. So `curl -sf` 404'd on EVERY attempt, the loop always
#   exhausted all 30 retries, warned, broke - and the installer then printed
#   "Maree-CareFlow installed successfully!" regardless. The readiness gate was
#   structurally incapable of passing and its failure was non-fatal, so success
#   was announced without ever being observed.
#
#   /health/ready is also the stronger check: it returns 200 only when the
#   database is reachable AND migrated to the code's head revision. /health
#   alone can answer 200 over an unmigrated database - the exact failure that
#   shipped on cPanel in v1.3.80.
RETRY=0
MAX_RETRIES=30
BACKEND_READY=0
until curl -sf http://localhost:8000/health/ready >/dev/null 2>&1; do
  RETRY=$((RETRY + 1))
  if [[ $RETRY -ge $MAX_RETRIES ]]; then
    break
  fi
  printf "  ${CYAN}  [%2d/%d] waiting...${NC}\r" "$RETRY" "$MAX_RETRIES"
  sleep 3
done
if curl -sf http://localhost:8000/health/ready >/dev/null 2>&1; then
  BACKEND_READY=1
  ok "Backend is ready (database reachable and migrated)"
fi
echo ""

# ── 27. Pull Ollama models ────────────────────────────────────────────────────
step "24/25 - Ollama AI models"
if [[ "${OLLAMA_AVAILABLE:-1}" != "1" ]]; then
  info "Ollama is not installed on this server - skipping model downloads"
elif [[ "${INSTALL_MODELS,,}" == "y" || "${INSTALL_MODELS,,}" == "yes" ]]; then
  info "Pulling qwen2.5:7b (~4.7 GB)..."
  _ollama_bin="$(command -v ollama || echo /usr/local/bin/ollama)"
  "$_ollama_bin" pull qwen2.5:7b     || warn "Failed to pull qwen2.5:7b - retry: ollama pull qwen2.5:7b"
  info "Pulling nomic-embed-text (~274 MB)..."
  "$_ollama_bin" pull nomic-embed-text || warn "Failed to pull nomic-embed-text"
  info "Pulling llava:7b (~4.1 GB)..."
  "$_ollama_bin" pull llava:7b        || warn "Failed to pull llava:7b"
  ok "AI models downloaded"
else
  info "Skipping AI model download."
  info "Pull later with: ollama pull qwen2.5:7b && ollama pull nomic-embed-text"
fi

# ── 28. Set file permissions ──────────────────────────────────────────────────
step "25/25 - Setting file permissions"

# Uploaded clinical documents live OUTSIDE the git clone (see UPLOAD_DIR above).
# Created here rather than left to the app so that ownership and mode are right
# from the first request, and so an operator can see the directory exists.
# 0700: this holds PHI and nothing but the service account should read it.
mkdir -p /var/lib/maree-careflow/uploads
chown -R careflow:careflow /var/lib/maree-careflow
chmod 700 /var/lib/maree-careflow /var/lib/maree-careflow/uploads

# /opt/maree-careflow (code + venv) stays ROOT-owned on purpose: the service
# reads it, world-read handles that, and a service user that can rewrite its
# own code or site-packages turns any exec bug into durable persistence
# (council finding). The units' ProtectSystem=strict makes it read-only in
# the service namespace too. /etc/maree-careflow is root:careflow 750/640 -
# see the backend.env step for the reasoning.
chown -R root:root /opt/maree-careflow
chown -R root:careflow /etc/maree-careflow
chmod 750 /etc/maree-careflow
chown -R minio:minio /var/lib/minio 2>/dev/null || true
chown -R ollama:ollama /var/lib/ollama 2>/dev/null || true
chown -R www-data:www-data /var/www/maree-careflow 2>/dev/null || true
chmod 640 /etc/maree-careflow/backend.env
chmod 640 /etc/maree-careflow/jwt_private.pem
chmod 640 /etc/maree-careflow/backup.env 2>/dev/null || true
chmod 644 /etc/maree-careflow/jwt_public.pem 2>/dev/null || true
ok "Permissions set (code root-owned read-only to the service; secrets root:careflow 640)"

# ── Done ──────────────────────────────────────────────────────────────────────
#
# REPORT WHAT WAS OBSERVED, NOT WHAT WAS ATTEMPTED.
#   This banner used to print "installed successfully!" unconditionally, after a
#   readiness probe that could never pass (it polled a route that does not
#   exist) and whose failure was only a warning. An operator was told the install
#   worked while the API was down.
echo ""
if [[ "${BACKEND_READY:-0}" -eq 1 ]]; then
  echo -e "${BOLD}${GREEN}"
  echo "  ╔════════════════════════════════════════════════════════════════════╗"
  echo "  ║        Maree-CareFlow installed successfully!                      ║"
  echo "  ╚════════════════════════════════════════════════════════════════════╝"
  echo -e "${NC}"
  echo -e "  Verified: the API answered /health/ready, so the database is"
  echo -e "  reachable and migrated to this release's schema."
else
  echo -e "${BOLD}${YELLOW}"
  echo "  ╔════════════════════════════════════════════════════════════════════╗"
  echo "  ║   Installed, but the API is NOT answering yet                      ║"
  echo "  ╚════════════════════════════════════════════════════════════════════╝"
  echo -e "${NC}"
  echo -e "  Everything was installed and configured, but the backend did not"
  echo -e "  report ready at http://localhost:8000/health/ready within 90"
  echo -e "  seconds. Do NOT treat this as a finished install."
  echo ""
  echo -e "  Check what it is saying:"
  echo -e "    journalctl -u careflow-api -n 50 --no-pager"
  echo -e "    systemctl status careflow-api"
  echo ""
  echo -e "  Then re-check with:"
  echo -e "    curl -sf http://localhost:8000/health/ready && echo READY"
fi
echo -e "  ${BOLD}Application URL:${NC}  https://${DOMAIN}"
echo -e "  ${BOLD}Admin email:${NC}      ${ADMIN_EMAIL}"
echo ""
# The first-run setup code. POST /auth/setup creates the install's first
# super_admin, and its only guard used to be "no users exist yet" - which stops
# elevation and does nothing about a race: whoever reached an unconfigured
# install first won an account that reads every participant record. The code is
# derived from this install's own SECRET_KEY, so finishing setup requires having
# read the server's configuration. Same principle as the cPanel wizard's
# setup-token.txt, applied to the step that creates the account.
#
# Derived rather than stored: a token file cannot be written on a read-only
# mount, and "nobody can ever finish setup" is a worse failure than the race it
# would close.
_setup_code="$(printf '%s' 'maree-careflow/first-run-setup/v1' \
  | openssl dgst -sha256 -hmac "$SECRET_KEY" -hex \
  | sed 's/.*= //' | cut -c1-8 | tr 'a-z' 'A-Z')"
echo -e "  ${BOLD}Setup code:${NC}       ${_setup_code}"
echo -e "  ${YELLOW}Needed once, on the first-run setup screen.${NC}"
echo -e "  ${CYAN}First-time setup:${NC}"
echo "    Visit https://${DOMAIN}/login to sign in (a fresh install will guide you through first-time setup)."
echo ""
echo -e "  ${CYAN}Service status:${NC}"
SVC_LIST="redis-server ollama careflow-api careflow-worker careflow-beat caddy"
[[ "$DB_MODE" == "local" ]] && SVC_LIST="postgresql $SVC_LIST"
for svc in $SVC_LIST; do
  STATUS=$(systemctl is-active "$svc" 2>/dev/null || echo "unknown")
  if [[ "$STATUS" == "active" ]]; then
    echo -e "    ${GREEN}${svc}: ${STATUS}${NC}"
  else
    echo -e "    ${YELLOW}${svc}: ${STATUS}${NC}"
  fi
done
echo ""
echo -e "  ${CYAN}Useful commands:${NC}"
echo "    journalctl -u careflow-api    -f   # API logs"
echo "    journalctl -u careflow-worker -f   # worker logs"
echo "    systemctl restart careflow-api     # restart API"
echo ""
echo -e "  ${CYAN}Install log:${NC}  ${LOG}"
echo ""

# Exit non-zero when the API never came up, so that automation and CI can tell a
# finished install from an unfinished one. A human reading the banner above
# already knows; a script piping this installer did not.
if [[ "${BACKEND_READY:-0}" -ne 1 ]]; then
  exit 1
fi
