#!/usr/bin/env bash
# =============================================================================
# Maree-CareFlow - Docker Installer (zero-friction)
# =============================================================================
# Usage: bash stack-docker/install.sh
# Requires: Docker 24+ and Docker Compose v2 (plugin)
# =============================================================================
set -euo pipefail

LOG=/tmp/careflow-docker-install.log
exec > >(tee -a "$LOG") 2>&1

# ── Colour helpers ────────────────────────────────────────────────────────────
RED='\033[0;31m'; GREEN='\033[0;32m'; YELLOW='\033[1;33m'
CYAN='\033[0;36m'; BOLD='\033[1m'; NC='\033[0m'

ok()   { echo -e "${GREEN}  ✔  $*${NC}"; }
info() { echo -e "${CYAN}  ℹ  $*${NC}"; }
warn() { echo -e "${YELLOW}  ⚠  $*${NC}"; }
err()  { echo -e "${RED}  ✘  $*${NC}" >&2; exit 1; }
step() { echo -e "\n${BOLD}${CYAN}━━━  $*  ━━━${NC}"; }

SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
REPO_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd)"

# Standalone download: the download page offers this .sh on its own, but
# `docker compose up --build` needs docker-compose.yml plus the backend/ and
# frontend/ build contexts. If they are not alongside this script, self-clone the
# repo (mirroring the bare-metal installer) so the compose build has its sources.
# Override the location with MCF_CLONE_DIR. (v1.3.72)
# WHICH VERSION THIS INSTALLER INSTALLS
#
# It used to `git clone` the default branch and `git pull origin main`, so a
# customer received whatever happened to be on main at the moment they ran it -
# including, on eight consecutive commits in one day, a tree whose Docker stack
# did not come up at all. Nothing recorded WHICH code they got, so a support
# call could not be traced back to a build.
#
# It now checks out a TAG by default. Set MCF_REF to install a different tag,
# branch or commit (`MCF_REF=main` restores the old moving-target behaviour, and
# is the right choice only for development).
MCF_VERSION="${MCF_VERSION:-1.7.15}"
MCF_PACKAGE_URL="${MCF_PACKAGE_URL:-https://maree-careflow.com.au/download/careflow-${MCF_VERSION}.tar.gz}"
# The expected SHA-256 of that package, stamped in at release time by
# scripts/stamp-package-checksum.sh and gated by CI against the published file.
# DO NOT hand-edit it: a wrong value fails every install, and a value that
# merely looks plausible fails none of them for the wrong reason.
MCF_PACKAGE_SHA256="${MCF_PACKAGE_SHA256:-8c5a0c961d555ac7e13af98d6610aeb9d8837da363ec730d3bce17be9caf22d7}"

if [[ ! -f "${REPO_ROOT}/docker-compose.yml" ]]; then
  APP_DIR="${MCF_CLONE_DIR:-/opt/maree-careflow}"

  # WHY THIS DOWNLOADS A PACKAGE INSTEAD OF CLONING THE REPOSITORY
  #
  # This used to run `git clone` against a PRIVATE repository with no token,
  # deploy key or credential helper anywhere in the script - a 404 for every
  # customer who ever ran it, behind an error message that blamed a missing ref
  # rather than the real cause.
  #
  # The two obvious repairs were to issue a deploy token or make the repo
  # public. Both would have put docs/CONFIDENTIAL/ - the financial model,
  # funding strategy and board papers - on every customer's disk, because
  # `git clone` copies everything tracked. The package is built from an
  # allowlist and cannot carry them (AGENTS.md guardrail 56).
  #
  # WHAT WE GIVE UP, AND HOW IT IS REPLACED: a clone verifies its own contents
  # through git's object hashes. A plain download does not, so swapping one for
  # the other WITHOUT verification would make this less safe, not more. The
  # SHA-256 above is checked before a single file is extracted; a mismatch is
  # fatal and is never worked around.
# >>> MCF_FETCH_BEGIN - extracted verbatim by scripts/test-install-from-package.sh
  command -v curl >/dev/null 2>&1 \
    || { echo "  ✘  curl is required to download Maree-CareFlow. Install curl and re-run." >&2; exit 1; }
  command -v sha256sum >/dev/null 2>&1 \
    || { echo "  ✘  sha256sum is required to verify the download. Install coreutils and re-run." >&2; exit 1; }

  _tmp="$(mktemp -d)"
  trap 'rm -rf "${_tmp}"' EXIT
  _tarball="${_tmp}/careflow.tar.gz"

  echo "  ℹ  Downloading Maree-CareFlow ${MCF_VERSION}"
  echo "     ${MCF_PACKAGE_URL}"
  curl -fsSL --retry 3 --retry-delay 2 -o "${_tarball}" "${MCF_PACKAGE_URL}" || {
    echo "  ✘  Could not download the package." >&2
    echo "     Check network/DNS, then confirm the version is listed on the" >&2
    echo "     downloads page at https://maree-careflow.com.au" >&2
    exit 1
  }

  _actual="$(sha256sum "${_tarball}" | cut -d' ' -f1)"
  if [[ "${_actual}" != "${MCF_PACKAGE_SHA256}" ]]; then
    echo "  ✘  DOWNLOAD VERIFICATION FAILED - refusing to install." >&2
    echo "     expected ${MCF_PACKAGE_SHA256}" >&2
    echo "     actual   ${_actual}" >&2
    echo "" >&2
    echo "     The file that arrived is not the one this installer was built to" >&2
    echo "     install. Most often that is a truncated download - try again." >&2
    echo "     If it repeats, do NOT work around it: contact" >&2
    echo "     support@maree-careflow.com.au before installing anything." >&2
    exit 1
  fi
  echo "  ✔  Download verified (sha256 ${_actual})"

  # Extract, then copy into place. Files NOT in the package - .env, secrets/,
  # uploads, anything the operator added - are never touched, because tar only
  # writes what the archive contains. That is what makes an upgrade safe here
  # without git's local-edit detection.
  tar -xzf "${_tarball}" -C "${_tmp}" || { echo "  ✘  Could not extract the package." >&2; exit 1; }
  _root="${_tmp}/careflow-${MCF_VERSION}"
  [[ -d "${_root}" ]] || { echo "  ✘  The package does not contain careflow-${MCF_VERSION}/ as expected." >&2; exit 1; }
  mkdir -p "${APP_DIR}"
  cp -a "${_root}/." "${APP_DIR}/" || { echo "  ✘  Could not write to ${APP_DIR}." >&2; exit 1; }

  echo "  ℹ  Installed Maree-CareFlow ${MCF_VERSION} into ${APP_DIR}"
  REPO_ROOT="${APP_DIR}"
# <<< MCF_FETCH_END

fi

# The path every function reads/writes install config at. Defined GLOBALLY,
# after REPO_ROOT is final: it used to be assigned only inside write_env()
# (fresh-install path) and as a `local` in heal_env(), so on the --upgrade
# path ensure_app_role() expanded an UNBOUND variable and set -u killed the
# installer right after `docker compose up` - the download page's exact
# upgrade command aborted on every run, past the pre-upgrade dump but before
# the health gate (B65/Council L, reproduced).
ENV_FILE="${REPO_ROOT}/.env"

# ── Banner ────────────────────────────────────────────────────────────────────
echo ""
echo -e "${BOLD}${CYAN}"
echo "  ╔══════════════════════════════════════════════════════════════╗"
echo "  ║          Maree-CareFlow - Docker Installer                  ║"
echo "  ║          Zero-friction • recommended for most users         ║"
echo "  ╚══════════════════════════════════════════════════════════════╝"
echo -e "${NC}"

# ── Check Docker ──────────────────────────────────────────────────────────────
check_docker() {
  step "Checking Docker prerequisites"

  if ! command -v docker &>/dev/null; then
    err "Docker is not installed.

  Install Docker Engine (Ubuntu/Debian):
    curl -fsSL https://get.docker.com | bash
    sudo usermod -aG docker \$USER   # then log out and back in

  Install Docker Desktop (Mac/Windows):
    https://www.docker.com/products/docker-desktop/

  Then re-run this script."
  fi

  DOCKER_VER=$(docker version --format '{{.Server.Version}}' 2>/dev/null || echo "unknown")
  ok "Docker Engine: ${DOCKER_VER}"

  # Check Docker is running
  docker info &>/dev/null || err "Docker daemon is not running. Start it with: sudo systemctl start docker"

  # Check Docker Compose v2 (plugin)
  if docker compose version &>/dev/null; then
    COMPOSE_VER=$(docker compose version --short 2>/dev/null || echo "v2")
    ok "Docker Compose: ${COMPOSE_VER}"
  elif command -v docker-compose &>/dev/null; then
    COMPOSE_VER=$(docker-compose version --short 2>/dev/null || echo "v1")
    warn "Found legacy docker-compose v1 (${COMPOSE_VER}). Docker Compose v2 plugin is recommended."
    warn "Install: sudo apt install docker-compose-plugin  OR  pip install docker-compose"
    # Override compose command to use legacy binary
    DOCKER_COMPOSE="docker-compose"
  else
    err "Docker Compose not found.

  Install Docker Compose v2 plugin:
    sudo apt install docker-compose-plugin       # Ubuntu/Debian
    sudo dnf install docker-compose-plugin       # RHEL/Fedora

  Then re-run this script."
  fi

  DOCKER_COMPOSE="${DOCKER_COMPOSE:-docker compose}"
}

# ── Gather config ─────────────────────────────────────────────────────────────
gather_config() {
  step "Configuration - 4 questions"

  echo ""
  echo -e "  ${CYAN}Press Enter to accept the default shown in [brackets]${NC}"
  echo ""

  # Question 1: Domain
  read -r -p "  Domain name [careflow.local]: " DOMAIN_INPUT
  DOMAIN="${DOMAIN_INPUT:-careflow.local}"
  ok "Domain: ${DOMAIN}"
  if [[ "$DOMAIN" == "careflow.local" || "$DOMAIN" == "localhost" || "$DOMAIN" != *.* ]]; then
    warn "'${DOMAIN}' is a LOCAL name. Reach the app at https://localhost/ , or add '127.0.0.1 ${DOMAIN}'"
    warn "to your hosts file to use the name. For a public, browser-trusted certificate you need a real"
    warn "domain pointed at this server - then replace the self-signed cert (see DEPLOYMENT.md)."
  fi

  # Question 2: Admin email
  while true; do
    read -r -p "  Admin email address: " ADMIN_EMAIL
    [[ -n "$ADMIN_EMAIL" ]] && break
    warn "Admin email cannot be empty."
  done
  ok "Admin email: ${ADMIN_EMAIL}"

  # Question 3: SMTP host (optional)
  echo ""
  echo -e "  ${CYAN}SMTP is used to send password reset and notification emails.${NC}"
  echo -e "  ${CYAN}Leave blank to skip (you can configure this later in .env).${NC}"
  read -r -p "  SMTP host [skip]: " SMTP_HOST
  SMTP_HOST="${SMTP_HOST:-}"

  if [[ -n "$SMTP_HOST" ]]; then
    read -r -p "  SMTP port [587]: " SMTP_PORT_INPUT
    SMTP_PORT="${SMTP_PORT_INPUT:-587}"
    read -r -p "  SMTP username [${ADMIN_EMAIL}]: " SMTP_USER_INPUT
    SMTP_USER="${SMTP_USER_INPUT:-${ADMIN_EMAIL}}"
    ok "SMTP: ${SMTP_HOST}:${SMTP_PORT}"
  else
    SMTP_PORT="587"
    SMTP_USER="${ADMIN_EMAIL}"
    warn "SMTP skipped - email sending disabled until configured in .env"
  fi

  # Question 4: Pull Ollama models
  echo ""
  echo -e "  ${CYAN}Ollama AI models enable clinical note summarisation and AI reports.${NC}"
  echo -e "  ${YELLOW}  Models are 4-8 GB each and take time to download.${NC}"
  read -r -p "  Pull Ollama AI models now? [n]: " PULL_MODELS_INPUT
  PULL_MODELS="${PULL_MODELS_INPUT:-n}"

  echo ""
  echo -e "${BOLD}  Configuration summary:${NC}"
  echo "  ┌─────────────────────────────────────────────┐"
  echo "  │  Domain     : ${DOMAIN}"
  echo "  │  Admin email: ${ADMIN_EMAIL}"
  echo "  │  SMTP host  : ${SMTP_HOST:-<disabled>}"
  echo "  │  Pull models: ${PULL_MODELS}"
  echo "  └─────────────────────────────────────────────┘"
  echo ""
  read -r -p "  Press Enter to continue (Ctrl+C to cancel): "
}

# ── Auto-generate all secrets ─────────────────────────────────────────────────
generate_secrets() {
  step "Generating secrets (auto)"

  SECRETS_DIR="${REPO_ROOT}/secrets"
  mkdir -p "$SECRETS_DIR"

  # JWT RS256 keypair
  if [[ ! -f "${SECRETS_DIR}/jwt_private_key.pem" ]]; then
    openssl genrsa -out "${SECRETS_DIR}/jwt_private_key.pem" 2048 2>/dev/null
    openssl rsa \
      -in "${SECRETS_DIR}/jwt_private_key.pem" \
      -pubout \
      -out "${SECRETS_DIR}/jwt_public_key.pem" 2>/dev/null
    chmod 600 "${SECRETS_DIR}/jwt_private_key.pem"
    chmod 644 "${SECRETS_DIR}/jwt_public_key.pem"
    ok "JWT RS256 key pair generated"
  else
    ok "JWT keys already exist - reusing"
  fi

  # PostgreSQL password
  if [[ ! -f "${SECRETS_DIR}/postgres_password.txt" ]]; then
    openssl rand -base64 32 | tr -dc 'a-zA-Z0-9' | head -c32 \
      > "${SECRETS_DIR}/postgres_password.txt"
    ok "PostgreSQL password generated"
  else
    ok "DB password already exists - reusing"
  fi
  DB_PASSWORD=$(cat "${SECRETS_DIR}/postgres_password.txt")

  # MinIO credentials
  if [[ ! -f "${SECRETS_DIR}/minio_access_key.txt" ]]; then
    openssl rand -base64 16 | tr -dc 'a-zA-Z0-9' | head -c20 \
      > "${SECRETS_DIR}/minio_access_key.txt"
    openssl rand -base64 32 | tr -dc 'a-zA-Z0-9' | head -c40 \
      > "${SECRETS_DIR}/minio_secret_key.txt"
    ok "MinIO credentials generated"
  else
    ok "MinIO credentials already exist - reusing"
  fi
  MINIO_ACCESS=$(cat "${SECRETS_DIR}/minio_access_key.txt")
  MINIO_SECRET=$(cat "${SECRETS_DIR}/minio_secret_key.txt")

  # TOTP encryption key
  if [[ ! -f "${SECRETS_DIR}/totp_encryption_key.txt" ]]; then
    openssl rand -base64 32 > "${SECRETS_DIR}/totp_encryption_key.txt"
    ok "TOTP encryption key generated"
  else
    ok "TOTP key already exists - reusing"
  fi
  TOTP_KEY=$(cat "${SECRETS_DIR}/totp_encryption_key.txt")

  # OAuth encryption key
  if [[ ! -f "${SECRETS_DIR}/oauth_encryption_key.txt" ]]; then
    openssl rand -base64 32 > "${SECRETS_DIR}/oauth_encryption_key.txt"
    ok "OAuth encryption key generated"
  else
    ok "OAuth key already exists - reusing"
  fi
  OAUTH_KEY=$(cat "${SECRETS_DIR}/oauth_encryption_key.txt")

  # PHI field-encryption key (Fernet; compose passes it via .env)
  #
  # DO NOT MINT A NEW PHI KEY OVER A DATABASE THAT ALREADY EXISTS.
  #
  # This is the commonest way a practice ends up locked out of its own records,
  # and it is entirely preventable here. The key file is per-CHECKOUT; the
  # database is a docker VOLUME that outlives it. So a re-install in a fresh
  # directory, a deleted checkout, or a restored volume leaves the volume holding
  # data encrypted under the old key while this block cheerfully generates a new
  # one - and the backend then refuses to start with "the PHI encryption key does
  # not match this database" (backend/app/core/phi_keycheck.py). The installer's
  # own comment already says as much about the database role: "A fresh install
  # over a PRE-EXISTING postgres volume (re-install in the same directory,
  # deleted checkout, restored volume) therefore never creates careflow_app."
  #
  # The signal is the same one this script already trusts for MinIO further down -
  # the VOLUME. If a postgres volume is on this host and the key that decrypts it
  # is not in this checkout, stop and say where to find the key, rather than
  # creating the condition and leaving the operator to recover from it.
  #
  # WHY STOP RATHER THAN GUESS: nothing available here can tell "this volume holds
  # a previous install's clinical records" from "this volume is an empty one a
  # failed run left behind", and the two need opposite actions. An operator who
  # genuinely wants a clean start says so with MCF_NEW_PHI_KEY=1 - one word, from
  # somebody who knows what is in the volume.
  if [[ ! -f "${SECRETS_DIR}/phi_encryption_key.txt" ]]; then
    local pg_volume=""
    pg_volume="$(docker volume ls -q 2>/dev/null | grep -E '(^|_)postgres_data$' | head -1 || true)"
    if [[ -n "$pg_volume" && "${MCF_NEW_PHI_KEY:-0}" != "1" ]]; then
      err "This host already has a PostgreSQL volume (${pg_volume}), but there is no
     PHI encryption key in ${SECRETS_DIR}.

     That combination means one of two things, and they need opposite actions:

       (a) The volume holds a PREVIOUS INSTALL'S RECORDS. Generating a new key now
           would leave every Medicare, NDIS and DVA number and every clinical note
           in it unreadable, and the backend would refuse to start. Put the
           original key back instead - it is in the backup taken by
           deploy/scripts/careflow-backup.sh, which captures key material
           alongside the database for exactly this reason:
             mkdir -p ${SECRETS_DIR}
             # write the ORIGINAL PHI_ENCRYPTION_KEY value into:
             ${SECRETS_DIR}/phi_encryption_key.txt
           Then run this installer again.

       (b) The volume is EMPTY - left behind by a failed run, or you intend to
           start over and discard whatever is in it. Then say so explicitly:
             MCF_NEW_PHI_KEY=1 bash install.sh

     To see what is actually in the volume before you decide:
       docker run --rm -v ${pg_volume}:/v alpine sh -c 'ls /v | head'"
    fi
    openssl rand -base64 32 > "${SECRETS_DIR}/phi_encryption_key.txt"
    if [[ -n "$pg_volume" ]]; then
      warn "A new PHI encryption key was generated over the existing volume
     ${pg_volume}, because MCF_NEW_PHI_KEY=1 was set. Anything that volume already
     held is encrypted under the OLD key and will not be readable."
    else
      ok "PHI encryption key generated"
    fi
  else
    ok "PHI key already exists - reusing"
  fi
  PHI_KEY=$(cat "${SECRETS_DIR}/phi_encryption_key.txt")

  # Application HMAC secret (SECRET_KEY) - keys the tamper-EVIDENT claim-integrity
  # ledger, and unsubscribe / calendar-ICS / intake token signing + OAuth-token
  # encryption. MUST be a per-install secret (never the shipped CHANGE_ME
  # placeholder) or the ledger's tamper-evidence is void. Reused on --upgrade.
  if [[ ! -f "${SECRETS_DIR}/app_secret_key.txt" ]]; then
    openssl rand -hex 32 > "${SECRETS_DIR}/app_secret_key.txt"
    ok "Application SECRET_KEY generated"
  else
    ok "Application SECRET_KEY already exists - reusing"
  fi
  APP_SECRET=$(cat "${SECRETS_DIR}/app_secret_key.txt")

  # Licence HMAC secret - signs/verifies Maree-CareFlow licence keys. Reused on
  # --upgrade so existing licence keys stay valid.
  if [[ ! -f "${SECRETS_DIR}/licence_secret.txt" ]]; then
    openssl rand -hex 32 > "${SECRETS_DIR}/licence_secret.txt"
    ok "Licence secret generated"
  else
    ok "Licence secret already exists - reusing"
  fi
  LICENCE_SECRET_VAL=$(cat "${SECRETS_DIR}/licence_secret.txt")

  # SMTP password secret file - docker-compose declares this secret, so the
  # file must exist even when SMTP is skipped (fill it in later to enable email)
  if [[ ! -f "${SECRETS_DIR}/smtp_password.txt" ]]; then
    touch "${SECRETS_DIR}/smtp_password.txt"
    chmod 600 "${SECRETS_DIR}/smtp_password.txt"
    ok "SMTP password placeholder created (set it later to enable email)"
  fi

  # Backup encryption passphrase - the nightly `backup` service REFUSES to write
  # an unencrypted archive, so without this file it cannot run at all.
  #
  # WHY THIS IS HERE AT ALL: until now the Docker stack scheduled NO nightly
  # backup, while the Data Processing and Privacy Agreement warranted that
  # "every installer schedules it nightly". Bare-metal enables a systemd timer
  # and the cPanel terminal installer writes a crontab line; Docker - the
  # recommended, default option - had only a pre-upgrade snapshot. Reused on
  # --upgrade, or every existing archive becomes undecryptable.
  if [[ ! -f "${SECRETS_DIR}/backup_passphrase.txt" ]]; then
    openssl rand -hex 32 > "${SECRETS_DIR}/backup_passphrase.txt"
    chmod 600 "${SECRETS_DIR}/backup_passphrase.txt"
    ok "Backup encryption passphrase generated"
  else
    ok "Backup encryption passphrase already exists - reusing"
  fi

  # nginx TLS certificate - nginx.conf hard-requires cert.pem + key.pem to start.
  # Without it the nginx container fails at startup and, with restart:
  # unless-stopped, crash-loops - so the app is UNREACHABLE (only nginx publishes
  # 80/443). Generate a self-signed certificate so the front door works out of the
  # box; replace it with a real (e.g. Let's Encrypt) certificate for production.
  local ssl_dir="${REPO_ROOT}/nginx/ssl"
  mkdir -p "$ssl_dir"
  if [[ ! -f "${ssl_dir}/cert.pem" || ! -f "${ssl_dir}/key.pem" ]]; then
    openssl req -x509 -newkey rsa:2048 -nodes -days 825 \
      -keyout "${ssl_dir}/key.pem" -out "${ssl_dir}/cert.pem" \
      -subj "/CN=${DOMAIN}" \
      -addext "subjectAltName=DNS:${DOMAIN},DNS:localhost,IP:127.0.0.1" 2>/dev/null \
    || openssl req -x509 -newkey rsa:2048 -nodes -days 825 \
         -keyout "${ssl_dir}/key.pem" -out "${ssl_dir}/cert.pem" \
         -subj "/CN=${DOMAIN}" 2>/dev/null
    chmod 600 "${ssl_dir}/key.pem"; chmod 644 "${ssl_dir}/cert.pem"
    ok "Self-signed TLS certificate generated (replace with a real cert for production)"
  else
    ok "TLS certificate already exists - reusing"
  fi

  # Secure the secrets directory
  chmod 700 "${SECRETS_DIR}"
  info "Secrets written to: ${SECRETS_DIR}/"
  warn "Keep the secrets/ directory private - never commit it to version control."
}

# ── Write .env ────────────────────────────────────────────────────────────────
write_env() {
  step "Writing .env"

  ENV_FILE="${REPO_ROOT}/.env"
  ENV_EXAMPLE="${REPO_ROOT}/.env.example"

  if [[ -f "$ENV_EXAMPLE" ]]; then
    cp "$ENV_EXAMPLE" "$ENV_FILE"
    info "Copied .env.example to .env"
  else
    info ".env.example not found - writing .env from scratch"
    touch "$ENV_FILE"
  fi

  # Helper: set or replace a variable in .env
  set_env() {
    local key="$1" val="$2"
    if grep -q "^${key}=" "$ENV_FILE" 2>/dev/null; then
      # Replace existing line (use | as delimiter to avoid issues with slashes)
      sed -i "s|^${key}=.*|${key}=${val}|" "$ENV_FILE"
    else
      echo "${key}=${val}" >> "$ENV_FILE"
    fi
  }

  set_env "CAREFLOW_DOMAIN"   "${DOMAIN}"
  set_env "ALLOWED_ORIGINS"   "https://${DOMAIN},http://${DOMAIN}"
  set_env "ADMIN_EMAIL"       "${ADMIN_EMAIL}"
  set_env "SMTP_HOST"         "${SMTP_HOST}"
  set_env "SMTP_PORT"         "${SMTP_PORT}"
  set_env "SMTP_USER"         "${SMTP_USER}"
  set_env "ENVIRONMENT"       "production"

  set_env "POSTGRES_PASSWORD" "${DB_PASSWORD}"
  # WHY set the URLs explicitly: the Postgres role is created with the RANDOM
  # generated password (via the postgres_password secret file), but .env.example
  # ships DATABASE_URL/DATABASE_SYNC_URL with the placeholder "CHANGE_ME". Without
  # rewriting them the backend authenticates with CHANGE_ME against a role that
  # has the random password -> auth failure, /health 503, non-functional install.
  # careflow_app, NOT careflow: the bootstrap POSTGRES_USER is a SUPERUSER,
  # which PostgreSQL exempts from every Row-Level Security policy - the app
  # must connect as the non-superuser role initdb creates so the tenant
  # walls (migration 0103) actually bind (F4-3a / GAP-015).
  set_env "DATABASE_URL"      "postgresql+asyncpg://careflow_app:${DB_PASSWORD}@postgres:5432/careflow"
  # The role the RUNNING APPLICATION connects as (B70). careflow_app above owns
  # the schema because it runs the migrations, and an owner can DROP POLICY on
  # its own tables - i.e. delete the tenant wall silently. careflow_runtime owns
  # nothing, so it cannot. Its password is DERIVED from the same secret (see
  # initdb/02-app-role.sh) so there is still only one secret to manage, while
  # the two credentials remain distinct.
  DB_RUNTIME_PASSWORD="$(printf 'careflow-runtime:%s' "${DB_PASSWORD}" | sha256sum | cut -d' ' -f1)"
  set_env "RUNTIME_DATABASE_URL" "postgresql+asyncpg://careflow_runtime:${DB_RUNTIME_PASSWORD}@postgres:5432/careflow"
  set_env "CAREFLOW_RUNTIME_DB_ROLE" "careflow_runtime"
  DB_SYSTEM_PASSWORD="$(printf 'careflow-system:%s' "${DB_PASSWORD}" | sha256sum | cut -d' ' -f1)"
  set_env "SYSTEM_DATABASE_URL" "postgresql+asyncpg://careflow_system:${DB_SYSTEM_PASSWORD}@postgres:5432/careflow"
  set_env "CAREFLOW_SYSTEM_DB_ROLE" "careflow_system"
  set_env "DATABASE_SYNC_URL" "postgresql+psycopg2://careflow_app:${DB_PASSWORD}@postgres:5432/careflow"
  # App + licence HMAC secrets (never leave the shipped CHANGE_ME placeholders).
  set_env "SECRET_KEY"        "${APP_SECRET}"
  set_env "LICENCE_SECRET"    "${LICENCE_SECRET_VAL}"
  set_env "MINIO_ROOT_USER"   "${MINIO_ACCESS}"
  set_env "MINIO_ROOT_PASSWORD" "${MINIO_SECRET}"

  heal_object_store_env
  set_env "TOTP_ENCRYPTION_KEY" "${TOTP_KEY}"
  set_env "OAUTH_ENCRYPTION_KEY" "${OAUTH_KEY}"
  set_env "PHI_ENCRYPTION_KEY"   "${PHI_KEY}"
  chmod 600 "$ENV_FILE"
  ok ".env written: ${ENV_FILE}"
}

# OBJECT STORE (v1.7.9). The minio service is an opt-in compose profile and
# documents live on the `uploads` volume by default. An install made BEFORE
# this release ran MinIO and holds its clinical documents in the minio_data
# volume, so on that host the store must keep running or every existing
# document is unreachable. The signal is the volume itself; the image must
# already be on the host (it can no longer be pulled).
#
# RUNS ON --upgrade TOO. The first cut of this lived inside write_env, which a
# fresh install runs and an upgrade never does - so every v1.7.8 Docker practice
# that upgraded found the store off, MinIO not started and every existing
# document "missing", while docs/UPGRADE_GUIDE.md said the upgrade handled it
# (release-cut council C, v1.7.9). An operator's own explicit value is never
# overwritten: the detection runs only when .env has no OBJECT_STORE_ENABLED.
heal_object_store_env() {
  if [[ -f "$ENV_FILE" ]] && grep -q '^OBJECT_STORE_ENABLED=' "$ENV_FILE" 2>/dev/null; then
    return 0
  fi
  local minio_volume minio_image
  minio_volume="$(docker volume ls -q 2>/dev/null | grep -E '(^|_)minio_data$' | head -1 || true)"
  minio_image="$(docker image ls -q 'minio/minio' 2>/dev/null | head -1 || true)"
  if [[ -n "$minio_volume" && -n "$minio_image" ]]; then
    set_env "COMPOSE_PROFILES" "objectstore"
    set_env "OBJECT_STORE_ENABLED" "true"
    warn "An existing MinIO volume (${minio_volume}) holds this install's documents,
     so the object store stays ON (COMPOSE_PROFILES=objectstore, OBJECT_STORE_ENABLED=true).
     MinIO's upstream is no longer maintained (no images, no security fixes since
     February 2026), and nothing in this product needs it any more - so move the
     documents to this server's own disk WHILE THE STORE IS STILL RUNNING:
       docker compose exec backend python scripts/migrate_object_store_to_local.py
       docker compose exec backend python scripts/migrate_object_store_to_local.py --apply
     It copies, hash-verifies every file and NEVER deletes the source. Once the
     application is serving them from disk, set OBJECT_STORE_ENABLED=false and drop
     objectstore from COMPOSE_PROFILES. See docs/UPGRADE_GUIDE.md."
  elif [[ -n "$minio_volume" ]]; then
    set_env "OBJECT_STORE_ENABLED" "false"
    warn "A MinIO volume (${minio_volume}) exists but no minio/minio image is on this host,
     so it cannot be started and the documents in it are NOT reachable. New documents go to
     local disk. To reach the old ones: set MINIO_IMAGE to an image you can pull, then set
     COMPOSE_PROFILES=objectstore and OBJECT_STORE_ENABLED=true in .env, re-run, then
     copy them off immediately with
       docker compose exec backend python scripts/migrate_object_store_to_local.py --apply"
  else
    set_env "OBJECT_STORE_ENABLED" "false"
  fi
}

# ── Heal a pre-existing .env (upgrade path) ───────────────────────────────────
# An install made with an OLDER installer may still carry the shipped CHANGE_ME
# placeholders for DATABASE_URL/SYNC_URL/SECRET_KEY/LICENCE_SECRET. Repair ONLY
# empty or CHANGE_ME values so a real operator-set value is never overwritten.
heal_env() {
  [[ -f "$ENV_FILE" ]] || return 0
  local pg_user pg_db
  pg_user="$(sed -n 's/^POSTGRES_USER=//p' "$ENV_FILE" | head -n1)"; pg_user="${pg_user:-careflow}"
  pg_db="$(sed -n 's/^POSTGRES_DB=//p' "$ENV_FILE" | head -n1)"; pg_db="${pg_db:-careflow}"

  _heal() {
    local key="$1" val="$2" cur
    cur="$(sed -n "s/^${key}=//p" "$ENV_FILE" | head -n1)"
    if [[ -z "$cur" || "$cur" == *CHANGE_ME* ]]; then
      if grep -q "^${key}=" "$ENV_FILE" 2>/dev/null; then
        sed -i "s|^${key}=.*|${key}=${val}|" "$ENV_FILE"
      else
        echo "${key}=${val}" >> "$ENV_FILE"
      fi
      ok "Healed ${key} (was placeholder/empty)"
    fi
  }
  _heal "DATABASE_URL"      "postgresql+asyncpg://${pg_user}:${DB_PASSWORD}@postgres:5432/${pg_db}"
  _heal "DATABASE_SYNC_URL" "postgresql+psycopg2://${pg_user}:${DB_PASSWORD}@postgres:5432/${pg_db}"
  _heal "SECRET_KEY"        "${APP_SECRET}"
  _heal "LICENCE_SECRET"    "${LICENCE_SECRET_VAL}"
  # The runtime role's URL and name (B72). Without these two lines an
  # --upgrade never provisioned anything: write_env runs only on a FRESH
  # install, so ensure_runtime_role found no password in .env and returned
  # immediately, compose substituted an empty string, and the release notes'
  # claim that Docker upgrades get the separation was false for every existing
  # customer. Three councils found this independently.
  #
  # The password is DERIVED from the same secret (see initdb/02-app-role.sh),
  # so healing it here reproduces exactly what a fresh install would write and
  # cannot desynchronise from the role the database already has.
  DB_RUNTIME_PASSWORD="$(printf 'careflow-runtime:%s' "${DB_PASSWORD}" | sha256sum | cut -d' ' -f1)"
  _heal "RUNTIME_DATABASE_URL"     "postgresql+asyncpg://careflow_runtime:${DB_RUNTIME_PASSWORD}@postgres:5432/${pg_db}"
  _heal "CAREFLOW_RUNTIME_DB_ROLE" "careflow_runtime"
  DB_SYSTEM_PASSWORD="$(printf 'careflow-system:%s' "${DB_PASSWORD}" | sha256sum | cut -d' ' -f1)"
  _heal "SYSTEM_DATABASE_URL"      "postgresql+asyncpg://careflow_system:${DB_SYSTEM_PASSWORD}@postgres:5432/${pg_db}"
  _heal "CAREFLOW_SYSTEM_DB_ROLE"  "careflow_system"
  # A v1.7.8 install kept its documents in MinIO; the upgrade must keep the
  # store on for it (see heal_object_store_env). Only when .env is silent.
  heal_object_store_env
  # Honesty on upgraded installs: an install whose volume predates the
  # careflow_app role connects as the bootstrap SUPERUSER, which PostgreSQL
  # exempts from every Row-Level Security policy - the tenant walls
  # (migration 0103) do not bind for it. Fresh installs get careflow_app
  # from initdb automatically; existing ones need the operator to create it
  # (see docs/UPGRADE_GUIDE.md "Docker and Row-Level Security"). Say so, loudly.
  if grep -q "^DATABASE_URL=postgresql+asyncpg://careflow:" "$ENV_FILE" 2>/dev/null; then
    warn "DATABASE_URL connects as the SUPERUSER 'careflow', so database row-level security does NOT bind on this install. To enable it, FINISH THIS UPGRADE FIRST (so all migrations have run), then follow docs/UPGRADE_GUIDE.md 'Docker and Row-Level Security' - it creates the careflow_app role AND transfers table ownership to it (grants alone are not enough: the next release's migrations would fail with 'must be owner of table'), then switches DATABASE_URL/DATABASE_SYNC_URL from careflow: to careflow_app: and restarts. Application-layer isolation remains in force either way."
  fi
}

# ── Ensure the non-superuser app role exists (B64) ────────────────────────────
# initdb scripts run ONLY on an empty data directory. A fresh install over a
# PRE-EXISTING postgres volume (re-install in the same directory, deleted
# checkout, restored volume) therefore never creates careflow_app, while
# write_env unconditionally points DATABASE_URL at it - the app fails
# password auth for a role that does not exist and nothing healed it. This
# probes for the role with the bootstrap superuser and creates it - including
# OWNERSHIP of any existing tables/sequences, without which the next
# release's migrations (and RLS DDL) fail with "must be owner of table".
apply_runtime_grants() {
  # Re-apply the runtime role's privileges, idempotently, as the OWNER.
  #
  # WHY THIS IS NOT LEFT TO THE MIGRATION (B72): migration 0109 grants them
  # once and is then stamped forever. The backend container runs its startup
  # migrations during `start_services`, i.e. BEFORE ensure_runtime_role above
  # has had a chance to create the role on a re-install over a pre-existing
  # volume - so 0109 sees no role, skips, stamps, and nothing would ever grant
  # again. The install would come up with the application pointed at a role
  # holding CONNECT and nothing else, and re-running the installer could not
  # repair it. This step closes that, and is a no-op on an install with no
  # runtime role.
  #
  # WAIT FOR THE BACKEND TO BE HEALTHY FIRST (F4-3c part 2). The backend
  # migrates DURING start_services, so this function races those migrations -
  # and before part 2 losing the race was benign (0109's grants are
  # re-runnable and the app worked regardless). With a system identity
  # configured it is NOT: "GRANT ... ON ALL TABLES" against a not-yet-
  # migrated schema grants NOTHING, and a GRANT issued WHILE the chain runs
  # can DEADLOCK against migration DDL and kill the script mid-grants -
  # both shapes were caught live by the installer-smoke CI job (first the
  # empty-schema race, then, after a first fix waited only for the `users`
  # table - which commits at migration 0001 while a hundred more revisions
  # follow - the pg_class deadlock). The only barrier that actually means
  # "the migrations are DONE" is the backend's own healthcheck: compose
  # marks it healthy when /health answers, and uvicorn serves only after
  # the startup lifespan (which runs the migrations) has completed.
  local health_tries=0 backend_cid backend_health
  while :; do
    backend_cid="$($DOCKER_COMPOSE ps -q backend 2>/dev/null | head -n1)"
    if [[ -n "$backend_cid" ]]; then
      backend_health="$(docker inspect --format '{{.State.Health.Status}}' "$backend_cid" 2>/dev/null || echo unknown)"
      [[ "$backend_health" == "healthy" ]] && break
    fi
    health_tries=$((health_tries+1))
    if [[ $health_tries -ge 60 ]]; then
      warn "The backend did not report healthy within 300s, so its startup"
      warn "migrations may still be running. Applying what can be applied;"
      warn "if the health check below fails, wait for the backend and re-run"
      warn "the grant command it prints."
      break
    fi
    sleep 5
  done
  $DOCKER_COMPOSE exec -T backend python /app/scripts/grant_runtime_role.py \
    && return 0
  # One whole-script retry: the script is idempotent and its per-statement
  # retries already absorb the transient catalog races (concurrent DDL,
  # post-migration autoanalyze) - but if a race still won every attempt,
  # ten seconds later the autovacuum storm on a fresh install is over.
  warn "Grant pass failed - retrying once in 10s (the script is idempotent)."
  sleep 10
  $DOCKER_COMPOSE exec -T backend python /app/scripts/grant_runtime_role.py \
    && return 0
  warn "Could not apply the runtime role's privileges. The application may not"
  warn "be able to read its own data. Run this as the database owner and re-run"
  warn "the health check (from the install directory, where docker-compose.yml"
  warn "lives - 'docker compose' resolves the project from the working directory,"
  warn "so the cd is required):"
  warn "  cd ${REPO_ROOT} && docker compose exec backend python /app/scripts/grant_runtime_role.py"
  return 0
}

ensure_runtime_role() {
  # Idempotent provisioning of the NON-OWNER runtime role on a database that
  # already exists (an upgrade, or a re-install over a previous volume).
  # initdb/02-app-role.sh only ever runs on a FRESH volume, so without this an
  # upgraded install would carry RUNTIME_DATABASE_URL in its .env naming a role
  # that does not exist - the app would fail to connect at the health gate,
  # which is exactly the class of upgrade breakage B65 and B69 were about.
  local pg_user pg_db runtime_pw tries
  # Read the same way ensure_app_role does - from .env, with the compose
  # defaults as the fallback. (An earlier draft of this function called a
  # `compose_env` helper that does not exist in this script; the Docker E2E
  # job caught it as "command not found" and killed the installer under set -e
  # AFTER the containers were up, which is exactly the half-provisioned
  # failure shape B65 was about.)
  pg_user="$(sed -n 's/^POSTGRES_USER=//p' "$ENV_FILE" | head -n1)"; pg_user="${pg_user:-careflow}"
  pg_db="$(sed -n 's/^POSTGRES_DB=//p' "$ENV_FILE" | head -n1)"; pg_db="${pg_db:-careflow}"
  runtime_pw="$(sed -n 's#^RUNTIME_DATABASE_URL=postgresql+asyncpg://careflow_runtime:\([^@]*\)@.*#\1#p' "$ENV_FILE" | head -n1)"
  [[ -n "$runtime_pw" ]] || return 0
  tries=0
  until $DOCKER_COMPOSE exec -T postgres pg_isready -U "$pg_user" -d "$pg_db" >/dev/null 2>&1; do
    tries=$((tries+1))
    [[ $tries -ge 30 ]] && { warn "PostgreSQL not ready after 60s - skipping the runtime role check."; return 0; }
    sleep 2
  done
  if $DOCKER_COMPOSE exec -T postgres psql -U "$pg_user" -d "$pg_db" \
      -v ON_ERROR_STOP=1 -v rt_pw="$runtime_pw" -v db="$pg_db" <<'SQL'
DO $$
BEGIN
  IF NOT EXISTS (SELECT 1 FROM pg_roles WHERE rolname = 'careflow_runtime') THEN
    CREATE ROLE careflow_runtime LOGIN NOSUPERUSER NOBYPASSRLS NOCREATEDB NOCREATEROLE;
  END IF;
END $$;
ALTER ROLE careflow_runtime PASSWORD :'rt_pw';
-- CONNECT only. TEMP is deliberately NOT granted, and is revoked below.
GRANT CONNECT ON DATABASE :"db" TO careflow_runtime;
-- Close TEMP table shadowing (B73). This psql runs as POSTGRES_USER - the
-- bootstrap SUPERUSER, which OWNS the database - so the revoke takes effect
-- here. Migration 0109's revoke does not: it runs as careflow_app, which owns
-- the tables but not the database, and REVOKE ... ON DATABASE answers a
-- non-owner with a WARNING that every driver discards while still reporting
-- success. Two councils found that independently and it was reproduced in the
-- shipped ownership shape; before this line, EVERY Docker install could run
-- `CREATE TEMP TABLE clients` and silently redirect every later read and write
-- on that pooled connection. PUBLIC must be named: it holds TEMPORARY on every
-- database by default and a member revoke does not remove a PUBLIC grant.
REVOKE TEMPORARY ON DATABASE :"db" FROM PUBLIC;
REVOKE TEMPORARY ON DATABASE :"db" FROM careflow_runtime;
-- CREATE ON DATABASE is the other shadowing door (B74) - revoke it too, from
-- PUBLIC and the runtime role. This runs as POSTGRES_USER (the bootstrap
-- superuser), so it takes effect on the re-install-over-existing-volume path.
REVOKE CREATE ON DATABASE :"db" FROM PUBLIC;
REVOKE CREATE ON DATABASE :"db" FROM careflow_runtime;
-- careflow_app runs the migrations and pg_dump, and legitimately needs TEMP.
DO $$
BEGIN
  IF EXISTS (SELECT 1 FROM pg_roles WHERE rolname = 'careflow_app') THEN
    EXECUTE format('GRANT TEMPORARY ON DATABASE %I TO careflow_app',
                   current_database());
  END IF;
END $$;
SQL
  then
    ok "Runtime role careflow_runtime ready (owns nothing - cannot remove the tenant wall; TEMP revoked, so it cannot shadow a table)"
  else
    warn "Could not provision careflow_runtime. The application will continue to"
    warn "connect as the owner, which is the pre-v1.3.94 behaviour and works, but"
    warn "the database wall stays advisory. Re-run the installer, or create the"
    warn "role by hand - see docs/UPGRADE_GUIDE.md."
    # Do NOT leave a URL naming a role that does not exist: that would break the
    # app at startup. Falling back to the owner keeps the install working.
    sed -i '/^RUNTIME_DATABASE_URL=/d' "$ENV_FILE"
    # The services are ALREADY RUNNING with the deleted URL in their
    # environment (start_services ran before this block) - editing .env alone
    # changes nothing until they restart, so the running app would keep
    # dialing the missing role until someone happened to bounce it (Council
    # JJ F5). Recreate them so the fallback actually takes effect.
    $DOCKER_COMPOSE up -d backend celery_worker celery_beat >/dev/null 2>&1 || \
      warn "Could not restart the app services - run: $DOCKER_COMPOSE up -d"
  fi

  # ── The SYSTEM identity + PHI marker on upgraded volumes (F4-3c part 2) ──
  # Fresh volumes get these in initdb; a volume that predates part 2 gets
  # them here, by the same superuser. Marker 2 is NOT granted to the runtime
  # role - arming belongs exclusively to grant_runtime_role.py, which first
  # proves a real authenticated connection on SYSTEM_DATABASE_URL.
  local system_pw
  system_pw="$(sed -n 's#^SYSTEM_DATABASE_URL=postgresql+asyncpg://careflow_system:\([^@]*\)@.*#\1#p' "$ENV_FILE" | head -n1)"
  if [[ -n "$system_pw" ]]; then
    if $DOCKER_COMPOSE exec -T postgres psql -U "$pg_user" -d "$pg_db" \
        -v ON_ERROR_STOP=1 -v sys_pw="$system_pw" -v db="$pg_db" <<'SQL'
DO $$
BEGIN
  IF NOT EXISTS (SELECT 1 FROM pg_roles WHERE rolname = 'careflow_system') THEN
    CREATE ROLE careflow_system LOGIN NOSUPERUSER NOBYPASSRLS NOCREATEDB NOCREATEROLE;
  END IF;
END $$;
ALTER ROLE careflow_system PASSWORD :'sys_pw';
GRANT CONNECT ON DATABASE :"db" TO careflow_system;
REVOKE TEMPORARY ON DATABASE :"db" FROM careflow_system;
REVOKE CREATE ON DATABASE :"db" FROM careflow_system;
DO $$
BEGIN
  IF to_regrole('careflow_bound_phi') IS NULL THEN
    CREATE ROLE "careflow_bound_phi" NOLOGIN;
  END IF;
END $$;
DO $$
BEGIN
  IF EXISTS (SELECT 1 FROM pg_roles WHERE rolname = 'careflow_app') THEN
    GRANT "careflow_bound_phi" TO careflow_app WITH ADMIN OPTION, INHERIT FALSE, SET FALSE;
  END IF;
END $$;
SQL
    then
      ok "System identity careflow_system + PHI marker ready (the grant script arms the PHI wall after its authenticated probe)"
    else
      # CRITICAL (Council auditor B, finding B-1): deleting SYSTEM_DATABASE_URL
      # is a SAFE fallback ONLY when the PHI wall is NOT already armed. If the
      # runtime role already inherits careflow_bound_phi (this volume was armed
      # by a previous release), dropping the system URL sends every contextless
      # login query onto the bound runtime role, users fail closed, and the
      # WHOLE INSTALL can no longer log in - over a database that is completely
      # intact. The installer cannot un-arm marker 2 here, so it must NOT delete
      # the URL in that state: keep it, fail loudly, and let the operator repair
      # careflow_system (the app keeps retrying the recoverable system identity;
      # /ready reports state 'bound-identity' until it is fixed).
      local phi_armed
      phi_armed="$($DOCKER_COMPOSE exec -T postgres psql -U "$pg_user" -d "$pg_db" \
        -tAc "SELECT pg_has_role('careflow_runtime', 'careflow_bound_phi', 'USAGE')" 2>/dev/null | tr -d '[:space:]')"
      if [[ "$phi_armed" == "t" ]]; then
        err "Could not provision careflow_system, but the PHI wall is ALREADY ARMED"
        err "on this volume (careflow_runtime inherits careflow_bound_phi). Deleting"
        err "SYSTEM_DATABASE_URL now would lock EVERY user out of an intact database."
        err "Leaving SYSTEM_DATABASE_URL in place - the app keeps retrying the system"
        err "identity, and /ready will report 'bound-identity' until it is repaired."
        err "FIX: restore or re-create the careflow_system role and password to match"
        err "SYSTEM_DATABASE_URL in .env, then re-run this installer. See UPGRADE_GUIDE.md."
        # Recreate the services so they pick up the (unchanged) URL and keep
        # retrying the recoverable system identity rather than a stale handle.
        $DOCKER_COMPOSE up -d backend celery_worker celery_beat >/dev/null 2>&1 || \
          warn "Could not restart the app services - run: $DOCKER_COMPOSE up -d"
      else
        warn "Could not provision careflow_system. Logins keep working on the"
        warn "runtime/owner engines and the PHI wall stays part-1 (unarmed)."
        sed -i '/^SYSTEM_DATABASE_URL=/d' "$ENV_FILE"
        # Same as the runtime branch above (Council JJ F5): the services were
        # started WITH the now-deleted URL, so the running app keeps dialing a
        # role that does not exist until recreated. Safe here ONLY because the
        # wall is unarmed (checked above), so the bound-runtime fallback reads
        # rows normally.
        $DOCKER_COMPOSE up -d backend celery_worker celery_beat >/dev/null 2>&1 || \
          warn "Could not restart the app services - run: $DOCKER_COMPOSE up -d"
      fi
    fi
  fi
}

ensure_app_role() {
  grep -q "^DATABASE_URL=postgresql+asyncpg://careflow_app:" "$ENV_FILE" 2>/dev/null || return 0
  local pg_user pg_db app_pw tries
  pg_user="$(sed -n 's/^POSTGRES_USER=//p' "$ENV_FILE" | head -n1)"; pg_user="${pg_user:-careflow}"
  pg_db="$(sed -n 's/^POSTGRES_DB=//p' "$ENV_FILE" | head -n1)"; pg_db="${pg_db:-careflow}"
  app_pw="$(sed -n 's#^DATABASE_URL=postgresql+asyncpg://careflow_app:\([^@]*\)@.*#\1#p' "$ENV_FILE" | head -n1)"
  [[ -n "$app_pw" ]] || { warn "Could not read the careflow_app password from .env - skipping role check."; return 0; }
  tries=0
  until $DOCKER_COMPOSE exec -T postgres pg_isready -U "$pg_user" -d "$pg_db" >/dev/null 2>&1; do
    tries=$((tries+1)); [[ $tries -ge 30 ]] && { warn "PostgreSQL not ready after 60s - skipping careflow_app role check (health gate below will surface any auth failure)."; return 0; }
    sleep 2
  done
  if $DOCKER_COMPOSE exec -T postgres psql -U "$pg_user" -d "$pg_db" -tAc \
      "SELECT 1 FROM pg_roles WHERE rolname='careflow_app'" 2>/dev/null | grep -q 1; then
    return 0
  fi
  info "Creating the careflow_app role on a pre-existing database volume..."
  # Password AND database name via psql variables, not string interpolation:
  # a metacharacter in the generated password cannot break or inject into the
  # SQL, and a custom POSTGRES_DB no longer hits a hardcoded 'careflow' GRANT
  # (B65/Council J - the mismatch aborted the installer half-provisioned).
  # `if !` rather than a bare call: under set -e a failing psql would kill
  # the whole installer HERE, making the guidance below dead code - the
  # exact failure it was written for (B65/Councils J+L, reproduced).
  if $DOCKER_COMPOSE exec -T postgres psql -U "$pg_user" -d "$pg_db" \
      -v ON_ERROR_STOP=1 -v app_pw="$app_pw" -v db="$pg_db" <<'SQL'
CREATE ROLE careflow_app LOGIN NOSUPERUSER NOBYPASSRLS NOCREATEDB NOCREATEROLE PASSWORD :'app_pw';
GRANT CONNECT, TEMP ON DATABASE :"db" TO careflow_app;
GRANT USAGE, CREATE ON SCHEMA public TO careflow_app;
GRANT ALL ON ALL TABLES IN SCHEMA public TO careflow_app;
GRANT USAGE, SELECT ON ALL SEQUENCES IN SCHEMA public TO careflow_app;
DO $$
DECLARE r record;
BEGIN
  FOR r IN SELECT tablename FROM pg_tables WHERE schemaname = 'public' LOOP
    EXECUTE format('ALTER TABLE public.%I OWNER TO careflow_app', r.tablename);
  END LOOP;
  FOR r IN SELECT sequencename FROM pg_sequences WHERE schemaname = 'public' LOOP
    EXECUTE format('ALTER SEQUENCE public.%I OWNER TO careflow_app', r.sequencename);
  END LOOP;
END $$;
SQL
  then
    ok "careflow_app role created and given ownership of the existing schema"
  else
    warn "Could not create the careflow_app role automatically - see docs/UPGRADE_GUIDE.md 'Docker and Row-Level Security' for the manual steps. Continuing so the health gate can report the install's real state."
  fi
}

# ── docker compose up ─────────────────────────────────────────────────────────
start_services() {
  step "Starting Maree-CareFlow services"

  cd "$REPO_ROOT"

  info "Pulling latest images..."
  $DOCKER_COMPOSE pull --quiet 2>/dev/null || info "Some images will be built locally"

  info "Starting containers..."
  $DOCKER_COMPOSE up -d --build

  # nginx resolves the backend's address ONCE, at its own startup, and caches
  # it forever. When `up --build` RECREATES the backend (every upgrade, and
  # any re-run over an existing stack) the new container gets a new address
  # while nginx keeps proxying the old one - the app boots healthy but every
  # front-door request answers "connection refused" until nginx restarts.
  # Caught by the docker-upgrade CI leg on its FIRST run (B65 follow-up):
  # the readiness gate below timed out against a perfectly healthy backend.
  # Restarting nginx after `up` is a ~2s no-op on a fresh stack and makes it
  # re-resolve on a recreated one; the readiness gate re-verifies either way.
  $DOCKER_COMPOSE restart nginx >/dev/null 2>&1 || true

  ok "Containers started"
  info "Object storage (MinIO) is not started by default - documents live on this
     server's own disk (STORAGE_PROVIDER=local). It is an opt-in compose profile;
     see the minio service in docker-compose.yml."
}

# ── Health check ──────────────────────────────────────────────────────────────
# ── Container health gate ─────────────────────────────────────────────────────
#
# WHY THE READINESS POLL ABOVE IS NOT ENOUGH:
#   /health/ready proves nginx and the backend are up. It says NOTHING about the
#   Celery worker or beat - and both of their healthchecks were meaningless until
#   v1.3.85. The worker's probe interpolated $HOSTNAME on the HOST at compose
#   parse time, emitting `--destination careflow-worker@` (confirmed with
#   `docker compose config`), so it could never match a node and the worker was
#   PERMANENTLY unhealthy. Beat's probe ran `inspect ping` with no --destination,
#   so the WORKER answered for it and a crashlooping beat stayed healthy forever.
#
#   Both are fixed. But an installer that never LOOKS at container health would
#   still have printed "installation complete" over either of them - which is how
#   they survived. So it looks now.
#
#   Reported per container rather than as one verdict, because "something is
#   unhealthy" is not an actionable message and this runs on a customer's server.
# WHY THIS WAITS INSTEAD OF JUDGING ONCE
#   The first version read `docker compose ps` exactly once, the moment
#   /health/ready answered, and treated `starting` as a failure. Docker reports
#   `starting` until a container's first probe has RUN - and the Celery probes
#   run on a 30-second interval, while the backend answers ready at around t+43s
#   on a CI runner. So a completely healthy stack was failed by this gate for
#   being asked too early: measured in CI, the Docker installer aborted on the
#   very commit that added this function.
#
#   `starting` means NOT YET KNOWN. Treating not-yet-known as failed is the same
#   mistake as treating "cannot check" as "failed the check" - the class of
#   defect this codebase keeps having to remove. So: wait for it to resolve,
#   with a bound, and only then judge.
_container_health_probe() {
  local unhealthy=0 pending=0 name state health
  HEALTH_REPORT=""
  while IFS=$'\t' read -r name state health; do
    [ -z "$name" ] && continue
    # A container with no healthcheck reports an empty string, not "healthy".
    # Treat that as "running is all we can assert" rather than silently passing
    # it off as verified.
    case "$health" in
      healthy)   HEALTH_REPORT="$HEALTH_REPORT
  ok $name: healthy" ;;
      "")        if [ "$state" = "running" ]; then
                   HEALTH_REPORT="$HEALTH_REPORT
  info $name: running (no healthcheck defined)"
                 else
                   HEALTH_REPORT="$HEALTH_REPORT
  warn $name: $state"; unhealthy=1
                 fi ;;
      starting)  HEALTH_REPORT="$HEALTH_REPORT
  warn $name: still starting"; pending=1 ;;
      *)         HEALTH_REPORT="$HEALTH_REPORT
  warn $name: $health ($state)"; unhealthy=1 ;;
    esac
  done < <($DOCKER_COMPOSE ps --format '{{.Service}}\t{{.State}}\t{{.Health}}' 2>/dev/null)

  # 0 = every container settled and good, 1 = something is genuinely wrong,
  # 2 = at least one is still starting and nothing is wrong yet.
  if [ "$unhealthy" -ne 0 ]; then return 1; fi
  if [ "$pending" -ne 0 ]; then return 2; fi
  return 0
}

container_health_check() {
  step "Checking every container is actually healthy"
  cd "$REPO_ROOT" || return 1

  # 180s: three times the 60s start_period the Celery containers declare, so a
  # slow image start on a small VPS still settles inside it.
  local deadline=$((SECONDS + 180)) rc=2 unhealthy=0
  while [ "$SECONDS" -lt "$deadline" ]; do
    _container_health_probe; rc=$?
    [ "$rc" -eq 2 ] || break
    printf "  ${CYAN}  waiting for containers to finish starting...${NC}\r"
    sleep 5
  done
  echo ""

  # shellcheck disable=SC2001
  echo "$HEALTH_REPORT" | while IFS= read -r line; do
    case "$line" in
      "  ok "*)   ok "  ${line#  ok }" ;;
      "  info "*) info "  ${line#  info }" ;;
      "  warn "*) warn "  ${line#  warn }" ;;
    esac
  done

  # A container still `starting` when the clock runs out has genuinely failed to
  # come up - by then its probe has had several chances to run.
  [ "$rc" -eq 0 ] || unhealthy=1

  if [ "$unhealthy" -ne 0 ]; then
    warn "One or more containers are not healthy. The web application answered,
     but a background service did not - scheduled work (appointment reminders,
     credential-expiry alerts, the retention sweep) may not run."
    warn "Container status:"; $DOCKER_COMPOSE ps 2>/dev/null || true
    warn "Last 30 log lines (celery_worker + celery_beat):"
    $DOCKER_COMPOSE logs --tail=30 celery_worker celery_beat 2>/dev/null || true
    return 1
  fi

  ok "All containers healthy"
  return 0
}

health_check() {
  step "Waiting for the application to be reachable (through the real TLS front door)"

  # Poll the REAL readiness endpoint THROUGH nginx over HTTPS. -k accepts the
  # self-signed cert; -L follows the 80->443 redirect; we require the backend to
  # actually report {"ready":true}. A bare redirect or the SPA index page is NOT
  # accepted - that is exactly how the old check passed while the app was down.
  local url="https://localhost/health/ready"
  local max=40 wait=5 i=0 body=""

  info "Polling ${url} (up to $((max * wait))s)..."
  while (( i < max )); do
    body="$(curl -skfL -m 5 "$url" 2>/dev/null || true)"
    if [[ "$body" == *'"ready":true'* ]]; then
      echo ""
      ok "Application is READY and reachable over HTTPS (${url})"
      container_health_check || return 1
      return 0
    fi
    i=$((i + 1))
    printf "  ${CYAN}  [%2d/%d] waiting...${NC}\r" "$i" "$max"
    sleep "$wait"
  done

  echo ""
  warn "The application did not report ready after $((max * wait)) seconds."
  cd "$REPO_ROOT"
  warn "Container status:"; $DOCKER_COMPOSE ps 2>/dev/null || true
  echo ""
  warn "Last 30 log lines (nginx + backend):"
  $DOCKER_COMPOSE logs --tail=30 nginx backend 2>/dev/null || $DOCKER_COMPOSE logs --tail=30 2>/dev/null || true
  return 1
}

# ── Pull Ollama models ────────────────────────────────────────────────────────
pull_models() {
  if [[ "${PULL_MODELS,,}" == "y" || "${PULL_MODELS,,}" == "yes" ]]; then
    step "Pulling Ollama AI models (this may take several minutes)"
    cd "$REPO_ROOT"

    info "Pulling qwen2.5:7b (language model, ~4.7 GB)..."
    $DOCKER_COMPOSE exec -T ollama ollama pull qwen2.5:7b || \
      warn "Could not pull qwen2.5:7b - retry with: docker compose exec ollama ollama pull qwen2.5:7b"

    info "Pulling nomic-embed-text (embeddings, ~274 MB)..."
    $DOCKER_COMPOSE exec -T ollama ollama pull nomic-embed-text || \
      warn "Could not pull nomic-embed-text - retry manually"

    ok "AI models ready"
  else
    info "Skipping AI model download."
    info "Pull later with: docker compose exec ollama ollama pull qwen2.5:7b"
  fi
}

# ── Failure summary (honest: never claim "running" when it is not) ────────────
print_failure_summary() {
  echo ""
  echo -e "${BOLD}${RED}"
  echo "  ╔══════════════════════════════════════════════════════════════╗"
  echo "  ║   Install did NOT reach a ready state                        ║"
  echo "  ╚══════════════════════════════════════════════════════════════╝"
  echo -e "${NC}"
  echo -e "  The containers were started, but ${BOLD}https://${DOMAIN}/health/ready${NC} did not report ready."
  echo -e "  Nothing is confirmed working yet. Diagnose with:"
  echo "    docker compose ps"
  echo "    docker compose logs -f nginx backend"
  echo -e "  Install log: ${LOG}"
  echo ""
}

# ── Final summary ─────────────────────────────────────────────────────────────
print_summary() {
  # nginx terminates TLS on 443 for every domain (including local), so the app is
  # always reached over HTTPS. A self-signed cert shows a browser warning until
  # you install a real certificate.
  local PROTO="https"

  echo ""
  echo -e "${BOLD}${GREEN}"
  echo "  ╔══════════════════════════════════════════════════════════════╗"
  echo "  ║       Maree-CareFlow is running!                            ║"
  echo "  ╚══════════════════════════════════════════════════════════════╝"
  echo -e "${NC}"
  echo -e "  ${BOLD}Application URL:${NC}  ${PROTO}://${DOMAIN}/"
  echo -e "  ${BOLD}API Docs:${NC}         ${PROTO}://${DOMAIN}/api/v1/docs"
  echo -e "  ${BOLD}Admin setup:${NC}      ${PROTO}://${DOMAIN}/login"
  if [[ "$DOMAIN" == "careflow.local" || "$DOMAIN" == "localhost" || "$DOMAIN" != *.* ]]; then
    echo -e "  ${YELLOW}(local name - if it does not resolve, use https://localhost/ )${NC}"
  fi
  echo -e "  ${YELLOW}The TLS certificate is self-signed; your browser will warn until you install a real one.${NC}"
  echo ""
  # The first-run setup code. `POST /auth/setup` creates the install's first
  # super_admin, and its only guard used to be "no users exist yet" - a correct
  # defence against elevation and none at all against a race. Whoever reached the
  # open install first won an account that reads every participant record. The
  # code is derived from this install's own SECRET_KEY, so completing setup now
  # requires having read the server's configuration.
  #
  # Derived, not stored: a token written to a file cannot be created on a
  # read-only mount, and "nobody can ever finish setup" is a worse failure than
  # the race it would close.
  # Read the key from the file rather than a variable: ${REPO_ROOT} is provably in
  # scope here (the lines below print it), whereas APP_SECRET is assigned inside an
  # earlier function and relying on it leaking out is the kind of assumption that
  # works until somebody adds a `local`.
  setup_code="$(printf '%s' 'maree-careflow/first-run-setup/v1' \
    | openssl dgst -sha256 -hmac "$(cat "${REPO_ROOT}/secrets/app_secret_key.txt")" -hex \
    | sed 's/.*= //' | cut -c1-8 | tr 'a-z' 'A-Z')"
  echo -e "  ${BOLD}Setup code:${NC}       ${setup_code}"
  echo -e "  ${YELLOW}You need this once, on the first-run setup screen. To see it again:${NC}"
  echo "                     printf '%s' 'maree-careflow/first-run-setup/v1' | openssl dgst -sha256 -hmac \"\$SECRET_KEY\" -hex"
  echo -e "  ${BOLD}Admin email:${NC}      ${ADMIN_EMAIL}"
  echo -e "  ${BOLD}Config file:${NC}      ${REPO_ROOT}/.env"
  echo -e "  ${BOLD}Secrets dir:${NC}      ${REPO_ROOT}/secrets/"
  echo -e "  ${BOLD}Install log:${NC}      ${LOG}"
  echo ""
  # Report the backup the way bare-metal and cPanel do, and do NOT claim it is
  # scheduled unless the container that schedules it is actually running.
  if $DOCKER_COMPOSE ps --status running --services 2>/dev/null | grep -qx backup; then
    echo -e "  ${BOLD}Nightly backup:${NC}   scheduled 02:15 (service 'backup'), encrypted, 30-day prune"
    echo -e "  ${BOLD}Backup archives:${NC}  docker volume 'backups' - retrieve with:"
    echo "                     docker compose cp backup:/backups ./backups"
    echo -e "  ${YELLOW}KEEP secrets/backup_passphrase.txt SOMEWHERE ELSE. Without it the${NC}"
    echo -e "  ${YELLOW}archives cannot be decrypted, and storing it beside them defeats${NC}"
    echo -e "  ${YELLOW}the encryption entirely.${NC}"
  else
    echo -e "  ${RED}Nightly backup:   NOT RUNNING. The 'backup' service did not start,${NC}"
    echo -e "  ${RED}                  so this install has NO scheduled backup. Check${NC}"
    echo -e "  ${RED}                  'docker compose logs backup' before going live.${NC}"
  fi
  echo ""
  echo -e "  ${CYAN}Useful commands:${NC}"
  echo "    docker compose ps                 # container status"
  echo "    docker compose logs -f backup     # nightly backup scheduler"
  echo "    docker compose run --rm backup bash /opt/careflow/scripts/careflow-backup.sh"
  echo "                                      # run a backup right now"
  echo "    docker compose logs -f            # follow all logs"
  echo "    docker compose logs -f careflow-api  # API logs only"
  echo "    docker compose restart careflow-api  # restart API"
  echo "    docker compose down               # stop everything"
  echo "    docker compose up -d              # start again"
  echo ""
  echo -e "  ${CYAN}Support:${NC}"
  echo "    https://github.com/supportcall/Maree-CareFlow-2026/issues"
  echo ""
}

# ── Fail-closed pre-upgrade DB snapshot (upgrade path) ────────────────────────
# A failed migration with no backup is the top data-loss risk. Before an in-place
# upgrade rebuilds the backend (which runs `alembic upgrade head` on boot), take a
# custom-format dump of the live database. Encryption keys are preserved in
# secrets/ and documents persist in the minio volume, so this snapshot + those is
# a full recovery point. Fail-closed: abort the upgrade if the snapshot fails.
preupgrade_backup_docker() {
  step "Fail-closed pre-upgrade database backup"
  if [[ "${CAREFLOW_SKIP_PREUPGRADE_BACKUP:-0}" == "1" ]]; then
    warn "CAREFLOW_SKIP_PREUPGRADE_BACKUP=1 - skipping the pre-upgrade backup (NOT recommended)"
    return 0
  fi
  cd "$REPO_ROOT"

  # PREFER THE REAL, ENCRYPTED BACKUP. THE RAW DUMP BELOW IS PLAINTEXT PHI.
  #
  # `pg_dump -Fc` is compressed, not encrypted. The dump this function writes
  # therefore holds every participant's NAME, DATE OF BIRTH, email and mobile in
  # the clear - those columns are not field-encrypted (see
  # docs/CONFIDENTIAL/COMPLIANCE_GUIDE.md) - in the checkout directory, at the
  # shell's default umask, for ever. `deploy/scripts/careflow-backup.sh` exists
  # precisely to prevent that: it REFUSES to write an unencrypted archive. The
  # bare-metal installer routes its pre-upgrade snapshot through it
  # (`--preupgrade`) and so does cPanel. This avenue did not, and it is the
  # flagship.
  #
  # WHY THIS IS NOT SIMPLY "ALWAYS USE careflow-backup.sh". That script dies when
  # no encryption is configured, and this function aborts the upgrade when the
  # backup fails. Wiring the two together unconditionally would make an upgrade
  # IMPOSSIBLE for every existing Docker install that has never set a backup
  # passphrase - stranding practices on an old version, which is its own security
  # problem. So: use the encrypted path when the operator has configured one, and
  # otherwise take the dump but stop leaving it lying around readable.
  # The keys live in .env - that is where the warning below tells the operator
  # to put them - and this script never sources .env, so read them the way
  # POSTGRES_USER is read further down. Without this the encrypted path could
  # not be reached by following its own instruction (release-cut council C).
  local _bk _bv
  for _bk in CAREFLOW_BACKUP_AGE_RECIPIENT CAREFLOW_BACKUP_PASSPHRASE CAREFLOW_BACKUP_KEYFILE; do
    if [[ -z "${!_bk:-}" && -f "$ENV_FILE" ]]; then
      _bv="$(sed -n "s/^${_bk}=//p" "$ENV_FILE" 2>/dev/null | head -n1)"
      # An explicit `if`: under `set -e` a bare `[[ ... ]] && export` that fails on
      # the LAST key would end the loop non-zero and abort the whole upgrade.
      if [[ -n "$_bv" ]]; then export "${_bk}=${_bv}"; fi
    fi
  done
  if [[ -n "${CAREFLOW_BACKUP_AGE_RECIPIENT:-}${CAREFLOW_BACKUP_PASSPHRASE:-}${CAREFLOW_BACKUP_KEYFILE:-}" \
        && -f "${REPO_ROOT}/deploy/scripts/careflow-backup.sh" ]]; then
    info "Backup encryption is configured - taking the full encrypted recovery point"
    if bash "${REPO_ROOT}/deploy/scripts/careflow-backup.sh" --preupgrade; then
      ok "Pre-upgrade snapshot: encrypted archive written by careflow-backup.sh"
      return 0
    fi
    err "Pre-upgrade backup FAILED (careflow-backup.sh) - aborting the upgrade (NO migration run)."
  fi

  # 0700, and the dump itself 0600 below. The directory sits inside the checkout,
  # which on a default Docker host is world-readable.
  mkdir -p "${REPO_ROOT}/backups"
  chmod 700 "${REPO_ROOT}/backups" 2>/dev/null || true
  # Honour a customised POSTGRES_USER/POSTGRES_DB, and dump over the container
  # SOCKET (no -h): the bootstrap superuser's socket auth is trust inside the
  # container, so a restored volume whose password diverged from the secrets
  # file no longer refuses the snapshot for the wrong reason (B65/Council L).
  local pg_user pg_db
  pg_user="$(sed -n 's/^POSTGRES_USER=//p' "$ENV_FILE" 2>/dev/null | head -n1)"; pg_user="${pg_user:-careflow}"
  pg_db="$(sed -n 's/^POSTGRES_DB=//p' "$ENV_FILE" 2>/dev/null | head -n1)"; pg_db="${pg_db:-careflow}"
  # Ensure the database container is up so it can be dumped (upgrade may start 'down').
  $DOCKER_COMPOSE up -d postgres >/dev/null 2>&1 || true
  local i=0
  until $DOCKER_COMPOSE exec -T postgres pg_isready -U "$pg_user" >/dev/null 2>&1; do
    i=$((i + 1)); [[ $i -ge 20 ]] && break; sleep 2
  done
  local out
  out="${REPO_ROOT}/backups/pre-upgrade-$(date -u +%Y%m%dT%H%M%SZ).dump"
  # CREATE THE FILE 0600 BEFORE ANY PHI GOES INTO IT. `>` truncates an existing
  # file and leaves its mode alone, so making it here under a private umask is
  # what fixes the permissions - a chmod after the redirect would leave the dump
  # group- and world-readable for as long as pg_dump takes to stream it, which on
  # a real practice database is minutes. The subshell keeps the umask local.
  ( umask 077; : > "$out" ) 2>/dev/null \
    || err "Cannot write ${out} - check the permissions on ${REPO_ROOT}/backups."
  if $DOCKER_COMPOSE exec -T postgres \
       pg_dump -U "$pg_user" -Fc "$pg_db" > "$out" 2>/dev/null && [[ -s "$out" ]]; then
    ok "Pre-upgrade snapshot: ${out} ($(du -h "$out" | cut -f1))"
    info "Keys preserved in secrets/; documents persist in their own volume (uploads, or minio_data on an install that still runs the object store) - this is a full recovery point."
    # THIS FILE IS NOT ENCRYPTED. Say so, and give the one line that fixes it,
    # because the operator reading this is the only person who can.
    warn "That snapshot is COMPRESSED BUT NOT ENCRYPTED and holds participant names, dates of birth and contact details in the clear."
    warn "To make every future pre-upgrade snapshot encrypted instead, add one line to ${ENV_FILE} and re-run:  CAREFLOW_BACKUP_PASSPHRASE=<a long passphrase you keep OFF this server>"
    # PRUNE. An unencrypted copy of the whole clinical record is a liability that
    # grows with every upgrade, so keep the newest few and delete the rest. The
    # prune runs only AFTER the new dump succeeded, so it can never leave the
    # host with no recovery point at all.
    local retain
    retain="${CAREFLOW_PREUPGRADE_RETAIN:-2}"
    # Never below one: a value of 0 would delete the dump just taken.
    [[ "$retain" =~ ^[0-9]+$ ]] && (( retain >= 1 )) || retain=1
    local -a dumps=()
    local f
    for f in "${REPO_ROOT}"/backups/pre-upgrade-*.dump; do
      [[ -e "$f" ]] && dumps+=("$f")
    done
    if ((${#dumps[@]} > retain)); then
      # Newest first. The names this function writes are ISO-8601 UTC stamps, so
      # reverse lexical order IS reverse chronological order - no stat needed.
      mapfile -t dumps < <(printf '%s\n' "${dumps[@]}" | sort -r)
      local i
      for ((i = retain; i < ${#dumps[@]}; i++)); do
        rm -f "${dumps[i]}" \
          && info "Pruned older unencrypted snapshot: $(basename "${dumps[i]}") (keeping the newest ${retain}; set CAREFLOW_PREUPGRADE_RETAIN to change)"
      done
    fi
  else
    rm -f "$out"
    err "Pre-upgrade backup FAILED - aborting the upgrade (NO migration run). Set CAREFLOW_SKIP_PREUPGRADE_BACKUP=1 to override at your own risk."
  fi
}

# ── Parse arguments ────────────────────────────────────────────────────────────
UPGRADE_MODE=0
for arg in "$@"; do
  case "$arg" in
    --upgrade) UPGRADE_MODE=1 ;;
    -h|--help)
      echo "Usage: bash $0 [--upgrade]"
      echo ""
      echo "  (no flag)   Fresh install, or a safe re-run. Every secret in secrets/"
      echo "              (including the PHI encryption key) is reused if already"
      echo "              present, never rotated, so stored data stays readable."
      echo "  --upgrade   Update an existing install in place: pull the new images,"
      echo "              rebuild, run database migrations and restart the containers,"
      echo "              reusing the existing .env and secrets unchanged."
      exit 0 ;;
    *) err "Unknown argument: $arg (try --help)" ;;
  esac
done

# ── Main ──────────────────────────────────────────────────────────────────────
main() {
  check_docker
  if [[ "$UPGRADE_MODE" -eq 1 ]]; then
    [[ -f "${REPO_ROOT}/.env" ]] || err "--upgrade needs an existing install, but ${REPO_ROOT}/.env was not found. Run a fresh install first."
    step "Upgrade mode - reusing existing .env and secrets (nothing regenerated)"
    DOMAIN="$(sed -n 's/^CAREFLOW_DOMAIN=//p' "${REPO_ROOT}/.env" | head -n1)"; DOMAIN="${DOMAIN:-careflow.local}"
    ADMIN_EMAIL="$(sed -n 's/^ADMIN_EMAIL=//p' "${REPO_ROOT}/.env" | head -n1)"
    PULL_MODELS=n
    # generate_secrets is idempotent: it fills in any secret file that is missing
    # but NEVER rewrites one that already exists (incl. the PHI encryption key).
    generate_secrets
    heal_env
    ok "Existing configuration preserved (Domain: ${DOMAIN})"
    preupgrade_backup_docker
  else
    gather_config
    generate_secrets
    write_env
  fi
  start_services
  ensure_app_role
  ensure_runtime_role
  apply_runtime_grants
  if health_check; then
    pull_models
    print_summary
  else
    print_failure_summary
    exit 1
  fi
}

main "$@"
