#!/usr/bin/env bash
#
# Olakai On-Prem (Tier 2) — one-command install script.
#
# Headline usage:
#   curl -fsSL https://get.olakai.ai | bash
#
# Review-first usage (recommended):
#   curl -fsSL https://get.olakai.ai -o get.sh
#   less get.sh
#   bash get.sh
#
# What this script does (audit summary):
#   1. Pre-flight: OS / RAM / disk / ports / container runtime. Default runtime
#      is Docker; --runtime=podman-rootless prepares rootless Podman instead
#      (packages, compose provider, service user, subordinate ids, linger,
#      sysctl, user-scope socket) and runs the bundle as the service user.
#      --no-root (podman-rootless only) runs as the service user itself on a
#      host an admin prepared: every prerequisite is verified, none is
#      installed, and a missing one prints the root command the admin runs.
#   2. Bootstrap Sigstore cosign if missing (verifies its own SHA256).
#   3. Collect license key, domain, admin email (interactive, with --flag
#      pre-fills; the license key is also read from OLAKAI_LICENSE_KEY so
#      automation can keep it off argv).
#   4. POST /api/onprem/install-handshake to relay.olakai.ai with the license key
#      and the domain. Receives a 15-min signed download URL plus the bundle's
#      expected SHA256 + cosign signature + cert.
#   5. Download + sha256-verify + cosign-verify the bundle tarball.
#   6. Extract to /opt/olakai-onprem (configurable).
#   7. Materialize .env: secrets via openssl rand, OLAKAI_DOMAIN/AUTH_URL from
#      --domain, OLAKAI_EMAIL_RELAY_KEY from the handshake bearer; then the
#      KEY=VALUE overrides from --env-file (managed PostgreSQL / object
#      storage / Redis) are merged in, so the bundle's install.sh sees the
#      final topology on the FIRST bring-up and starts no stateful container
#      the customer did not ask for. --private-ca stages the CA PEM and the
#      private-CA overlay in the install dir for the same reason.
#   8. Run the bundle's own install.sh, which verifies the four image
#      signatures and brings the stack up via docker compose. Under
#      --runtime=podman-rootless this runs as the service user via runuser.
#   9. Wait for app health; print the success banner with the dashboard URL +
#      magic-link delivery confirmation + (per D-022 Option B) the bootstrap-log
#      magic-link fallback URL.
#
# Source repo:    https://github.com/olakai-ai/olakai-installer
# Public verify:  see SECURITY.md in that repo
# Issues / docs:  https://docs.olakai.ai/on-prem/install
#
# This script supports many bundle versions. Pin one with --version=v1.0.0;
# default is the latest non-yanked release the relay knows about. Verification
# (sha256 + cosign keyless) is enforced regardless of which version is
# installed; --skip-verify is an air-gapped escape hatch with a loud warning.
#
# Trust model for image verification:
#   The default posture skips per-image cosign verification at install time —
#   the four image references in docker-compose.yml are digest-pinned (@sha256
#   immutable refs) inside the bundle tarball, and the bundle tarball itself
#   is cosign-verified above. Pass --verify-images (or set
#   OLAKAI_VERIFY_IMAGES=true) to additionally run per-image cosign verify in
#   the bundle's install.sh — strict mode for customers with policy
#   requirements. --skip-verify remains orthogonal: it disables ALL cosign
#   verification (bundle + images) for air-gapped installs.
#
# License: MIT (see LICENSE).

set -euo pipefail
IFS=$'\n\t'

# ──────────────────────────────────────────────────────────────────────────────
# Versioning + chain-of-trust constants
# ──────────────────────────────────────────────────────────────────────────────

GET_SH_VERSION="2026.09.22"

# Pinned cosign version + SHA256 sums for the cosign binaries we may bootstrap.
# Update procedure: bump COSIGN_VERSION and refresh both SHA256s from the
# matching release's cosign_checksums.txt:
#   https://github.com/sigstore/cosign/releases/download/<COSIGN_VERSION>/cosign_checksums.txt
# Last refreshed: 2026-05-06 (cosign v3.0.6).
COSIGN_VERSION="v3.0.6"
COSIGN_SHA256_LINUX_AMD64="c956e5dfcac53d52bcf058360d579472f0c1d2d9b69f55209e256fe7783f4c74"
COSIGN_SHA256_LINUX_ARM64="bedac92e8c3729864e13d4a17048007cfafa79d5deca993a43a90ffe018ef2b8"

# Bundle-tarball cosign identity. Set on every release by the
# onprem-publish.yml workflow's `cosign sign-blob` step (W2 / OLA-146).
# The regexp matches every stable release tag (vX.Y.Z); pre-release tags
# (-rc, -beta, etc) are intentionally excluded so customers can't be tricked
# into installing a non-stable build that happens to have a valid signature.
COSIGN_CERT_IDENTITY_REGEXP='^https://github\.com/olakai-ai/localnode-app/\.github/workflows/onprem-publish\.yml@refs/tags/v[0-9]+\.[0-9]+\.[0-9]+$'
COSIGN_OIDC_ISSUER='https://token.actions.githubusercontent.com'

# Endpoints. Override via env for staging/testing.
RELAY_HANDSHAKE_URL_DEFAULT="${OLAKAI_RELAY_HANDSHAKE_URL:-https://relay.olakai.ai/api/onprem/install-handshake}"

# OS support matrix (humans-readable copy used in error messages).
SUPPORTED_OSES_HUMAN="Ubuntu 22.04+, Debian 12+, RHEL 9+, Oracle Linux 9+, Amazon Linux 2023"

# Resource thresholds.
#
# RAM is checked in MB, not GB, on purpose. MemTotal in /proc/meminfo is the
# memory left for userspace after the kernel, firmware and initrd take their
# reservation, so it is always below the size the provider advertises: every
# "8 GB" cloud VM (AWS, GCP, Azure, Hetzner, DigitalOcean) reports ~7.5-7.7 GiB.
# Comparing whole truncated GB rejected all of them. MIN_RAM_MB sits below that
# band but well above a genuinely undersized 4 GB box; MIN_RAM_ADVERTISED_GB is
# the number operators actually provision against and is what error messages
# quote back to them.
MIN_RAM_MB=7500
MIN_RAM_ADVERTISED_GB=8
# Free disk is checked per file system, not on / alone: hardened images
# (e.g. CIS-style Oracle Linux 9) keep / small and mount /home, /var, /var/tmp
# and /opt separately. Container images need MIN_FREE_CONTAINER_STORAGE_GB
# where the runtime stores them (~5 GB per bundle version; upgrades keep the
# previous images); the install dir needs MIN_FREE_INSTALL_DIR_GB (config,
# pg_dump backups, bundled PostgreSQL / MinIO data); rootless Podman stages
# pulls in /var/tmp (MIN_FREE_IMAGE_TMP_GB). Needs on one file system are
# summed, with the /var/tmp share counted inside the other two, so a
# flat-root host still needs exactly 80 GB free on /. See _disk_needs.
MIN_FREE_CONTAINER_STORAGE_GB=60
MIN_FREE_INSTALL_DIR_GB=20
MIN_FREE_IMAGE_TMP_GB=5
REQUIRED_PORTS=(80 443)
MIN_DOCKER_MAJOR=24

# Rootless Podman runtime (S4 / OLA-1294).
#
# Podman 5 is the qualified floor: the OL 9.8 qualification ran on 5.8.2, and
# Ubuntu 24.04's 4.9 is not qualified. `podman compose` delegates to an
# external provider; the bundle needs the real docker-compose (podman-compose
# lacks the service_healthy / service_completed_successfully conditions).
# The provider binary is pinned by version and SHA256 like cosign above.
# Update procedure: bump COMPOSE_VERSION and refresh both sums from
#   https://github.com/docker/compose/releases/download/<COMPOSE_VERSION>/checksums.txt
# Last refreshed: 2026-09-03 (docker-compose v5.5.1).
MIN_PODMAN_MAJOR=5
COMPOSE_VERSION="v5.5.1"
COMPOSE_SHA256_LINUX_X86_64="db1889184726840f75c4f9c001048430d4f25b3be3cb084d3ddd762bc0aed576"
COMPOSE_SHA256_LINUX_AARCH64="732e3a84c1a0f67256ce80bc2598a24546b10ca05f9faa97efceb1171ece2ef7"
# Subordinate uid/gid range for the service user. 65536 ids is Podman's
# documented minimum for images that use many uids. The search starts at
# 524288 (= 8 x 65536) because useradd hands the first interactive user
# 100000:65536 (SUB_UID_MIN in login.defs on both families), so `opc`,
# `ubuntu` or `ec2-user` usually own 100000-165535 already; two users sharing
# a range can read and write each other's container files. The search steps
# by 65536 past every existing entry, so the picked start stays aligned.
SUBID_COUNT=65536
SUBID_START_MIN=524288
SUBID_MAX_PROBES=1024

# ──────────────────────────────────────────────────────────────────────────────
# Mutable globals (filled by parse_args / prompts / handshake)
# ──────────────────────────────────────────────────────────────────────────────

# License key precedence: --license-key flag > OLAKAI_LICENSE_KEY env > silent
# TTY prompt. The env path exists for automation (e.g. an Ansible task with
# `environment:` + `no_log: true`): a flag value lands in this process's argv
# and is readable in `ps` and /proc/<pid>/cmdline on the target host; an env
# value is not visible there (it does remain in /proc/<pid>/environ, which is
# readable by root only). We consume the variable here and unset it at once so no child
# process (python3, curl, cosign, get.docker.com, the bundle's install.sh)
# inherits it. The key itself only ever reaches curl through a 0600 -K config
# file (see do_handshake).
LICENSE_KEY="${OLAKAI_LICENSE_KEY:-}"
unset OLAKAI_LICENSE_KEY
DOMAIN=""
ADMIN_EMAIL=""
INSTALL_DIR="/opt/olakai-onprem"
NO_TLS="false"
SKIP_VERIFY="false"
# VERIFY_IMAGES toggles per-image cosign verification inside the bundle's
# install.sh. Default false: the bundle is already cosign-verified above and
# images are digest-pinned. Set true via --verify-images or
# OLAKAI_VERIFY_IMAGES=<truthy> for strict mode (CLI flag wins over env).
#
# Truthy values normalized below: `true`, `1`, `yes`, `TRUE`, `True`. Anything
# else (including the common `OLAKAI_VERIFY_IMAGES=0` or unset) leaves
# VERIFY_IMAGES=false. We accept multiple truthy spellings because a silent
# disable on a security flag is the worst-possible failure mode — operators
# trying `=1` and getting silently ignored would think they were verifying.
case "${OLAKAI_VERIFY_IMAGES:-false}" in
  true|TRUE|True|1|yes|YES|Yes) VERIFY_IMAGES="true" ;;
  *)                            VERIFY_IMAGES="false" ;;
esac
# FORCE_OVERLAY answers "yes" to the non-empty --install-dir confirmation so
# automation can re-run get.sh without a TTY. Existing .env is preserved on
# that path regardless. Same truthy normalization as OLAKAI_VERIFY_IMAGES
# above; CLI flag --force-overlay wins over env.
case "${OLAKAI_FORCE_OVERLAY:-false}" in
  true|TRUE|True|1|yes|YES|Yes) FORCE_OVERLAY="true" ;;
  *)                            FORCE_OVERLAY="false" ;;
esac
REKOR_BUNDLE=""
VERSION_PIN=""
NON_INTERACTIVE="false"
QUIET="false"
AUTO_INSTALL_DOCKER="false"
RELAY_HANDSHAKE_URL="$RELAY_HANDSHAKE_URL_DEFAULT"
# Container runtime (S4 / OLA-1294): `docker` (default on every distro) or
# `podman-rootless` (explicit opt-in). Rootful Podman is not an installer
# mode. SERVICE_USER only matters under podman-rootless.
RUNTIME="docker"
SERVICE_USER="olakai"
# --no-root (OLA-1443): run get.sh as the service user on a host an admin
# already prepared (podman, subordinate ids, linger, sysctl, install dir).
# Nothing is installed system-wide and get.sh never escalates: every missing
# prerequisite is a refusal that prints the one-line root command to run.
# The service user is the invoking user; cosign and docker-compose go to
# $HOME/.local/bin and the compose provider is registered in the user's
# containers.conf.d. Flag only, like --env-file: no env-var equivalent.
NO_ROOT="false"

# --env-file=PATH: KEY=VALUE overrides merged into .env after the secret
# fill and before the hand-off (managed-state installs: customer-managed
# PostgreSQL, object storage and Redis, so the bundle's install.sh starts no
# bundled stateful container on the first bring-up). An explicit flag only,
# never an environment variable: the file is read as root and its values
# land in .env, so the path must come from argv the operator can review.
# --private-ca=PATH: a CA certificate (PEM) for managed PostgreSQL / Redis
# whose certificates do not chain to a public root; staged as
# <install-dir>/certs/private-ca.pem plus the bundle's private-CA overlay.
# The two flags are independent.
ENV_FILE=""
PRIVATE_CA=""
# Keys get.sh derives itself; an --env-file that names one is refused in
# pre-flight (the value would be overwritten, or would break the install).
# OLAKAI_SKIP_VERIFY and OLAKAI_VERIFY_IMAGES are refused for a different
# reason: they are --skip-verify / --verify-images, and get.sh resolves the
# conflict between the two in bring_up. install.sh sources .env, so a value
# smuggled in through the file would bypass that resolution.
GET_SH_OWNED_ENV_KEYS=(OLAKAI_DOMAIN AUTH_URL OLAKAI_ADMIN_EMAIL OLAKAI_EMAIL_RELAY_KEY OLAKAI_VERSION OLAKAI_SKIP_VERIFY OLAKAI_VERIFY_IMAGES)
# Filled by check_env_file (pre-flight): the parsed KEY / VALUE pairs of
# --env-file, in file order. The file is read exactly once, here; the merge
# in materialize_env works from these arrays and never copies the file.
# Values are held in this process only (never exported, never printed).
ENV_FILE_KEYS=()
ENV_FILE_VALUES=()
ENV_FILE_PARSED="false"
# Set by _merge_env_file while its 0600 temp file exists next to .env, so
# cleanup (the EXIT trap) removes it if the merge dies midway; cleared once
# the rename into place succeeded.
ENV_MERGE_TMP=""

# Filled by ensure_service_user: numeric uid of SERVICE_USER. _runtime_exec
# keys the user's XDG_RUNTIME_DIR and D-Bus address on it.
SERVICE_UID=""
# Filled by _pick_subid_start (output-variable convention, like
# TARBALL_PATH_OUT): the subordinate id range chosen for SERVICE_USER.
SUBID_START_OUT=""
SUBID_COUNT_OUT=""

# Filled by detect_os: the ID value from os-release (e.g. "ol").
# check_or_install_docker reads it to pick the Docker install path.
OS_ID=""

# Filled by ensure_cosign: absolute path of the cosign binary every cosign
# call uses. Never call `cosign` by bare name: the RHEL family's sudo ships
# `Defaults secure_path = /sbin:/bin:/usr/sbin:/usr/bin` (no /usr/local/bin),
# so a binary we just installed there is invisible to `command -v` /
# PATH lookup even though the install succeeded (seen live on Oracle Linux
# 9.8, OLA-1292). Ubuntu's secure_path includes /usr/local/bin, which is why
# this never surfaced there.
COSIGN_BIN=""

# Filled by handshake.
HANDSHAKE_DEPLOYMENT_BEARER=""
HANDSHAKE_DOWNLOAD_URL=""
HANDSHAKE_EXPECTED_SHA256=""
HANDSHAKE_EXPECTED_SIG_B64=""
HANDSHAKE_EXPECTED_CERT_PEM=""
HANDSHAKE_VERSION=""

# Workspace dir (mktemp); cleaned up on exit.
WORK_DIR=""

# ──────────────────────────────────────────────────────────────────────────────
# Print helpers (color-aware; degrade when not a TTY)
# ──────────────────────────────────────────────────────────────────────────────

if [[ -t 1 ]] && [[ -z "${NO_COLOR:-}" ]]; then
  C_RESET=$'\033[0m'
  C_BOLD=$'\033[1m'
  C_DIM=$'\033[2m'
  C_GREEN=$'\033[32m'
  C_YELLOW=$'\033[33m'
  C_RED=$'\033[31m'
  C_CYAN=$'\033[36m'
else
  C_RESET=""
  C_BOLD=""
  C_DIM=""
  C_GREEN=""
  C_YELLOW=""
  C_RED=""
  C_CYAN=""
fi

print_step()  { printf '%s\n' "${C_BOLD}${C_CYAN}==>${C_RESET} ${C_BOLD}$*${C_RESET}"; }
print_info()  { printf '    %s\n' "$*"; }
print_dim()   { printf '    %s%s%s\n' "$C_DIM" "$*" "$C_RESET"; }
print_warn()  { printf '%swarn:%s %s\n' "${C_YELLOW}" "${C_RESET}" "$*" >&2; }
print_error() { printf '%serror:%s %s\n' "${C_RED}" "${C_RESET}" "$*" >&2; }
print_ok()    { printf '    %s✓%s %s\n' "$C_GREEN" "$C_RESET" "$*"; }

die() {
  print_error "$*"
  exit 1
}

# ──────────────────────────────────────────────────────────────────────────────
# Cleanup
# ──────────────────────────────────────────────────────────────────────────────

cleanup() {
  local rc=$?
  if [[ -n "$WORK_DIR" && -d "$WORK_DIR" ]]; then
    rm -rf "$WORK_DIR"
  fi
  # A merge that died between writing its temp file and renaming it must not
  # leave a second copy of the credentials next to .env.
  if [[ -n "${ENV_MERGE_TMP:-}" && -e "$ENV_MERGE_TMP" ]]; then
    rm -f "$ENV_MERGE_TMP"
  fi
  # Tell the operator about preserved partial state so they don't have to
  # guess why a re-run hits the "non-empty install dir" prompt or finds an
  # existing .env. We only print this on non-zero exits past the point
  # where we may have created files in INSTALL_DIR.
  if (( rc != 0 )) && [[ -n "$INSTALL_DIR" ]] && [[ -d "$INSTALL_DIR" ]] \
      && [[ -n "$(ls -A "$INSTALL_DIR" 2>/dev/null)" ]]; then
    printf '\n%snote:%s partial install state at %s is preserved. Re-run get.sh to retry; existing .env values (if any) will not be overwritten.\n' \
      "${C_DIM}" "${C_RESET}" "${INSTALL_DIR}" >&2
  fi
  exit "$rc"
}
trap cleanup EXIT INT TERM

# ──────────────────────────────────────────────────────────────────────────────
# TTY binding for interactive prompts (D-024 / OLA-169).
#
# Under `curl … | bash`, bash reads the SCRIPT ITSELF from fd 0 (the pipe).
# The old idiom `exec < /dev/tty` rebound fd 0 to the terminal — which made
# bash read the REST of the script (every function definition and the final
# `main` call) from the now-idle terminal, hanging silently with zero output
# even in --non-interactive mode (OLA-301). Running from a file never hit this
# because bash reads the script from the file, never from fd 0.
#
# Fix: never touch fd 0. Bind the controlling terminal to a dedicated
# descriptor (fd 3) and read every prompt from it. TTY_FD records where
# interactive reads should come from:
#   - fd 0  when stdin is already a terminal (review-first `bash get.sh` path)
#   - fd 3  when we successfully borrowed /dev/tty (the `curl | bash` path)
#   - empty when no terminal is available (true CI, `docker exec` without -t,
#           a pipe with no controlling tty) → the script runs strictly
#           non-interactive and every prompt hard-fails on missing input.
#
# The probe tolerates hosts where /dev/tty exists but has no controlling
# terminal: `exec 3</dev/tty` fails, 2>/dev/null swallows bash's "cannot open
# /dev/tty" diagnostic, and TTY_FD stays empty. The `-t 3` guard rejects the
# case where the open succeeds but doesn't yield an actual terminal.
# ──────────────────────────────────────────────────────────────────────────────

TTY_FD=""
if [[ -t 0 ]]; then
  TTY_FD=0
elif { exec 3</dev/tty; } 2>/dev/null && [[ -t 3 ]]; then
  TTY_FD=3
fi

# ──────────────────────────────────────────────────────────────────────────────
# Help / usage
# ──────────────────────────────────────────────────────────────────────────────

print_help() {
  cat <<'EOF'
Olakai on-prem one-command install.

Usage:
  curl -fsSL https://get.olakai.ai | bash
  curl -fsSL https://get.olakai.ai | bash -s -- [flags]

Supported OS: Ubuntu 22.04+, Debian 12+, RHEL 9+ (incl. Rocky, AlmaLinux,
  CentOS Stream), Oracle Linux 9+, Amazon Linux 2023. x86_64 / arm64.

Flags (all optional in interactive mode; required with --non-interactive):
  --license-key=KEY        License from Olakai sales (silent prompt if unset).
                           Equivalent env var: OLAKAI_LICENSE_KEY=KEY
                           Precedence: flag > env > prompt. Prefer the env var
                           from automation (e.g. Ansible `environment:` with
                           `no_log: true`): a flag value is visible in `ps` and
                           /proc/<pid>/cmdline on the host; the env var is not
                           (it stays only in root-only /proc/<pid>/environ) and
                           is unset before any child process starts.
  --domain=DOMAIN          Customer-facing domain (becomes appUrl + AUTH_URL).
  --admin-email=EMAIL      First admin user; receives a magic-link login.
  --install-dir=DIR        Where to extract the bundle. Default: /opt/olakai-onprem
                           Needs 20 GB free (config, backups, bundled database
                           and object storage data), checked in pre-flight.
  --no-tls                 Skip Caddy + Let's Encrypt (use behind a load balancer).
  --skip-verify            Skip cosign verify (air-gapped escape hatch — UNSAFE).
  --verify-images          Run per-image cosign verification at install (strict mode).
                           Default: skip (images are digest-pinned in the verified bundle).
                           Equivalent env var: OLAKAI_VERIFY_IMAGES=true
  --rekor-bundle=PATH      Use an offline cosign --bundle for verification.
  --version=VERSION        Pin to a specific bundle version (e.g. v1.0.0).
  --non-interactive        Hard-fail on missing required flags. Implied by CI=true.
  --force-overlay          Overlay a non-empty --install-dir without asking
                           (existing .env is always preserved). Without it,
                           --non-interactive aborts on a non-empty directory.
                           Automation should either guard on the presence of
                           <install-dir>/.env (e.g. Ansible `creates:`) or
                           pass this flag.
                           Equivalent env var: OLAKAI_FORCE_OVERLAY=true
  --auto-install-docker    Allow get.sh to install Docker when missing: via
                           get.docker.com, or from Docker's CentOS package repo
                           on Oracle Linux (get.docker.com rejects ID=ol; Docker
                           CE on OL is community-supported, Oracle's supported
                           runtime is Podman). Default in interactive mode is to prompt;
                           in non-interactive mode this flag is REQUIRED to opt in
                           (otherwise we fail with manual-install instructions —
                           the get.docker.com installer is not signature-verified).
  --quiet                  Suppress the magic-link fallback URL on the success banner
                           (OLAKAI_SETUP_URL= is then printed empty; see below).
  --runtime=RUNTIME        Container runtime: docker (default on every distro) or
                           podman-rootless (explicit opt-in). See "Container
                           runtime" below.
                           Rootful Podman is not an installer mode (the bundle
                           README documents it as a manual compatibility setup).
  --service-user=NAME      Unprivileged user that owns the stack under
                           --runtime=podman-rootless. Created if missing.
                           Default: olakai. Ignored under --runtime=docker.
                           A re-run must name the same user; the deployment
                           is not moved between users in place.
  --no-root                Run as the service user on a host an admin already
                           prepared (see "Prepared host" below). Only with
                           --runtime=podman-rootless. get.sh must then run
                           WITHOUT sudo, as the user that will own the stack;
                           --service-user, if given, must name that user.
                           Nothing is installed system-wide and get.sh never
                           calls sudo: a missing prerequisite stops the run
                           before the license handshake and prints the root
                           command an admin runs. No environment-variable
                           equivalent.
  --env-file=PATH          KEY=VALUE lines merged into .env before the bundle's
                           install.sh runs (managed-state installs; see
                           "Managed state" below). Comments and blank lines
                           are allowed; values are copied verbatim, quotes
                           included, like .env. Single-quote a value that
                           contains $, a backtick, whitespace or " #"
                           (KEY='pa$w0rd'): .env is sourced by install.sh,
                           interpolated by compose and read by the bundle's
                           env reader, and only single quotes mean the same
                           literal to all of them (unquoted pa$w0rd sources
                           as "pa"; " #" after an unquoted value starts a
                           comment; double quotes do not protect $ or
                           backticks from the shell). Such a value is warned
                           about by key and line, never refused (${VAR}
                           interpolation may be intended). Each key replaces
                           the existing line in .env (same position) or is
                           appended. The file must be owned by root (by the
                           invoking user under --no-root) with mode 0600 or
                           0400 (it carries credentials); it is read once and
                           never copied. Keys get.sh sets
                           itself (OLAKAI_DOMAIN, AUTH_URL,
                           OLAKAI_ADMIN_EMAIL, OLAKAI_EMAIL_RELAY_KEY,
                           OLAKAI_VERSION) and the verification switches
                           (OLAKAI_SKIP_VERIFY: use --skip-verify;
                           OLAKAI_VERIFY_IMAGES: use --verify-images) are
                           refused. Checked in pre-flight, before the license
                           handshake. No environment-variable equivalent.
  --private-ca=PATH        CA certificate (PEM) for a managed PostgreSQL or
                           Redis whose certificate is not signed by a public
                           CA. Copied to <install-dir>/certs/private-ca.pem
                           (mode 0644) together with the bundle's
                           examples/docker-compose.private-ca.yml overlay,
                           after extraction and before the hand-off; the
                           bundle's install.sh and upgrade.sh auto-detect the
                           overlay. Needs bundle 1.5.8 or later. Independent
                           of --env-file.
  -h, --help               This text.

Container runtime:
  --runtime=docker          Docker 24+ with Compose v2 (installed on request, see
                            --auto-install-docker). The stack runs under the
                            Docker daemon, i.e. as root.
  --runtime=podman-rootless get.sh still runs as root for the OS preparation,
                            then hands the bundle to the service user. In order:
                            preflight (systemd, cgroup v2, user namespaces);
                            Podman 5 + podman-docker from the distro repos (dnf
                            or apt-get; Ubuntu 24.04 ships Podman 4.9 and is
                            not qualified); the docker-compose provider pinned
                            by SHA256 at /usr/local/bin/docker-compose and
                            named in /etc/containers/containers.conf.d;
                            `useradd -r -m` for the service user, a free
                            65536-wide range in /etc/subuid + /etc/subgid,
                            `loginctl enable-linger`; /etc/sysctl.d/90-olakai.conf
                            (net.ipv4.ip_unprivileged_port_start=80, so the
                            service user can bind 80/443 for Caddy); the
                            user-scope podman.socket and podman-restart.service;
                            then, after extraction, the install dir is handed
                            to the service user (owner, mode 0750,
                            .olakai-runtime marker, docker-compose.podman.yml
                            overlay) and install.sh runs via
                            `runuser -u <user>`. Needs bundle 1.5.10 or later.
                            Every later bundle script (upgrade.sh,
                            support-bundle.sh, scripts/restore.sh) must run the
                            same way; the banner prints the form. Files you add
                            later (certs/private-ca.pem, overlays) must be
                            readable by the service user (--private-ca and
                            --env-file are staged before the hand-off, so
                            they already are).
                            Qualified on Oracle Linux 9.8 (Podman 5.8.2); other
                            RHEL 9 family distributions ship the same Podman 5
                            and are accepted but not qualified. The docker
                            runtime remains the default.
                            An existing deployment is never converted in place:
                            a .olakai-runtime marker (or a data/ tree from a
                            pre-marker Docker install) that names another
                            runtime stops the run before the license handshake.
  Security summary: under rootless Podman a container escape yields the service user, not root.

Prepared host (--no-root, podman-rootless only):
  For an operator with no sudo. An admin prepares the host once, as root:
    dnf -y install podman podman-docker      (apt: podman podman-docker uidmap dbus-user-session)
    useradd -m -s /bin/bash olakai           (OL9/RHEL 9 allocate the subordinate
                                              ids; verify: grep olakai /etc/subuid)
    loginctl enable-linger olakai
    printf 'net.ipv4.ip_unprivileged_port_start=80\n' > /etc/sysctl.d/90-olakai.conf && sysctl --system
    install -d -o olakai -g olakai -m 750 /opt/olakai-onprem
    firewall-cmd --permanent --add-service=http --add-service=https && firewall-cmd --reload
  The operator then logs in over SSH as that user (a real login: the systemd
  user session must exist) and runs:
    curl -fsSL https://get.olakai.ai | bash -s -- --no-root --runtime=podman-rootless \
      --domain=olakai.example.com --admin-email=admin@example.com \
      [--env-file=$HOME/olakai-managed.env] [--private-ca=$HOME/ca.pem]
  The --env-file must be owned by that user, mode 0600. get.sh verifies each
  item above and installs cosign and docker-compose under $HOME/.local/bin
  (the compose provider is registered in $HOME/.config/containers). Every
  missing item is a refusal that prints the root command, before the license
  handshake. Later bundle scripts run as that user directly (no runuser).
  Qualified: not yet. Validated on the root path; the --no-root path is
  verified by the test suite only until an end-to-end run is recorded.

Managed state (customer-managed PostgreSQL, object storage and Redis):
  The bundle's install.sh reads the topology from .env: when DATABASE_URL,
  OBJECT_STORAGE_* and REDIS_URL all point at managed services it starts zero
  stateful containers (bundle 1.5.8 or later). Without --env-file, get.sh
  materializes .env from .env.example, so the FIRST bring-up would start the
  bundled PostgreSQL, MinIO and Redis. Pass the managed settings up front:

    # /root/olakai-managed.env  (chown root: … && chmod 600 …)
    DATABASE_URL='postgresql://USER:PASSWORD@db.example.com:5432/olakai?sslmode=verify-full'
    OBJECT_STORAGE_KIND=s3
    OBJECT_STORAGE_ENDPOINT=https://your-endpoint
    OBJECT_STORAGE_REGION=your-region
    OBJECT_STORAGE_FORCE_PATH_STYLE=true
    OBJECT_STORAGE_ACCESS_KEY='...'
    OBJECT_STORAGE_SECRET_KEY='...'
    OBJECT_STORAGE_BUCKET=customer-olakai-documents
    REDIS_URL='rediss://:PASSWORD@cache.example.com:6380'

    curl -fsSL https://get.olakai.ai | sudo bash -s -- \
      --env-file=/root/olakai-managed.env [--private-ca=/root/corp-ca.pem] \
      --domain=olakai.example.com --admin-email=admin@example.com

  The passwords and secrets are single-quoted on purpose: a password with $,
  a backtick, whitespace or " #" survives every reader of .env only that way
  (see --env-file).

  Use DNS names, not IPs (verify-full checks the hostname). If the PostgreSQL
  or Redis certificate is not signed by a public CA, add --private-ca=PATH:
  the application verifies against Node's built-in roots, not the host trust
  store, and without the CA `migrate` succeeds while the app then fails with
  TlsConnectionError.

Automation contract:
  On a successful install the last line on stdout is `OLAKAI_SETUP_URL=<url>`:
  the admin setup URL read from the bootstrap log, or empty when it could not
  be read within 30s or --quiet is set. Capture it with a grep (e.g. Ansible
  `register:`). The URL carries a one-time admin token: keep it out of shared
  job logs.

Failure modes (each prints an actionable message before exiting non-zero):
  License invalid / already consumed   relay returns 401/409 → re-confirm with sales.
  Tarball SHA256 mismatch              corruption or MITM → re-run; if persistent, bail.
  Cosign verify failure                signature mismatch → DO NOT continue; report it.
  Docker not present / too old         install Docker 24+ from get.docker.com
                                       (Oracle Linux: from Docker's CentOS repo).
  Compose v2 missing                   ensure 'docker compose version' works.
  Podman too old (podman-rootless)     need Podman 5+ (qualified on Oracle Linux 9.8,
                                       Podman 5.8.2; other RHEL 9 family distros are
                                       accepted, not qualified) or --runtime=docker.
  No cgroup v2 / user namespaces       podman-rootless needs both; fix the host
                                       or use --runtime=docker.
  Subordinate id range overlaps        another user already owns the range in
                                       /etc/subuid or /etc/subgid; fix and re-run.
  Host not prepared (--no-root)        podman, the subordinate ids, linger, the
                                       sysctl or the install dir is missing; the
                                       message ends with the root command to run.
  Existing deployment, other runtime   in-place runtime migration is not supported:
                                       back up, fresh-install, restore.
  Bundle older than 1.5.9              no Podman support in the bundle; the key is
                                       consumed → contact support for a new one.
  --env-file unreadable / wrong mode   pre-flight, before the handshake: the file
                                       must exist, be owned by root, mode 0600/0400,
                                       and not set a key get.sh owns (including
                                       OLAKAI_SKIP_VERIFY / OLAKAI_VERIFY_IMAGES).
  --env-file value warning             not fatal: a value that is not single-quoted
                                       and contains $, a backtick, whitespace or
                                       " #" is reported by key and line. Single-quote
                                       it unless ${VAR} interpolation is intended.
  --private-ca not a PEM certificate   pre-flight; a -----BEGIN CERTIFICATE-----
                                       header must appear in the first 4096 bytes.
  Bundle older than 1.5.8              no private-CA overlay in the bundle
                                       (--private-ca); the key is consumed.
  Ports 80/443 in use                  free the ports (or use --no-tls behind an LB).
  VM under 8 GB RAM                    provision a larger VM and re-run.
  Free disk too low                    pre-flight, per file system (df -BG: GiB,
                                       rounded up): 60 GB where container images
                                       live (docker: DockerRootDir, default
                                       /var/lib/docker, and /var/lib/containerd
                                       unless the classic store is confirmed;
                                       podman-rootless: the service user's home),
                                       20 GB for --install-dir, 5 GB for /var/tmp
                                       under podman-rootless when it is a separate
                                       file system. Needs on one file system are
                                       summed (a flat / needs 80 GB). A missing
                                       path is measured at its nearest existing
                                       parent. The message names the file system
                                       to grow or the path to mount a volume at.
  Stack didn't go healthy in 5 min     run /opt/olakai-onprem/support-bundle.sh.

Exit code is non-zero on any failure. Re-running this script is safe: secrets
in .env are never overwritten on a second pass.

Source: https://github.com/olakai-ai/olakai-installer
EOF
}

# ──────────────────────────────────────────────────────────────────────────────
# Argument parsing
# ──────────────────────────────────────────────────────────────────────────────

parse_args() {
  local service_user_given="false"
  while [[ $# -gt 0 ]]; do
    case "$1" in
      --license-key=*)   LICENSE_KEY="${1#*=}"; shift ;;
      --license-key)     LICENSE_KEY="${2:-}"; shift 2 ;;
      --domain=*)        DOMAIN="${1#*=}"; shift ;;
      --domain)          DOMAIN="${2:-}"; shift 2 ;;
      --admin-email=*)   ADMIN_EMAIL="${1#*=}"; shift ;;
      --admin-email)     ADMIN_EMAIL="${2:-}"; shift 2 ;;
      --install-dir=*)   INSTALL_DIR="${1#*=}"; shift ;;
      --install-dir)     INSTALL_DIR="${2:-}"; shift 2 ;;
      --no-tls)          NO_TLS="true"; shift ;;
      --skip-verify)     SKIP_VERIFY="true"; shift ;;
      --verify-images)   VERIFY_IMAGES="true"; shift ;;
      --rekor-bundle=*)  REKOR_BUNDLE="${1#*=}"; shift ;;
      --rekor-bundle)    REKOR_BUNDLE="${2:-}"; shift 2 ;;
      --version=*)       VERSION_PIN="${1#*=}"; shift ;;
      --version)         VERSION_PIN="${2:-}"; shift 2 ;;
      --non-interactive) NON_INTERACTIVE="true"; shift ;;
      --force-overlay)   FORCE_OVERLAY="true"; shift ;;
      --auto-install-docker) AUTO_INSTALL_DOCKER="true"; shift ;;
      --quiet)           QUIET="true"; shift ;;
      --runtime=*)       RUNTIME="${1#*=}"; shift ;;
      --runtime)         RUNTIME="${2:-}"; shift 2 ;;
      --service-user=*)  SERVICE_USER="${1#*=}"; service_user_given="true"; shift ;;
      --service-user)    SERVICE_USER="${2:-}"; service_user_given="true"; shift 2 ;;
      --no-root)         NO_ROOT="true"; shift ;;
      --env-file=*)      ENV_FILE="${1#*=}"; shift ;;
      --env-file)        ENV_FILE="${2:-}"; shift 2 ;;
      --private-ca=*)    PRIVATE_CA="${1#*=}"; shift ;;
      --private-ca)      PRIVATE_CA="${2:-}"; shift 2 ;;
      -h|--help)         print_help; exit 0 ;;
      --)                shift; break ;;
      *)                 die "unknown flag: $1 (run with --help)" ;;
    esac
  done

  case "$RUNTIME" in
    docker|podman-rootless) ;;
    podman)
      die "--runtime=podman (rootful) is not an installer mode. Use --runtime=podman-rootless (a container escape then yields the service user, not root); the bundle README documents rootful Podman as a manual compatibility setup only."
      ;;
    *) die "unknown --runtime '${RUNTIME}' (expected docker or podman-rootless)." ;;
  esac
  # --no-root (D-1, D-2): only meaningful where root was used for host
  # preparation alone, and the service user IS the invoking user.
  if [[ "$NO_ROOT" == "true" ]]; then
    if [[ "$RUNTIME" != "podman-rootless" ]]; then
      die "--no-root needs --runtime=podman-rootless: under --runtime=docker the stack runs under the Docker daemon, as root."
    fi
    local me
    me="$(id -un 2>/dev/null || true)"
    [[ -n "$me" ]] || die "could not resolve the invoking user ('id -un' printed nothing)."
    if [[ "$service_user_given" == "true" && "$SERVICE_USER" != "$me" ]]; then
      die "--service-user=${SERVICE_USER} but get.sh is running as '${me:-?}': under --no-root the service user is the invoking user. Drop --service-user, or log in as ${SERVICE_USER} and re-run."
    fi
    SERVICE_USER="$me"
  fi
  # The name reaches useradd, chown, runuser and two /etc files: keep it to
  # the portable POSIX user-name shape. root is refused explicitly because
  # it would silently turn the rootless runtime back into a root-owned stack.
  # Under --no-root the name is whatever `id -un` says (an AD/SSSD account
  # like jsmith@corp.example.com is legal there) and only reaches id, getent,
  # stat and message text, so the shape check does not apply.
  if [[ "$NO_ROOT" != "true" && ! "$SERVICE_USER" =~ ^[a-z_][a-z0-9_-]{0,31}$ ]]; then
    die "invalid --service-user '${SERVICE_USER}' (lowercase letters, digits, '_' and '-', max 32 chars)."
  fi
  # An empty --install-dir= would make every later step (pre-flight disk
  # check, extraction, chown) act on the current directory.
  if [[ -z "$INSTALL_DIR" ]]; then
    die "--install-dir must not be empty (default: /opt/olakai-onprem)."
  fi
  if [[ "$SERVICE_USER" == "root" ]]; then
    die "--service-user=root defeats --runtime=podman-rootless; pick an unprivileged name (default: olakai)."
  fi

  # No usable terminal (true CI, `docker exec` without -t, or a pipe with no
  # controlling tty) flips us into non-interactive mode; CI=true does too.
  # TTY_FD is set above and is empty exactly when no terminal could be bound.
  if [[ "${CI:-}" == "true" || -z "$TTY_FD" ]]; then
    NON_INTERACTIVE="true"
  fi
}

# ──────────────────────────────────────────────────────────────────────────────
# Pre-flight: OS / privileges / hardware / Docker
# ──────────────────────────────────────────────────────────────────────────────

# detect_os [OS_RELEASE_FILE]
# The optional argument exists for the test suite only (synthetic distros);
# main always calls it bare. It is deliberately NOT an environment variable:
# get.sh runs as root and sources this file, so an env-controlled path would
# let anything in the caller's environment inject shell into the installer.
# shellcheck disable=SC2120  # the optional argument is a test hook; main calls it bare
detect_os() {
  local os_release="${1:-/etc/os-release}"
  if [[ ! -r "$os_release" ]]; then
    die "cannot detect OS (no ${os_release}). Supported: ${SUPPORTED_OSES_HUMAN}."
  fi
  # shellcheck disable=SC1090,SC1091
  . "$os_release"

  local id="${ID:-}"
  local id_like="${ID_LIKE:-}"
  local version_id="${VERSION_ID:-}"
  local human="${PRETTY_NAME:-${id} ${version_id}}"
  OS_ID="$id"

  case "$id" in
    ubuntu)
      _semver_ge "$version_id" "22.04" \
        || die "Ubuntu ${version_id} is too old (need 22.04+). Detected: ${human}."
      ;;
    debian)
      _semver_ge "$version_id" "12" \
        || die "Debian ${version_id} is too old (need 12+). Detected: ${human}."
      ;;
    rhel|rocky|almalinux|centos)
      _semver_ge "$version_id" "9" \
        || die "RHEL-family ${version_id} is too old (need 9+). Detected: ${human}."
      ;;
    ol)
      # Oracle Linux gets its own arm: OL 9.8 reports ID_LIKE="fedora" only
      # (verified 2026-09-03), not "rhel fedora", so the ID_LIKE fallback
      # below never matched it. get.docker.com also rejects ID="ol"; see
      # _docker_needs_centos_repo for the Docker install path.
      _semver_ge "$version_id" "9" \
        || die "Oracle Linux ${version_id} is too old (need 9+). Detected: ${human}."
      ;;
    amzn)
      [[ "$version_id" == "2023" ]] \
        || die "Amazon Linux ${version_id} is not supported (need 2023). Detected: ${human}."
      ;;
    *)
      # Unknown ID: accept Debian-, RHEL- and Fedora-family derivatives with
      # a warning. "fedora" is included because some RHEL rebuilds report it
      # as their only ID_LIKE (Oracle Linux 9 does), and future rebuilds may
      # do the same.
      if [[ "$id_like" == *"rhel"* || "$id_like" == *"fedora"* || "$id_like" == *"debian"* ]]; then
        print_warn "Unsupported distro ${human}; ID_LIKE suggests it may work. Proceeding."
      else
        die "unsupported OS '${human}'. Supported: ${SUPPORTED_OSES_HUMAN}."
      fi
      ;;
  esac

  print_ok "OS: ${human}"
}

# Compares two dotted version strings: returns success iff $1 >= $2.
# Pure-bash; no `bc`, `python`, or `dpkg --compare-versions` required.
# Major.minor only — adequate for the OS-version gate (we never compare
# sub-minor like "9.5 vs 9.10"). If a future caller needs three components,
# extend this rather than assuming the existing semantics.
_semver_ge() {
  local lhs="$1" rhs="$2"
  local lhs_major lhs_minor rhs_major rhs_minor
  lhs_major="${lhs%%.*}"; lhs_minor="${lhs#*.}"; [[ "$lhs_minor" == "$lhs" ]] && lhs_minor="0"
  rhs_major="${rhs%%.*}"; rhs_minor="${rhs#*.}"; [[ "$rhs_minor" == "$rhs" ]] && rhs_minor="0"
  # Strip any trailing minor noise ("22.04.1" → "04").
  lhs_minor="${lhs_minor%%.*}"
  rhs_minor="${rhs_minor%%.*}"
  if (( lhs_major > rhs_major )); then return 0; fi
  if (( lhs_major < rhs_major )); then return 1; fi
  (( lhs_minor >= rhs_minor ))
}

check_root() {
  local uid
  uid="$(id -u)"
  if [[ "$NO_ROOT" == "true" ]]; then
    if (( uid == 0 )); then
      die "--no-root was passed but get.sh is running as root (uid 0). Drop --no-root, or run without sudo as the user that will own the stack."
    fi
    SERVICE_UID="$uid"
    print_ok "Running as ${SERVICE_USER} (--no-root): host preparation is verified, not performed"
    return 0
  fi
  if (( uid != 0 )); then
    if [[ "$RUNTIME" == "podman-rootless" ]]; then
      die "must run as root (use 'sudo bash get.sh' or curl … | sudo bash); or, on a host an admin already prepared, pass --no-root (see --help)."
    fi
    die "must run as root (use 'sudo bash get.sh' or curl … | sudo bash)."
  fi
  print_ok "Running as root"
}

check_resources() {
  local mem_kb mem_mb mem_gib
  mem_kb="$(awk '/^MemTotal:/ {print $2}' /proc/meminfo)"
  mem_mb=$(( mem_kb / 1024 ))
  # Reported to one decimal so an operator can see the kernel reservation
  # rather than a truncated whole number that looks a full GB short.
  mem_gib="$(awk -v kb="$mem_kb" 'BEGIN { printf "%.1f", kb / 1024 / 1024 }')"
  if (( mem_mb < MIN_RAM_MB )); then
    die "RAM too low (${mem_gib} GiB usable; need a ${MIN_RAM_ADVERTISED_GB} GB VM)."
  fi
  print_ok "RAM: ${mem_gib} GiB"

  check_disk_space

  if [[ "$NO_TLS" == "true" ]]; then
    print_dim "Skipping port checks (--no-tls)"
    return 0
  fi
  if ! command -v ss >/dev/null 2>&1; then
    print_warn "'ss' not found; skipping port-conflict check (install iproute2)."
    return 0
  fi
  local p
  for p in "${REQUIRED_PORTS[@]}"; do
    if ss -lnt "sport = :${p}" 2>/dev/null | tail -n +2 | grep -q .; then
      die "port ${p} is already in use; free it or pass --no-tls."
    fi
  done
  print_ok "Ports $(_ports_human) are free"
}

# ── Free-disk pre-flight ─────────────────────────────────────────────────────
#
# Each place the install writes to is a "need": POOL|LABEL|PATH, where POOL
# is storage (container images), install (--install-dir) or tmp (Podman's
# image pull staging dir). Needs whose paths resolve to the same file system
# are checked together against one requirement:
#   storage MIN_FREE_CONTAINER_STORAGE_GB (once, however many storage paths)
# + install MIN_FREE_INSTALL_DIR_GB
# + tmp     MIN_FREE_IMAGE_TMP_GB, only when nothing else is on that file
#           system (otherwise it is counted inside the budget above).
# So a flat / needs exactly 80 GB on every runtime, and a split layout needs
# 60 GB on the container storage mount, 20 GB on the install dir mount and,
# under podman-rootless, 5 GB on a separate /var/tmp mount.

# _canonical_path PATH → PATH with every symlink resolved, including a
# dangling one (GNU `realpath -m`: no component needs to exist), so an
# --install-dir symlink pointing at a mount that has no directory yet is
# measured where the data will land. PATH unchanged when `realpath -m` is not
# available (a non-GNU userland; _df_field needs GNU df anyway).
_canonical_path() {
  local resolved=""
  resolved="$(realpath -m -- "$1" 2>/dev/null || true)"
  printf '%s' "${resolved:-$1}"
}

# Nearest existing path at or above PATH (PATH itself when it exists). The
# pre-flight measures a directory that does not exist yet (the service user's
# home before useradd, a fresh --install-dir) on the file system that will
# hold it, and never creates anything.
_nearest_existing_path() {
  local path="$1"
  while [[ ! -e "$path" && "$path" != "/" && "$path" != "." ]]; do
    path="$(dirname "$path")"
  done
  printf '%s' "$path"
}

# _df_field PATH FIELD → one GNU df --output column for PATH, trimmed. avail
# is in df -BG units: GiB, rounded UP, "<n>G" with the G stripped. A reading
# of 60 can therefore mean up to 1 GiB less; the thresholds carry margin.
_df_field() {
  local path="$1" field="$2" value
  if [[ "$field" == "avail" ]]; then
    value="$(df -BG --output=avail "$path" 2>/dev/null | tail -n 1 | tr -dc '0-9' || true)"
  else
    value="$(df --output="$field" "$path" 2>/dev/null | tail -n 1 || true)"
    value="${value#"${value%%[![:space:]]*}"}"
    value="${value%"${value##*[![:space:]]}"}"
  fi
  printf '%s' "$value"
}

# Base directory `useradd -m` creates new homes under (HOME= in
# `useradd -D`, from /etc/default/useradd); /home when it cannot be read.
_useradd_home_base() {
  local base=""
  base="$(useradd -D 2>/dev/null | sed -n 's/^HOME=//p' | head -n 1 || true)"
  if [[ "$base" != /* ]]; then
    base="/home"
  fi
  printf '%s' "$base"
}

# Home of SERVICE_USER from getent (local files, SSSD, LDAP), or where
# ensure_service_user's `useradd -m` will create it when the account does not
# exist yet (the disk check runs before the account is created).
_service_user_home() {
  local home="" base
  home="$(getent passwd "$SERVICE_USER" 2>/dev/null | cut -d: -f6 || true)"
  if [[ -z "$home" ]]; then
    base="$(_useradd_home_base)"
    home="${base%/}/${SERVICE_USER}"
  fi
  printf '%s' "$home"
}

# _disk_needs → one POOL|LABEL|PATH line per place the install writes to.
#
# podman-rootless: image storage under the service user's home
# (~/.local/share/containers), and /var/tmp, where Podman stages layer
# downloads before committing them (containers.conf image_copy_tmp_dir,
# default /var/tmp; measured on Oracle Linux 9.8 / Podman 5.8.2 at ~580 MB
# for one 2 GB image, and compose pulls several in parallel). A non-default
# image_copy_tmp_dir or graphroot in containers.conf / storage.conf is not
# read: asking Podman would mean running it before it is installed or as
# the not-yet-created service user.
#
# docker: DockerRootDir from a running Docker Engine (asked under a 15 s
# timeout), else /var/lib/docker. With the containerd image store (the
# default for fresh installs since Docker Engine 29; `docker info` reports
# driver-type io.containerd.snapshotter.v1) image content lives in
# containerd's own root, /var/lib/containerd by default, which DockerRootDir
# does not move. That path is charged too whenever the store is in use or
# cannot be ruled out (no daemon yet, daemon down, `docker` is the Podman
# shim). The 60 GB storage budget is counted once per file system, so the
# conservative extra path only costs anything when it sits on a different
# file system from DockerRootDir, where both must then hold 60 GB.
_disk_needs() {
  if [[ "$RUNTIME" == "podman-rootless" ]]; then
    printf 'storage|rootless Podman container storage of %s|%s/.local/share/containers\n' \
      "$SERVICE_USER" "$(_service_user_home)"
    printf 'tmp|Podman image pull staging|/var/tmp\n'
  else
    local info="" root="" status="" containerd="true"
    if command -v docker >/dev/null 2>&1 && ! _docker_is_podman_shim; then
      if command -v timeout >/dev/null 2>&1; then
        info="$(timeout 15 docker info --format '{{.DockerRootDir}}|{{json .DriverStatus}}' 2>/dev/null || true)"
      else
        info="$(docker info --format '{{.DockerRootDir}}|{{json .DriverStatus}}' 2>/dev/null || true)"
      fi
      info="$(printf '%s' "$info" | head -n 1)"
      root="${info%%|*}"
      status="${info#*|}"
    fi
    if [[ "$root" == /* && "$info" == *"|"* ]]; then
      if [[ "$status" != *io.containerd.snapshotter* ]]; then
        containerd="false"
      fi
    else
      root="/var/lib/docker"
    fi
    printf 'storage|Docker data root|%s\n' "$root"
    if [[ "$containerd" == "true" ]]; then
      printf 'storage|containerd image store|/var/lib/containerd\n'
    fi
  fi
  printf 'install|install dir|%s\n' "$INSTALL_DIR"
}

# _disk_mount_advice POOL PATH → the "mount a volume" half of a failure.
_disk_mount_advice() {
  local pool="$1" path="$2" home
  case "$pool" in
    storage)
      if [[ "$RUNTIME" == "podman-rootless" ]]; then
        home="${path%/.local/share/containers}"
        printf 'mount a volume of at least %s GB at the home directory %s itself (then run '\''chown %s: %s'\'' and, on SELinux hosts such as Oracle Linux 9, '\''restorecon -R %s'\'': an unlabeled home makes rootless containers fail), or pass --service-user=NAME for an account whose home is on a larger file system' \
          "$MIN_FREE_CONTAINER_STORAGE_GB" "$home" "$SERVICE_USER" "$home" "$home"
      else
        printf 'mount a volume of at least %s GB at %s' "$MIN_FREE_CONTAINER_STORAGE_GB" "$path"
      fi
      ;;
    install)
      printf 'mount a volume of at least %s GB at %s (or pass a different --install-dir)' "$MIN_FREE_INSTALL_DIR_GB" "$path"
      ;;
    tmp)
      printf 'give %s at least %s GB (or set image_copy_tmp_dir in /etc/containers/containers.conf to a directory on a larger file system)' "$path" "$MIN_FREE_IMAGE_TMP_GB"
      ;;
  esac
}

# _check_disk_needs NEED... (POOL|LABEL|PATH lines, see _disk_needs)
#
# Test hook: main reaches it through check_disk_space with the real needs.
# Every path is canonicalized, then measured at its nearest existing
# ancestor. Two needs share a file system when df reports the same mount
# point, or the same /dev source (a bind mount or a btrfs subvolume shares
# the free space of its source). Every file system is reported before the
# run stops, so an operator with two undersized mounts fixes both in one pass.
_check_disk_needs() {
  local -a pools=() labels=() paths=() targets=() srcs=() avails=() groups=()
  local need_line rest measured target src avail
  local -i n=0 i j g
  for need_line in "$@"; do
    pools+=("${need_line%%|*}")
    rest="${need_line#*|}"
    labels+=("${rest%%|*}")
    paths+=("${rest#*|}")
    measured="$(_nearest_existing_path "$(_canonical_path "${paths[n]}")")"
    target="$(_df_field "$measured" target)"
    src="$(_df_field "$measured" source)"
    avail="$(_df_field "$measured" avail)"
    if [[ -z "$target" || ! "$avail" =~ ^[0-9]+$ ]]; then
      die "could not read free disk space for ${paths[n]} (df -BG --output=target,source,avail ${measured} failed; GNU coreutils df is required)."
    fi
    targets+=("$target")
    srcs+=("$src")
    avails+=("$avail")
    groups+=("$n")
    for (( j = 0; j < n; j++ )); do
      if [[ "${targets[j]}" == "$target" ]] \
        || [[ -n "$src" && "${srcs[j]}" == "$src" && "$src" == /dev/* ]]; then
        groups[n]="${groups[j]}"
        break
      fi
    done
    n=$(( n + 1 ))
  done

  local short="false" has_storage has_install storage_named need what advice part
  for (( g = 0; g < n; g++ )); do
    (( groups[g] == g )) || continue
    has_storage="false"; has_install="false"
    for (( i = 0; i < n; i++ )); do
      (( groups[i] == g )) || continue
      case "${pools[i]}" in
        storage) has_storage="true" ;;
        install) has_install="true" ;;
      esac
    done
    need=0
    [[ "$has_storage" == "true" ]] && need=$(( need + MIN_FREE_CONTAINER_STORAGE_GB ))
    [[ "$has_install" == "true" ]] && need=$(( need + MIN_FREE_INSTALL_DIR_GB ))
    (( need > 0 )) || need="$MIN_FREE_IMAGE_TMP_GB"

    what=""; advice=""; storage_named="false"
    for (( i = 0; i < n; i++ )); do
      (( groups[i] == g )) || continue
      case "${pools[i]}" in
        storage)
          if [[ "$storage_named" == "true" ]]; then
            part="${labels[i]} ${paths[i]} (same ${MIN_FREE_CONTAINER_STORAGE_GB} GB)"
          else
            part="${labels[i]} ${paths[i]} (${MIN_FREE_CONTAINER_STORAGE_GB} GB)"
            storage_named="true"
          fi
          advice+="${advice:+; or }$(_disk_mount_advice storage "${paths[i]}")"
          ;;
        install)
          part="${labels[i]} ${paths[i]} (${MIN_FREE_INSTALL_DIR_GB} GB)"
          advice+="${advice:+; or }$(_disk_mount_advice install "${paths[i]}")"
          ;;
        tmp)
          if [[ "$has_storage" == "true" || "$has_install" == "true" ]]; then
            part="${labels[i]} ${paths[i]} (within that budget)"
          else
            part="${labels[i]} ${paths[i]} (${MIN_FREE_IMAGE_TMP_GB} GB)"
            advice+="${advice:+; or }$(_disk_mount_advice tmp "${paths[i]}")"
          fi
          ;;
      esac
      what+="${what:+ + }${part}"
    done

    if (( avails[g] < need )); then
      print_error "free disk too low on the file system mounted at ${targets[g]}: ${avails[g]} GB free, need ${need} GB for ${what}. Fix: grow the file system mounted at ${targets[g]}; or ${advice}. Then re-run."
      short="true"
    else
      print_ok "Free disk on ${targets[g]}: ${avails[g]} GB (need ${need} GB: ${what})"
    fi
  done
  if [[ "$short" == "true" ]]; then
    die "free disk pre-flight failed (see above); nothing was changed and the license key was not used."
  fi
}

# Pre-flight free-disk check for the selected runtime and --install-dir.
# Runs inside check_resources, before the license handshake.
check_disk_space() {
  local -a needs=()
  local need_line
  while IFS= read -r need_line; do
    if [[ -n "$need_line" ]]; then
      needs+=("$need_line")
    fi
  done < <(_disk_needs)
  _check_disk_needs "${needs[@]}"
}

# "80, 443". Joined explicitly: `"${REQUIRED_PORTS[*]}"` joins with the first
# IFS character, which under our strict IFS is a newline and split the
# "Ports … are free" line in two.
_ports_human() {
  local joined
  joined="$(printf '%s, ' "${REQUIRED_PORTS[@]}")"
  printf '%s' "${joined%, }"
}

check_or_install_docker() {
  local need_install="false" docker_major
  if ! command -v docker >/dev/null 2>&1; then
    need_install="true"
  else
    docker_major="$(docker --version | awk '{print $3}' | cut -d. -f1)"
    if [[ -z "$docker_major" ]] || (( docker_major < MIN_DOCKER_MAJOR )); then
      print_warn "Docker is too old (need ${MIN_DOCKER_MAJOR}+; have ${docker_major:-?}). Will reinstall."
      need_install="true"
    fi
  fi
  if [[ "$need_install" == "false" ]] && ! docker compose version >/dev/null 2>&1; then
    print_warn "docker compose v2 not detected; will reinstall via the official Docker repo."
    need_install="true"
  fi

  if [[ "$need_install" == "false" ]]; then
    print_ok "Docker $(docker --version | awk '{print $3}' | tr -d ',') + Compose v2 present"
    return 0
  fi

  # Docker auto-install runs an arbitrary upstream script (`get.docker.com`)
  # as root with no signature verification. We treat opting into that as a
  # conscious choice: prompt in interactive mode, REQUIRE --auto-install-docker
  # in non-interactive mode. Otherwise direct the operator to the signed
  # apt/dnf repos via Docker's documented install page.
  local use_centos_repo="false"
  if _docker_needs_centos_repo; then
    use_centos_repo="true"
  fi

  if [[ "$AUTO_INSTALL_DOCKER" != "true" ]]; then
    if [[ "$NON_INTERACTIVE" == "true" ]]; then
      die "Docker 24+ with Compose v2 is required; auto-install is opt-in via --auto-install-docker. Install Docker per https://docs.docker.com/engine/install/ (signed apt/dnf repos) and re-run."
    fi
    if [[ "$use_centos_repo" == "true" ]]; then
      # TODO(OLA-1292): pin Docker's repo GPG key fingerprint instead of
      # trusting whatever the .repo file served over TLS points at.
      print_warn "get.sh can install Docker CE from Docker's CentOS package repository over HTTPS (no arbitrary upstream script is executed as root) because get.docker.com does not support this distro."
      if ! _confirm "Install Docker CE now from Docker's CentOS repository?" "y"; then
        die "aborted: install Docker per https://docs.docker.com/engine/install/ and re-run."
      fi
    else
      print_warn "get.sh can install Docker via https://get.docker.com — note that the upstream installer script is NOT signature-verified, so this extends the install's trust boundary to docker.com's HTTPS-served script."
      if ! _confirm "Install Docker now via the official get.docker.com installer?" "y"; then
        die "aborted: install Docker per https://docs.docker.com/engine/install/ and re-run."
      fi
    fi
  fi

  # Oracle Linux takes Docker's CentOS repository; every other distro stays
  # on the upstream installer below, unchanged.
  if [[ "$use_centos_repo" == "true" ]]; then
    _install_docker_from_centos_repo
    return 0
  fi

  print_info "Installing Docker via https://get.docker.com (may take a few minutes)…"
  if ! curl -fsSL https://get.docker.com -o /tmp/get-docker.sh; then
    die "failed to download https://get.docker.com — check network."
  fi
  # Defense-in-depth (OLA-301): run the upstream installer with stdin detached
  # from any pipe/terminal (< /dev/null) so it can never block on an
  # interactive read, and force apt/dnf into non-interactive mode so a
  # configuration prompt deep in the dependency chain can't stall the install.
  if ! DEBIAN_FRONTEND=noninteractive sh /tmp/get-docker.sh < /dev/null 3<&-; then
    rm -f /tmp/get-docker.sh
    die "Docker installation failed; install manually per https://docs.docker.com/engine/install/ and re-run."
  fi
  rm -f /tmp/get-docker.sh

  systemctl enable --now docker >/dev/null 2>&1 || true
  print_ok "Docker installed: $(docker --version | awk '{print $3}' | tr -d ',')"
}

# True when Docker must come from Docker's CentOS repository rather than
# get.docker.com. Upstream's dnf case arms are only centos|rhel|rocky and
# fedora.* (checked 2026-09-03), so Oracle Linux (ID="ol") fails there with
# "ERROR: Unsupported distribution 'ol'". Only ol is routed: an ID_LIKE-based
# guess misroutes Amazon Linux 2023 (ID_LIKE=fedora, $releasever=2023 makes
# the CentOS repo 404) and AlmaLinux, so anything else keeps the upstream
# path exactly as before.
_docker_needs_centos_repo() {
  case "$OS_ID" in
    ol) return 0 ;;
    *)  return 1 ;;
  esac
}

# Docker CE from Docker's CentOS repository, following Docker's documented
# RHEL/CentOS procedure (dnf-plugins-core → config-manager --add-repo →
# install). Verified 2026-09-03 on Oracle Linux 9.8: Docker 29.7.2 + Compose
# v5.5.1. Every dnf call is -y with stdin detached and the borrowed /dev/tty
# fd closed (3<&-, OLA-301) so nothing can prompt.
_install_docker_from_centos_repo() {
  local repo_url="https://download.docker.com/linux/centos/docker-ce.repo"
  print_warn "Docker CE on Oracle Linux is community-supported (neither Docker Inc. nor Oracle lists it). Oracle's supported container runtime is Podman."
  print_info "Installing Docker CE from ${repo_url} (may take a few minutes)…"
  if ! dnf -y install dnf-plugins-core < /dev/null 3<&-; then
    die "failed to install dnf-plugins-core (needed for 'dnf config-manager'); install Docker per https://docs.docker.com/engine/install/centos/ and re-run."
  fi
  if ! dnf -y config-manager --add-repo "$repo_url" < /dev/null 3<&-; then
    die "failed to add Docker's CentOS repository (${repo_url}) — check network."
  fi
  if ! dnf -y install docker-ce docker-ce-cli containerd.io docker-compose-plugin < /dev/null 3<&-; then
    die "Docker CE installation from ${repo_url} failed; install Docker per https://docs.docker.com/engine/install/centos/ and re-run."
  fi
  if ! systemctl enable --now docker >/dev/null 2>&1; then
    die "Docker CE installed but 'systemctl enable --now docker' failed; start the docker service and re-run."
  fi
  print_ok "Docker installed: $(docker --version | awk '{print $3}' | tr -d ',')"
}

# ──────────────────────────────────────────────────────────────────────────────
# Rootless Podman runtime (S4 / OLA-1294)
#
# `--runtime=podman-rootless` keeps root for the OS preparation and hands the
# bundle to an unprivileged service user. In-container root then maps to that
# user's subordinate uid range, so a container escape yields the service user,
# not root. Every step below mirrors the manual procedure proven live on
# Oracle Linux 9.8 (localnode-app onprem-bundle/README.md, "Running under
# Podman") and the olakai-ansible role's runtime_podman_rootless.yml. The
# bundle side reads `<install-dir>/.olakai-runtime` (scripts/lib/compose-files.sh)
# and refuses to run as root when it says podman-rootless.
#
# Several functions take optional path arguments. They exist for the test
# suite only (fixtures under a scratch dir); main calls every one of them
# bare. They are deliberately NOT environment variables: get.sh runs as root,
# and an env-controlled path would let the caller's environment redirect
# which files the installer reads and writes.
# ──────────────────────────────────────────────────────────────────────────────

# check_rootless_prereqs [MAX_USERNS_FILE] [SYSTEMD_RUN_DIR]
# shellcheck disable=SC2120  # the optional arguments are test hooks; main calls it bare
check_rootless_prereqs() {
  local max_userns_file="${1:-/proc/sys/user/max_user_namespaces}"
  local systemd_run_dir="${2:-/run/systemd/system}"
  local cgroup_fs max_userns

  # systemd: user units (podman.socket, podman-restart.service, the
  # healthcheck timers) and linger are what keep the stack alive with no
  # login. /run/systemd/system exists only when systemd is PID 1.
  if [[ ! -d "$systemd_run_dir" ]] || ! command -v systemctl >/dev/null 2>&1 \
      || ! command -v loginctl >/dev/null 2>&1; then
    die "--runtime=podman-rootless needs systemd (user units + linger); this host does not run it. Use --runtime=docker."
  fi
  # runuser is the root path's hand-off; --no-root never calls it, and
  # /usr/sbin (where Debian keeps it) is not on a non-root PATH.
  if [[ "$NO_ROOT" != "true" ]]; then
    command -v runuser >/dev/null 2>&1 \
      || die "--runtime=podman-rootless needs 'runuser' (util-linux) to run the bundle as ${SERVICE_USER}."
  fi

  # cgroup v2: rootless Podman needs the unified hierarchy for resource
  # delegation and healthchecks. OL9 / RHEL 9 / Ubuntu 22.04+ default to it.
  cgroup_fs="$(stat -fc %T /sys/fs/cgroup 2>/dev/null || true)"
  if [[ "$cgroup_fs" != "cgroup2fs" ]]; then
    die "--runtime=podman-rootless needs cgroup v2 (unified hierarchy); /sys/fs/cgroup is '${cgroup_fs:-absent}'. Boot with systemd.unified_cgroup_hierarchy=1 or use --runtime=docker."
  fi

  # User namespaces: the mechanism that maps in-container root to the
  # service user. 0 (or a missing file) means the kernel refuses them.
  max_userns="$(cat "$max_userns_file" 2>/dev/null || true)"
  if [[ ! "$max_userns" =~ ^[0-9]+$ ]] || (( max_userns == 0 )); then
    die "--runtime=podman-rootless needs unprivileged user namespaces; ${max_userns_file} is '${max_userns:-absent}' (need > 0). Set user.max_user_namespaces via sysctl or use --runtime=docker."
  fi

  print_ok "Rootless prerequisites: systemd, cgroup v2, user namespaces (max ${max_userns})"
}

# "podman version 5.8.2" → "5". Empty when podman is absent or the output is
# not in that shape.
_podman_major() {
  command -v podman >/dev/null 2>&1 || return 0
  # `|| true`: under pipefail a broken podman would otherwise make this the
  # function's exit status and stop the caller silently; the caller's regex
  # guard turns empty output into a clear message instead.
  podman --version 2>/dev/null | awk '{print $3}' | cut -d. -f1 || true
}

# True when the `docker` on PATH is the podman-docker shim, not Docker
# Engine. A real Docker next to Podman would make the bundle's scripts drive
# Docker while the marker says rootless Podman.
_docker_is_podman_shim() {
  command -v docker >/dev/null 2>&1 || return 1
  docker --version 2>/dev/null | grep -qi podman
}

_docker_version_human() {
  docker --version 2>/dev/null | head -n 1
}

# _die_docker_engine_beside_podman MAJOR: a real Docker next to Podman MAJOR.
_die_docker_engine_beside_podman() {
  die "Docker Engine is installed ('docker --version' reports '$(_docker_version_human)') next to Podman $1. The bundle's scripts call 'docker' and would drive Docker while the deployment is marked rootless Podman. Remove Docker Engine, or use --runtime=docker."
}

# _prepend_path DIR: DIR first on PATH for the rest of the run (and for the
# bundle's install.sh, which inherits it), unless it is there already.
_prepend_path() {
  case ":${PATH}:" in
    *":$1:"*) ;;
    *) export PATH="$1:${PATH}" ;;
  esac
}

# Major version apt would install for a package ("Candidate:" line of
# `apt-cache policy`), epoch stripped. Empty when the package is unknown or
# the candidate is "(none)".
_apt_candidate_major() {
  apt-cache policy "$1" 2>/dev/null \
    | awk '/^ *Candidate:/ {print $2}' \
    | sed -E 's/^[0-9]+://' \
    | grep -oE '^[0-9]+' || true
}

_podman_version_human() {
  podman --version 2>/dev/null | awk '{print $3}'
}

check_or_install_podman() {
  local major need_install="false"
  major="$(_podman_major)"
  if [[ ! "$major" =~ ^[0-9]+$ ]]; then
    need_install="true"
  elif (( major < MIN_PODMAN_MAJOR )); then
    print_warn "Podman is too old (need ${MIN_PODMAN_MAJOR}+; have ${major}). Will install from the distro repository."
    need_install="true"
  elif ! _docker_is_podman_shim; then
    # podman-docker provides the `docker` CLI shim the bundle's scripts call.
    if command -v docker >/dev/null 2>&1; then
      _die_docker_engine_beside_podman "$major"
    fi
    print_warn "Podman ${major} present but the podman-docker shim is missing; will install it."
    need_install="true"
  fi

  if [[ "$need_install" == "false" ]]; then
    print_ok "Podman $(_podman_version_human) + podman-docker present"
    return 0
  fi

  # Distro packages come from the distro's signed repositories, so unlike the
  # get.docker.com path there is no trust-boundary extension to opt into.
  # Every call is -y with stdin detached and the borrowed /dev/tty fd closed
  # (3<&-, OLA-301) so nothing can prompt.
  if command -v dnf >/dev/null 2>&1; then
    print_info "Installing podman + podman-docker via dnf (may take a few minutes)…"
    if ! dnf -y install podman podman-docker < /dev/null 3<&-; then
      die "Podman installation via dnf failed; install podman + podman-docker (${MIN_PODMAN_MAJOR}+) manually and re-run."
    fi
  elif command -v apt-get >/dev/null 2>&1; then
    # Check what apt WOULD install before installing anything, so a host
    # whose repository only offers Podman 4 is left untouched.
    if ! DEBIAN_FRONTEND=noninteractive apt-get update -y < /dev/null 3<&-; then
      die "apt-get update failed — check network and the apt sources."
    fi
    local candidate
    candidate="$(_apt_candidate_major podman)"
    if [[ ! "$candidate" =~ ^[0-9]+$ ]] || (( candidate < MIN_PODMAN_MAJOR )); then
      die "this host's apt repository offers Podman ${candidate:-(none)}; --runtime=podman-rootless needs ${MIN_PODMAN_MAJOR}+. Ubuntu 24.04 ships Podman 4.9 and is not qualified (qualified on Oracle Linux 9.8 (Podman 5.8.2); other RHEL 9 family distributions ship the same Podman 5 and are accepted but not qualified). Nothing was installed. Use --runtime=docker on this host."
    fi
    print_warn "Podman on Debian/Ubuntu is not qualified for --runtime=podman-rootless (qualified on Oracle Linux 9.8 (Podman 5.8.2); other RHEL 9 family distributions ship the same Podman 5 and are accepted but not qualified); proceeding with the repository's Podman ${candidate}."
    print_info "Installing podman + podman-docker + uidmap + dbus-user-session via apt-get (may take a few minutes)…"
    if ! DEBIAN_FRONTEND=noninteractive apt-get install -y podman podman-docker uidmap dbus-user-session < /dev/null 3<&-; then
      die "Podman installation via apt-get failed; install podman + podman-docker + uidmap + dbus-user-session (${MIN_PODMAN_MAJOR}+) manually and re-run."
    fi
  else
    die "neither dnf nor apt-get found; install Podman ${MIN_PODMAN_MAJOR}+ with podman-docker manually and re-run."
  fi

  major="$(_podman_major)"
  if [[ ! "$major" =~ ^[0-9]+$ ]] || (( major < MIN_PODMAN_MAJOR )); then
    die "Podman ${major:-?} installed but ${MIN_PODMAN_MAJOR}+ is required for --runtime=podman-rootless (Ubuntu 24.04 ships 4.9 and is not qualified; qualified on Oracle Linux 9.8 (Podman 5.8.2); other RHEL 9 family distributions ship the same Podman 5 and are accepted but not qualified). Use --runtime=docker on this host."
  fi
  command -v docker >/dev/null 2>&1 \
    || die "podman-docker installed but the 'docker' shim is not on PATH (${PATH}); re-run in a fresh root shell."
  if ! _docker_is_podman_shim; then
    die "Docker Engine is installed ('docker --version' reports '$(_docker_version_human)') and shadows the podman-docker shim. The bundle's scripts call 'docker' and would drive Docker while the deployment is marked rootless Podman. Remove Docker Engine, or use --runtime=docker."
  fi
  print_ok "Podman installed: $(_podman_version_human)"
}

# ensure_compose_provider [BIN_DIR] [CONTAINERS_ETC]
#
# Installs the pinned docker-compose binary (same download + SHA256 pattern as
# ensure_cosign) and names it as Podman's compose provider. The explicit
# provider path matters: `podman compose` looks the provider up on PATH, and
# the RHEL family's sudo secure_path lacks /usr/local/bin (the trap that
# bit cosign, OLA-1292), so under plain sudo the lookup finds nothing. The
# containers.conf.d drop-in applies to every user, including the service
# user. /etc/containers/nodocker silences the shim's "emulate Docker" notice
# on every `docker` call (the shim looks at that one path only; the marker
# written under a user-scope CONTAINERS_ETC is inert and harmless).
# shellcheck disable=SC2120  # the optional arguments are test hooks; main calls it bare
ensure_compose_provider() {
  local bin_dir="${1:-/usr/local/bin}" containers_etc="${2:-/etc/containers}"
  local target="${bin_dir}/docker-compose"
  local arch expected_sha url tmp_path actual_sha
  case "$(uname -m)" in
    x86_64|amd64)  arch="x86_64";  expected_sha="$COMPOSE_SHA256_LINUX_X86_64" ;;
    aarch64|arm64) arch="aarch64"; expected_sha="$COMPOSE_SHA256_LINUX_AARCH64" ;;
    *) die "unsupported CPU arch '$(uname -m)'; docker-compose builds for x86_64 / aarch64 only." ;;
  esac

  if [[ -x "$target" ]] && [[ "$(sha256sum "$target" | awk '{print $1}')" == "$expected_sha" ]]; then
    print_ok "docker-compose ${COMPOSE_VERSION} present at ${target} (SHA256 matches)"
  else
    url="https://github.com/docker/compose/releases/download/${COMPOSE_VERSION}/docker-compose-linux-${arch}"
    tmp_path="$(mktemp)"
    print_info "Downloading docker-compose ${COMPOSE_VERSION} (${arch})…"
    if ! curl -fsSL -o "$tmp_path" "$url"; then
      rm -f "$tmp_path"
      die "failed to download docker-compose from ${url}"
    fi
    actual_sha="$(sha256sum "$tmp_path" | awk '{print $1}')"
    if [[ "$actual_sha" != "$expected_sha" ]]; then
      rm -f "$tmp_path"
      die "docker-compose SHA256 mismatch (expected ${expected_sha}, got ${actual_sha}). Refusing to install — possible MITM."
    fi
    print_ok "docker-compose SHA256 matches embedded constant"
    install -m 0755 "$tmp_path" "$target"
    rm -f "$tmp_path"
    print_ok "docker-compose installed at ${target}"
  fi

  mkdir -p "${containers_etc}/containers.conf.d"
  printf '# Written by get.sh (--runtime=podman-rootless). Names the compose provider\n# by absolute path: sudo secure_path on the RHEL family lacks /usr/local/bin.\n[engine]\ncompose_providers = ["%s"]\ncompose_warning_logs = false\n' \
    "$target" > "${containers_etc}/containers.conf.d/olakai-compose.conf"
  : > "${containers_etc}/nodocker"
  print_ok "Compose provider registered in ${containers_etc}/containers.conf.d/olakai-compose.conf"
}

# Prints the well-formed entries of a subuid/subgid file as `name:start:count`,
# one per line. Blank lines, comments and malformed lines are skipped, and so
# are numbers with leading zeros (bash arithmetic would read them as octal).
# A missing file yields nothing (Ubuntu minimal images may lack it until
# uidmap is installed).
#
# Entries are matched by NAME only, in both files. A numeric first field is
# legal there but means a uid in /etc/subuid and a GID in /etc/subgid, so
# treating `987:…` in subgid as the service user's own range would adopt a
# stranger's. useradd/usermod and the olakai-ansible role write names.
_subid_entries() {
  local file="$1" line name start count
  [[ -r "$file" ]] || return 0
  while IFS= read -r line || [[ -n "$line" ]]; do
    line="${line%%#*}"
    line="${line//[[:space:]]/}"
    [[ -n "$line" ]] || continue
    IFS=: read -r name start count <<<"$line"
    if [[ -n "$name" && "$start" =~ ^(0|[1-9][0-9]*)$ && "$count" =~ ^(0|[1-9][0-9]*)$ ]]; then
      printf '%s:%s:%s\n' "$name" "$start" "$count"
    fi
  done < "$file"
}

# _own_subid_entry USER FILE → prints `start:count` of USER's own entry
# (matched by name), empty when there is none.
_own_subid_entry() {
  local user="$1" file="$2" entry name start count
  while IFS= read -r entry; do
    [[ -n "$entry" ]] || continue
    IFS=: read -r name start count <<<"$entry"
    if [[ "$name" == "$user" ]]; then
      printf '%s:%s\n' "$start" "$count"
      return 0
    fi
  done < <(_subid_entries "$file")
  return 0
}

# _subid_conflicts START COUNT USER FILE... → prints every entry not named
# USER whose range overlaps [START, START+COUNT-1], as `file: entry`.
_subid_conflicts() {
  local start="$1" count="$2" user="$3"
  shift 3
  local end=$(( start + count - 1 )) file entry name o_start o_count o_end
  for file in "$@"; do
    while IFS= read -r entry; do
      [[ -n "$entry" ]] || continue
      IFS=: read -r name o_start o_count <<<"$entry"
      [[ "$name" == "$user" ]] && continue
      o_end=$(( o_start + o_count - 1 ))
      if (( o_start <= end && o_end >= start )); then
        printf '%s: %s\n' "$file" "$entry"
      fi
    done < <(_subid_entries "$file")
  done
}

# _die_prepared ROOT_MESSAGE ADMIN_MESSAGE: the root path fixes the host
# itself or names the fix as a plain command; under --no-root the message
# must end with the sudo one-liner an admin runs (D-4). Root-path text is
# unchanged.
_die_prepared() {
  if [[ "$NO_ROOT" == "true" ]]; then die "$2"; else die "$1"; fi
}

# _free_subid_start USER SUBUID_FILE SUBGID_FILE → the first SUBID_COUNT-wide
# slot at or above SUBID_START_MIN that overlaps nobody else's entry (the
# user's own entries are ignored), or nothing after SUBID_MAX_PROBES tries.
_free_subid_start() {
  local user="$1" subuid_file="$2" subgid_file="$3" candidate i
  candidate="$SUBID_START_MIN"
  for (( i = 0; i < SUBID_MAX_PROBES; i++ )); do
    if [[ -z "$(_subid_conflicts "$candidate" "$SUBID_COUNT" "$user" "$subuid_file" "$subgid_file")" ]]; then
      printf '%s' "$candidate"
      return 0
    fi
    candidate=$(( candidate + SUBID_COUNT ))
  done
}

# _subid_admin_fix USER SUBUID_FILE SUBGID_FILE OWN_UID OWN_GID MODE → the
# "Run as an admin: sudo usermod …" tail of a --no-root refusal. OWN_* are
# the user's entries as `start:count` (empty when absent). MODE `add` adds
# only the missing half (usermod refuses to re-add an existing range) at the
# start the other half already uses, or at a free start when both are
# missing; MODE `replace` deletes the existing entries and adds a free
# range in both files. When no free start can be computed (unreadable files,
# every probe taken) the default start is proposed with a note.
_subid_admin_fix() {
  local user="$1" subuid_file="$2" subgid_file="$3" own_uid="$4" own_gid="$5" mode="$6"
  local start="" end note="" flags="" s c
  if [[ "$mode" == "add" && -n "${own_uid}${own_gid}" ]]; then
    start="${own_uid:-$own_gid}"
    start="${start%%:*}"
  else
    start="$(_free_subid_start "$user" "$subuid_file" "$subgid_file")"
  fi
  if [[ -z "$start" ]]; then
    start="$SUBID_START_MIN"
    note=" (no free range was found in ${subuid_file} / ${subgid_file}; adjust the range if it is taken)"
  elif [[ ! -r "$subuid_file" || ! -r "$subgid_file" ]]; then
    note=" (${subuid_file} or ${subgid_file} is not readable here; adjust the range if it is taken)"
  fi
  end=$(( start + SUBID_COUNT - 1 ))
  if [[ "$mode" == "replace" ]]; then
    if [[ -n "$own_uid" ]]; then
      s="${own_uid%%:*}"; c="${own_uid##*:}"
      flags+="--del-subuids ${s}-$(( s + c - 1 )) "
    fi
    if [[ -n "$own_gid" ]]; then
      s="${own_gid%%:*}"; c="${own_gid##*:}"
      flags+="--del-subgids ${s}-$(( s + c - 1 )) "
    fi
    flags+="--add-subuids ${start}-${end} --add-subgids ${start}-${end}"
  else
    [[ -z "$own_uid" ]] && flags+="--add-subuids ${start}-${end} "
    [[ -z "$own_gid" ]] && flags+="--add-subgids ${start}-${end} "
    flags="${flags% }"
  fi
  printf 'Run as an admin: sudo usermod %s %s%s' "$flags" "$user" "$note"
}

# _pick_subid_start USER SUBUID_FILE SUBGID_FILE
#
# Sets SUBID_START_OUT / SUBID_COUNT_OUT. A range the user already owns is
# reused (re-run safety) after checking it is wide enough and overlaps
# nobody else's; otherwise the first free SUBID_COUNT-wide slot at or above
# SUBID_START_MIN wins. An existing overlap or a narrow range for the service
# user is fatal, not fixed: two users sharing a range can read and write each
# other's container files, and silently moving or widening the range would
# orphan any storage the user already has. Under --no-root the refusal ends
# with the usermod one-liner (_subid_admin_fix), which the admin applies
# knowing the host: on a first install there is no storage to orphan.
_pick_subid_start() {
  local user="$1" subuid_file="$2" subgid_file="$3"
  local own_uid own_gid own start count conflicts candidate
  own_uid="$(_own_subid_entry "$user" "$subuid_file")"
  own_gid="$(_own_subid_entry "$user" "$subgid_file")"
  if [[ -n "$own_uid" && -n "$own_gid" && "$own_uid" != "$own_gid" ]]; then
    _die_prepared "subuid (${own_uid}) and subgid (${own_gid}) ranges for ${user} differ; make ${subuid_file} and ${subgid_file} agree and re-run." \
      "subuid (${own_uid}) and subgid (${own_gid}) ranges for ${user} differ; --no-root does not edit ${subuid_file} / ${subgid_file}. $(_subid_admin_fix "$user" "$subuid_file" "$subgid_file" "$own_uid" "$own_gid" replace)"
  fi
  own="${own_uid:-$own_gid}"
  if [[ -n "$own" ]]; then
    start="${own%%:*}"
    count="${own##*:}"
    if (( count < SUBID_COUNT )); then
      _die_prepared "the subordinate id range ${start}:${count} already assigned to ${user} is narrower than ${SUBID_COUNT} ids (images with many uids fail to unpack). Fix ${subuid_file} and ${subgid_file} by hand (widen it to ${user}:${start}:${SUBID_COUNT} if those ids are free, or assign a free ${SUBID_COUNT}-wide range) and re-run." \
        "the subordinate id range ${start}:${count} already assigned to ${user} is narrower than ${SUBID_COUNT} ids (images with many uids fail to unpack); --no-root does not edit ${subuid_file} / ${subgid_file}. $(_subid_admin_fix "$user" "$subuid_file" "$subgid_file" "$own_uid" "$own_gid" replace)"
    fi
    conflicts="$(_subid_conflicts "$start" "$count" "$user" "$subuid_file" "$subgid_file")"
    if [[ -n "$conflicts" ]]; then
      _die_prepared "the subordinate id range ${start}:${count} already assigned to ${user} overlaps another user's entry ($(printf '%s' "$conflicts" | tr '\n' ';')). Two users sharing a range can read and write each other's container files; fix ${subuid_file} / ${subgid_file} by hand and re-run." \
        "the subordinate id range ${start}:${count} already assigned to ${user} overlaps another user's entry ($(printf '%s' "$conflicts" | tr '\n' ';')). Two users sharing a range can read and write each other's container files; --no-root does not edit ${subuid_file} / ${subgid_file}. $(_subid_admin_fix "$user" "$subuid_file" "$subgid_file" "$own_uid" "$own_gid" replace)"
    fi
    SUBID_START_OUT="$start"
    SUBID_COUNT_OUT="$count"
    return 0
  fi

  start="$(_free_subid_start "$user" "$subuid_file" "$subgid_file")"
  if [[ -n "$start" ]]; then
    SUBID_START_OUT="$start"
    SUBID_COUNT_OUT="$SUBID_COUNT"
    return 0
  fi
  candidate=$(( SUBID_START_MIN + SUBID_MAX_PROBES * SUBID_COUNT ))
  _die_prepared "no free ${SUBID_COUNT}-wide subordinate id range found for ${user} between ${SUBID_START_MIN} and ${candidate} in ${subuid_file} / ${subgid_file}; free one by hand and re-run." \
    "no free ${SUBID_COUNT}-wide subordinate id range found for ${user} between ${SUBID_START_MIN} and ${candidate} in ${subuid_file} / ${subgid_file}; --no-root does not edit them. $(_subid_admin_fix "$user" "$subuid_file" "$subgid_file" "" "" add)"
}

# _ensure_subid_line USER START COUNT FILE: appends `USER:START:COUNT`
# unless the user already has an entry (validated by _pick_subid_start).
_ensure_subid_line() {
  local user="$1" start="$2" count="$3" file="$4"
  if [[ -n "$(_own_subid_entry "$user" "$file")" ]]; then
    return 0
  fi
  if [[ ! -f "$file" ]]; then
    : > "$file"
    chmod 644 "$file"
  elif [[ -s "$file" && -n "$(tail -c 1 "$file")" ]]; then
    # No trailing newline: appending would glue our entry onto the last line.
    printf '\n' >> "$file"
  fi
  printf '%s:%s:%s\n' "$user" "$start" "$count" >> "$file"
}

# ensure_service_user [SUBUID_FILE] [SUBGID_FILE]
# shellcheck disable=SC2120  # the optional arguments are test hooks; main calls it bare
ensure_service_user() {
  local subuid_file="${1:-/etc/subuid}" subgid_file="${2:-/etc/subgid}"

  if getent passwd "$SERVICE_USER" >/dev/null 2>&1; then
    print_ok "Service user ${SERVICE_USER} exists"
  else
    print_info "Creating service user ${SERVICE_USER}…"
    # -r: system account (no aging, uid from the system range). -m: a home
    # is required — rootless Podman keeps its storage and config there.
    # /bin/bash: not needed by the hand-off (`runuser -- CMD` execs CMD
    # directly, no login shell); kept so an operator can debug as the
    # service user with `sudo -u olakai bash`.
    if ! useradd -r -m -s /bin/bash "$SERVICE_USER"; then
      die "useradd ${SERVICE_USER} failed; create the user by hand (useradd -r -m -s /bin/bash ${SERVICE_USER}) and re-run."
    fi
    print_ok "Service user ${SERVICE_USER} created"
  fi

  SERVICE_UID="$(id -u "$SERVICE_USER" 2>/dev/null || true)"
  if [[ ! "$SERVICE_UID" =~ ^[0-9]+$ ]]; then
    die "could not resolve the uid of ${SERVICE_USER} (id -u returned '${SERVICE_UID}')."
  fi
  if (( SERVICE_UID == 0 )); then
    die "${SERVICE_USER} resolves to uid 0; --runtime=podman-rootless needs an unprivileged user."
  fi

  _check_service_user_account

  _pick_subid_start "$SERVICE_USER" "$subuid_file" "$subgid_file"
  _ensure_subid_line "$SERVICE_USER" "$SUBID_START_OUT" "$SUBID_COUNT_OUT" "$subuid_file"
  _ensure_subid_line "$SERVICE_USER" "$SUBID_START_OUT" "$SUBID_COUNT_OUT" "$subgid_file"
  print_ok "Subordinate ids for ${SERVICE_USER}: ${SUBID_START_OUT}:${SUBID_COUNT_OUT} (${subuid_file}, ${subgid_file})"

  # Linger starts user@<uid>.service at boot and keeps it after logout, so
  # the user's podman.socket, podman-restart.service and the healthcheck
  # timers run with nobody logged in.
  if ! loginctl enable-linger "$SERVICE_USER"; then
    die "loginctl enable-linger ${SERVICE_USER} failed; the stack would stop at logout. Fix systemd-logind and re-run."
  fi
  print_ok "systemd linger enabled for ${SERVICE_USER}"
}

# Primary group and home of SERVICE_USER, checked pre-handshake rather than
# discovered at the hand-off after the key is spent. A pre-existing account
# passed via --service-user may have a primary group that is not named after
# the user (chown uses `user:`, the primary group, for that reason) or no
# usable home; rootless Podman keeps its storage and configuration under the
# home, and runuser fails opaquely without one. Shared by ensure_service_user
# (root path) and check_prepared_host (--no-root); needs SERVICE_UID set.
_check_service_user_account() {
  local group home home_owner
  group="$(id -gn "$SERVICE_USER" 2>/dev/null || true)"
  if [[ -z "$group" ]] || ! getent group "$group" >/dev/null 2>&1; then
    _die_prepared "the primary group of ${SERVICE_USER} ('${group:-?}') does not resolve; fix the account (usermod -g <group> ${SERVICE_USER}) and re-run." \
      "the primary group of ${SERVICE_USER} ('${group:-?}') does not resolve; --no-root cannot fix the account. Run as an admin: sudo groupadd ${SERVICE_USER} && sudo usermod -g ${SERVICE_USER} ${SERVICE_USER}"
  fi
  home="$(getent passwd "$SERVICE_USER" 2>/dev/null | cut -d: -f6 || true)"
  if [[ -z "$home" ]]; then
    _die_prepared "the passwd entry of ${SERVICE_USER} has no home directory; rootless Podman keeps its storage there. Set one (usermod -d /home/${SERVICE_USER} -m ${SERVICE_USER}) and re-run." \
      "the passwd entry of ${SERVICE_USER} has no home directory; rootless Podman keeps its storage there. Run as an admin: sudo usermod -d /home/${SERVICE_USER} -m ${SERVICE_USER}"
  fi
  if [[ ! -d "$home" ]]; then
    _die_prepared "the home directory of ${SERVICE_USER}, ${home}, is missing or not a directory; rootless Podman keeps its storage there. Create it (mkdir -p ${home} && chown ${SERVICE_USER}: ${home}) and re-run." \
      "the home directory of ${SERVICE_USER}, ${home}, is missing or not a directory; rootless Podman keeps its storage there. Run as an admin: sudo mkdir -p ${home} && sudo chown ${SERVICE_USER}: ${home}"
  fi
  home_owner="$(stat -c %U "$home" 2>/dev/null || true)"
  if [[ "$home_owner" != "$SERVICE_USER" ]]; then
    _die_prepared "the home directory of ${SERVICE_USER}, ${home}, is owned by '${home_owner:-?}'; Podman cannot use it. Run 'chown ${SERVICE_USER}: ${home}' and re-run." \
      "the home directory of ${SERVICE_USER}, ${home}, is owned by '${home_owner:-?}'; Podman cannot use it. Run as an admin: sudo chown ${SERVICE_USER}: ${home}"
  fi
  print_ok "Service user ${SERVICE_USER}: uid ${SERVICE_UID}, group ${group}, home ${home}"
}

# Under --no-root the run writes under $HOME (cosign and docker-compose in
# $HOME/.local/bin, the compose provider in $XDG_CONFIG_HOME or
# $HOME/.config). These are the file's only env-derived paths, and they are
# acceptable exactly because they are what Podman itself keys on for its
# storage and its user-scope containers.conf.d: an env value here chooses
# nothing the caller could not already choose for Podman (unlike detect_os,
# where an env path would pick what the installer sources as root). The
# passwd home must agree with $HOME: Podman's user session uses the passwd
# one, and a `su`/`sudo -u` shell can carry someone else's HOME.
_check_no_root_home() {
  local home="${HOME:-}" passwd_home
  if [[ -z "$home" ]]; then
    die "HOME is not set; --no-root installs under \$HOME/.local/bin and \$HOME/.config. Log in again as ${SERVICE_USER} (a real login session) and re-run."
  fi
  passwd_home="$(getent passwd "$SERVICE_USER" 2>/dev/null | cut -d: -f6 || true)"
  if [[ -n "$passwd_home" && "$passwd_home" != "$home" ]]; then
    die "HOME is ${home} but the passwd entry of ${SERVICE_USER} says ${passwd_home}; Podman keys its storage on the passwd home. Log in again as ${SERVICE_USER} (a real login session, not su or sudo -u) and re-run."
  fi
  print_ok "HOME ${home} matches the passwd entry of ${SERVICE_USER}"
}

# ensure_unprivileged_ports [SYSCTL_DIR]
#
# Host-wide: every unprivileged process may then bind ports 80-1023. Always
# applied: the Caddy overlay binds 80/443 and the base file binds the app on
# 80 without it, so skipping it guarantees a failed first `up -d`. A
# load-balancer topology (Caddy remapped to 8080/8443 behind a customer LB,
# no sysctl) is not an installer mode (D-007: single VM, direct ingress).
# shellcheck disable=SC2120  # the optional argument is a test hook; main calls it bare
ensure_unprivileged_ports() {
  local sysctl_dir="${1:-/etc/sysctl.d}"
  local conf="${sysctl_dir}/90-olakai.conf"
  mkdir -p "$sysctl_dir"
  printf '# Written by get.sh (--runtime=podman-rootless): lets the unprivileged service\n# user bind ports 80 and 443 for Caddy. Host-wide setting.\nnet.ipv4.ip_unprivileged_port_start=80\n' > "$conf"
  if ! sysctl --system >/dev/null; then
    die "sysctl --system failed after writing ${conf}; fix the sysctl configuration and re-run."
  fi
  print_ok "net.ipv4.ip_unprivileged_port_start=80 (${conf})"
}

# Runs a command as the owner of the container runtime: directly under
# Docker, as the service user under rootless Podman. `runuser` (not `su`,
# not `sudo`) so no PAM session or password path is involved; the user's
# XDG_RUNTIME_DIR and D-Bus address are what `systemctl --user` and Podman's
# socket lookup key on, and a runuser shell does not set them. PATH is passed
# explicitly so the cosign dir ensure_cosign prepended reaches the bundle's
# install.sh. HOME is set by runuser itself. Under --no-root we already are
# the service user: same environment, no runuser.
_runtime_exec() {
  if [[ "$RUNTIME" != "podman-rootless" ]]; then
    "$@"
    return $?
  fi
  if [[ ! "$SERVICE_UID" =~ ^[0-9]+$ ]]; then
    die "service user uid not resolved (expected ensure_service_user, or check_root under --no-root, to set SERVICE_UID; have '${SERVICE_UID}')."
  fi
  if [[ "$NO_ROOT" == "true" ]]; then
    env \
      "XDG_RUNTIME_DIR=/run/user/${SERVICE_UID}" \
      "DBUS_SESSION_BUS_ADDRESS=unix:path=/run/user/${SERVICE_UID}/bus" \
      "PATH=${PATH}" \
      "$@"
    return $?
  fi
  runuser -u "$SERVICE_USER" -- env \
    "XDG_RUNTIME_DIR=/run/user/${SERVICE_UID}" \
    "DBUS_SESSION_BUS_ADDRESS=unix:path=/run/user/${SERVICE_UID}/bus" \
    "PATH=${PATH}" \
    "$@"
}

# ensure_user_podman_units [RUN_USER_DIR]
#
# The socket is what the support-bundle sidecar mounts and what `docker
# compose` (via podman compose) talks to; podman-restart.service brings the
# containers back after a reboot. Both are user-scope units of the service
# user. The optional argument relocates only the wait-for-files probe (bus,
# socket); the argv handed to runuser always names the real /run/user path.
# shellcheck disable=SC2120  # the optional argument is a test hook; main calls it bare
ensure_user_podman_units() {
  local run_user_dir="${1:-/run/user}"
  local runtime_dir="${run_user_dir}/${SERVICE_UID}"
  local sock="${runtime_dir}/podman/podman.sock" i
  # Recovery advice: the runuser form from a root shell, or the plain
  # systemctl --user form when the operator is the service user (--no-root),
  # gated like print_success_banner's as_user.
  local by_hand="runuser -u ${SERVICE_USER} -- env XDG_RUNTIME_DIR=/run/user/${SERVICE_UID} systemctl --user …"
  local status_cmd="runuser -u ${SERVICE_USER} -- env XDG_RUNTIME_DIR=/run/user/${SERVICE_UID} systemctl --user status podman.socket"
  if [[ "$NO_ROOT" == "true" ]]; then
    by_hand="systemctl --user enable --now podman.socket podman-restart.service"
    status_cmd="systemctl --user status podman.socket"
  fi

  # enable-linger starts user@<uid>.service asynchronously; its bus is what
  # `systemctl --user` connects to. 6 checks, 5 sleeps x 5s = 25s budget.
  for i in 1 2 3 4 5 6; do
    [[ -S "${runtime_dir}/bus" ]] && break
    if (( i < 6 )); then sleep 5; fi
  done
  if [[ ! -S "${runtime_dir}/bus" ]]; then
    die "the systemd user manager for ${SERVICE_USER} (uid ${SERVICE_UID}) did not come up: ${runtime_dir}/bus is missing after 25s. Check 'loginctl enable-linger ${SERVICE_USER}' and 'systemctl status user@${SERVICE_UID}', then re-run."
  fi

  if ! _runtime_exec systemctl --user enable --now podman.socket podman-restart.service; then
    die "'systemctl --user enable --now podman.socket podman-restart.service' failed as ${SERVICE_USER}. Run it by hand as that user (${by_hand}) and re-run."
  fi

  # 6 checks, 5 sleeps x 2s = 10s budget for the socket to appear.
  for i in 1 2 3 4 5 6; do
    [[ -S "$sock" ]] && break
    if (( i < 6 )); then sleep 2; fi
  done
  if [[ ! -S "$sock" ]]; then
    die "${sock} did not appear within 10s. The user podman.socket is enabled but not listening: check 'loginctl enable-linger ${SERVICE_USER}' (without linger the unit stops at logout) and '${status_cmd}', then re-run."
  fi
  print_ok "User-scope podman.socket + podman-restart.service enabled for ${SERVICE_USER} (${sock})"
}

# Hands the extracted, root-written install dir to the service user: the
# runtime marker the bundle's scripts read, the Podman overlay they
# auto-detect, ownership and a 0750 mode. Called from bring_up after the
# Caddy overlay copy and .env materialization, so everything root wrote in
# this run is covered.
#
# `data/` is left alone on purpose. On a --force-overlay re-run it holds the
# containers' state, owned by ids inside the subordinate range (postgres's
# files belong to <start>+69, for instance); a recursive chown to the service
# user would break the user-namespace ownership the stack relies on. On a
# fresh install the directory does not exist yet: install.sh creates it as
# the service user.
_stage_rootless_install_dir() {
  local marker="${INSTALL_DIR}/.olakai-runtime"
  local overlay_example="${INSTALL_DIR}/examples/docker-compose.podman.yml"
  local overlay_active="${INSTALL_DIR}/docker-compose.podman.yml"
  local entry

  # check_bundle_runtime_support verified the overlay exists right after
  # extraction, before anything was written.
  printf 'podman-rootless\n' > "$marker"
  if [[ ! -f "$overlay_active" ]]; then
    cp "$overlay_example" "$overlay_active"
    print_ok "Activated Podman overlay (docker-compose.podman.yml)."
  fi

  # Under --no-root the invoking user wrote the tree; there is nothing to
  # hand over and chown would need root.
  if [[ "$NO_ROOT" == "true" ]]; then
    chmod 750 "$INSTALL_DIR"
    print_ok "${INSTALL_DIR} owned by ${SERVICE_USER} (mode 0750, .olakai-runtime=podman-rootless)"
    return 0
  fi

  # Dotfiles included (.env, .olakai-runtime); unmatched globs stay literal
  # and fail the -e test, so no nullglob dance is needed. `user:` (trailing
  # colon) is the user's PRIMARY group, whatever its name: a pre-existing
  # --service-user account need not have a same-named group
  # (ensure_service_user verified the group resolves, pre-handshake).
  for entry in "$INSTALL_DIR"/* "$INSTALL_DIR"/.[!.]* "$INSTALL_DIR"/..?*; do
    [[ -e "$entry" || -L "$entry" ]] || continue
    [[ "${entry##*/}" == "data" ]] && continue
    chown -R "${SERVICE_USER}:" "$entry"
  done
  chown "${SERVICE_USER}:" "$INSTALL_DIR"
  chmod 750 "$INSTALL_DIR"
  print_ok "${INSTALL_DIR} handed to ${SERVICE_USER} (mode 0750, .olakai-runtime=podman-rootless)"
}

# ──────────────────────────────────────────────────────────────────────────────
# Prepared host (--no-root, OLA-1443)
#
# The root path above PERFORMS the host preparation; this path VERIFIES it
# as the service user and installs nothing system-wide. Each check dies with
# the one root command an admin runs to fix it (D-4). Everything here runs
# before the license handshake. No podman command that creates state is run
# in pre-flight: a first `podman info` as the user is what initialises its
# storage, so the subordinate id check is file-based.
# ──────────────────────────────────────────────────────────────────────────────

# The distro's package command for Podman + the podman-docker shim.
_podman_install_hint() {
  if command -v dnf >/dev/null 2>&1; then
    printf 'sudo dnf -y install podman podman-docker'
  else
    printf 'sudo apt-get install -y podman podman-docker uidmap dbus-user-session'
  fi
}

_check_prepared_podman() {
  local major hint
  hint="$(_podman_install_hint)"
  major="$(_podman_major)"
  if [[ ! "$major" =~ ^[0-9]+$ ]]; then
    die "Podman is not installed (or 'podman --version' is unreadable); --no-root does not install packages. Run as an admin: ${hint}"
  fi
  if (( major < MIN_PODMAN_MAJOR )); then
    die "Podman ${major} is too old (need ${MIN_PODMAN_MAJOR}+); --no-root does not install packages. Run as an admin: ${hint}"
  fi
  if ! _docker_is_podman_shim; then
    if command -v docker >/dev/null 2>&1; then
      _die_docker_engine_beside_podman "$major"
    fi
    die "Podman ${major} present but the podman-docker shim ('docker') is missing; --no-root does not install packages. Run as an admin: ${hint}"
  fi
  print_ok "Podman $(_podman_version_human) + podman-docker present"
}

# _check_prepared_subids SUBUID_FILE SUBGID_FILE: the invoking user needs a
# SUBID_COUNT-wide entry, by name, in both files. `useradd -m` on OL9 /
# RHEL 9 allocates one; a hand-made account may lack it. Only the missing
# half is proposed (usermod refuses to re-add a range the user has), at the
# start the other half already uses.
_check_prepared_subids() {
  local subuid_file="$1" subgid_file="$2" own_uid own_gid where=""
  own_uid="$(_own_subid_entry "$SERVICE_USER" "$subuid_file")"
  own_gid="$(_own_subid_entry "$SERVICE_USER" "$subgid_file")"
  # Reuses the root path's validation: differing, narrow or overlapping
  # ranges are refused there, under --no-root with the usermod one-liner.
  _pick_subid_start "$SERVICE_USER" "$subuid_file" "$subgid_file"
  if [[ -n "$own_uid" && -n "$own_gid" ]]; then
    print_ok "Subordinate ids for ${SERVICE_USER}: ${SUBID_START_OUT}:${SUBID_COUNT_OUT} (${subuid_file}, ${subgid_file})"
    return 0
  fi
  [[ -z "$own_uid" ]] && where="$subuid_file"
  [[ -z "$own_gid" ]] && where="${where:+${where} and }${subgid_file}"
  die "${SERVICE_USER} has no subordinate id entry in ${where} (need ${SUBID_COUNT} ids in ${subuid_file} and ${subgid_file}); --no-root does not edit them. $(_subid_admin_fix "$SERVICE_USER" "$subuid_file" "$subgid_file" "$own_uid" "$own_gid" add)"
}

# Linger keeps the user's podman.socket, podman-restart.service and the
# healthcheck timers running with nobody logged in. `loginctl enable-linger`
# with no argument acts on the caller, which polkit allows for an active
# session by default; when it does not, the admin runs it.
_check_prepared_linger() {
  local linger
  linger="$(loginctl show-user "$SERVICE_USER" -p Linger --value 2>/dev/null || true)"
  if [[ "$linger" == "yes" ]]; then
    print_ok "systemd linger enabled for ${SERVICE_USER}"
    return 0
  fi
  if loginctl enable-linger >/dev/null 2>&1; then
    print_ok "systemd linger enabled for ${SERVICE_USER} (loginctl enable-linger)"
    return 0
  fi
  die "systemd linger is off for ${SERVICE_USER} and 'loginctl enable-linger' was refused; the stack would stop at logout. Run as an admin: sudo loginctl enable-linger ${SERVICE_USER}"
}

# _check_prepared_ports [PORT_START_FILE]: the sysctl the root path writes.
# Required even with --no-tls (D-5): the base compose file binds the app on
# 80. The optional argument is a test hook; check_prepared_host passes it on.
_check_prepared_ports() {
  local port_start_file="${1:-/proc/sys/net/ipv4/ip_unprivileged_port_start}" value
  local fix="printf 'net.ipv4.ip_unprivileged_port_start=80\\n' | sudo tee /etc/sysctl.d/90-olakai.conf && sudo sysctl --system"
  value="$(cat "$port_start_file" 2>/dev/null || true)"
  if [[ ! "$value" =~ ^[0-9]+$ ]]; then
    die "could not read ${port_start_file} (got '${value}'); the service user must be allowed to bind port 80. Run as an admin: ${fix}"
  fi
  if (( value > 80 )); then
    die "unprivileged processes cannot bind port 80 (net.ipv4.ip_unprivileged_port_start is ${value}, need 80 or lower); --no-root does not change sysctl. Run as an admin: ${fix}"
  fi
  print_ok "net.ipv4.ip_unprivileged_port_start=${value}"
}

# The install dir must already be ours, or creatable by us (D-6: the default
# /opt/olakai-onprem is pre-created by the admin, or the operator passes a
# dir under $HOME). _check_existing_runtime (check_install_dir) has already
# refused a foreign marker or a non-empty dir owned by someone else.
_check_prepared_install_dir() {
  local owner parent group
  group="$(id -gn "$SERVICE_USER" 2>/dev/null || true)"
  if [[ -d "$INSTALL_DIR" ]]; then
    owner="$(stat -c %U "$INSTALL_DIR" 2>/dev/null || true)"
    if [[ "$owner" != "$SERVICE_USER" ]]; then
      die "${INSTALL_DIR} is owned by '${owner:-?}', not by ${SERVICE_USER}; --no-root cannot take it over. Run as an admin: sudo chown ${SERVICE_USER}: ${INSTALL_DIR}"
    fi
    print_ok "Install dir ${INSTALL_DIR} is owned by ${SERVICE_USER}"
    return 0
  fi
  parent="$(_nearest_existing_path "$INSTALL_DIR")"
  if [[ ! -w "$parent" ]]; then
    die "${INSTALL_DIR} does not exist and ${SERVICE_USER} cannot create it (${parent} is not writable); --no-root does not escalate. Run as an admin: sudo install -d -o ${SERVICE_USER} -g ${group:-$SERVICE_USER} -m 750 ${INSTALL_DIR}"
  fi
  print_ok "Install dir ${INSTALL_DIR} can be created under ${parent}"
}

# check_prepared_host [SUBUID_FILE] [SUBGID_FILE] [PORT_START_FILE] [RUN_USER_DIR]
#
# Replaces the six ensure_* calls of the root path under --no-root. The
# optional arguments are test hooks (fixtures), handed to the helpers that
# read those files; main calls it bare. Not env vars, as everywhere in
# this file. cosign is bootstrapped by main (step 2) into the same
# $HOME/.local/bin.
# shellcheck disable=SC2120  # the optional arguments are test hooks; main calls it bare
check_prepared_host() {
  local subuid_file="${1:-/etc/subuid}" subgid_file="${2:-/etc/subgid}"
  local port_start_file="${3:-}" run_user_dir="${4:-}"
  # $HOME / $XDG_CONFIG_HOME: validated by _check_no_root_home below (see
  # its comment for why env is acceptable for these two paths).
  local user_bin="${HOME:-}/.local/bin"
  local user_containers="${XDG_CONFIG_HOME:-${HOME:-}/.config}/containers"

  if [[ ! "$SERVICE_UID" =~ ^[0-9]+$ ]] || (( SERVICE_UID == 0 )); then
    die "service user uid not resolved (expected check_root to set SERVICE_UID under --no-root; have '${SERVICE_UID}')."
  fi
  _check_prepared_podman
  _check_service_user_account
  _check_no_root_home
  _check_prepared_subids "$subuid_file" "$subgid_file"
  _check_prepared_linger
  _check_prepared_ports ${port_start_file:+"$port_start_file"}
  _check_prepared_install_dir

  # User-scope compose provider (D-3): Podman reads the user's
  # containers.conf.d. The nodocker marker ensure_compose_provider writes
  # there silences nothing (the shim reads /etc/containers/nodocker only)
  # and is harmless; the README lists the /etc one as optional admin prep.
  # The bin dir is on PATH for the rest of the run so the bundle's
  # install.sh finds cosign by name.
  mkdir -p "$user_bin"
  _prepend_path "$user_bin"
  ensure_compose_provider "$user_bin" "$user_containers"

  # shellcheck disable=SC2119  # bare call on the real /run/user unless the test passes a fixture
  ensure_user_podman_units ${run_user_dir:+"$run_user_dir"}
}

# ──────────────────────────────────────────────────────────────────────────────
# Cosign bootstrap (W2 / OLA-146)
# ──────────────────────────────────────────────────────────────────────────────

# shellcheck disable=SC2120  # the optional argument is a test hook; main calls it bare
ensure_cosign() {
  if [[ "$SKIP_VERIFY" == "true" ]]; then
    print_warn "--skip-verify is set: ALL cosign verification is disabled — both the bundle tarball blob signature AND the four image signatures performed by the bundle's install.sh. Only use this for air-gapped installs where you have already verified out-of-band."
    return 0
  fi

  if command -v cosign >/dev/null 2>&1; then
    COSIGN_BIN="$(command -v cosign)"
    print_ok "cosign present at ${COSIGN_BIN}: $("$COSIGN_BIN" version --json 2>/dev/null | grep -oE '"GitVersion":"[^"]+"' | head -1 | cut -d'"' -f4 || echo unknown)"
    return 0
  fi

  # ensure_cosign [INSTALL_DIR]: the argument exists for the test suite only
  # (scratch install dir); main calls it bare. Not an env var on purpose:
  # the directory is prepended to PATH for the rest of a root run.
  local install_dir="${1:-/usr/local/bin}"
  local arch expected_sha url tmp_path target_path
  case "$(uname -m)" in
    x86_64|amd64) arch="amd64"; expected_sha="$COSIGN_SHA256_LINUX_AMD64" ;;
    aarch64|arm64) arch="arm64"; expected_sha="$COSIGN_SHA256_LINUX_ARM64" ;;
    *) die "unsupported CPU arch '$(uname -m)'; cosign builds for amd64 / arm64 only." ;;
  esac
  url="https://github.com/sigstore/cosign/releases/download/${COSIGN_VERSION}/cosign-linux-${arch}"
  tmp_path="$(mktemp)"
  target_path="${install_dir}/cosign"

  print_info "Downloading cosign ${COSIGN_VERSION} (${arch})…"
  if ! curl -fsSL -o "$tmp_path" "$url"; then
    rm -f "$tmp_path"
    die "failed to download cosign from ${url}"
  fi

  local actual_sha
  actual_sha="$(sha256sum "$tmp_path" | awk '{print $1}')"
  if [[ "$actual_sha" != "$expected_sha" ]]; then
    rm -f "$tmp_path"
    die "cosign SHA256 mismatch (expected ${expected_sha}, got ${actual_sha}). Refusing to install — possible MITM."
  fi
  print_ok "cosign SHA256 matches embedded constant"

  install -m 0755 "$tmp_path" "$target_path"
  rm -f "$tmp_path"
  COSIGN_BIN="$target_path"
  # Belt and braces on top of COSIGN_BIN: the bundle's install.sh (bring_up)
  # calls `cosign` by name for per-image verification and inherits our PATH,
  # which under RHEL-family sudo secure_path lacks /usr/local/bin.
  _prepend_path "$install_dir"
  print_ok "cosign installed at ${target_path}"
}

# ──────────────────────────────────────────────────────────────────────────────
# Interactive collection (D-024 / OLA-169)
# ──────────────────────────────────────────────────────────────────────────────

# Prompt with optional default, returning value via stdout. Hard-fails in
# non-interactive mode if the var is empty.
_prompt() {
  local var_name="$1" prompt_text="$2" default_value="${3:-}" silent="${4:-false}"
  local current="${!var_name}" reply

  if [[ -n "$current" ]]; then
    return 0
  fi

  if [[ "$NON_INTERACTIVE" == "true" ]]; then
    # Optional inputs (those with a default) fall back silently in
    # non-interactive mode; required inputs hard-fail with a flag hint.
    if [[ -n "$default_value" ]]; then
      printf -v "$var_name" '%s' "$default_value"
      return 0
    fi
    die "missing required input '${var_name}' (pass --${var_name//_/-} or run interactively)."
  fi

  # Read from TTY_FD (fd 0 in the review-first path, fd 3 under `curl | bash`).
  # This branch is only reached when NON_INTERACTIVE is false, which the
  # parse_args guard guarantees implies TTY_FD is non-empty.
  if [[ "$silent" == "true" ]]; then
    printf '    %s%s%s ' "$C_BOLD" "$prompt_text" "$C_RESET"
    read -rs -u "$TTY_FD" reply
    printf '\n'
  else
    if [[ -n "$default_value" ]]; then
      printf '    %s%s%s [%s]: ' "$C_BOLD" "$prompt_text" "$C_RESET" "$default_value"
    else
      printf '    %s%s%s: ' "$C_BOLD" "$prompt_text" "$C_RESET"
    fi
    read -r -u "$TTY_FD" reply
  fi
  if [[ -z "$reply" ]]; then
    reply="$default_value"
  fi
  printf -v "$var_name" '%s' "$reply"
}

_confirm() {
  local prompt_text="$1" default="${2:-y}"
  if [[ "$NON_INTERACTIVE" == "true" ]]; then
    [[ "$default" == "y" ]]
    return $?
  fi
  local reply
  printf '    %s%s%s [%s/n]: ' "$C_BOLD" "$prompt_text" "$C_RESET" "$default"
  read -r -u "$TTY_FD" reply
  reply="${reply:-$default}"
  [[ "$reply" =~ ^[Yy]([Ee][Ss])?$ ]]
}

_validate_domain() {
  local d="$1"
  # Reject empty, embedded scheme, or whitespace.
  [[ -n "$d" ]]                                              || return 1
  [[ "$d" != *"://"* ]]                                      || return 1
  [[ "$d" != *" "* ]]                                        || return 1
  # Permissive DNS-name shape; relay does the authoritative check.
  [[ "$d" =~ ^[A-Za-z0-9]([A-Za-z0-9.-]*[A-Za-z0-9])?$ ]]    || return 1
}

_validate_email() {
  [[ "$1" =~ ^[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}$ ]]
}

collect_inputs() {
  _prompt LICENSE_KEY "License key" "" "true"
  [[ -n "$LICENSE_KEY" ]] || die "license key cannot be empty."

  if [[ -z "$DOMAIN" && "$NON_INTERACTIVE" == "false" ]]; then
    print_dim "DNS for the domain should already point at this VM. If it doesn't yet, that's OK — set it now and we'll wait for propagation before TLS issuance."
  fi
  _prompt DOMAIN "Customer-facing domain (e.g. olakai.example.com)"
  _validate_domain "$DOMAIN" || die "invalid domain '${DOMAIN}'."

  _prompt ADMIN_EMAIL "First admin email (will receive a magic-link login)"
  _validate_email "$ADMIN_EMAIL" || die "invalid email '${ADMIN_EMAIL}'."

  _prompt INSTALL_DIR "Install directory" "/opt/olakai-onprem"

  if [[ "$NO_TLS" != "true" && "$NON_INTERACTIVE" == "false" ]]; then
    if ! _confirm "Issue TLS certificates via Let's Encrypt (recommended)?" "y"; then
      NO_TLS="true"
      print_warn "TLS issuance disabled — make sure you terminate TLS upstream."
    fi
  fi

  print_ok "Inputs collected"
}

# ──────────────────────────────────────────────────────────────────────────────
# Helper: parse a JSON field via Python (every supported distro ships python3).
# Stdin is the JSON body; field is a top-level key. Empty string if missing.
# ──────────────────────────────────────────────────────────────────────────────

_json_field() {
  local field="$1"
  python3 -c "
import json, sys
try:
    d = json.load(sys.stdin)
except Exception:
    sys.exit(0)
v = d.get('${field}')
if v is None:
    sys.exit(0)
sys.stdout.write(str(v))
"
}

# ──────────────────────────────────────────────────────────────────────────────
# Handshake (W5 / OLA-156, with D-021 appUrl handling)
# ──────────────────────────────────────────────────────────────────────────────

do_handshake() {
  if ! command -v python3 >/dev/null 2>&1; then
    die "python3 not found; required for parsing the handshake response."
  fi

  local app_url body http_status response_file curl_config body_file
  app_url="https://${DOMAIN}"
  response_file="${WORK_DIR}/handshake.json"
  curl_config="${WORK_DIR}/curl.cfg"
  body_file="${WORK_DIR}/handshake-body.json"

  # Build the request body via Python's json.dumps so the result is
  # guaranteed to be a well-formed JSON value regardless of what's in
  # DOMAIN / VERSION_PIN. The validators upstream of here narrow these
  # inputs further, but defense-in-depth here costs nothing.
  body="$(VERSION_PIN="$VERSION_PIN" APP_URL="$app_url" python3 -c '
import json, os
v = os.environ.get("VERSION_PIN", "")
body = {"instanceMetadata": {"appUrl": os.environ["APP_URL"]}}
if v:
    body["version"] = v.lstrip("v")
print(json.dumps(body))
')"
  printf '%s' "$body" > "$body_file"

  # Pass auth via curl --config so the license key never appears on the
  # command line (and therefore never in /proc/<pid>/cmdline or `ps auxww`).
  # The config file is created mode 0600 inside an mktemp work dir, and
  # removed as soon as the request returns (success or failure) rather than
  # lingering until the EXIT trap. xtrace is silenced around the heredoc so
  # an inherited `set -x` (e.g. `bash -x get.sh`) can never echo the bearer.
  local xtrace_was_on="false"
  if [[ "$-" == *x* ]]; then xtrace_was_on="true"; fi
  { set +x; } 2>/dev/null
  umask 077
  cat > "$curl_config" <<EOF
header = "Authorization: Bearer ${LICENSE_KEY}"
header = "Content-Type: application/json"
EOF
  umask 022
  if [[ "$xtrace_was_on" == "true" ]]; then set -x; fi

  print_info "POST ${RELAY_HANDSHAKE_URL}"
  http_status="$(curl -sS -o "$response_file" -w '%{http_code}' \
    -X POST \
    -K "$curl_config" \
    --data-binary "@${body_file}" \
    "$RELAY_HANDSHAKE_URL" || echo "000")"
  rm -f "$curl_config"

  case "$http_status" in
    200) ;;
    400) die "handshake rejected (400): malformed request. Re-check --domain / --version." ;;
    401) die "handshake rejected (401): license invalid. Confirm the key with Olakai sales." ;;
    409) die "handshake rejected (409): license already consumed by a previous install. Contact Olakai sales for a fresh key." ;;
    410) die "handshake rejected (410): requested release version is yanked or unknown. Try without --version, or pin to the value 'latestAvailable' from the response: $(cat "$response_file")" ;;
    503) die "handshake rejected (503): no release available yet. Olakai is still publishing the bundle — try again later or contact support." ;;
    000) die "handshake transport failure: could not reach ${RELAY_HANDSHAKE_URL}. Check outbound network access." ;;
    *) die "handshake failed (HTTP ${http_status}): $(cat "$response_file")" ;;
  esac

  HANDSHAKE_DEPLOYMENT_BEARER="$(_json_field deploymentBearer < "$response_file")"
  HANDSHAKE_DOWNLOAD_URL="$(_json_field downloadUrl < "$response_file")"
  HANDSHAKE_EXPECTED_SHA256="$(_json_field expectedSha256 < "$response_file")"
  HANDSHAKE_EXPECTED_SIG_B64="$(_json_field expectedSignature < "$response_file")"
  HANDSHAKE_EXPECTED_CERT_PEM="$(_json_field expectedCertificate < "$response_file")"
  HANDSHAKE_VERSION="$(_json_field version < "$response_file")"

  [[ -n "$HANDSHAKE_DEPLOYMENT_BEARER" ]] \
    || die "handshake response missing deploymentBearer (relay bug; contact support)."
  [[ -n "$HANDSHAKE_DOWNLOAD_URL" ]] \
    || die "handshake response has no downloadUrl. The release bucket may not be provisioned yet (OLA-154); contact support."
  [[ -n "$HANDSHAKE_EXPECTED_SHA256" ]] \
    || die "handshake response missing expectedSha256."
  [[ -n "$HANDSHAKE_VERSION" ]] \
    || die "handshake response missing version."

  # Sig + cert are required for cosign verify-blob unless verification is
  # explicitly skipped or running in offline-bundle mode.
  if [[ "$SKIP_VERIFY" != "true" && -z "$REKOR_BUNDLE" ]]; then
    [[ -n "$HANDSHAKE_EXPECTED_SIG_B64" ]] \
      || die "handshake response missing expectedSignature. Pass --skip-verify to bypass (UNSAFE) or wait for the relay to be updated."
    [[ -n "$HANDSHAKE_EXPECTED_CERT_PEM" ]] \
      || die "handshake response missing expectedCertificate. Pass --skip-verify to bypass (UNSAFE) or wait for the relay to be updated."
  fi

  print_ok "License accepted; bundle version ${HANDSHAKE_VERSION} resolved"
}

# ──────────────────────────────────────────────────────────────────────────────
# Download + verify (W2 + W5)
# ──────────────────────────────────────────────────────────────────────────────

download_and_verify() {
  local tarball="${WORK_DIR}/olakai-bundle.tar.gz"
  local sig_path="${WORK_DIR}/olakai-bundle.tar.gz.sig"
  local cert_path="${WORK_DIR}/olakai-bundle.tar.gz.cert"

  print_info "Downloading bundle (signed URL, 15-min TTL)…"
  if ! curl -fsSL -o "$tarball" "$HANDSHAKE_DOWNLOAD_URL"; then
    die "tarball download failed. The signed URL may have expired (15 min TTL); re-run get.sh to refresh."
  fi

  local actual_sha
  actual_sha="$(sha256sum "$tarball" | awk '{print $1}')"
  if [[ "$actual_sha" != "$HANDSHAKE_EXPECTED_SHA256" ]]; then
    die "tarball SHA256 mismatch (expected ${HANDSHAKE_EXPECTED_SHA256}, got ${actual_sha}). Possible corruption or MITM — DO NOT proceed; contact Olakai support."
  fi
  print_ok "Tarball SHA256 verified"

  if [[ "$SKIP_VERIFY" == "true" ]]; then
    print_warn "Skipping cosign verify-blob (--skip-verify)."
    TARBALL_PATH_OUT="$tarball"
    return 0
  fi

  # ensure_cosign always runs before us in main; the guard turns a broken
  # call order into a clear message instead of "command not found".
  [[ -n "$COSIGN_BIN" && -x "$COSIGN_BIN" ]] \
    || die "cosign binary not resolved (expected ensure_cosign to set COSIGN_BIN; have '${COSIGN_BIN}'). Re-run get.sh; if it persists, contact Olakai support."

  if [[ -n "$REKOR_BUNDLE" ]]; then
    [[ -f "$REKOR_BUNDLE" ]] || die "--rekor-bundle path '${REKOR_BUNDLE}' does not exist."
    print_info "Verifying with offline Rekor bundle: ${REKOR_BUNDLE}"
    if ! "$COSIGN_BIN" verify-blob \
      --certificate-identity-regexp "$COSIGN_CERT_IDENTITY_REGEXP" \
      --certificate-oidc-issuer "$COSIGN_OIDC_ISSUER" \
      --bundle "$REKOR_BUNDLE" \
      --offline \
      "$tarball" >/dev/null 2>&1; then
      die "cosign verify-blob (offline) FAILED. The bundle may not match the published signature; refusing to install."
    fi
  else
    printf '%s' "$HANDSHAKE_EXPECTED_SIG_B64" > "$sig_path"
    printf '%s' "$HANDSHAKE_EXPECTED_CERT_PEM" > "$cert_path"

    # Defense-in-depth: catch a malformed handshake response BEFORE running
    # cosign so the operator gets a "relay returned a malformed cert" error
    # instead of an opaque cosign-internal failure.
    local first_cert_line
    first_cert_line="$(head -n 1 "$cert_path" 2>/dev/null || true)"
    if [[ "$first_cert_line" != "-----BEGIN CERTIFICATE-----" ]]; then
      die "handshake's expectedCertificate is not a PEM certificate (got '${first_cert_line:0:80}'). Relay response is malformed; contact Olakai support."
    fi
    if [[ ! "$HANDSHAKE_EXPECTED_SIG_B64" =~ ^[A-Za-z0-9+/=[:space:]]+$ ]]; then
      die "handshake's expectedSignature is not valid base64. Relay response is malformed; contact Olakai support."
    fi

    print_info "Verifying cosign keyless signature against published GitHub workflow identity…"
    if ! "$COSIGN_BIN" verify-blob \
      --certificate-identity-regexp "$COSIGN_CERT_IDENTITY_REGEXP" \
      --certificate-oidc-issuer "$COSIGN_OIDC_ISSUER" \
      --signature "$sig_path" \
      --certificate "$cert_path" \
      "$tarball" >/dev/null; then
      die "cosign verify-blob FAILED. Signature/identity mismatch — refusing to install."
    fi
  fi
  print_ok "cosign signature verified (signer: olakai-ai/localnode-app onprem-publish.yml)"

  TARBALL_PATH_OUT="$tarball"
}

# ──────────────────────────────────────────────────────────────────────────────
# Extract
# ──────────────────────────────────────────────────────────────────────────────

# Pre-flight (runs BEFORE the license handshake): a non-empty install dir
# must abort before the single-use key is consumed, otherwise a re-run
# without --force-overlay burns the key and then stops at extraction.
check_install_dir() {
  if [[ -e "$INSTALL_DIR" && ! -d "$INSTALL_DIR" ]]; then
    die "${INSTALL_DIR} exists and is not a directory."
  fi
  if [[ -d "$INSTALL_DIR" ]] && [[ -n "$(ls -A "$INSTALL_DIR" 2>/dev/null)" ]]; then
    _check_existing_runtime
    if [[ "$FORCE_OVERLAY" == "true" ]]; then
      print_warn "${INSTALL_DIR} is non-empty; overlaying (--force-overlay). Existing .env (if any) will be preserved."
    elif [[ "$NON_INTERACTIVE" == "true" ]]; then
      # Default stays "abort" so a blind re-run never touches a live install.
      # Name the escape hatch so automation authors do not have to guess.
      die "aborted: install directory ${INSTALL_DIR} is non-empty. Pass --force-overlay (or OLAKAI_FORCE_OVERLAY=true) to overlay it (existing .env is preserved), or guard your automation on the presence of ${INSTALL_DIR}/.env."
    elif ! _confirm "${INSTALL_DIR} is non-empty; overlay anyway? Existing .env (if any) will be preserved." "n"; then
      die "aborted: install directory is non-empty."
    fi
    return 0
  fi
  print_ok "Install dir ${INSTALL_DIR} is empty or absent"
}

# First non-blank, non-comment line of a `.olakai-runtime` marker, trimmed
# (the same rule as the bundle's olakai_runtime in scripts/lib/compose-files.sh).
_read_runtime_marker() {
  local marker="$1" line token=""
  while IFS= read -r line || [[ -n "$line" ]]; do
    line="${line%%#*}"
    line="${line#"${line%%[![:space:]]*}"}"
    line="${line%"${line##*[![:space:]]}"}"
    if [[ -n "$line" ]]; then token="$line"; break; fi
  done < "$marker"
  printf '%s' "$token"
}

# Pre-flight guard on a non-empty install dir; runs BEFORE the handshake so
# a refusal never consumes the license key. Refuses:
#   - a `.olakai-runtime` marker naming a runtime other than --runtime.
#     Docker state and rootless Podman state share nothing (daemon storage
#     vs the service user's storage), so overwriting the marker, activating
#     the overlay and re-owning the tree would silently break the running
#     deployment. In-place migration is not supported in either direction;
#   - no marker but a data/ tree under --runtime=podman-rootless: a Docker
#     deployment from before the marker existed (bundles < 1.5.9);
#   - under --runtime=podman-rootless, a dir owned by another unprivileged
#     user than --service-user: a changed name on re-run would leave the
#     running stack under the old user (N16).
_check_existing_runtime() {
  local marker="${INSTALL_DIR}/.olakai-runtime" existing="" evidence="" owner
  if [[ -f "$marker" ]]; then
    existing="$(_read_runtime_marker "$marker")"
    existing="${existing:-docker}"
    evidence="${marker} says '${existing}'"
  elif [[ -d "${INSTALL_DIR}/data" ]]; then
    existing="docker"
    evidence="${INSTALL_DIR}/data exists and there is no .olakai-runtime marker, so this predates Podman support"
  fi
  if [[ -n "$existing" && "$existing" != "$RUNTIME" ]]; then
    die "${INSTALL_DIR} holds a '${existing}' deployment (${evidence}) and --runtime=${RUNTIME} was requested. In-place runtime migration is not supported: back up (scripts/backup.sh), fresh-install on the new runtime, then restore."
  fi
  if [[ "$RUNTIME" == "podman-rootless" ]]; then
    owner="$(stat -c %U "$INSTALL_DIR" 2>/dev/null || true)"
    if [[ "$NO_ROOT" == "true" ]]; then
      # We cannot switch users; the dir must be ours (root-owned included).
      if [[ -n "$owner" && "$owner" != "$SERVICE_USER" ]]; then
        die "${INSTALL_DIR} is owned by '${owner}', not by ${SERVICE_USER}; --no-root cannot take it over. Run as an admin: sudo chown ${SERVICE_USER}: ${INSTALL_DIR}"
      fi
    elif [[ -n "$owner" && "$owner" != "root" && "$owner" != "$SERVICE_USER" ]]; then
      die "${INSTALL_DIR} is owned by '${owner}' but --service-user=${SERVICE_USER} was requested. Re-run with --service-user=${owner}; a deployment is not moved between users in place."
    fi
  fi
}

extract_bundle() {
  # Enforce, not just assert, the ".env is preserved" promise: a bundle that
  # ships a top-level .env would overwrite the operator's secrets on overlay.
  # Layout is `onprem-bundle/<files>` (hence --strip-components=1 below), so
  # only `<topdir>/.env` (optionally `./`-prefixed) is the dangerous entry.
  # The listing goes through a variable rather than `| grep -q` so grep's
  # early exit cannot SIGPIPE tar under pipefail and hide a hit.
  local bundled_env
  bundled_env="$(tar -tzf "$TARBALL_PATH_OUT" | grep -xE '(\./)?[^/]+/\.env' || true)"
  if [[ -n "$bundled_env" ]]; then
    die "bundle contains a top-level .env (${bundled_env}); refusing to extract over ${INSTALL_DIR} because it would overwrite existing secrets. Report this to Olakai support."
  fi

  mkdir -p "$INSTALL_DIR"
  print_info "Extracting bundle to ${INSTALL_DIR}…"
  # Bundle tarball convention (per W1 / OLA-145): top-level dir is
  # `onprem-bundle/`. Strip the leading component so files land directly
  # in INSTALL_DIR (matches today's manual scp+tar flow).
  if ! tar -xzf "$TARBALL_PATH_OUT" --strip-components=1 -C "$INSTALL_DIR"; then
    die "failed to extract bundle to ${INSTALL_DIR}."
  fi
  print_ok "Extracted to ${INSTALL_DIR}"
}

# Runs immediately after extract_bundle and before anything is written to
# the tree. The overlay is the one file the rootless runtime cannot do
# without, and it first shipped in bundle 1.5.9. By this point the license
# key is consumed (the handshake precedes the download), which the message
# says plainly so the operator does not burn a second key on a retry.
check_bundle_runtime_support() {
  [[ "$RUNTIME" == "podman-rootless" ]] || return 0
  local overlay_example="${INSTALL_DIR}/examples/docker-compose.podman.yml"
  if [[ ! -f "$overlay_example" ]]; then
    die "bundle ${HANDSHAKE_VERSION:-?} does not ship examples/docker-compose.podman.yml: --runtime=podman-rootless needs bundle 1.5.10 or later. The license key has already been consumed by this run; contact support@olakai.ai for a new one, then re-run without --version (or pin v1.5.10 or later), or use --runtime=docker."
  fi
  print_ok "Bundle ${HANDSHAKE_VERSION:-?} supports the podman-rootless runtime (examples/docker-compose.podman.yml present)"
}

# ──────────────────────────────────────────────────────────────────────────────
# Managed state: --env-file and --private-ca
#
# The bundle's install.sh reads the state topology from .env
# (scripts/lib/state-topology.sh, bundle 1.5.8+): when DATABASE_URL,
# OBJECT_STORAGE_* and REDIS_URL all point at managed services it starts no
# bundled PostgreSQL / MinIO / Redis. get.sh materializes .env from
# .env.example and runs install.sh at once, so without a way to inject those
# settings the FIRST bring-up always started the bundled stateful containers,
# which a customer whose policy forbids third-party stateful containers
# cannot accept even once. --env-file closes that gap: its KEY=VALUE lines
# are merged into .env after the secret fill and before the hand-off.
# --private-ca stages the CA certificate + overlay that a managed PostgreSQL
# or Redis behind a non-public CA needs, for the same first-bring-up reason.
#
# Both files are named by explicit flags only. get.sh runs as root and its
# functions are sourced by the test suite; an env-controlled path would let
# anything in the caller's environment choose what the installer reads
# (see detect_os). The --env-file values are held in this process only:
# never exported, never printed (only key names are logged), never copied
# to disk other than into .env itself, and xtrace is silenced while they
# are handled so `bash -x get.sh` cannot echo them.
# ──────────────────────────────────────────────────────────────────────────────

# Pre-flight (runs BEFORE the license handshake, from step 1/8): a missing,
# unreadable, world-readable or malformed --env-file must fail before the
# single-use key is consumed. Parses the file once into ENV_FILE_KEYS /
# ENV_FILE_VALUES; materialize_env merges from those arrays.
check_env_file() {
  [[ -n "$ENV_FILE" ]] || return 0
  [[ -e "$ENV_FILE" ]] || die "--env-file ${ENV_FILE} does not exist."
  [[ -f "$ENV_FILE" ]] || die "--env-file ${ENV_FILE} is not a regular file."
  [[ -r "$ENV_FILE" ]] || die "--env-file ${ENV_FILE} is not readable."

  # The file carries database and object-storage credentials: refuse anything
  # another user could read or replace. Exact modes, so a setuid/sticky bit
  # or a group-readable 0640 is refused too.
  # Under --no-root the file is read as the invoking user, so that user (not
  # root, who could not be asked to fix it) must own it.
  local mode_owner mode owner want_owner="root"
  if [[ "$NO_ROOT" == "true" ]]; then want_owner="$SERVICE_USER"; fi
  mode_owner="$(stat -c '%a %U' "$ENV_FILE" 2>/dev/null || true)"
  mode="${mode_owner%% *}"
  owner="${mode_owner#* }"
  if [[ "$owner" != "$want_owner" || ( "$mode" != "600" && "$mode" != "400" ) ]]; then
    die "--env-file ${ENV_FILE} carries credentials and must be owned by ${want_owner} with mode 0600 or 0400 (have owner '${owner:-?}', mode ${mode:-?}). Fix: chown ${want_owner}: ${ENV_FILE} && chmod 600 ${ENV_FILE}"
  fi

  _parse_env_file "$ENV_FILE"
  local keys_human
  keys_human="$(printf '%s, ' "${ENV_FILE_KEYS[@]}")"
  print_ok "--env-file ${ENV_FILE}: ${#ENV_FILE_KEYS[@]} key(s) will be applied to .env (${keys_human%, })"
}

# Which get.sh input sets an owned key; named in the refusal so the operator
# knows where the value belongs instead.
_env_file_owned_key_source() {
  case "$1" in
    OLAKAI_DOMAIN)          printf -- '--domain' ;;
    AUTH_URL)               printf -- '--domain (and --no-tls for the scheme)' ;;
    OLAKAI_ADMIN_EMAIL)     printf -- '--admin-email' ;;
    OLAKAI_EMAIL_RELAY_KEY) printf -- 'the license handshake (no flag; it is the deployment bearer)' ;;
    OLAKAI_VERSION)         printf -- '--version (resolved by the license handshake)' ;;
    OLAKAI_SKIP_VERIFY)     printf -- '--skip-verify (get.sh resolves its conflict with --verify-images itself)' ;;
    OLAKAI_VERIFY_IMAGES)   printf -- '--verify-images, or OLAKAI_VERIFY_IMAGES in the environment of get.sh' ;;
    *)                      printf -- 'get.sh' ;;
  esac
}

# _parse_env_file FILE: KEY=VALUE per line; blank lines and `#` comments are
# skipped; CRLF is tolerated; a UTF-8 byte-order mark at the start of the
# file (a Windows editor adds one) is stripped with a warning, since left in
# place it would glue itself to the first key. The value is everything after
# the first `=`, verbatim (quotes included), so the bundle's env parser sees
# exactly what the operator wrote, as if typed into .env. Refuses a line
# without `=`, an invalid key, a key get.sh owns, and a key that appears
# twice (the file is an explicit override list; an ambiguous one is a
# mistake, not a precedence puzzle). A value that is not single-quoted and
# contains `$`, a backtick, whitespace or ` #` gets a warning
# (_warn_env_file_value_quoting), not a refusal. Messages name line numbers
# and keys, never values.
_parse_env_file() {
  local file="$1" line key value lineno=0 i owned
  local xtrace_was_on="false"
  if [[ "$-" == *x* ]]; then xtrace_was_on="true"; fi
  { set +x; } 2>/dev/null

  ENV_FILE_KEYS=()
  ENV_FILE_VALUES=()
  # shellcheck disable=SC2094  # $file is only named in a warning; nothing in the loop writes to it
  while IFS= read -r line || [[ -n "$line" ]]; do
    lineno=$(( lineno + 1 ))
    if (( lineno == 1 )) && [[ "$line" == $'\xEF\xBB\xBF'* ]]; then
      line="${line#$'\xEF\xBB\xBF'}"
      print_warn "--env-file ${file}: a UTF-8 byte-order mark at the start of line 1 was ignored; save the file without a BOM."
    fi
    line="${line%$'\r'}"
    line="${line#"${line%%[![:space:]]*}"}"
    if [[ -z "$line" || "$line" == \#* ]]; then continue; fi
    if [[ "$line" != *=* ]]; then
      die "--env-file ${file}: line ${lineno} is not a KEY=VALUE line (no '=')."
    fi
    key="${line%%=*}"
    value="${line#*=}"
    if [[ ! "$key" =~ ^[A-Za-z_][A-Za-z0-9_]*$ ]]; then
      die "--env-file ${file}: line ${lineno} has an invalid key (letters, digits and '_' only; must not start with a digit; no spaces around '=')."
    fi
    for owned in "${GET_SH_OWNED_ENV_KEYS[@]}"; do
      if [[ "$key" == "$owned" ]]; then
        die "--env-file ${file}: ${key} (line ${lineno}) is set by get.sh from $(_env_file_owned_key_source "$key"); remove it from the file."
      fi
    done
    for (( i = 0; i < ${#ENV_FILE_KEYS[@]}; i++ )); do
      if [[ "${ENV_FILE_KEYS[i]}" == "$key" ]]; then
        die "--env-file ${file}: ${key} appears twice (line ${lineno}); keep one line per key."
      fi
    done
    _warn_env_file_value_quoting "$file" "$key" "$lineno" "$value"
    ENV_FILE_KEYS+=("$key")
    ENV_FILE_VALUES+=("$value")
  done < "$file"

  if [[ "$xtrace_was_on" == "true" ]]; then set -x; fi
  if (( ${#ENV_FILE_KEYS[@]} == 0 )); then
    die "--env-file ${file} sets no keys (only blank lines and comments)."
  fi
  ENV_FILE_PARSED="true"
}

# _warn_env_file_value_quoting FILE KEY LINENO VALUE: .env is read three
# ways, by install.sh (bash `source`), by compose (interpolation, env_file)
# and by the bundle's olakai_read_env_var, and only a single-quoted value is
# the same literal to all three. Unquoted `pa$w0rd` sources as `pa`; an
# unquoted value with whitespace sources as a command; ` #` after an
# unquoted value starts a comment for the bundle's readers; and double
# quotes do not protect `$` or a backtick from the shell (whitespace and
# ` #` are safe inside them). Warns, never dies: `${VAR}` interpolation may
# be exactly what the operator meant. Prints the key and line, never the
# value.
_warn_env_file_value_quoting() {
  local file="$1" key="$2" lineno="$3" value="$4" quoting hazard=""
  if [[ "$value" == \'*\' && ${#value} -ge 2 ]]; then
    return 0
  fi
  if [[ "$value" == \"*\" && ${#value} -ge 2 ]]; then
    quoting="double-quoted"
  else
    quoting="unquoted"
  fi
  if [[ "$value" == *'$'* ]]; then
    hazard='$'
  elif [[ "$value" == *'`'* ]]; then
    hazard='a backtick'
  elif [[ "$quoting" == "unquoted" ]]; then
    if [[ "$value" == *[[:space:]]'#'* ]]; then
      hazard="' #' (the bundle's readers treat what follows as a comment)"
    elif [[ "$value" == *[[:space:]]* ]]; then
      hazard='whitespace'
    fi
  fi
  [[ -n "$hazard" ]] || return 0
  print_warn "--env-file ${file}: ${key} (line ${lineno}) is ${quoting} and contains ${hazard}; install.sh sources .env, so the value may not survive as written. Single-quote it (${key}='...') unless \${VAR} interpolation is intended."
}

# _merge_env_file ENV_TARGET: applies the parsed --env-file pairs to the
# materialized .env. Every uncommented `KEY=` line of a named key is
# rewritten in place (position kept; every occurrence, because compose
# resolves duplicate keys last-wins and a stray later duplicate would win
# over the override); a key with no line is appended. The rewrite goes
# through a 0600 temp file next to .env and a rename, never through sed, so
# no character of a value needs escaping. Called from materialize_env after
# the secret fill and the get.sh-owned keys, before the 0600 chmod and the
# hand-off; under --runtime=podman-rootless that is before the chown, so the
# result is owned by the service user like the rest of the tree. The temp
# file never outlives a failure: ENV_MERGE_TMP names it for the cleanup trap
# while it exists, and a failed rename removes it explicitly.
_merge_env_file() {
  local env_file="$1"
  [[ -n "$ENV_FILE" ]] || return 0
  if [[ "$ENV_FILE_PARSED" != "true" ]]; then
    die "--env-file was not parsed (expected check_env_file to run in pre-flight). Re-run get.sh; if it persists, contact Olakai support."
  fi
  local n=${#ENV_FILE_KEYS[@]} tmp line key i
  local -a applied=()
  for (( i = 0; i < n; i++ )); do applied[i]="false"; done

  local xtrace_was_on="false"
  if [[ "$-" == *x* ]]; then xtrace_was_on="true"; fi
  { set +x; } 2>/dev/null
  tmp="${env_file}.merge.tmp"
  ENV_MERGE_TMP="$tmp"
  umask 077
  {
    while IFS= read -r line || [[ -n "$line" ]]; do
      if [[ "$line" =~ ^([A-Za-z_][A-Za-z0-9_]*)= ]]; then
        key="${BASH_REMATCH[1]}"
        for (( i = 0; i < n; i++ )); do
          if [[ "${ENV_FILE_KEYS[i]}" == "$key" ]]; then
            line="${key}=${ENV_FILE_VALUES[i]}"
            applied[i]="true"
            break
          fi
        done
      fi
      printf '%s\n' "$line"
    done < "$env_file"
    for (( i = 0; i < n; i++ )); do
      [[ "${applied[i]}" == "true" ]] && continue
      printf '%s=%s\n' "${ENV_FILE_KEYS[i]}" "${ENV_FILE_VALUES[i]}"
    done
  } > "$tmp"
  umask 022
  if ! mv -f "$tmp" "$env_file"; then
    rm -f "$tmp"
    ENV_MERGE_TMP=""
    if [[ "$xtrace_was_on" == "true" ]]; then set -x; fi
    die "--env-file merge: could not rename ${tmp} over ${env_file}; .env is unchanged and the temp file was removed."
  fi
  ENV_MERGE_TMP=""
  if [[ "$xtrace_was_on" == "true" ]]; then set -x; fi

  for (( i = 0; i < n; i++ )); do
    print_info ".env: set ${ENV_FILE_KEYS[i]} (from --env-file)"
  done
}

# Pre-flight (step 1/8, before the handshake): the --private-ca file must be
# a readable, non-empty PEM certificate. A DER `.crt` passes `-f` and then
# contributes nothing to the trust store, which fails exactly like no CA at
# all, so the PEM header is checked here: anywhere within the first 4096
# bytes, as the bundle's olakai_require_private_ca_cert does, because a
# `openssl x509 -text` dump or a PKCS#12 export puts a text preamble before
# it. `tr -d` drops the NUL bytes of a DER file so the command substitution
# stays quiet, and drains head's output so head never takes a SIGPIPE under
# pipefail.
check_private_ca() {
  [[ -n "$PRIVATE_CA" ]] || return 0
  [[ -e "$PRIVATE_CA" ]] || die "--private-ca ${PRIVATE_CA} does not exist."
  [[ -f "$PRIVATE_CA" ]] || die "--private-ca ${PRIVATE_CA} is not a regular file."
  [[ -r "$PRIVATE_CA" ]] || die "--private-ca ${PRIVATE_CA} is not readable."
  [[ -s "$PRIVATE_CA" ]] || die "--private-ca ${PRIVATE_CA} is empty; an empty file adds nothing to the trust store."
  local header_probe
  header_probe="$(head -c 4096 "$PRIVATE_CA" 2>/dev/null | tr -d '\000' || true)"
  if [[ "$header_probe" != *"-----BEGIN CERTIFICATE-----"* ]]; then
    die "--private-ca ${PRIVATE_CA} is not a PEM certificate (no -----BEGIN CERTIFICATE----- header within the first 4096 bytes). Convert DER with: openssl x509 -inform der -in ca.crt -out ca.pem"
  fi
  print_ok "--private-ca ${PRIVATE_CA}: PEM certificate (staged as certs/private-ca.pem after extraction)"
}

# Runs after extraction and before the hand-off: copies the CA to
# <install-dir>/certs/private-ca.pem (0644: the containers read it through a
# bind mount, and under rootless Podman as the service user) and activates
# examples/docker-compose.private-ca.yml as docker-compose.private-ca.yml,
# which the bundle's install.sh and upgrade.sh auto-detect
# (scripts/lib/compose-files.sh). PEM first, overlay second: the bundle
# refuses an active overlay with no certificate behind it, and a container
# runtime would otherwise create a DIRECTORY at the bind-mount source. The
# overlay first shipped in bundle 1.5.8; by now the license key is consumed
# (the handshake precedes the download), which the message says plainly.
stage_private_ca() {
  [[ -n "$PRIVATE_CA" ]] || return 0
  local overlay_example="${INSTALL_DIR}/examples/docker-compose.private-ca.yml"
  local overlay_active="${INSTALL_DIR}/docker-compose.private-ca.yml"
  local certs_dir="${INSTALL_DIR}/certs"
  local pem="${certs_dir}/private-ca.pem"
  if [[ ! -f "$overlay_example" ]]; then
    die "bundle ${HANDSHAKE_VERSION:-?} does not ship examples/docker-compose.private-ca.yml: --private-ca needs bundle 1.5.8 or later. The license key has already been consumed by this run; contact support@olakai.ai for a new one, then re-run without --version (or pin v1.5.8 or later)."
  fi
  if [[ -d "$pem" ]]; then
    die "${pem} is a directory (a container runtime created it because the overlay was active before the certificate was in place). Remove it (rm -rf ${pem}) and re-run."
  fi
  mkdir -p "$certs_dir"
  install -m 0644 "$PRIVATE_CA" "$pem"
  if [[ ! -f "$overlay_active" ]]; then
    cp "$overlay_example" "$overlay_active"
  fi
  print_ok "Private CA staged at ${pem} (mode 0644); docker-compose.private-ca.yml active (install.sh and upgrade.sh auto-detect it)"
}

# ──────────────────────────────────────────────────────────────────────────────
# Materialize .env
# ──────────────────────────────────────────────────────────────────────────────

# Replace `^KEY=$` with `KEY=<generated>` only when the value is empty.
# Mirrors the helper in localnode-app/onprem-bundle/install.sh so behavior is
# identical: URL-safe base64, idempotent, no overwrite when re-running.
_fill_secret_if_empty() {
  local key="$1" bytes="$2" env_file="$3" encoding="${4:-urlsafe}"
  local value escaped
  if grep -qE "^${key}=$" "$env_file"; then
    if [[ "$encoding" == "standard" ]]; then
      # NEXT_SERVER_ACTIONS_ENCRYPTION_KEY round-trips through Next.js's
      # atob() decrypt path (OLA-132); a URL-safe value fails the boot
      # validator (it requires the standard alphabet A-Za-z0-9+/=). Emit
      # standard base64 with padding for this key. The sed replacement
      # below uses `|` as the delimiter and escapes \ & |, so the +/=
      # characters in standard base64 are inserted safely.
      value="$(openssl rand -base64 "$bytes")"
    else
      value="$(openssl rand -base64 "$bytes" | tr '+/' '-_' | tr -d '=')"
    fi
    escaped="$(printf '%s\n' "$value" | sed -e 's/[\\&|]/\\&/g')"
    sed -i.bak -e "s|^${key}=$|${key}=${escaped}|" "$env_file"
    rm -f "${env_file}.bak"
  fi
}

# Set KEY=<value> in env file. Handles the "key is present, possibly with a
# value" case by replacing in place; otherwise appends. Value is sed-escaped.
_set_env_var() {
  local key="$1" value="$2" env_file="$3" escaped
  escaped="$(printf '%s\n' "$value" | sed -e 's/[\\&|]/\\&/g')"
  if grep -qE "^${key}=" "$env_file"; then
    sed -i.bak -e "s|^${key}=.*|${key}=${escaped}|" "$env_file"
    rm -f "${env_file}.bak"
  else
    printf '\n%s=%s\n' "$key" "$value" >> "$env_file"
  fi
}

materialize_env() {
  local env_file="${INSTALL_DIR}/.env"
  local env_template="${INSTALL_DIR}/.env.example"

  if [[ -f "$env_file" ]]; then
    print_warn ".env already exists — preserving existing values (re-run safety)."
  else
    [[ -f "$env_template" ]] || die "bundle is incomplete: ${env_template} missing."
    cp "$env_template" "$env_file"
  fi

  _fill_secret_if_empty NEXTAUTH_SECRET 32 "$env_file"
  _fill_secret_if_empty AUTH_SECRET 32 "$env_file"
  _fill_secret_if_empty DEFAULT_ENCRYPTION_KEY 32 "$env_file"
  # Standard base64 (NOT url-safe) — see OLA-132 note in _fill_secret_if_empty.
  _fill_secret_if_empty NEXT_SERVER_ACTIONS_ENCRYPTION_KEY 32 "$env_file" standard
  _fill_secret_if_empty ENCRYPTION_SALT 16 "$env_file"
  _fill_secret_if_empty POSTGRES_PASSWORD 24 "$env_file"
  _fill_secret_if_empty MINIO_ROOT_PASSWORD 24 "$env_file"
  _fill_secret_if_empty REDIS_PASSWORD 24 "$env_file"
  # Shared secret between main-app and the support-bundle sidecar (OLA-177).
  # Required by the on-prem env validator; without it the bootstrap container
  # exits 1 on first boot. install.sh seeds this on its own first-run path,
  # but get.sh writes a fully-populated .env so that path never runs here.
  _fill_secret_if_empty OLAKAI_SUPPORT_BUNDLE_SECRET 32 "$env_file"

  _set_env_var OLAKAI_DOMAIN "$DOMAIN" "$env_file"
  if [[ "$NO_TLS" == "true" ]]; then
    _set_env_var AUTH_URL "http://${DOMAIN}" "$env_file"
  else
    _set_env_var AUTH_URL "https://${DOMAIN}" "$env_file"
  fi
  _set_env_var OLAKAI_ADMIN_EMAIL "$ADMIN_EMAIL" "$env_file"
  _set_env_var OLAKAI_EMAIL_RELAY_KEY "$HANDSHAKE_DEPLOYMENT_BEARER" "$env_file"
  _set_env_var OLAKAI_VERSION "$HANDSHAKE_VERSION" "$env_file"

  # --env-file overrides go in LAST, after the generated secrets and the
  # get.sh-owned keys (which it cannot name), so the bundle's install.sh
  # reads the final topology on the first bring-up. No-op without the flag.
  _merge_env_file "$env_file"

  chmod 600 "$env_file"
  print_ok ".env materialized at ${env_file} (mode 0600)"
}

# ──────────────────────────────────────────────────────────────────────────────
# Bring-up: hand off to the bundle's install.sh.
# install.sh sees that .env is fully populated, runs cosign verify on the four
# images, then `docker compose up -d` and waits for app health (5 min).
# ──────────────────────────────────────────────────────────────────────────────

bring_up() {
  local installer="${INSTALL_DIR}/install.sh"
  [[ -x "$installer" ]] || die "bundle is incomplete: ${installer} missing or not executable."

  # Activate the bundled Caddy TLS sidecar unless the operator opted out with
  # --no-tls. The bundle ships docker-compose.caddy.yml.example; install.sh
  # auto-detects the (non-.example) docker-compose.caddy.yml and includes it
  # via -f. Without this copy the stack comes up HTTP-only, so the https
  # AUTH_URL and the magic-link setup URL are unreachable and first-login
  # silently breaks. Idempotent: skip if the operator already created one.
  if [[ "$NO_TLS" != "true" ]]; then
    local caddy_example="${INSTALL_DIR}/docker-compose.caddy.yml.example"
    local caddy_active="${INSTALL_DIR}/docker-compose.caddy.yml"
    if [[ -f "$caddy_example" && ! -f "$caddy_active" ]]; then
      cp "$caddy_example" "$caddy_active"
      print_ok "Activated Caddy TLS sidecar (docker-compose.caddy.yml)."
    fi
  fi

  if [[ "$SKIP_VERIFY" == "true" ]]; then
    export OLAKAI_SKIP_VERIFY=true
    print_warn "Propagating --skip-verify to bundle install.sh (image cosign verify will be skipped)."
  fi

  # Resolve --skip-verify + --verify-images conflict at the user-facing layer
  # rather than punting it to the bundle. --skip-verify is the air-gapped
  # escape hatch that disables ALL verification; --verify-images opts INTO
  # extra per-image cosign. Setting both is contradictory, so honor the
  # safer-looking-but-stricter --skip-verify (matches the existing trust-model
  # header) and warn so the operator notices the contradiction.
  if [[ "$SKIP_VERIFY" == "true" && "$VERIFY_IMAGES" == "true" ]]; then
    print_warn "--skip-verify overrides --verify-images; per-image cosign verify will not run."
    VERIFY_IMAGES="false"
  fi

  if [[ "$VERIFY_IMAGES" == "true" ]]; then
    export OLAKAI_VERIFY_IMAGES=true
    print_info "Propagating --verify-images to bundle install.sh (per-image cosign verify enabled on top of bundle verification)."
  fi

  # Close our borrowed /dev/tty (fd 3, only open under `curl | bash`) before
  # handing off so the bundle's install.sh starts with a clean fd table.
  # `3<&-` is a harmless no-op when fd 3 was never opened (TTY_FD=0 / "").
  if [[ "$RUNTIME" == "podman-rootless" ]]; then
    _stage_rootless_install_dir
    # The two verification switches are exported above for the Docker path;
    # they are also passed as explicit `env` assignments here so the
    # hand-off does not depend on runuser's environment handling.
    local -a handoff_env=()
    if [[ "$SKIP_VERIFY" == "true" ]]; then handoff_env+=(OLAKAI_SKIP_VERIFY=true); fi
    if [[ "$VERIFY_IMAGES" == "true" ]]; then handoff_env+=(OLAKAI_VERIFY_IMAGES=true); fi
    print_info "Handing off to ${installer} as ${SERVICE_USER}…"
    ( cd "$INSTALL_DIR" && _runtime_exec ${handoff_env[@]+"${handoff_env[@]}"} bash install.sh 3<&- )
    return 0
  fi
  print_info "Handing off to ${installer}…"
  ( cd "$INSTALL_DIR" && bash install.sh 3<&- )
}

# ──────────────────────────────────────────────────────────────────────────────
# Success banner (D-022 Option B — always print both magic-link sent +
# bootstrap-log fallback URL, unless --quiet)
# ──────────────────────────────────────────────────────────────────────────────

_extract_setup_url() {
  # The bootstrap container always logs `Setup URL: …` on first boot
  # (per OLA-120). `docker compose logs` takes the compose SERVICE name,
  # which in the bundle is `bootstrap` (`docker compose config --services`:
  # migrate support-bundle bootstrap app worker); the container it creates
  # is `olakai-bootstrap-1` under COMPOSE_PROJECT_NAME=olakai. Passing the
  # container name here fails with "no such service" and yields no URL
  # (seen live on the first OL9 install, OLA-1292).
  #
  # The bundle's install.sh returns when the *app* is healthy, but the
  # bootstrap container may still be minting the magic-link token at that
  # moment. Retry with a short timeout so the success banner reliably
  # contains the fallback URL (per D-022 Option B). 30s is well within
  # human attention span and gives bootstrap ample time after app-healthy.
  #
  # Under rootless Podman the containers live in the service user's storage;
  # root's `docker compose logs` sees an empty project. _runtime_exec runs
  # the command as that user (and is a plain pass-through under Docker).
  local url="" i
  for i in 1 2 3 4 5 6 7; do
    url="$( ( cd "$INSTALL_DIR" \
        && _runtime_exec docker compose logs --tail 200 bootstrap 2>/dev/null \
        | grep -oE 'https?://[^[:space:]]+/setup/admin\?token=[^[:space:]]+' \
        | tail -n 1 ) )"
    if [[ -n "$url" ]]; then
      printf '%s' "$url"
      return 0
    fi
    # 7 attempts, 6 sleeps × 5s = 30s total budget (matches --help and
    # README). The bundle's install.sh waits for app health before
    # returning; bootstrap is typically just behind it.
    if (( i < 7 )); then sleep 5; fi
  done
  return 1
}

print_success_banner() {
  local proto="https"
  [[ "$NO_TLS" == "true" ]] && proto="http"
  local dashboard_url="${proto}://${DOMAIN}"
  local fallback_url=""
  if [[ "$QUIET" != "true" ]]; then
    fallback_url="$(_extract_setup_url || true)"
  fi
  # How a later bundle script is run under rootless Podman: via runuser from
  # a root shell, or directly when the operator is the service user (--no-root).
  local as_user=""
  if [[ "$RUNTIME" == "podman-rootless" && "$NO_ROOT" != "true" ]]; then
    as_user="sudo runuser -u ${SERVICE_USER} -- "
  fi

  printf '\n'
  printf '%s%s✓ Olakai is up at %s%s\n' "$C_BOLD" "$C_GREEN" "$dashboard_url" "$C_RESET"
  printf '\n'
  printf '  Magic link sent to %s%s%s.\n' "$C_BOLD" "$ADMIN_EMAIL" "$C_RESET"
  if [[ -n "$fallback_url" ]]; then
    printf "  If it doesn't arrive within a minute, you can also use:\n"
    printf '    %s\n' "$fallback_url"
    printf '\n'
    printf '  This link expires in 24 hours.\n'
  elif [[ "$QUIET" != "true" ]]; then
    # Per D-022 Option B we promise both lines by default. If the grep
    # genuinely found nothing (rare race or non-default bootstrap
    # configuration), tell the operator how to retrieve it manually
    # rather than silently dropping the contract.
    printf "  If the magic link doesn't arrive, retrieve the fallback URL with:\n"
    printf '    cd %s && %sdocker compose logs bootstrap | grep "Setup URL"\n' "$INSTALL_DIR" "$as_user"
  fi
  printf '\n'
  if [[ "$RUNTIME" == "podman-rootless" ]]; then
    printf '  Runtime: rootless Podman. The stack belongs to the %s%s%s user; root has no\n' "$C_BOLD" "$SERVICE_USER" "$C_RESET"
    printf '  containers. Run every later bundle script as that user, from the install dir:\n'
    printf '    cd %s && %s./upgrade.sh --to=X.Y.Z\n' "$INSTALL_DIR" "$as_user"
    printf '  (air-gapped hosts: add --bundle=<tarball>). Same form for ./support-bundle.sh\n'
    printf '  and ./scripts/restore.sh.\n'
    printf '\n'
  fi
  printf '  Docs:    https://docs.olakai.ai/on-prem/install\n'
  printf '  Support: support@olakai.ai\n'
  printf '\n'
  # Machine-readable contract for automation (Ansible `register:` + grep):
  # exactly one line, always the last one on stdout, nothing else on it.
  # Empty when the URL could not be read within the timeout or --quiet is
  # set (the human banner above already explains how to retrieve it).
  printf 'OLAKAI_SETUP_URL=%s\n' "$fallback_url"
}

# ──────────────────────────────────────────────────────────────────────────────
# Main
# ──────────────────────────────────────────────────────────────────────────────

main() {
  parse_args "$@"

  printf '%solakai on-prem installer%s (script v%s, cosign %s)\n\n' \
    "$C_BOLD" "$C_RESET" "$GET_SH_VERSION" "$COSIGN_VERSION"

  WORK_DIR="$(mktemp -d -t olakai-install.XXXXXX)"

  print_step "1/8 Pre-flight"
  # shellcheck disable=SC2119
  detect_os
  check_root
  check_resources
  check_install_dir
  # Managed-state inputs are validated (and --env-file parsed, once) here
  # so a bad path, mode or key fails before the license handshake.
  check_env_file
  check_private_ca
  if [[ "$RUNTIME" == "podman-rootless" ]]; then
    # OS preparation for the rootless runtime. All of it is idempotent and
    # runs BEFORE the handshake so a failure here never burns the
    # single-use license key. Under --no-root it is verified, not performed.
    # shellcheck disable=SC2119
    check_rootless_prereqs
    if [[ "$NO_ROOT" == "true" ]]; then
      # shellcheck disable=SC2119
      check_prepared_host
    else
      check_or_install_podman
      # shellcheck disable=SC2119
      ensure_compose_provider
      # shellcheck disable=SC2119
      ensure_service_user
      # shellcheck disable=SC2119
      ensure_unprivileged_ports
      # shellcheck disable=SC2119
      ensure_user_podman_units
    fi
  else
    check_or_install_docker
  fi

  print_step "2/8 Cosign bootstrap"
  if [[ "$NO_ROOT" == "true" ]]; then
    # $HOME validated by check_prepared_host (_check_no_root_home).
    ensure_cosign "${HOME:-}/.local/bin"
  else
    # shellcheck disable=SC2119
    ensure_cosign
  fi

  print_step "3/8 Inputs"
  collect_inputs

  print_step "4/8 License handshake"
  do_handshake
  # The single-use key has done its job; drop it before download, extract
  # and the install.sh hand-off so nothing downstream can log or inherit it.
  LICENSE_KEY=""

  print_step "5/8 Download + verify"
  download_and_verify

  print_step "6/8 Extract"
  extract_bundle
  check_bundle_runtime_support
  stage_private_ca

  print_step "7/8 .env"
  materialize_env

  print_step "8/8 Bring up"
  bring_up

  print_success_banner
}

main "$@"
