#!/bin/sh
# gpuk — installer + control CLI for the GPU Kitchen host daemon (gpu-kitchen-worker).
#
# Since C1 a compute node runs NO backend. The single privileged host component is
# the Rust `gpu-kitchen-worker` daemon (workerd), running as root DIRECTLY on the
# host (not in a container). It owns NVML clock/power locks, whitelisted host
# browsing, the model-cache staging pipeline, the signed `gpu-kitchen-bench`
# runner, and — on a controller node — the app container's lifecycle (create,
# health-gate, roll back, pull+recreate to update) via its manifest. There is no
# separate host daemon and no unix control socket: workerd is driven over the
# wss+mTLS channel by the controller, and locally by this CLI.
#
# Two roles:
#   worker      a headless compute node. Installs the binary + systemd unit + a
#               manifest (cacheDisks) with NO app container, NO
#               Postgres, NO docker app image. It ENROLLS over mTLS with a
#               enrollment token, or waits for LAN discovery admission, then runs.
#   controller  the full app + UI. Installs the binary + systemd unit + an
#               app-container manifest (image, ports, env, secretsRef) that workerd
#               applies (`gpu-kitchen-worker apply`, then cmd:host_apply in steady
#               state).
#
# Install (root):
#   # controller:
#   curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/gpuk \
#     | sudo sh -s -- install --mode controller \
#       --image repo.byterain.io/gpukitchen/gpukitchen-controller:vX.Y.Z
#   # worker (enroll against a controller with a single-use token from its UI):
#   sudo ./deployments/install/gpuk install --mode worker \
#       --controller wss://controller.lan:8443 --enroll-token gk_enroll_... \
#       --cache-dir /mnt/models --binary apps/worker/target/release/gpu-kitchen-worker
#
# Control:
#   gpuk status | apply | update | enroll | logs | manifest | uninstall
#
set -eu

# ── Paths ────────────────────────────────────────────────────────────────────
# Overridable, so the daemon can be driven against a prefix a normal user owns —
# the only way any of this is testable without handing a test suite root on the
# host. Unset (the real install) they are exactly the systemd defaults workerd uses
# (apps/worker/src/module.rs, apps/worker/src/identity.rs).
BIN_DEST="${GPUK_BIN_DEST:-/usr/local/bin/gpu-kitchen-worker}"
ETC_DIR="${GPUK_ETC_DIR:-/etc/gpu-kitchen}"
MANIFEST="$ETC_DIR/manifest.json"
IDENTITY_DIR="${GPUK_IDENTITY_DIR:-$ETC_DIR/identity}"
MACHINE_ID_FILE="${GPUK_MACHINE_ID_FILE:-$ETC_DIR/machine-id}"
WORKER_ENV="$ETC_DIR/worker.env"
SERVICE_NAME="gpu-kitchen-worker"
UNIT_DEST="${GPUK_UNIT_DEST:-/etc/systemd/system/$SERVICE_NAME.service}"

# Where to download the binary from when no --binary is given.
GPUK_RELEASE_BASE="${GPUK_RELEASE_BASE:-}"

die() { echo "gpuk: $*" >&2; exit 1; }

# Root, or able to do the job anyway. Fail with a clear message BEFORE touching
# /etc, /usr/local/bin or systemd. Against a user-owned prefix it simply works.
need_root() {
  [ "$(id -u)" -eq 0 ] && return 0
  [ -w "$ETC_DIR" ] && return 0
  die "this command must run as root (use sudo)"
}

# ── JSON helpers (controlled inputs: paths + identifiers) ──────────────────────
json_str() { printf '%s' "$1" | sed 's/\\/\\\\/g; s/"/\\"/g'; }

json_array() { # args → ["a","b",...]
  _out=""
  for _p in "$@"; do
    _e=$(json_str "$_p")
    if [ -z "$_out" ]; then _out="\"$_e\""; else _out="$_out,\"$_e\""; fi
  done
  printf '[%s]' "$_out"
}

# ── install ────────────────────────────────────────────────────────────────────
arch_asset() {
  case "$(uname -m)" in
    x86_64|amd64)  echo "gpu-kitchen-worker-x86_64" ;;
    aarch64|arm64) echo "gpu-kitchen-worker-aarch64" ;;
    *) die "unsupported architecture: $(uname -m)" ;;
  esac
}

install_binary() { # [local-path]
  if [ -n "${1:-}" ]; then
    [ -f "$1" ] || die "binary not found: $1"
    install -m 0755 "$1" "$BIN_DEST"
    echo "==> installed $BIN_DEST from $1"
  elif [ -n "$GPUK_RELEASE_BASE" ]; then
    command -v curl >/dev/null || die "curl is required to download the binary"
    _url="$GPUK_RELEASE_BASE/$(arch_asset)"
    echo "==> downloading $_url"
    curl -fsSL "$_url" -o "$BIN_DEST.new"
    chmod 0755 "$BIN_DEST.new"
    mv "$BIN_DEST.new" "$BIN_DEST"
  elif [ -x "$BIN_DEST" ]; then
    echo "==> reusing existing $BIN_DEST"
  else
    die "no binary: pass --binary <path> or set GPUK_RELEASE_BASE=<url base>"
  fi
}

gen_secret() { head -c 32 /dev/urandom | base64 | tr '+/' '-_' | tr -d '='; }

seed_secrets() { # DATA_ROOT
  _sd="$1/secrets"
  mkdir -p "$_sd"; chmod 0700 "$_sd"
  [ -f "$_sd/encryption_key" ] || { umask 077; gen_secret > "$_sd/encryption_key"; }
  [ -f "$_sd/node_id" ] || { umask 077; (cat /proc/sys/kernel/random/uuid 2>/dev/null || uuidgen) > "$_sd/node_id"; }
  # No nominal first-run password is generated. An operator may pre-provision
  # this documented break-glass file; only then is it injected into the app.
  [ ! -f "$_sd/bootstrap_admin_password" ] || chmod 0600 "$_sd/bootstrap_admin_password"
}

prepare_claim_code() {
  # The file is the operator's recoverable proof of machine possession. Create
  # it once, preserve it across reinstalls, and let the backend unlink it after
  # the atomic first-account claim. A missing file beside an existing manifest
  # therefore means "consumed", never "rotate the credential".
  if [ -f "$CLAIM_CODE_FILE" ]; then
    chmod 0600 "$CLAIM_CODE_FILE"
  elif [ ! -f "$MANIFEST" ]; then
    umask 077
    gen_secret > "$CLAIM_CODE_FILE"
    chmod 0600 "$CLAIM_CODE_FILE"
  fi
}

# Write the systemd unit generated from the worker's canonical template. The
# generator injects the Rust lock-contention exit code here too, so the binary
# and systemd restart policy cannot silently drift apart.
write_unit() {
  # BEGIN GENERATED WORKER SYSTEMD UNIT
  cat > "$UNIT_DEST" <<UNIT
[Unit]
Description=GPU Kitchen worker daemon (gpu-kitchen-worker)
Documentation=https://repo.byterain.io/Sebastien/GPU-Manager
After=network-online.target docker.service
Wants=network-online.target docker.service

[Service]
Type=simple
# The worker runs as root on the host (NVML, docker.sock, mounts) — NOT in a
# container. Identity (keypair/cert/CA) lives under $IDENTITY_DIR; enroll once
# before starting this unit. Optional overrides live in $WORKER_ENV.
EnvironmentFile=-$WORKER_ENV
ExecStart=$BIN_DEST
# WRK-173: the old MainPID transfers supervision to the verified replacement
# before it exits. Notifications remain local to this service cgroup.
NotifyAccess=all
# Transient operation state only. The node lock has its own stable inode under
# /run/lock, outside this systemd-managed directory.
RuntimeDirectory=gpu-kitchen
RuntimeDirectoryMode=0750
Restart=always
RestartSec=2
# Lock contention is an operator error, not a crash: do not retry forever while
# another directly installed worker owns the WRK-167 host lock.
RestartPreventExitStatus=75
User=root
# Fail-closed renewal: on a refused cert renewal the daemon exits and systemd
# restarts it into an enroll-required state.
KillSignal=SIGTERM
TimeoutStopSec=15

[Install]
WantedBy=multi-user.target
UNIT
  # END GENERATED WORKER SYSTEMD UNIT
}

# TLS reverse-proxy example, FILLED with the operator's domain (INS-47) — written
# only under `--profile public --domain <d>`. Kept aligned with
# deployments/controller/Caddyfile.example (the compose variant); this copy
# targets the all-in-one image, where nginx on the UI port is the single front
# door (INS-09) so one upstream carries pages, /api and the /ws upgrade alike.
# We write a file and NOTHING more: no package install, no service start, no
# other program's config read or touched — putting the proxy in service stays an
# operator act (OPS-13), and its presence stays unverifiable (OPS-67).
write_caddyfile() {
  mkdir -p "$DATA_ROOT/caddy"
  cat > "$DATA_ROOT/caddy/Caddyfile" <<CADDY
# TLS in front of GPU Kitchen — generated by the installer for --profile public
# (specs/plateforme/installation.md INS-47). Caddy provisions and renews the
# certificate itself once DNS for $DOMAIN points at this machine.
#
#   sudo cp $DATA_ROOT/caddy/Caddyfile /etc/caddy/Caddyfile
#   sudo systemctl reload caddy
#
# The app already runs with GPUK_HSTS=true and GPUK_SESSION_COOKIE_SECURE=true
# (set by the public profile). What does NOT go through this proxy:
#   - the worker mTLS channel (:8443): workers pin the controller CA and must
#     reach it DIRECTLY — terminating it here would break the pin.
#   - worker<->worker data transfers (:8300): LAN-only by contract (OPS-68).

$DOMAIN {
	encode zstd gzip
	reverse_proxy localhost:$HTTP_PORT
}

# OpenAI-compatible inference endpoint (gpuk-proxy) — uncomment when inference
# clients live beyond the trusted LAN; TLS keeps their API keys off the wire.
#
# inference.$DOMAIN {
# 	encode zstd gzip
# 	reverse_proxy localhost:8200
# }
CADDY
  chmod 0644 "$DATA_ROOT/caddy/Caddyfile"
  echo "==> wrote $DATA_ROOT/caddy/Caddyfile (filled TLS reverse-proxy example for $DOMAIN)"
}

# Pre-existing model caches (B81): a server that installs GPU Kitchen usually already
# holds tens or hundreds of GB of weights. We print a hint and nothing more — cache
# questions belong to the first-run wizard (INS-48, REG-41), never to the CLI; no
# config of any other program is read or touched, and referencing a cache stays an
# explicit choice made in the UI.
hint_existing_caches() {
  hec_chosen=$(readlink -f "$1" 2>/dev/null || echo "$1")
  hec_found=""
  for hec_dir in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do
    [ -n "$hec_dir" ] || continue
    [ -d "$hec_dir/hub" ] || continue
    hec_real=$(readlink -f "$hec_dir" 2>/dev/null || echo "$hec_dir")
    [ "$hec_real" != "$hec_chosen" ] || continue
    # Hub layout only (models--*) — matches what the scan can actually reference.
    ls -d "$hec_dir"/hub/models--* >/dev/null 2>&1 || continue
    hec_found="$hec_found $hec_real"
  done
  [ -n "$hec_found" ] || return 0
  echo
  for hec_dir in $hec_found; do
    echo "==> existing model cache found at $hec_dir"
  done
  echo "    GPU Kitchen will offer to reuse those models at first launch, and any"
  echo "    time from Nodes & GPU -> Storage. A reused cache is referenced in"
  echo "    place. Nothing is moved or deleted."
}

cmd_install() {
  IMAGE=""; MODE="worker"; CLUSTER="default"; DATA_ROOT="/var/lib/gpu-kitchen"
  CACHE_DIR=""; NETWORK="host"; CONTROLLER_URL=""; BIN_SRC=""; HEALTH_PORT="8001"
  # 1337, not 8080: kept in lockstep with install.sh's default (INS-01).
  ENROLL_TOKEN=""; HTTP_PORT="1337"; PROFILE=""; DOMAIN=""; DRY_RUN=0
  while [ $# -gt 0 ]; do
    case "$1" in
      --image)        IMAGE="$2"; shift 2 ;;
      --mode)         MODE="$2"; shift 2 ;;
      --cluster)      CLUSTER="$2"; shift 2 ;;
      --profile)      PROFILE="$2"; shift 2 ;;
      --domain)       DOMAIN="$2"; shift 2 ;;
      --data-root)    DATA_ROOT="$2"; shift 2 ;;
      --cache-dir)    CACHE_DIR="$2"; shift 2 ;;
      --network)      NETWORK="$2"; shift 2 ;;
      --controller)   CONTROLLER_URL="$2"; shift 2 ;;
      # `--token` is an ENROLLMENT token (single-use or shared), NOT a bearer:
      # the worker↔controller channel is cert-only since D11. `--token` is kept as
      # a spelling of `--enroll-token`.
      --enroll-token|--token) ENROLL_TOKEN="$2"; shift 2 ;;
      --health-port)  HEALTH_PORT="$2"; shift 2 ;;
      --http-port)    HTTP_PORT="$2"; shift 2 ;;
      --binary)       BIN_SRC="$2"; shift 2 ;;
      --dry-run)      DRY_RUN=1; shift ;;
      *) die "unknown install option: $1" ;;
    esac
  done

  # Roles, as the backend names them (multi-server/config.ts): "controller" (full
  # app + UI, accepts workers) and "worker" (headless compute node). Historical
  # spellings still work.
  case "$MODE" in
    controller|server|standalone|manager) MODE="controller" ;;
    worker|agent) MODE="worker" ;;
    *) die "--mode must be controller or worker (got '$MODE')" ;;
  esac

  if [ "$MODE" = "worker" ] && [ -n "$PROFILE" ]; then
    die "--profile applies only to --mode controller; workers do not have an installation profile"
  fi
  case "$PROFILE" in
    ""|homelab|studio|enterprise|public) ;;
    *) die "--profile must be homelab, studio, enterprise or public (got '$PROFILE')" ;;
  esac

  if [ -n "$DOMAIN" ]; then
    [ "$PROFILE" = "public" ] \
      || die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)"
    case "$DOMAIN" in
      *[!A-Za-z0-9.-]*) die "--domain must be a bare domain name (got '$DOMAIN')" ;;
    esac
  fi

  if [ "$MODE" = "controller" ]; then
    if [ -z "$IMAGE" ] && [ "$DRY_RUN" -eq 1 ]; then
      IMAGE="example.invalid/gpukitchen-controller:v0.0.0-dry-run"
    fi
    [ -n "$IMAGE" ] || die "--image registry/gpukitchen-controller:<tag> is required for a controller"
    case "$IMAGE" in
      *@sha256:*) ;;
      *:latest) die "refusing floating image tag '$IMAGE' — use an explicit release tag or digest" ;;
      *)
        _image_tag="${IMAGE##*:}"
        case "$_image_tag" in
          "$IMAGE"|*/*) die "controller image must carry an explicit tag or sha256 digest (got '$IMAGE')" ;;
        esac
        ;;
    esac
  fi
  [ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf"

  CACHE_JSON="[{\"hostPath\":\"$(json_str "$CACHE_DIR")\",\"shared\":false}]"
  SELF_ENROLL_FILE="$DATA_ROOT/self-enroll-token"
  BOOTSTRAP_PASSWORD_FILE="$DATA_ROOT/secrets/bootstrap_admin_password"
  CLAIM_CODE_FILE="$DATA_ROOT/secrets/claim_code"
  CLAIM_CODE_AVAILABLE=0

  if [ "$DRY_RUN" -eq 1 ]; then
    [ "$MODE" != "controller" ] || CLAIM_CODE_AVAILABLE=1
    echo "==> dry run: no file, service or container was changed"
    if [ "$MODE" = "worker" ]; then render_worker_manifest; else render_controller_manifest; fi
    [ -z "$DOMAIN" ] || echo "==> dry run: would write $DATA_ROOT/caddy/Caddyfile for $DOMAIN"
    return 0
  fi

  # ── Port conflicts (INS-46) — the mutator's own guard ────────────────────────
  # install.sh's preflight already checks these, and on a terminal it can offer
  # an alternative port. gpuk is the actual mutator and contributors call it
  # DIRECTLY, so it re-checks and refuses, non-interactively. Same helpers as
  # install.sh (both scripts ship standalone from the channel). A listener owned
  # by an existing install is not a conflict: a manifest on disk means the
  # re-run is the update path, and every checked port is then our own.
  if [ ! -f "$MANIFEST" ]; then
    _port_tool=""
    if command -v ss >/dev/null 2>&1; then _port_tool="ss"
    elif command -v netstat >/dev/null 2>&1; then _port_tool="netstat"; fi
    if [ -z "$_port_tool" ]; then
      echo "==> warning: cannot check for port conflicts (no ss or netstat)"
    else
      if [ "$MODE" = "controller" ]; then
        set -- "$HTTP_PORT" 8443 8200
      else
        # Worker data-plane ports (OPS-68) plus the local health listener.
        set -- "$HEALTH_PORT" 8300 8301 8302
      fi
      for _p in "$@"; do
        _busy=1
        case "$_port_tool" in
          ss) [ -n "$(ss -ltnH "sport = :$_p" 2>/dev/null)" ] || _busy=0 ;;
          netstat)
            netstat -ltn 2>/dev/null \
              | awk -v p="$_p" '{n=split($4,a,":"); if (a[n]==p) found=1} END{exit found?0:1}' \
              || _busy=0
            ;;
        esac
        [ "$_busy" -eq 0 ] || die "port $_p is already in use. Free it first, then run the install again."
      done
    fi
  fi

  need_root
  install_binary "$BIN_SRC"

  mkdir -p "$ETC_DIR" "$DATA_ROOT" "$CACHE_DIR"
  seed_secrets "$DATA_ROOT"
  if [ "$MODE" = "controller" ]; then
    prepare_claim_code
    [ ! -s "$CLAIM_CODE_FILE" ] || CLAIM_CODE_AVAILABLE=1
  fi

  umask 077
  if [ "$MODE" = "worker" ]; then
    write_worker_manifest
    # Optional env overrides for the daemon (cluster grouping, display name, an
    # explicit controller URL). The enrolled identity carries the controller URL
    # + CA too — this is belt-and-braces / pre-enroll discovery grouping.
    {
      echo "GPUK_CLUSTER=$CLUSTER"
      echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT"
      echo "GPUK_MACHINE_ID_FILE=$MACHINE_ID_FILE"
      echo "NODE_DISPLAY_NAME=$(hostname)"
      [ -n "$CONTROLLER_URL" ] && echo "GPUK_CONTROLLER_URLS=$CONTROLLER_URL"
    } > "$WORKER_ENV"
    chmod 0600 "$WORKER_ENV"
    echo "==> wrote $MANIFEST (worker: no app container)"
  else
    write_controller_manifest
    # The host daemon and app container share DATA_ROOT. Only controller-mode
    # workerd gets this private bootstrap channel; remote workers stay tokenless.
    {
      echo "GPUK_CLUSTER=$CLUSTER"
      echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT"
      echo "GPUK_MACHINE_ID_FILE=$MACHINE_ID_FILE"
      echo "GPUK_CONTROLLER_URLS=wss://127.0.0.1:8443"
      echo "GPUK_SELF_ENROLL_FILE=$SELF_ENROLL_FILE"
      echo "NODE_DISPLAY_NAME=$(hostname)"
    } > "$WORKER_ENV"
    chmod 0600 "$WORKER_ENV"
    echo "==> wrote $MANIFEST (controller: app container $IMAGE)"
  fi
  chmod 0600 "$MANIFEST"
  [ -z "$DOMAIN" ] || write_caddyfile

  write_unit
  systemctl daemon-reload
  echo "==> wrote $UNIT_DEST"

  # ── Worker: token enrollment before start, or unattended LAN discovery ──
  if [ "$MODE" = "worker" ]; then
    if [ -n "$ENROLL_TOKEN" ]; then
      [ -n "$CONTROLLER_URL" ] || die "--controller wss://<controller>:<port> is required to enroll"
      echo "==> enrolling against $CONTROLLER_URL ..."
      GPUK_IDENTITY_DIR="$IDENTITY_DIR" GPUK_MACHINE_ID_FILE="$MACHINE_ID_FILE" "$BIN_DEST" enroll \
        --controller "$CONTROLLER_URL" --token "$ENROLL_TOKEN" \
        || die "enrollment failed (bad/expired token, or controller unreachable)"
    else
      if [ -n "$CONTROLLER_URL" ]; then
        echo "==> no token given: the daemon will request admission from $CONTROLLER_URL and wait"
        echo "    for automatic admission or administrator approval."
      else
        echo "==> no token or controller given: the daemon will discover its cluster on the LAN and wait"
        echo "    for automatic admission or administrator approval."
      fi
    fi
    systemctl enable --now "$SERVICE_NAME"
    echo "==> $SERVICE_NAME enabled and started"
    echo
    echo "Done. The worker daemon is running${ENROLL_TOKEN:+ and enrolled}."
    echo "  Status : gpuk status        Logs: gpuk logs"
    hint_existing_caches "$CACHE_DIR"
    return 0
  fi

  # ── Controller: start the daemon, then bring up the app container ──
  systemctl enable --now "$SERVICE_NAME"
  echo "==> $SERVICE_NAME enabled and started"
  echo "==> applying manifest (first app-container start) ..."
  # No unix socket any more: workerd reconciles the app container in-process from
  # the on-disk manifest (there is no backend to relay through on the very first
  # boot). Steady-state updates go through the daemon over WS.
  "$BIN_DEST" apply || die "apply failed. Check: gpuk logs"
  echo
  echo "Done. The worker daemon is running and the app container is up."
  echo "  Status : gpuk status        App logs: gpuk logs        Daemon: journalctl -u $SERVICE_NAME"
  hint_existing_caches "$CACHE_DIR"
}

# Worker manifest: cacheDisks and an EMPTY image so
# has_app_container() is false (apps/worker/src/modules/manifest.rs). No ports, no env,
# no secretsRef — a worker runs no app container.
render_worker_manifest() {
  cat <<JSON
{
  "schemaVersion": 1,
  "image": "",
  "mode": "worker",
  "cluster": "$(json_str "$CLUSTER")",
  "gpus": "all",
  "dataRoot": "$(json_str "$DATA_ROOT")",
  "cacheDisks": $CACHE_JSON
}
JSON
}

write_worker_manifest() { render_worker_manifest > "$MANIFEST"; }

# Controller manifest: the declarative app-container description workerd applies.
render_controller_manifest() {
  # NOTE there is deliberately no PORT here: PORT is the backend's own port, a
  # loopback-only 8000 behind nginx in the all-in-one image. What the outside world
  # dials is GPUK_PUBLIC_PORT (the published host port), which falls back to nginx's
  # own GPUK_PORT when nothing republishes it.
  EXTRA_ENV=""
  EXTRA_ENV="$EXTRA_ENV,\"GPUK_DATA_ROOT\":\"$(json_str "$DATA_ROOT")\""
  [ -z "$PROFILE" ] || EXTRA_ENV="$EXTRA_ENV,\"GPUK_INSTALL_PROFILE\":\"$(json_str "$PROFILE")\""
  if [ "$PROFILE" = "public" ]; then
    EXTRA_ENV="$EXTRA_ENV,\"GPUK_HSTS\":\"true\""
    EXTRA_ENV="$EXTRA_ENV,\"GPUK_SESSION_COOKIE_SECURE\":\"true\""
  fi
  BOOTSTRAP_SECRET_JSON=""
  if [ -s "$BOOTSTRAP_PASSWORD_FILE" ]; then
    EXTRA_ENV="$EXTRA_ENV,\"GPUK_BOOTSTRAP_MUST_CHANGE\":\"1\""
    BOOTSTRAP_SECRET_JSON=",\"GPUK_BOOTSTRAP_ADMIN_PASSWORD\":\"$(json_str "$BOOTSTRAP_PASSWORD_FILE")\""
  fi
  CLAIM_SECRET_JSON=""
  if [ "$CLAIM_CODE_AVAILABLE" -eq 1 ]; then
    CLAIM_SECRET_JSON=",\"GPUK_CLAIM_CODE\":\"$(json_str "$CLAIM_CODE_FILE")\""
  fi
  [ "$NETWORK" = "host" ] || EXTRA_ENV="$EXTRA_ENV,\"BACKEND_DOCKER_NETWORK\":\"$(json_str "$NETWORK")\""

  # Published != bound (INS-43). On a bridged network the container keeps the image's
  # FIXED listeners — nginx 8080, worker mTLS 8443 — and --http-port only moves the HOST
  # side of the publication; Settings -> Network moves it later by patching this same
  # manifest, so the container port must never become a variable. With host networking
  # nothing is published and the listener itself takes the port.
  PORTS_JSON="{}"
  if [ "$NETWORK" = "host" ]; then
    EXTRA_ENV="$EXTRA_ENV,\"GPUK_PORT\":\"$(json_str "$HTTP_PORT")\""
  else
    EXTRA_ENV="$EXTRA_ENV,\"GPUK_PORT\":\"8080\""
    EXTRA_ENV="$EXTRA_ENV,\"GPUK_PUBLIC_PORT\":\"$(json_str "$HTTP_PORT")\""
    PORTS_JSON="{\"8080\":$HTTP_PORT,\"8200\":8200,\"8443\":8443}"
  fi

  cat <<JSON
{
  "schemaVersion": 1,
  "image": "$(json_str "$IMAGE")",
  "containerName": "gpu-kitchen",
  "mode": "controller",
  "cluster": "$(json_str "$CLUSTER")",
  "networkMode": "$(json_str "$NETWORK")",
  "gpus": "all",
  "restartPolicy": "unless-stopped",
  "dataRoot": "$(json_str "$DATA_ROOT")",
  "cacheDisks": $CACHE_JSON,
  "ports": $PORTS_JSON,
  "env": {
    "NODE_ENV": "production",
    "GPUK_MODE": "controller",
    "GPUK_CLUSTER": "$(json_str "$CLUSTER")",
    "GPUK_SELF_ENROLL_FILE": "$(json_str "$SELF_ENROLL_FILE")",
    "NODE_DISPLAY_NAME": "$(json_str "$(hostname)")",
    "HF_HOME": "$(json_str "$CACHE_DIR")"$EXTRA_ENV
  },
  "secretsRef": {
    "ENCRYPTION_KEY": "$(json_str "$DATA_ROOT/secrets/encryption_key")",
    "NODE_ID": "$(json_str "$DATA_ROOT/secrets/node_id")"$BOOTSTRAP_SECRET_JSON$CLAIM_SECRET_JSON
  }
}
JSON
}

write_controller_manifest() { render_controller_manifest > "$MANIFEST"; }

# ── control subcommands ────────────────────────────────────────────────────────
container_name() {
  sed -n 's/.*"containerName"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1
}

manifest_data_root() {
  sed -n 's/.*"dataRoot"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1
}

manifest_image() {
  sed -n 's/.*"image"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1
}

health_port() {
  sed -n 's/.*GPUK_WORKER_HEALTH_PORT=\([0-9]*\).*/\1/p' "$WORKER_ENV" 2>/dev/null | head -1
}

# Split an image reference into repository and tag.
#
# `${img%%:*}` cuts at the FIRST colon and is WRONG:
# `registry.internal:5000/gpuk/controller:v1.2.3` would yield repo
# `registry.internal`. The colon in a registry's host:port is not a tag separator.
# Rule: it is a tag only if the last colon comes after the last slash.
# (Same logic as apps/controller/api/src/core/release-channel.ts — see its unit tests.)
image_repo() {
  # A digest suffix (…@sha256:…) never carries the repo; drop it, then apply
  # the tag logic — `repo:tag@sha256:…` and `repo@sha256:…` both reduce right.
  set -- "${1%@*}"
  _t="${1##*:}"
  case "$_t" in
    "$1") printf '%s' "$1" ;;    # no colon at all → untagged
    */*)  printf '%s' "$1" ;;    # the last colon is inside a path → host:port, untagged
    *)    printf '%s' "${1%:*}" ;;
  esac
}

image_tag() {
  # `repo:tag@sha256:…` keeps the human-readable tag next to the content pin —
  # docker resolves by digest and ignores the tag. Strip the digest, then parse.
  set -- "${1%@*}"
  _t="${1##*:}"
  case "$_t" in
    "$1") return ;;
    */*)  return ;;
    *)    printf '%s' "$_t" ;;
  esac
}

# The release channel: one flat JSON document served next to the installer. The
# UI's update check reads the same one (apps/controller/api/src/core/release-channel.ts) — one
# source of truth. Interim default: the public Gitea channel repo — flips to
# https://gpu.kitchen/latest.json once the hub exists (specs/developpement/ci-cd.md).
CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen/channel/raw/branch/main/latest.json}"

# The release public key pinned in THIS copy of gpuk (OPS-20). The channel copy
# gets the real key substituted at publish time; the operator override
# (GPUK_UPDATE_PUBKEY) covers a self-hosted channel with its own keypair. The
# first install fetched gpuk itself over HTTPS from the channel — that moment is
# trust-on-first-use, like a worker's enrolment token pin; every later `update`
# is verified against the key pinned HERE, so whoever controls latest.json can
# no longer pick what an existing install runs.
CHANNEL_PUBKEY="${GPUK_UPDATE_PUBKEY:-RWQ7BKXJqGX2jdKXu1GxeSPVAN3JDRTefpImM/mFRjtFwq4E7mhnQtA7}"

# Fetch latest.json AND its minisign signature, verify, and leave the verified
# document at $CHANNEL_DOC. Fail-closed: no signature, bad signature, no
# minisign CLI or no pinned key are all fatal — GPUK_CHANNEL_INSECURE=1 is the
# explicit, logged opt-out (a private mirror that does not sign).
CHANNEL_DOC=""
channel_fetch() {
  command -v curl >/dev/null 2>&1 || return 1
  CHANNEL_DOC=$(mktemp) || return 1
  curl -fsSL --max-time 20 "$CHANNEL_URL" -o "$CHANNEL_DOC" 2>/dev/null || return 1
  if [ "${GPUK_CHANNEL_INSECURE:-}" = "1" ]; then
    echo "WARNING: GPUK_CHANNEL_INSECURE=1 — release channel signature NOT verified" >&2
    return 0
  fi
  case "$CHANNEL_PUBKEY" in
    ""|RWQ7BKXJqGX2jdKXu1GxeSPVAN3JDRTefpImM/mFRjtFwq4E7mhnQtA7*)
      die "this gpuk carries no pinned release public key — set GPUK_UPDATE_PUBKEY (the minisign public-key line), or GPUK_CHANNEL_INSECURE=1 to skip verification" ;;
  esac
  command -v minisign >/dev/null 2>&1 \
    || die "minisign is required to verify the release channel (apt install minisign), or set GPUK_CHANNEL_INSECURE=1"
  _sig=$(mktemp)
  if ! curl -fsSL --max-time 20 "${CHANNEL_URL}.minisig" -o "$_sig" 2>/dev/null; then
    rm -f "$_sig"
    die "no signature at ${CHANNEL_URL}.minisig — refusing an unsigned channel document (OPS-20)"
  fi
  if ! minisign -Vq -m "$CHANNEL_DOC" -x "$_sig" -P "$CHANNEL_PUBKEY" >/dev/null 2>&1; then
    rm -f "$_sig"
    die "latest.json signature verification FAILED — refusing the channel document (OPS-20)"
  fi
  rm -f "$_sig"
}

channel_field() {
  sed -n 's/.*"'"$2"'"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$1" | head -1
}

channel_version() {
  channel_fetch || return 1
  channel_field "$CHANNEL_DOC" version
}

# ── backup ─────────────────────────────────────────────────────────────────────
# OPS-10: the embedded Postgres is only backed up COLD — hot-copying pgdata with
# a file tool is forbidden (torn pages). Order matters: stop the DAEMON first
# (its reconciler would immediately restart a stopped app container), then the
# container, snapshot, and restarting the service re-applies the manifest.
# An install on an external DATABASE_URL is refused here: gpuk only owns the
# embedded pgdata — back the real database up with pg_dump/backup-compose.sh.
# The finished directory still has to be copied to encrypted off-host storage,
# next to the recovery set (OPS-04/OPS-05: ENCRYPTION_KEY above all).
cmd_backup() {
  need_root
  _img=$(manifest_image)
  [ -n "$_img" ] || die "this is a worker node — no controller data to back up here"
  if grep -q '"DATABASE_URL"' "$MANIFEST" 2>/dev/null; then
    die "this install uses an external DATABASE_URL — back THAT database up (pg_dump, or the compose procedures in specs/plateforme/operations.md); gpuk backup only snapshots the embedded pgdata"
  fi
  _root=$(manifest_data_root)
  [ -n "$_root" ] || die "no dataRoot in $MANIFEST"
  [ -d "$_root/pgdata" ] || die "no embedded pgdata under $_root — nothing to snapshot"
  command -v sha256sum >/dev/null 2>&1 || die "sha256sum is required"

  _out="${1:-$_root/backups/$(date -u +%Y%m%dT%H%M%SZ)}"
  [ -e "$_out" ] && die "refusing to overwrite existing $_out"
  mkdir -p "$(dirname "$_out")"
  _tmp="$_out.partial"
  rm -rf "$_tmp"; mkdir -p "$_tmp"

  _cn=$(container_name)
  echo "==> Stopping $SERVICE_NAME (its reconciler would restart the container mid-snapshot)..."
  systemctl stop "$SERVICE_NAME" || die "could not stop $SERVICE_NAME"
  _restart_daemon() { systemctl start "$SERVICE_NAME" 2>/dev/null || true; }
  trap _restart_daemon EXIT
  if [ -n "$_cn" ]; then
    echo "==> Stopping $_cn (cold snapshot — OPS-10)..."
    docker stop "$_cn" >/dev/null 2>&1 || true
    _state=$(docker inspect -f '{{.State.Status}}' "$_cn" 2>/dev/null || echo absent)
    case "$_state" in
      running) die "container $_cn is still running — refusing a hot snapshot" ;;
    esac
  fi

  echo "==> Snapshotting $_root/pgdata..."
  tar -C "$_root" -czf "$_tmp/pgdata.tar.gz" pgdata || die "snapshot failed"
  cp "$MANIFEST" "$_tmp/host-manifest.json" 2>/dev/null || true
  {
    echo "{"
    echo "  \"created_utc\": \"$(date -u +%Y-%m-%dT%H:%M:%SZ)\","
    echo "  \"image\": \"$(json_str "$_img")\","
    echo "  \"data_root\": \"$(json_str "$_root")\","
    echo "  \"kind\": \"cold-pgdata-snapshot\""
    echo "}"
  } > "$_tmp/backup-manifest.json"
  (cd "$_tmp" && sha256sum ./* > SHA256SUMS) || die "checksums failed"
  chmod 0700 "$_tmp"
  mv "$_tmp" "$_out"

  echo "==> Restarting $SERVICE_NAME (re-applies the manifest, container included)..."
  systemctl start "$SERVICE_NAME" || die "could not restart $SERVICE_NAME — start it manually"
  trap - EXIT
  echo "backup: $_out"
  echo "Copy it to encrypted OFF-HOST storage together with the recovery set"
  echo "(ENCRYPTION_KEY above all — without it the data is unrecoverable, OPS-04)."
  echo "A physical pgdata restore requires the same Postgres major and a throwaway"
  echo "host rehearsal first (OPS-10)."
}

# ── channel ────────────────────────────────────────────────────────────────────
# Diagnostic (no root): fetch + VERIFY the channel document, print what it
# offers. Exercises exactly the trust chain `update` relies on — the CI probes
# it with a throwaway keypair, an operator uses it to debug a mirror.
cmd_channel() {
  channel_fetch || die "cannot fetch $CHANNEL_URL"
  echo "channel   : $CHANNEL_URL"
  echo "version   : $(channel_field "$CHANNEL_DOC" version)"
  _d=$(channel_field "$CHANNEL_DOC" controllerImageDigest)
  [ -n "$_d" ] && echo "digest    : $_d"
  _d=$(channel_field "$CHANNEL_DOC" controllerImageDigestEnterprise)
  [ -n "$_d" ] && echo "digest ee : $_d"
  if [ "${GPUK_CHANNEL_INSECURE:-}" = "1" ]; then
    echo "signature : SKIPPED (GPUK_CHANNEL_INSECURE=1)"
  else
    echo "signature : verified"
  fi
}

# ── status ─────────────────────────────────────────────────────────────────────
# No control socket any more. Status = the systemd unit state + the daemon's own
# /health endpoint + whether a worker has enrolled (identity present).
cmd_status() {
  _active=$(systemctl is-active "$SERVICE_NAME" 2>/dev/null || true)
  echo "service   : $_active"
  if [ -f "$IDENTITY_DIR/identity.json" ]; then
    echo "enrolled  : yes ($IDENTITY_DIR)"
  else
    echo "enrolled  : no (daemon waits for LAN admission; token enrollment is also available)"
  fi
  _hp=$(health_port); [ -n "$_hp" ] || _hp=8001
  if command -v curl >/dev/null 2>&1; then
    _h=$(curl -fsS --max-time 3 "http://127.0.0.1:$_hp/health" 2>/dev/null || true)
    [ -n "$_h" ] && echo "health    : $_h" || echo "health    : (no answer on :$_hp)"
  fi
  _img=$(manifest_image)
  if [ -n "$_img" ]; then
    echo "app image : $_img"
    echo "app cont. : $(docker inspect -f '{{.State.Status}}' "$(container_name)" 2>/dev/null || echo 'not running')"
  else
    echo "role      : worker (no app container)"
  fi
}

# ── enroll ─────────────────────────────────────────────────────────────────────
cmd_enroll() {
  need_root
  _url=""; _tok=""
  while [ $# -gt 0 ]; do
    case "$1" in
      --controller)               _url="$2"; shift 2 ;;
      --enroll-token|--token)     _tok="$2"; shift 2 ;;
      *) die "unknown enroll option: $1" ;;
    esac
  done
  [ -n "$_url" ] || die "enroll needs --controller wss://<controller>:<port>"
  [ -n "$_tok" ] || die "enroll needs --token gk_enroll_..."
  GPUK_IDENTITY_DIR="$IDENTITY_DIR" GPUK_MACHINE_ID_FILE="$MACHINE_ID_FILE" \
    "$BIN_DEST" enroll --controller "$_url" --token "$_tok"
  systemctl restart "$SERVICE_NAME" 2>/dev/null || true
}

# ── apply ──────────────────────────────────────────────────────────────────────
# Controller: reconcile the app container from the manifest, in-process. A worker
# has no app container — `apply` there is a no-op with a clear message.
cmd_apply() {
  need_root
  "$BIN_DEST" apply
}

# ── update ─────────────────────────────────────────────────────────────────────
# Controller: decide WHICH image TAG to pin, write it into the manifest, then let
# workerd pull + recreate (health-gate + rollback are the daemon's — apply()).
# Worker: the signed-binary self-update is DRIVEN FROM THE CONTROLLER (its Update
# button → POST /api/nodes/:id/host/update → cmd:host_update → verified swap).
# There is no local unverified swap path.
cmd_update() {
  need_root
  _want=""; _check=0
  while [ $# -gt 0 ]; do
    case "$1" in
      --version) _want="$2"; shift 2 ;;
      --check)   _check=1; shift ;;
      *) die "unknown update option: $1" ;;
    esac
  done

  _image=$(manifest_image)
  if [ -z "$_image" ]; then
    echo "This is a worker node. Worker self-update is driven from the controller UI"
    echo "(the node's Update button), which pushes a minisign-verified binary swap."
    return 0
  fi

  _repo=$(image_repo "$_image")
  _current=$(image_tag "$_image")
  [ -n "$_current" ] || _current="(untagged)"

  if [ "$_check" -eq 1 ]; then
    _latest=$(channel_version) || true
    echo "installed : $_current"
    if [ -z "$_latest" ]; then
      echo "available : unknown (cannot reach $CHANNEL_URL)"
      exit 1
    fi
    echo "available : $_latest"
    if [ "$_latest" = "$_current" ]; then echo "up to date."; else echo "run 'gpuk update' to move to $_latest"; fi
    return 0
  fi

  _digest=""
  if [ -z "$_want" ]; then
    # channel_fetch runs in THIS shell (not a $(…) subshell) so a signature
    # failure is fatal here — fail-closed — and $CHANNEL_DOC survives. The
    # verified signature closes the document half of OPS-20; the digest read
    # from it pins CONTENT, closing the mutable-tag half.
    if channel_fetch; then
      _want=$(channel_field "$CHANNEL_DOC" version)
    fi
    if [ -z "$_want" ]; then
      echo "==> cannot reach $CHANNEL_URL — re-applying the pinned $_current"
      "$BIN_DEST" apply
      return 0
    fi
    # Pick the digest matching the installed edition by image basename — the
    # repo itself may be a mirror, the basename is the edition marker.
    case "${_repo##*/}" in
      *-ee) _digest=$(channel_field "$CHANNEL_DOC" controllerImageDigestEnterprise) ;;
      *)    _digest=$(channel_field "$CHANNEL_DOC" controllerImageDigest) ;;
    esac
  fi

  _new="$_repo:$_want"
  [ -n "$_digest" ] && _new="$_repo:$_want@$_digest"
  _installed=$(manifest_image)
  if [ "$_new" != "$_installed" ]; then
    echo "==> $_current → $_want${_digest:+ (pinned by digest)}"
    # Pin the new reference into the manifest, then apply. sed edits the single
    # "image" line in place (atomic tmp + move).
    _tmp="$MANIFEST.new"
    sed "s#\"image\"[[:space:]]*:[[:space:]]*\"[^\"]*\"#\"image\": \"$(json_str "$_new")\"#" "$MANIFEST" > "$_tmp" \
      || die "could not rewrite the image in $MANIFEST"
    chmod 0600 "$_tmp"; mv "$_tmp" "$MANIFEST"
  else
    echo "==> already on $_current — re-pulling and recreating"
  fi
  "$BIN_DEST" apply
}

cmd_uninstall() {
  need_root
  systemctl disable --now "$SERVICE_NAME" 2>/dev/null || true
  rm -f "$UNIT_DEST"; systemctl daemon-reload 2>/dev/null || true
  echo "Removed the systemd service. Left in place: $BIN_DEST, $ETC_DIR (incl. identity),"
  echo "the data root and any app container. Delete them manually for a full cleanup."
}

usage() {
  cat <<EOF
gpuk — GPU Kitchen host daemon (gpu-kitchen-worker)

  gpuk install --mode controller --image <ref> [--profile homelab|studio|enterprise|public]
               [--cluster N] [--cache-dir P]
               [--data-root P] [--http-port P] [--network host|bridge|<net>] [--binary <path>]
  gpuk install --mode worker --controller wss://<host>:<port> --enroll-token gk_enroll_...
               [--cluster N] [--cache-dir P] [--binary <path>]
  gpuk install ... --dry-run     Validate inputs and print the manifest without changing the host
  gpuk status            Service state, enrollment, /health, app container status
  gpuk enroll --controller wss://<host>:<port> --token gk_enroll_...
  gpuk apply             (controller) Reconcile the app container from the manifest
  gpuk update            (controller) Move to the current release (pull + recreate, rollback)
  gpuk update --check    (controller) Compare the installed version with the release
  gpuk update --version <tag>          (controller) Move to a specific release
  gpuk channel           Fetch + VERIFY the release channel and print what it offers
  gpuk backup [DIR]      (controller) Cold snapshot of the embedded pgdata (stop → tar → restart)
  gpuk manifest          Print the current manifest
  gpuk logs              Follow the app container logs (controller) or the daemon journal
  gpuk uninstall         Remove the systemd service

Most people never run this directly: the channel's install.sh installs it
(https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh).
EOF
}

# ── dispatch ────────────────────────────────────────────────────────────────────
cmd="${1:-help}"; if [ $# -gt 0 ]; then shift; fi
case "$cmd" in
  install)      cmd_install "$@" ;;
  status)       cmd_status ;;
  enroll)       cmd_enroll "$@" ;;
  apply)        cmd_apply ;;
  update)       cmd_update "$@" ;;
  channel)      cmd_channel ;;
  backup)       cmd_backup "${1:-}" ;;
  manifest)     cat "$MANIFEST" ;;
  logs)
    _img=$(manifest_image)
    if [ -n "$_img" ]; then exec docker logs -f "$(container_name)"; else exec journalctl -u "$SERVICE_NAME" -f; fi
    ;;
  uninstall)    cmd_uninstall ;;
  help|-h|--help) usage ;;
  *) usage; exit 1 ;;
esac
