#!/bin/sh
# gpuk — installer + control CLI for the GPU Kitchen host daemon (gpu-kitchen-worker).
#
# Since C1 a compute node runs NO backend. The single privileged host component is
# the Rust `gpu-kitchen-worker` daemon (workerd), running as root DIRECTLY on the
# host (not in a container). It owns NVML clock/power locks, whitelisted host
# browsing, the model-cache staging pipeline, the signed `gpu-kitchen-bench`
# runner, and — on a controller node — the app container's lifecycle (create,
# health-gate, roll back, pull+recreate to update) via its manifest. There is no
# separate host daemon and no unix control socket: workerd is driven over the
# wss+mTLS channel by the controller, and locally by this CLI.
#
# Two roles:
#   worker      a headless compute node. Installs the binary + systemd unit + a
#               manifest (cacheDisks, browseRoots) with NO app container, NO
#               Postgres, NO docker app image. It ENROLLS over mTLS with a
#               enrollment token, or waits for LAN discovery admission, then runs.
#   controller  the full app + UI. Installs the binary + systemd unit + an
#               app-container manifest (image, ports, env, secretsRef) that workerd
#               applies (`gpu-kitchen-worker apply`, then cmd:host_apply in steady
#               state).
#
# Install (root):
#   # controller:
#   curl -fsSL https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/gpuk \
#     | sudo sh -s -- install --mode controller \
#       --image repo.byterain.io/gpukitchen-public/gpukitchen-controller:vX.Y.Z
#   # worker (enroll against a controller with a single-use token from its UI):
#   sudo ./deployments/install/gpuk install --mode worker \
#       --controller wss://controller.lan:8443 --enroll-token gk_enroll_... \
#       --cache-dir /mnt/models --binary apps/worker/target/release/gpu-kitchen-worker
#
# Control:
#   gpuk status | apply | update | enroll | logs | manifest | uninstall
#
set -eu

# ── Paths ────────────────────────────────────────────────────────────────────
# Overridable, so the daemon can be driven against a prefix a normal user owns —
# the only way any of this is testable without handing a test suite root on the
# host. Unset (the real install) they are exactly the systemd defaults workerd uses
# (apps/worker/src/module.rs, apps/worker/src/identity.rs).
BIN_DEST="${GPUK_BIN_DEST:-/usr/local/bin/gpu-kitchen-worker}"
ETC_DIR="${GPUK_ETC_DIR:-/etc/gpu-kitchen}"
MANIFEST="$ETC_DIR/manifest.json"
IDENTITY_DIR="${GPUK_IDENTITY_DIR:-$ETC_DIR/identity}"
WORKER_ENV="$ETC_DIR/worker.env"
SERVICE_NAME="gpu-kitchen-worker"
UNIT_DEST="/etc/systemd/system/$SERVICE_NAME.service"

# Where to download the binary from when no --binary is given.
GPUK_RELEASE_BASE="${GPUK_RELEASE_BASE:-}"

die() { echo "gpuk: $*" >&2; exit 1; }

# Root, or able to do the job anyway. Fail with a clear message BEFORE touching
# /etc, /usr/local/bin or systemd. Against a user-owned prefix it simply works.
need_root() {
  [ "$(id -u)" -eq 0 ] && return 0
  [ -w "$ETC_DIR" ] && return 0
  die "this command must run as root (use sudo)"
}

# ── JSON helpers (controlled inputs: paths + identifiers) ──────────────────────
json_str() { printf '%s' "$1" | sed 's/\\/\\\\/g; s/"/\\"/g'; }

json_array() { # args → ["a","b",...]
  _out=""
  for _p in "$@"; do
    _e=$(json_str "$_p")
    if [ -z "$_out" ]; then _out="\"$_e\""; else _out="$_out,\"$_e\""; fi
  done
  printf '[%s]' "$_out"
}

# ── install ────────────────────────────────────────────────────────────────────
arch_asset() {
  case "$(uname -m)" in
    x86_64|amd64)  echo "gpu-kitchen-worker-x86_64" ;;
    aarch64|arm64) echo "gpu-kitchen-worker-aarch64" ;;
    *) die "unsupported architecture: $(uname -m)" ;;
  esac
}

install_binary() { # [local-path]
  if [ -n "${1:-}" ]; then
    [ -f "$1" ] || die "binary not found: $1"
    install -m 0755 "$1" "$BIN_DEST"
    echo "==> installed $BIN_DEST from $1"
  elif [ -n "$GPUK_RELEASE_BASE" ]; then
    command -v curl >/dev/null || die "curl is required to download the binary"
    _url="$GPUK_RELEASE_BASE/$(arch_asset)"
    echo "==> downloading $_url"
    curl -fsSL "$_url" -o "$BIN_DEST.new"
    chmod 0755 "$BIN_DEST.new"
    mv "$BIN_DEST.new" "$BIN_DEST"
  elif [ -x "$BIN_DEST" ]; then
    echo "==> reusing existing $BIN_DEST"
  else
    die "no binary: pass --binary <path> or set GPUK_RELEASE_BASE=<url base>"
  fi
}

gen_secret() { head -c 32 /dev/urandom | base64 | tr '+/' '-_' | tr -d '='; }
# A password a human retypes once, from a terminal. Ambiguous glyphs removed.
gen_password() { head -c 24 /dev/urandom | base64 | tr -d '=+/OIl01' | cut -c1-16; }

seed_secrets() { # DATA_ROOT MODE
  _sd="$1/secrets"
  mkdir -p "$_sd"; chmod 0700 "$_sd"
  [ -f "$_sd/encryption_key" ] || { umask 077; gen_secret > "$_sd/encryption_key"; }
  [ -f "$_sd/node_id" ] || { umask 077; (cat /proc/sys/kernel/random/uuid 2>/dev/null || uuidgen) > "$_sd/node_id"; }
  # The controller's first-run password. Generated here — NOT left to the backend
  # to print into a log nobody watches when the install is one piped command. The
  # installer prints it once, and the first-run wizard makes the operator replace
  # it (GPUK_BOOTSTRAP_MUST_CHANGE).
  if [ "$2" = "controller" ] && [ ! -f "$_sd/bootstrap_admin_password" ]; then
    umask 077; gen_password > "$_sd/bootstrap_admin_password"
  fi
}

# Write the systemd unit: prefer a sibling file, else embed. The daemon runs the
# binary with NO arguments (steady-state); role/controller URL/CA come from the
# enrolled identity (worker) and the optional EnvironmentFile.
write_unit() {
  _src_unit="$(dirname "$0")/../../apps/worker/install/$SERVICE_NAME.service"
  [ -f "$_src_unit" ] || _src_unit="$(dirname "$0")/$SERVICE_NAME.service"
  if [ -f "$_src_unit" ]; then
    install -m 0644 "$_src_unit" "$UNIT_DEST"
  else
    cat > "$UNIT_DEST" <<UNIT
[Unit]
Description=GPU Kitchen worker daemon (gpu-kitchen-worker)
After=network-online.target docker.service
Wants=network-online.target docker.service

[Service]
Type=simple
# Runs as root on the host (NVML, docker.sock, mounts) — NOT in a container.
# Identity (keypair/cert/CA) lives under $IDENTITY_DIR; a worker enrolls once
# before this unit starts. Optional overrides in $WORKER_ENV.
EnvironmentFile=-$WORKER_ENV
ExecStart=$BIN_DEST
RuntimeDirectory=gpu-kitchen
RuntimeDirectoryMode=0750
Restart=always
RestartSec=2
User=root
KillSignal=SIGTERM
TimeoutStopSec=15

[Install]
WantedBy=multi-user.target
UNIT
  fi
}

# Pre-existing model caches (B81): a server that installs GPU Kitchen usually already
# holds tens or hundreds of GB of weights. We print a hint and nothing more — the
# install stays NON-INTERACTIVE, no config of any other program is read or touched,
# and referencing a cache is an explicit choice made later in the UI.
hint_existing_caches() {
  hec_chosen=$(readlink -f "$1" 2>/dev/null || echo "$1")
  hec_found=""
  for hec_dir in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do
    [ -n "$hec_dir" ] || continue
    [ -d "$hec_dir/hub" ] || continue
    hec_real=$(readlink -f "$hec_dir" 2>/dev/null || echo "$hec_dir")
    [ "$hec_real" != "$hec_chosen" ] || continue
    # Hub layout only (models--*) — matches what the scan can actually reference.
    ls -d "$hec_dir"/hub/models--* >/dev/null 2>&1 || continue
    hec_found="$hec_found $hec_real"
  done
  [ -n "$hec_found" ] || return 0
  echo
  for hec_dir in $hec_found; do
    echo "==> existing model cache found at $hec_dir"
  done
  echo "    Reference it from Settings -> Cache folders to reuse those models."
  echo "    GPU Kitchen only READS a referenced cache: nothing is moved or deleted."
}

cmd_install() {
  need_root
  IMAGE=""; MODE="worker"; CLUSTER="default"; DATA_ROOT="/var/lib/gpu-kitchen"
  CACHE_DIR=""; NETWORK="host"; CONTROLLER_URL=""; BIN_SRC=""; HEALTH_PORT="8001"
  BROWSE_ROOTS=""; ENROLL_TOKEN=""; HTTP_PORT="8080"
  while [ $# -gt 0 ]; do
    case "$1" in
      --image)        IMAGE="$2"; shift 2 ;;
      --mode)         MODE="$2"; shift 2 ;;
      --cluster)      CLUSTER="$2"; shift 2 ;;
      --data-root)    DATA_ROOT="$2"; shift 2 ;;
      --cache-dir)    CACHE_DIR="$2"; shift 2 ;;
      --network)      NETWORK="$2"; shift 2 ;;
      --controller)   CONTROLLER_URL="$2"; shift 2 ;;
      # `--token` is an ENROLLMENT token (single-use or shared), NOT a bearer:
      # the worker↔controller channel is cert-only since D11. `--token` is kept as
      # a spelling of `--enroll-token`.
      --enroll-token|--token) ENROLL_TOKEN="$2"; shift 2 ;;
      --health-port)  HEALTH_PORT="$2"; shift 2 ;;
      --http-port)    HTTP_PORT="$2"; shift 2 ;;
      --binary)       BIN_SRC="$2"; shift 2 ;;
      --browse-root)  BROWSE_ROOTS="$BROWSE_ROOTS $2"; shift 2 ;;
      *) die "unknown install option: $1" ;;
    esac
  done

  # Roles, as the backend names them (multi-server/config.ts): "controller" (full
  # app + UI, accepts workers) and "worker" (headless compute node). Historical
  # spellings still work.
  case "$MODE" in
    controller|server|standalone|manager) MODE="controller" ;;
    worker|agent) MODE="worker" ;;
    *) die "--mode must be controller or worker (got '$MODE')" ;;
  esac

  [ "$MODE" != "controller" ] || [ -n "$IMAGE" ] \
    || die "--image registry/gpukitchen-controller:<tag> is required for a controller"
  [ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf"

  install_binary "$BIN_SRC"

  mkdir -p "$ETC_DIR" "$DATA_ROOT" "$CACHE_DIR"
  seed_secrets "$DATA_ROOT" "$MODE"

  # Default browse roots: common mount points + the dirs we already use.
  [ -n "$BROWSE_ROOTS" ] || BROWSE_ROOTS="/mnt /data /srv $DATA_ROOT $(dirname "$CACHE_DIR")"
  # shellcheck disable=SC2086
  BROWSE_JSON=$(json_array $BROWSE_ROOTS)
  CACHE_JSON="[{\"hostPath\":\"$(json_str "$CACHE_DIR")\",\"shared\":false}]"

  umask 077
  if [ "$MODE" = "worker" ]; then
    write_worker_manifest
    # Optional env overrides for the daemon (cluster grouping, display name, an
    # explicit controller URL). The enrolled identity carries the controller URL
    # + CA too — this is belt-and-braces / pre-enroll discovery grouping.
    {
      echo "GPUK_CLUSTER=$CLUSTER"
      echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT"
      echo "NODE_DISPLAY_NAME=$(hostname)"
      [ -n "$CONTROLLER_URL" ] && echo "GPUK_CONTROLLER_URLS=$CONTROLLER_URL"
    } > "$WORKER_ENV"
    chmod 0600 "$WORKER_ENV"
    echo "==> wrote $MANIFEST (worker: no app container)"
  else
    write_controller_manifest
    echo "==> wrote $MANIFEST (controller: app container $IMAGE)"
  fi
  chmod 0600 "$MANIFEST"

  write_unit
  systemctl daemon-reload
  echo "==> wrote $UNIT_DEST"

  # ── Worker: token enrollment before start, or unattended LAN discovery ──
  if [ "$MODE" = "worker" ]; then
    if [ -n "$ENROLL_TOKEN" ]; then
      [ -n "$CONTROLLER_URL" ] || die "--controller wss://<controller>:<port> is required to enroll"
      echo "==> enrolling against $CONTROLLER_URL ..."
      GPUK_IDENTITY_DIR="$IDENTITY_DIR" "$BIN_DEST" enroll \
        --controller "$CONTROLLER_URL" --token "$ENROLL_TOKEN" \
        || die "enrollment failed (bad/expired token, or controller unreachable)"
    else
      echo "==> no token given: the daemon will discover its cluster on the LAN and wait"
      echo "    for automatic admission or administrator approval."
    fi
    systemctl enable --now "$SERVICE_NAME"
    echo "==> $SERVICE_NAME enabled and started"
    echo
    echo "Done. The worker daemon is running${ENROLL_TOKEN:+ and enrolled}."
    echo "  Status : gpuk status        Logs: gpuk logs"
    hint_existing_caches "$CACHE_DIR"
    return 0
  fi

  # ── Controller: start the daemon, then bring up the app container ──
  systemctl enable --now "$SERVICE_NAME"
  echo "==> $SERVICE_NAME enabled and started"
  echo "==> applying manifest (first app-container start) ..."
  # No unix socket any more: workerd reconciles the app container in-process from
  # the on-disk manifest (there is no backend to relay through on the very first
  # boot). Steady-state updates go through the daemon over WS.
  "$BIN_DEST" apply || die "apply failed — check: gpuk logs"
  echo
  echo "Done. The worker daemon is running and the app container is up."
  echo "  Status : gpuk status        App logs: gpuk logs        Daemon: journalctl -u $SERVICE_NAME"
  hint_existing_caches "$CACHE_DIR"
}

# Worker manifest: cacheDisks + browseRoots, and an EMPTY image so
# has_app_container() is false (apps/worker/src/modules/manifest.rs). No ports, no env,
# no secretsRef — a worker runs no app container.
write_worker_manifest() {
  cat > "$MANIFEST" <<JSON
{
  "schemaVersion": 1,
  "image": "",
  "mode": "worker",
  "cluster": "$(json_str "$CLUSTER")",
  "gpus": "all",
  "dataRoot": "$(json_str "$DATA_ROOT")",
  "cacheDisks": $CACHE_JSON,
  "browseRoots": $BROWSE_JSON
}
JSON
}

# Controller manifest: the declarative app-container description workerd applies.
write_controller_manifest() {
  # NOTE there is deliberately no PORT here: PORT is the backend's own port, a
  # loopback-only 8000 behind nginx in the all-in-one image. What the outside
  # world dials is GPUK_PORT.
  EXTRA_ENV=""
  EXTRA_ENV="$EXTRA_ENV,\"GPUK_PORT\":\"$(json_str "$HTTP_PORT")\""
  EXTRA_ENV="$EXTRA_ENV,\"GPUK_DATA_ROOT\":\"$(json_str "$DATA_ROOT")\""
  EXTRA_ENV="$EXTRA_ENV,\"GPUK_BOOTSTRAP_MUST_CHANGE\":\"1\""
  [ "$NETWORK" = "host" ] || EXTRA_ENV="$EXTRA_ENV,\"BACKEND_DOCKER_NETWORK\":\"$(json_str "$NETWORK")\""

  PORTS_JSON="{}"
  if [ "$NETWORK" != "host" ]; then
    PORTS_JSON="{\"$HTTP_PORT\":$HTTP_PORT,\"8200\":8200}"
  fi

  cat > "$MANIFEST" <<JSON
{
  "schemaVersion": 1,
  "image": "$(json_str "$IMAGE")",
  "containerName": "gpu-kitchen",
  "mode": "controller",
  "cluster": "$(json_str "$CLUSTER")",
  "networkMode": "$(json_str "$NETWORK")",
  "gpus": "all",
  "restartPolicy": "unless-stopped",
  "dataRoot": "$(json_str "$DATA_ROOT")",
  "cacheDisks": $CACHE_JSON,
  "browseRoots": $BROWSE_JSON,
  "ports": $PORTS_JSON,
  "env": {
    "NODE_ENV": "production",
    "GPUK_MODE": "controller",
    "GPUK_CLUSTER": "$(json_str "$CLUSTER")",
    "NODE_DISPLAY_NAME": "$(json_str "$(hostname)")",
    "HF_HOME": "$(json_str "$CACHE_DIR")"$EXTRA_ENV
  },
  "secretsRef": {
    "ENCRYPTION_KEY": "$(json_str "$DATA_ROOT/secrets/encryption_key")",
    "NODE_ID": "$(json_str "$DATA_ROOT/secrets/node_id")",
    "GPUK_BOOTSTRAP_ADMIN_PASSWORD": "$(json_str "$DATA_ROOT/secrets/bootstrap_admin_password")"
  }
}
JSON
}

# ── control subcommands ────────────────────────────────────────────────────────
container_name() {
  sed -n 's/.*"containerName"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1
}

manifest_image() {
  sed -n 's/.*"image"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1
}

health_port() {
  sed -n 's/.*GPUK_WORKER_HEALTH_PORT=\([0-9]*\).*/\1/p' "$WORKER_ENV" 2>/dev/null | head -1
}

# Split an image reference into repository and tag.
#
# `${img%%:*}` cuts at the FIRST colon and is WRONG:
# `registry.internal:5000/gpuk/controller:v1.2.3` would yield repo
# `registry.internal`. The colon in a registry's host:port is not a tag separator.
# Rule: it is a tag only if the last colon comes after the last slash.
# (Same logic as apps/controller/api/src/core/release-channel.ts — see its unit tests.)
image_repo() {
  case "$1" in
    *@*) printf '%s' "${1%@*}"; return ;;   # digest pin → repo@sha256:…
  esac
  _t="${1##*:}"
  case "$_t" in
    "$1") printf '%s' "$1" ;;    # no colon at all → untagged
    */*)  printf '%s' "$1" ;;    # the last colon is inside a path → host:port, untagged
    *)    printf '%s' "${1%:*}" ;;
  esac
}

image_tag() {
  case "$1" in
    *@*) return ;;               # digest pin: no version to speak of
  esac
  _t="${1##*:}"
  case "$_t" in
    "$1") return ;;
    */*)  return ;;
    *)    printf '%s' "$_t" ;;
  esac
}

# The release channel: one flat JSON document served next to the installer. The
# UI's update check reads the same one (apps/controller/api/src/core/release-channel.ts) — one
# source of truth. Interim default: the public Gitea channel repo — flips to
# https://gpu.kitchen/latest.json once the hub exists (specs/developpement/ci-cd.md).
CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/latest.json}"

channel_version() {
  command -v curl >/dev/null 2>&1 || return 1
  curl -fsSL --max-time 20 "$CHANNEL_URL" 2>/dev/null \
    | sed -n 's/.*"version"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' | head -1
}

# ── status ─────────────────────────────────────────────────────────────────────
# No control socket any more. Status = the systemd unit state + the daemon's own
# /health endpoint + whether a worker has enrolled (identity present).
cmd_status() {
  _active=$(systemctl is-active "$SERVICE_NAME" 2>/dev/null || true)
  echo "service   : $_active"
  if [ -f "$IDENTITY_DIR/identity.json" ]; then
    echo "enrolled  : yes ($IDENTITY_DIR)"
  else
    echo "enrolled  : no (daemon waits for LAN admission; token enrollment is also available)"
  fi
  _hp=$(health_port); [ -n "$_hp" ] || _hp=8001
  if command -v curl >/dev/null 2>&1; then
    _h=$(curl -fsS --max-time 3 "http://127.0.0.1:$_hp/health" 2>/dev/null || true)
    [ -n "$_h" ] && echo "health    : $_h" || echo "health    : (no answer on :$_hp)"
  fi
  _img=$(manifest_image)
  if [ -n "$_img" ]; then
    echo "app image : $_img"
    echo "app cont. : $(docker inspect -f '{{.State.Status}}' "$(container_name)" 2>/dev/null || echo 'not running')"
  else
    echo "role      : worker (no app container)"
  fi
}

# ── enroll ─────────────────────────────────────────────────────────────────────
cmd_enroll() {
  need_root
  _url=""; _tok=""
  while [ $# -gt 0 ]; do
    case "$1" in
      --controller)               _url="$2"; shift 2 ;;
      --enroll-token|--token)     _tok="$2"; shift 2 ;;
      *) die "unknown enroll option: $1" ;;
    esac
  done
  [ -n "$_url" ] || die "enroll needs --controller wss://<controller>:<port>"
  [ -n "$_tok" ] || die "enroll needs --token gk_enroll_..."
  GPUK_IDENTITY_DIR="$IDENTITY_DIR" "$BIN_DEST" enroll --controller "$_url" --token "$_tok"
  systemctl restart "$SERVICE_NAME" 2>/dev/null || true
}

# ── apply ──────────────────────────────────────────────────────────────────────
# Controller: reconcile the app container from the manifest, in-process. A worker
# has no app container — `apply` there is a no-op with a clear message.
cmd_apply() {
  need_root
  "$BIN_DEST" apply
}

# ── update ─────────────────────────────────────────────────────────────────────
# Controller: decide WHICH image TAG to pin, write it into the manifest, then let
# workerd pull + recreate (health-gate + rollback are the daemon's — apply()).
# Worker: the signed-binary self-update is DRIVEN FROM THE CONTROLLER (its Update
# button → POST /api/nodes/:id/host/update → cmd:host_update → verified swap).
# There is no local unverified swap path.
cmd_update() {
  need_root
  _want=""; _check=0
  while [ $# -gt 0 ]; do
    case "$1" in
      --version) _want="$2"; shift 2 ;;
      --check)   _check=1; shift ;;
      *) die "unknown update option: $1" ;;
    esac
  done

  _image=$(manifest_image)
  if [ -z "$_image" ]; then
    echo "This is a worker node. Worker self-update is driven from the controller UI"
    echo "(the node's Update button), which pushes a minisign-verified binary swap."
    return 0
  fi

  _repo=$(image_repo "$_image")
  _current=$(image_tag "$_image")
  [ -n "$_current" ] || _current="(untagged)"

  if [ "$_check" -eq 1 ]; then
    _latest=$(channel_version) || true
    echo "installed : $_current"
    if [ -z "$_latest" ]; then
      echo "available : unknown (cannot reach $CHANNEL_URL)"
      exit 1
    fi
    echo "available : $_latest"
    if [ "$_latest" = "$_current" ]; then echo "up to date."; else echo "run 'gpuk update' to move to $_latest"; fi
    return 0
  fi

  if [ -z "$_want" ]; then
    _want=$(channel_version) || true
    if [ -z "$_want" ]; then
      echo "==> cannot reach $CHANNEL_URL — re-applying the pinned $_current"
      "$BIN_DEST" apply
      return 0
    fi
  fi

  if [ "$_want" != "$_current" ]; then
    echo "==> $_current → $_want"
    # Pin the new tag into the manifest, then apply. sed edits the single "image"
    # line in place (atomic tmp + move).
    _new="$_repo:$_want"
    _tmp="$MANIFEST.new"
    sed "s#\"image\"[[:space:]]*:[[:space:]]*\"[^\"]*\"#\"image\": \"$(json_str "$_new")\"#" "$MANIFEST" > "$_tmp" \
      || die "could not rewrite the image in $MANIFEST"
    chmod 0600 "$_tmp"; mv "$_tmp" "$MANIFEST"
  else
    echo "==> already on $_current — re-pulling and recreating"
  fi
  "$BIN_DEST" apply
}

cmd_uninstall() {
  need_root
  systemctl disable --now "$SERVICE_NAME" 2>/dev/null || true
  rm -f "$UNIT_DEST"; systemctl daemon-reload 2>/dev/null || true
  echo "Removed the systemd service. Left in place: $BIN_DEST, $ETC_DIR (incl. identity),"
  echo "the data root and any app container. Delete them manually for a full cleanup."
}

usage() {
  cat <<EOF
gpuk — GPU Kitchen host daemon (gpu-kitchen-worker)

  gpuk install --mode controller --image <ref> [--cluster N] [--cache-dir P]
               [--data-root P] [--http-port P] [--network host|bridge|<net>] [--binary <path>]
  gpuk install --mode worker --controller wss://<host>:<port> --enroll-token gk_enroll_...
               [--cluster N] [--cache-dir P] [--browse-root P]... [--binary <path>]
  gpuk status            Service state, enrollment, /health, app container status
  gpuk enroll --controller wss://<host>:<port> --token gk_enroll_...
  gpuk apply             (controller) Reconcile the app container from the manifest
  gpuk update            (controller) Move to the current release (pull + recreate, rollback)
  gpuk update --check    (controller) Compare the installed version with the release
  gpuk update --version <tag>          (controller) Move to a specific release
  gpuk manifest          Print the current manifest
  gpuk logs              Follow the app container logs (controller) or the daemon journal
  gpuk uninstall         Remove the systemd service

Most people never run this directly: the channel's install.sh installs it
(https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/install.sh).
EOF
}

# ── dispatch ────────────────────────────────────────────────────────────────────
cmd="${1:-help}"; if [ $# -gt 0 ]; then shift; fi
case "$cmd" in
  install)      cmd_install "$@" ;;
  status)       cmd_status ;;
  enroll)       cmd_enroll "$@" ;;
  apply)        cmd_apply ;;
  update)       cmd_update "$@" ;;
  manifest)     cat "$MANIFEST" ;;
  logs)
    _img=$(manifest_image)
    if [ -n "$_img" ]; then exec docker logs -f "$(container_name)"; else exec journalctl -u "$SERVICE_NAME" -f; fi
    ;;
  uninstall)    cmd_uninstall ;;
  help|-h|--help) usage ;;
  *) usage; exit 1 ;;
esac
