Files
channel/gpuk
T
2026-08-03 09:02:45 +00:00

574 lines
24 KiB
Bash
Executable File

#!/bin/sh
# gpuk — installer + control CLI for the GPU Kitchen host daemon (gpu-kitchen-worker).
#
# Since C1 a compute node runs NO backend. The single privileged host component is
# the Rust `gpu-kitchen-worker` daemon (workerd), running as root DIRECTLY on the
# host (not in a container). It owns NVML clock/power locks, whitelisted host
# browsing, the model-cache staging pipeline, the signed `gpu-kitchen-bench`
# runner, and — on a controller node — the app container's lifecycle (create,
# health-gate, roll back, pull+recreate to update) via its manifest. There is no
# separate host daemon and no unix control socket: workerd is driven over the
# wss+mTLS channel by the controller, and locally by this CLI.
#
# Two roles:
# worker a headless compute node. Installs the binary + systemd unit + a
# manifest (cacheDisks, browseRoots) with NO app container, NO
# Postgres, NO docker app image. It ENROLLS over mTLS with a
# enrollment token, or waits for LAN discovery admission, then runs.
# controller the full app + UI. Installs the binary + systemd unit + an
# app-container manifest (image, ports, env, secretsRef) that workerd
# applies (`gpu-kitchen-worker apply`, then cmd:host_apply in steady
# state).
#
# Install (root):
# # controller:
# curl -fsSL https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/gpuk \
# | sudo sh -s -- install --mode controller \
# --image repo.byterain.io/gpukitchen-public/gpukitchen-controller:vX.Y.Z
# # worker (enroll against a controller with a single-use token from its UI):
# sudo ./deployments/install/gpuk install --mode worker \
# --controller wss://controller.lan:8443 --enroll-token gk_enroll_... \
# --cache-dir /mnt/models --binary apps/worker/target/release/gpu-kitchen-worker
#
# Control:
# gpuk status | apply | update | enroll | logs | manifest | uninstall
#
set -eu
# ── Paths ────────────────────────────────────────────────────────────────────
# Overridable, so the daemon can be driven against a prefix a normal user owns —
# the only way any of this is testable without handing a test suite root on the
# host. Unset (the real install) they are exactly the systemd defaults workerd uses
# (apps/worker/src/module.rs, apps/worker/src/identity.rs).
BIN_DEST="${GPUK_BIN_DEST:-/usr/local/bin/gpu-kitchen-worker}"
ETC_DIR="${GPUK_ETC_DIR:-/etc/gpu-kitchen}"
MANIFEST="$ETC_DIR/manifest.json"
IDENTITY_DIR="${GPUK_IDENTITY_DIR:-$ETC_DIR/identity}"
WORKER_ENV="$ETC_DIR/worker.env"
SERVICE_NAME="gpu-kitchen-worker"
UNIT_DEST="/etc/systemd/system/$SERVICE_NAME.service"
# Where to download the binary from when no --binary is given.
GPUK_RELEASE_BASE="${GPUK_RELEASE_BASE:-}"
die() { echo "gpuk: $*" >&2; exit 1; }
# Root, or able to do the job anyway. Fail with a clear message BEFORE touching
# /etc, /usr/local/bin or systemd. Against a user-owned prefix it simply works.
need_root() {
[ "$(id -u)" -eq 0 ] && return 0
[ -w "$ETC_DIR" ] && return 0
die "this command must run as root (use sudo)"
}
# ── JSON helpers (controlled inputs: paths + identifiers) ──────────────────────
json_str() { printf '%s' "$1" | sed 's/\\/\\\\/g; s/"/\\"/g'; }
json_array() { # args → ["a","b",...]
_out=""
for _p in "$@"; do
_e=$(json_str "$_p")
if [ -z "$_out" ]; then _out="\"$_e\""; else _out="$_out,\"$_e\""; fi
done
printf '[%s]' "$_out"
}
# ── install ────────────────────────────────────────────────────────────────────
arch_asset() {
case "$(uname -m)" in
x86_64|amd64) echo "gpu-kitchen-worker-x86_64" ;;
aarch64|arm64) echo "gpu-kitchen-worker-aarch64" ;;
*) die "unsupported architecture: $(uname -m)" ;;
esac
}
install_binary() { # [local-path]
if [ -n "${1:-}" ]; then
[ -f "$1" ] || die "binary not found: $1"
install -m 0755 "$1" "$BIN_DEST"
echo "==> installed $BIN_DEST from $1"
elif [ -n "$GPUK_RELEASE_BASE" ]; then
command -v curl >/dev/null || die "curl is required to download the binary"
_url="$GPUK_RELEASE_BASE/$(arch_asset)"
echo "==> downloading $_url"
curl -fsSL "$_url" -o "$BIN_DEST.new"
chmod 0755 "$BIN_DEST.new"
mv "$BIN_DEST.new" "$BIN_DEST"
elif [ -x "$BIN_DEST" ]; then
echo "==> reusing existing $BIN_DEST"
else
die "no binary: pass --binary <path> or set GPUK_RELEASE_BASE=<url base>"
fi
}
gen_secret() { head -c 32 /dev/urandom | base64 | tr '+/' '-_' | tr -d '='; }
# A password a human retypes once, from a terminal. Ambiguous glyphs removed.
gen_password() { head -c 24 /dev/urandom | base64 | tr -d '=+/OIl01' | cut -c1-16; }
seed_secrets() { # DATA_ROOT MODE
_sd="$1/secrets"
mkdir -p "$_sd"; chmod 0700 "$_sd"
[ -f "$_sd/encryption_key" ] || { umask 077; gen_secret > "$_sd/encryption_key"; }
[ -f "$_sd/node_id" ] || { umask 077; (cat /proc/sys/kernel/random/uuid 2>/dev/null || uuidgen) > "$_sd/node_id"; }
# The controller's first-run password. Generated here — NOT left to the backend
# to print into a log nobody watches when the install is one piped command. The
# installer prints it once, and the first-run wizard makes the operator replace
# it (GPUK_BOOTSTRAP_MUST_CHANGE).
if [ "$2" = "controller" ] && [ ! -f "$_sd/bootstrap_admin_password" ]; then
umask 077; gen_password > "$_sd/bootstrap_admin_password"
fi
}
# Write the systemd unit: prefer a sibling file, else embed. The daemon runs the
# binary with NO arguments (steady-state); role/controller URL/CA come from the
# enrolled identity (worker) and the optional EnvironmentFile.
write_unit() {
_src_unit="$(dirname "$0")/../../apps/worker/install/$SERVICE_NAME.service"
[ -f "$_src_unit" ] || _src_unit="$(dirname "$0")/$SERVICE_NAME.service"
if [ -f "$_src_unit" ]; then
install -m 0644 "$_src_unit" "$UNIT_DEST"
else
cat > "$UNIT_DEST" <<UNIT
[Unit]
Description=GPU Kitchen worker daemon (gpu-kitchen-worker)
After=network-online.target docker.service
Wants=network-online.target docker.service
[Service]
Type=simple
# Runs as root on the host (NVML, docker.sock, mounts) — NOT in a container.
# Identity (keypair/cert/CA) lives under $IDENTITY_DIR; a worker enrolls once
# before this unit starts. Optional overrides in $WORKER_ENV.
EnvironmentFile=-$WORKER_ENV
ExecStart=$BIN_DEST
RuntimeDirectory=gpu-kitchen
RuntimeDirectoryMode=0750
Restart=always
RestartSec=2
User=root
KillSignal=SIGTERM
TimeoutStopSec=15
[Install]
WantedBy=multi-user.target
UNIT
fi
}
# Pre-existing model caches (B81): a server that installs GPU Kitchen usually already
# holds tens or hundreds of GB of weights. We print a hint and nothing more — the
# install stays NON-INTERACTIVE, no config of any other program is read or touched,
# and referencing a cache is an explicit choice made later in the UI.
hint_existing_caches() {
hec_chosen=$(readlink -f "$1" 2>/dev/null || echo "$1")
hec_found=""
for hec_dir in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do
[ -n "$hec_dir" ] || continue
[ -d "$hec_dir/hub" ] || continue
hec_real=$(readlink -f "$hec_dir" 2>/dev/null || echo "$hec_dir")
[ "$hec_real" != "$hec_chosen" ] || continue
# Hub layout only (models--*) — matches what the scan can actually reference.
ls -d "$hec_dir"/hub/models--* >/dev/null 2>&1 || continue
hec_found="$hec_found $hec_real"
done
[ -n "$hec_found" ] || return 0
echo
for hec_dir in $hec_found; do
echo "==> existing model cache found at $hec_dir"
done
echo " Reference it from Settings -> Cache folders to reuse those models."
echo " GPU Kitchen only READS a referenced cache: nothing is moved or deleted."
}
cmd_install() {
need_root
IMAGE=""; MODE="worker"; CLUSTER="default"; DATA_ROOT="/var/lib/gpu-kitchen"
CACHE_DIR=""; NETWORK="host"; CONTROLLER_URL=""; BIN_SRC=""; HEALTH_PORT="8001"
BROWSE_ROOTS=""; ENROLL_TOKEN=""; HTTP_PORT="8080"
while [ $# -gt 0 ]; do
case "$1" in
--image) IMAGE="$2"; shift 2 ;;
--mode) MODE="$2"; shift 2 ;;
--cluster) CLUSTER="$2"; shift 2 ;;
--data-root) DATA_ROOT="$2"; shift 2 ;;
--cache-dir) CACHE_DIR="$2"; shift 2 ;;
--network) NETWORK="$2"; shift 2 ;;
--controller) CONTROLLER_URL="$2"; shift 2 ;;
# `--token` is an ENROLLMENT token (single-use or shared), NOT a bearer:
# the worker↔controller channel is cert-only since D11. `--token` is kept as
# a spelling of `--enroll-token`.
--enroll-token|--token) ENROLL_TOKEN="$2"; shift 2 ;;
--health-port) HEALTH_PORT="$2"; shift 2 ;;
--http-port) HTTP_PORT="$2"; shift 2 ;;
--binary) BIN_SRC="$2"; shift 2 ;;
--browse-root) BROWSE_ROOTS="$BROWSE_ROOTS $2"; shift 2 ;;
*) die "unknown install option: $1" ;;
esac
done
# Roles, as the backend names them (multi-server/config.ts): "controller" (full
# app + UI, accepts workers) and "worker" (headless compute node). Historical
# spellings still work.
case "$MODE" in
controller|server|standalone|manager) MODE="controller" ;;
worker|agent) MODE="worker" ;;
*) die "--mode must be controller or worker (got '$MODE')" ;;
esac
[ "$MODE" != "controller" ] || [ -n "$IMAGE" ] \
|| die "--image registry/gpukitchen-controller:<tag> is required for a controller"
[ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf"
install_binary "$BIN_SRC"
mkdir -p "$ETC_DIR" "$DATA_ROOT" "$CACHE_DIR"
seed_secrets "$DATA_ROOT" "$MODE"
# Default browse roots: common mount points + the dirs we already use.
[ -n "$BROWSE_ROOTS" ] || BROWSE_ROOTS="/mnt /data /srv $DATA_ROOT $(dirname "$CACHE_DIR")"
# shellcheck disable=SC2086
BROWSE_JSON=$(json_array $BROWSE_ROOTS)
CACHE_JSON="[{\"hostPath\":\"$(json_str "$CACHE_DIR")\",\"shared\":false}]"
umask 077
if [ "$MODE" = "worker" ]; then
write_worker_manifest
# Optional env overrides for the daemon (cluster grouping, display name, an
# explicit controller URL). The enrolled identity carries the controller URL
# + CA too — this is belt-and-braces / pre-enroll discovery grouping.
{
echo "GPUK_CLUSTER=$CLUSTER"
echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT"
echo "NODE_DISPLAY_NAME=$(hostname)"
[ -n "$CONTROLLER_URL" ] && echo "GPUK_CONTROLLER_URLS=$CONTROLLER_URL"
} > "$WORKER_ENV"
chmod 0600 "$WORKER_ENV"
echo "==> wrote $MANIFEST (worker: no app container)"
else
write_controller_manifest
echo "==> wrote $MANIFEST (controller: app container $IMAGE)"
fi
chmod 0600 "$MANIFEST"
write_unit
systemctl daemon-reload
echo "==> wrote $UNIT_DEST"
# ── Worker: token enrollment before start, or unattended LAN discovery ──
if [ "$MODE" = "worker" ]; then
if [ -n "$ENROLL_TOKEN" ]; then
[ -n "$CONTROLLER_URL" ] || die "--controller wss://<controller>:<port> is required to enroll"
echo "==> enrolling against $CONTROLLER_URL ..."
GPUK_IDENTITY_DIR="$IDENTITY_DIR" "$BIN_DEST" enroll \
--controller "$CONTROLLER_URL" --token "$ENROLL_TOKEN" \
|| die "enrollment failed (bad/expired token, or controller unreachable)"
else
echo "==> no token given: the daemon will discover its cluster on the LAN and wait"
echo " for automatic admission or administrator approval."
fi
systemctl enable --now "$SERVICE_NAME"
echo "==> $SERVICE_NAME enabled and started"
echo
echo "Done. The worker daemon is running${ENROLL_TOKEN:+ and enrolled}."
echo " Status : gpuk status Logs: gpuk logs"
hint_existing_caches "$CACHE_DIR"
return 0
fi
# ── Controller: start the daemon, then bring up the app container ──
systemctl enable --now "$SERVICE_NAME"
echo "==> $SERVICE_NAME enabled and started"
echo "==> applying manifest (first app-container start) ..."
# No unix socket any more: workerd reconciles the app container in-process from
# the on-disk manifest (there is no backend to relay through on the very first
# boot). Steady-state updates go through the daemon over WS.
"$BIN_DEST" apply || die "apply failed — check: gpuk logs"
echo
echo "Done. The worker daemon is running and the app container is up."
echo " Status : gpuk status App logs: gpuk logs Daemon: journalctl -u $SERVICE_NAME"
hint_existing_caches "$CACHE_DIR"
}
# Worker manifest: cacheDisks + browseRoots, and an EMPTY image so
# has_app_container() is false (apps/worker/src/modules/manifest.rs). No ports, no env,
# no secretsRef — a worker runs no app container.
write_worker_manifest() {
cat > "$MANIFEST" <<JSON
{
"schemaVersion": 1,
"image": "",
"mode": "worker",
"cluster": "$(json_str "$CLUSTER")",
"gpus": "all",
"dataRoot": "$(json_str "$DATA_ROOT")",
"cacheDisks": $CACHE_JSON,
"browseRoots": $BROWSE_JSON
}
JSON
}
# Controller manifest: the declarative app-container description workerd applies.
write_controller_manifest() {
# NOTE there is deliberately no PORT here: PORT is the backend's own port, a
# loopback-only 8000 behind nginx in the all-in-one image. What the outside
# world dials is GPUK_PORT.
EXTRA_ENV=""
EXTRA_ENV="$EXTRA_ENV,\"GPUK_PORT\":\"$(json_str "$HTTP_PORT")\""
EXTRA_ENV="$EXTRA_ENV,\"GPUK_DATA_ROOT\":\"$(json_str "$DATA_ROOT")\""
EXTRA_ENV="$EXTRA_ENV,\"GPUK_BOOTSTRAP_MUST_CHANGE\":\"1\""
[ "$NETWORK" = "host" ] || EXTRA_ENV="$EXTRA_ENV,\"BACKEND_DOCKER_NETWORK\":\"$(json_str "$NETWORK")\""
PORTS_JSON="{}"
if [ "$NETWORK" != "host" ]; then
PORTS_JSON="{\"$HTTP_PORT\":$HTTP_PORT,\"8200\":8200}"
fi
cat > "$MANIFEST" <<JSON
{
"schemaVersion": 1,
"image": "$(json_str "$IMAGE")",
"containerName": "gpu-kitchen",
"mode": "controller",
"cluster": "$(json_str "$CLUSTER")",
"networkMode": "$(json_str "$NETWORK")",
"gpus": "all",
"restartPolicy": "unless-stopped",
"dataRoot": "$(json_str "$DATA_ROOT")",
"cacheDisks": $CACHE_JSON,
"browseRoots": $BROWSE_JSON,
"ports": $PORTS_JSON,
"env": {
"NODE_ENV": "production",
"GPUK_MODE": "controller",
"GPUK_CLUSTER": "$(json_str "$CLUSTER")",
"NODE_DISPLAY_NAME": "$(json_str "$(hostname)")",
"HF_HOME": "$(json_str "$CACHE_DIR")"$EXTRA_ENV
},
"secretsRef": {
"ENCRYPTION_KEY": "$(json_str "$DATA_ROOT/secrets/encryption_key")",
"NODE_ID": "$(json_str "$DATA_ROOT/secrets/node_id")",
"GPUK_BOOTSTRAP_ADMIN_PASSWORD": "$(json_str "$DATA_ROOT/secrets/bootstrap_admin_password")"
}
}
JSON
}
# ── control subcommands ────────────────────────────────────────────────────────
container_name() {
sed -n 's/.*"containerName"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1
}
manifest_image() {
sed -n 's/.*"image"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1
}
health_port() {
sed -n 's/.*GPUK_WORKER_HEALTH_PORT=\([0-9]*\).*/\1/p' "$WORKER_ENV" 2>/dev/null | head -1
}
# Split an image reference into repository and tag.
#
# `${img%%:*}` cuts at the FIRST colon and is WRONG:
# `registry.internal:5000/gpuk/controller:v1.2.3` would yield repo
# `registry.internal`. The colon in a registry's host:port is not a tag separator.
# Rule: it is a tag only if the last colon comes after the last slash.
# (Same logic as apps/controller/api/src/core/release-channel.ts — see its unit tests.)
image_repo() {
case "$1" in
*@*) printf '%s' "${1%@*}"; return ;; # digest pin → repo@sha256:…
esac
_t="${1##*:}"
case "$_t" in
"$1") printf '%s' "$1" ;; # no colon at all → untagged
*/*) printf '%s' "$1" ;; # the last colon is inside a path → host:port, untagged
*) printf '%s' "${1%:*}" ;;
esac
}
image_tag() {
case "$1" in
*@*) return ;; # digest pin: no version to speak of
esac
_t="${1##*:}"
case "$_t" in
"$1") return ;;
*/*) return ;;
*) printf '%s' "$_t" ;;
esac
}
# The release channel: one flat JSON document served next to the installer. The
# UI's update check reads the same one (apps/controller/api/src/core/release-channel.ts) — one
# source of truth. Interim default: the public Gitea channel repo — flips to
# https://gpu.kitchen/latest.json once the hub exists (specs/developpement/ci-cd.md).
CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/latest.json}"
channel_version() {
command -v curl >/dev/null 2>&1 || return 1
curl -fsSL --max-time 20 "$CHANNEL_URL" 2>/dev/null \
| sed -n 's/.*"version"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' | head -1
}
# ── status ─────────────────────────────────────────────────────────────────────
# No control socket any more. Status = the systemd unit state + the daemon's own
# /health endpoint + whether a worker has enrolled (identity present).
cmd_status() {
_active=$(systemctl is-active "$SERVICE_NAME" 2>/dev/null || true)
echo "service : $_active"
if [ -f "$IDENTITY_DIR/identity.json" ]; then
echo "enrolled : yes ($IDENTITY_DIR)"
else
echo "enrolled : no (daemon waits for LAN admission; token enrollment is also available)"
fi
_hp=$(health_port); [ -n "$_hp" ] || _hp=8001
if command -v curl >/dev/null 2>&1; then
_h=$(curl -fsS --max-time 3 "http://127.0.0.1:$_hp/health" 2>/dev/null || true)
[ -n "$_h" ] && echo "health : $_h" || echo "health : (no answer on :$_hp)"
fi
_img=$(manifest_image)
if [ -n "$_img" ]; then
echo "app image : $_img"
echo "app cont. : $(docker inspect -f '{{.State.Status}}' "$(container_name)" 2>/dev/null || echo 'not running')"
else
echo "role : worker (no app container)"
fi
}
# ── enroll ─────────────────────────────────────────────────────────────────────
cmd_enroll() {
need_root
_url=""; _tok=""
while [ $# -gt 0 ]; do
case "$1" in
--controller) _url="$2"; shift 2 ;;
--enroll-token|--token) _tok="$2"; shift 2 ;;
*) die "unknown enroll option: $1" ;;
esac
done
[ -n "$_url" ] || die "enroll needs --controller wss://<controller>:<port>"
[ -n "$_tok" ] || die "enroll needs --token gk_enroll_..."
GPUK_IDENTITY_DIR="$IDENTITY_DIR" "$BIN_DEST" enroll --controller "$_url" --token "$_tok"
systemctl restart "$SERVICE_NAME" 2>/dev/null || true
}
# ── apply ──────────────────────────────────────────────────────────────────────
# Controller: reconcile the app container from the manifest, in-process. A worker
# has no app container — `apply` there is a no-op with a clear message.
cmd_apply() {
need_root
"$BIN_DEST" apply
}
# ── update ─────────────────────────────────────────────────────────────────────
# Controller: decide WHICH image TAG to pin, write it into the manifest, then let
# workerd pull + recreate (health-gate + rollback are the daemon's — apply()).
# Worker: the signed-binary self-update is DRIVEN FROM THE CONTROLLER (its Update
# button → POST /api/nodes/:id/host/update → cmd:host_update → verified swap).
# There is no local unverified swap path.
cmd_update() {
need_root
_want=""; _check=0
while [ $# -gt 0 ]; do
case "$1" in
--version) _want="$2"; shift 2 ;;
--check) _check=1; shift ;;
*) die "unknown update option: $1" ;;
esac
done
_image=$(manifest_image)
if [ -z "$_image" ]; then
echo "This is a worker node. Worker self-update is driven from the controller UI"
echo "(the node's Update button), which pushes a minisign-verified binary swap."
return 0
fi
_repo=$(image_repo "$_image")
_current=$(image_tag "$_image")
[ -n "$_current" ] || _current="(untagged)"
if [ "$_check" -eq 1 ]; then
_latest=$(channel_version) || true
echo "installed : $_current"
if [ -z "$_latest" ]; then
echo "available : unknown (cannot reach $CHANNEL_URL)"
exit 1
fi
echo "available : $_latest"
if [ "$_latest" = "$_current" ]; then echo "up to date."; else echo "run 'gpuk update' to move to $_latest"; fi
return 0
fi
if [ -z "$_want" ]; then
_want=$(channel_version) || true
if [ -z "$_want" ]; then
echo "==> cannot reach $CHANNEL_URL — re-applying the pinned $_current"
"$BIN_DEST" apply
return 0
fi
fi
if [ "$_want" != "$_current" ]; then
echo "==> $_current → $_want"
# Pin the new tag into the manifest, then apply. sed edits the single "image"
# line in place (atomic tmp + move).
_new="$_repo:$_want"
_tmp="$MANIFEST.new"
sed "s#\"image\"[[:space:]]*:[[:space:]]*\"[^\"]*\"#\"image\": \"$(json_str "$_new")\"#" "$MANIFEST" > "$_tmp" \
|| die "could not rewrite the image in $MANIFEST"
chmod 0600 "$_tmp"; mv "$_tmp" "$MANIFEST"
else
echo "==> already on $_current — re-pulling and recreating"
fi
"$BIN_DEST" apply
}
cmd_uninstall() {
need_root
systemctl disable --now "$SERVICE_NAME" 2>/dev/null || true
rm -f "$UNIT_DEST"; systemctl daemon-reload 2>/dev/null || true
echo "Removed the systemd service. Left in place: $BIN_DEST, $ETC_DIR (incl. identity),"
echo "the data root and any app container. Delete them manually for a full cleanup."
}
usage() {
cat <<EOF
gpuk — GPU Kitchen host daemon (gpu-kitchen-worker)
gpuk install --mode controller --image <ref> [--cluster N] [--cache-dir P]
[--data-root P] [--http-port P] [--network host|bridge|<net>] [--binary <path>]
gpuk install --mode worker --controller wss://<host>:<port> --enroll-token gk_enroll_...
[--cluster N] [--cache-dir P] [--browse-root P]... [--binary <path>]
gpuk status Service state, enrollment, /health, app container status
gpuk enroll --controller wss://<host>:<port> --token gk_enroll_...
gpuk apply (controller) Reconcile the app container from the manifest
gpuk update (controller) Move to the current release (pull + recreate, rollback)
gpuk update --check (controller) Compare the installed version with the release
gpuk update --version <tag> (controller) Move to a specific release
gpuk manifest Print the current manifest
gpuk logs Follow the app container logs (controller) or the daemon journal
gpuk uninstall Remove the systemd service
Most people never run this directly: the channel's install.sh installs it
(https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/install.sh).
EOF
}
# ── dispatch ────────────────────────────────────────────────────────────────────
cmd="${1:-help}"; if [ $# -gt 0 ]; then shift; fi
case "$cmd" in
install) cmd_install "$@" ;;
status) cmd_status ;;
enroll) cmd_enroll "$@" ;;
apply) cmd_apply ;;
update) cmd_update "$@" ;;
manifest) cat "$MANIFEST" ;;
logs)
_img=$(manifest_image)
if [ -n "$_img" ]; then exec docker logs -f "$(container_name)"; else exec journalctl -u "$SERVICE_NAME" -f; fi
;;
uninstall) cmd_uninstall ;;
help|-h|--help) usage ;;
*) usage; exit 1 ;;
esac