#!/bin/sh # gpuk — installer + control CLI for the GPU Kitchen host daemon (gpu-kitchen-worker). # # Since C1 a compute node runs NO backend. The single privileged host component is # the Rust `gpu-kitchen-worker` daemon (workerd), running as root DIRECTLY on the # host (not in a container). It owns NVML clock/power locks, whitelisted host # browsing, the model-cache staging pipeline, the signed `gpu-kitchen-bench` # runner, and — on a controller node — the app container's lifecycle (create, # health-gate, roll back, pull+recreate to update) via its manifest. There is no # separate host daemon and no unix control socket: workerd is driven over the # wss+mTLS channel by the controller, and locally by this CLI. # # Two roles: # worker a headless compute node. Installs the binary + systemd unit + a # manifest (cacheDisks, browseRoots) with NO app container, NO # Postgres, NO docker app image. It ENROLLS over mTLS with a # enrollment token, or waits for LAN discovery admission, then runs. # controller the full app + UI. Installs the binary + systemd unit + an # app-container manifest (image, ports, env, secretsRef) that workerd # applies (`gpu-kitchen-worker apply`, then cmd:host_apply in steady # state). # # Install (root): # # controller: # curl -fsSL https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/gpuk \ # | sudo sh -s -- install --mode controller \ # --image repo.byterain.io/gpukitchen-public/gpukitchen-controller:vX.Y.Z # # worker (enroll against a controller with a single-use token from its UI): # sudo ./deployments/install/gpuk install --mode worker \ # --controller wss://controller.lan:8443 --enroll-token gk_enroll_... \ # --cache-dir /mnt/models --binary apps/worker/target/release/gpu-kitchen-worker # # Control: # gpuk status | apply | update | enroll | logs | manifest | uninstall # set -eu # ── Paths ──────────────────────────────────────────────────────────────────── # Overridable, so the daemon can be driven against a prefix a normal user owns — # the only way any of this is testable without handing a test suite root on the # host. Unset (the real install) they are exactly the systemd defaults workerd uses # (apps/worker/src/module.rs, apps/worker/src/identity.rs). BIN_DEST="${GPUK_BIN_DEST:-/usr/local/bin/gpu-kitchen-worker}" ETC_DIR="${GPUK_ETC_DIR:-/etc/gpu-kitchen}" MANIFEST="$ETC_DIR/manifest.json" IDENTITY_DIR="${GPUK_IDENTITY_DIR:-$ETC_DIR/identity}" WORKER_ENV="$ETC_DIR/worker.env" SERVICE_NAME="gpu-kitchen-worker" UNIT_DEST="/etc/systemd/system/$SERVICE_NAME.service" # Where to download the binary from when no --binary is given. GPUK_RELEASE_BASE="${GPUK_RELEASE_BASE:-}" die() { echo "gpuk: $*" >&2; exit 1; } # Root, or able to do the job anyway. Fail with a clear message BEFORE touching # /etc, /usr/local/bin or systemd. Against a user-owned prefix it simply works. need_root() { [ "$(id -u)" -eq 0 ] && return 0 [ -w "$ETC_DIR" ] && return 0 die "this command must run as root (use sudo)" } # ── JSON helpers (controlled inputs: paths + identifiers) ────────────────────── json_str() { printf '%s' "$1" | sed 's/\\/\\\\/g; s/"/\\"/g'; } json_array() { # args → ["a","b",...] _out="" for _p in "$@"; do _e=$(json_str "$_p") if [ -z "$_out" ]; then _out="\"$_e\""; else _out="$_out,\"$_e\""; fi done printf '[%s]' "$_out" } # ── install ──────────────────────────────────────────────────────────────────── arch_asset() { case "$(uname -m)" in x86_64|amd64) echo "gpu-kitchen-worker-x86_64" ;; aarch64|arm64) echo "gpu-kitchen-worker-aarch64" ;; *) die "unsupported architecture: $(uname -m)" ;; esac } install_binary() { # [local-path] if [ -n "${1:-}" ]; then [ -f "$1" ] || die "binary not found: $1" install -m 0755 "$1" "$BIN_DEST" echo "==> installed $BIN_DEST from $1" elif [ -n "$GPUK_RELEASE_BASE" ]; then command -v curl >/dev/null || die "curl is required to download the binary" _url="$GPUK_RELEASE_BASE/$(arch_asset)" echo "==> downloading $_url" curl -fsSL "$_url" -o "$BIN_DEST.new" chmod 0755 "$BIN_DEST.new" mv "$BIN_DEST.new" "$BIN_DEST" elif [ -x "$BIN_DEST" ]; then echo "==> reusing existing $BIN_DEST" else die "no binary: pass --binary or set GPUK_RELEASE_BASE=" fi } gen_secret() { head -c 32 /dev/urandom | base64 | tr '+/' '-_' | tr -d '='; } # A password a human retypes once, from a terminal. Ambiguous glyphs removed. gen_password() { head -c 24 /dev/urandom | base64 | tr -d '=+/OIl01' | cut -c1-16; } seed_secrets() { # DATA_ROOT MODE _sd="$1/secrets" mkdir -p "$_sd"; chmod 0700 "$_sd" [ -f "$_sd/encryption_key" ] || { umask 077; gen_secret > "$_sd/encryption_key"; } [ -f "$_sd/node_id" ] || { umask 077; (cat /proc/sys/kernel/random/uuid 2>/dev/null || uuidgen) > "$_sd/node_id"; } # The controller's first-run password. Generated here — NOT left to the backend # to print into a log nobody watches when the install is one piped command. The # installer prints it once, and the first-run wizard makes the operator replace # it (GPUK_BOOTSTRAP_MUST_CHANGE). if [ "$2" = "controller" ] && [ ! -f "$_sd/bootstrap_admin_password" ]; then umask 077; gen_password > "$_sd/bootstrap_admin_password" fi } # Write the systemd unit: prefer a sibling file, else embed. The daemon runs the # binary with NO arguments (steady-state); role/controller URL/CA come from the # enrolled identity (worker) and the optional EnvironmentFile. write_unit() { _src_unit="$(dirname "$0")/../../apps/worker/install/$SERVICE_NAME.service" [ -f "$_src_unit" ] || _src_unit="$(dirname "$0")/$SERVICE_NAME.service" if [ -f "$_src_unit" ]; then install -m 0644 "$_src_unit" "$UNIT_DEST" else cat > "$UNIT_DEST" </dev/null || echo "$1") hec_found="" for hec_dir in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do [ -n "$hec_dir" ] || continue [ -d "$hec_dir/hub" ] || continue hec_real=$(readlink -f "$hec_dir" 2>/dev/null || echo "$hec_dir") [ "$hec_real" != "$hec_chosen" ] || continue # Hub layout only (models--*) — matches what the scan can actually reference. ls -d "$hec_dir"/hub/models--* >/dev/null 2>&1 || continue hec_found="$hec_found $hec_real" done [ -n "$hec_found" ] || return 0 echo for hec_dir in $hec_found; do echo "==> existing model cache found at $hec_dir" done echo " Reference it from Settings -> Cache folders to reuse those models." echo " GPU Kitchen only READS a referenced cache: nothing is moved or deleted." } cmd_install() { need_root IMAGE=""; MODE="worker"; CLUSTER="default"; DATA_ROOT="/var/lib/gpu-kitchen" CACHE_DIR=""; NETWORK="host"; CONTROLLER_URL=""; BIN_SRC=""; HEALTH_PORT="8001" BROWSE_ROOTS=""; ENROLL_TOKEN=""; HTTP_PORT="8080" while [ $# -gt 0 ]; do case "$1" in --image) IMAGE="$2"; shift 2 ;; --mode) MODE="$2"; shift 2 ;; --cluster) CLUSTER="$2"; shift 2 ;; --data-root) DATA_ROOT="$2"; shift 2 ;; --cache-dir) CACHE_DIR="$2"; shift 2 ;; --network) NETWORK="$2"; shift 2 ;; --controller) CONTROLLER_URL="$2"; shift 2 ;; # `--token` is an ENROLLMENT token (single-use or shared), NOT a bearer: # the worker↔controller channel is cert-only since D11. `--token` is kept as # a spelling of `--enroll-token`. --enroll-token|--token) ENROLL_TOKEN="$2"; shift 2 ;; --health-port) HEALTH_PORT="$2"; shift 2 ;; --http-port) HTTP_PORT="$2"; shift 2 ;; --binary) BIN_SRC="$2"; shift 2 ;; --browse-root) BROWSE_ROOTS="$BROWSE_ROOTS $2"; shift 2 ;; *) die "unknown install option: $1" ;; esac done # Roles, as the backend names them (multi-server/config.ts): "controller" (full # app + UI, accepts workers) and "worker" (headless compute node). Historical # spellings still work. case "$MODE" in controller|server|standalone|manager) MODE="controller" ;; worker|agent) MODE="worker" ;; *) die "--mode must be controller or worker (got '$MODE')" ;; esac [ "$MODE" != "controller" ] || [ -n "$IMAGE" ] \ || die "--image registry/gpukitchen-controller: is required for a controller" [ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf" install_binary "$BIN_SRC" mkdir -p "$ETC_DIR" "$DATA_ROOT" "$CACHE_DIR" seed_secrets "$DATA_ROOT" "$MODE" # Default browse roots: common mount points + the dirs we already use. [ -n "$BROWSE_ROOTS" ] || BROWSE_ROOTS="/mnt /data /srv $DATA_ROOT $(dirname "$CACHE_DIR")" # shellcheck disable=SC2086 BROWSE_JSON=$(json_array $BROWSE_ROOTS) CACHE_JSON="[{\"hostPath\":\"$(json_str "$CACHE_DIR")\",\"shared\":false}]" umask 077 if [ "$MODE" = "worker" ]; then write_worker_manifest # Optional env overrides for the daemon (cluster grouping, display name, an # explicit controller URL). The enrolled identity carries the controller URL # + CA too — this is belt-and-braces / pre-enroll discovery grouping. { echo "GPUK_CLUSTER=$CLUSTER" echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT" echo "NODE_DISPLAY_NAME=$(hostname)" [ -n "$CONTROLLER_URL" ] && echo "GPUK_CONTROLLER_URLS=$CONTROLLER_URL" } > "$WORKER_ENV" chmod 0600 "$WORKER_ENV" echo "==> wrote $MANIFEST (worker: no app container)" else write_controller_manifest echo "==> wrote $MANIFEST (controller: app container $IMAGE)" fi chmod 0600 "$MANIFEST" write_unit systemctl daemon-reload echo "==> wrote $UNIT_DEST" # ── Worker: token enrollment before start, or unattended LAN discovery ── if [ "$MODE" = "worker" ]; then if [ -n "$ENROLL_TOKEN" ]; then [ -n "$CONTROLLER_URL" ] || die "--controller wss://: is required to enroll" echo "==> enrolling against $CONTROLLER_URL ..." GPUK_IDENTITY_DIR="$IDENTITY_DIR" "$BIN_DEST" enroll \ --controller "$CONTROLLER_URL" --token "$ENROLL_TOKEN" \ || die "enrollment failed (bad/expired token, or controller unreachable)" else echo "==> no token given: the daemon will discover its cluster on the LAN and wait" echo " for automatic admission or administrator approval." fi systemctl enable --now "$SERVICE_NAME" echo "==> $SERVICE_NAME enabled and started" echo echo "Done. The worker daemon is running${ENROLL_TOKEN:+ and enrolled}." echo " Status : gpuk status Logs: gpuk logs" hint_existing_caches "$CACHE_DIR" return 0 fi # ── Controller: start the daemon, then bring up the app container ── systemctl enable --now "$SERVICE_NAME" echo "==> $SERVICE_NAME enabled and started" echo "==> applying manifest (first app-container start) ..." # No unix socket any more: workerd reconciles the app container in-process from # the on-disk manifest (there is no backend to relay through on the very first # boot). Steady-state updates go through the daemon over WS. "$BIN_DEST" apply || die "apply failed — check: gpuk logs" echo echo "Done. The worker daemon is running and the app container is up." echo " Status : gpuk status App logs: gpuk logs Daemon: journalctl -u $SERVICE_NAME" hint_existing_caches "$CACHE_DIR" } # Worker manifest: cacheDisks + browseRoots, and an EMPTY image so # has_app_container() is false (apps/worker/src/modules/manifest.rs). No ports, no env, # no secretsRef — a worker runs no app container. write_worker_manifest() { cat > "$MANIFEST" < "$MANIFEST" </dev/null | head -1 } manifest_image() { sed -n 's/.*"image"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1 } health_port() { sed -n 's/.*GPUK_WORKER_HEALTH_PORT=\([0-9]*\).*/\1/p' "$WORKER_ENV" 2>/dev/null | head -1 } # Split an image reference into repository and tag. # # `${img%%:*}` cuts at the FIRST colon and is WRONG: # `registry.internal:5000/gpuk/controller:v1.2.3` would yield repo # `registry.internal`. The colon in a registry's host:port is not a tag separator. # Rule: it is a tag only if the last colon comes after the last slash. # (Same logic as apps/controller/api/src/core/release-channel.ts — see its unit tests.) image_repo() { case "$1" in *@*) printf '%s' "${1%@*}"; return ;; # digest pin → repo@sha256:… esac _t="${1##*:}" case "$_t" in "$1") printf '%s' "$1" ;; # no colon at all → untagged */*) printf '%s' "$1" ;; # the last colon is inside a path → host:port, untagged *) printf '%s' "${1%:*}" ;; esac } image_tag() { case "$1" in *@*) return ;; # digest pin: no version to speak of esac _t="${1##*:}" case "$_t" in "$1") return ;; */*) return ;; *) printf '%s' "$_t" ;; esac } # The release channel: one flat JSON document served next to the installer. The # UI's update check reads the same one (apps/controller/api/src/core/release-channel.ts) — one # source of truth. Interim default: the public Gitea channel repo — flips to # https://gpu.kitchen/latest.json once the hub exists (specs/developpement/ci-cd.md). CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/latest.json}" channel_version() { command -v curl >/dev/null 2>&1 || return 1 curl -fsSL --max-time 20 "$CHANNEL_URL" 2>/dev/null \ | sed -n 's/.*"version"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' | head -1 } # ── status ───────────────────────────────────────────────────────────────────── # No control socket any more. Status = the systemd unit state + the daemon's own # /health endpoint + whether a worker has enrolled (identity present). cmd_status() { _active=$(systemctl is-active "$SERVICE_NAME" 2>/dev/null || true) echo "service : $_active" if [ -f "$IDENTITY_DIR/identity.json" ]; then echo "enrolled : yes ($IDENTITY_DIR)" else echo "enrolled : no (daemon waits for LAN admission; token enrollment is also available)" fi _hp=$(health_port); [ -n "$_hp" ] || _hp=8001 if command -v curl >/dev/null 2>&1; then _h=$(curl -fsS --max-time 3 "http://127.0.0.1:$_hp/health" 2>/dev/null || true) [ -n "$_h" ] && echo "health : $_h" || echo "health : (no answer on :$_hp)" fi _img=$(manifest_image) if [ -n "$_img" ]; then echo "app image : $_img" echo "app cont. : $(docker inspect -f '{{.State.Status}}' "$(container_name)" 2>/dev/null || echo 'not running')" else echo "role : worker (no app container)" fi } # ── enroll ───────────────────────────────────────────────────────────────────── cmd_enroll() { need_root _url=""; _tok="" while [ $# -gt 0 ]; do case "$1" in --controller) _url="$2"; shift 2 ;; --enroll-token|--token) _tok="$2"; shift 2 ;; *) die "unknown enroll option: $1" ;; esac done [ -n "$_url" ] || die "enroll needs --controller wss://:" [ -n "$_tok" ] || die "enroll needs --token gk_enroll_..." GPUK_IDENTITY_DIR="$IDENTITY_DIR" "$BIN_DEST" enroll --controller "$_url" --token "$_tok" systemctl restart "$SERVICE_NAME" 2>/dev/null || true } # ── apply ────────────────────────────────────────────────────────────────────── # Controller: reconcile the app container from the manifest, in-process. A worker # has no app container — `apply` there is a no-op with a clear message. cmd_apply() { need_root "$BIN_DEST" apply } # ── update ───────────────────────────────────────────────────────────────────── # Controller: decide WHICH image TAG to pin, write it into the manifest, then let # workerd pull + recreate (health-gate + rollback are the daemon's — apply()). # Worker: the signed-binary self-update is DRIVEN FROM THE CONTROLLER (its Update # button → POST /api/nodes/:id/host/update → cmd:host_update → verified swap). # There is no local unverified swap path. cmd_update() { need_root _want=""; _check=0 while [ $# -gt 0 ]; do case "$1" in --version) _want="$2"; shift 2 ;; --check) _check=1; shift ;; *) die "unknown update option: $1" ;; esac done _image=$(manifest_image) if [ -z "$_image" ]; then echo "This is a worker node. Worker self-update is driven from the controller UI" echo "(the node's Update button), which pushes a minisign-verified binary swap." return 0 fi _repo=$(image_repo "$_image") _current=$(image_tag "$_image") [ -n "$_current" ] || _current="(untagged)" if [ "$_check" -eq 1 ]; then _latest=$(channel_version) || true echo "installed : $_current" if [ -z "$_latest" ]; then echo "available : unknown (cannot reach $CHANNEL_URL)" exit 1 fi echo "available : $_latest" if [ "$_latest" = "$_current" ]; then echo "up to date."; else echo "run 'gpuk update' to move to $_latest"; fi return 0 fi if [ -z "$_want" ]; then _want=$(channel_version) || true if [ -z "$_want" ]; then echo "==> cannot reach $CHANNEL_URL — re-applying the pinned $_current" "$BIN_DEST" apply return 0 fi fi if [ "$_want" != "$_current" ]; then echo "==> $_current → $_want" # Pin the new tag into the manifest, then apply. sed edits the single "image" # line in place (atomic tmp + move). _new="$_repo:$_want" _tmp="$MANIFEST.new" sed "s#\"image\"[[:space:]]*:[[:space:]]*\"[^\"]*\"#\"image\": \"$(json_str "$_new")\"#" "$MANIFEST" > "$_tmp" \ || die "could not rewrite the image in $MANIFEST" chmod 0600 "$_tmp"; mv "$_tmp" "$MANIFEST" else echo "==> already on $_current — re-pulling and recreating" fi "$BIN_DEST" apply } cmd_uninstall() { need_root systemctl disable --now "$SERVICE_NAME" 2>/dev/null || true rm -f "$UNIT_DEST"; systemctl daemon-reload 2>/dev/null || true echo "Removed the systemd service. Left in place: $BIN_DEST, $ETC_DIR (incl. identity)," echo "the data root and any app container. Delete them manually for a full cleanup." } usage() { cat < [--cluster N] [--cache-dir P] [--data-root P] [--http-port P] [--network host|bridge|] [--binary ] gpuk install --mode worker --controller wss://: --enroll-token gk_enroll_... [--cluster N] [--cache-dir P] [--browse-root P]... [--binary ] gpuk status Service state, enrollment, /health, app container status gpuk enroll --controller wss://: --token gk_enroll_... gpuk apply (controller) Reconcile the app container from the manifest gpuk update (controller) Move to the current release (pull + recreate, rollback) gpuk update --check (controller) Compare the installed version with the release gpuk update --version (controller) Move to a specific release gpuk manifest Print the current manifest gpuk logs Follow the app container logs (controller) or the daemon journal gpuk uninstall Remove the systemd service Most people never run this directly: the channel's install.sh installs it (https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/install.sh). EOF } # ── dispatch ──────────────────────────────────────────────────────────────────── cmd="${1:-help}"; if [ $# -gt 0 ]; then shift; fi case "$cmd" in install) cmd_install "$@" ;; status) cmd_status ;; enroll) cmd_enroll "$@" ;; apply) cmd_apply ;; update) cmd_update "$@" ;; manifest) cat "$MANIFEST" ;; logs) _img=$(manifest_image) if [ -n "$_img" ]; then exec docker logs -f "$(container_name)"; else exec journalctl -u "$SERVICE_NAME" -f; fi ;; uninstall) cmd_uninstall ;; help|-h|--help) usage ;; *) usage; exit 1 ;; esac