commit 6336279269a967b07764853b1f5e13be3f25598f Author: gpuk-release Date: Mon Aug 3 09:02:45 2026 +0000 release v0.1.1 diff --git a/gpuk b/gpuk new file mode 100755 index 0000000..c7f812a --- /dev/null +++ b/gpuk @@ -0,0 +1,573 @@ +#!/bin/sh +# gpuk — installer + control CLI for the GPU Kitchen host daemon (gpu-kitchen-worker). +# +# Since C1 a compute node runs NO backend. The single privileged host component is +# the Rust `gpu-kitchen-worker` daemon (workerd), running as root DIRECTLY on the +# host (not in a container). It owns NVML clock/power locks, whitelisted host +# browsing, the model-cache staging pipeline, the signed `gpu-kitchen-bench` +# runner, and — on a controller node — the app container's lifecycle (create, +# health-gate, roll back, pull+recreate to update) via its manifest. There is no +# separate host daemon and no unix control socket: workerd is driven over the +# wss+mTLS channel by the controller, and locally by this CLI. +# +# Two roles: +# worker a headless compute node. Installs the binary + systemd unit + a +# manifest (cacheDisks, browseRoots) with NO app container, NO +# Postgres, NO docker app image. It ENROLLS over mTLS with a +# enrollment token, or waits for LAN discovery admission, then runs. +# controller the full app + UI. Installs the binary + systemd unit + an +# app-container manifest (image, ports, env, secretsRef) that workerd +# applies (`gpu-kitchen-worker apply`, then cmd:host_apply in steady +# state). +# +# Install (root): +# # controller: +# curl -fsSL https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/gpuk \ +# | sudo sh -s -- install --mode controller \ +# --image repo.byterain.io/gpukitchen-public/gpukitchen-controller:vX.Y.Z +# # worker (enroll against a controller with a single-use token from its UI): +# sudo ./deployments/install/gpuk install --mode worker \ +# --controller wss://controller.lan:8443 --enroll-token gk_enroll_... \ +# --cache-dir /mnt/models --binary apps/worker/target/release/gpu-kitchen-worker +# +# Control: +# gpuk status | apply | update | enroll | logs | manifest | uninstall +# +set -eu + +# ── Paths ──────────────────────────────────────────────────────────────────── +# Overridable, so the daemon can be driven against a prefix a normal user owns — +# the only way any of this is testable without handing a test suite root on the +# host. Unset (the real install) they are exactly the systemd defaults workerd uses +# (apps/worker/src/module.rs, apps/worker/src/identity.rs). +BIN_DEST="${GPUK_BIN_DEST:-/usr/local/bin/gpu-kitchen-worker}" +ETC_DIR="${GPUK_ETC_DIR:-/etc/gpu-kitchen}" +MANIFEST="$ETC_DIR/manifest.json" +IDENTITY_DIR="${GPUK_IDENTITY_DIR:-$ETC_DIR/identity}" +WORKER_ENV="$ETC_DIR/worker.env" +SERVICE_NAME="gpu-kitchen-worker" +UNIT_DEST="/etc/systemd/system/$SERVICE_NAME.service" + +# Where to download the binary from when no --binary is given. +GPUK_RELEASE_BASE="${GPUK_RELEASE_BASE:-}" + +die() { echo "gpuk: $*" >&2; exit 1; } + +# Root, or able to do the job anyway. Fail with a clear message BEFORE touching +# /etc, /usr/local/bin or systemd. Against a user-owned prefix it simply works. +need_root() { + [ "$(id -u)" -eq 0 ] && return 0 + [ -w "$ETC_DIR" ] && return 0 + die "this command must run as root (use sudo)" +} + +# ── JSON helpers (controlled inputs: paths + identifiers) ────────────────────── +json_str() { printf '%s' "$1" | sed 's/\\/\\\\/g; s/"/\\"/g'; } + +json_array() { # args → ["a","b",...] + _out="" + for _p in "$@"; do + _e=$(json_str "$_p") + if [ -z "$_out" ]; then _out="\"$_e\""; else _out="$_out,\"$_e\""; fi + done + printf '[%s]' "$_out" +} + +# ── install ──────────────────────────────────────────────────────────────────── +arch_asset() { + case "$(uname -m)" in + x86_64|amd64) echo "gpu-kitchen-worker-x86_64" ;; + aarch64|arm64) echo "gpu-kitchen-worker-aarch64" ;; + *) die "unsupported architecture: $(uname -m)" ;; + esac +} + +install_binary() { # [local-path] + if [ -n "${1:-}" ]; then + [ -f "$1" ] || die "binary not found: $1" + install -m 0755 "$1" "$BIN_DEST" + echo "==> installed $BIN_DEST from $1" + elif [ -n "$GPUK_RELEASE_BASE" ]; then + command -v curl >/dev/null || die "curl is required to download the binary" + _url="$GPUK_RELEASE_BASE/$(arch_asset)" + echo "==> downloading $_url" + curl -fsSL "$_url" -o "$BIN_DEST.new" + chmod 0755 "$BIN_DEST.new" + mv "$BIN_DEST.new" "$BIN_DEST" + elif [ -x "$BIN_DEST" ]; then + echo "==> reusing existing $BIN_DEST" + else + die "no binary: pass --binary or set GPUK_RELEASE_BASE=" + fi +} + +gen_secret() { head -c 32 /dev/urandom | base64 | tr '+/' '-_' | tr -d '='; } +# A password a human retypes once, from a terminal. Ambiguous glyphs removed. +gen_password() { head -c 24 /dev/urandom | base64 | tr -d '=+/OIl01' | cut -c1-16; } + +seed_secrets() { # DATA_ROOT MODE + _sd="$1/secrets" + mkdir -p "$_sd"; chmod 0700 "$_sd" + [ -f "$_sd/encryption_key" ] || { umask 077; gen_secret > "$_sd/encryption_key"; } + [ -f "$_sd/node_id" ] || { umask 077; (cat /proc/sys/kernel/random/uuid 2>/dev/null || uuidgen) > "$_sd/node_id"; } + # The controller's first-run password. Generated here — NOT left to the backend + # to print into a log nobody watches when the install is one piped command. The + # installer prints it once, and the first-run wizard makes the operator replace + # it (GPUK_BOOTSTRAP_MUST_CHANGE). + if [ "$2" = "controller" ] && [ ! -f "$_sd/bootstrap_admin_password" ]; then + umask 077; gen_password > "$_sd/bootstrap_admin_password" + fi +} + +# Write the systemd unit: prefer a sibling file, else embed. The daemon runs the +# binary with NO arguments (steady-state); role/controller URL/CA come from the +# enrolled identity (worker) and the optional EnvironmentFile. +write_unit() { + _src_unit="$(dirname "$0")/../../apps/worker/install/$SERVICE_NAME.service" + [ -f "$_src_unit" ] || _src_unit="$(dirname "$0")/$SERVICE_NAME.service" + if [ -f "$_src_unit" ]; then + install -m 0644 "$_src_unit" "$UNIT_DEST" + else + cat > "$UNIT_DEST" </dev/null || echo "$1") + hec_found="" + for hec_dir in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do + [ -n "$hec_dir" ] || continue + [ -d "$hec_dir/hub" ] || continue + hec_real=$(readlink -f "$hec_dir" 2>/dev/null || echo "$hec_dir") + [ "$hec_real" != "$hec_chosen" ] || continue + # Hub layout only (models--*) — matches what the scan can actually reference. + ls -d "$hec_dir"/hub/models--* >/dev/null 2>&1 || continue + hec_found="$hec_found $hec_real" + done + [ -n "$hec_found" ] || return 0 + echo + for hec_dir in $hec_found; do + echo "==> existing model cache found at $hec_dir" + done + echo " Reference it from Settings -> Cache folders to reuse those models." + echo " GPU Kitchen only READS a referenced cache: nothing is moved or deleted." +} + +cmd_install() { + need_root + IMAGE=""; MODE="worker"; CLUSTER="default"; DATA_ROOT="/var/lib/gpu-kitchen" + CACHE_DIR=""; NETWORK="host"; CONTROLLER_URL=""; BIN_SRC=""; HEALTH_PORT="8001" + BROWSE_ROOTS=""; ENROLL_TOKEN=""; HTTP_PORT="8080" + while [ $# -gt 0 ]; do + case "$1" in + --image) IMAGE="$2"; shift 2 ;; + --mode) MODE="$2"; shift 2 ;; + --cluster) CLUSTER="$2"; shift 2 ;; + --data-root) DATA_ROOT="$2"; shift 2 ;; + --cache-dir) CACHE_DIR="$2"; shift 2 ;; + --network) NETWORK="$2"; shift 2 ;; + --controller) CONTROLLER_URL="$2"; shift 2 ;; + # `--token` is an ENROLLMENT token (single-use or shared), NOT a bearer: + # the worker↔controller channel is cert-only since D11. `--token` is kept as + # a spelling of `--enroll-token`. + --enroll-token|--token) ENROLL_TOKEN="$2"; shift 2 ;; + --health-port) HEALTH_PORT="$2"; shift 2 ;; + --http-port) HTTP_PORT="$2"; shift 2 ;; + --binary) BIN_SRC="$2"; shift 2 ;; + --browse-root) BROWSE_ROOTS="$BROWSE_ROOTS $2"; shift 2 ;; + *) die "unknown install option: $1" ;; + esac + done + + # Roles, as the backend names them (multi-server/config.ts): "controller" (full + # app + UI, accepts workers) and "worker" (headless compute node). Historical + # spellings still work. + case "$MODE" in + controller|server|standalone|manager) MODE="controller" ;; + worker|agent) MODE="worker" ;; + *) die "--mode must be controller or worker (got '$MODE')" ;; + esac + + [ "$MODE" != "controller" ] || [ -n "$IMAGE" ] \ + || die "--image registry/gpukitchen-controller: is required for a controller" + [ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf" + + install_binary "$BIN_SRC" + + mkdir -p "$ETC_DIR" "$DATA_ROOT" "$CACHE_DIR" + seed_secrets "$DATA_ROOT" "$MODE" + + # Default browse roots: common mount points + the dirs we already use. + [ -n "$BROWSE_ROOTS" ] || BROWSE_ROOTS="/mnt /data /srv $DATA_ROOT $(dirname "$CACHE_DIR")" + # shellcheck disable=SC2086 + BROWSE_JSON=$(json_array $BROWSE_ROOTS) + CACHE_JSON="[{\"hostPath\":\"$(json_str "$CACHE_DIR")\",\"shared\":false}]" + + umask 077 + if [ "$MODE" = "worker" ]; then + write_worker_manifest + # Optional env overrides for the daemon (cluster grouping, display name, an + # explicit controller URL). The enrolled identity carries the controller URL + # + CA too — this is belt-and-braces / pre-enroll discovery grouping. + { + echo "GPUK_CLUSTER=$CLUSTER" + echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT" + echo "NODE_DISPLAY_NAME=$(hostname)" + [ -n "$CONTROLLER_URL" ] && echo "GPUK_CONTROLLER_URLS=$CONTROLLER_URL" + } > "$WORKER_ENV" + chmod 0600 "$WORKER_ENV" + echo "==> wrote $MANIFEST (worker: no app container)" + else + write_controller_manifest + echo "==> wrote $MANIFEST (controller: app container $IMAGE)" + fi + chmod 0600 "$MANIFEST" + + write_unit + systemctl daemon-reload + echo "==> wrote $UNIT_DEST" + + # ── Worker: token enrollment before start, or unattended LAN discovery ── + if [ "$MODE" = "worker" ]; then + if [ -n "$ENROLL_TOKEN" ]; then + [ -n "$CONTROLLER_URL" ] || die "--controller wss://: is required to enroll" + echo "==> enrolling against $CONTROLLER_URL ..." + GPUK_IDENTITY_DIR="$IDENTITY_DIR" "$BIN_DEST" enroll \ + --controller "$CONTROLLER_URL" --token "$ENROLL_TOKEN" \ + || die "enrollment failed (bad/expired token, or controller unreachable)" + else + echo "==> no token given: the daemon will discover its cluster on the LAN and wait" + echo " for automatic admission or administrator approval." + fi + systemctl enable --now "$SERVICE_NAME" + echo "==> $SERVICE_NAME enabled and started" + echo + echo "Done. The worker daemon is running${ENROLL_TOKEN:+ and enrolled}." + echo " Status : gpuk status Logs: gpuk logs" + hint_existing_caches "$CACHE_DIR" + return 0 + fi + + # ── Controller: start the daemon, then bring up the app container ── + systemctl enable --now "$SERVICE_NAME" + echo "==> $SERVICE_NAME enabled and started" + echo "==> applying manifest (first app-container start) ..." + # No unix socket any more: workerd reconciles the app container in-process from + # the on-disk manifest (there is no backend to relay through on the very first + # boot). Steady-state updates go through the daemon over WS. + "$BIN_DEST" apply || die "apply failed — check: gpuk logs" + echo + echo "Done. The worker daemon is running and the app container is up." + echo " Status : gpuk status App logs: gpuk logs Daemon: journalctl -u $SERVICE_NAME" + hint_existing_caches "$CACHE_DIR" +} + +# Worker manifest: cacheDisks + browseRoots, and an EMPTY image so +# has_app_container() is false (apps/worker/src/modules/manifest.rs). No ports, no env, +# no secretsRef — a worker runs no app container. +write_worker_manifest() { + cat > "$MANIFEST" < "$MANIFEST" </dev/null | head -1 +} + +manifest_image() { + sed -n 's/.*"image"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1 +} + +health_port() { + sed -n 's/.*GPUK_WORKER_HEALTH_PORT=\([0-9]*\).*/\1/p' "$WORKER_ENV" 2>/dev/null | head -1 +} + +# Split an image reference into repository and tag. +# +# `${img%%:*}` cuts at the FIRST colon and is WRONG: +# `registry.internal:5000/gpuk/controller:v1.2.3` would yield repo +# `registry.internal`. The colon in a registry's host:port is not a tag separator. +# Rule: it is a tag only if the last colon comes after the last slash. +# (Same logic as apps/controller/api/src/core/release-channel.ts — see its unit tests.) +image_repo() { + case "$1" in + *@*) printf '%s' "${1%@*}"; return ;; # digest pin → repo@sha256:… + esac + _t="${1##*:}" + case "$_t" in + "$1") printf '%s' "$1" ;; # no colon at all → untagged + */*) printf '%s' "$1" ;; # the last colon is inside a path → host:port, untagged + *) printf '%s' "${1%:*}" ;; + esac +} + +image_tag() { + case "$1" in + *@*) return ;; # digest pin: no version to speak of + esac + _t="${1##*:}" + case "$_t" in + "$1") return ;; + */*) return ;; + *) printf '%s' "$_t" ;; + esac +} + +# The release channel: one flat JSON document served next to the installer. The +# UI's update check reads the same one (apps/controller/api/src/core/release-channel.ts) — one +# source of truth. Interim default: the public Gitea channel repo — flips to +# https://gpu.kitchen/latest.json once the hub exists (specs/developpement/ci-cd.md). +CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/latest.json}" + +channel_version() { + command -v curl >/dev/null 2>&1 || return 1 + curl -fsSL --max-time 20 "$CHANNEL_URL" 2>/dev/null \ + | sed -n 's/.*"version"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' | head -1 +} + +# ── status ───────────────────────────────────────────────────────────────────── +# No control socket any more. Status = the systemd unit state + the daemon's own +# /health endpoint + whether a worker has enrolled (identity present). +cmd_status() { + _active=$(systemctl is-active "$SERVICE_NAME" 2>/dev/null || true) + echo "service : $_active" + if [ -f "$IDENTITY_DIR/identity.json" ]; then + echo "enrolled : yes ($IDENTITY_DIR)" + else + echo "enrolled : no (daemon waits for LAN admission; token enrollment is also available)" + fi + _hp=$(health_port); [ -n "$_hp" ] || _hp=8001 + if command -v curl >/dev/null 2>&1; then + _h=$(curl -fsS --max-time 3 "http://127.0.0.1:$_hp/health" 2>/dev/null || true) + [ -n "$_h" ] && echo "health : $_h" || echo "health : (no answer on :$_hp)" + fi + _img=$(manifest_image) + if [ -n "$_img" ]; then + echo "app image : $_img" + echo "app cont. : $(docker inspect -f '{{.State.Status}}' "$(container_name)" 2>/dev/null || echo 'not running')" + else + echo "role : worker (no app container)" + fi +} + +# ── enroll ───────────────────────────────────────────────────────────────────── +cmd_enroll() { + need_root + _url=""; _tok="" + while [ $# -gt 0 ]; do + case "$1" in + --controller) _url="$2"; shift 2 ;; + --enroll-token|--token) _tok="$2"; shift 2 ;; + *) die "unknown enroll option: $1" ;; + esac + done + [ -n "$_url" ] || die "enroll needs --controller wss://:" + [ -n "$_tok" ] || die "enroll needs --token gk_enroll_..." + GPUK_IDENTITY_DIR="$IDENTITY_DIR" "$BIN_DEST" enroll --controller "$_url" --token "$_tok" + systemctl restart "$SERVICE_NAME" 2>/dev/null || true +} + +# ── apply ────────────────────────────────────────────────────────────────────── +# Controller: reconcile the app container from the manifest, in-process. A worker +# has no app container — `apply` there is a no-op with a clear message. +cmd_apply() { + need_root + "$BIN_DEST" apply +} + +# ── update ───────────────────────────────────────────────────────────────────── +# Controller: decide WHICH image TAG to pin, write it into the manifest, then let +# workerd pull + recreate (health-gate + rollback are the daemon's — apply()). +# Worker: the signed-binary self-update is DRIVEN FROM THE CONTROLLER (its Update +# button → POST /api/nodes/:id/host/update → cmd:host_update → verified swap). +# There is no local unverified swap path. +cmd_update() { + need_root + _want=""; _check=0 + while [ $# -gt 0 ]; do + case "$1" in + --version) _want="$2"; shift 2 ;; + --check) _check=1; shift ;; + *) die "unknown update option: $1" ;; + esac + done + + _image=$(manifest_image) + if [ -z "$_image" ]; then + echo "This is a worker node. Worker self-update is driven from the controller UI" + echo "(the node's Update button), which pushes a minisign-verified binary swap." + return 0 + fi + + _repo=$(image_repo "$_image") + _current=$(image_tag "$_image") + [ -n "$_current" ] || _current="(untagged)" + + if [ "$_check" -eq 1 ]; then + _latest=$(channel_version) || true + echo "installed : $_current" + if [ -z "$_latest" ]; then + echo "available : unknown (cannot reach $CHANNEL_URL)" + exit 1 + fi + echo "available : $_latest" + if [ "$_latest" = "$_current" ]; then echo "up to date."; else echo "run 'gpuk update' to move to $_latest"; fi + return 0 + fi + + if [ -z "$_want" ]; then + _want=$(channel_version) || true + if [ -z "$_want" ]; then + echo "==> cannot reach $CHANNEL_URL — re-applying the pinned $_current" + "$BIN_DEST" apply + return 0 + fi + fi + + if [ "$_want" != "$_current" ]; then + echo "==> $_current → $_want" + # Pin the new tag into the manifest, then apply. sed edits the single "image" + # line in place (atomic tmp + move). + _new="$_repo:$_want" + _tmp="$MANIFEST.new" + sed "s#\"image\"[[:space:]]*:[[:space:]]*\"[^\"]*\"#\"image\": \"$(json_str "$_new")\"#" "$MANIFEST" > "$_tmp" \ + || die "could not rewrite the image in $MANIFEST" + chmod 0600 "$_tmp"; mv "$_tmp" "$MANIFEST" + else + echo "==> already on $_current — re-pulling and recreating" + fi + "$BIN_DEST" apply +} + +cmd_uninstall() { + need_root + systemctl disable --now "$SERVICE_NAME" 2>/dev/null || true + rm -f "$UNIT_DEST"; systemctl daemon-reload 2>/dev/null || true + echo "Removed the systemd service. Left in place: $BIN_DEST, $ETC_DIR (incl. identity)," + echo "the data root and any app container. Delete them manually for a full cleanup." +} + +usage() { + cat < [--cluster N] [--cache-dir P] + [--data-root P] [--http-port P] [--network host|bridge|] [--binary ] + gpuk install --mode worker --controller wss://: --enroll-token gk_enroll_... + [--cluster N] [--cache-dir P] [--browse-root P]... [--binary ] + gpuk status Service state, enrollment, /health, app container status + gpuk enroll --controller wss://: --token gk_enroll_... + gpuk apply (controller) Reconcile the app container from the manifest + gpuk update (controller) Move to the current release (pull + recreate, rollback) + gpuk update --check (controller) Compare the installed version with the release + gpuk update --version (controller) Move to a specific release + gpuk manifest Print the current manifest + gpuk logs Follow the app container logs (controller) or the daemon journal + gpuk uninstall Remove the systemd service + +Most people never run this directly: the channel's install.sh installs it +(https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/install.sh). +EOF +} + +# ── dispatch ──────────────────────────────────────────────────────────────────── +cmd="${1:-help}"; if [ $# -gt 0 ]; then shift; fi +case "$cmd" in + install) cmd_install "$@" ;; + status) cmd_status ;; + enroll) cmd_enroll "$@" ;; + apply) cmd_apply ;; + update) cmd_update "$@" ;; + manifest) cat "$MANIFEST" ;; + logs) + _img=$(manifest_image) + if [ -n "$_img" ]; then exec docker logs -f "$(container_name)"; else exec journalctl -u "$SERVICE_NAME" -f; fi + ;; + uninstall) cmd_uninstall ;; + help|-h|--help) usage ;; + *) usage; exit 1 ;; +esac diff --git a/install.sh b/install.sh new file mode 100644 index 0000000..d11dd62 --- /dev/null +++ b/install.sh @@ -0,0 +1,368 @@ +#!/bin/sh +# GPU Kitchen — one-command install (specs/plateforme/installation.md). +# +# curl -fsSL https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/install.sh | sudo sh +# +# (Interim URL — becomes https://gpu.kitchen/install.sh once the hub exists; the +# channel repo is written by .gitea/workflows/release.yml, see specs/developpement/ci-cd.md.) +# +# What it does, and nothing more: +# 1. preflight docker, the NVIDIA driver, and a REAL `--gpus all` smoke test +# 2. resolve the current release from the channel (a TAG — never a floating +# `latest`: an install that silently changes version under you is +# not an install, it is a surprise) +# 3. install `gpu-kitchen-worker` (workerd), the privileged host daemon, via +# the existing `gpuk` installer — on a controller node it OWNS the +# app container's lifecycle +# 4. hand over workerd pulls the pinned all-in-one controller image and starts it +# 5. print the UI URL on the real host, and the first-run password +# +# Re-running is how you UPDATE: same command, newer tag, `docker pull` + recreate, +# data untouched (it lives in the data root, not the container). +# +# This installs a CONTROLLER node (the full app + UI). A headless compute node is +# `gpuk install --mode worker …` — see deployments/install/gpuk and +# specs/plateforme/installation.md. +# +# Design note — this script starts nothing itself. workerd owns the container's +# lifecycle (create, health-gate, roll back, update); the UI's update button and +# `gpuk update` drive that same daemon. One updater, three front doors. +set -eu + +# ── Defaults (every one overridable by flag or env) ─────────────────────────── +# The channel is the single source of truth for "what is the current release": +# it is served next to this script, and the backend's update check reads the SAME +# document (apps/controller/api/src/core/release-channel.ts). One file, one answer. Interim +# default: the public Gitea channel repo — flips to https://gpu.kitchen/latest.json +# once the hub exists (keep the three defaults in sync, see specs/developpement/ci-cd.md). +CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/latest.json}" +EDITION="${GPUK_EDITION:-community}" +DATA_ROOT="${GPUK_DATA_ROOT:-/var/lib/gpu-kitchen}" +CACHE_DIR="${GPUK_CACHE_DIR:-}" +PORT="${GPUK_PORT:-8080}" +CLUSTER="${GPUK_CLUSTER:-default}" +VERSION="" +IMAGE="" +WORKER_BINARY="" +GPUK_SCRIPT="" +SKIP_GPU_CHECK=0 +SKIP_PREFLIGHT=0 +DRY_RUN=0 + +GREEN=''; RED=''; YELLOW=''; BOLD=''; RESET='' +if [ -t 1 ]; then + GREEN=$(printf '\033[32m'); RED=$(printf '\033[31m') + YELLOW=$(printf '\033[33m'); BOLD=$(printf '\033[1m'); RESET=$(printf '\033[0m') +fi + +ok() { echo " ${GREEN}✓${RESET} $*"; } +warn() { echo " ${YELLOW}!${RESET} $*"; } +step() { echo; echo "${BOLD}$*${RESET}"; } +die() { echo " ${RED}✗${RESET} $*" >&2; exit 1; } + +usage() { + cat < Install this release instead of the channel's current one + --image Use this controller image outright (implies --version none) + --edition community (default) | enterprise + --port

Port the UI listens on (default 8080) + --data-root Where the database and secrets live (default /var/lib/gpu-kitchen) + --cache-dir Model cache (default /hf) + --cluster Cluster name workers join (default "default") + --worker-binary

Use a locally-built gpu-kitchen-worker instead of downloading one + --gpuk-script

Use a local copy of the gpuk installer + --skip-gpu-check Skip the 'docker run --gpus all' smoke test + --skip-preflight Skip the host checks entirely (CI: no docker, no GPU) + --dry-run Run the preflight and resolve the release, change nothing + -h, --help This +EOF +} + +while [ $# -gt 0 ]; do + case "$1" in + --version) VERSION="$2"; shift 2 ;; + --image) IMAGE="$2"; shift 2 ;; + --edition) EDITION="$2"; shift 2 ;; + --port) PORT="$2"; shift 2 ;; + --data-root) DATA_ROOT="$2"; shift 2 ;; + --cache-dir) CACHE_DIR="$2"; shift 2 ;; + --cluster) CLUSTER="$2"; shift 2 ;; + --worker-binary) WORKER_BINARY="$2"; shift 2 ;; + --gpuk-script) GPUK_SCRIPT="$2"; shift 2 ;; + --skip-gpu-check) SKIP_GPU_CHECK=1; shift ;; + --skip-preflight) SKIP_PREFLIGHT=1; shift ;; + --dry-run) DRY_RUN=1; shift ;; + -h|--help) usage; exit 0 ;; + *) die "unknown option: $1 (try --help)" ;; + esac +done + +[ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf" + +case "$EDITION" in + community|enterprise) ;; + *) die "--edition must be community or enterprise (got '$EDITION')" ;; +esac + +echo +echo "${BOLD}GPU Kitchen — installing the $EDITION controller${RESET}" + +# ── 1. Preflight ───────────────────────────────────────────────────────────── +# The same checks tools/provision-feeder.sh makes, minus the compose ones: the +# all-in-one image is driven by workerd through the plain docker CLI, so there is +# no compose dependency to satisfy any more. +if [ "$SKIP_PREFLIGHT" -eq 1 ]; then + +step "Preflight — skipped (--skip-preflight)" +warn "the host is NOT being checked for docker, a driver or a GPU" + +else + +step "Preflight — docker, NVIDIA driver, container toolkit" + +[ "$DRY_RUN" -eq 1 ] || [ "$(id -u)" -eq 0 ] \ + || die "run as root: curl -fsSL … | sudo sh" + +command -v curl >/dev/null 2>&1 || die "curl not found — install it first" + +if command -v docker >/dev/null 2>&1; then + if docker info >/dev/null 2>&1; then + ok "docker daemon reachable ($(docker --version | cut -d' ' -f3 | tr -d ,))" + else + die "docker is installed but its daemon is unreachable (is it running? are you root?)" + fi +else + die "docker not found — install Docker Engine first: https://docs.docker.com/engine/install/" +fi + +if command -v nvidia-smi >/dev/null 2>&1; then + DRIVER=$(nvidia-smi --query-gpu=driver_version --format=csv,noheader 2>/dev/null | head -1 || true) + GPU_COUNT=$(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | grep -c . || true) + if [ -n "$DRIVER" ] && [ "${GPU_COUNT:-0}" -gt 0 ]; then + ok "NVIDIA driver $DRIVER — ${GPU_COUNT} GPU(s): $(nvidia-smi --query-gpu=name --format=csv,noheader | sort -u | paste -sd', ')" + else + die "nvidia-smi is present but reports no GPU" + fi +else + die "nvidia-smi not found — install the NVIDIA driver first" +fi + +if [ "$SKIP_GPU_CHECK" -eq 1 ]; then + warn "nvidia-container-toolkit check skipped (--skip-gpu-check)" +elif docker info 2>/dev/null | grep -qiE 'runtimes:.*nvidia'; then + # The runtime being REGISTERED is not the same as it working. With the toolkit + # installed, --gpus injects the driver and nvidia-smi into a plain image; that + # is the exact mechanism the controller container relies on, so test it rather + # than infer it. + if docker run --rm --gpus all ubuntu:24.04 nvidia-smi -L >/dev/null 2>&1; then + ok "nvidia-container-toolkit works (a container can see the GPUs)" + else + die "the nvidia runtime is registered but 'docker run --gpus all' cannot see the GPUs — reinstall nvidia-container-toolkit" + fi +else + die "nvidia-container-toolkit is not registered with docker — install it, then restart dockerd: + https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html" +fi + +# Model weights are large and the failure mode (a download dying at 90%) is +# miserable, so say so up front. A warning, not a refusal: it is the user's disk. +CACHE_PARENT="$CACHE_DIR" +while [ ! -d "$CACHE_PARENT" ] && [ "$CACHE_PARENT" != "/" ]; do + CACHE_PARENT=$(dirname "$CACHE_PARENT") +done +FREE_GB=$(df -BG --output=avail "$CACHE_PARENT" 2>/dev/null | tail -1 | tr -dc '0-9' || true) +if [ "${FREE_GB:-0}" -ge 100 ]; then + ok "model cache $CACHE_DIR — ${FREE_GB}G free" +else + warn "only ${FREE_GB:-?}G free under $CACHE_PARENT — model weights want 100G+" +fi + +fi # end preflight + +# ── 2. Resolve the release ─────────────────────────────────────────────────── +step "Release — resolving the version to install" + +# One tiny JSON document, fetched over TLS, holding what the current release IS. +# Parsed with sed rather than jq: `curl … | sudo sh` cannot assume jq exists, and +# the document is ours and flat. +json_field() { sed -n "s/.*\"$1\"[[:space:]]*:[[:space:]]*\"\([^\"]*\)\".*/\1/p" | head -1; } + +# Strip a tag off an image reference WITHOUT mangling a registry's host:port. +# `${ref%%:*}` cuts at the FIRST colon and is WRONG: a +# `registry.internal:5000/gpuk/controller:v1.2.3` (or an untagged +# `registry.internal:5000/gpuk/controller`) would collapse to `registry.internal`. +# A colon is a tag separator only when the last colon comes AFTER the last slash. +# Same rule as deployments/install/gpuk's image_repo() and core/release-channel.ts (unit-tested). +image_repo() { + case "$1" in + *@*) printf '%s' "${1%@*}"; return ;; # digest pin → repo@sha256:… + esac + _t="${1##*:}" + case "$_t" in + "$1") printf '%s' "$1" ;; # no colon at all → already untagged + */*) printf '%s' "$1" ;; # last colon is inside a path → host:port, untagged + *) printf '%s' "${1%:*}" ;; + esac +} + +if [ -n "$IMAGE" ]; then + ok "using the image given on the command line: $IMAGE" + [ -n "$VERSION" ] || VERSION="(pinned by --image)" +else + CHANNEL=$(curl -fsSL --max-time 20 "$CHANNEL_URL" 2>/dev/null) \ + || die "cannot reach the release channel at $CHANNEL_URL + (offline? pass --image to install a specific image directly)" + + [ -n "$VERSION" ] || VERSION=$(echo "$CHANNEL" | json_field version) + [ -n "$VERSION" ] || die "the release channel returned no version: $CHANNEL_URL" + + if [ "$EDITION" = "enterprise" ]; then + IMAGE_TEMPLATE=$(echo "$CHANNEL" | json_field controllerImageEnterprise) + else + IMAGE_TEMPLATE=$(echo "$CHANNEL" | json_field controllerImage) + fi + [ -n "$IMAGE_TEMPLATE" ] \ + || die "the release channel names no $EDITION controller image: $CHANNEL_URL" + + # The channel gives the repository; WE pin the tag. A floating `:latest` would + # make every container recreate a silent, unrequested upgrade. Strip any tag the + # channel already carries with image_repo (host:port-safe), then pin OUR version. + IMAGE="$(image_repo "$IMAGE_TEMPLATE"):${VERSION}" + ok "release $VERSION" + ok "image $IMAGE" + + [ -n "$WORKER_BINARY" ] || WORKER_RELEASE_BASE=$(echo "$CHANNEL" | json_field workerBase) + [ -n "${GPUK_SCRIPT}" ] || GPUK_SCRIPT_URL=$(echo "$CHANNEL" | json_field gpukScript) +fi + +if [ "$DRY_RUN" -eq 1 ]; then + step "Dry run — stopping here" + if [ "$SKIP_PREFLIGHT" -eq 1 ]; then + ok "the release resolved; the host was not checked; nothing was installed" + else + ok "preflight passed and the release resolved; nothing was installed" + fi + echo + echo " would install : $IMAGE" + echo " data root : $DATA_ROOT" + echo " model cache : $CACHE_DIR" + echo " UI port : $PORT" + exit 0 +fi + +# ── 3. The host daemon ─────────────────────────────────────────────────────── +step "Host daemon — gpu-kitchen-worker" + +TMP=$(mktemp -d) +# shellcheck disable=SC2064 # expand TMP now: it must be removed even if it changes +trap "rm -rf '$TMP'" EXIT INT TERM + +if [ -z "$GPUK_SCRIPT" ]; then + # A checkout right here beats a download (that is how contributors run it). + _local="$(dirname "$0")/gpuk" + if [ -f "$_local" ]; then + GPUK_SCRIPT="$_local" + ok "using the gpuk installer from this checkout" + else + [ -n "${GPUK_SCRIPT_URL:-}" ] \ + || die "the release channel names no gpuk installer, and none was found locally" + curl -fsSL --max-time 60 "$GPUK_SCRIPT_URL" -o "$TMP/gpuk" \ + || die "cannot download the gpuk installer from $GPUK_SCRIPT_URL" + chmod +x "$TMP/gpuk" + GPUK_SCRIPT="$TMP/gpuk" + ok "downloaded the gpuk installer" + fi +fi + +# `gpuk install` does the rest: it drops the binary, writes the systemd unit, +# writes the manifest (the declarative description of the app container) and +# applies it. Everything below is passed straight through to it. +set -- install \ + --mode controller \ + --image "$IMAGE" \ + --cluster "$CLUSTER" \ + --data-root "$DATA_ROOT" \ + --cache-dir "$CACHE_DIR" \ + --http-port "$PORT" + +if [ -n "$WORKER_BINARY" ]; then + [ -f "$WORKER_BINARY" ] || die "no such worker binary: $WORKER_BINARY" + set -- "$@" --binary "$WORKER_BINARY" + ok "using a locally-built gpu-kitchen-worker" +elif [ -n "${WORKER_RELEASE_BASE:-}" ]; then + GPUK_RELEASE_BASE="$WORKER_RELEASE_BASE" + export GPUK_RELEASE_BASE + ok "gpu-kitchen-worker will be downloaded from the release" +fi +# else: gpuk reuses an already-installed binary, or fails with its own message. + +step "Installing — this pulls the image, so it can take a few minutes" +sh "$GPUK_SCRIPT" "$@" || die "the install failed — see: journalctl -u gpu-kitchen-worker" + +# ── 4. Wait for the app, then say where it is ──────────────────────────────── +step "Waiting for the controller to answer" + +i=0 +until curl -fsS -o /dev/null --max-time 3 "http://127.0.0.1:$PORT/health" 2>/dev/null; do + i=$((i + 1)) + [ "$i" -lt 90 ] || die "the controller did not answer on :$PORT within 90s — see: gpuk logs" + sleep 2 +done +ok "the controller is up" + +# The URL must name the REAL host: the person installing this is very often not +# sitting at the machine, and "localhost" would be a lie on every box but theirs +# (same reason the backend resolves its own hostname — core/host-name.ts). +HOSTNAME_FQDN=$(hostname -f 2>/dev/null || hostname 2>/dev/null || echo localhost) +LAN_IP=$(ip route get 1.1.1.1 2>/dev/null | sed -n 's/.* src \([0-9.]*\).*/\1/p' | head -1) + +PW_FILE="$DATA_ROOT/secrets/bootstrap_admin_password" +ADMIN_PW="" +[ -f "$PW_FILE" ] && ADMIN_PW=$(cat "$PW_FILE" 2>/dev/null || true) + +echo +echo "${BOLD}${GREEN}GPU Kitchen is running.${RESET}" +echo +echo " ${BOLD}Open:${RESET} http://${HOSTNAME_FQDN}:${PORT}" +[ -n "$LAN_IP" ] && echo " http://${LAN_IP}:${PORT}" +echo +if [ -n "$ADMIN_PW" ]; then + echo " ${BOLD}Sign in:${RESET} admin@local" + echo " ${BOLD}Password:${RESET} ${ADMIN_PW}" + echo " (the first-run wizard asks you to change it)" + echo +fi +echo " Inference endpoint : http://${HOSTNAME_FQDN}:8200/v1" +echo " Version : ${VERSION}" +echo + +# Pre-existing model caches (B81). `gpuk install` prints the same hint, but that +# scrolls past mid-install; this banner is where people actually look. Detection +# only — the install stays non-interactive and nothing outside $CACHE_DIR is +# touched, read as configuration, or modified. +EXISTING_CACHE="" +CHOSEN_CACHE=$(readlink -f "$CACHE_DIR" 2>/dev/null || echo "$CACHE_DIR") +for d in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do + [ -n "$d" ] && [ -d "$d/hub" ] || continue + real=$(readlink -f "$d" 2>/dev/null || echo "$d") + [ "$real" != "$CHOSEN_CACHE" ] || continue + ls -d "$d"/hub/models--* >/dev/null 2>&1 || continue + EXISTING_CACHE="$EXISTING_CACHE $real" +done +if [ -n "$EXISTING_CACHE" ]; then + for d in $EXISTING_CACHE; do + echo " ${BOLD}Existing model cache:${RESET} $d" + done + echo " Reference it from Settings -> Cache folders to reuse those models" + echo " without downloading again. GPU Kitchen only READS a referenced cache." + echo +fi +echo " Update : re-run this command, or press Update in the UI, or: gpuk update" +echo " Status : gpuk status Logs: gpuk logs Remove: gpuk uninstall" +echo diff --git a/latest.json b/latest.json new file mode 100644 index 0000000..7dcd470 --- /dev/null +++ b/latest.json @@ -0,0 +1,9 @@ +{ + "version": "v0.1.1", + "semver": "0.1.1", + "controllerImage": "repo.byterain.io/gpukitchen-public/gpukitchen-controller", + "controllerImageEnterprise": "repo.byterain.io/gpukitchen/gpukitchen-controller-ee", + "workerBase": "https://repo.byterain.io/api/packages/gpukitchen-public/generic/gpu-kitchen-worker/v0.1.1", + "gpukScript": "https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/gpuk", + "releaseNotes": "https://repo.byterain.io/gpukitchen-public/channel" +}