#!/bin/sh # GPU Kitchen — one-command install (specs/plateforme/installation.md). # # curl -fsSL https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/install.sh | sudo sh # # (Interim URL — becomes https://gpu.kitchen/install.sh once the hub exists; the # channel repo is written by .gitea/workflows/release.yml, see specs/developpement/ci-cd.md.) # # What it does, and nothing more: # 1. preflight docker, the NVIDIA driver, and a REAL `--gpus all` smoke test # 2. resolve the current release from the channel (a TAG — never a floating # `latest`: an install that silently changes version under you is # not an install, it is a surprise) # 3. install `gpu-kitchen-worker` (workerd), the privileged host daemon, via # the existing `gpuk` installer — on a controller node it OWNS the # app container's lifecycle # 4. hand over workerd pulls the pinned all-in-one controller image and starts it # 5. print the UI URL on the real host, and the first-run password # # Re-running is how you UPDATE: same command, newer tag, `docker pull` + recreate, # data untouched (it lives in the data root, not the container). # # This installs a CONTROLLER node (the full app + UI). A headless compute node is # `gpuk install --mode worker …` — see deployments/install/gpuk and # specs/plateforme/installation.md. # # Design note — this script starts nothing itself. workerd owns the container's # lifecycle (create, health-gate, roll back, update); the UI's update button and # `gpuk update` drive that same daemon. One updater, three front doors. set -eu # ── Defaults (every one overridable by flag or env) ─────────────────────────── # The channel is the single source of truth for "what is the current release": # it is served next to this script, and the backend's update check reads the SAME # document (apps/controller/api/src/core/release-channel.ts). One file, one answer. Interim # default: the public Gitea channel repo — flips to https://gpu.kitchen/latest.json # once the hub exists (keep the three defaults in sync, see specs/developpement/ci-cd.md). CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/latest.json}" EDITION="${GPUK_EDITION:-community}" DATA_ROOT="${GPUK_DATA_ROOT:-/var/lib/gpu-kitchen}" CACHE_DIR="${GPUK_CACHE_DIR:-}" PORT="${GPUK_PORT:-8080}" CLUSTER="${GPUK_CLUSTER:-default}" VERSION="" IMAGE="" WORKER_BINARY="" GPUK_SCRIPT="" SKIP_GPU_CHECK=0 SKIP_PREFLIGHT=0 DRY_RUN=0 GREEN=''; RED=''; YELLOW=''; BOLD=''; RESET='' if [ -t 1 ]; then GREEN=$(printf '\033[32m'); RED=$(printf '\033[31m') YELLOW=$(printf '\033[33m'); BOLD=$(printf '\033[1m'); RESET=$(printf '\033[0m') fi ok() { echo " ${GREEN}✓${RESET} $*"; } warn() { echo " ${YELLOW}!${RESET} $*"; } step() { echo; echo "${BOLD}$*${RESET}"; } die() { echo " ${RED}✗${RESET} $*" >&2; exit 1; } usage() { cat < Install this release instead of the channel's current one --image Use this controller image outright (implies --version none) --edition community (default) | enterprise --port

Port the UI listens on (default 8080) --data-root Where the database and secrets live (default /var/lib/gpu-kitchen) --cache-dir Model cache (default /hf) --cluster Cluster name workers join (default "default") --worker-binary

Use a locally-built gpu-kitchen-worker instead of downloading one --gpuk-script

Use a local copy of the gpuk installer --skip-gpu-check Skip the 'docker run --gpus all' smoke test --skip-preflight Skip the host checks entirely (CI: no docker, no GPU) --dry-run Run the preflight and resolve the release, change nothing -h, --help This EOF } while [ $# -gt 0 ]; do case "$1" in --version) VERSION="$2"; shift 2 ;; --image) IMAGE="$2"; shift 2 ;; --edition) EDITION="$2"; shift 2 ;; --port) PORT="$2"; shift 2 ;; --data-root) DATA_ROOT="$2"; shift 2 ;; --cache-dir) CACHE_DIR="$2"; shift 2 ;; --cluster) CLUSTER="$2"; shift 2 ;; --worker-binary) WORKER_BINARY="$2"; shift 2 ;; --gpuk-script) GPUK_SCRIPT="$2"; shift 2 ;; --skip-gpu-check) SKIP_GPU_CHECK=1; shift ;; --skip-preflight) SKIP_PREFLIGHT=1; shift ;; --dry-run) DRY_RUN=1; shift ;; -h|--help) usage; exit 0 ;; *) die "unknown option: $1 (try --help)" ;; esac done [ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf" case "$EDITION" in community|enterprise) ;; *) die "--edition must be community or enterprise (got '$EDITION')" ;; esac echo echo "${BOLD}GPU Kitchen — installing the $EDITION controller${RESET}" # ── 1. Preflight ───────────────────────────────────────────────────────────── # The same checks tools/provision-feeder.sh makes, minus the compose ones: the # all-in-one image is driven by workerd through the plain docker CLI, so there is # no compose dependency to satisfy any more. if [ "$SKIP_PREFLIGHT" -eq 1 ]; then step "Preflight — skipped (--skip-preflight)" warn "the host is NOT being checked for docker, a driver or a GPU" else step "Preflight — docker, NVIDIA driver, container toolkit" [ "$DRY_RUN" -eq 1 ] || [ "$(id -u)" -eq 0 ] \ || die "run as root: curl -fsSL … | sudo sh" command -v curl >/dev/null 2>&1 || die "curl not found — install it first" if command -v docker >/dev/null 2>&1; then if docker info >/dev/null 2>&1; then ok "docker daemon reachable ($(docker --version | cut -d' ' -f3 | tr -d ,))" else die "docker is installed but its daemon is unreachable (is it running? are you root?)" fi else die "docker not found — install Docker Engine first: https://docs.docker.com/engine/install/" fi if command -v nvidia-smi >/dev/null 2>&1; then DRIVER=$(nvidia-smi --query-gpu=driver_version --format=csv,noheader 2>/dev/null | head -1 || true) GPU_COUNT=$(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | grep -c . || true) if [ -n "$DRIVER" ] && [ "${GPU_COUNT:-0}" -gt 0 ]; then ok "NVIDIA driver $DRIVER — ${GPU_COUNT} GPU(s): $(nvidia-smi --query-gpu=name --format=csv,noheader | sort -u | paste -sd', ')" else die "nvidia-smi is present but reports no GPU" fi else die "nvidia-smi not found — install the NVIDIA driver first" fi if [ "$SKIP_GPU_CHECK" -eq 1 ]; then warn "nvidia-container-toolkit check skipped (--skip-gpu-check)" elif docker info 2>/dev/null | grep -qiE 'runtimes:.*nvidia'; then # The runtime being REGISTERED is not the same as it working. With the toolkit # installed, --gpus injects the driver and nvidia-smi into a plain image; that # is the exact mechanism the controller container relies on, so test it rather # than infer it. if docker run --rm --gpus all ubuntu:24.04 nvidia-smi -L >/dev/null 2>&1; then ok "nvidia-container-toolkit works (a container can see the GPUs)" else die "the nvidia runtime is registered but 'docker run --gpus all' cannot see the GPUs — reinstall nvidia-container-toolkit" fi else die "nvidia-container-toolkit is not registered with docker — install it, then restart dockerd: https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html" fi # Model weights are large and the failure mode (a download dying at 90%) is # miserable, so say so up front. A warning, not a refusal: it is the user's disk. CACHE_PARENT="$CACHE_DIR" while [ ! -d "$CACHE_PARENT" ] && [ "$CACHE_PARENT" != "/" ]; do CACHE_PARENT=$(dirname "$CACHE_PARENT") done FREE_GB=$(df -BG --output=avail "$CACHE_PARENT" 2>/dev/null | tail -1 | tr -dc '0-9' || true) if [ "${FREE_GB:-0}" -ge 100 ]; then ok "model cache $CACHE_DIR — ${FREE_GB}G free" else warn "only ${FREE_GB:-?}G free under $CACHE_PARENT — model weights want 100G+" fi fi # end preflight # ── 2. Resolve the release ─────────────────────────────────────────────────── step "Release — resolving the version to install" # One tiny JSON document, fetched over TLS, holding what the current release IS. # Parsed with sed rather than jq: `curl … | sudo sh` cannot assume jq exists, and # the document is ours and flat. json_field() { sed -n "s/.*\"$1\"[[:space:]]*:[[:space:]]*\"\([^\"]*\)\".*/\1/p" | head -1; } # Strip a tag off an image reference WITHOUT mangling a registry's host:port. # `${ref%%:*}` cuts at the FIRST colon and is WRONG: a # `registry.internal:5000/gpuk/controller:v1.2.3` (or an untagged # `registry.internal:5000/gpuk/controller`) would collapse to `registry.internal`. # A colon is a tag separator only when the last colon comes AFTER the last slash. # Same rule as deployments/install/gpuk's image_repo() and core/release-channel.ts (unit-tested). image_repo() { case "$1" in *@*) printf '%s' "${1%@*}"; return ;; # digest pin → repo@sha256:… esac _t="${1##*:}" case "$_t" in "$1") printf '%s' "$1" ;; # no colon at all → already untagged */*) printf '%s' "$1" ;; # last colon is inside a path → host:port, untagged *) printf '%s' "${1%:*}" ;; esac } if [ -n "$IMAGE" ]; then ok "using the image given on the command line: $IMAGE" [ -n "$VERSION" ] || VERSION="(pinned by --image)" else CHANNEL=$(curl -fsSL --max-time 20 "$CHANNEL_URL" 2>/dev/null) \ || die "cannot reach the release channel at $CHANNEL_URL (offline? pass --image to install a specific image directly)" [ -n "$VERSION" ] || VERSION=$(echo "$CHANNEL" | json_field version) [ -n "$VERSION" ] || die "the release channel returned no version: $CHANNEL_URL" if [ "$EDITION" = "enterprise" ]; then IMAGE_TEMPLATE=$(echo "$CHANNEL" | json_field controllerImageEnterprise) else IMAGE_TEMPLATE=$(echo "$CHANNEL" | json_field controllerImage) fi [ -n "$IMAGE_TEMPLATE" ] \ || die "the release channel names no $EDITION controller image: $CHANNEL_URL" # The channel gives the repository; WE pin the tag. A floating `:latest` would # make every container recreate a silent, unrequested upgrade. Strip any tag the # channel already carries with image_repo (host:port-safe), then pin OUR version. IMAGE="$(image_repo "$IMAGE_TEMPLATE"):${VERSION}" ok "release $VERSION" ok "image $IMAGE" [ -n "$WORKER_BINARY" ] || WORKER_RELEASE_BASE=$(echo "$CHANNEL" | json_field workerBase) [ -n "${GPUK_SCRIPT}" ] || GPUK_SCRIPT_URL=$(echo "$CHANNEL" | json_field gpukScript) fi if [ "$DRY_RUN" -eq 1 ]; then step "Dry run — stopping here" if [ "$SKIP_PREFLIGHT" -eq 1 ]; then ok "the release resolved; the host was not checked; nothing was installed" else ok "preflight passed and the release resolved; nothing was installed" fi echo echo " would install : $IMAGE" echo " data root : $DATA_ROOT" echo " model cache : $CACHE_DIR" echo " UI port : $PORT" exit 0 fi # ── 3. The host daemon ─────────────────────────────────────────────────────── step "Host daemon — gpu-kitchen-worker" TMP=$(mktemp -d) # shellcheck disable=SC2064 # expand TMP now: it must be removed even if it changes trap "rm -rf '$TMP'" EXIT INT TERM if [ -z "$GPUK_SCRIPT" ]; then # A checkout right here beats a download (that is how contributors run it). _local="$(dirname "$0")/gpuk" if [ -f "$_local" ]; then GPUK_SCRIPT="$_local" ok "using the gpuk installer from this checkout" else [ -n "${GPUK_SCRIPT_URL:-}" ] \ || die "the release channel names no gpuk installer, and none was found locally" curl -fsSL --max-time 60 "$GPUK_SCRIPT_URL" -o "$TMP/gpuk" \ || die "cannot download the gpuk installer from $GPUK_SCRIPT_URL" chmod +x "$TMP/gpuk" GPUK_SCRIPT="$TMP/gpuk" ok "downloaded the gpuk installer" fi fi # `gpuk install` does the rest: it drops the binary, writes the systemd unit, # writes the manifest (the declarative description of the app container) and # applies it. Everything below is passed straight through to it. set -- install \ --mode controller \ --image "$IMAGE" \ --cluster "$CLUSTER" \ --data-root "$DATA_ROOT" \ --cache-dir "$CACHE_DIR" \ --http-port "$PORT" if [ -n "$WORKER_BINARY" ]; then [ -f "$WORKER_BINARY" ] || die "no such worker binary: $WORKER_BINARY" set -- "$@" --binary "$WORKER_BINARY" ok "using a locally-built gpu-kitchen-worker" elif [ -n "${WORKER_RELEASE_BASE:-}" ]; then GPUK_RELEASE_BASE="$WORKER_RELEASE_BASE" export GPUK_RELEASE_BASE ok "gpu-kitchen-worker will be downloaded from the release" fi # else: gpuk reuses an already-installed binary, or fails with its own message. step "Installing — this pulls the image, so it can take a few minutes" sh "$GPUK_SCRIPT" "$@" || die "the install failed — see: journalctl -u gpu-kitchen-worker" # ── 4. Wait for the app, then say where it is ──────────────────────────────── step "Waiting for the controller to answer" i=0 until curl -fsS -o /dev/null --max-time 3 "http://127.0.0.1:$PORT/health" 2>/dev/null; do i=$((i + 1)) [ "$i" -lt 90 ] || die "the controller did not answer on :$PORT within 90s — see: gpuk logs" sleep 2 done ok "the controller is up" # The URL must name the REAL host: the person installing this is very often not # sitting at the machine, and "localhost" would be a lie on every box but theirs # (same reason the backend resolves its own hostname — core/host-name.ts). HOSTNAME_FQDN=$(hostname -f 2>/dev/null || hostname 2>/dev/null || echo localhost) LAN_IP=$(ip route get 1.1.1.1 2>/dev/null | sed -n 's/.* src \([0-9.]*\).*/\1/p' | head -1) PW_FILE="$DATA_ROOT/secrets/bootstrap_admin_password" ADMIN_PW="" [ -f "$PW_FILE" ] && ADMIN_PW=$(cat "$PW_FILE" 2>/dev/null || true) echo echo "${BOLD}${GREEN}GPU Kitchen is running.${RESET}" echo echo " ${BOLD}Open:${RESET} http://${HOSTNAME_FQDN}:${PORT}" [ -n "$LAN_IP" ] && echo " http://${LAN_IP}:${PORT}" echo if [ -n "$ADMIN_PW" ]; then echo " ${BOLD}Sign in:${RESET} admin@local" echo " ${BOLD}Password:${RESET} ${ADMIN_PW}" echo " (the first-run wizard asks you to change it)" echo fi echo " Inference endpoint : http://${HOSTNAME_FQDN}:8200/v1" echo " Version : ${VERSION}" echo # Pre-existing model caches (B81). `gpuk install` prints the same hint, but that # scrolls past mid-install; this banner is where people actually look. Detection # only — the install stays non-interactive and nothing outside $CACHE_DIR is # touched, read as configuration, or modified. EXISTING_CACHE="" CHOSEN_CACHE=$(readlink -f "$CACHE_DIR" 2>/dev/null || echo "$CACHE_DIR") for d in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do [ -n "$d" ] && [ -d "$d/hub" ] || continue real=$(readlink -f "$d" 2>/dev/null || echo "$d") [ "$real" != "$CHOSEN_CACHE" ] || continue ls -d "$d"/hub/models--* >/dev/null 2>&1 || continue EXISTING_CACHE="$EXISTING_CACHE $real" done if [ -n "$EXISTING_CACHE" ]; then for d in $EXISTING_CACHE; do echo " ${BOLD}Existing model cache:${RESET} $d" done echo " Reference it from Settings -> Cache folders to reuse those models" echo " without downloading again. GPU Kitchen only READS a referenced cache." echo fi echo " Update : re-run this command, or press Update in the UI, or: gpuk update" echo " Status : gpuk status Logs: gpuk logs Remove: gpuk uninstall" echo