Files
channel/install.sh
T
2026-08-03 09:02:45 +00:00

369 lines
16 KiB
Bash

#!/bin/sh
# GPU Kitchen — one-command install (specs/plateforme/installation.md).
#
# curl -fsSL https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/install.sh | sudo sh
#
# (Interim URL — becomes https://gpu.kitchen/install.sh once the hub exists; the
# channel repo is written by .gitea/workflows/release.yml, see specs/developpement/ci-cd.md.)
#
# What it does, and nothing more:
# 1. preflight docker, the NVIDIA driver, and a REAL `--gpus all` smoke test
# 2. resolve the current release from the channel (a TAG — never a floating
# `latest`: an install that silently changes version under you is
# not an install, it is a surprise)
# 3. install `gpu-kitchen-worker` (workerd), the privileged host daemon, via
# the existing `gpuk` installer — on a controller node it OWNS the
# app container's lifecycle
# 4. hand over workerd pulls the pinned all-in-one controller image and starts it
# 5. print the UI URL on the real host, and the first-run password
#
# Re-running is how you UPDATE: same command, newer tag, `docker pull` + recreate,
# data untouched (it lives in the data root, not the container).
#
# This installs a CONTROLLER node (the full app + UI). A headless compute node is
# `gpuk install --mode worker …` — see deployments/install/gpuk and
# specs/plateforme/installation.md.
#
# Design note — this script starts nothing itself. workerd owns the container's
# lifecycle (create, health-gate, roll back, update); the UI's update button and
# `gpuk update` drive that same daemon. One updater, three front doors.
set -eu
# ── Defaults (every one overridable by flag or env) ───────────────────────────
# The channel is the single source of truth for "what is the current release":
# it is served next to this script, and the backend's update check reads the SAME
# document (apps/controller/api/src/core/release-channel.ts). One file, one answer. Interim
# default: the public Gitea channel repo — flips to https://gpu.kitchen/latest.json
# once the hub exists (keep the three defaults in sync, see specs/developpement/ci-cd.md).
CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/latest.json}"
EDITION="${GPUK_EDITION:-community}"
DATA_ROOT="${GPUK_DATA_ROOT:-/var/lib/gpu-kitchen}"
CACHE_DIR="${GPUK_CACHE_DIR:-}"
PORT="${GPUK_PORT:-8080}"
CLUSTER="${GPUK_CLUSTER:-default}"
VERSION=""
IMAGE=""
WORKER_BINARY=""
GPUK_SCRIPT=""
SKIP_GPU_CHECK=0
SKIP_PREFLIGHT=0
DRY_RUN=0
GREEN=''; RED=''; YELLOW=''; BOLD=''; RESET=''
if [ -t 1 ]; then
GREEN=$(printf '\033[32m'); RED=$(printf '\033[31m')
YELLOW=$(printf '\033[33m'); BOLD=$(printf '\033[1m'); RESET=$(printf '\033[0m')
fi
ok() { echo " ${GREEN}✓${RESET} $*"; }
warn() { echo " ${YELLOW}!${RESET} $*"; }
step() { echo; echo "${BOLD}$*${RESET}"; }
die() { echo " ${RED}✗${RESET} $*" >&2; exit 1; }
usage() {
cat <<EOF
GPU Kitchen installer
curl -fsSL https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/install.sh | sudo sh
curl -fsSL https://repo.byterain.io/gpukitchen-public/channel/raw/branch/main/install.sh | sudo sh -s -- [options]
Options:
--version <tag> Install this release instead of the channel's current one
--image <ref> Use this controller image outright (implies --version none)
--edition <ed> community (default) | enterprise
--port <p> Port the UI listens on (default 8080)
--data-root <path> Where the database and secrets live (default /var/lib/gpu-kitchen)
--cache-dir <path> Model cache (default <data-root>/hf)
--cluster <name> Cluster name workers join (default "default")
--worker-binary <p> Use a locally-built gpu-kitchen-worker instead of downloading one
--gpuk-script <p> Use a local copy of the gpuk installer
--skip-gpu-check Skip the 'docker run --gpus all' smoke test
--skip-preflight Skip the host checks entirely (CI: no docker, no GPU)
--dry-run Run the preflight and resolve the release, change nothing
-h, --help This
EOF
}
while [ $# -gt 0 ]; do
case "$1" in
--version) VERSION="$2"; shift 2 ;;
--image) IMAGE="$2"; shift 2 ;;
--edition) EDITION="$2"; shift 2 ;;
--port) PORT="$2"; shift 2 ;;
--data-root) DATA_ROOT="$2"; shift 2 ;;
--cache-dir) CACHE_DIR="$2"; shift 2 ;;
--cluster) CLUSTER="$2"; shift 2 ;;
--worker-binary) WORKER_BINARY="$2"; shift 2 ;;
--gpuk-script) GPUK_SCRIPT="$2"; shift 2 ;;
--skip-gpu-check) SKIP_GPU_CHECK=1; shift ;;
--skip-preflight) SKIP_PREFLIGHT=1; shift ;;
--dry-run) DRY_RUN=1; shift ;;
-h|--help) usage; exit 0 ;;
*) die "unknown option: $1 (try --help)" ;;
esac
done
[ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf"
case "$EDITION" in
community|enterprise) ;;
*) die "--edition must be community or enterprise (got '$EDITION')" ;;
esac
echo
echo "${BOLD}GPU Kitchen — installing the $EDITION controller${RESET}"
# ── 1. Preflight ─────────────────────────────────────────────────────────────
# The same checks tools/provision-feeder.sh makes, minus the compose ones: the
# all-in-one image is driven by workerd through the plain docker CLI, so there is
# no compose dependency to satisfy any more.
if [ "$SKIP_PREFLIGHT" -eq 1 ]; then
step "Preflight — skipped (--skip-preflight)"
warn "the host is NOT being checked for docker, a driver or a GPU"
else
step "Preflight — docker, NVIDIA driver, container toolkit"
[ "$DRY_RUN" -eq 1 ] || [ "$(id -u)" -eq 0 ] \
|| die "run as root: curl -fsSL … | sudo sh"
command -v curl >/dev/null 2>&1 || die "curl not found — install it first"
if command -v docker >/dev/null 2>&1; then
if docker info >/dev/null 2>&1; then
ok "docker daemon reachable ($(docker --version | cut -d' ' -f3 | tr -d ,))"
else
die "docker is installed but its daemon is unreachable (is it running? are you root?)"
fi
else
die "docker not found — install Docker Engine first: https://docs.docker.com/engine/install/"
fi
if command -v nvidia-smi >/dev/null 2>&1; then
DRIVER=$(nvidia-smi --query-gpu=driver_version --format=csv,noheader 2>/dev/null | head -1 || true)
GPU_COUNT=$(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | grep -c . || true)
if [ -n "$DRIVER" ] && [ "${GPU_COUNT:-0}" -gt 0 ]; then
ok "NVIDIA driver $DRIVER — ${GPU_COUNT} GPU(s): $(nvidia-smi --query-gpu=name --format=csv,noheader | sort -u | paste -sd', ')"
else
die "nvidia-smi is present but reports no GPU"
fi
else
die "nvidia-smi not found — install the NVIDIA driver first"
fi
if [ "$SKIP_GPU_CHECK" -eq 1 ]; then
warn "nvidia-container-toolkit check skipped (--skip-gpu-check)"
elif docker info 2>/dev/null | grep -qiE 'runtimes:.*nvidia'; then
# The runtime being REGISTERED is not the same as it working. With the toolkit
# installed, --gpus injects the driver and nvidia-smi into a plain image; that
# is the exact mechanism the controller container relies on, so test it rather
# than infer it.
if docker run --rm --gpus all ubuntu:24.04 nvidia-smi -L >/dev/null 2>&1; then
ok "nvidia-container-toolkit works (a container can see the GPUs)"
else
die "the nvidia runtime is registered but 'docker run --gpus all' cannot see the GPUs — reinstall nvidia-container-toolkit"
fi
else
die "nvidia-container-toolkit is not registered with docker — install it, then restart dockerd:
https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html"
fi
# Model weights are large and the failure mode (a download dying at 90%) is
# miserable, so say so up front. A warning, not a refusal: it is the user's disk.
CACHE_PARENT="$CACHE_DIR"
while [ ! -d "$CACHE_PARENT" ] && [ "$CACHE_PARENT" != "/" ]; do
CACHE_PARENT=$(dirname "$CACHE_PARENT")
done
FREE_GB=$(df -BG --output=avail "$CACHE_PARENT" 2>/dev/null | tail -1 | tr -dc '0-9' || true)
if [ "${FREE_GB:-0}" -ge 100 ]; then
ok "model cache $CACHE_DIR — ${FREE_GB}G free"
else
warn "only ${FREE_GB:-?}G free under $CACHE_PARENT — model weights want 100G+"
fi
fi # end preflight
# ── 2. Resolve the release ───────────────────────────────────────────────────
step "Release — resolving the version to install"
# One tiny JSON document, fetched over TLS, holding what the current release IS.
# Parsed with sed rather than jq: `curl … | sudo sh` cannot assume jq exists, and
# the document is ours and flat.
json_field() { sed -n "s/.*\"$1\"[[:space:]]*:[[:space:]]*\"\([^\"]*\)\".*/\1/p" | head -1; }
# Strip a tag off an image reference WITHOUT mangling a registry's host:port.
# `${ref%%:*}` cuts at the FIRST colon and is WRONG: a
# `registry.internal:5000/gpuk/controller:v1.2.3` (or an untagged
# `registry.internal:5000/gpuk/controller`) would collapse to `registry.internal`.
# A colon is a tag separator only when the last colon comes AFTER the last slash.
# Same rule as deployments/install/gpuk's image_repo() and core/release-channel.ts (unit-tested).
image_repo() {
case "$1" in
*@*) printf '%s' "${1%@*}"; return ;; # digest pin → repo@sha256:…
esac
_t="${1##*:}"
case "$_t" in
"$1") printf '%s' "$1" ;; # no colon at all → already untagged
*/*) printf '%s' "$1" ;; # last colon is inside a path → host:port, untagged
*) printf '%s' "${1%:*}" ;;
esac
}
if [ -n "$IMAGE" ]; then
ok "using the image given on the command line: $IMAGE"
[ -n "$VERSION" ] || VERSION="(pinned by --image)"
else
CHANNEL=$(curl -fsSL --max-time 20 "$CHANNEL_URL" 2>/dev/null) \
|| die "cannot reach the release channel at $CHANNEL_URL
(offline? pass --image <ref> to install a specific image directly)"
[ -n "$VERSION" ] || VERSION=$(echo "$CHANNEL" | json_field version)
[ -n "$VERSION" ] || die "the release channel returned no version: $CHANNEL_URL"
if [ "$EDITION" = "enterprise" ]; then
IMAGE_TEMPLATE=$(echo "$CHANNEL" | json_field controllerImageEnterprise)
else
IMAGE_TEMPLATE=$(echo "$CHANNEL" | json_field controllerImage)
fi
[ -n "$IMAGE_TEMPLATE" ] \
|| die "the release channel names no $EDITION controller image: $CHANNEL_URL"
# The channel gives the repository; WE pin the tag. A floating `:latest` would
# make every container recreate a silent, unrequested upgrade. Strip any tag the
# channel already carries with image_repo (host:port-safe), then pin OUR version.
IMAGE="$(image_repo "$IMAGE_TEMPLATE"):${VERSION}"
ok "release $VERSION"
ok "image $IMAGE"
[ -n "$WORKER_BINARY" ] || WORKER_RELEASE_BASE=$(echo "$CHANNEL" | json_field workerBase)
[ -n "${GPUK_SCRIPT}" ] || GPUK_SCRIPT_URL=$(echo "$CHANNEL" | json_field gpukScript)
fi
if [ "$DRY_RUN" -eq 1 ]; then
step "Dry run — stopping here"
if [ "$SKIP_PREFLIGHT" -eq 1 ]; then
ok "the release resolved; the host was not checked; nothing was installed"
else
ok "preflight passed and the release resolved; nothing was installed"
fi
echo
echo " would install : $IMAGE"
echo " data root : $DATA_ROOT"
echo " model cache : $CACHE_DIR"
echo " UI port : $PORT"
exit 0
fi
# ── 3. The host daemon ───────────────────────────────────────────────────────
step "Host daemon — gpu-kitchen-worker"
TMP=$(mktemp -d)
# shellcheck disable=SC2064 # expand TMP now: it must be removed even if it changes
trap "rm -rf '$TMP'" EXIT INT TERM
if [ -z "$GPUK_SCRIPT" ]; then
# A checkout right here beats a download (that is how contributors run it).
_local="$(dirname "$0")/gpuk"
if [ -f "$_local" ]; then
GPUK_SCRIPT="$_local"
ok "using the gpuk installer from this checkout"
else
[ -n "${GPUK_SCRIPT_URL:-}" ] \
|| die "the release channel names no gpuk installer, and none was found locally"
curl -fsSL --max-time 60 "$GPUK_SCRIPT_URL" -o "$TMP/gpuk" \
|| die "cannot download the gpuk installer from $GPUK_SCRIPT_URL"
chmod +x "$TMP/gpuk"
GPUK_SCRIPT="$TMP/gpuk"
ok "downloaded the gpuk installer"
fi
fi
# `gpuk install` does the rest: it drops the binary, writes the systemd unit,
# writes the manifest (the declarative description of the app container) and
# applies it. Everything below is passed straight through to it.
set -- install \
--mode controller \
--image "$IMAGE" \
--cluster "$CLUSTER" \
--data-root "$DATA_ROOT" \
--cache-dir "$CACHE_DIR" \
--http-port "$PORT"
if [ -n "$WORKER_BINARY" ]; then
[ -f "$WORKER_BINARY" ] || die "no such worker binary: $WORKER_BINARY"
set -- "$@" --binary "$WORKER_BINARY"
ok "using a locally-built gpu-kitchen-worker"
elif [ -n "${WORKER_RELEASE_BASE:-}" ]; then
GPUK_RELEASE_BASE="$WORKER_RELEASE_BASE"
export GPUK_RELEASE_BASE
ok "gpu-kitchen-worker will be downloaded from the release"
fi
# else: gpuk reuses an already-installed binary, or fails with its own message.
step "Installing — this pulls the image, so it can take a few minutes"
sh "$GPUK_SCRIPT" "$@" || die "the install failed — see: journalctl -u gpu-kitchen-worker"
# ── 4. Wait for the app, then say where it is ────────────────────────────────
step "Waiting for the controller to answer"
i=0
until curl -fsS -o /dev/null --max-time 3 "http://127.0.0.1:$PORT/health" 2>/dev/null; do
i=$((i + 1))
[ "$i" -lt 90 ] || die "the controller did not answer on :$PORT within 90s — see: gpuk logs"
sleep 2
done
ok "the controller is up"
# The URL must name the REAL host: the person installing this is very often not
# sitting at the machine, and "localhost" would be a lie on every box but theirs
# (same reason the backend resolves its own hostname — core/host-name.ts).
HOSTNAME_FQDN=$(hostname -f 2>/dev/null || hostname 2>/dev/null || echo localhost)
LAN_IP=$(ip route get 1.1.1.1 2>/dev/null | sed -n 's/.* src \([0-9.]*\).*/\1/p' | head -1)
PW_FILE="$DATA_ROOT/secrets/bootstrap_admin_password"
ADMIN_PW=""
[ -f "$PW_FILE" ] && ADMIN_PW=$(cat "$PW_FILE" 2>/dev/null || true)
echo
echo "${BOLD}${GREEN}GPU Kitchen is running.${RESET}"
echo
echo " ${BOLD}Open:${RESET} http://${HOSTNAME_FQDN}:${PORT}"
[ -n "$LAN_IP" ] && echo " http://${LAN_IP}:${PORT}"
echo
if [ -n "$ADMIN_PW" ]; then
echo " ${BOLD}Sign in:${RESET} admin@local"
echo " ${BOLD}Password:${RESET} ${ADMIN_PW}"
echo " (the first-run wizard asks you to change it)"
echo
fi
echo " Inference endpoint : http://${HOSTNAME_FQDN}:8200/v1"
echo " Version : ${VERSION}"
echo
# Pre-existing model caches (B81). `gpuk install` prints the same hint, but that
# scrolls past mid-install; this banner is where people actually look. Detection
# only — the install stays non-interactive and nothing outside $CACHE_DIR is
# touched, read as configuration, or modified.
EXISTING_CACHE=""
CHOSEN_CACHE=$(readlink -f "$CACHE_DIR" 2>/dev/null || echo "$CACHE_DIR")
for d in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do
[ -n "$d" ] && [ -d "$d/hub" ] || continue
real=$(readlink -f "$d" 2>/dev/null || echo "$d")
[ "$real" != "$CHOSEN_CACHE" ] || continue
ls -d "$d"/hub/models--* >/dev/null 2>&1 || continue
EXISTING_CACHE="$EXISTING_CACHE $real"
done
if [ -n "$EXISTING_CACHE" ]; then
for d in $EXISTING_CACHE; do
echo " ${BOLD}Existing model cache:${RESET} $d"
done
echo " Reference it from Settings -> Cache folders to reuse those models"
echo " without downloading again. GPU Kitchen only READS a referenced cache."
echo
fi
echo " Update : re-run this command, or press Update in the UI, or: gpuk update"
echo " Status : gpuk status Logs: gpuk logs Remove: gpuk uninstall"
echo