Files
channel/install.sh
T

1135 lines
55 KiB
Bash

#!/bin/sh
# GPU Kitchen — one-command install (specs/plateforme/installation.md).
#
# curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh | sudo sh
#
# (Interim URL — becomes https://gpu.kitchen/install.sh once the hub exists; the
# channel repo is written by .gitea/workflows/release.yml, see specs/developpement/ci-cd.md.)
#
# What it does, and nothing more:
# 0. existing detect an install already on this host (manifest, or the traces
# a previous one left) and KEEP its settings — ports, data root,
# profile — unless a flag says otherwise (INS-03)
# 1. preflight docker, the NVIDIA driver, a REAL `--gpus all` smoke test, and
# the listening ports (INS-46 — a taken port is resolved HERE, on
# the terminal, not three minutes later in a health-check timeout)
# 2. resolve the current release from the channel (a TAG — never a floating
# `latest`: an install that silently changes version under you is
# not an install, it is a surprise)
# 3. ask the security profile — and, under `public`, a domain — when a
# terminal is attached (INS-45); no terminal, no questions, and
# the first-run wizard asks instead (PRF-05)
# 4. install `gpu-kitchen-worker` (workerd), the privileged host daemon, via
# the existing `gpuk` installer — on a controller node it OWNS the
# app container's lifecycle
# 5. hand over workerd pulls the pinned all-in-one controller image and starts it
# 6. print the UI URL on the real host, plus the profile-specific next step
#
# Re-running is how you UPDATE: on a host that already runs GPU Kitchen, the same
# command hands over to the installed `gpuk update` — the signed channel checked
# against the key that install pinned, the image pinned by digest, data untouched
# (it lives in the data root, not the container). A flag that changes a setting,
# --reinstall or --reset-manifest reinstall over it instead (INS-03).
#
# This installs a CONTROLLER node (the full app + UI). A headless compute node is
# `gpuk install --mode worker …` — see deployments/install/gpuk and
# specs/plateforme/installation.md.
#
# Design note — this script starts nothing itself. workerd owns the container's
# lifecycle (create, health-gate, roll back, update); the UI's update button and
# `gpuk update` drive that same daemon. One updater, three front doors.
#
# The whole script is the body of main(), called on its very last line: `sh`
# reads a piped script as it arrives, so a download cut short would otherwise
# run as root up to wherever it stopped. Truncated, main is never called.
set -eu
main() {
# ── Defaults (every one overridable by flag or env) ───────────────────────────
# The channel is the single source of truth for "what is the current release":
# it is served next to this script, and the backend's update check reads the SAME
# document (apps/controller/api/src/core/release-channel.ts). One file, one answer. Interim
# default: the public Gitea channel repo — flips to https://gpu.kitchen/latest.json
# once the hub exists (keep the three defaults in sync, see specs/developpement/ci-cd.md).
CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen/channel/raw/branch/main/latest.json}"
EDITION="${GPUK_EDITION:-community}"
DATA_ROOT="${GPUK_DATA_ROOT:-/var/lib/gpu-kitchen}"
CACHE_DIR="${GPUK_CACHE_DIR:-}"
# 1337, not 8080: the single most-squatted port in existence would make the
# conflict preflight fire on half the lab boxes out there (INS-01).
PORT="${GPUK_PORT:-1337}"
# The worker mTLS channel and the gpuk-proxy inference endpoint (defaults mirror
# core/cluster-settings.ts). Movable at install time like the UI port (INS-46):
# 8443 is every second appliance's HTTPS alias and 8200 is HashiCorp Vault's.
MTLS_PORT="${GPUK_MTLS_PORT:-8443}"
INFERENCE_PORT="${GPUK_INFERENCE_PORT:-8200}"
CLUSTER="${GPUK_CLUSTER:-default}"
# The app container's docker network. Empty = gpuk's default (bridge, INS-49);
# an existing install keeps the mode it runs in (INS-03) — flipping the default
# must never move a host-mode install to bridge on its next re-run.
NETWORK="${GPUK_NETWORK:-}"
# Which of those came from the operator (flag or env) — an existing install keeps
# its own value for everything the operator did not ask to change (INS-03).
PORT_GIVEN=0; [ -z "${GPUK_PORT:-}" ] || PORT_GIVEN=1
MTLS_GIVEN=0; [ -z "${GPUK_MTLS_PORT:-}" ] || MTLS_GIVEN=1
INFERENCE_GIVEN=0; [ -z "${GPUK_INFERENCE_PORT:-}" ] || INFERENCE_GIVEN=1
DATA_ROOT_GIVEN=0; [ -z "${GPUK_DATA_ROOT:-}" ] || DATA_ROOT_GIVEN=1
CACHE_GIVEN=0; [ -z "${GPUK_CACHE_DIR:-}" ] || CACHE_GIVEN=1
CLUSTER_GIVEN=0; [ -z "${GPUK_CLUSTER:-}" ] || CLUSTER_GIVEN=1
NETWORK_GIVEN=0; [ -z "${GPUK_NETWORK:-}" ] || NETWORK_GIVEN=1
# Where an existing install keeps its manifest. Same override as gpuk's, and for
# the same reason: it is the only way to exercise the re-run path without root.
ETC_DIR="${GPUK_ETC_DIR:-/etc/gpu-kitchen}"
MANIFEST="$ETC_DIR/manifest.json"
UNIT_DEST="${GPUK_UNIT_DEST:-/etc/systemd/system/gpu-kitchen-worker.service}"
BIN_DEST="${GPUK_BIN_DEST:-/usr/local/bin/gpu-kitchen-worker}"
# The gpuk CLI `gpuk install` leaves on the host, next to the daemon binary: the
# one that carries the release key this install pinned, and the one a re-run
# hands the update to (same override as gpuk's).
GPUK_CLI="${GPUK_CLI_DEST:-$(dirname "$BIN_DEST")/gpuk}"
# The release public key pinned in THIS copy of install.sh (OPS-20). The channel
# copy gets the real key substituted at publish time, exactly like gpuk's; the
# operator override covers a self-hosted channel with its own keypair. The
# script itself arrives over HTTPS (trust on first use); everything it then
# takes from the channel — latest.json, the gpuk installer, the daemon binary,
# the image digest — is verified against this key before it is used.
CHANNEL_PUBKEY="${GPUK_UPDATE_PUBKEY:-RWQ7BKXJqGX2jdKXu1GxeSPVAN3JDRTefpImM/mFRjtFwq4E7mhnQtA7}"
EDITION_GIVEN=0; [ -z "${GPUK_EDITION:-}" ] || EDITION_GIVEN=1
PROFILE=""
RESET_MANIFEST=0
REINSTALL=0
DOMAIN=""
VERSION=""
IMAGE=""
WORKER_BINARY=""
GPUK_SCRIPT=""
SKIP_GPU_CHECK=0
SKIP_PREFLIGHT=0
NON_INTERACTIVE=0
DRY_RUN=0
GREEN=''; RED=''; YELLOW=''; BOLD=''; RESET=''
if [ -t 1 ]; then
GREEN=$(printf '\033[32m'); RED=$(printf '\033[31m')
YELLOW=$(printf '\033[33m'); BOLD=$(printf '\033[1m'); RESET=$(printf '\033[0m')
fi
ok() { echo " ${GREEN}✓${RESET} $*"; }
warn() { echo " ${YELLOW}!${RESET} $*"; }
step() { echo; echo "${BOLD}$*${RESET}"; }
die() { echo " ${RED}✗${RESET} $*" >&2; exit 1; }
# `curl … | sudo sh` leaves stdin holding the script itself, so questions are
# asked and answered on the controlling terminal — /dev/tty — when there is one
# (INS-45). No terminal (CI, provisioning), --non-interactive or --dry-run keep
# every historical flags-only behaviour.
can_prompt() {
[ "$NON_INTERACTIVE" -eq 0 ] || return 1
[ "$DRY_RUN" -eq 0 ] || return 1
(: < /dev/tty) 2>/dev/null
}
ask() { # $1 = prompt → $REPLY (empty on EOF)
printf '%s' "$1" > /dev/tty
IFS= read -r REPLY < /dev/tty || REPLY=""
}
usage() {
cat <<EOF
GPU Kitchen installer
curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh | sudo sh
curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh | sudo sh -s -- [options]
Options:
--version <tag> The release the channel names (another one: --image <ref>@sha256:…)
--image <ref> Use this controller image outright (implies --version none)
--edition <ed> community (default) | enterprise
--port <p> Port the UI listens on (default 1337)
--mtls-port <p> Port workers dial to join this controller (default 8443)
--inference-port <p> Port of the OpenAI-compatible inference endpoint (default 8200)
--data-root <path> Where the database and secrets live (default /var/lib/gpu-kitchen)
--cache-dir <path> Model cache (default <data-root>/hf)
--cluster <name> Cluster name workers join (default "default")
--network <mode> bridge (default) | host. Bridged, only the three ports above
touch the host; host makes every listener of the container a
host-wide claim. An existing install keeps its mode.
--profile <p> homelab | studio | enterprise | public (no flag + a terminal
= the script asks; no flag + no terminal = first-run asks)
--domain <d> Name the UI is reached by through a reverse proxy, any
profile: allowed as a host of the install, proxy trusted for
the client address, and a filled TLS reverse-proxy example
written to <data-root>/caddy/Caddyfile
--reinstall On a host that already runs GPU Kitchen, reinstall over it
(settings kept, merged into its manifest) instead of the
default re-run, which is 'gpuk update'
--reset-manifest Rebuild /etc/gpu-kitchen/manifest.json from this run's settings
when the daemon refuses the existing one (a field an older
release wrote); the old file is archived beside it and what
it carried that the rebuild does not is listed
--non-interactive Never ask anything, even with a terminal attached
--worker-binary <p> Use a locally-built gpu-kitchen-worker instead of downloading one
--gpuk-script <p> Use a local copy of the gpuk installer
--skip-gpu-check Skip the 'docker run --gpus all' smoke test
--skip-preflight Skip the docker, driver, GPU and disk checks (CI: no docker,
no GPU); the listening ports are still checked
--dry-run Run the preflight and resolve the release, change nothing
-h, --help This
EOF
}
while [ $# -gt 0 ]; do
case "$1" in
--version) VERSION="$2"; shift 2 ;;
--image) IMAGE="$2"; shift 2 ;;
--edition) EDITION="$2"; EDITION_GIVEN=1; shift 2 ;;
--port) PORT="$2"; PORT_GIVEN=1; shift 2 ;;
--mtls-port) MTLS_PORT="$2"; MTLS_GIVEN=1; shift 2 ;;
--inference-port) INFERENCE_PORT="$2"; INFERENCE_GIVEN=1; shift 2 ;;
--data-root) DATA_ROOT="$2"; DATA_ROOT_GIVEN=1; shift 2 ;;
--cache-dir) CACHE_DIR="$2"; CACHE_GIVEN=1; shift 2 ;;
--cluster) CLUSTER="$2"; CLUSTER_GIVEN=1; shift 2 ;;
--network) NETWORK="$2"; NETWORK_GIVEN=1; shift 2 ;;
--profile) PROFILE="$2"; shift 2 ;;
--domain) DOMAIN="$2"; shift 2 ;;
--non-interactive) NON_INTERACTIVE=1; shift ;;
--reset-manifest) RESET_MANIFEST=1; shift ;;
--reinstall) REINSTALL=1; shift ;;
--worker-binary) WORKER_BINARY="$2"; shift 2 ;;
--gpuk-script) GPUK_SCRIPT="$2"; shift 2 ;;
--skip-gpu-check) SKIP_GPU_CHECK=1; shift ;;
--skip-preflight) SKIP_PREFLIGHT=1; shift ;;
--dry-run) DRY_RUN=1; shift ;;
-h|--help) usage; exit 0 ;;
*) die "unknown option: $1 (try --help)" ;;
esac
done
case "$EDITION" in
community|enterprise) ;;
*) die "--edition must be community or enterprise (got '$EDITION')" ;;
esac
case "$PROFILE" in
""|homelab|studio|enterprise|public) ;;
*) die "--profile must be homelab, studio, enterprise or public (got '$PROFILE')" ;;
esac
case "$DOMAIN" in
*[!A-Za-z0-9.-]*) die "--domain must be a bare domain name (got '$DOMAIN')" ;;
esac
# bridge, host, or the name of a docker network gpuk hands to `docker create
# --network` — never a value that could be read as another flag or as whitespace.
case "$NETWORK" in
""|bridge|host) ;;
-*|*[!A-Za-z0-9_.-]*) die "--network must be bridge, host or a docker network name (got '$NETWORK')" ;;
esac
for _pv in "$PORT" "$MTLS_PORT" "$INFERENCE_PORT"; do
case "$_pv" in
''|*[!0-9]*) die "not a port number: '$_pv'" ;;
esac
[ "$_pv" -ge 1 ] && [ "$_pv" -le 65535 ] || die "port out of range: $_pv"
done
echo
echo "${BOLD}GPU Kitchen — installing the $EDITION controller${RESET}"
# ── 0. An existing install (INS-03) ──────────────────────────────────────────
# Re-running this script is the update path, so before checking anything it
# reads what is already here — and KEEPS it. An update that silently moved the
# UI to another port, re-asked the profile (Enter = homelab would downgrade a
# public install) or pointed at a fresh data root beside the real one is not an
# update. Read-only. The manifest is the authority (WRK-55); without it, the
# traces a previous install leaves (container, unit, binary, data root) still
# mean "take over", never "start beside", and the container's own ports are ours.
manifest_str() { # $1 = key of a string field, anywhere in the manifest
sed -n "s/.*\"$1\"[[:space:]]*:[[:space:]]*\"\([^\"]*\)\".*/\1/p" "$MANIFEST" 2>/dev/null | head -1
}
manifest_binding() { # $1 = container port → the host port the ports table publishes it on
sed -n "s/.*\"$1\(\/tcp\)\{0,1\}\"[[:space:]]*:[[:space:]]*\([0-9][0-9]*\).*/\2/p" "$MANIFEST" 2>/dev/null | head -1
}
C_STATUS=""; C_NETMODE=""
E_PORT=""; E_MTLS_PORT=""; E_LISTEN_ADDR=""
E_PUBLIC_PORT=""; E_PUBLIC_MTLS_PORT=""; E_PROXY_PUBLIC_PORT=""
B_8080=""; B_8443=""; B_8200=""
container_facts() { # $1 = name → C_*, E_*, B_* from docker; 1 when absent
command -v docker >/dev/null 2>&1 || return 1
_f=$(docker inspect -f '{{.State.Status}}|{{.HostConfig.NetworkMode}}|{{range $p, $b := .HostConfig.PortBindings}}{{range $b}}{{$p}}={{.HostPort}} {{end}}{{end}}{{"\n"}}{{range .Config.Env}}{{.}}{{"\n"}}{{end}}' "$1" 2>/dev/null) \
|| return 1
[ -n "$_f" ] || return 1
_head=$(printf '%s\n' "$_f" | head -1)
C_STATUS=${_head%%|*}; _r=${_head#*|}; C_NETMODE=${_r%%|*}; _bind=${_r#*|}
E_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PORT=//p' | head -1)
E_MTLS_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_MTLS_PORT=//p' | head -1)
E_LISTEN_ADDR=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_LISTEN_ADDR=//p' | head -1)
E_PUBLIC_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PUBLIC_PORT=//p' | head -1)
E_PUBLIC_MTLS_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PUBLIC_MTLS_PORT=//p' | head -1)
E_PROXY_PUBLIC_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PROXY_PUBLIC_PORT=//p' | head -1)
B_8080=$(printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n 's/^8080\/tcp=//p' | head -1)
B_8443=$(printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n 's/^8443\/tcp=//p' | head -1)
B_8200=$(printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n 's/^8200\/tcp=//p' | head -1)
}
# The ports an install is REACHED on, with the precedence the controller itself
# applies (core/published-ports.ts, INS-43): host networking moves the listeners
# (GPUK_PORT, GPUK_MTLS_PORT, GPUK_LISTEN_ADDR); anything else keeps the image's
# fixed listeners and publishes them (GPUK_PUBLIC_*, then the ports table).
published_ports() { # $1 = network mode → OURS_UI OURS_MTLS OURS_INF
if [ "$1" = "host" ]; then
OURS_UI="${E_PORT:-8080}"
OURS_MTLS="${E_MTLS_PORT:-8443}"
OURS_INF="${E_PROXY_PUBLIC_PORT:-${E_LISTEN_ADDR##*:}}"
else
OURS_UI="${E_PUBLIC_PORT:-${B_8080:-8080}}"
OURS_MTLS="${E_PUBLIC_MTLS_PORT:-${B_8443:-8443}}"
OURS_INF="${E_PROXY_PUBLIC_PORT:-${B_8200:-8200}}"
fi
[ -n "$OURS_INF" ] || OURS_INF=8200
}
EXISTING=""; OUR_PORTS=""; CONTAINER_NAME="gpu-kitchen"; RECONFIGURE=""
reconfigure() { RECONFIGURE="${RECONFIGURE:+$RECONFIGURE, }$1"; }
OURS_UI=""; OURS_MTLS=""; OURS_INF=""
if [ -f "$MANIFEST" ] && [ ! -r "$MANIFEST" ]; then
EXISTING="unreadable"
elif [ -f "$MANIFEST" ]; then
EXISTING="manifest"
_cn=$(manifest_str containerName); [ -z "$_cn" ] || CONTAINER_NAME="$_cn"
C_NETMODE=$(manifest_str networkMode)
E_PORT=$(manifest_str GPUK_PORT); E_MTLS_PORT=$(manifest_str GPUK_MTLS_PORT)
E_LISTEN_ADDR=$(manifest_str GPUK_LISTEN_ADDR)
E_PUBLIC_PORT=$(manifest_str GPUK_PUBLIC_PORT)
E_PUBLIC_MTLS_PORT=$(manifest_str GPUK_PUBLIC_MTLS_PORT)
E_PROXY_PUBLIC_PORT=$(manifest_str GPUK_PROXY_PUBLIC_PORT)
B_8080=$(manifest_binding 8080); B_8443=$(manifest_binding 8443); B_8200=$(manifest_binding 8200)
published_ports "${C_NETMODE:-host}"
OUR_PORTS="$OURS_UI $OURS_MTLS $OURS_INF"
elif container_facts "$CONTAINER_NAME"; then
EXISTING="leftovers"
published_ports "$C_NETMODE"
OUR_PORTS="$OURS_UI $OURS_MTLS $OURS_INF"
elif [ -f "$UNIT_DEST" ] || [ -x "$BIN_DEST" ] || [ -d "$DATA_ROOT/secrets" ]; then
EXISTING="leftovers"
fi
case "$EXISTING" in
manifest)
step "Existing install — $MANIFEST"
ok "image $(manifest_str image)"
_root=$(manifest_str dataRoot); _cache=$(manifest_str hostPath)
_cluster=$(manifest_str GPUK_CLUSTER); _profile=$(manifest_str GPUK_INSTALL_PROFILE)
# The edition an install runs is its image's basename (gpukitchen-controller-ee).
_edition=$(manifest_str image); _edition=${_edition%@*}; _edition=${_edition##*/}
case "${_edition%%:*}" in *-ee) _edition=enterprise ;; *) _edition=community ;; esac
# What this run asks to CHANGE, before anything is inherited: a flag restating
# the installed value changes nothing, so the same command line re-run later
# is still an update. Anything else is an explicit reinstall.
[ "$PORT_GIVEN" -eq 0 ] || [ "$PORT" = "$OURS_UI" ] || reconfigure "--port"
[ "$MTLS_GIVEN" -eq 0 ] || [ "$MTLS_PORT" = "$OURS_MTLS" ] || reconfigure "--mtls-port"
[ "$INFERENCE_GIVEN" -eq 0 ] || [ "$INFERENCE_PORT" = "$OURS_INF" ] || reconfigure "--inference-port"
[ "$DATA_ROOT_GIVEN" -eq 0 ] || [ "$DATA_ROOT" = "$_root" ] || reconfigure "--data-root"
[ "$CACHE_GIVEN" -eq 0 ] || [ "$CACHE_DIR" = "$_cache" ] || reconfigure "--cache-dir"
[ "$CLUSTER_GIVEN" -eq 0 ] || [ "$CLUSTER" = "$_cluster" ] || reconfigure "--cluster"
[ "$NETWORK_GIVEN" -eq 0 ] || [ "$NETWORK" = "${C_NETMODE:-host}" ] || reconfigure "--network"
[ "$EDITION_GIVEN" -eq 0 ] || [ "$EDITION" = "$_edition" ] || reconfigure "--edition"
[ -z "$PROFILE" ] || [ "$PROFILE" = "$_profile" ] || reconfigure "--profile"
[ -z "$DOMAIN" ] || reconfigure "--domain"
[ -z "$IMAGE" ] || reconfigure "--image"
[ -z "$VERSION" ] || reconfigure "--version"
[ -z "$WORKER_BINARY" ] || reconfigure "--worker-binary"
[ -z "$GPUK_SCRIPT" ] || reconfigure "--gpuk-script"
[ "$REINSTALL" -eq 0 ] || reconfigure "--reinstall"
[ "$RESET_MANIFEST" -eq 0 ] || reconfigure "--reset-manifest"
[ "$DATA_ROOT_GIVEN" -eq 1 ] || [ -z "$_root" ] || DATA_ROOT="$_root"
[ "$CACHE_GIVEN" -eq 1 ] || [ -z "$_cache" ] || CACHE_DIR="$_cache"
[ "$CLUSTER_GIVEN" -eq 1 ] || [ -z "$_cluster" ] || CLUSTER="$_cluster"
[ "$PORT_GIVEN" -eq 1 ] || PORT="$OURS_UI"
[ "$MTLS_GIVEN" -eq 1 ] || MTLS_PORT="$OURS_MTLS"
[ "$INFERENCE_GIVEN" -eq 1 ] || INFERENCE_PORT="$OURS_INF"
# The network mode is a setting like the ports: an install that runs on the
# host network stays there when the default is bridge (INS-49), and the
# reverse — only --network moves it. An old manifest without the field is
# read as the daemon reads it (host).
[ "$NETWORK_GIVEN" -eq 1 ] || NETWORK="${C_NETMODE:-host}"
case "$_profile" in
homelab|studio|enterprise|public) [ -n "$PROFILE" ] || PROFILE="$_profile" ;;
esac
ok "data root $DATA_ROOT"
ok "ports UI $OURS_UI, worker channel $OURS_MTLS, inference $OURS_INF"
ok "network ${C_NETMODE:-host}"
[ -z "$_profile" ] || ok "profile $_profile"
if [ -n "$RECONFIGURE" ]; then
ok "reinstalling over it ($RECONFIGURE). Its other settings are kept."
else
ok "re-running updates it in place, through the installed gpuk. Its settings are kept."
fi
;;
unreadable)
step "Existing install — $MANIFEST"
warn "present, but not readable from here: run as root to keep its settings"
;;
leftovers)
step "Existing install — traces of a previous install, no manifest"
if [ -n "$C_STATUS" ]; then
ok "container $CONTAINER_NAME ($C_STATUS; UI $OURS_UI, worker channel $OURS_MTLS, inference $OURS_INF) — the install replaces it"
fi
[ ! -f "$UNIT_DEST" ] || ok "systemd unit $UNIT_DEST — rewritten"
[ ! -x "$BIN_DEST" ] || ok "daemon binary $BIN_DEST — replaced"
[ ! -d "$DATA_ROOT/secrets" ] || ok "data root $DATA_ROOT — reused, nothing in it is touched"
warn "without $MANIFEST no setting can be kept: the flags and the defaults apply"
;;
esac
[ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf"
# ── 0b. A plain re-run is `gpuk update` (INS-03, OPS-20) ──────────────────────
# The gpuk this install left on the host carries the release key pinned when it
# was installed: it verifies the signed channel against THAT key, pins the image
# by digest, and lets the daemon pull, health-gate and roll back. A freshly
# downloaded install.sh is trust-on-first-use again; handing the update to the
# installed gpuk keeps a re-run inside the chain the first install established.
if [ "$EXISTING" = "manifest" ] && [ -z "$RECONFIGURE" ]; then
step "Update — handed to the installed gpuk"
if [ "$DRY_RUN" -eq 1 ]; then
step "Dry run — stopping here"
echo " would run : $GPUK_CLI update"
echo " existing : yes — updated in place"
exit 0
fi
[ -x "$GPUK_CLI" ] \
|| die "this install has no $GPUK_CLI to update it with. Re-run with --reinstall: it reinstalls
over the existing install, keeps its settings, and leaves gpuk on the host for the next update."
"$GPUK_CLI" update || die "the update failed — gpuk reported the cause just above"
echo
echo "${BOLD}${GREEN}GPU Kitchen is up to date.${RESET}"
echo
echo " Status : gpuk status Logs: gpuk logs"
echo " Change a setting: re-run with its flag (a reinstall), or use Settings in the UI"
echo
exit 0
fi
if [ "$PORT" = "$MTLS_PORT" ] || [ "$PORT" = "$INFERENCE_PORT" ] || [ "$MTLS_PORT" = "$INFERENCE_PORT" ]; then
die "the UI, worker channel and inference ports must differ (got $PORT, $MTLS_PORT, $INFERENCE_PORT)"
fi
# ── 1. Preflight ─────────────────────────────────────────────────────────────
# The same checks tools/provision-feeder.sh makes, minus the compose ones: the
# all-in-one image is driven by workerd through the plain docker CLI, so there is
# no compose dependency to satisfy any more.
if [ "$SKIP_PREFLIGHT" -eq 1 ]; then
step "Preflight — skipped (--skip-preflight)"
warn "the host is NOT being checked for docker, a driver or a GPU"
else
step "Preflight — docker, NVIDIA driver, container toolkit"
# Root, or a user-owned prefix (GPUK_ETC_DIR) — the same rule as gpuk's
# need_root, and the only way the full path is testable without handing root
# to a test suite.
[ "$DRY_RUN" -eq 1 ] || [ "$(id -u)" -eq 0 ] || [ -w "$ETC_DIR" ] \
|| die "run as root: curl -fsSL … | sudo sh"
command -v curl >/dev/null 2>&1 || die "curl not found. Install curl first."
if command -v docker >/dev/null 2>&1; then
if docker info >/dev/null 2>&1; then
ok "docker daemon reachable ($(docker --version | cut -d' ' -f3 | tr -d ,))"
else
die "docker is installed but its daemon does not answer. Start the daemon, or run as root."
fi
else
die "docker not found. Install Docker Engine first: https://docs.docker.com/engine/install/"
fi
if command -v nvidia-smi >/dev/null 2>&1; then
DRIVER=$(nvidia-smi --query-gpu=driver_version --format=csv,noheader 2>/dev/null | head -1 || true)
GPU_COUNT=$(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | grep -c . || true)
if [ -n "$DRIVER" ] && [ "${GPU_COUNT:-0}" -gt 0 ]; then
ok "NVIDIA driver $DRIVER — ${GPU_COUNT} GPU(s): $(nvidia-smi --query-gpu=name --format=csv,noheader | sort -u | paste -sd', ')"
else
die "nvidia-smi is present but reports no GPU"
fi
else
die "nvidia-smi not found. Install the NVIDIA driver first."
fi
if [ "$SKIP_GPU_CHECK" -eq 1 ]; then
warn "nvidia-container-toolkit check skipped (--skip-gpu-check)"
elif docker info 2>/dev/null | grep -qiE 'runtimes:.*nvidia'; then
# The runtime being REGISTERED is not the same as it working. With the toolkit
# installed, --gpus injects the driver and nvidia-smi into a plain image; that
# is the exact mechanism the controller container relies on, so test it rather
# than infer it.
if docker run --rm --gpus all ubuntu:24.04 nvidia-smi -L >/dev/null 2>&1; then
ok "nvidia-container-toolkit works (a container can see the GPUs)"
else
die "the nvidia runtime is registered but 'docker run --gpus all' cannot see the GPUs. Reinstall nvidia-container-toolkit."
fi
else
die "nvidia-container-toolkit is not registered with docker. Install it, then restart dockerd:
https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html"
fi
# Model weights are large and the failure mode (a download dying at 90%) is
# miserable, so say so up front. A warning, not a refusal: it is the user's disk.
CACHE_PARENT="$CACHE_DIR"
while [ ! -d "$CACHE_PARENT" ] && [ "$CACHE_PARENT" != "/" ]; do
CACHE_PARENT=$(dirname "$CACHE_PARENT")
done
FREE_GB=$(df -BG --output=avail "$CACHE_PARENT" 2>/dev/null | tail -1 | tr -dc '0-9' || true)
if [ "${FREE_GB:-0}" -ge 100 ]; then
ok "model cache $CACHE_DIR — ${FREE_GB}G free"
else
warn "only ${FREE_GB:-?}G free under $CACHE_PARENT. Model weights need 100G or more."
fi
fi # end preflight
# ── Listening ports (INS-46) ──────────────────────────────────────────────────
# Deliberately OUTSIDE the preflight branch: --skip-preflight skips docker, the
# driver, the GPU smoke test and the disk (things a runner or a VM cannot have),
# but a taken port is exactly as fatal there, and checking it costs nothing.
# Skipping it here only moved the failure to gpuk's non-interactive refusal.
if [ "$SKIP_PREFLIGHT" -eq 1 ]; then
step "Listening ports — checked even without the preflight"
fi
# A taken port must fail HERE, before anything mutates the host — today's
# alternative is a 3-minute health-check timeout with zero diagnosis. Best-effort
# detection (ss, then netstat); neither present is a warn, never a false red.
# A listener owned by an EXISTING GPU Kitchen install is not a conflict: re-running
# this script is the documented update path, and the manifest names our port.
# Test hook: force the detector. The netstat fallback is unreachable on any
# host that has ss (all of them, in practice), so the CI smoke pins it here to
# keep its parsing honest.
PORT_TOOL="${GPUK_PORT_CHECK_TOOL:-}"
case "$PORT_TOOL" in
""|ss|netstat) ;;
*) die "GPUK_PORT_CHECK_TOOL must be ss or netstat (got '$PORT_TOOL')" ;;
esac
if [ -z "$PORT_TOOL" ]; then
if command -v ss >/dev/null 2>&1; then PORT_TOOL="ss"
elif command -v netstat >/dev/null 2>&1; then PORT_TOOL="netstat"; fi
fi
port_busy() { # $1 = port → 0 iff something listens on TCP :$1
case "$PORT_TOOL" in
ss) [ -n "$(ss -ltnH "sport = :$1" 2>/dev/null)" ] ;;
netstat) netstat -ltn 2>/dev/null \
| awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) found=1} END{exit found?0:1}' ;;
*) return 1 ;;
esac
}
# Which container a listener belongs to, if any: the process's cgroup names the
# container id (host networking — the listener IS the container's process), and a
# bridged publication shows up as the container's port mapping in `docker ps`
# (the host-side holder is docker-proxy, which says nothing by itself). "nginx"
# is a riddle; "nginx in container gpu-kitchen-dev" is the answer.
port_container() { # $1 = port, $2 = pid ("" if unknown) → container name or ""
command -v docker >/dev/null 2>&1 || return 0
if [ -n "$2" ] && [ -r "/proc/$2/cgroup" ]; then
_cid=$(sed -n 's#.*docker[-/]\([0-9a-f]\{64\}\).*#\1#p' "/proc/$2/cgroup" 2>/dev/null | head -1)
if [ -n "$_cid" ]; then
docker inspect -f '{{.Name}}' "$_cid" 2>/dev/null | sed 's|^/||'
return 0
fi
fi
docker ps --format '{{.Names}} {{.Ports}}' 2>/dev/null \
| awk -v p=":$1->" 'index($0, p) { print $1; exit }'
}
port_owner() { # $1 = port → best-effort "process", "process in container NAME", or ""
_proc=""; _pid=""
case "$PORT_TOOL" in
ss)
_line=$(ss -ltnpH "sport = :$1" 2>/dev/null | head -1)
_proc=$(printf '%s' "$_line" | sed -n 's/.*users:((\"\([^"]*\)\".*/\1/p')
_pid=$(printf '%s' "$_line" | sed -n 's/.*pid=\([0-9]*\).*/\1/p')
;;
netstat)
_field=$(netstat -ltnp 2>/dev/null \
| awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) {print $NF; exit}}')
case "$_field" in
*/*) _pid=${_field%%/*}; _proc=${_field#*/} ;;
*) _proc="$_field" ;;
esac
;;
esac
case "$_pid" in *[!0-9]*|"") _pid="" ;; esac
_ctr=$(port_container "$1" "$_pid")
if [ -n "$_ctr" ]; then
printf '%s' "${_proc:-a process} in container $_ctr"
else
printf '%s' "$_proc"
fi
}
# A port is ours when the existing install (step 0) already holds it: the
# re-run replaces that container, so what it listens on is not a conflict.
port_is_ours() { case " $OUR_PORTS " in *" $1 "*) return 0 ;; esac; return 1; }
# Ports this run may not hand out twice: the three requested ones, plus every
# alternative already accepted. Without it, a busy 8442 would be offered 8443
# and collide with the worker channel one question later.
RESERVED_PORTS="$PORT $MTLS_PORT $INFERENCE_PORT"
port_available() { # $1 → free on the host AND not reserved by this run
case " $RESERVED_PORTS " in *" $1 "*) return 1 ;; esac
port_is_ours "$1" && return 1
! port_busy "$1"
}
# resolve_port <label> <port> <flag> → RESOLVED. Same rules for all three
# listeners: ours = fine; busy on a terminal = propose the next free port
# (Enter accepts, a number picks, q aborts) — never auto-pick silently, the URL
# printed at the end and the idempotent re-run both need the operator to KNOW
# the port; busy without a terminal = fail now, naming the process and the flag.
resolve_port() {
_label="$1"; _want="$2"; _flag="$3"
if port_is_ours "$_want"; then
ok "$_label port $_want — already ours (re-running is how you update)"
elif port_busy "$_want"; then
OWNER=$(port_owner "$_want")
OWNER="${OWNER:-an unknown process}"
if can_prompt; then
ALT=$((_want + 1))
while ! port_available "$ALT"; do ALT=$((ALT + 1)); done
ask " ${YELLOW}!${RESET} $_label port $_want is busy ($OWNER). Use $ALT instead? [$ALT], another port, or 'q' to abort: "
case "$REPLY" in
q|Q) die "$_label port $_want is in use by $OWNER. Run the install again with $_flag <p>." ;;
"") _want="$ALT" ;;
*)
case "$REPLY" in
*[!0-9]*) die "not a port number: $REPLY" ;;
esac
port_available "$REPLY" \
|| die "port $REPLY is busy, or already taken by another GPU Kitchen listener. Run the install again with $_flag <p>."
_want="$REPLY"
;;
esac
RESERVED_PORTS="$RESERVED_PORTS $_want"
ok "$_label port $_want is free"
else
die "$_label port $_want is already in use by $OWNER. Pass $_flag <p> to choose another port."
fi
else
ok "$_label port $_want is free"
fi
RESOLVED="$_want"
}
if [ -z "$PORT_TOOL" ]; then
warn "cannot check for port conflicts (neither ss nor netstat found)"
else
resolve_port "UI" "$PORT" "--port"; PORT="$RESOLVED"
resolve_port "worker channel" "$MTLS_PORT" "--mtls-port"; MTLS_PORT="$RESOLVED"
resolve_port "inference" "$INFERENCE_PORT" "--inference-port"; INFERENCE_PORT="$RESOLVED"
fi
# ── 2. Resolve the release ───────────────────────────────────────────────────
step "Release — resolving the version to install"
# Everything fetched from the channel lands here, and is verified here, before use.
TMP=$(mktemp -d)
# shellcheck disable=SC2064 # expand TMP now: it must be removed even if it changes
trap "rm -rf '$TMP'" EXIT INT TERM
# The trust chain (OPS-20): latest.json is signed with the release key, and so is
# every daemon binary; latest.json names the gpuk installer's sha256 and the
# image digests. Fail-closed: no pinned key, no minisign, no or a bad signature
# are all fatal. GPUK_CHANNEL_INSECURE=1 is the explicit, loudly reported
# opt-out for a private mirror that does not sign — gpuk's own opt-out.
INSECURE=0; [ "${GPUK_CHANNEL_INSECURE:-}" != "1" ] || INSECURE=1
# The minisign CLI is what verifies the release; a host without it gets it from
# its own package manager — detected, non-interactive, quiet — before anything
# else changes. No known manager, or an install that fails: stop, naming the
# manual command. GPUK_MINISIGN names another verifier binary (test seam).
MINISIGN="${GPUK_MINISIGN:-minisign}"
minisign_manual_command() {
if command -v apt-get >/dev/null 2>&1; then echo "apt-get install minisign"
elif command -v dnf >/dev/null 2>&1; then echo "dnf install minisign (EPEL on RHEL)"
elif command -v yum >/dev/null 2>&1; then echo "yum install minisign (EPEL)"
elif command -v zypper >/dev/null 2>&1; then echo "zypper install minisign"
elif command -v apk >/dev/null 2>&1; then echo "apk add minisign"
elif command -v pacman >/dev/null 2>&1; then echo "pacman -S minisign"
else echo "install minisign from https://jedisct1.github.io/minisign/"
fi
}
ensure_minisign() { # $1 = 1 when nothing may be installed (dry run)
command -v "$MINISIGN" >/dev/null 2>&1 && return 0
[ "${1:-0}" -eq 0 ] \
|| die "minisign is required to verify the release and a dry run installs nothing: $(minisign_manual_command), or set GPUK_CHANNEL_INSECURE=1"
_mlog=$(mktemp)
for _pm in apt-get dnf yum zypper apk pacman; do
command -v "$_pm" >/dev/null 2>&1 || continue
ok "minisign is missing — installing it with $_pm"
case "$_pm" in
apt-get) { DEBIAN_FRONTEND=noninteractive apt-get update -qq \
&& DEBIAN_FRONTEND=noninteractive apt-get install -y -qq minisign; } ;;
dnf) dnf install -y -q minisign ;;
yum) yum install -y -q minisign ;;
zypper) zypper --non-interactive --quiet install minisign ;;
apk) apk add --quiet minisign ;;
pacman) pacman -S --noconfirm --needed --quiet minisign ;;
esac >"$_mlog" 2>&1 || true
if command -v "$MINISIGN" >/dev/null 2>&1; then rm -f "$_mlog"; return 0; fi
done
tail -5 "$_mlog" >&2 2>/dev/null || true
rm -f "$_mlog"
die "minisign is required to verify the release and could not be installed automatically. Nothing was changed: run '$(minisign_manual_command)' as root, then re-run — or set GPUK_CHANNEL_INSECURE=1"
}
require_verifier() {
case "$CHANNEL_PUBKEY" in
# The unstamped placeholder, matched by its prefix only: the release stamps
# the key by substituting the whole placeholder wherever it appears, and a
# guard spelling it in full would become one refusing the very key it pinned.
""|__GPUK_UPDATE_*)
die "this install.sh carries no pinned release public key: use the channel's copy, set
GPUK_UPDATE_PUBKEY (the minisign public-key line), or GPUK_CHANNEL_INSECURE=1 to skip verification" ;;
esac
ensure_minisign "$DRY_RUN"
}
verify_signature() { # $1 = file, $2 = its .minisig, $3 = what it is
"$MINISIGN" -Vq -m "$1" -x "$2" -P "$CHANNEL_PUBKEY" >/dev/null 2>&1 \
|| die "$3: signature verification FAILED against the pinned release key — refusing it (OPS-20)"
}
# One tiny JSON document, fetched over TLS, holding what the current release IS.
# Parsed with sed rather than jq: `curl … | sudo sh` cannot assume jq exists, and
# the document is ours and flat.
json_field() { sed -n "s/.*\"$1\"[[:space:]]*:[[:space:]]*\"\([^\"]*\)\".*/\1/p" | head -1; }
# Strip a tag off an image reference WITHOUT mangling a registry's host:port.
# `${ref%%:*}` cuts at the FIRST colon and is WRONG: a
# `registry.internal:5000/gpuk/controller:v1.2.3` (or an untagged
# `registry.internal:5000/gpuk/controller`) would collapse to `registry.internal`.
# A colon is a tag separator only when the last colon comes AFTER the last slash.
# Same rule as deployments/install/gpuk's image_repo() and core/release-channel.ts (unit-tested).
image_repo() {
case "$1" in
*@*) printf '%s' "${1%@*}"; return ;; # digest pin → repo@sha256:…
esac
_t="${1##*:}"
case "$_t" in
"$1") printf '%s' "$1" ;; # no colon at all → already untagged
*/*) printf '%s' "$1" ;; # last colon is inside a path → host:port, untagged
*) printf '%s' "${1%:*}" ;;
esac
}
if [ -n "$IMAGE" ]; then
ok "using the image given on the command line: $IMAGE"
[ -n "$VERSION" ] || VERSION="(pinned by --image)"
else
CHANNEL_DOC="$TMP/latest.json"
curl -fsSL --max-time 20 "$CHANNEL_URL" -o "$CHANNEL_DOC" 2>/dev/null \
|| die "cannot reach the release channel at $CHANNEL_URL
If this host is offline, pass --image <ref> to install a specific image directly."
if [ "$INSECURE" -eq 1 ]; then
warn "GPUK_CHANNEL_INSECURE=1 — the release channel and what it names are NOT verified"
else
require_verifier
curl -fsSL --max-time 20 "$CHANNEL_URL.minisig" -o "$CHANNEL_DOC.minisig" 2>/dev/null \
|| die "no signature at $CHANNEL_URL.minisig — refusing an unsigned channel document (OPS-20)"
verify_signature "$CHANNEL_DOC" "$CHANNEL_DOC.minisig" "the release channel ($CHANNEL_URL)"
ok "release channel signature verified"
fi
CHANNEL_VERSION=$(json_field version < "$CHANNEL_DOC")
[ -n "$CHANNEL_VERSION" ] || die "the release channel returned no version: $CHANNEL_URL"
if [ "$EDITION" = "enterprise" ]; then
IMAGE_TEMPLATE=$(json_field controllerImageEnterprise < "$CHANNEL_DOC")
IMAGE_DIGEST=$(json_field controllerImageDigestEnterprise < "$CHANNEL_DOC")
else
IMAGE_TEMPLATE=$(json_field controllerImage < "$CHANNEL_DOC")
IMAGE_DIGEST=$(json_field controllerImageDigest < "$CHANNEL_DOC")
fi
[ -n "$IMAGE_TEMPLATE" ] \
|| die "the release channel names no $EDITION controller image: $CHANNEL_URL"
# The signed document vouches for ONE release: its digest belongs to that
# version and no other. Another release is installed by its full reference.
if [ -n "$VERSION" ] && [ "$VERSION" != "$CHANNEL_VERSION" ]; then
[ "$INSECURE" -eq 1 ] \
|| die "the signed channel vouches for $CHANNEL_VERSION only, not $VERSION. To install another
release, pass its full reference: --image <repo>:$VERSION@sha256:<digest>"
IMAGE_DIGEST=""
fi
[ -n "$VERSION" ] || VERSION="$CHANNEL_VERSION"
case "$IMAGE_DIGEST" in
sha256:*) _hex=${IMAGE_DIGEST#sha256:}
case "$_hex" in *[!0-9a-f]*) die "the release channel carries a malformed image digest: $IMAGE_DIGEST" ;; esac
[ "${#_hex}" -eq 64 ] || die "the release channel carries a malformed image digest: $IMAGE_DIGEST" ;;
"") [ "$INSECURE" -eq 1 ] \
|| die "the release channel names no $EDITION image digest: refusing to pin a mutable tag (OPS-20)" ;;
*) die "the release channel carries a malformed image digest: $IMAGE_DIGEST" ;;
esac
# The channel gives the repository; WE pin the tag and, from the signed
# document, the content: `repo:tag@sha256:…` keeps the tag readable while
# docker resolves by digest, so a registry cannot swap what the tag points at.
# A floating `:latest` would make every recreate a silent, unrequested upgrade.
IMAGE="$(image_repo "$IMAGE_TEMPLATE"):${VERSION}${IMAGE_DIGEST:+@$IMAGE_DIGEST}"
ok "release $VERSION"
ok "image $IMAGE"
[ -n "$IMAGE_DIGEST" ] || warn "the image is pinned by tag only (no digest in an unverified channel)"
[ -n "$WORKER_BINARY" ] || WORKER_RELEASE_BASE=$(json_field workerBase < "$CHANNEL_DOC")
if [ -z "${GPUK_SCRIPT}" ]; then
GPUK_SCRIPT_URL=$(json_field gpukScript < "$CHANNEL_DOC")
GPUK_SCRIPT_SHA256=$(json_field gpukScriptSha256 < "$CHANNEL_DOC")
fi
fi
# A direct --image and a channel version are held to the same immutable-image
# rule. A registry host:port is not a tag separator; only the last colon after
# the last slash counts. Digest pins are accepted too.
case "$IMAGE" in
*@sha256:*) ;;
*:latest) die "refusing floating image tag '$IMAGE'. Use an explicit release tag or digest." ;;
*)
IMAGE_TAG="${IMAGE##*:}"
case "$IMAGE_TAG" in
"$IMAGE"|*/*) die "controller image must carry an explicit tag or sha256 digest (got '$IMAGE')" ;;
esac
;;
esac
if [ "$DRY_RUN" -eq 1 ]; then
step "Dry run — stopping here"
if [ "$SKIP_PREFLIGHT" -eq 1 ]; then
ok "the release resolved; the host was not checked; nothing was installed"
else
ok "preflight passed and the release resolved; nothing was installed"
fi
echo
echo " would install : $IMAGE"
echo " data root : $DATA_ROOT"
echo " model cache : $CACHE_DIR"
echo " UI port : $PORT"
echo " worker channel: $MTLS_PORT"
echo " inference : $INFERENCE_PORT"
echo " network : ${NETWORK:-bridge}"
case "$EXISTING" in
manifest) echo " existing : yes — updated in place" ;;
leftovers) echo " existing : traces of a previous install — taken over" ;;
*) echo " existing : no" ;;
esac
if [ -n "$PROFILE" ]; then
echo " profile : $PROFILE"
else
echo " profile : (asked on the terminal, or chosen during first run)"
fi
[ -z "$DOMAIN" ] || echo " domain : $DOMAIN"
exit 0
fi
# ── 3. Resolve the installation profile (INS-45) ─────────────────────────────
# Only ever on a terminal, and only when --profile was not given. The answer is
# relayed verbatim as --profile: the backend stays the sole applier of profile
# defaults and floors. Without a terminal the historical path is untouched —
# profile unset, enterprise floors, the first-run wizard requires the choice
# (PRF-05).
if [ -z "$PROFILE" ] && can_prompt; then
step "Security profile — how will this kitchen be used?"
cat > /dev/tty <<'PROFILES'
1) Home lab a trusted home network
No sign-in on your home network. Nearby workers are found and join
without waiting for approval.
2) Studio one control station, shared compute
Control stays on this machine. Colleagues use API keys, while nearby
workers wait for your approval.
3) Enterprise a managed company network
Sign-in is required. The controller stays quiet on the network; known
workers can request your approval.
4) Public server direct internet exposure
Sign-in and hardened browser transport are required. Network discovery
is off and workers join only by token.
You can change this later in Settings. Stricter floors re-apply. Nothing
already issued is revoked.
PROFILES
while :; do
ask " Choose a profile [1-4, Enter = 1 (Home lab)]: "
case "$REPLY" in
""|1|homelab) PROFILE="homelab" ;;
2|studio) PROFILE="studio" ;;
3|enterprise) PROFILE="enterprise" ;;
4|public) PROFILE="public" ;;
*) printf '%s\n' " pick 1, 2, 3 or 4" > /dev/tty; continue ;;
esac
break
done
ok "profile: $PROFILE"
fi
# Under public, a domain lets gpuk write a FILLED TLS reverse-proxy example
# (INS-47). Optional: Enter skips, and the banner still points at the shipped
# Caddyfile.example.
if [ "$PROFILE" = "public" ] && [ -z "$DOMAIN" ] && can_prompt; then
while :; do
ask " Domain for HTTPS access (e.g. gpu.example.com — Enter to skip): "
# Accept the copy-paste reflex: strip a pasted scheme and anything after
# the first slash, then insist on a bare domain rather than skipping —
# a silently dropped answer would be discovered hours later, at DNS time.
REPLY="${REPLY#https://}"
REPLY="${REPLY#http://}"
REPLY="${REPLY%%/*}"
[ -n "$REPLY" ] || break
case "$REPLY" in
*[!A-Za-z0-9.-]*)
printf '%s\n' " not a bare domain name: $REPLY. Try again, or press Enter to skip." > /dev/tty
;;
*) DOMAIN="$REPLY"; break ;;
esac
done
fi
# ── 4. The host daemon ───────────────────────────────────────────────────────
step "Host daemon — gpu-kitchen-worker"
if [ -z "$GPUK_SCRIPT" ]; then
# A checkout right here beats a download (that is how contributors run it) —
# but ONLY a checkout. Piped into `sh`, $0 is `sh` and dirname "$0" is the
# current directory: a stray `./gpuk` in /tmp or a shared folder would run as
# root, unverified. So the local copy counts only when this script was run by
# its path, next to its gpuk, inside a repository checkout (dev.sh two levels
# up, deployments/install/); everything else downloads (INS-03).
_local=""
case "$0" in
install.sh|*/install.sh)
_here=$(dirname "$0")
[ -f "$_here/gpuk" ] && [ -f "$_here/../../dev.sh" ] && _local="$_here/gpuk"
;;
esac
if [ -n "$_local" ]; then
GPUK_SCRIPT="$_local"
ok "using the gpuk installer from this checkout ($_local)"
else
[ -n "${GPUK_SCRIPT_URL:-}" ] \
|| die "the release channel names no gpuk installer, and none was found locally"
curl -fsSL --max-time 60 "$GPUK_SCRIPT_URL" -o "$TMP/gpuk" \
|| die "cannot download the gpuk installer from $GPUK_SCRIPT_URL"
# The signed channel names the installer's sha256: the bytes that run as root
# next are the ones the release signed for, wherever they were served from.
if [ -n "${GPUK_SCRIPT_SHA256:-}" ]; then
_sum=$(sha256sum "$TMP/gpuk" | cut -d' ' -f1)
[ "$_sum" = "$GPUK_SCRIPT_SHA256" ] \
|| die "the gpuk installer from $GPUK_SCRIPT_URL does not match the signed channel (sha256 $_sum)"
ok "downloaded the gpuk installer (sha256 matches the signed channel)"
else
[ "$INSECURE" -eq 1 ] \
|| die "the release channel names no gpukScriptSha256: refusing an unverified gpuk installer (OPS-20)"
warn "downloaded the gpuk installer, NOT verified (GPUK_CHANNEL_INSECURE=1)"
fi
chmod +x "$TMP/gpuk"
GPUK_SCRIPT="$TMP/gpuk"
fi
fi
# `gpuk install` does the rest: it drops the binary, writes the systemd unit,
# writes the manifest (the declarative description of the app container) and
# applies it. Everything below is passed straight through to it.
set -- install \
--mode controller \
--image "$IMAGE" \
--cluster "$CLUSTER" \
--data-root "$DATA_ROOT" \
--cache-dir "$CACHE_DIR" \
--http-port "$PORT" \
--mtls-port "$MTLS_PORT" \
--inference-port "$INFERENCE_PORT"
[ -z "$PROFILE" ] || set -- "$@" --profile "$PROFILE"
[ -z "$DOMAIN" ] || set -- "$@" --domain "$DOMAIN"
[ "$RESET_MANIFEST" -eq 0 ] || set -- "$@" --reset-manifest
# Inherited from the manifest or given by flag; unset on a first install, so
# gpuk's own default (bridge, INS-49) applies and this script never restates it.
[ -z "$NETWORK" ] || set -- "$@" --network "$NETWORK"
if [ -n "$WORKER_BINARY" ]; then
[ -f "$WORKER_BINARY" ] || die "no such worker binary: $WORKER_BINARY"
set -- "$@" --binary "$WORKER_BINARY"
ok "using a locally-built gpu-kitchen-worker"
elif [ -n "${WORKER_RELEASE_BASE:-}" ]; then
# The privileged daemon: downloaded here, its minisign signature checked
# against the pinned release key, and only then handed to gpuk (OPS-20).
case "$(uname -m)" in
x86_64|amd64) _asset="gpu-kitchen-worker-x86_64"; _build_key="workerBuildIdX86_64" ;;
aarch64|arm64) _asset="gpu-kitchen-worker-aarch64"; _build_key="workerBuildIdAarch64" ;;
*) die "unsupported architecture: $(uname -m)" ;;
esac
ok "downloading $_asset from the release"
curl -fL --progress-bar "$WORKER_RELEASE_BASE/$_asset" -o "$TMP/gpu-kitchen-worker" \
|| die "cannot download $WORKER_RELEASE_BASE/$_asset"
if [ "$INSECURE" -eq 1 ]; then
warn "gpu-kitchen-worker NOT verified (GPUK_CHANNEL_INSECURE=1)"
else
curl -fsSL --max-time 60 "$WORKER_RELEASE_BASE/$_asset.minisig" -o "$TMP/gpu-kitchen-worker.minisig" \
|| die "no signature at $WORKER_RELEASE_BASE/$_asset.minisig — refusing an unsigned daemon binary"
verify_signature "$TMP/gpu-kitchen-worker" "$TMP/gpu-kitchen-worker.minisig" "gpu-kitchen-worker ($_asset)"
# The signature binds artifact AND build (trusted comment
# `gpu-kitchen-worker@<buildId>`, P2-29): the build must be the one the signed
# channel names for this release, so an older signed binary served in its
# place is refused, and never older than the daemon already installed.
_build=$(json_field "$_build_key" < "$CHANNEL_DOC")
[ -n "$_build" ] \
|| die "the release channel names no $_build_key: cannot tell which gpu-kitchen-worker build it vouches for (OPS-20)"
_comment=$("$MINISIGN" -V -m "$TMP/gpu-kitchen-worker" -x "$TMP/gpu-kitchen-worker.minisig" -P "$CHANNEL_PUBKEY" 2>/dev/null \
| sed -n 's/^Trusted comment: //p' | head -1)
[ "$_comment" = "gpu-kitchen-worker@$_build" ] \
|| die "gpu-kitchen-worker ($_asset) is signed for '${_comment:-nothing}', not gpu-kitchen-worker@$_build — refusing it (OPS-20)"
if [ -x "$BIN_DEST" ]; then
_installed=$("$BIN_DEST" --version 2>/dev/null | sed -n 's/^gpu-kitchen-worker //p' | head -1)
case "$_installed" in
[0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9]-*)
# buildId = YYYYMMDDHHMMSS-<commit>: the timestamp orders builds.
[ "${_build%%-*}" -ge "${_installed%%-*}" ] \
|| die "gpu-kitchen-worker $_build is OLDER than the installed $_installed — refusing a downgrade" ;;
esac
fi
ok "gpu-kitchen-worker signature verified (build $_build)"
fi
set -- "$@" --binary "$TMP/gpu-kitchen-worker"
fi
# else: gpuk reuses an already-installed binary, or fails with its own message.
step "Installing — this pulls the image, so it can take a few minutes"
# gpuk names its own failure on stderr before exiting (a refused manifest, a
# failed apply, a missing binary…), so the cause is the line right above this
# one. journalctl only has something to say once the daemon has started, and
# gpuk points at `gpuk logs` itself in that case.
sh "$GPUK_SCRIPT" "$@" || die "the install failed — gpuk reported the cause just above"
# ── 5. Wait for the app, then say where it is ────────────────────────────────
step "Waiting for the controller to answer"
i=0
until curl -fsS -o /dev/null --max-time 3 "http://127.0.0.1:$PORT/health" 2>/dev/null; do
i=$((i + 1))
[ "$i" -lt 90 ] || die "the controller did not answer on :$PORT within 90s. See: gpuk logs"
sleep 2
done
ok "the controller is up"
# The URL must name the REAL host: the person installing this is very often not
# sitting at the machine, and "localhost" would be a lie on every box but theirs
# (same reason the backend resolves its own hostname — core/host-name.ts).
HOSTNAME_FQDN=$(hostname -f 2>/dev/null || hostname 2>/dev/null || echo localhost)
LAN_IP=$(ip route get 1.1.1.1 2>/dev/null | sed -n 's/.* src \([0-9.]*\).*/\1/p' | head -1)
echo
echo "${BOLD}${GREEN}GPU Kitchen is running.${RESET}"
echo
echo " ${BOLD}Open:${RESET} http://${HOSTNAME_FQDN}:${PORT}"
[ -n "$LAN_IP" ] && echo " http://${LAN_IP}:${PORT}"
echo
# [PRF-12] [SEC-53] homelab and studio open straight into the app: the controller
# closed their claim window at first boot, so there is no code to show. Every other
# profile — or none yet — starts with the claim code (SEC-52).
case "$PROFILE" in
homelab|studio)
echo " ${BOLD}Next:${RESET} open the URL above: GPU Kitchen opens straight away."
;;
*)
CLAIM_CODE_FILE="$DATA_ROOT/secrets/claim_code"
CLAIM_CODE=""
[ ! -s "$CLAIM_CODE_FILE" ] || CLAIM_CODE=$(cat "$CLAIM_CODE_FILE" 2>/dev/null || true)
if [ -n "$CLAIM_CODE" ]; then
echo " ${BOLD}Claim code:${RESET} $CLAIM_CODE"
echo " ${BOLD}Read again:${RESET} $CLAIM_CODE_FILE (mode 0600; removed after claim)"
else
echo " ${BOLD}Claim code:${RESET} consumed (the first account already exists)"
fi
echo
echo " ${BOLD}Next:${RESET} open the URL above and enter the claim code."
;;
esac
# [INS-50] Linking to a GPU Kitchen account is never a first-run step: a banner in
# the app offers it, and the person approves on gpu.kitchen — no key is copied here,
# none ever travels in a URL (CPT-05/06).
echo " A banner in the app connects this installation to your GPU Kitchen account:"
echo " \"Connect to my account\" or \"Create an account\" on gpu.kitchen — nothing to copy."
case "$PROFILE" in
public)
echo " ${BOLD}Then:${RESET} create the first administrator in the UI"
echo
# The public profile REQUIRES a TLS reverse proxy (OPS-13, INS-47) — the UI
# port speaks plain HTTP. The product cannot verify the proxy's presence
# (OPS-67), so the closest thing to enforcement is saying it here, clearly.
if [ -n "$DOMAIN" ]; then
echo " ${BOLD}HTTPS:${RESET} https://${DOMAIN} answers after these steps:"
echo " 1. DNS: point ${DOMAIN} at this machine's public IP"
echo " 2. Install Caddy: https://caddyserver.com/docs/install"
echo " 3. sudo cp $DATA_ROOT/caddy/Caddyfile /etc/caddy/Caddyfile"
echo " sudo systemctl reload caddy"
echo " A filled example was written to $DATA_ROOT/caddy/Caddyfile."
echo " Until then the UI answers in cleartext on the URLs above."
else
echo " ${BOLD}HTTPS:${RESET} the public profile requires a TLS reverse proxy before any"
echo " public exposure. See deployments/controller/Caddyfile.example."
fi
echo
;;
enterprise)
echo " ${BOLD}Then:${RESET} create the first administrator in the UI"
echo
;;
homelab|studio)
echo
;;
"")
echo " ${BOLD}Then:${RESET} choose an installation profile in the first-run assistant"
echo
;;
esac
# Any profile may be reached by a name through a reverse proxy (INS-47); the
# public case above already printed its steps.
if [ -n "$DOMAIN" ] && [ "$PROFILE" != "public" ]; then
echo " ${BOLD}HTTPS:${RESET} https://${DOMAIN} once your reverse proxy serves it; a filled"
echo " Caddy example was written to $DATA_ROOT/caddy/Caddyfile."
echo
fi
echo " Inference endpoint : http://${HOSTNAME_FQDN}:${INFERENCE_PORT}/v1"
echo " Version : ${VERSION}"
echo
# Pre-existing model caches (B81). `gpuk install` prints the same hint, but that
# scrolls past mid-install; this banner is where people actually look. Detection
# only — cache questions belong to the first-run wizard (INS-48, REG-41), never
# to the CLI, and nothing outside $CACHE_DIR is touched, read as configuration,
# or modified.
EXISTING_CACHE=""
CHOSEN_CACHE=$(readlink -f "$CACHE_DIR" 2>/dev/null || echo "$CACHE_DIR")
for d in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do
[ -n "$d" ] && [ -d "$d/hub" ] || continue
real=$(readlink -f "$d" 2>/dev/null || echo "$d")
[ "$real" != "$CHOSEN_CACHE" ] || continue
ls -d "$d"/hub/models--* >/dev/null 2>&1 || continue
EXISTING_CACHE="$EXISTING_CACHE $real"
done
if [ -n "$EXISTING_CACHE" ]; then
for d in $EXISTING_CACHE; do
echo " ${BOLD}Existing model cache:${RESET} $d"
done
echo " GPU Kitchen will offer to reuse those models at first launch, and any"
echo " time from Nodes & GPU -> Storage. A reused cache is referenced in"
echo " place. Nothing is moved or deleted."
echo
fi
echo " Update : re-run this command, or press Update in the UI, or: gpuk update"
echo " Status : gpuk status Logs: gpuk logs"
echo " Remove : gpuk uninstall (service only) or gpuk uninstall --purge (all but the data root)"
echo
}
main "$@"