Files
channel/install.sh
T
2026-09-13 21:53:54 +00:00

651 lines
28 KiB
Bash

#!/bin/sh
# GPU Kitchen — one-command install (specs/plateforme/installation.md).
#
# curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh | sudo sh
#
# (Interim URL — becomes https://gpu.kitchen/install.sh once the hub exists; the
# channel repo is written by .gitea/workflows/release.yml, see specs/developpement/ci-cd.md.)
#
# What it does, and nothing more:
# 1. preflight docker, the NVIDIA driver, a REAL `--gpus all` smoke test, and
# the listening ports (INS-46 — a taken port fails HERE, not three
# minutes later in a health-check timeout)
# 2. resolve the current release from the channel (a TAG — never a floating
# `latest`: an install that silently changes version under you is
# not an install, it is a surprise)
# 3. ask the security profile — and, under `public`, a domain — when a
# terminal is attached (INS-45); no terminal, no questions, and
# the first-run wizard asks instead (PRF-05)
# 4. install `gpu-kitchen-worker` (workerd), the privileged host daemon, via
# the existing `gpuk` installer — on a controller node it OWNS the
# app container's lifecycle
# 5. hand over workerd pulls the pinned all-in-one controller image and starts it
# 6. print the UI URL on the real host, plus the profile-specific next step
#
# Re-running is how you UPDATE: same command, newer tag, `docker pull` + recreate,
# data untouched (it lives in the data root, not the container).
#
# This installs a CONTROLLER node (the full app + UI). A headless compute node is
# `gpuk install --mode worker …` — see deployments/install/gpuk and
# specs/plateforme/installation.md.
#
# Design note — this script starts nothing itself. workerd owns the container's
# lifecycle (create, health-gate, roll back, update); the UI's update button and
# `gpuk update` drive that same daemon. One updater, three front doors.
set -eu
# ── Defaults (every one overridable by flag or env) ───────────────────────────
# The channel is the single source of truth for "what is the current release":
# it is served next to this script, and the backend's update check reads the SAME
# document (apps/controller/api/src/core/release-channel.ts). One file, one answer. Interim
# default: the public Gitea channel repo — flips to https://gpu.kitchen/latest.json
# once the hub exists (keep the three defaults in sync, see specs/developpement/ci-cd.md).
CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen/channel/raw/branch/main/latest.json}"
EDITION="${GPUK_EDITION:-community}"
DATA_ROOT="${GPUK_DATA_ROOT:-/var/lib/gpu-kitchen}"
CACHE_DIR="${GPUK_CACHE_DIR:-}"
# 1337, not 8080: the single most-squatted port in existence would make the
# conflict preflight fire on half the lab boxes out there (INS-01).
PORT="${GPUK_PORT:-1337}"
# Fixed listeners: mirror of the gpuk-proxy default (core/cluster-settings.ts
# proxyPublicPort) and of the worker mTLS channel — no install-time flag moves
# them (INS-46).
INFERENCE_PORT=8200
MTLS_PORT=8443
CLUSTER="${GPUK_CLUSTER:-default}"
PROFILE=""
DOMAIN=""
VERSION=""
IMAGE=""
WORKER_BINARY=""
GPUK_SCRIPT=""
SKIP_GPU_CHECK=0
SKIP_PREFLIGHT=0
NON_INTERACTIVE=0
DRY_RUN=0
GREEN=''; RED=''; YELLOW=''; BOLD=''; RESET=''
if [ -t 1 ]; then
GREEN=$(printf '\033[32m'); RED=$(printf '\033[31m')
YELLOW=$(printf '\033[33m'); BOLD=$(printf '\033[1m'); RESET=$(printf '\033[0m')
fi
ok() { echo " ${GREEN}✓${RESET} $*"; }
warn() { echo " ${YELLOW}!${RESET} $*"; }
step() { echo; echo "${BOLD}$*${RESET}"; }
die() { echo " ${RED}✗${RESET} $*" >&2; exit 1; }
# `curl … | sudo sh` leaves stdin holding the script itself, so questions are
# asked and answered on the controlling terminal — /dev/tty — when there is one
# (INS-45). No terminal (CI, provisioning), --non-interactive or --dry-run keep
# every historical flags-only behaviour.
can_prompt() {
[ "$NON_INTERACTIVE" -eq 0 ] || return 1
[ "$DRY_RUN" -eq 0 ] || return 1
(: < /dev/tty) 2>/dev/null
}
ask() { # $1 = prompt → $REPLY (empty on EOF)
printf '%s' "$1" > /dev/tty
IFS= read -r REPLY < /dev/tty || REPLY=""
}
usage() {
cat <<EOF
GPU Kitchen installer
curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh | sudo sh
curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh | sudo sh -s -- [options]
Options:
--version <tag> Install this release instead of the channel's current one
--image <ref> Use this controller image outright (implies --version none)
--edition <ed> community (default) | enterprise
--port <p> Port the UI listens on (default 1337)
--data-root <path> Where the database and secrets live (default /var/lib/gpu-kitchen)
--cache-dir <path> Model cache (default <data-root>/hf)
--cluster <name> Cluster name workers join (default "default")
--profile <p> homelab | studio | enterprise | public (no flag + a terminal
= the script asks; no flag + no terminal = first-run asks)
--domain <d> Domain for the public profile: writes a filled TLS
reverse-proxy example to <data-root>/caddy/Caddyfile
--non-interactive Never ask anything, even with a terminal attached
--worker-binary <p> Use a locally-built gpu-kitchen-worker instead of downloading one
--gpuk-script <p> Use a local copy of the gpuk installer
--skip-gpu-check Skip the 'docker run --gpus all' smoke test
--skip-preflight Skip the host checks entirely (CI: no docker, no GPU)
--dry-run Run the preflight and resolve the release, change nothing
-h, --help This
EOF
}
while [ $# -gt 0 ]; do
case "$1" in
--version) VERSION="$2"; shift 2 ;;
--image) IMAGE="$2"; shift 2 ;;
--edition) EDITION="$2"; shift 2 ;;
--port) PORT="$2"; shift 2 ;;
--data-root) DATA_ROOT="$2"; shift 2 ;;
--cache-dir) CACHE_DIR="$2"; shift 2 ;;
--cluster) CLUSTER="$2"; shift 2 ;;
--profile) PROFILE="$2"; shift 2 ;;
--domain) DOMAIN="$2"; shift 2 ;;
--non-interactive) NON_INTERACTIVE=1; shift ;;
--worker-binary) WORKER_BINARY="$2"; shift 2 ;;
--gpuk-script) GPUK_SCRIPT="$2"; shift 2 ;;
--skip-gpu-check) SKIP_GPU_CHECK=1; shift ;;
--skip-preflight) SKIP_PREFLIGHT=1; shift ;;
--dry-run) DRY_RUN=1; shift ;;
-h|--help) usage; exit 0 ;;
*) die "unknown option: $1 (try --help)" ;;
esac
done
[ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf"
case "$EDITION" in
community|enterprise) ;;
*) die "--edition must be community or enterprise (got '$EDITION')" ;;
esac
case "$PROFILE" in
""|homelab|studio|enterprise|public) ;;
*) die "--profile must be homelab, studio, enterprise or public (got '$PROFILE')" ;;
esac
case "$DOMAIN" in
*[!A-Za-z0-9.-]*) die "--domain must be a bare domain name (got '$DOMAIN')" ;;
esac
echo
echo "${BOLD}GPU Kitchen — installing the $EDITION controller${RESET}"
# ── 1. Preflight ─────────────────────────────────────────────────────────────
# The same checks tools/provision-feeder.sh makes, minus the compose ones: the
# all-in-one image is driven by workerd through the plain docker CLI, so there is
# no compose dependency to satisfy any more.
if [ "$SKIP_PREFLIGHT" -eq 1 ]; then
step "Preflight — skipped (--skip-preflight)"
warn "the host is NOT being checked for docker, a driver or a GPU"
else
step "Preflight — docker, NVIDIA driver, container toolkit"
[ "$DRY_RUN" -eq 1 ] || [ "$(id -u)" -eq 0 ] \
|| die "run as root: curl -fsSL … | sudo sh"
command -v curl >/dev/null 2>&1 || die "curl not found. Install curl first."
if command -v docker >/dev/null 2>&1; then
if docker info >/dev/null 2>&1; then
ok "docker daemon reachable ($(docker --version | cut -d' ' -f3 | tr -d ,))"
else
die "docker is installed but its daemon does not answer. Start the daemon, or run as root."
fi
else
die "docker not found. Install Docker Engine first: https://docs.docker.com/engine/install/"
fi
if command -v nvidia-smi >/dev/null 2>&1; then
DRIVER=$(nvidia-smi --query-gpu=driver_version --format=csv,noheader 2>/dev/null | head -1 || true)
GPU_COUNT=$(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | grep -c . || true)
if [ -n "$DRIVER" ] && [ "${GPU_COUNT:-0}" -gt 0 ]; then
ok "NVIDIA driver $DRIVER — ${GPU_COUNT} GPU(s): $(nvidia-smi --query-gpu=name --format=csv,noheader | sort -u | paste -sd', ')"
else
die "nvidia-smi is present but reports no GPU"
fi
else
die "nvidia-smi not found. Install the NVIDIA driver first."
fi
if [ "$SKIP_GPU_CHECK" -eq 1 ]; then
warn "nvidia-container-toolkit check skipped (--skip-gpu-check)"
elif docker info 2>/dev/null | grep -qiE 'runtimes:.*nvidia'; then
# The runtime being REGISTERED is not the same as it working. With the toolkit
# installed, --gpus injects the driver and nvidia-smi into a plain image; that
# is the exact mechanism the controller container relies on, so test it rather
# than infer it.
if docker run --rm --gpus all ubuntu:24.04 nvidia-smi -L >/dev/null 2>&1; then
ok "nvidia-container-toolkit works (a container can see the GPUs)"
else
die "the nvidia runtime is registered but 'docker run --gpus all' cannot see the GPUs. Reinstall nvidia-container-toolkit."
fi
else
die "nvidia-container-toolkit is not registered with docker. Install it, then restart dockerd:
https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html"
fi
# Model weights are large and the failure mode (a download dying at 90%) is
# miserable, so say so up front. A warning, not a refusal: it is the user's disk.
CACHE_PARENT="$CACHE_DIR"
while [ ! -d "$CACHE_PARENT" ] && [ "$CACHE_PARENT" != "/" ]; do
CACHE_PARENT=$(dirname "$CACHE_PARENT")
done
FREE_GB=$(df -BG --output=avail "$CACHE_PARENT" 2>/dev/null | tail -1 | tr -dc '0-9' || true)
if [ "${FREE_GB:-0}" -ge 100 ]; then
ok "model cache $CACHE_DIR — ${FREE_GB}G free"
else
warn "only ${FREE_GB:-?}G free under $CACHE_PARENT. Model weights need 100G or more."
fi
# ── Port conflicts (INS-46) ──
# A taken port must fail HERE, before anything mutates the host — today's
# alternative is a 3-minute health-check timeout with zero diagnosis. Best-effort
# detection (ss, then netstat); neither present is a warn, never a false red.
# A listener owned by an EXISTING GPU Kitchen install is not a conflict: re-running
# this script is the documented update path, and the manifest names our port.
# Test hook: force the detector. The netstat fallback is unreachable on any
# host that has ss (all of them, in practice), so the CI smoke pins it here to
# keep its parsing honest.
PORT_TOOL="${GPUK_PORT_CHECK_TOOL:-}"
case "$PORT_TOOL" in
""|ss|netstat) ;;
*) die "GPUK_PORT_CHECK_TOOL must be ss or netstat (got '$PORT_TOOL')" ;;
esac
if [ -z "$PORT_TOOL" ]; then
if command -v ss >/dev/null 2>&1; then PORT_TOOL="ss"
elif command -v netstat >/dev/null 2>&1; then PORT_TOOL="netstat"; fi
fi
port_busy() { # $1 = port → 0 iff something listens on TCP :$1
case "$PORT_TOOL" in
ss) [ -n "$(ss -ltnH "sport = :$1" 2>/dev/null)" ] ;;
netstat) netstat -ltn 2>/dev/null \
| awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) found=1} END{exit found?0:1}' ;;
*) return 1 ;;
esac
}
port_owner() { # $1 = port → best-effort process name (needs root for -p)
case "$PORT_TOOL" in
ss) ss -ltnpH "sport = :$1" 2>/dev/null \
| sed -n 's/.*users:((\"\([^"]*\)\".*/\1/p' | head -1 ;;
netstat) netstat -ltnp 2>/dev/null \
| awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) {print $NF; exit}}' \
| sed 's|^[0-9]*/||' ;;
esac
}
manifest_ui_port() { # the port an existing install already owns, if any
[ -f /etc/gpu-kitchen/manifest.json ] || return 0
# Bridge publishes GPUK_PUBLIC_PORT over the fixed container 8080; host
# networking moves the listener itself (GPUK_PORT). Same precedence as gpuk.
_p=$(sed -n 's/.*"GPUK_PUBLIC_PORT"[[:space:]]*:[[:space:]]*"\([0-9]*\)".*/\1/p' \
/etc/gpu-kitchen/manifest.json | head -1)
[ -n "$_p" ] || _p=$(sed -n 's/.*"GPUK_PORT"[[:space:]]*:[[:space:]]*"\([0-9]*\)".*/\1/p' \
/etc/gpu-kitchen/manifest.json | head -1)
printf '%s' "$_p"
}
if [ -z "$PORT_TOOL" ]; then
warn "cannot check for port conflicts (neither ss nor netstat found)"
else
HAVE_MANIFEST=0
[ ! -f /etc/gpu-kitchen/manifest.json ] || HAVE_MANIFEST=1
if [ "$HAVE_MANIFEST" -eq 1 ] && [ "$(manifest_ui_port)" = "$PORT" ]; then
ok "UI port $PORT — already ours (re-running is how you update)"
elif port_busy "$PORT"; then
OWNER=$(port_owner "$PORT")
OWNER="${OWNER:-an unknown process}"
if can_prompt; then
ALT=$((PORT + 1))
while port_busy "$ALT"; do ALT=$((ALT + 1)); done
# Propose, never auto-pick: the URL printed at the end and the idempotent
# re-run both need the operator to KNOW which port they chose.
ask " ${YELLOW}!${RESET} port $PORT is busy ($OWNER). Use $ALT instead? [$ALT], another port, or 'q' to abort: "
case "$REPLY" in
q|Q) die "port $PORT is in use by $OWNER. Run the install again with --port <p>." ;;
"") PORT="$ALT" ;;
*)
case "$REPLY" in
*[!0-9]*) die "not a port number: $REPLY" ;;
esac
if port_busy "$REPLY"; then
die "port $REPLY is busy too. Run the install again with --port <p>."
fi
PORT="$REPLY"
;;
esac
ok "UI port $PORT is free"
else
die "port $PORT is already in use by $OWNER. Pass --port <p> to choose another port."
fi
else
ok "UI port $PORT is free"
fi
# The mTLS and inference listeners have no install-time flag — assumed
# limitation (INS-46): the published mTLS port moves later via
# Settings -> Network. With a manifest present they are our own listeners.
if [ "$HAVE_MANIFEST" -eq 0 ]; then
if port_busy "$MTLS_PORT"; then
OWNER=$(port_owner "$MTLS_PORT")
die "port $MTLS_PORT (worker channel) is in use by ${OWNER:-an unknown process}. Free it first."
fi
if port_busy "$INFERENCE_PORT"; then
OWNER=$(port_owner "$INFERENCE_PORT")
die "port $INFERENCE_PORT (inference endpoint) is in use by ${OWNER:-an unknown process}. Free it first.
($INFERENCE_PORT is also HashiCorp Vault's default port.)"
fi
ok "worker channel port $MTLS_PORT and inference port $INFERENCE_PORT are free"
fi
fi
fi # end preflight
# ── 2. Resolve the release ───────────────────────────────────────────────────
step "Release — resolving the version to install"
# One tiny JSON document, fetched over TLS, holding what the current release IS.
# Parsed with sed rather than jq: `curl … | sudo sh` cannot assume jq exists, and
# the document is ours and flat.
json_field() { sed -n "s/.*\"$1\"[[:space:]]*:[[:space:]]*\"\([^\"]*\)\".*/\1/p" | head -1; }
# Strip a tag off an image reference WITHOUT mangling a registry's host:port.
# `${ref%%:*}` cuts at the FIRST colon and is WRONG: a
# `registry.internal:5000/gpuk/controller:v1.2.3` (or an untagged
# `registry.internal:5000/gpuk/controller`) would collapse to `registry.internal`.
# A colon is a tag separator only when the last colon comes AFTER the last slash.
# Same rule as deployments/install/gpuk's image_repo() and core/release-channel.ts (unit-tested).
image_repo() {
case "$1" in
*@*) printf '%s' "${1%@*}"; return ;; # digest pin → repo@sha256:…
esac
_t="${1##*:}"
case "$_t" in
"$1") printf '%s' "$1" ;; # no colon at all → already untagged
*/*) printf '%s' "$1" ;; # last colon is inside a path → host:port, untagged
*) printf '%s' "${1%:*}" ;;
esac
}
if [ -n "$IMAGE" ]; then
ok "using the image given on the command line: $IMAGE"
[ -n "$VERSION" ] || VERSION="(pinned by --image)"
else
CHANNEL=$(curl -fsSL --max-time 20 "$CHANNEL_URL" 2>/dev/null) \
|| die "cannot reach the release channel at $CHANNEL_URL
If this host is offline, pass --image <ref> to install a specific image directly."
[ -n "$VERSION" ] || VERSION=$(echo "$CHANNEL" | json_field version)
[ -n "$VERSION" ] || die "the release channel returned no version: $CHANNEL_URL"
if [ "$EDITION" = "enterprise" ]; then
IMAGE_TEMPLATE=$(echo "$CHANNEL" | json_field controllerImageEnterprise)
else
IMAGE_TEMPLATE=$(echo "$CHANNEL" | json_field controllerImage)
fi
[ -n "$IMAGE_TEMPLATE" ] \
|| die "the release channel names no $EDITION controller image: $CHANNEL_URL"
# The channel gives the repository; WE pin the tag. A floating `:latest` would
# make every container recreate a silent, unrequested upgrade. Strip any tag the
# channel already carries with image_repo (host:port-safe), then pin OUR version.
IMAGE="$(image_repo "$IMAGE_TEMPLATE"):${VERSION}"
ok "release $VERSION"
ok "image $IMAGE"
[ -n "$WORKER_BINARY" ] || WORKER_RELEASE_BASE=$(echo "$CHANNEL" | json_field workerBase)
[ -n "${GPUK_SCRIPT}" ] || GPUK_SCRIPT_URL=$(echo "$CHANNEL" | json_field gpukScript)
fi
# A direct --image and a channel version are held to the same immutable-image
# rule. A registry host:port is not a tag separator; only the last colon after
# the last slash counts. Digest pins are accepted too.
case "$IMAGE" in
*@sha256:*) ;;
*:latest) die "refusing floating image tag '$IMAGE'. Use an explicit release tag or digest." ;;
*)
IMAGE_TAG="${IMAGE##*:}"
case "$IMAGE_TAG" in
"$IMAGE"|*/*) die "controller image must carry an explicit tag or sha256 digest (got '$IMAGE')" ;;
esac
;;
esac
if [ -n "$DOMAIN" ] && [ -n "$PROFILE" ] && [ "$PROFILE" != "public" ]; then
die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)"
fi
if [ "$DRY_RUN" -eq 1 ]; then
step "Dry run — stopping here"
if [ "$SKIP_PREFLIGHT" -eq 1 ]; then
ok "the release resolved; the host was not checked; nothing was installed"
else
ok "preflight passed and the release resolved; nothing was installed"
fi
echo
echo " would install : $IMAGE"
echo " data root : $DATA_ROOT"
echo " model cache : $CACHE_DIR"
echo " UI port : $PORT"
if [ -n "$PROFILE" ]; then
echo " profile : $PROFILE"
else
echo " profile : (asked on the terminal, or chosen during first run)"
fi
[ -z "$DOMAIN" ] || echo " domain : $DOMAIN"
exit 0
fi
# ── 3. Resolve the installation profile (INS-45) ─────────────────────────────
# Only ever on a terminal, and only when --profile was not given. The answer is
# relayed verbatim as --profile: the backend stays the sole applier of profile
# defaults and floors. Without a terminal the historical path is untouched —
# profile unset, enterprise floors, the first-run wizard requires the choice
# (PRF-05).
if [ -z "$PROFILE" ] && can_prompt; then
step "Security profile — how will this kitchen be used?"
cat > /dev/tty <<'PROFILES'
1) Home lab a trusted home network
No sign-in on your home network. Nearby workers are found and join
without waiting for approval.
2) Studio one control station, shared compute
Control stays on this machine. Colleagues use API keys, while nearby
workers wait for your approval.
3) Enterprise a managed company network
Sign-in is required. The controller stays quiet on the network; known
workers can request your approval.
4) Public server direct internet exposure
Sign-in and hardened browser transport are required. Network discovery
is off and workers join only by token.
You can change this later in Settings. Stricter floors re-apply. Nothing
already issued is revoked.
PROFILES
while :; do
ask " Choose a profile [1-4, Enter = 1 (Home lab)]: "
case "$REPLY" in
""|1|homelab) PROFILE="homelab" ;;
2|studio) PROFILE="studio" ;;
3|enterprise) PROFILE="enterprise" ;;
4|public) PROFILE="public" ;;
*) printf '%s\n' " pick 1, 2, 3 or 4" > /dev/tty; continue ;;
esac
break
done
ok "profile: $PROFILE"
fi
# Under public, a domain lets gpuk write a FILLED TLS reverse-proxy example
# (INS-47). Optional: Enter skips, and the banner still points at the shipped
# Caddyfile.example.
if [ "$PROFILE" = "public" ] && [ -z "$DOMAIN" ] && can_prompt; then
while :; do
ask " Domain for HTTPS access (e.g. gpu.example.com — Enter to skip): "
# Accept the copy-paste reflex: strip a pasted scheme and anything after
# the first slash, then insist on a bare domain rather than skipping —
# a silently dropped answer would be discovered hours later, at DNS time.
REPLY="${REPLY#https://}"
REPLY="${REPLY#http://}"
REPLY="${REPLY%%/*}"
[ -n "$REPLY" ] || break
case "$REPLY" in
*[!A-Za-z0-9.-]*)
printf '%s\n' " not a bare domain name: $REPLY. Try again, or press Enter to skip." > /dev/tty
;;
*) DOMAIN="$REPLY"; break ;;
esac
done
fi
if [ -n "$DOMAIN" ] && [ "$PROFILE" != "public" ]; then
die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)"
fi
# ── 4. The host daemon ───────────────────────────────────────────────────────
step "Host daemon — gpu-kitchen-worker"
TMP=$(mktemp -d)
# shellcheck disable=SC2064 # expand TMP now: it must be removed even if it changes
trap "rm -rf '$TMP'" EXIT INT TERM
if [ -z "$GPUK_SCRIPT" ]; then
# A checkout right here beats a download (that is how contributors run it).
_local="$(dirname "$0")/gpuk"
if [ -f "$_local" ]; then
GPUK_SCRIPT="$_local"
ok "using the gpuk installer from this checkout"
else
[ -n "${GPUK_SCRIPT_URL:-}" ] \
|| die "the release channel names no gpuk installer, and none was found locally"
curl -fsSL --max-time 60 "$GPUK_SCRIPT_URL" -o "$TMP/gpuk" \
|| die "cannot download the gpuk installer from $GPUK_SCRIPT_URL"
chmod +x "$TMP/gpuk"
GPUK_SCRIPT="$TMP/gpuk"
ok "downloaded the gpuk installer"
fi
fi
# `gpuk install` does the rest: it drops the binary, writes the systemd unit,
# writes the manifest (the declarative description of the app container) and
# applies it. Everything below is passed straight through to it.
set -- install \
--mode controller \
--image "$IMAGE" \
--cluster "$CLUSTER" \
--data-root "$DATA_ROOT" \
--cache-dir "$CACHE_DIR" \
--http-port "$PORT"
[ -z "$PROFILE" ] || set -- "$@" --profile "$PROFILE"
[ -z "$DOMAIN" ] || set -- "$@" --domain "$DOMAIN"
if [ -n "$WORKER_BINARY" ]; then
[ -f "$WORKER_BINARY" ] || die "no such worker binary: $WORKER_BINARY"
set -- "$@" --binary "$WORKER_BINARY"
ok "using a locally-built gpu-kitchen-worker"
elif [ -n "${WORKER_RELEASE_BASE:-}" ]; then
GPUK_RELEASE_BASE="$WORKER_RELEASE_BASE"
export GPUK_RELEASE_BASE
ok "gpu-kitchen-worker will be downloaded from the release"
fi
# else: gpuk reuses an already-installed binary, or fails with its own message.
step "Installing — this pulls the image, so it can take a few minutes"
sh "$GPUK_SCRIPT" "$@" || die "the install failed. See: journalctl -u gpu-kitchen-worker"
# ── 5. Wait for the app, then say where it is ────────────────────────────────
step "Waiting for the controller to answer"
i=0
until curl -fsS -o /dev/null --max-time 3 "http://127.0.0.1:$PORT/health" 2>/dev/null; do
i=$((i + 1))
[ "$i" -lt 90 ] || die "the controller did not answer on :$PORT within 90s. See: gpuk logs"
sleep 2
done
ok "the controller is up"
# The URL must name the REAL host: the person installing this is very often not
# sitting at the machine, and "localhost" would be a lie on every box but theirs
# (same reason the backend resolves its own hostname — core/host-name.ts).
HOSTNAME_FQDN=$(hostname -f 2>/dev/null || hostname 2>/dev/null || echo localhost)
LAN_IP=$(ip route get 1.1.1.1 2>/dev/null | sed -n 's/.* src \([0-9.]*\).*/\1/p' | head -1)
echo
echo "${BOLD}${GREEN}GPU Kitchen is running.${RESET}"
echo
echo " ${BOLD}Open:${RESET} http://${HOSTNAME_FQDN}:${PORT}"
[ -n "$LAN_IP" ] && echo " http://${LAN_IP}:${PORT}"
echo
CLAIM_CODE_FILE="$DATA_ROOT/secrets/claim_code"
CLAIM_CODE=""
[ ! -s "$CLAIM_CODE_FILE" ] || CLAIM_CODE=$(cat "$CLAIM_CODE_FILE" 2>/dev/null || true)
if [ -n "$CLAIM_CODE" ]; then
echo " ${BOLD}Claim code:${RESET} $CLAIM_CODE"
echo " ${BOLD}Read again:${RESET} $CLAIM_CODE_FILE (mode 0600; removed after claim)"
else
echo " ${BOLD}Claim code:${RESET} consumed (the first account already exists)"
fi
echo
case "$PROFILE" in
public)
echo " ${BOLD}Next:${RESET} create the first administrator in the UI"
echo
# The public profile REQUIRES a TLS reverse proxy (OPS-13, INS-47) — the UI
# port speaks plain HTTP. The product cannot verify the proxy's presence
# (OPS-67), so the closest thing to enforcement is saying it here, clearly.
if [ -n "$DOMAIN" ]; then
echo " ${BOLD}HTTPS:${RESET} https://${DOMAIN} answers after these steps:"
echo " 1. DNS: point ${DOMAIN} at this machine's public IP"
echo " 2. Install Caddy: https://caddyserver.com/docs/install"
echo " 3. sudo cp $DATA_ROOT/caddy/Caddyfile /etc/caddy/Caddyfile"
echo " sudo systemctl reload caddy"
echo " A filled example was written to $DATA_ROOT/caddy/Caddyfile."
echo " Until then the UI answers in cleartext on the URLs above."
else
echo " ${BOLD}HTTPS:${RESET} the public profile requires a TLS reverse proxy before any"
echo " public exposure. See deployments/controller/Caddyfile.example."
fi
echo
;;
enterprise)
echo " ${BOLD}Next:${RESET} create the first administrator in the UI"
echo
;;
homelab|studio)
echo " ${BOLD}Next:${RESET} finish first-run in the UI"
echo
;;
"")
echo " ${BOLD}Next:${RESET} choose an installation profile in the first-run assistant"
echo
;;
esac
echo " Inference endpoint : http://${HOSTNAME_FQDN}:${INFERENCE_PORT}/v1"
echo " Version : ${VERSION}"
echo
# Pre-existing model caches (B81). `gpuk install` prints the same hint, but that
# scrolls past mid-install; this banner is where people actually look. Detection
# only — cache questions belong to the first-run wizard (INS-48, REG-41), never
# to the CLI, and nothing outside $CACHE_DIR is touched, read as configuration,
# or modified.
EXISTING_CACHE=""
CHOSEN_CACHE=$(readlink -f "$CACHE_DIR" 2>/dev/null || echo "$CACHE_DIR")
for d in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do
[ -n "$d" ] && [ -d "$d/hub" ] || continue
real=$(readlink -f "$d" 2>/dev/null || echo "$d")
[ "$real" != "$CHOSEN_CACHE" ] || continue
ls -d "$d"/hub/models--* >/dev/null 2>&1 || continue
EXISTING_CACHE="$EXISTING_CACHE $real"
done
if [ -n "$EXISTING_CACHE" ]; then
for d in $EXISTING_CACHE; do
echo " ${BOLD}Existing model cache:${RESET} $d"
done
echo " GPU Kitchen will offer to reuse those models at first launch, and any"
echo " time from Nodes & GPU -> Storage. A reused cache is referenced in"
echo " place. Nothing is moved or deleted."
echo
fi
echo " Update : re-run this command, or press Update in the UI, or: gpuk update"
echo " Status : gpuk status Logs: gpuk logs Remove: gpuk uninstall"
echo