651 lines
28 KiB
Bash
651 lines
28 KiB
Bash
#!/bin/sh
|
|
# GPU Kitchen — one-command install (specs/plateforme/installation.md).
|
|
#
|
|
# curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh | sudo sh
|
|
#
|
|
# (Interim URL — becomes https://gpu.kitchen/install.sh once the hub exists; the
|
|
# channel repo is written by .gitea/workflows/release.yml, see specs/developpement/ci-cd.md.)
|
|
#
|
|
# What it does, and nothing more:
|
|
# 1. preflight docker, the NVIDIA driver, a REAL `--gpus all` smoke test, and
|
|
# the listening ports (INS-46 — a taken port fails HERE, not three
|
|
# minutes later in a health-check timeout)
|
|
# 2. resolve the current release from the channel (a TAG — never a floating
|
|
# `latest`: an install that silently changes version under you is
|
|
# not an install, it is a surprise)
|
|
# 3. ask the security profile — and, under `public`, a domain — when a
|
|
# terminal is attached (INS-45); no terminal, no questions, and
|
|
# the first-run wizard asks instead (PRF-05)
|
|
# 4. install `gpu-kitchen-worker` (workerd), the privileged host daemon, via
|
|
# the existing `gpuk` installer — on a controller node it OWNS the
|
|
# app container's lifecycle
|
|
# 5. hand over workerd pulls the pinned all-in-one controller image and starts it
|
|
# 6. print the UI URL on the real host, plus the profile-specific next step
|
|
#
|
|
# Re-running is how you UPDATE: same command, newer tag, `docker pull` + recreate,
|
|
# data untouched (it lives in the data root, not the container).
|
|
#
|
|
# This installs a CONTROLLER node (the full app + UI). A headless compute node is
|
|
# `gpuk install --mode worker …` — see deployments/install/gpuk and
|
|
# specs/plateforme/installation.md.
|
|
#
|
|
# Design note — this script starts nothing itself. workerd owns the container's
|
|
# lifecycle (create, health-gate, roll back, update); the UI's update button and
|
|
# `gpuk update` drive that same daemon. One updater, three front doors.
|
|
set -eu
|
|
|
|
# ── Defaults (every one overridable by flag or env) ───────────────────────────
|
|
# The channel is the single source of truth for "what is the current release":
|
|
# it is served next to this script, and the backend's update check reads the SAME
|
|
# document (apps/controller/api/src/core/release-channel.ts). One file, one answer. Interim
|
|
# default: the public Gitea channel repo — flips to https://gpu.kitchen/latest.json
|
|
# once the hub exists (keep the three defaults in sync, see specs/developpement/ci-cd.md).
|
|
CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen/channel/raw/branch/main/latest.json}"
|
|
EDITION="${GPUK_EDITION:-community}"
|
|
DATA_ROOT="${GPUK_DATA_ROOT:-/var/lib/gpu-kitchen}"
|
|
CACHE_DIR="${GPUK_CACHE_DIR:-}"
|
|
# 1337, not 8080: the single most-squatted port in existence would make the
|
|
# conflict preflight fire on half the lab boxes out there (INS-01).
|
|
PORT="${GPUK_PORT:-1337}"
|
|
# Fixed listeners: mirror of the gpuk-proxy default (core/cluster-settings.ts
|
|
# proxyPublicPort) and of the worker mTLS channel — no install-time flag moves
|
|
# them (INS-46).
|
|
INFERENCE_PORT=8200
|
|
MTLS_PORT=8443
|
|
CLUSTER="${GPUK_CLUSTER:-default}"
|
|
PROFILE=""
|
|
DOMAIN=""
|
|
VERSION=""
|
|
IMAGE=""
|
|
WORKER_BINARY=""
|
|
GPUK_SCRIPT=""
|
|
SKIP_GPU_CHECK=0
|
|
SKIP_PREFLIGHT=0
|
|
NON_INTERACTIVE=0
|
|
DRY_RUN=0
|
|
|
|
GREEN=''; RED=''; YELLOW=''; BOLD=''; RESET=''
|
|
if [ -t 1 ]; then
|
|
GREEN=$(printf '\033[32m'); RED=$(printf '\033[31m')
|
|
YELLOW=$(printf '\033[33m'); BOLD=$(printf '\033[1m'); RESET=$(printf '\033[0m')
|
|
fi
|
|
|
|
ok() { echo " ${GREEN}✓${RESET} $*"; }
|
|
warn() { echo " ${YELLOW}!${RESET} $*"; }
|
|
step() { echo; echo "${BOLD}$*${RESET}"; }
|
|
die() { echo " ${RED}✗${RESET} $*" >&2; exit 1; }
|
|
|
|
# `curl … | sudo sh` leaves stdin holding the script itself, so questions are
|
|
# asked and answered on the controlling terminal — /dev/tty — when there is one
|
|
# (INS-45). No terminal (CI, provisioning), --non-interactive or --dry-run keep
|
|
# every historical flags-only behaviour.
|
|
can_prompt() {
|
|
[ "$NON_INTERACTIVE" -eq 0 ] || return 1
|
|
[ "$DRY_RUN" -eq 0 ] || return 1
|
|
(: < /dev/tty) 2>/dev/null
|
|
}
|
|
|
|
ask() { # $1 = prompt → $REPLY (empty on EOF)
|
|
printf '%s' "$1" > /dev/tty
|
|
IFS= read -r REPLY < /dev/tty || REPLY=""
|
|
}
|
|
|
|
usage() {
|
|
cat <<EOF
|
|
GPU Kitchen installer
|
|
|
|
curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh | sudo sh
|
|
curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh | sudo sh -s -- [options]
|
|
|
|
Options:
|
|
--version <tag> Install this release instead of the channel's current one
|
|
--image <ref> Use this controller image outright (implies --version none)
|
|
--edition <ed> community (default) | enterprise
|
|
--port <p> Port the UI listens on (default 1337)
|
|
--data-root <path> Where the database and secrets live (default /var/lib/gpu-kitchen)
|
|
--cache-dir <path> Model cache (default <data-root>/hf)
|
|
--cluster <name> Cluster name workers join (default "default")
|
|
--profile <p> homelab | studio | enterprise | public (no flag + a terminal
|
|
= the script asks; no flag + no terminal = first-run asks)
|
|
--domain <d> Domain for the public profile: writes a filled TLS
|
|
reverse-proxy example to <data-root>/caddy/Caddyfile
|
|
--non-interactive Never ask anything, even with a terminal attached
|
|
--worker-binary <p> Use a locally-built gpu-kitchen-worker instead of downloading one
|
|
--gpuk-script <p> Use a local copy of the gpuk installer
|
|
--skip-gpu-check Skip the 'docker run --gpus all' smoke test
|
|
--skip-preflight Skip the host checks entirely (CI: no docker, no GPU)
|
|
--dry-run Run the preflight and resolve the release, change nothing
|
|
-h, --help This
|
|
EOF
|
|
}
|
|
|
|
while [ $# -gt 0 ]; do
|
|
case "$1" in
|
|
--version) VERSION="$2"; shift 2 ;;
|
|
--image) IMAGE="$2"; shift 2 ;;
|
|
--edition) EDITION="$2"; shift 2 ;;
|
|
--port) PORT="$2"; shift 2 ;;
|
|
--data-root) DATA_ROOT="$2"; shift 2 ;;
|
|
--cache-dir) CACHE_DIR="$2"; shift 2 ;;
|
|
--cluster) CLUSTER="$2"; shift 2 ;;
|
|
--profile) PROFILE="$2"; shift 2 ;;
|
|
--domain) DOMAIN="$2"; shift 2 ;;
|
|
--non-interactive) NON_INTERACTIVE=1; shift ;;
|
|
--worker-binary) WORKER_BINARY="$2"; shift 2 ;;
|
|
--gpuk-script) GPUK_SCRIPT="$2"; shift 2 ;;
|
|
--skip-gpu-check) SKIP_GPU_CHECK=1; shift ;;
|
|
--skip-preflight) SKIP_PREFLIGHT=1; shift ;;
|
|
--dry-run) DRY_RUN=1; shift ;;
|
|
-h|--help) usage; exit 0 ;;
|
|
*) die "unknown option: $1 (try --help)" ;;
|
|
esac
|
|
done
|
|
|
|
[ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf"
|
|
|
|
case "$EDITION" in
|
|
community|enterprise) ;;
|
|
*) die "--edition must be community or enterprise (got '$EDITION')" ;;
|
|
esac
|
|
|
|
case "$PROFILE" in
|
|
""|homelab|studio|enterprise|public) ;;
|
|
*) die "--profile must be homelab, studio, enterprise or public (got '$PROFILE')" ;;
|
|
esac
|
|
|
|
case "$DOMAIN" in
|
|
*[!A-Za-z0-9.-]*) die "--domain must be a bare domain name (got '$DOMAIN')" ;;
|
|
esac
|
|
|
|
echo
|
|
echo "${BOLD}GPU Kitchen — installing the $EDITION controller${RESET}"
|
|
|
|
# ── 1. Preflight ─────────────────────────────────────────────────────────────
|
|
# The same checks tools/provision-feeder.sh makes, minus the compose ones: the
|
|
# all-in-one image is driven by workerd through the plain docker CLI, so there is
|
|
# no compose dependency to satisfy any more.
|
|
if [ "$SKIP_PREFLIGHT" -eq 1 ]; then
|
|
|
|
step "Preflight — skipped (--skip-preflight)"
|
|
warn "the host is NOT being checked for docker, a driver or a GPU"
|
|
|
|
else
|
|
|
|
step "Preflight — docker, NVIDIA driver, container toolkit"
|
|
|
|
[ "$DRY_RUN" -eq 1 ] || [ "$(id -u)" -eq 0 ] \
|
|
|| die "run as root: curl -fsSL … | sudo sh"
|
|
|
|
command -v curl >/dev/null 2>&1 || die "curl not found. Install curl first."
|
|
|
|
if command -v docker >/dev/null 2>&1; then
|
|
if docker info >/dev/null 2>&1; then
|
|
ok "docker daemon reachable ($(docker --version | cut -d' ' -f3 | tr -d ,))"
|
|
else
|
|
die "docker is installed but its daemon does not answer. Start the daemon, or run as root."
|
|
fi
|
|
else
|
|
die "docker not found. Install Docker Engine first: https://docs.docker.com/engine/install/"
|
|
fi
|
|
|
|
if command -v nvidia-smi >/dev/null 2>&1; then
|
|
DRIVER=$(nvidia-smi --query-gpu=driver_version --format=csv,noheader 2>/dev/null | head -1 || true)
|
|
GPU_COUNT=$(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | grep -c . || true)
|
|
if [ -n "$DRIVER" ] && [ "${GPU_COUNT:-0}" -gt 0 ]; then
|
|
ok "NVIDIA driver $DRIVER — ${GPU_COUNT} GPU(s): $(nvidia-smi --query-gpu=name --format=csv,noheader | sort -u | paste -sd', ')"
|
|
else
|
|
die "nvidia-smi is present but reports no GPU"
|
|
fi
|
|
else
|
|
die "nvidia-smi not found. Install the NVIDIA driver first."
|
|
fi
|
|
|
|
if [ "$SKIP_GPU_CHECK" -eq 1 ]; then
|
|
warn "nvidia-container-toolkit check skipped (--skip-gpu-check)"
|
|
elif docker info 2>/dev/null | grep -qiE 'runtimes:.*nvidia'; then
|
|
# The runtime being REGISTERED is not the same as it working. With the toolkit
|
|
# installed, --gpus injects the driver and nvidia-smi into a plain image; that
|
|
# is the exact mechanism the controller container relies on, so test it rather
|
|
# than infer it.
|
|
if docker run --rm --gpus all ubuntu:24.04 nvidia-smi -L >/dev/null 2>&1; then
|
|
ok "nvidia-container-toolkit works (a container can see the GPUs)"
|
|
else
|
|
die "the nvidia runtime is registered but 'docker run --gpus all' cannot see the GPUs. Reinstall nvidia-container-toolkit."
|
|
fi
|
|
else
|
|
die "nvidia-container-toolkit is not registered with docker. Install it, then restart dockerd:
|
|
https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html"
|
|
fi
|
|
|
|
# Model weights are large and the failure mode (a download dying at 90%) is
|
|
# miserable, so say so up front. A warning, not a refusal: it is the user's disk.
|
|
CACHE_PARENT="$CACHE_DIR"
|
|
while [ ! -d "$CACHE_PARENT" ] && [ "$CACHE_PARENT" != "/" ]; do
|
|
CACHE_PARENT=$(dirname "$CACHE_PARENT")
|
|
done
|
|
FREE_GB=$(df -BG --output=avail "$CACHE_PARENT" 2>/dev/null | tail -1 | tr -dc '0-9' || true)
|
|
if [ "${FREE_GB:-0}" -ge 100 ]; then
|
|
ok "model cache $CACHE_DIR — ${FREE_GB}G free"
|
|
else
|
|
warn "only ${FREE_GB:-?}G free under $CACHE_PARENT. Model weights need 100G or more."
|
|
fi
|
|
|
|
# ── Port conflicts (INS-46) ──
|
|
# A taken port must fail HERE, before anything mutates the host — today's
|
|
# alternative is a 3-minute health-check timeout with zero diagnosis. Best-effort
|
|
# detection (ss, then netstat); neither present is a warn, never a false red.
|
|
# A listener owned by an EXISTING GPU Kitchen install is not a conflict: re-running
|
|
# this script is the documented update path, and the manifest names our port.
|
|
# Test hook: force the detector. The netstat fallback is unreachable on any
|
|
# host that has ss (all of them, in practice), so the CI smoke pins it here to
|
|
# keep its parsing honest.
|
|
PORT_TOOL="${GPUK_PORT_CHECK_TOOL:-}"
|
|
case "$PORT_TOOL" in
|
|
""|ss|netstat) ;;
|
|
*) die "GPUK_PORT_CHECK_TOOL must be ss or netstat (got '$PORT_TOOL')" ;;
|
|
esac
|
|
if [ -z "$PORT_TOOL" ]; then
|
|
if command -v ss >/dev/null 2>&1; then PORT_TOOL="ss"
|
|
elif command -v netstat >/dev/null 2>&1; then PORT_TOOL="netstat"; fi
|
|
fi
|
|
|
|
port_busy() { # $1 = port → 0 iff something listens on TCP :$1
|
|
case "$PORT_TOOL" in
|
|
ss) [ -n "$(ss -ltnH "sport = :$1" 2>/dev/null)" ] ;;
|
|
netstat) netstat -ltn 2>/dev/null \
|
|
| awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) found=1} END{exit found?0:1}' ;;
|
|
*) return 1 ;;
|
|
esac
|
|
}
|
|
|
|
port_owner() { # $1 = port → best-effort process name (needs root for -p)
|
|
case "$PORT_TOOL" in
|
|
ss) ss -ltnpH "sport = :$1" 2>/dev/null \
|
|
| sed -n 's/.*users:((\"\([^"]*\)\".*/\1/p' | head -1 ;;
|
|
netstat) netstat -ltnp 2>/dev/null \
|
|
| awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) {print $NF; exit}}' \
|
|
| sed 's|^[0-9]*/||' ;;
|
|
esac
|
|
}
|
|
|
|
manifest_ui_port() { # the port an existing install already owns, if any
|
|
[ -f /etc/gpu-kitchen/manifest.json ] || return 0
|
|
# Bridge publishes GPUK_PUBLIC_PORT over the fixed container 8080; host
|
|
# networking moves the listener itself (GPUK_PORT). Same precedence as gpuk.
|
|
_p=$(sed -n 's/.*"GPUK_PUBLIC_PORT"[[:space:]]*:[[:space:]]*"\([0-9]*\)".*/\1/p' \
|
|
/etc/gpu-kitchen/manifest.json | head -1)
|
|
[ -n "$_p" ] || _p=$(sed -n 's/.*"GPUK_PORT"[[:space:]]*:[[:space:]]*"\([0-9]*\)".*/\1/p' \
|
|
/etc/gpu-kitchen/manifest.json | head -1)
|
|
printf '%s' "$_p"
|
|
}
|
|
|
|
if [ -z "$PORT_TOOL" ]; then
|
|
warn "cannot check for port conflicts (neither ss nor netstat found)"
|
|
else
|
|
HAVE_MANIFEST=0
|
|
[ ! -f /etc/gpu-kitchen/manifest.json ] || HAVE_MANIFEST=1
|
|
if [ "$HAVE_MANIFEST" -eq 1 ] && [ "$(manifest_ui_port)" = "$PORT" ]; then
|
|
ok "UI port $PORT — already ours (re-running is how you update)"
|
|
elif port_busy "$PORT"; then
|
|
OWNER=$(port_owner "$PORT")
|
|
OWNER="${OWNER:-an unknown process}"
|
|
if can_prompt; then
|
|
ALT=$((PORT + 1))
|
|
while port_busy "$ALT"; do ALT=$((ALT + 1)); done
|
|
# Propose, never auto-pick: the URL printed at the end and the idempotent
|
|
# re-run both need the operator to KNOW which port they chose.
|
|
ask " ${YELLOW}!${RESET} port $PORT is busy ($OWNER). Use $ALT instead? [$ALT], another port, or 'q' to abort: "
|
|
case "$REPLY" in
|
|
q|Q) die "port $PORT is in use by $OWNER. Run the install again with --port <p>." ;;
|
|
"") PORT="$ALT" ;;
|
|
*)
|
|
case "$REPLY" in
|
|
*[!0-9]*) die "not a port number: $REPLY" ;;
|
|
esac
|
|
if port_busy "$REPLY"; then
|
|
die "port $REPLY is busy too. Run the install again with --port <p>."
|
|
fi
|
|
PORT="$REPLY"
|
|
;;
|
|
esac
|
|
ok "UI port $PORT is free"
|
|
else
|
|
die "port $PORT is already in use by $OWNER. Pass --port <p> to choose another port."
|
|
fi
|
|
else
|
|
ok "UI port $PORT is free"
|
|
fi
|
|
# The mTLS and inference listeners have no install-time flag — assumed
|
|
# limitation (INS-46): the published mTLS port moves later via
|
|
# Settings -> Network. With a manifest present they are our own listeners.
|
|
if [ "$HAVE_MANIFEST" -eq 0 ]; then
|
|
if port_busy "$MTLS_PORT"; then
|
|
OWNER=$(port_owner "$MTLS_PORT")
|
|
die "port $MTLS_PORT (worker channel) is in use by ${OWNER:-an unknown process}. Free it first."
|
|
fi
|
|
if port_busy "$INFERENCE_PORT"; then
|
|
OWNER=$(port_owner "$INFERENCE_PORT")
|
|
die "port $INFERENCE_PORT (inference endpoint) is in use by ${OWNER:-an unknown process}. Free it first.
|
|
($INFERENCE_PORT is also HashiCorp Vault's default port.)"
|
|
fi
|
|
ok "worker channel port $MTLS_PORT and inference port $INFERENCE_PORT are free"
|
|
fi
|
|
fi
|
|
|
|
fi # end preflight
|
|
|
|
# ── 2. Resolve the release ───────────────────────────────────────────────────
|
|
step "Release — resolving the version to install"
|
|
|
|
# One tiny JSON document, fetched over TLS, holding what the current release IS.
|
|
# Parsed with sed rather than jq: `curl … | sudo sh` cannot assume jq exists, and
|
|
# the document is ours and flat.
|
|
json_field() { sed -n "s/.*\"$1\"[[:space:]]*:[[:space:]]*\"\([^\"]*\)\".*/\1/p" | head -1; }
|
|
|
|
# Strip a tag off an image reference WITHOUT mangling a registry's host:port.
|
|
# `${ref%%:*}` cuts at the FIRST colon and is WRONG: a
|
|
# `registry.internal:5000/gpuk/controller:v1.2.3` (or an untagged
|
|
# `registry.internal:5000/gpuk/controller`) would collapse to `registry.internal`.
|
|
# A colon is a tag separator only when the last colon comes AFTER the last slash.
|
|
# Same rule as deployments/install/gpuk's image_repo() and core/release-channel.ts (unit-tested).
|
|
image_repo() {
|
|
case "$1" in
|
|
*@*) printf '%s' "${1%@*}"; return ;; # digest pin → repo@sha256:…
|
|
esac
|
|
_t="${1##*:}"
|
|
case "$_t" in
|
|
"$1") printf '%s' "$1" ;; # no colon at all → already untagged
|
|
*/*) printf '%s' "$1" ;; # last colon is inside a path → host:port, untagged
|
|
*) printf '%s' "${1%:*}" ;;
|
|
esac
|
|
}
|
|
|
|
if [ -n "$IMAGE" ]; then
|
|
ok "using the image given on the command line: $IMAGE"
|
|
[ -n "$VERSION" ] || VERSION="(pinned by --image)"
|
|
else
|
|
CHANNEL=$(curl -fsSL --max-time 20 "$CHANNEL_URL" 2>/dev/null) \
|
|
|| die "cannot reach the release channel at $CHANNEL_URL
|
|
If this host is offline, pass --image <ref> to install a specific image directly."
|
|
|
|
[ -n "$VERSION" ] || VERSION=$(echo "$CHANNEL" | json_field version)
|
|
[ -n "$VERSION" ] || die "the release channel returned no version: $CHANNEL_URL"
|
|
|
|
if [ "$EDITION" = "enterprise" ]; then
|
|
IMAGE_TEMPLATE=$(echo "$CHANNEL" | json_field controllerImageEnterprise)
|
|
else
|
|
IMAGE_TEMPLATE=$(echo "$CHANNEL" | json_field controllerImage)
|
|
fi
|
|
[ -n "$IMAGE_TEMPLATE" ] \
|
|
|| die "the release channel names no $EDITION controller image: $CHANNEL_URL"
|
|
|
|
# The channel gives the repository; WE pin the tag. A floating `:latest` would
|
|
# make every container recreate a silent, unrequested upgrade. Strip any tag the
|
|
# channel already carries with image_repo (host:port-safe), then pin OUR version.
|
|
IMAGE="$(image_repo "$IMAGE_TEMPLATE"):${VERSION}"
|
|
ok "release $VERSION"
|
|
ok "image $IMAGE"
|
|
|
|
[ -n "$WORKER_BINARY" ] || WORKER_RELEASE_BASE=$(echo "$CHANNEL" | json_field workerBase)
|
|
[ -n "${GPUK_SCRIPT}" ] || GPUK_SCRIPT_URL=$(echo "$CHANNEL" | json_field gpukScript)
|
|
fi
|
|
|
|
# A direct --image and a channel version are held to the same immutable-image
|
|
# rule. A registry host:port is not a tag separator; only the last colon after
|
|
# the last slash counts. Digest pins are accepted too.
|
|
case "$IMAGE" in
|
|
*@sha256:*) ;;
|
|
*:latest) die "refusing floating image tag '$IMAGE'. Use an explicit release tag or digest." ;;
|
|
*)
|
|
IMAGE_TAG="${IMAGE##*:}"
|
|
case "$IMAGE_TAG" in
|
|
"$IMAGE"|*/*) die "controller image must carry an explicit tag or sha256 digest (got '$IMAGE')" ;;
|
|
esac
|
|
;;
|
|
esac
|
|
|
|
if [ -n "$DOMAIN" ] && [ -n "$PROFILE" ] && [ "$PROFILE" != "public" ]; then
|
|
die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)"
|
|
fi
|
|
|
|
if [ "$DRY_RUN" -eq 1 ]; then
|
|
step "Dry run — stopping here"
|
|
if [ "$SKIP_PREFLIGHT" -eq 1 ]; then
|
|
ok "the release resolved; the host was not checked; nothing was installed"
|
|
else
|
|
ok "preflight passed and the release resolved; nothing was installed"
|
|
fi
|
|
echo
|
|
echo " would install : $IMAGE"
|
|
echo " data root : $DATA_ROOT"
|
|
echo " model cache : $CACHE_DIR"
|
|
echo " UI port : $PORT"
|
|
if [ -n "$PROFILE" ]; then
|
|
echo " profile : $PROFILE"
|
|
else
|
|
echo " profile : (asked on the terminal, or chosen during first run)"
|
|
fi
|
|
[ -z "$DOMAIN" ] || echo " domain : $DOMAIN"
|
|
exit 0
|
|
fi
|
|
|
|
# ── 3. Resolve the installation profile (INS-45) ─────────────────────────────
|
|
# Only ever on a terminal, and only when --profile was not given. The answer is
|
|
# relayed verbatim as --profile: the backend stays the sole applier of profile
|
|
# defaults and floors. Without a terminal the historical path is untouched —
|
|
# profile unset, enterprise floors, the first-run wizard requires the choice
|
|
# (PRF-05).
|
|
if [ -z "$PROFILE" ] && can_prompt; then
|
|
step "Security profile — how will this kitchen be used?"
|
|
cat > /dev/tty <<'PROFILES'
|
|
1) Home lab a trusted home network
|
|
No sign-in on your home network. Nearby workers are found and join
|
|
without waiting for approval.
|
|
|
|
2) Studio one control station, shared compute
|
|
Control stays on this machine. Colleagues use API keys, while nearby
|
|
workers wait for your approval.
|
|
|
|
3) Enterprise a managed company network
|
|
Sign-in is required. The controller stays quiet on the network; known
|
|
workers can request your approval.
|
|
|
|
4) Public server direct internet exposure
|
|
Sign-in and hardened browser transport are required. Network discovery
|
|
is off and workers join only by token.
|
|
|
|
You can change this later in Settings. Stricter floors re-apply. Nothing
|
|
already issued is revoked.
|
|
|
|
PROFILES
|
|
while :; do
|
|
ask " Choose a profile [1-4, Enter = 1 (Home lab)]: "
|
|
case "$REPLY" in
|
|
""|1|homelab) PROFILE="homelab" ;;
|
|
2|studio) PROFILE="studio" ;;
|
|
3|enterprise) PROFILE="enterprise" ;;
|
|
4|public) PROFILE="public" ;;
|
|
*) printf '%s\n' " pick 1, 2, 3 or 4" > /dev/tty; continue ;;
|
|
esac
|
|
break
|
|
done
|
|
ok "profile: $PROFILE"
|
|
fi
|
|
|
|
# Under public, a domain lets gpuk write a FILLED TLS reverse-proxy example
|
|
# (INS-47). Optional: Enter skips, and the banner still points at the shipped
|
|
# Caddyfile.example.
|
|
if [ "$PROFILE" = "public" ] && [ -z "$DOMAIN" ] && can_prompt; then
|
|
while :; do
|
|
ask " Domain for HTTPS access (e.g. gpu.example.com — Enter to skip): "
|
|
# Accept the copy-paste reflex: strip a pasted scheme and anything after
|
|
# the first slash, then insist on a bare domain rather than skipping —
|
|
# a silently dropped answer would be discovered hours later, at DNS time.
|
|
REPLY="${REPLY#https://}"
|
|
REPLY="${REPLY#http://}"
|
|
REPLY="${REPLY%%/*}"
|
|
[ -n "$REPLY" ] || break
|
|
case "$REPLY" in
|
|
*[!A-Za-z0-9.-]*)
|
|
printf '%s\n' " not a bare domain name: $REPLY. Try again, or press Enter to skip." > /dev/tty
|
|
;;
|
|
*) DOMAIN="$REPLY"; break ;;
|
|
esac
|
|
done
|
|
fi
|
|
|
|
if [ -n "$DOMAIN" ] && [ "$PROFILE" != "public" ]; then
|
|
die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)"
|
|
fi
|
|
|
|
# ── 4. The host daemon ───────────────────────────────────────────────────────
|
|
step "Host daemon — gpu-kitchen-worker"
|
|
|
|
TMP=$(mktemp -d)
|
|
# shellcheck disable=SC2064 # expand TMP now: it must be removed even if it changes
|
|
trap "rm -rf '$TMP'" EXIT INT TERM
|
|
|
|
if [ -z "$GPUK_SCRIPT" ]; then
|
|
# A checkout right here beats a download (that is how contributors run it).
|
|
_local="$(dirname "$0")/gpuk"
|
|
if [ -f "$_local" ]; then
|
|
GPUK_SCRIPT="$_local"
|
|
ok "using the gpuk installer from this checkout"
|
|
else
|
|
[ -n "${GPUK_SCRIPT_URL:-}" ] \
|
|
|| die "the release channel names no gpuk installer, and none was found locally"
|
|
curl -fsSL --max-time 60 "$GPUK_SCRIPT_URL" -o "$TMP/gpuk" \
|
|
|| die "cannot download the gpuk installer from $GPUK_SCRIPT_URL"
|
|
chmod +x "$TMP/gpuk"
|
|
GPUK_SCRIPT="$TMP/gpuk"
|
|
ok "downloaded the gpuk installer"
|
|
fi
|
|
fi
|
|
|
|
# `gpuk install` does the rest: it drops the binary, writes the systemd unit,
|
|
# writes the manifest (the declarative description of the app container) and
|
|
# applies it. Everything below is passed straight through to it.
|
|
set -- install \
|
|
--mode controller \
|
|
--image "$IMAGE" \
|
|
--cluster "$CLUSTER" \
|
|
--data-root "$DATA_ROOT" \
|
|
--cache-dir "$CACHE_DIR" \
|
|
--http-port "$PORT"
|
|
|
|
[ -z "$PROFILE" ] || set -- "$@" --profile "$PROFILE"
|
|
[ -z "$DOMAIN" ] || set -- "$@" --domain "$DOMAIN"
|
|
|
|
if [ -n "$WORKER_BINARY" ]; then
|
|
[ -f "$WORKER_BINARY" ] || die "no such worker binary: $WORKER_BINARY"
|
|
set -- "$@" --binary "$WORKER_BINARY"
|
|
ok "using a locally-built gpu-kitchen-worker"
|
|
elif [ -n "${WORKER_RELEASE_BASE:-}" ]; then
|
|
GPUK_RELEASE_BASE="$WORKER_RELEASE_BASE"
|
|
export GPUK_RELEASE_BASE
|
|
ok "gpu-kitchen-worker will be downloaded from the release"
|
|
fi
|
|
# else: gpuk reuses an already-installed binary, or fails with its own message.
|
|
|
|
step "Installing — this pulls the image, so it can take a few minutes"
|
|
sh "$GPUK_SCRIPT" "$@" || die "the install failed. See: journalctl -u gpu-kitchen-worker"
|
|
|
|
# ── 5. Wait for the app, then say where it is ────────────────────────────────
|
|
step "Waiting for the controller to answer"
|
|
|
|
i=0
|
|
until curl -fsS -o /dev/null --max-time 3 "http://127.0.0.1:$PORT/health" 2>/dev/null; do
|
|
i=$((i + 1))
|
|
[ "$i" -lt 90 ] || die "the controller did not answer on :$PORT within 90s. See: gpuk logs"
|
|
sleep 2
|
|
done
|
|
ok "the controller is up"
|
|
|
|
# The URL must name the REAL host: the person installing this is very often not
|
|
# sitting at the machine, and "localhost" would be a lie on every box but theirs
|
|
# (same reason the backend resolves its own hostname — core/host-name.ts).
|
|
HOSTNAME_FQDN=$(hostname -f 2>/dev/null || hostname 2>/dev/null || echo localhost)
|
|
LAN_IP=$(ip route get 1.1.1.1 2>/dev/null | sed -n 's/.* src \([0-9.]*\).*/\1/p' | head -1)
|
|
|
|
echo
|
|
echo "${BOLD}${GREEN}GPU Kitchen is running.${RESET}"
|
|
echo
|
|
echo " ${BOLD}Open:${RESET} http://${HOSTNAME_FQDN}:${PORT}"
|
|
[ -n "$LAN_IP" ] && echo " http://${LAN_IP}:${PORT}"
|
|
echo
|
|
CLAIM_CODE_FILE="$DATA_ROOT/secrets/claim_code"
|
|
CLAIM_CODE=""
|
|
[ ! -s "$CLAIM_CODE_FILE" ] || CLAIM_CODE=$(cat "$CLAIM_CODE_FILE" 2>/dev/null || true)
|
|
if [ -n "$CLAIM_CODE" ]; then
|
|
echo " ${BOLD}Claim code:${RESET} $CLAIM_CODE"
|
|
echo " ${BOLD}Read again:${RESET} $CLAIM_CODE_FILE (mode 0600; removed after claim)"
|
|
else
|
|
echo " ${BOLD}Claim code:${RESET} consumed (the first account already exists)"
|
|
fi
|
|
echo
|
|
|
|
case "$PROFILE" in
|
|
public)
|
|
echo " ${BOLD}Next:${RESET} create the first administrator in the UI"
|
|
echo
|
|
# The public profile REQUIRES a TLS reverse proxy (OPS-13, INS-47) — the UI
|
|
# port speaks plain HTTP. The product cannot verify the proxy's presence
|
|
# (OPS-67), so the closest thing to enforcement is saying it here, clearly.
|
|
if [ -n "$DOMAIN" ]; then
|
|
echo " ${BOLD}HTTPS:${RESET} https://${DOMAIN} answers after these steps:"
|
|
echo " 1. DNS: point ${DOMAIN} at this machine's public IP"
|
|
echo " 2. Install Caddy: https://caddyserver.com/docs/install"
|
|
echo " 3. sudo cp $DATA_ROOT/caddy/Caddyfile /etc/caddy/Caddyfile"
|
|
echo " sudo systemctl reload caddy"
|
|
echo " A filled example was written to $DATA_ROOT/caddy/Caddyfile."
|
|
echo " Until then the UI answers in cleartext on the URLs above."
|
|
else
|
|
echo " ${BOLD}HTTPS:${RESET} the public profile requires a TLS reverse proxy before any"
|
|
echo " public exposure. See deployments/controller/Caddyfile.example."
|
|
fi
|
|
echo
|
|
;;
|
|
enterprise)
|
|
echo " ${BOLD}Next:${RESET} create the first administrator in the UI"
|
|
echo
|
|
;;
|
|
homelab|studio)
|
|
echo " ${BOLD}Next:${RESET} finish first-run in the UI"
|
|
echo
|
|
;;
|
|
"")
|
|
echo " ${BOLD}Next:${RESET} choose an installation profile in the first-run assistant"
|
|
echo
|
|
;;
|
|
esac
|
|
echo " Inference endpoint : http://${HOSTNAME_FQDN}:${INFERENCE_PORT}/v1"
|
|
echo " Version : ${VERSION}"
|
|
echo
|
|
|
|
# Pre-existing model caches (B81). `gpuk install` prints the same hint, but that
|
|
# scrolls past mid-install; this banner is where people actually look. Detection
|
|
# only — cache questions belong to the first-run wizard (INS-48, REG-41), never
|
|
# to the CLI, and nothing outside $CACHE_DIR is touched, read as configuration,
|
|
# or modified.
|
|
EXISTING_CACHE=""
|
|
CHOSEN_CACHE=$(readlink -f "$CACHE_DIR" 2>/dev/null || echo "$CACHE_DIR")
|
|
for d in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do
|
|
[ -n "$d" ] && [ -d "$d/hub" ] || continue
|
|
real=$(readlink -f "$d" 2>/dev/null || echo "$d")
|
|
[ "$real" != "$CHOSEN_CACHE" ] || continue
|
|
ls -d "$d"/hub/models--* >/dev/null 2>&1 || continue
|
|
EXISTING_CACHE="$EXISTING_CACHE $real"
|
|
done
|
|
if [ -n "$EXISTING_CACHE" ]; then
|
|
for d in $EXISTING_CACHE; do
|
|
echo " ${BOLD}Existing model cache:${RESET} $d"
|
|
done
|
|
echo " GPU Kitchen will offer to reuse those models at first launch, and any"
|
|
echo " time from Nodes & GPU -> Storage. A reused cache is referenced in"
|
|
echo " place. Nothing is moved or deleted."
|
|
echo
|
|
fi
|
|
echo " Update : re-run this command, or press Update in the UI, or: gpuk update"
|
|
echo " Status : gpuk status Logs: gpuk logs Remove: gpuk uninstall"
|
|
echo
|