845 lines
38 KiB
Bash
845 lines
38 KiB
Bash
#!/bin/sh
|
|
# GPU Kitchen — one-command install (specs/plateforme/installation.md).
|
|
#
|
|
# curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh | sudo sh
|
|
#
|
|
# (Interim URL — becomes https://gpu.kitchen/install.sh once the hub exists; the
|
|
# channel repo is written by .gitea/workflows/release.yml, see specs/developpement/ci-cd.md.)
|
|
#
|
|
# What it does, and nothing more:
|
|
# 0. existing detect an install already on this host (manifest, or the traces
|
|
# a previous one left) and KEEP its settings — ports, data root,
|
|
# profile — unless a flag says otherwise (INS-03)
|
|
# 1. preflight docker, the NVIDIA driver, a REAL `--gpus all` smoke test, and
|
|
# the listening ports (INS-46 — a taken port is resolved HERE, on
|
|
# the terminal, not three minutes later in a health-check timeout)
|
|
# 2. resolve the current release from the channel (a TAG — never a floating
|
|
# `latest`: an install that silently changes version under you is
|
|
# not an install, it is a surprise)
|
|
# 3. ask the security profile — and, under `public`, a domain — when a
|
|
# terminal is attached (INS-45); no terminal, no questions, and
|
|
# the first-run wizard asks instead (PRF-05)
|
|
# 4. install `gpu-kitchen-worker` (workerd), the privileged host daemon, via
|
|
# the existing `gpuk` installer — on a controller node it OWNS the
|
|
# app container's lifecycle
|
|
# 5. hand over workerd pulls the pinned all-in-one controller image and starts it
|
|
# 6. print the UI URL on the real host, plus the profile-specific next step
|
|
#
|
|
# Re-running is how you UPDATE: same command, newer tag, `docker pull` + recreate,
|
|
# data untouched (it lives in the data root, not the container).
|
|
#
|
|
# This installs a CONTROLLER node (the full app + UI). A headless compute node is
|
|
# `gpuk install --mode worker …` — see deployments/install/gpuk and
|
|
# specs/plateforme/installation.md.
|
|
#
|
|
# Design note — this script starts nothing itself. workerd owns the container's
|
|
# lifecycle (create, health-gate, roll back, update); the UI's update button and
|
|
# `gpuk update` drive that same daemon. One updater, three front doors.
|
|
set -eu
|
|
|
|
# ── Defaults (every one overridable by flag or env) ───────────────────────────
|
|
# The channel is the single source of truth for "what is the current release":
|
|
# it is served next to this script, and the backend's update check reads the SAME
|
|
# document (apps/controller/api/src/core/release-channel.ts). One file, one answer. Interim
|
|
# default: the public Gitea channel repo — flips to https://gpu.kitchen/latest.json
|
|
# once the hub exists (keep the three defaults in sync, see specs/developpement/ci-cd.md).
|
|
CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen/channel/raw/branch/main/latest.json}"
|
|
EDITION="${GPUK_EDITION:-community}"
|
|
DATA_ROOT="${GPUK_DATA_ROOT:-/var/lib/gpu-kitchen}"
|
|
CACHE_DIR="${GPUK_CACHE_DIR:-}"
|
|
# 1337, not 8080: the single most-squatted port in existence would make the
|
|
# conflict preflight fire on half the lab boxes out there (INS-01).
|
|
PORT="${GPUK_PORT:-1337}"
|
|
# The worker mTLS channel and the gpuk-proxy inference endpoint (defaults mirror
|
|
# core/cluster-settings.ts). Movable at install time like the UI port (INS-46):
|
|
# 8443 is every second appliance's HTTPS alias and 8200 is HashiCorp Vault's.
|
|
MTLS_PORT="${GPUK_MTLS_PORT:-8443}"
|
|
INFERENCE_PORT="${GPUK_INFERENCE_PORT:-8200}"
|
|
CLUSTER="${GPUK_CLUSTER:-default}"
|
|
# Which of those came from the operator (flag or env) — an existing install keeps
|
|
# its own value for everything the operator did not ask to change (INS-03).
|
|
PORT_GIVEN=0; [ -z "${GPUK_PORT:-}" ] || PORT_GIVEN=1
|
|
MTLS_GIVEN=0; [ -z "${GPUK_MTLS_PORT:-}" ] || MTLS_GIVEN=1
|
|
INFERENCE_GIVEN=0; [ -z "${GPUK_INFERENCE_PORT:-}" ] || INFERENCE_GIVEN=1
|
|
DATA_ROOT_GIVEN=0; [ -z "${GPUK_DATA_ROOT:-}" ] || DATA_ROOT_GIVEN=1
|
|
CACHE_GIVEN=0; [ -z "${GPUK_CACHE_DIR:-}" ] || CACHE_GIVEN=1
|
|
CLUSTER_GIVEN=0; [ -z "${GPUK_CLUSTER:-}" ] || CLUSTER_GIVEN=1
|
|
# Where an existing install keeps its manifest. Same override as gpuk's, and for
|
|
# the same reason: it is the only way to exercise the re-run path without root.
|
|
ETC_DIR="${GPUK_ETC_DIR:-/etc/gpu-kitchen}"
|
|
MANIFEST="$ETC_DIR/manifest.json"
|
|
UNIT_DEST="${GPUK_UNIT_DEST:-/etc/systemd/system/gpu-kitchen-worker.service}"
|
|
BIN_DEST="${GPUK_BIN_DEST:-/usr/local/bin/gpu-kitchen-worker}"
|
|
PROFILE=""
|
|
DOMAIN=""
|
|
VERSION=""
|
|
IMAGE=""
|
|
WORKER_BINARY=""
|
|
GPUK_SCRIPT=""
|
|
SKIP_GPU_CHECK=0
|
|
SKIP_PREFLIGHT=0
|
|
NON_INTERACTIVE=0
|
|
DRY_RUN=0
|
|
|
|
GREEN=''; RED=''; YELLOW=''; BOLD=''; RESET=''
|
|
if [ -t 1 ]; then
|
|
GREEN=$(printf '\033[32m'); RED=$(printf '\033[31m')
|
|
YELLOW=$(printf '\033[33m'); BOLD=$(printf '\033[1m'); RESET=$(printf '\033[0m')
|
|
fi
|
|
|
|
ok() { echo " ${GREEN}✓${RESET} $*"; }
|
|
warn() { echo " ${YELLOW}!${RESET} $*"; }
|
|
step() { echo; echo "${BOLD}$*${RESET}"; }
|
|
die() { echo " ${RED}✗${RESET} $*" >&2; exit 1; }
|
|
|
|
# `curl … | sudo sh` leaves stdin holding the script itself, so questions are
|
|
# asked and answered on the controlling terminal — /dev/tty — when there is one
|
|
# (INS-45). No terminal (CI, provisioning), --non-interactive or --dry-run keep
|
|
# every historical flags-only behaviour.
|
|
can_prompt() {
|
|
[ "$NON_INTERACTIVE" -eq 0 ] || return 1
|
|
[ "$DRY_RUN" -eq 0 ] || return 1
|
|
(: < /dev/tty) 2>/dev/null
|
|
}
|
|
|
|
ask() { # $1 = prompt → $REPLY (empty on EOF)
|
|
printf '%s' "$1" > /dev/tty
|
|
IFS= read -r REPLY < /dev/tty || REPLY=""
|
|
}
|
|
|
|
usage() {
|
|
cat <<EOF
|
|
GPU Kitchen installer
|
|
|
|
curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh | sudo sh
|
|
curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh | sudo sh -s -- [options]
|
|
|
|
Options:
|
|
--version <tag> Install this release instead of the channel's current one
|
|
--image <ref> Use this controller image outright (implies --version none)
|
|
--edition <ed> community (default) | enterprise
|
|
--port <p> Port the UI listens on (default 1337)
|
|
--mtls-port <p> Port workers dial to join this controller (default 8443)
|
|
--inference-port <p> Port of the OpenAI-compatible inference endpoint (default 8200)
|
|
--data-root <path> Where the database and secrets live (default /var/lib/gpu-kitchen)
|
|
--cache-dir <path> Model cache (default <data-root>/hf)
|
|
--cluster <name> Cluster name workers join (default "default")
|
|
--profile <p> homelab | studio | enterprise | public (no flag + a terminal
|
|
= the script asks; no flag + no terminal = first-run asks)
|
|
--domain <d> Domain for the public profile: writes a filled TLS
|
|
reverse-proxy example to <data-root>/caddy/Caddyfile
|
|
--non-interactive Never ask anything, even with a terminal attached
|
|
--worker-binary <p> Use a locally-built gpu-kitchen-worker instead of downloading one
|
|
--gpuk-script <p> Use a local copy of the gpuk installer
|
|
--skip-gpu-check Skip the 'docker run --gpus all' smoke test
|
|
--skip-preflight Skip the docker, driver, GPU and disk checks (CI: no docker,
|
|
no GPU); the listening ports are still checked
|
|
--dry-run Run the preflight and resolve the release, change nothing
|
|
-h, --help This
|
|
EOF
|
|
}
|
|
|
|
while [ $# -gt 0 ]; do
|
|
case "$1" in
|
|
--version) VERSION="$2"; shift 2 ;;
|
|
--image) IMAGE="$2"; shift 2 ;;
|
|
--edition) EDITION="$2"; shift 2 ;;
|
|
--port) PORT="$2"; PORT_GIVEN=1; shift 2 ;;
|
|
--mtls-port) MTLS_PORT="$2"; MTLS_GIVEN=1; shift 2 ;;
|
|
--inference-port) INFERENCE_PORT="$2"; INFERENCE_GIVEN=1; shift 2 ;;
|
|
--data-root) DATA_ROOT="$2"; DATA_ROOT_GIVEN=1; shift 2 ;;
|
|
--cache-dir) CACHE_DIR="$2"; CACHE_GIVEN=1; shift 2 ;;
|
|
--cluster) CLUSTER="$2"; CLUSTER_GIVEN=1; shift 2 ;;
|
|
--profile) PROFILE="$2"; shift 2 ;;
|
|
--domain) DOMAIN="$2"; shift 2 ;;
|
|
--non-interactive) NON_INTERACTIVE=1; shift ;;
|
|
--worker-binary) WORKER_BINARY="$2"; shift 2 ;;
|
|
--gpuk-script) GPUK_SCRIPT="$2"; shift 2 ;;
|
|
--skip-gpu-check) SKIP_GPU_CHECK=1; shift ;;
|
|
--skip-preflight) SKIP_PREFLIGHT=1; shift ;;
|
|
--dry-run) DRY_RUN=1; shift ;;
|
|
-h|--help) usage; exit 0 ;;
|
|
*) die "unknown option: $1 (try --help)" ;;
|
|
esac
|
|
done
|
|
|
|
case "$EDITION" in
|
|
community|enterprise) ;;
|
|
*) die "--edition must be community or enterprise (got '$EDITION')" ;;
|
|
esac
|
|
|
|
case "$PROFILE" in
|
|
""|homelab|studio|enterprise|public) ;;
|
|
*) die "--profile must be homelab, studio, enterprise or public (got '$PROFILE')" ;;
|
|
esac
|
|
|
|
case "$DOMAIN" in
|
|
*[!A-Za-z0-9.-]*) die "--domain must be a bare domain name (got '$DOMAIN')" ;;
|
|
esac
|
|
|
|
for _pv in "$PORT" "$MTLS_PORT" "$INFERENCE_PORT"; do
|
|
case "$_pv" in
|
|
''|*[!0-9]*) die "not a port number: '$_pv'" ;;
|
|
esac
|
|
[ "$_pv" -ge 1 ] && [ "$_pv" -le 65535 ] || die "port out of range: $_pv"
|
|
done
|
|
|
|
echo
|
|
echo "${BOLD}GPU Kitchen — installing the $EDITION controller${RESET}"
|
|
|
|
# ── 0. An existing install (INS-03) ──────────────────────────────────────────
|
|
# Re-running this script is the update path, so before checking anything it
|
|
# reads what is already here — and KEEPS it. An update that silently moved the
|
|
# UI to another port, re-asked the profile (Enter = homelab would downgrade a
|
|
# public install) or pointed at a fresh data root beside the real one is not an
|
|
# update. Read-only. The manifest is the authority (WRK-55); without it, the
|
|
# traces a previous install leaves (container, unit, binary, data root) still
|
|
# mean "take over", never "start beside", and the container's own ports are ours.
|
|
manifest_str() { # $1 = key of a string field, anywhere in the manifest
|
|
sed -n "s/.*\"$1\"[[:space:]]*:[[:space:]]*\"\([^\"]*\)\".*/\1/p" "$MANIFEST" 2>/dev/null | head -1
|
|
}
|
|
manifest_binding() { # $1 = container port → the host port the ports table publishes it on
|
|
sed -n "s/.*\"$1\(\/tcp\)\{0,1\}\"[[:space:]]*:[[:space:]]*\([0-9][0-9]*\).*/\2/p" "$MANIFEST" 2>/dev/null | head -1
|
|
}
|
|
C_STATUS=""; C_NETMODE=""
|
|
E_PORT=""; E_MTLS_PORT=""; E_LISTEN_ADDR=""
|
|
E_PUBLIC_PORT=""; E_PUBLIC_MTLS_PORT=""; E_PROXY_PUBLIC_PORT=""
|
|
B_8080=""; B_8443=""; B_8200=""
|
|
container_facts() { # $1 = name → C_*, E_*, B_* from docker; 1 when absent
|
|
command -v docker >/dev/null 2>&1 || return 1
|
|
_f=$(docker inspect -f '{{.State.Status}}|{{.HostConfig.NetworkMode}}|{{range $p, $b := .HostConfig.PortBindings}}{{range $b}}{{$p}}={{.HostPort}} {{end}}{{end}}{{"\n"}}{{range .Config.Env}}{{.}}{{"\n"}}{{end}}' "$1" 2>/dev/null) \
|
|
|| return 1
|
|
[ -n "$_f" ] || return 1
|
|
_head=$(printf '%s\n' "$_f" | head -1)
|
|
C_STATUS=${_head%%|*}; _r=${_head#*|}; C_NETMODE=${_r%%|*}; _bind=${_r#*|}
|
|
E_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PORT=//p' | head -1)
|
|
E_MTLS_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_MTLS_PORT=//p' | head -1)
|
|
E_LISTEN_ADDR=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_LISTEN_ADDR=//p' | head -1)
|
|
E_PUBLIC_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PUBLIC_PORT=//p' | head -1)
|
|
E_PUBLIC_MTLS_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PUBLIC_MTLS_PORT=//p' | head -1)
|
|
E_PROXY_PUBLIC_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PROXY_PUBLIC_PORT=//p' | head -1)
|
|
B_8080=$(printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n 's/^8080\/tcp=//p' | head -1)
|
|
B_8443=$(printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n 's/^8443\/tcp=//p' | head -1)
|
|
B_8200=$(printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n 's/^8200\/tcp=//p' | head -1)
|
|
}
|
|
# The ports an install is REACHED on, with the precedence the controller itself
|
|
# applies (core/published-ports.ts, INS-43): host networking moves the listeners
|
|
# (GPUK_PORT, GPUK_MTLS_PORT, GPUK_LISTEN_ADDR); anything else keeps the image's
|
|
# fixed listeners and publishes them (GPUK_PUBLIC_*, then the ports table).
|
|
published_ports() { # $1 = network mode → OURS_UI OURS_MTLS OURS_INF
|
|
if [ "$1" = "host" ]; then
|
|
OURS_UI="${E_PORT:-8080}"
|
|
OURS_MTLS="${E_MTLS_PORT:-8443}"
|
|
OURS_INF="${E_PROXY_PUBLIC_PORT:-${E_LISTEN_ADDR##*:}}"
|
|
else
|
|
OURS_UI="${E_PUBLIC_PORT:-${B_8080:-8080}}"
|
|
OURS_MTLS="${E_PUBLIC_MTLS_PORT:-${B_8443:-8443}}"
|
|
OURS_INF="${E_PROXY_PUBLIC_PORT:-${B_8200:-8200}}"
|
|
fi
|
|
[ -n "$OURS_INF" ] || OURS_INF=8200
|
|
}
|
|
EXISTING=""; OUR_PORTS=""; CONTAINER_NAME="gpu-kitchen"
|
|
OURS_UI=""; OURS_MTLS=""; OURS_INF=""
|
|
if [ -f "$MANIFEST" ] && [ ! -r "$MANIFEST" ]; then
|
|
EXISTING="unreadable"
|
|
elif [ -f "$MANIFEST" ]; then
|
|
EXISTING="manifest"
|
|
_cn=$(manifest_str containerName); [ -z "$_cn" ] || CONTAINER_NAME="$_cn"
|
|
C_NETMODE=$(manifest_str networkMode)
|
|
E_PORT=$(manifest_str GPUK_PORT); E_MTLS_PORT=$(manifest_str GPUK_MTLS_PORT)
|
|
E_LISTEN_ADDR=$(manifest_str GPUK_LISTEN_ADDR)
|
|
E_PUBLIC_PORT=$(manifest_str GPUK_PUBLIC_PORT)
|
|
E_PUBLIC_MTLS_PORT=$(manifest_str GPUK_PUBLIC_MTLS_PORT)
|
|
E_PROXY_PUBLIC_PORT=$(manifest_str GPUK_PROXY_PUBLIC_PORT)
|
|
B_8080=$(manifest_binding 8080); B_8443=$(manifest_binding 8443); B_8200=$(manifest_binding 8200)
|
|
published_ports "${C_NETMODE:-host}"
|
|
OUR_PORTS="$OURS_UI $OURS_MTLS $OURS_INF"
|
|
elif container_facts "$CONTAINER_NAME"; then
|
|
EXISTING="leftovers"
|
|
published_ports "$C_NETMODE"
|
|
OUR_PORTS="$OURS_UI $OURS_MTLS $OURS_INF"
|
|
elif [ -f "$UNIT_DEST" ] || [ -x "$BIN_DEST" ] || [ -d "$DATA_ROOT/secrets" ]; then
|
|
EXISTING="leftovers"
|
|
fi
|
|
|
|
case "$EXISTING" in
|
|
manifest)
|
|
step "Existing install — $MANIFEST"
|
|
ok "image $(manifest_str image)"
|
|
_root=$(manifest_str dataRoot); _cache=$(manifest_str hostPath)
|
|
_cluster=$(manifest_str GPUK_CLUSTER); _profile=$(manifest_str GPUK_INSTALL_PROFILE)
|
|
[ "$DATA_ROOT_GIVEN" -eq 1 ] || [ -z "$_root" ] || DATA_ROOT="$_root"
|
|
[ "$CACHE_GIVEN" -eq 1 ] || [ -z "$_cache" ] || CACHE_DIR="$_cache"
|
|
[ "$CLUSTER_GIVEN" -eq 1 ] || [ -z "$_cluster" ] || CLUSTER="$_cluster"
|
|
[ "$PORT_GIVEN" -eq 1 ] || PORT="$OURS_UI"
|
|
[ "$MTLS_GIVEN" -eq 1 ] || MTLS_PORT="$OURS_MTLS"
|
|
[ "$INFERENCE_GIVEN" -eq 1 ] || INFERENCE_PORT="$OURS_INF"
|
|
case "$_profile" in
|
|
homelab|studio|enterprise|public) [ -n "$PROFILE" ] || PROFILE="$_profile" ;;
|
|
esac
|
|
ok "data root $DATA_ROOT"
|
|
ok "ports UI $OURS_UI, worker channel $OURS_MTLS, inference $OURS_INF"
|
|
[ -z "$_profile" ] || ok "profile $_profile"
|
|
ok "re-running updates it in place. Its settings are kept unless a flag says otherwise."
|
|
;;
|
|
unreadable)
|
|
step "Existing install — $MANIFEST"
|
|
warn "present, but not readable from here: run as root to keep its settings"
|
|
;;
|
|
leftovers)
|
|
step "Existing install — traces of a previous install, no manifest"
|
|
if [ -n "$C_STATUS" ]; then
|
|
ok "container $CONTAINER_NAME ($C_STATUS; UI $OURS_UI, worker channel $OURS_MTLS, inference $OURS_INF) — the install replaces it"
|
|
fi
|
|
[ ! -f "$UNIT_DEST" ] || ok "systemd unit $UNIT_DEST — rewritten"
|
|
[ ! -x "$BIN_DEST" ] || ok "daemon binary $BIN_DEST — replaced"
|
|
[ ! -d "$DATA_ROOT/secrets" ] || ok "data root $DATA_ROOT — reused, nothing in it is touched"
|
|
warn "without $MANIFEST no setting can be kept: the flags and the defaults apply"
|
|
;;
|
|
esac
|
|
[ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf"
|
|
|
|
if [ "$PORT" = "$MTLS_PORT" ] || [ "$PORT" = "$INFERENCE_PORT" ] || [ "$MTLS_PORT" = "$INFERENCE_PORT" ]; then
|
|
die "the UI, worker channel and inference ports must differ (got $PORT, $MTLS_PORT, $INFERENCE_PORT)"
|
|
fi
|
|
|
|
# ── 1. Preflight ─────────────────────────────────────────────────────────────
|
|
# The same checks tools/provision-feeder.sh makes, minus the compose ones: the
|
|
# all-in-one image is driven by workerd through the plain docker CLI, so there is
|
|
# no compose dependency to satisfy any more.
|
|
if [ "$SKIP_PREFLIGHT" -eq 1 ]; then
|
|
|
|
step "Preflight — skipped (--skip-preflight)"
|
|
warn "the host is NOT being checked for docker, a driver or a GPU"
|
|
|
|
else
|
|
|
|
step "Preflight — docker, NVIDIA driver, container toolkit"
|
|
|
|
# Root, or a user-owned prefix (GPUK_ETC_DIR) — the same rule as gpuk's
|
|
# need_root, and the only way the full path is testable without handing root
|
|
# to a test suite.
|
|
[ "$DRY_RUN" -eq 1 ] || [ "$(id -u)" -eq 0 ] || [ -w "$ETC_DIR" ] \
|
|
|| die "run as root: curl -fsSL … | sudo sh"
|
|
|
|
command -v curl >/dev/null 2>&1 || die "curl not found. Install curl first."
|
|
|
|
if command -v docker >/dev/null 2>&1; then
|
|
if docker info >/dev/null 2>&1; then
|
|
ok "docker daemon reachable ($(docker --version | cut -d' ' -f3 | tr -d ,))"
|
|
else
|
|
die "docker is installed but its daemon does not answer. Start the daemon, or run as root."
|
|
fi
|
|
else
|
|
die "docker not found. Install Docker Engine first: https://docs.docker.com/engine/install/"
|
|
fi
|
|
|
|
if command -v nvidia-smi >/dev/null 2>&1; then
|
|
DRIVER=$(nvidia-smi --query-gpu=driver_version --format=csv,noheader 2>/dev/null | head -1 || true)
|
|
GPU_COUNT=$(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | grep -c . || true)
|
|
if [ -n "$DRIVER" ] && [ "${GPU_COUNT:-0}" -gt 0 ]; then
|
|
ok "NVIDIA driver $DRIVER — ${GPU_COUNT} GPU(s): $(nvidia-smi --query-gpu=name --format=csv,noheader | sort -u | paste -sd', ')"
|
|
else
|
|
die "nvidia-smi is present but reports no GPU"
|
|
fi
|
|
else
|
|
die "nvidia-smi not found. Install the NVIDIA driver first."
|
|
fi
|
|
|
|
if [ "$SKIP_GPU_CHECK" -eq 1 ]; then
|
|
warn "nvidia-container-toolkit check skipped (--skip-gpu-check)"
|
|
elif docker info 2>/dev/null | grep -qiE 'runtimes:.*nvidia'; then
|
|
# The runtime being REGISTERED is not the same as it working. With the toolkit
|
|
# installed, --gpus injects the driver and nvidia-smi into a plain image; that
|
|
# is the exact mechanism the controller container relies on, so test it rather
|
|
# than infer it.
|
|
if docker run --rm --gpus all ubuntu:24.04 nvidia-smi -L >/dev/null 2>&1; then
|
|
ok "nvidia-container-toolkit works (a container can see the GPUs)"
|
|
else
|
|
die "the nvidia runtime is registered but 'docker run --gpus all' cannot see the GPUs. Reinstall nvidia-container-toolkit."
|
|
fi
|
|
else
|
|
die "nvidia-container-toolkit is not registered with docker. Install it, then restart dockerd:
|
|
https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html"
|
|
fi
|
|
|
|
# Model weights are large and the failure mode (a download dying at 90%) is
|
|
# miserable, so say so up front. A warning, not a refusal: it is the user's disk.
|
|
CACHE_PARENT="$CACHE_DIR"
|
|
while [ ! -d "$CACHE_PARENT" ] && [ "$CACHE_PARENT" != "/" ]; do
|
|
CACHE_PARENT=$(dirname "$CACHE_PARENT")
|
|
done
|
|
FREE_GB=$(df -BG --output=avail "$CACHE_PARENT" 2>/dev/null | tail -1 | tr -dc '0-9' || true)
|
|
if [ "${FREE_GB:-0}" -ge 100 ]; then
|
|
ok "model cache $CACHE_DIR — ${FREE_GB}G free"
|
|
else
|
|
warn "only ${FREE_GB:-?}G free under $CACHE_PARENT. Model weights need 100G or more."
|
|
fi
|
|
|
|
fi # end preflight
|
|
|
|
# ── Listening ports (INS-46) ──────────────────────────────────────────────────
|
|
# Deliberately OUTSIDE the preflight branch: --skip-preflight skips docker, the
|
|
# driver, the GPU smoke test and the disk (things a runner or a VM cannot have),
|
|
# but a taken port is exactly as fatal there, and checking it costs nothing.
|
|
# Skipping it here only moved the failure to gpuk's non-interactive refusal.
|
|
if [ "$SKIP_PREFLIGHT" -eq 1 ]; then
|
|
step "Listening ports — checked even without the preflight"
|
|
fi
|
|
# A taken port must fail HERE, before anything mutates the host — today's
|
|
# alternative is a 3-minute health-check timeout with zero diagnosis. Best-effort
|
|
# detection (ss, then netstat); neither present is a warn, never a false red.
|
|
# A listener owned by an EXISTING GPU Kitchen install is not a conflict: re-running
|
|
# this script is the documented update path, and the manifest names our port.
|
|
# Test hook: force the detector. The netstat fallback is unreachable on any
|
|
# host that has ss (all of them, in practice), so the CI smoke pins it here to
|
|
# keep its parsing honest.
|
|
PORT_TOOL="${GPUK_PORT_CHECK_TOOL:-}"
|
|
case "$PORT_TOOL" in
|
|
""|ss|netstat) ;;
|
|
*) die "GPUK_PORT_CHECK_TOOL must be ss or netstat (got '$PORT_TOOL')" ;;
|
|
esac
|
|
if [ -z "$PORT_TOOL" ]; then
|
|
if command -v ss >/dev/null 2>&1; then PORT_TOOL="ss"
|
|
elif command -v netstat >/dev/null 2>&1; then PORT_TOOL="netstat"; fi
|
|
fi
|
|
|
|
port_busy() { # $1 = port → 0 iff something listens on TCP :$1
|
|
case "$PORT_TOOL" in
|
|
ss) [ -n "$(ss -ltnH "sport = :$1" 2>/dev/null)" ] ;;
|
|
netstat) netstat -ltn 2>/dev/null \
|
|
| awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) found=1} END{exit found?0:1}' ;;
|
|
*) return 1 ;;
|
|
esac
|
|
}
|
|
|
|
# Which container a listener belongs to, if any: the process's cgroup names the
|
|
# container id (host networking — the listener IS the container's process), and a
|
|
# bridged publication shows up as the container's port mapping in `docker ps`
|
|
# (the host-side holder is docker-proxy, which says nothing by itself). "nginx"
|
|
# is a riddle; "nginx in container gpu-kitchen-dev" is the answer.
|
|
port_container() { # $1 = port, $2 = pid ("" if unknown) → container name or ""
|
|
command -v docker >/dev/null 2>&1 || return 0
|
|
if [ -n "$2" ] && [ -r "/proc/$2/cgroup" ]; then
|
|
_cid=$(sed -n 's#.*docker[-/]\([0-9a-f]\{64\}\).*#\1#p' "/proc/$2/cgroup" 2>/dev/null | head -1)
|
|
if [ -n "$_cid" ]; then
|
|
docker inspect -f '{{.Name}}' "$_cid" 2>/dev/null | sed 's|^/||'
|
|
return 0
|
|
fi
|
|
fi
|
|
docker ps --format '{{.Names}} {{.Ports}}' 2>/dev/null \
|
|
| awk -v p=":$1->" 'index($0, p) { print $1; exit }'
|
|
}
|
|
|
|
port_owner() { # $1 = port → best-effort "process", "process in container NAME", or ""
|
|
_proc=""; _pid=""
|
|
case "$PORT_TOOL" in
|
|
ss)
|
|
_line=$(ss -ltnpH "sport = :$1" 2>/dev/null | head -1)
|
|
_proc=$(printf '%s' "$_line" | sed -n 's/.*users:((\"\([^"]*\)\".*/\1/p')
|
|
_pid=$(printf '%s' "$_line" | sed -n 's/.*pid=\([0-9]*\).*/\1/p')
|
|
;;
|
|
netstat)
|
|
_field=$(netstat -ltnp 2>/dev/null \
|
|
| awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) {print $NF; exit}}')
|
|
case "$_field" in
|
|
*/*) _pid=${_field%%/*}; _proc=${_field#*/} ;;
|
|
*) _proc="$_field" ;;
|
|
esac
|
|
;;
|
|
esac
|
|
case "$_pid" in *[!0-9]*|"") _pid="" ;; esac
|
|
_ctr=$(port_container "$1" "$_pid")
|
|
if [ -n "$_ctr" ]; then
|
|
printf '%s' "${_proc:-a process} in container $_ctr"
|
|
else
|
|
printf '%s' "$_proc"
|
|
fi
|
|
}
|
|
|
|
# A port is ours when the existing install (step 0) already holds it: the
|
|
# re-run replaces that container, so what it listens on is not a conflict.
|
|
port_is_ours() { case " $OUR_PORTS " in *" $1 "*) return 0 ;; esac; return 1; }
|
|
|
|
# Ports this run may not hand out twice: the three requested ones, plus every
|
|
# alternative already accepted. Without it, a busy 8442 would be offered 8443
|
|
# and collide with the worker channel one question later.
|
|
RESERVED_PORTS="$PORT $MTLS_PORT $INFERENCE_PORT"
|
|
port_available() { # $1 → free on the host AND not reserved by this run
|
|
case " $RESERVED_PORTS " in *" $1 "*) return 1 ;; esac
|
|
port_is_ours "$1" && return 1
|
|
! port_busy "$1"
|
|
}
|
|
|
|
# resolve_port <label> <port> <flag> → RESOLVED. Same rules for all three
|
|
# listeners: ours = fine; busy on a terminal = propose the next free port
|
|
# (Enter accepts, a number picks, q aborts) — never auto-pick silently, the URL
|
|
# printed at the end and the idempotent re-run both need the operator to KNOW
|
|
# the port; busy without a terminal = fail now, naming the process and the flag.
|
|
resolve_port() {
|
|
_label="$1"; _want="$2"; _flag="$3"
|
|
if port_is_ours "$_want"; then
|
|
ok "$_label port $_want — already ours (re-running is how you update)"
|
|
elif port_busy "$_want"; then
|
|
OWNER=$(port_owner "$_want")
|
|
OWNER="${OWNER:-an unknown process}"
|
|
if can_prompt; then
|
|
ALT=$((_want + 1))
|
|
while ! port_available "$ALT"; do ALT=$((ALT + 1)); done
|
|
ask " ${YELLOW}!${RESET} $_label port $_want is busy ($OWNER). Use $ALT instead? [$ALT], another port, or 'q' to abort: "
|
|
case "$REPLY" in
|
|
q|Q) die "$_label port $_want is in use by $OWNER. Run the install again with $_flag <p>." ;;
|
|
"") _want="$ALT" ;;
|
|
*)
|
|
case "$REPLY" in
|
|
*[!0-9]*) die "not a port number: $REPLY" ;;
|
|
esac
|
|
port_available "$REPLY" \
|
|
|| die "port $REPLY is busy, or already taken by another GPU Kitchen listener. Run the install again with $_flag <p>."
|
|
_want="$REPLY"
|
|
;;
|
|
esac
|
|
RESERVED_PORTS="$RESERVED_PORTS $_want"
|
|
ok "$_label port $_want is free"
|
|
else
|
|
die "$_label port $_want is already in use by $OWNER. Pass $_flag <p> to choose another port."
|
|
fi
|
|
else
|
|
ok "$_label port $_want is free"
|
|
fi
|
|
RESOLVED="$_want"
|
|
}
|
|
|
|
if [ -z "$PORT_TOOL" ]; then
|
|
warn "cannot check for port conflicts (neither ss nor netstat found)"
|
|
else
|
|
resolve_port "UI" "$PORT" "--port"; PORT="$RESOLVED"
|
|
resolve_port "worker channel" "$MTLS_PORT" "--mtls-port"; MTLS_PORT="$RESOLVED"
|
|
resolve_port "inference" "$INFERENCE_PORT" "--inference-port"; INFERENCE_PORT="$RESOLVED"
|
|
fi
|
|
|
|
# ── 2. Resolve the release ───────────────────────────────────────────────────
|
|
step "Release — resolving the version to install"
|
|
|
|
# One tiny JSON document, fetched over TLS, holding what the current release IS.
|
|
# Parsed with sed rather than jq: `curl … | sudo sh` cannot assume jq exists, and
|
|
# the document is ours and flat.
|
|
json_field() { sed -n "s/.*\"$1\"[[:space:]]*:[[:space:]]*\"\([^\"]*\)\".*/\1/p" | head -1; }
|
|
|
|
# Strip a tag off an image reference WITHOUT mangling a registry's host:port.
|
|
# `${ref%%:*}` cuts at the FIRST colon and is WRONG: a
|
|
# `registry.internal:5000/gpuk/controller:v1.2.3` (or an untagged
|
|
# `registry.internal:5000/gpuk/controller`) would collapse to `registry.internal`.
|
|
# A colon is a tag separator only when the last colon comes AFTER the last slash.
|
|
# Same rule as deployments/install/gpuk's image_repo() and core/release-channel.ts (unit-tested).
|
|
image_repo() {
|
|
case "$1" in
|
|
*@*) printf '%s' "${1%@*}"; return ;; # digest pin → repo@sha256:…
|
|
esac
|
|
_t="${1##*:}"
|
|
case "$_t" in
|
|
"$1") printf '%s' "$1" ;; # no colon at all → already untagged
|
|
*/*) printf '%s' "$1" ;; # last colon is inside a path → host:port, untagged
|
|
*) printf '%s' "${1%:*}" ;;
|
|
esac
|
|
}
|
|
|
|
if [ -n "$IMAGE" ]; then
|
|
ok "using the image given on the command line: $IMAGE"
|
|
[ -n "$VERSION" ] || VERSION="(pinned by --image)"
|
|
else
|
|
CHANNEL=$(curl -fsSL --max-time 20 "$CHANNEL_URL" 2>/dev/null) \
|
|
|| die "cannot reach the release channel at $CHANNEL_URL
|
|
If this host is offline, pass --image <ref> to install a specific image directly."
|
|
|
|
[ -n "$VERSION" ] || VERSION=$(echo "$CHANNEL" | json_field version)
|
|
[ -n "$VERSION" ] || die "the release channel returned no version: $CHANNEL_URL"
|
|
|
|
if [ "$EDITION" = "enterprise" ]; then
|
|
IMAGE_TEMPLATE=$(echo "$CHANNEL" | json_field controllerImageEnterprise)
|
|
else
|
|
IMAGE_TEMPLATE=$(echo "$CHANNEL" | json_field controllerImage)
|
|
fi
|
|
[ -n "$IMAGE_TEMPLATE" ] \
|
|
|| die "the release channel names no $EDITION controller image: $CHANNEL_URL"
|
|
|
|
# The channel gives the repository; WE pin the tag. A floating `:latest` would
|
|
# make every container recreate a silent, unrequested upgrade. Strip any tag the
|
|
# channel already carries with image_repo (host:port-safe), then pin OUR version.
|
|
IMAGE="$(image_repo "$IMAGE_TEMPLATE"):${VERSION}"
|
|
ok "release $VERSION"
|
|
ok "image $IMAGE"
|
|
|
|
[ -n "$WORKER_BINARY" ] || WORKER_RELEASE_BASE=$(echo "$CHANNEL" | json_field workerBase)
|
|
[ -n "${GPUK_SCRIPT}" ] || GPUK_SCRIPT_URL=$(echo "$CHANNEL" | json_field gpukScript)
|
|
fi
|
|
|
|
# A direct --image and a channel version are held to the same immutable-image
|
|
# rule. A registry host:port is not a tag separator; only the last colon after
|
|
# the last slash counts. Digest pins are accepted too.
|
|
case "$IMAGE" in
|
|
*@sha256:*) ;;
|
|
*:latest) die "refusing floating image tag '$IMAGE'. Use an explicit release tag or digest." ;;
|
|
*)
|
|
IMAGE_TAG="${IMAGE##*:}"
|
|
case "$IMAGE_TAG" in
|
|
"$IMAGE"|*/*) die "controller image must carry an explicit tag or sha256 digest (got '$IMAGE')" ;;
|
|
esac
|
|
;;
|
|
esac
|
|
|
|
if [ -n "$DOMAIN" ] && [ -n "$PROFILE" ] && [ "$PROFILE" != "public" ]; then
|
|
die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)"
|
|
fi
|
|
|
|
if [ "$DRY_RUN" -eq 1 ]; then
|
|
step "Dry run — stopping here"
|
|
if [ "$SKIP_PREFLIGHT" -eq 1 ]; then
|
|
ok "the release resolved; the host was not checked; nothing was installed"
|
|
else
|
|
ok "preflight passed and the release resolved; nothing was installed"
|
|
fi
|
|
echo
|
|
echo " would install : $IMAGE"
|
|
echo " data root : $DATA_ROOT"
|
|
echo " model cache : $CACHE_DIR"
|
|
echo " UI port : $PORT"
|
|
echo " worker channel: $MTLS_PORT"
|
|
echo " inference : $INFERENCE_PORT"
|
|
case "$EXISTING" in
|
|
manifest) echo " existing : yes — updated in place" ;;
|
|
leftovers) echo " existing : traces of a previous install — taken over" ;;
|
|
*) echo " existing : no" ;;
|
|
esac
|
|
if [ -n "$PROFILE" ]; then
|
|
echo " profile : $PROFILE"
|
|
else
|
|
echo " profile : (asked on the terminal, or chosen during first run)"
|
|
fi
|
|
[ -z "$DOMAIN" ] || echo " domain : $DOMAIN"
|
|
exit 0
|
|
fi
|
|
|
|
# ── 3. Resolve the installation profile (INS-45) ─────────────────────────────
|
|
# Only ever on a terminal, and only when --profile was not given. The answer is
|
|
# relayed verbatim as --profile: the backend stays the sole applier of profile
|
|
# defaults and floors. Without a terminal the historical path is untouched —
|
|
# profile unset, enterprise floors, the first-run wizard requires the choice
|
|
# (PRF-05).
|
|
if [ -z "$PROFILE" ] && can_prompt; then
|
|
step "Security profile — how will this kitchen be used?"
|
|
cat > /dev/tty <<'PROFILES'
|
|
1) Home lab a trusted home network
|
|
No sign-in on your home network. Nearby workers are found and join
|
|
without waiting for approval.
|
|
|
|
2) Studio one control station, shared compute
|
|
Control stays on this machine. Colleagues use API keys, while nearby
|
|
workers wait for your approval.
|
|
|
|
3) Enterprise a managed company network
|
|
Sign-in is required. The controller stays quiet on the network; known
|
|
workers can request your approval.
|
|
|
|
4) Public server direct internet exposure
|
|
Sign-in and hardened browser transport are required. Network discovery
|
|
is off and workers join only by token.
|
|
|
|
You can change this later in Settings. Stricter floors re-apply. Nothing
|
|
already issued is revoked.
|
|
|
|
PROFILES
|
|
while :; do
|
|
ask " Choose a profile [1-4, Enter = 1 (Home lab)]: "
|
|
case "$REPLY" in
|
|
""|1|homelab) PROFILE="homelab" ;;
|
|
2|studio) PROFILE="studio" ;;
|
|
3|enterprise) PROFILE="enterprise" ;;
|
|
4|public) PROFILE="public" ;;
|
|
*) printf '%s\n' " pick 1, 2, 3 or 4" > /dev/tty; continue ;;
|
|
esac
|
|
break
|
|
done
|
|
ok "profile: $PROFILE"
|
|
fi
|
|
|
|
# Under public, a domain lets gpuk write a FILLED TLS reverse-proxy example
|
|
# (INS-47). Optional: Enter skips, and the banner still points at the shipped
|
|
# Caddyfile.example.
|
|
if [ "$PROFILE" = "public" ] && [ -z "$DOMAIN" ] && can_prompt; then
|
|
while :; do
|
|
ask " Domain for HTTPS access (e.g. gpu.example.com — Enter to skip): "
|
|
# Accept the copy-paste reflex: strip a pasted scheme and anything after
|
|
# the first slash, then insist on a bare domain rather than skipping —
|
|
# a silently dropped answer would be discovered hours later, at DNS time.
|
|
REPLY="${REPLY#https://}"
|
|
REPLY="${REPLY#http://}"
|
|
REPLY="${REPLY%%/*}"
|
|
[ -n "$REPLY" ] || break
|
|
case "$REPLY" in
|
|
*[!A-Za-z0-9.-]*)
|
|
printf '%s\n' " not a bare domain name: $REPLY. Try again, or press Enter to skip." > /dev/tty
|
|
;;
|
|
*) DOMAIN="$REPLY"; break ;;
|
|
esac
|
|
done
|
|
fi
|
|
|
|
if [ -n "$DOMAIN" ] && [ "$PROFILE" != "public" ]; then
|
|
die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)"
|
|
fi
|
|
|
|
# ── 4. The host daemon ───────────────────────────────────────────────────────
|
|
step "Host daemon — gpu-kitchen-worker"
|
|
|
|
TMP=$(mktemp -d)
|
|
# shellcheck disable=SC2064 # expand TMP now: it must be removed even if it changes
|
|
trap "rm -rf '$TMP'" EXIT INT TERM
|
|
|
|
if [ -z "$GPUK_SCRIPT" ]; then
|
|
# A checkout right here beats a download (that is how contributors run it).
|
|
_local="$(dirname "$0")/gpuk"
|
|
if [ -f "$_local" ]; then
|
|
GPUK_SCRIPT="$_local"
|
|
ok "using the gpuk installer from this checkout"
|
|
else
|
|
[ -n "${GPUK_SCRIPT_URL:-}" ] \
|
|
|| die "the release channel names no gpuk installer, and none was found locally"
|
|
curl -fsSL --max-time 60 "$GPUK_SCRIPT_URL" -o "$TMP/gpuk" \
|
|
|| die "cannot download the gpuk installer from $GPUK_SCRIPT_URL"
|
|
chmod +x "$TMP/gpuk"
|
|
GPUK_SCRIPT="$TMP/gpuk"
|
|
ok "downloaded the gpuk installer"
|
|
fi
|
|
fi
|
|
|
|
# `gpuk install` does the rest: it drops the binary, writes the systemd unit,
|
|
# writes the manifest (the declarative description of the app container) and
|
|
# applies it. Everything below is passed straight through to it.
|
|
set -- install \
|
|
--mode controller \
|
|
--image "$IMAGE" \
|
|
--cluster "$CLUSTER" \
|
|
--data-root "$DATA_ROOT" \
|
|
--cache-dir "$CACHE_DIR" \
|
|
--http-port "$PORT" \
|
|
--mtls-port "$MTLS_PORT" \
|
|
--inference-port "$INFERENCE_PORT"
|
|
|
|
[ -z "$PROFILE" ] || set -- "$@" --profile "$PROFILE"
|
|
[ -z "$DOMAIN" ] || set -- "$@" --domain "$DOMAIN"
|
|
|
|
if [ -n "$WORKER_BINARY" ]; then
|
|
[ -f "$WORKER_BINARY" ] || die "no such worker binary: $WORKER_BINARY"
|
|
set -- "$@" --binary "$WORKER_BINARY"
|
|
ok "using a locally-built gpu-kitchen-worker"
|
|
elif [ -n "${WORKER_RELEASE_BASE:-}" ]; then
|
|
GPUK_RELEASE_BASE="$WORKER_RELEASE_BASE"
|
|
export GPUK_RELEASE_BASE
|
|
ok "gpu-kitchen-worker will be downloaded from the release"
|
|
fi
|
|
# else: gpuk reuses an already-installed binary, or fails with its own message.
|
|
|
|
step "Installing — this pulls the image, so it can take a few minutes"
|
|
sh "$GPUK_SCRIPT" "$@" || die "the install failed. See: journalctl -u gpu-kitchen-worker"
|
|
|
|
# ── 5. Wait for the app, then say where it is ────────────────────────────────
|
|
step "Waiting for the controller to answer"
|
|
|
|
i=0
|
|
until curl -fsS -o /dev/null --max-time 3 "http://127.0.0.1:$PORT/health" 2>/dev/null; do
|
|
i=$((i + 1))
|
|
[ "$i" -lt 90 ] || die "the controller did not answer on :$PORT within 90s. See: gpuk logs"
|
|
sleep 2
|
|
done
|
|
ok "the controller is up"
|
|
|
|
# The URL must name the REAL host: the person installing this is very often not
|
|
# sitting at the machine, and "localhost" would be a lie on every box but theirs
|
|
# (same reason the backend resolves its own hostname — core/host-name.ts).
|
|
HOSTNAME_FQDN=$(hostname -f 2>/dev/null || hostname 2>/dev/null || echo localhost)
|
|
LAN_IP=$(ip route get 1.1.1.1 2>/dev/null | sed -n 's/.* src \([0-9.]*\).*/\1/p' | head -1)
|
|
|
|
echo
|
|
echo "${BOLD}${GREEN}GPU Kitchen is running.${RESET}"
|
|
echo
|
|
echo " ${BOLD}Open:${RESET} http://${HOSTNAME_FQDN}:${PORT}"
|
|
[ -n "$LAN_IP" ] && echo " http://${LAN_IP}:${PORT}"
|
|
echo
|
|
CLAIM_CODE_FILE="$DATA_ROOT/secrets/claim_code"
|
|
CLAIM_CODE=""
|
|
[ ! -s "$CLAIM_CODE_FILE" ] || CLAIM_CODE=$(cat "$CLAIM_CODE_FILE" 2>/dev/null || true)
|
|
if [ -n "$CLAIM_CODE" ]; then
|
|
echo " ${BOLD}Claim code:${RESET} $CLAIM_CODE"
|
|
echo " ${BOLD}Read again:${RESET} $CLAIM_CODE_FILE (mode 0600; removed after claim)"
|
|
else
|
|
echo " ${BOLD}Claim code:${RESET} consumed (the first account already exists)"
|
|
fi
|
|
echo
|
|
|
|
case "$PROFILE" in
|
|
public)
|
|
echo " ${BOLD}Next:${RESET} create the first administrator in the UI"
|
|
echo
|
|
# The public profile REQUIRES a TLS reverse proxy (OPS-13, INS-47) — the UI
|
|
# port speaks plain HTTP. The product cannot verify the proxy's presence
|
|
# (OPS-67), so the closest thing to enforcement is saying it here, clearly.
|
|
if [ -n "$DOMAIN" ]; then
|
|
echo " ${BOLD}HTTPS:${RESET} https://${DOMAIN} answers after these steps:"
|
|
echo " 1. DNS: point ${DOMAIN} at this machine's public IP"
|
|
echo " 2. Install Caddy: https://caddyserver.com/docs/install"
|
|
echo " 3. sudo cp $DATA_ROOT/caddy/Caddyfile /etc/caddy/Caddyfile"
|
|
echo " sudo systemctl reload caddy"
|
|
echo " A filled example was written to $DATA_ROOT/caddy/Caddyfile."
|
|
echo " Until then the UI answers in cleartext on the URLs above."
|
|
else
|
|
echo " ${BOLD}HTTPS:${RESET} the public profile requires a TLS reverse proxy before any"
|
|
echo " public exposure. See deployments/controller/Caddyfile.example."
|
|
fi
|
|
echo
|
|
;;
|
|
enterprise)
|
|
echo " ${BOLD}Next:${RESET} create the first administrator in the UI"
|
|
echo
|
|
;;
|
|
homelab|studio)
|
|
echo " ${BOLD}Next:${RESET} finish first-run in the UI"
|
|
echo
|
|
;;
|
|
"")
|
|
echo " ${BOLD}Next:${RESET} choose an installation profile in the first-run assistant"
|
|
echo
|
|
;;
|
|
esac
|
|
echo " Inference endpoint : http://${HOSTNAME_FQDN}:${INFERENCE_PORT}/v1"
|
|
echo " Version : ${VERSION}"
|
|
echo
|
|
|
|
# Pre-existing model caches (B81). `gpuk install` prints the same hint, but that
|
|
# scrolls past mid-install; this banner is where people actually look. Detection
|
|
# only — cache questions belong to the first-run wizard (INS-48, REG-41), never
|
|
# to the CLI, and nothing outside $CACHE_DIR is touched, read as configuration,
|
|
# or modified.
|
|
EXISTING_CACHE=""
|
|
CHOSEN_CACHE=$(readlink -f "$CACHE_DIR" 2>/dev/null || echo "$CACHE_DIR")
|
|
for d in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do
|
|
[ -n "$d" ] && [ -d "$d/hub" ] || continue
|
|
real=$(readlink -f "$d" 2>/dev/null || echo "$d")
|
|
[ "$real" != "$CHOSEN_CACHE" ] || continue
|
|
ls -d "$d"/hub/models--* >/dev/null 2>&1 || continue
|
|
EXISTING_CACHE="$EXISTING_CACHE $real"
|
|
done
|
|
if [ -n "$EXISTING_CACHE" ]; then
|
|
for d in $EXISTING_CACHE; do
|
|
echo " ${BOLD}Existing model cache:${RESET} $d"
|
|
done
|
|
echo " GPU Kitchen will offer to reuse those models at first launch, and any"
|
|
echo " time from Nodes & GPU -> Storage. A reused cache is referenced in"
|
|
echo " place. Nothing is moved or deleted."
|
|
echo
|
|
fi
|
|
echo " Update : re-run this command, or press Update in the UI, or: gpuk update"
|
|
echo " Status : gpuk status Logs: gpuk logs"
|
|
echo " Remove : gpuk uninstall (service only) or gpuk uninstall --purge (all but the data root)"
|
|
echo
|