release v0.1.3
This commit is contained in:
+314
-32
@@ -7,15 +7,20 @@
|
||||
# channel repo is written by .gitea/workflows/release.yml, see specs/developpement/ci-cd.md.)
|
||||
#
|
||||
# What it does, and nothing more:
|
||||
# 1. preflight docker, the NVIDIA driver, and a REAL `--gpus all` smoke test
|
||||
# 1. preflight docker, the NVIDIA driver, a REAL `--gpus all` smoke test, and
|
||||
# the listening ports (INS-46 — a taken port fails HERE, not three
|
||||
# minutes later in a health-check timeout)
|
||||
# 2. resolve the current release from the channel (a TAG — never a floating
|
||||
# `latest`: an install that silently changes version under you is
|
||||
# not an install, it is a surprise)
|
||||
# 3. install `gpu-kitchen-worker` (workerd), the privileged host daemon, via
|
||||
# 3. ask the security profile — and, under `public`, a domain — when a
|
||||
# terminal is attached (INS-45); no terminal, no questions, and
|
||||
# the first-run wizard asks instead (PRF-05)
|
||||
# 4. install `gpu-kitchen-worker` (workerd), the privileged host daemon, via
|
||||
# the existing `gpuk` installer — on a controller node it OWNS the
|
||||
# app container's lifecycle
|
||||
# 4. hand over workerd pulls the pinned all-in-one controller image and starts it
|
||||
# 5. print the UI URL on the real host, and the first-run password
|
||||
# 5. hand over workerd pulls the pinned all-in-one controller image and starts it
|
||||
# 6. print the UI URL on the real host, plus the profile-specific next step
|
||||
#
|
||||
# Re-running is how you UPDATE: same command, newer tag, `docker pull` + recreate,
|
||||
# data untouched (it lives in the data root, not the container).
|
||||
@@ -39,14 +44,24 @@ CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen/channel/raw
|
||||
EDITION="${GPUK_EDITION:-community}"
|
||||
DATA_ROOT="${GPUK_DATA_ROOT:-/var/lib/gpu-kitchen}"
|
||||
CACHE_DIR="${GPUK_CACHE_DIR:-}"
|
||||
PORT="${GPUK_PORT:-8080}"
|
||||
# 1337, not 8080: the single most-squatted port in existence would make the
|
||||
# conflict preflight fire on half the lab boxes out there (INS-01).
|
||||
PORT="${GPUK_PORT:-1337}"
|
||||
# Fixed listeners: mirror of the gpuk-proxy default (core/cluster-settings.ts
|
||||
# proxyPublicPort) and of the worker mTLS channel — no install-time flag moves
|
||||
# them (INS-46).
|
||||
INFERENCE_PORT=8200
|
||||
MTLS_PORT=8443
|
||||
CLUSTER="${GPUK_CLUSTER:-default}"
|
||||
PROFILE=""
|
||||
DOMAIN=""
|
||||
VERSION=""
|
||||
IMAGE=""
|
||||
WORKER_BINARY=""
|
||||
GPUK_SCRIPT=""
|
||||
SKIP_GPU_CHECK=0
|
||||
SKIP_PREFLIGHT=0
|
||||
NON_INTERACTIVE=0
|
||||
DRY_RUN=0
|
||||
|
||||
GREEN=''; RED=''; YELLOW=''; BOLD=''; RESET=''
|
||||
@@ -60,6 +75,21 @@ warn() { echo " ${YELLOW}!${RESET} $*"; }
|
||||
step() { echo; echo "${BOLD}$*${RESET}"; }
|
||||
die() { echo " ${RED}✗${RESET} $*" >&2; exit 1; }
|
||||
|
||||
# `curl … | sudo sh` leaves stdin holding the script itself, so questions are
|
||||
# asked and answered on the controlling terminal — /dev/tty — when there is one
|
||||
# (INS-45). No terminal (CI, provisioning), --non-interactive or --dry-run keep
|
||||
# every historical flags-only behaviour.
|
||||
can_prompt() {
|
||||
[ "$NON_INTERACTIVE" -eq 0 ] || return 1
|
||||
[ "$DRY_RUN" -eq 0 ] || return 1
|
||||
(: < /dev/tty) 2>/dev/null
|
||||
}
|
||||
|
||||
ask() { # $1 = prompt → $REPLY (empty on EOF)
|
||||
printf '%s' "$1" > /dev/tty
|
||||
IFS= read -r REPLY < /dev/tty || REPLY=""
|
||||
}
|
||||
|
||||
usage() {
|
||||
cat <<EOF
|
||||
GPU Kitchen installer
|
||||
@@ -71,10 +101,15 @@ Options:
|
||||
--version <tag> Install this release instead of the channel's current one
|
||||
--image <ref> Use this controller image outright (implies --version none)
|
||||
--edition <ed> community (default) | enterprise
|
||||
--port <p> Port the UI listens on (default 8080)
|
||||
--port <p> Port the UI listens on (default 1337)
|
||||
--data-root <path> Where the database and secrets live (default /var/lib/gpu-kitchen)
|
||||
--cache-dir <path> Model cache (default <data-root>/hf)
|
||||
--cluster <name> Cluster name workers join (default "default")
|
||||
--profile <p> homelab | studio | enterprise | public (no flag + a terminal
|
||||
= the script asks; no flag + no terminal = first-run asks)
|
||||
--domain <d> Domain for the public profile: writes a filled TLS
|
||||
reverse-proxy example to <data-root>/caddy/Caddyfile
|
||||
--non-interactive Never ask anything, even with a terminal attached
|
||||
--worker-binary <p> Use a locally-built gpu-kitchen-worker instead of downloading one
|
||||
--gpuk-script <p> Use a local copy of the gpuk installer
|
||||
--skip-gpu-check Skip the 'docker run --gpus all' smoke test
|
||||
@@ -93,6 +128,9 @@ while [ $# -gt 0 ]; do
|
||||
--data-root) DATA_ROOT="$2"; shift 2 ;;
|
||||
--cache-dir) CACHE_DIR="$2"; shift 2 ;;
|
||||
--cluster) CLUSTER="$2"; shift 2 ;;
|
||||
--profile) PROFILE="$2"; shift 2 ;;
|
||||
--domain) DOMAIN="$2"; shift 2 ;;
|
||||
--non-interactive) NON_INTERACTIVE=1; shift ;;
|
||||
--worker-binary) WORKER_BINARY="$2"; shift 2 ;;
|
||||
--gpuk-script) GPUK_SCRIPT="$2"; shift 2 ;;
|
||||
--skip-gpu-check) SKIP_GPU_CHECK=1; shift ;;
|
||||
@@ -110,6 +148,15 @@ case "$EDITION" in
|
||||
*) die "--edition must be community or enterprise (got '$EDITION')" ;;
|
||||
esac
|
||||
|
||||
case "$PROFILE" in
|
||||
""|homelab|studio|enterprise|public) ;;
|
||||
*) die "--profile must be homelab, studio, enterprise or public (got '$PROFILE')" ;;
|
||||
esac
|
||||
|
||||
case "$DOMAIN" in
|
||||
*[!A-Za-z0-9.-]*) die "--domain must be a bare domain name (got '$DOMAIN')" ;;
|
||||
esac
|
||||
|
||||
echo
|
||||
echo "${BOLD}GPU Kitchen — installing the $EDITION controller${RESET}"
|
||||
|
||||
@@ -129,16 +176,16 @@ step "Preflight — docker, NVIDIA driver, container toolkit"
|
||||
[ "$DRY_RUN" -eq 1 ] || [ "$(id -u)" -eq 0 ] \
|
||||
|| die "run as root: curl -fsSL … | sudo sh"
|
||||
|
||||
command -v curl >/dev/null 2>&1 || die "curl not found — install it first"
|
||||
command -v curl >/dev/null 2>&1 || die "curl not found. Install curl first."
|
||||
|
||||
if command -v docker >/dev/null 2>&1; then
|
||||
if docker info >/dev/null 2>&1; then
|
||||
ok "docker daemon reachable ($(docker --version | cut -d' ' -f3 | tr -d ,))"
|
||||
else
|
||||
die "docker is installed but its daemon is unreachable (is it running? are you root?)"
|
||||
die "docker is installed but its daemon does not answer. Start the daemon, or run as root."
|
||||
fi
|
||||
else
|
||||
die "docker not found — install Docker Engine first: https://docs.docker.com/engine/install/"
|
||||
die "docker not found. Install Docker Engine first: https://docs.docker.com/engine/install/"
|
||||
fi
|
||||
|
||||
if command -v nvidia-smi >/dev/null 2>&1; then
|
||||
@@ -150,7 +197,7 @@ if command -v nvidia-smi >/dev/null 2>&1; then
|
||||
die "nvidia-smi is present but reports no GPU"
|
||||
fi
|
||||
else
|
||||
die "nvidia-smi not found — install the NVIDIA driver first"
|
||||
die "nvidia-smi not found. Install the NVIDIA driver first."
|
||||
fi
|
||||
|
||||
if [ "$SKIP_GPU_CHECK" -eq 1 ]; then
|
||||
@@ -163,10 +210,10 @@ elif docker info 2>/dev/null | grep -qiE 'runtimes:.*nvidia'; then
|
||||
if docker run --rm --gpus all ubuntu:24.04 nvidia-smi -L >/dev/null 2>&1; then
|
||||
ok "nvidia-container-toolkit works (a container can see the GPUs)"
|
||||
else
|
||||
die "the nvidia runtime is registered but 'docker run --gpus all' cannot see the GPUs — reinstall nvidia-container-toolkit"
|
||||
die "the nvidia runtime is registered but 'docker run --gpus all' cannot see the GPUs. Reinstall nvidia-container-toolkit."
|
||||
fi
|
||||
else
|
||||
die "nvidia-container-toolkit is not registered with docker — install it, then restart dockerd:
|
||||
die "nvidia-container-toolkit is not registered with docker. Install it, then restart dockerd:
|
||||
https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html"
|
||||
fi
|
||||
|
||||
@@ -180,7 +227,109 @@ FREE_GB=$(df -BG --output=avail "$CACHE_PARENT" 2>/dev/null | tail -1 | tr -dc '
|
||||
if [ "${FREE_GB:-0}" -ge 100 ]; then
|
||||
ok "model cache $CACHE_DIR — ${FREE_GB}G free"
|
||||
else
|
||||
warn "only ${FREE_GB:-?}G free under $CACHE_PARENT — model weights want 100G+"
|
||||
warn "only ${FREE_GB:-?}G free under $CACHE_PARENT. Model weights need 100G or more."
|
||||
fi
|
||||
|
||||
# ── Port conflicts (INS-46) ──
|
||||
# A taken port must fail HERE, before anything mutates the host — today's
|
||||
# alternative is a 3-minute health-check timeout with zero diagnosis. Best-effort
|
||||
# detection (ss, then netstat); neither present is a warn, never a false red.
|
||||
# A listener owned by an EXISTING GPU Kitchen install is not a conflict: re-running
|
||||
# this script is the documented update path, and the manifest names our port.
|
||||
# Test hook: force the detector. The netstat fallback is unreachable on any
|
||||
# host that has ss (all of them, in practice), so the CI smoke pins it here to
|
||||
# keep its parsing honest.
|
||||
PORT_TOOL="${GPUK_PORT_CHECK_TOOL:-}"
|
||||
case "$PORT_TOOL" in
|
||||
""|ss|netstat) ;;
|
||||
*) die "GPUK_PORT_CHECK_TOOL must be ss or netstat (got '$PORT_TOOL')" ;;
|
||||
esac
|
||||
if [ -z "$PORT_TOOL" ]; then
|
||||
if command -v ss >/dev/null 2>&1; then PORT_TOOL="ss"
|
||||
elif command -v netstat >/dev/null 2>&1; then PORT_TOOL="netstat"; fi
|
||||
fi
|
||||
|
||||
port_busy() { # $1 = port → 0 iff something listens on TCP :$1
|
||||
case "$PORT_TOOL" in
|
||||
ss) [ -n "$(ss -ltnH "sport = :$1" 2>/dev/null)" ] ;;
|
||||
netstat) netstat -ltn 2>/dev/null \
|
||||
| awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) found=1} END{exit found?0:1}' ;;
|
||||
*) return 1 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
port_owner() { # $1 = port → best-effort process name (needs root for -p)
|
||||
case "$PORT_TOOL" in
|
||||
ss) ss -ltnpH "sport = :$1" 2>/dev/null \
|
||||
| sed -n 's/.*users:((\"\([^"]*\)\".*/\1/p' | head -1 ;;
|
||||
netstat) netstat -ltnp 2>/dev/null \
|
||||
| awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) {print $NF; exit}}' \
|
||||
| sed 's|^[0-9]*/||' ;;
|
||||
esac
|
||||
}
|
||||
|
||||
manifest_ui_port() { # the port an existing install already owns, if any
|
||||
[ -f /etc/gpu-kitchen/manifest.json ] || return 0
|
||||
# Bridge publishes GPUK_PUBLIC_PORT over the fixed container 8080; host
|
||||
# networking moves the listener itself (GPUK_PORT). Same precedence as gpuk.
|
||||
_p=$(sed -n 's/.*"GPUK_PUBLIC_PORT"[[:space:]]*:[[:space:]]*"\([0-9]*\)".*/\1/p' \
|
||||
/etc/gpu-kitchen/manifest.json | head -1)
|
||||
[ -n "$_p" ] || _p=$(sed -n 's/.*"GPUK_PORT"[[:space:]]*:[[:space:]]*"\([0-9]*\)".*/\1/p' \
|
||||
/etc/gpu-kitchen/manifest.json | head -1)
|
||||
printf '%s' "$_p"
|
||||
}
|
||||
|
||||
if [ -z "$PORT_TOOL" ]; then
|
||||
warn "cannot check for port conflicts (neither ss nor netstat found)"
|
||||
else
|
||||
HAVE_MANIFEST=0
|
||||
[ ! -f /etc/gpu-kitchen/manifest.json ] || HAVE_MANIFEST=1
|
||||
if [ "$HAVE_MANIFEST" -eq 1 ] && [ "$(manifest_ui_port)" = "$PORT" ]; then
|
||||
ok "UI port $PORT — already ours (re-running is how you update)"
|
||||
elif port_busy "$PORT"; then
|
||||
OWNER=$(port_owner "$PORT")
|
||||
OWNER="${OWNER:-an unknown process}"
|
||||
if can_prompt; then
|
||||
ALT=$((PORT + 1))
|
||||
while port_busy "$ALT"; do ALT=$((ALT + 1)); done
|
||||
# Propose, never auto-pick: the URL printed at the end and the idempotent
|
||||
# re-run both need the operator to KNOW which port they chose.
|
||||
ask " ${YELLOW}!${RESET} port $PORT is busy ($OWNER). Use $ALT instead? [$ALT], another port, or 'q' to abort: "
|
||||
case "$REPLY" in
|
||||
q|Q) die "port $PORT is in use by $OWNER. Run the install again with --port <p>." ;;
|
||||
"") PORT="$ALT" ;;
|
||||
*)
|
||||
case "$REPLY" in
|
||||
*[!0-9]*) die "not a port number: $REPLY" ;;
|
||||
esac
|
||||
if port_busy "$REPLY"; then
|
||||
die "port $REPLY is busy too. Run the install again with --port <p>."
|
||||
fi
|
||||
PORT="$REPLY"
|
||||
;;
|
||||
esac
|
||||
ok "UI port $PORT is free"
|
||||
else
|
||||
die "port $PORT is already in use by $OWNER. Pass --port <p> to choose another port."
|
||||
fi
|
||||
else
|
||||
ok "UI port $PORT is free"
|
||||
fi
|
||||
# The mTLS and inference listeners have no install-time flag — assumed
|
||||
# limitation (INS-46): the published mTLS port moves later via
|
||||
# Settings -> Network. With a manifest present they are our own listeners.
|
||||
if [ "$HAVE_MANIFEST" -eq 0 ]; then
|
||||
if port_busy "$MTLS_PORT"; then
|
||||
OWNER=$(port_owner "$MTLS_PORT")
|
||||
die "port $MTLS_PORT (worker channel) is in use by ${OWNER:-an unknown process}. Free it first."
|
||||
fi
|
||||
if port_busy "$INFERENCE_PORT"; then
|
||||
OWNER=$(port_owner "$INFERENCE_PORT")
|
||||
die "port $INFERENCE_PORT (inference endpoint) is in use by ${OWNER:-an unknown process}. Free it first.
|
||||
($INFERENCE_PORT is also HashiCorp Vault's default port.)"
|
||||
fi
|
||||
ok "worker channel port $MTLS_PORT and inference port $INFERENCE_PORT are free"
|
||||
fi
|
||||
fi
|
||||
|
||||
fi # end preflight
|
||||
@@ -217,7 +366,7 @@ if [ -n "$IMAGE" ]; then
|
||||
else
|
||||
CHANNEL=$(curl -fsSL --max-time 20 "$CHANNEL_URL" 2>/dev/null) \
|
||||
|| die "cannot reach the release channel at $CHANNEL_URL
|
||||
(offline? pass --image <ref> to install a specific image directly)"
|
||||
If this host is offline, pass --image <ref> to install a specific image directly."
|
||||
|
||||
[ -n "$VERSION" ] || VERSION=$(echo "$CHANNEL" | json_field version)
|
||||
[ -n "$VERSION" ] || die "the release channel returned no version: $CHANNEL_URL"
|
||||
@@ -241,6 +390,24 @@ else
|
||||
[ -n "${GPUK_SCRIPT}" ] || GPUK_SCRIPT_URL=$(echo "$CHANNEL" | json_field gpukScript)
|
||||
fi
|
||||
|
||||
# A direct --image and a channel version are held to the same immutable-image
|
||||
# rule. A registry host:port is not a tag separator; only the last colon after
|
||||
# the last slash counts. Digest pins are accepted too.
|
||||
case "$IMAGE" in
|
||||
*@sha256:*) ;;
|
||||
*:latest) die "refusing floating image tag '$IMAGE'. Use an explicit release tag or digest." ;;
|
||||
*)
|
||||
IMAGE_TAG="${IMAGE##*:}"
|
||||
case "$IMAGE_TAG" in
|
||||
"$IMAGE"|*/*) die "controller image must carry an explicit tag or sha256 digest (got '$IMAGE')" ;;
|
||||
esac
|
||||
;;
|
||||
esac
|
||||
|
||||
if [ -n "$DOMAIN" ] && [ -n "$PROFILE" ] && [ "$PROFILE" != "public" ]; then
|
||||
die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)"
|
||||
fi
|
||||
|
||||
if [ "$DRY_RUN" -eq 1 ]; then
|
||||
step "Dry run — stopping here"
|
||||
if [ "$SKIP_PREFLIGHT" -eq 1 ]; then
|
||||
@@ -253,10 +420,85 @@ if [ "$DRY_RUN" -eq 1 ]; then
|
||||
echo " data root : $DATA_ROOT"
|
||||
echo " model cache : $CACHE_DIR"
|
||||
echo " UI port : $PORT"
|
||||
if [ -n "$PROFILE" ]; then
|
||||
echo " profile : $PROFILE"
|
||||
else
|
||||
echo " profile : (asked on the terminal, or chosen during first run)"
|
||||
fi
|
||||
[ -z "$DOMAIN" ] || echo " domain : $DOMAIN"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# ── 3. The host daemon ───────────────────────────────────────────────────────
|
||||
# ── 3. Resolve the installation profile (INS-45) ─────────────────────────────
|
||||
# Only ever on a terminal, and only when --profile was not given. The answer is
|
||||
# relayed verbatim as --profile: the backend stays the sole applier of profile
|
||||
# defaults and floors. Without a terminal the historical path is untouched —
|
||||
# profile unset, enterprise floors, the first-run wizard requires the choice
|
||||
# (PRF-05).
|
||||
if [ -z "$PROFILE" ] && can_prompt; then
|
||||
step "Security profile — how will this kitchen be used?"
|
||||
cat > /dev/tty <<'PROFILES'
|
||||
1) Home lab a trusted home network
|
||||
No sign-in on your home network. Nearby workers are found and join
|
||||
without waiting for approval.
|
||||
|
||||
2) Studio one control station, shared compute
|
||||
Control stays on this machine. Colleagues use API keys, while nearby
|
||||
workers wait for your approval.
|
||||
|
||||
3) Enterprise a managed company network
|
||||
Sign-in is required. The controller stays quiet on the network; known
|
||||
workers can request your approval.
|
||||
|
||||
4) Public server direct internet exposure
|
||||
Sign-in and hardened browser transport are required. Network discovery
|
||||
is off and workers join only by token.
|
||||
|
||||
You can change this later in Settings. Stricter floors re-apply. Nothing
|
||||
already issued is revoked.
|
||||
|
||||
PROFILES
|
||||
while :; do
|
||||
ask " Choose a profile [1-4, Enter = 1 (Home lab)]: "
|
||||
case "$REPLY" in
|
||||
""|1|homelab) PROFILE="homelab" ;;
|
||||
2|studio) PROFILE="studio" ;;
|
||||
3|enterprise) PROFILE="enterprise" ;;
|
||||
4|public) PROFILE="public" ;;
|
||||
*) printf '%s\n' " pick 1, 2, 3 or 4" > /dev/tty; continue ;;
|
||||
esac
|
||||
break
|
||||
done
|
||||
ok "profile: $PROFILE"
|
||||
fi
|
||||
|
||||
# Under public, a domain lets gpuk write a FILLED TLS reverse-proxy example
|
||||
# (INS-47). Optional: Enter skips, and the banner still points at the shipped
|
||||
# Caddyfile.example.
|
||||
if [ "$PROFILE" = "public" ] && [ -z "$DOMAIN" ] && can_prompt; then
|
||||
while :; do
|
||||
ask " Domain for HTTPS access (e.g. gpu.example.com — Enter to skip): "
|
||||
# Accept the copy-paste reflex: strip a pasted scheme and anything after
|
||||
# the first slash, then insist on a bare domain rather than skipping —
|
||||
# a silently dropped answer would be discovered hours later, at DNS time.
|
||||
REPLY="${REPLY#https://}"
|
||||
REPLY="${REPLY#http://}"
|
||||
REPLY="${REPLY%%/*}"
|
||||
[ -n "$REPLY" ] || break
|
||||
case "$REPLY" in
|
||||
*[!A-Za-z0-9.-]*)
|
||||
printf '%s\n' " not a bare domain name: $REPLY. Try again, or press Enter to skip." > /dev/tty
|
||||
;;
|
||||
*) DOMAIN="$REPLY"; break ;;
|
||||
esac
|
||||
done
|
||||
fi
|
||||
|
||||
if [ -n "$DOMAIN" ] && [ "$PROFILE" != "public" ]; then
|
||||
die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)"
|
||||
fi
|
||||
|
||||
# ── 4. The host daemon ───────────────────────────────────────────────────────
|
||||
step "Host daemon — gpu-kitchen-worker"
|
||||
|
||||
TMP=$(mktemp -d)
|
||||
@@ -291,6 +533,9 @@ set -- install \
|
||||
--cache-dir "$CACHE_DIR" \
|
||||
--http-port "$PORT"
|
||||
|
||||
[ -z "$PROFILE" ] || set -- "$@" --profile "$PROFILE"
|
||||
[ -z "$DOMAIN" ] || set -- "$@" --domain "$DOMAIN"
|
||||
|
||||
if [ -n "$WORKER_BINARY" ]; then
|
||||
[ -f "$WORKER_BINARY" ] || die "no such worker binary: $WORKER_BINARY"
|
||||
set -- "$@" --binary "$WORKER_BINARY"
|
||||
@@ -303,15 +548,15 @@ fi
|
||||
# else: gpuk reuses an already-installed binary, or fails with its own message.
|
||||
|
||||
step "Installing — this pulls the image, so it can take a few minutes"
|
||||
sh "$GPUK_SCRIPT" "$@" || die "the install failed — see: journalctl -u gpu-kitchen-worker"
|
||||
sh "$GPUK_SCRIPT" "$@" || die "the install failed. See: journalctl -u gpu-kitchen-worker"
|
||||
|
||||
# ── 4. Wait for the app, then say where it is ────────────────────────────────
|
||||
# ── 5. Wait for the app, then say where it is ────────────────────────────────
|
||||
step "Waiting for the controller to answer"
|
||||
|
||||
i=0
|
||||
until curl -fsS -o /dev/null --max-time 3 "http://127.0.0.1:$PORT/health" 2>/dev/null; do
|
||||
i=$((i + 1))
|
||||
[ "$i" -lt 90 ] || die "the controller did not answer on :$PORT within 90s — see: gpuk logs"
|
||||
[ "$i" -lt 90 ] || die "the controller did not answer on :$PORT within 90s. See: gpuk logs"
|
||||
sleep 2
|
||||
done
|
||||
ok "the controller is up"
|
||||
@@ -322,30 +567,66 @@ ok "the controller is up"
|
||||
HOSTNAME_FQDN=$(hostname -f 2>/dev/null || hostname 2>/dev/null || echo localhost)
|
||||
LAN_IP=$(ip route get 1.1.1.1 2>/dev/null | sed -n 's/.* src \([0-9.]*\).*/\1/p' | head -1)
|
||||
|
||||
PW_FILE="$DATA_ROOT/secrets/bootstrap_admin_password"
|
||||
ADMIN_PW=""
|
||||
[ -f "$PW_FILE" ] && ADMIN_PW=$(cat "$PW_FILE" 2>/dev/null || true)
|
||||
|
||||
echo
|
||||
echo "${BOLD}${GREEN}GPU Kitchen is running.${RESET}"
|
||||
echo
|
||||
echo " ${BOLD}Open:${RESET} http://${HOSTNAME_FQDN}:${PORT}"
|
||||
[ -n "$LAN_IP" ] && echo " http://${LAN_IP}:${PORT}"
|
||||
echo
|
||||
if [ -n "$ADMIN_PW" ]; then
|
||||
echo " ${BOLD}Sign in:${RESET} admin@local"
|
||||
echo " ${BOLD}Password:${RESET} ${ADMIN_PW}"
|
||||
echo " (the first-run wizard asks you to change it)"
|
||||
echo
|
||||
CLAIM_CODE_FILE="$DATA_ROOT/secrets/claim_code"
|
||||
CLAIM_CODE=""
|
||||
[ ! -s "$CLAIM_CODE_FILE" ] || CLAIM_CODE=$(cat "$CLAIM_CODE_FILE" 2>/dev/null || true)
|
||||
if [ -n "$CLAIM_CODE" ]; then
|
||||
echo " ${BOLD}Claim code:${RESET} $CLAIM_CODE"
|
||||
echo " ${BOLD}Read again:${RESET} $CLAIM_CODE_FILE (mode 0600; removed after claim)"
|
||||
else
|
||||
echo " ${BOLD}Claim code:${RESET} consumed (the first account already exists)"
|
||||
fi
|
||||
echo " Inference endpoint : http://${HOSTNAME_FQDN}:8200/v1"
|
||||
echo
|
||||
|
||||
case "$PROFILE" in
|
||||
public)
|
||||
echo " ${BOLD}Next:${RESET} create the first administrator in the UI"
|
||||
echo
|
||||
# The public profile REQUIRES a TLS reverse proxy (OPS-13, INS-47) — the UI
|
||||
# port speaks plain HTTP. The product cannot verify the proxy's presence
|
||||
# (OPS-67), so the closest thing to enforcement is saying it here, clearly.
|
||||
if [ -n "$DOMAIN" ]; then
|
||||
echo " ${BOLD}HTTPS:${RESET} https://${DOMAIN} answers after these steps:"
|
||||
echo " 1. DNS: point ${DOMAIN} at this machine's public IP"
|
||||
echo " 2. Install Caddy: https://caddyserver.com/docs/install"
|
||||
echo " 3. sudo cp $DATA_ROOT/caddy/Caddyfile /etc/caddy/Caddyfile"
|
||||
echo " sudo systemctl reload caddy"
|
||||
echo " A filled example was written to $DATA_ROOT/caddy/Caddyfile."
|
||||
echo " Until then the UI answers in cleartext on the URLs above."
|
||||
else
|
||||
echo " ${BOLD}HTTPS:${RESET} the public profile requires a TLS reverse proxy before any"
|
||||
echo " public exposure. See deployments/controller/Caddyfile.example."
|
||||
fi
|
||||
echo
|
||||
;;
|
||||
enterprise)
|
||||
echo " ${BOLD}Next:${RESET} create the first administrator in the UI"
|
||||
echo
|
||||
;;
|
||||
homelab|studio)
|
||||
echo " ${BOLD}Next:${RESET} finish first-run in the UI"
|
||||
echo
|
||||
;;
|
||||
"")
|
||||
echo " ${BOLD}Next:${RESET} choose an installation profile in the first-run assistant"
|
||||
echo
|
||||
;;
|
||||
esac
|
||||
echo " Inference endpoint : http://${HOSTNAME_FQDN}:${INFERENCE_PORT}/v1"
|
||||
echo " Version : ${VERSION}"
|
||||
echo
|
||||
|
||||
# Pre-existing model caches (B81). `gpuk install` prints the same hint, but that
|
||||
# scrolls past mid-install; this banner is where people actually look. Detection
|
||||
# only — the install stays non-interactive and nothing outside $CACHE_DIR is
|
||||
# touched, read as configuration, or modified.
|
||||
# only — cache questions belong to the first-run wizard (INS-48, REG-41), never
|
||||
# to the CLI, and nothing outside $CACHE_DIR is touched, read as configuration,
|
||||
# or modified.
|
||||
EXISTING_CACHE=""
|
||||
CHOSEN_CACHE=$(readlink -f "$CACHE_DIR" 2>/dev/null || echo "$CACHE_DIR")
|
||||
for d in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do
|
||||
@@ -359,8 +640,9 @@ if [ -n "$EXISTING_CACHE" ]; then
|
||||
for d in $EXISTING_CACHE; do
|
||||
echo " ${BOLD}Existing model cache:${RESET} $d"
|
||||
done
|
||||
echo " Reference it from Settings -> Cache folders to reuse those models"
|
||||
echo " without downloading again. GPU Kitchen only READS a referenced cache."
|
||||
echo " GPU Kitchen will offer to reuse those models at first launch, and any"
|
||||
echo " time from Nodes & GPU -> Storage. A reused cache is referenced in"
|
||||
echo " place. Nothing is moved or deleted."
|
||||
echo
|
||||
fi
|
||||
echo " Update : re-run this command, or press Update in the UI, or: gpuk update"
|
||||
|
||||
Reference in New Issue
Block a user