#!/bin/sh # GPU Kitchen — one-command install (specs/plateforme/installation.md). # # curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh | sudo sh # # (Interim URL — becomes https://gpu.kitchen/install.sh once the hub exists; the # channel repo is written by .gitea/workflows/release.yml, see specs/developpement/ci-cd.md.) # # What it does, and nothing more: # 0. existing detect an install already on this host (manifest, or the traces # a previous one left) and KEEP its settings — ports, data root, # profile — unless a flag says otherwise (INS-03) # 1. preflight docker, the NVIDIA driver, a REAL `--gpus all` smoke test, and # the listening ports (INS-46 — a taken port is resolved HERE, on # the terminal, not three minutes later in a health-check timeout) # 2. resolve the current release from the channel (a TAG — never a floating # `latest`: an install that silently changes version under you is # not an install, it is a surprise) # 3. ask the security profile — and, under `public`, a domain — when a # terminal is attached (INS-45); no terminal, no questions, and # the first-run wizard asks instead (PRF-05) # 4. install `gpu-kitchen-worker` (workerd), the privileged host daemon, via # the existing `gpuk` installer — on a controller node it OWNS the # app container's lifecycle # 5. hand over workerd pulls the pinned all-in-one controller image and starts it # 6. print the UI URL on the real host, plus the profile-specific next step # # Re-running is how you UPDATE: same command, newer tag, `docker pull` + recreate, # data untouched (it lives in the data root, not the container). # # This installs a CONTROLLER node (the full app + UI). A headless compute node is # `gpuk install --mode worker …` — see deployments/install/gpuk and # specs/plateforme/installation.md. # # Design note — this script starts nothing itself. workerd owns the container's # lifecycle (create, health-gate, roll back, update); the UI's update button and # `gpuk update` drive that same daemon. One updater, three front doors. set -eu # ── Defaults (every one overridable by flag or env) ─────────────────────────── # The channel is the single source of truth for "what is the current release": # it is served next to this script, and the backend's update check reads the SAME # document (apps/controller/api/src/core/release-channel.ts). One file, one answer. Interim # default: the public Gitea channel repo — flips to https://gpu.kitchen/latest.json # once the hub exists (keep the three defaults in sync, see specs/developpement/ci-cd.md). CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen/channel/raw/branch/main/latest.json}" EDITION="${GPUK_EDITION:-community}" DATA_ROOT="${GPUK_DATA_ROOT:-/var/lib/gpu-kitchen}" CACHE_DIR="${GPUK_CACHE_DIR:-}" # 1337, not 8080: the single most-squatted port in existence would make the # conflict preflight fire on half the lab boxes out there (INS-01). PORT="${GPUK_PORT:-1337}" # The worker mTLS channel and the gpuk-proxy inference endpoint (defaults mirror # core/cluster-settings.ts). Movable at install time like the UI port (INS-46): # 8443 is every second appliance's HTTPS alias and 8200 is HashiCorp Vault's. MTLS_PORT="${GPUK_MTLS_PORT:-8443}" INFERENCE_PORT="${GPUK_INFERENCE_PORT:-8200}" CLUSTER="${GPUK_CLUSTER:-default}" # The app container's docker network. Empty = gpuk's default (bridge, INS-49); # an existing install keeps the mode it runs in (INS-03) — flipping the default # must never move a host-mode install to bridge on its next re-run. NETWORK="${GPUK_NETWORK:-}" # Which of those came from the operator (flag or env) — an existing install keeps # its own value for everything the operator did not ask to change (INS-03). PORT_GIVEN=0; [ -z "${GPUK_PORT:-}" ] || PORT_GIVEN=1 MTLS_GIVEN=0; [ -z "${GPUK_MTLS_PORT:-}" ] || MTLS_GIVEN=1 INFERENCE_GIVEN=0; [ -z "${GPUK_INFERENCE_PORT:-}" ] || INFERENCE_GIVEN=1 DATA_ROOT_GIVEN=0; [ -z "${GPUK_DATA_ROOT:-}" ] || DATA_ROOT_GIVEN=1 CACHE_GIVEN=0; [ -z "${GPUK_CACHE_DIR:-}" ] || CACHE_GIVEN=1 CLUSTER_GIVEN=0; [ -z "${GPUK_CLUSTER:-}" ] || CLUSTER_GIVEN=1 NETWORK_GIVEN=0; [ -z "${GPUK_NETWORK:-}" ] || NETWORK_GIVEN=1 # Where an existing install keeps its manifest. Same override as gpuk's, and for # the same reason: it is the only way to exercise the re-run path without root. ETC_DIR="${GPUK_ETC_DIR:-/etc/gpu-kitchen}" MANIFEST="$ETC_DIR/manifest.json" UNIT_DEST="${GPUK_UNIT_DEST:-/etc/systemd/system/gpu-kitchen-worker.service}" BIN_DEST="${GPUK_BIN_DEST:-/usr/local/bin/gpu-kitchen-worker}" PROFILE="" RESET_MANIFEST=0 DOMAIN="" VERSION="" IMAGE="" WORKER_BINARY="" GPUK_SCRIPT="" SKIP_GPU_CHECK=0 SKIP_PREFLIGHT=0 NON_INTERACTIVE=0 DRY_RUN=0 GREEN=''; RED=''; YELLOW=''; BOLD=''; RESET='' if [ -t 1 ]; then GREEN=$(printf '\033[32m'); RED=$(printf '\033[31m') YELLOW=$(printf '\033[33m'); BOLD=$(printf '\033[1m'); RESET=$(printf '\033[0m') fi ok() { echo " ${GREEN}✓${RESET} $*"; } warn() { echo " ${YELLOW}!${RESET} $*"; } step() { echo; echo "${BOLD}$*${RESET}"; } die() { echo " ${RED}✗${RESET} $*" >&2; exit 1; } # `curl … | sudo sh` leaves stdin holding the script itself, so questions are # asked and answered on the controlling terminal — /dev/tty — when there is one # (INS-45). No terminal (CI, provisioning), --non-interactive or --dry-run keep # every historical flags-only behaviour. can_prompt() { [ "$NON_INTERACTIVE" -eq 0 ] || return 1 [ "$DRY_RUN" -eq 0 ] || return 1 (: < /dev/tty) 2>/dev/null } ask() { # $1 = prompt → $REPLY (empty on EOF) printf '%s' "$1" > /dev/tty IFS= read -r REPLY < /dev/tty || REPLY="" } usage() { cat < Install this release instead of the channel's current one --image Use this controller image outright (implies --version none) --edition community (default) | enterprise --port

Port the UI listens on (default 1337) --mtls-port

Port workers dial to join this controller (default 8443) --inference-port

Port of the OpenAI-compatible inference endpoint (default 8200) --data-root Where the database and secrets live (default /var/lib/gpu-kitchen) --cache-dir Model cache (default /hf) --cluster Cluster name workers join (default "default") --network bridge (default) | host. Bridged, only the three ports above touch the host; host makes every listener of the container a host-wide claim. An existing install keeps its mode. --profile

homelab | studio | enterprise | public (no flag + a terminal = the script asks; no flag + no terminal = first-run asks) --domain Domain for the public profile: writes a filled TLS reverse-proxy example to /caddy/Caddyfile --reset-manifest Rebuild /etc/gpu-kitchen/manifest.json from this run's settings when the daemon refuses the existing one (a field an older release wrote); the old file is archived beside it and what it carried that the rebuild does not is listed --non-interactive Never ask anything, even with a terminal attached --worker-binary

Use a locally-built gpu-kitchen-worker instead of downloading one --gpuk-script

Use a local copy of the gpuk installer --skip-gpu-check Skip the 'docker run --gpus all' smoke test --skip-preflight Skip the docker, driver, GPU and disk checks (CI: no docker, no GPU); the listening ports are still checked --dry-run Run the preflight and resolve the release, change nothing -h, --help This EOF } while [ $# -gt 0 ]; do case "$1" in --version) VERSION="$2"; shift 2 ;; --image) IMAGE="$2"; shift 2 ;; --edition) EDITION="$2"; shift 2 ;; --port) PORT="$2"; PORT_GIVEN=1; shift 2 ;; --mtls-port) MTLS_PORT="$2"; MTLS_GIVEN=1; shift 2 ;; --inference-port) INFERENCE_PORT="$2"; INFERENCE_GIVEN=1; shift 2 ;; --data-root) DATA_ROOT="$2"; DATA_ROOT_GIVEN=1; shift 2 ;; --cache-dir) CACHE_DIR="$2"; CACHE_GIVEN=1; shift 2 ;; --cluster) CLUSTER="$2"; CLUSTER_GIVEN=1; shift 2 ;; --network) NETWORK="$2"; NETWORK_GIVEN=1; shift 2 ;; --profile) PROFILE="$2"; shift 2 ;; --domain) DOMAIN="$2"; shift 2 ;; --non-interactive) NON_INTERACTIVE=1; shift ;; --reset-manifest) RESET_MANIFEST=1; shift ;; --worker-binary) WORKER_BINARY="$2"; shift 2 ;; --gpuk-script) GPUK_SCRIPT="$2"; shift 2 ;; --skip-gpu-check) SKIP_GPU_CHECK=1; shift ;; --skip-preflight) SKIP_PREFLIGHT=1; shift ;; --dry-run) DRY_RUN=1; shift ;; -h|--help) usage; exit 0 ;; *) die "unknown option: $1 (try --help)" ;; esac done case "$EDITION" in community|enterprise) ;; *) die "--edition must be community or enterprise (got '$EDITION')" ;; esac case "$PROFILE" in ""|homelab|studio|enterprise|public) ;; *) die "--profile must be homelab, studio, enterprise or public (got '$PROFILE')" ;; esac case "$DOMAIN" in *[!A-Za-z0-9.-]*) die "--domain must be a bare domain name (got '$DOMAIN')" ;; esac # bridge, host, or the name of a docker network gpuk hands to `docker create # --network` — never a value that could be read as another flag or as whitespace. case "$NETWORK" in ""|bridge|host) ;; -*|*[!A-Za-z0-9_.-]*) die "--network must be bridge, host or a docker network name (got '$NETWORK')" ;; esac for _pv in "$PORT" "$MTLS_PORT" "$INFERENCE_PORT"; do case "$_pv" in ''|*[!0-9]*) die "not a port number: '$_pv'" ;; esac [ "$_pv" -ge 1 ] && [ "$_pv" -le 65535 ] || die "port out of range: $_pv" done echo echo "${BOLD}GPU Kitchen — installing the $EDITION controller${RESET}" # ── 0. An existing install (INS-03) ────────────────────────────────────────── # Re-running this script is the update path, so before checking anything it # reads what is already here — and KEEPS it. An update that silently moved the # UI to another port, re-asked the profile (Enter = homelab would downgrade a # public install) or pointed at a fresh data root beside the real one is not an # update. Read-only. The manifest is the authority (WRK-55); without it, the # traces a previous install leaves (container, unit, binary, data root) still # mean "take over", never "start beside", and the container's own ports are ours. manifest_str() { # $1 = key of a string field, anywhere in the manifest sed -n "s/.*\"$1\"[[:space:]]*:[[:space:]]*\"\([^\"]*\)\".*/\1/p" "$MANIFEST" 2>/dev/null | head -1 } manifest_binding() { # $1 = container port → the host port the ports table publishes it on sed -n "s/.*\"$1\(\/tcp\)\{0,1\}\"[[:space:]]*:[[:space:]]*\([0-9][0-9]*\).*/\2/p" "$MANIFEST" 2>/dev/null | head -1 } C_STATUS=""; C_NETMODE="" E_PORT=""; E_MTLS_PORT=""; E_LISTEN_ADDR="" E_PUBLIC_PORT=""; E_PUBLIC_MTLS_PORT=""; E_PROXY_PUBLIC_PORT="" B_8080=""; B_8443=""; B_8200="" container_facts() { # $1 = name → C_*, E_*, B_* from docker; 1 when absent command -v docker >/dev/null 2>&1 || return 1 _f=$(docker inspect -f '{{.State.Status}}|{{.HostConfig.NetworkMode}}|{{range $p, $b := .HostConfig.PortBindings}}{{range $b}}{{$p}}={{.HostPort}} {{end}}{{end}}{{"\n"}}{{range .Config.Env}}{{.}}{{"\n"}}{{end}}' "$1" 2>/dev/null) \ || return 1 [ -n "$_f" ] || return 1 _head=$(printf '%s\n' "$_f" | head -1) C_STATUS=${_head%%|*}; _r=${_head#*|}; C_NETMODE=${_r%%|*}; _bind=${_r#*|} E_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PORT=//p' | head -1) E_MTLS_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_MTLS_PORT=//p' | head -1) E_LISTEN_ADDR=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_LISTEN_ADDR=//p' | head -1) E_PUBLIC_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PUBLIC_PORT=//p' | head -1) E_PUBLIC_MTLS_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PUBLIC_MTLS_PORT=//p' | head -1) E_PROXY_PUBLIC_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PROXY_PUBLIC_PORT=//p' | head -1) B_8080=$(printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n 's/^8080\/tcp=//p' | head -1) B_8443=$(printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n 's/^8443\/tcp=//p' | head -1) B_8200=$(printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n 's/^8200\/tcp=//p' | head -1) } # The ports an install is REACHED on, with the precedence the controller itself # applies (core/published-ports.ts, INS-43): host networking moves the listeners # (GPUK_PORT, GPUK_MTLS_PORT, GPUK_LISTEN_ADDR); anything else keeps the image's # fixed listeners and publishes them (GPUK_PUBLIC_*, then the ports table). published_ports() { # $1 = network mode → OURS_UI OURS_MTLS OURS_INF if [ "$1" = "host" ]; then OURS_UI="${E_PORT:-8080}" OURS_MTLS="${E_MTLS_PORT:-8443}" OURS_INF="${E_PROXY_PUBLIC_PORT:-${E_LISTEN_ADDR##*:}}" else OURS_UI="${E_PUBLIC_PORT:-${B_8080:-8080}}" OURS_MTLS="${E_PUBLIC_MTLS_PORT:-${B_8443:-8443}}" OURS_INF="${E_PROXY_PUBLIC_PORT:-${B_8200:-8200}}" fi [ -n "$OURS_INF" ] || OURS_INF=8200 } EXISTING=""; OUR_PORTS=""; CONTAINER_NAME="gpu-kitchen" OURS_UI=""; OURS_MTLS=""; OURS_INF="" if [ -f "$MANIFEST" ] && [ ! -r "$MANIFEST" ]; then EXISTING="unreadable" elif [ -f "$MANIFEST" ]; then EXISTING="manifest" _cn=$(manifest_str containerName); [ -z "$_cn" ] || CONTAINER_NAME="$_cn" C_NETMODE=$(manifest_str networkMode) E_PORT=$(manifest_str GPUK_PORT); E_MTLS_PORT=$(manifest_str GPUK_MTLS_PORT) E_LISTEN_ADDR=$(manifest_str GPUK_LISTEN_ADDR) E_PUBLIC_PORT=$(manifest_str GPUK_PUBLIC_PORT) E_PUBLIC_MTLS_PORT=$(manifest_str GPUK_PUBLIC_MTLS_PORT) E_PROXY_PUBLIC_PORT=$(manifest_str GPUK_PROXY_PUBLIC_PORT) B_8080=$(manifest_binding 8080); B_8443=$(manifest_binding 8443); B_8200=$(manifest_binding 8200) published_ports "${C_NETMODE:-host}" OUR_PORTS="$OURS_UI $OURS_MTLS $OURS_INF" elif container_facts "$CONTAINER_NAME"; then EXISTING="leftovers" published_ports "$C_NETMODE" OUR_PORTS="$OURS_UI $OURS_MTLS $OURS_INF" elif [ -f "$UNIT_DEST" ] || [ -x "$BIN_DEST" ] || [ -d "$DATA_ROOT/secrets" ]; then EXISTING="leftovers" fi case "$EXISTING" in manifest) step "Existing install — $MANIFEST" ok "image $(manifest_str image)" _root=$(manifest_str dataRoot); _cache=$(manifest_str hostPath) _cluster=$(manifest_str GPUK_CLUSTER); _profile=$(manifest_str GPUK_INSTALL_PROFILE) [ "$DATA_ROOT_GIVEN" -eq 1 ] || [ -z "$_root" ] || DATA_ROOT="$_root" [ "$CACHE_GIVEN" -eq 1 ] || [ -z "$_cache" ] || CACHE_DIR="$_cache" [ "$CLUSTER_GIVEN" -eq 1 ] || [ -z "$_cluster" ] || CLUSTER="$_cluster" [ "$PORT_GIVEN" -eq 1 ] || PORT="$OURS_UI" [ "$MTLS_GIVEN" -eq 1 ] || MTLS_PORT="$OURS_MTLS" [ "$INFERENCE_GIVEN" -eq 1 ] || INFERENCE_PORT="$OURS_INF" # The network mode is a setting like the ports: an install that runs on the # host network stays there when the default is bridge (INS-49), and the # reverse — only --network moves it. An old manifest without the field is # read as the daemon reads it (host). [ "$NETWORK_GIVEN" -eq 1 ] || NETWORK="${C_NETMODE:-host}" case "$_profile" in homelab|studio|enterprise|public) [ -n "$PROFILE" ] || PROFILE="$_profile" ;; esac ok "data root $DATA_ROOT" ok "ports UI $OURS_UI, worker channel $OURS_MTLS, inference $OURS_INF" ok "network ${C_NETMODE:-host}" [ -z "$_profile" ] || ok "profile $_profile" ok "re-running updates it in place. Its settings are kept unless a flag says otherwise." ;; unreadable) step "Existing install — $MANIFEST" warn "present, but not readable from here: run as root to keep its settings" ;; leftovers) step "Existing install — traces of a previous install, no manifest" if [ -n "$C_STATUS" ]; then ok "container $CONTAINER_NAME ($C_STATUS; UI $OURS_UI, worker channel $OURS_MTLS, inference $OURS_INF) — the install replaces it" fi [ ! -f "$UNIT_DEST" ] || ok "systemd unit $UNIT_DEST — rewritten" [ ! -x "$BIN_DEST" ] || ok "daemon binary $BIN_DEST — replaced" [ ! -d "$DATA_ROOT/secrets" ] || ok "data root $DATA_ROOT — reused, nothing in it is touched" warn "without $MANIFEST no setting can be kept: the flags and the defaults apply" ;; esac [ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf" if [ "$PORT" = "$MTLS_PORT" ] || [ "$PORT" = "$INFERENCE_PORT" ] || [ "$MTLS_PORT" = "$INFERENCE_PORT" ]; then die "the UI, worker channel and inference ports must differ (got $PORT, $MTLS_PORT, $INFERENCE_PORT)" fi # ── 1. Preflight ───────────────────────────────────────────────────────────── # The same checks tools/provision-feeder.sh makes, minus the compose ones: the # all-in-one image is driven by workerd through the plain docker CLI, so there is # no compose dependency to satisfy any more. if [ "$SKIP_PREFLIGHT" -eq 1 ]; then step "Preflight — skipped (--skip-preflight)" warn "the host is NOT being checked for docker, a driver or a GPU" else step "Preflight — docker, NVIDIA driver, container toolkit" # Root, or a user-owned prefix (GPUK_ETC_DIR) — the same rule as gpuk's # need_root, and the only way the full path is testable without handing root # to a test suite. [ "$DRY_RUN" -eq 1 ] || [ "$(id -u)" -eq 0 ] || [ -w "$ETC_DIR" ] \ || die "run as root: curl -fsSL … | sudo sh" command -v curl >/dev/null 2>&1 || die "curl not found. Install curl first." if command -v docker >/dev/null 2>&1; then if docker info >/dev/null 2>&1; then ok "docker daemon reachable ($(docker --version | cut -d' ' -f3 | tr -d ,))" else die "docker is installed but its daemon does not answer. Start the daemon, or run as root." fi else die "docker not found. Install Docker Engine first: https://docs.docker.com/engine/install/" fi if command -v nvidia-smi >/dev/null 2>&1; then DRIVER=$(nvidia-smi --query-gpu=driver_version --format=csv,noheader 2>/dev/null | head -1 || true) GPU_COUNT=$(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | grep -c . || true) if [ -n "$DRIVER" ] && [ "${GPU_COUNT:-0}" -gt 0 ]; then ok "NVIDIA driver $DRIVER — ${GPU_COUNT} GPU(s): $(nvidia-smi --query-gpu=name --format=csv,noheader | sort -u | paste -sd', ')" else die "nvidia-smi is present but reports no GPU" fi else die "nvidia-smi not found. Install the NVIDIA driver first." fi if [ "$SKIP_GPU_CHECK" -eq 1 ]; then warn "nvidia-container-toolkit check skipped (--skip-gpu-check)" elif docker info 2>/dev/null | grep -qiE 'runtimes:.*nvidia'; then # The runtime being REGISTERED is not the same as it working. With the toolkit # installed, --gpus injects the driver and nvidia-smi into a plain image; that # is the exact mechanism the controller container relies on, so test it rather # than infer it. if docker run --rm --gpus all ubuntu:24.04 nvidia-smi -L >/dev/null 2>&1; then ok "nvidia-container-toolkit works (a container can see the GPUs)" else die "the nvidia runtime is registered but 'docker run --gpus all' cannot see the GPUs. Reinstall nvidia-container-toolkit." fi else die "nvidia-container-toolkit is not registered with docker. Install it, then restart dockerd: https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html" fi # Model weights are large and the failure mode (a download dying at 90%) is # miserable, so say so up front. A warning, not a refusal: it is the user's disk. CACHE_PARENT="$CACHE_DIR" while [ ! -d "$CACHE_PARENT" ] && [ "$CACHE_PARENT" != "/" ]; do CACHE_PARENT=$(dirname "$CACHE_PARENT") done FREE_GB=$(df -BG --output=avail "$CACHE_PARENT" 2>/dev/null | tail -1 | tr -dc '0-9' || true) if [ "${FREE_GB:-0}" -ge 100 ]; then ok "model cache $CACHE_DIR — ${FREE_GB}G free" else warn "only ${FREE_GB:-?}G free under $CACHE_PARENT. Model weights need 100G or more." fi fi # end preflight # ── Listening ports (INS-46) ────────────────────────────────────────────────── # Deliberately OUTSIDE the preflight branch: --skip-preflight skips docker, the # driver, the GPU smoke test and the disk (things a runner or a VM cannot have), # but a taken port is exactly as fatal there, and checking it costs nothing. # Skipping it here only moved the failure to gpuk's non-interactive refusal. if [ "$SKIP_PREFLIGHT" -eq 1 ]; then step "Listening ports — checked even without the preflight" fi # A taken port must fail HERE, before anything mutates the host — today's # alternative is a 3-minute health-check timeout with zero diagnosis. Best-effort # detection (ss, then netstat); neither present is a warn, never a false red. # A listener owned by an EXISTING GPU Kitchen install is not a conflict: re-running # this script is the documented update path, and the manifest names our port. # Test hook: force the detector. The netstat fallback is unreachable on any # host that has ss (all of them, in practice), so the CI smoke pins it here to # keep its parsing honest. PORT_TOOL="${GPUK_PORT_CHECK_TOOL:-}" case "$PORT_TOOL" in ""|ss|netstat) ;; *) die "GPUK_PORT_CHECK_TOOL must be ss or netstat (got '$PORT_TOOL')" ;; esac if [ -z "$PORT_TOOL" ]; then if command -v ss >/dev/null 2>&1; then PORT_TOOL="ss" elif command -v netstat >/dev/null 2>&1; then PORT_TOOL="netstat"; fi fi port_busy() { # $1 = port → 0 iff something listens on TCP :$1 case "$PORT_TOOL" in ss) [ -n "$(ss -ltnH "sport = :$1" 2>/dev/null)" ] ;; netstat) netstat -ltn 2>/dev/null \ | awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) found=1} END{exit found?0:1}' ;; *) return 1 ;; esac } # Which container a listener belongs to, if any: the process's cgroup names the # container id (host networking — the listener IS the container's process), and a # bridged publication shows up as the container's port mapping in `docker ps` # (the host-side holder is docker-proxy, which says nothing by itself). "nginx" # is a riddle; "nginx in container gpu-kitchen-dev" is the answer. port_container() { # $1 = port, $2 = pid ("" if unknown) → container name or "" command -v docker >/dev/null 2>&1 || return 0 if [ -n "$2" ] && [ -r "/proc/$2/cgroup" ]; then _cid=$(sed -n 's#.*docker[-/]\([0-9a-f]\{64\}\).*#\1#p' "/proc/$2/cgroup" 2>/dev/null | head -1) if [ -n "$_cid" ]; then docker inspect -f '{{.Name}}' "$_cid" 2>/dev/null | sed 's|^/||' return 0 fi fi docker ps --format '{{.Names}} {{.Ports}}' 2>/dev/null \ | awk -v p=":$1->" 'index($0, p) { print $1; exit }' } port_owner() { # $1 = port → best-effort "process", "process in container NAME", or "" _proc=""; _pid="" case "$PORT_TOOL" in ss) _line=$(ss -ltnpH "sport = :$1" 2>/dev/null | head -1) _proc=$(printf '%s' "$_line" | sed -n 's/.*users:((\"\([^"]*\)\".*/\1/p') _pid=$(printf '%s' "$_line" | sed -n 's/.*pid=\([0-9]*\).*/\1/p') ;; netstat) _field=$(netstat -ltnp 2>/dev/null \ | awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) {print $NF; exit}}') case "$_field" in */*) _pid=${_field%%/*}; _proc=${_field#*/} ;; *) _proc="$_field" ;; esac ;; esac case "$_pid" in *[!0-9]*|"") _pid="" ;; esac _ctr=$(port_container "$1" "$_pid") if [ -n "$_ctr" ]; then printf '%s' "${_proc:-a process} in container $_ctr" else printf '%s' "$_proc" fi } # A port is ours when the existing install (step 0) already holds it: the # re-run replaces that container, so what it listens on is not a conflict. port_is_ours() { case " $OUR_PORTS " in *" $1 "*) return 0 ;; esac; return 1; } # Ports this run may not hand out twice: the three requested ones, plus every # alternative already accepted. Without it, a busy 8442 would be offered 8443 # and collide with the worker channel one question later. RESERVED_PORTS="$PORT $MTLS_PORT $INFERENCE_PORT" port_available() { # $1 → free on the host AND not reserved by this run case " $RESERVED_PORTS " in *" $1 "*) return 1 ;; esac port_is_ours "$1" && return 1 ! port_busy "$1" } # resolve_port

." ;; "") _want="$ALT" ;; *) case "$REPLY" in *[!0-9]*) die "not a port number: $REPLY" ;; esac port_available "$REPLY" \ || die "port $REPLY is busy, or already taken by another GPU Kitchen listener. Run the install again with $_flag

." _want="$REPLY" ;; esac RESERVED_PORTS="$RESERVED_PORTS $_want" ok "$_label port $_want is free" else die "$_label port $_want is already in use by $OWNER. Pass $_flag

to choose another port." fi else ok "$_label port $_want is free" fi RESOLVED="$_want" } if [ -z "$PORT_TOOL" ]; then warn "cannot check for port conflicts (neither ss nor netstat found)" else resolve_port "UI" "$PORT" "--port"; PORT="$RESOLVED" resolve_port "worker channel" "$MTLS_PORT" "--mtls-port"; MTLS_PORT="$RESOLVED" resolve_port "inference" "$INFERENCE_PORT" "--inference-port"; INFERENCE_PORT="$RESOLVED" fi # ── 2. Resolve the release ─────────────────────────────────────────────────── step "Release — resolving the version to install" # One tiny JSON document, fetched over TLS, holding what the current release IS. # Parsed with sed rather than jq: `curl … | sudo sh` cannot assume jq exists, and # the document is ours and flat. json_field() { sed -n "s/.*\"$1\"[[:space:]]*:[[:space:]]*\"\([^\"]*\)\".*/\1/p" | head -1; } # Strip a tag off an image reference WITHOUT mangling a registry's host:port. # `${ref%%:*}` cuts at the FIRST colon and is WRONG: a # `registry.internal:5000/gpuk/controller:v1.2.3` (or an untagged # `registry.internal:5000/gpuk/controller`) would collapse to `registry.internal`. # A colon is a tag separator only when the last colon comes AFTER the last slash. # Same rule as deployments/install/gpuk's image_repo() and core/release-channel.ts (unit-tested). image_repo() { case "$1" in *@*) printf '%s' "${1%@*}"; return ;; # digest pin → repo@sha256:… esac _t="${1##*:}" case "$_t" in "$1") printf '%s' "$1" ;; # no colon at all → already untagged */*) printf '%s' "$1" ;; # last colon is inside a path → host:port, untagged *) printf '%s' "${1%:*}" ;; esac } if [ -n "$IMAGE" ]; then ok "using the image given on the command line: $IMAGE" [ -n "$VERSION" ] || VERSION="(pinned by --image)" else CHANNEL=$(curl -fsSL --max-time 20 "$CHANNEL_URL" 2>/dev/null) \ || die "cannot reach the release channel at $CHANNEL_URL If this host is offline, pass --image to install a specific image directly." [ -n "$VERSION" ] || VERSION=$(echo "$CHANNEL" | json_field version) [ -n "$VERSION" ] || die "the release channel returned no version: $CHANNEL_URL" if [ "$EDITION" = "enterprise" ]; then IMAGE_TEMPLATE=$(echo "$CHANNEL" | json_field controllerImageEnterprise) else IMAGE_TEMPLATE=$(echo "$CHANNEL" | json_field controllerImage) fi [ -n "$IMAGE_TEMPLATE" ] \ || die "the release channel names no $EDITION controller image: $CHANNEL_URL" # The channel gives the repository; WE pin the tag. A floating `:latest` would # make every container recreate a silent, unrequested upgrade. Strip any tag the # channel already carries with image_repo (host:port-safe), then pin OUR version. IMAGE="$(image_repo "$IMAGE_TEMPLATE"):${VERSION}" ok "release $VERSION" ok "image $IMAGE" [ -n "$WORKER_BINARY" ] || WORKER_RELEASE_BASE=$(echo "$CHANNEL" | json_field workerBase) [ -n "${GPUK_SCRIPT}" ] || GPUK_SCRIPT_URL=$(echo "$CHANNEL" | json_field gpukScript) fi # A direct --image and a channel version are held to the same immutable-image # rule. A registry host:port is not a tag separator; only the last colon after # the last slash counts. Digest pins are accepted too. case "$IMAGE" in *@sha256:*) ;; *:latest) die "refusing floating image tag '$IMAGE'. Use an explicit release tag or digest." ;; *) IMAGE_TAG="${IMAGE##*:}" case "$IMAGE_TAG" in "$IMAGE"|*/*) die "controller image must carry an explicit tag or sha256 digest (got '$IMAGE')" ;; esac ;; esac if [ -n "$DOMAIN" ] && [ -n "$PROFILE" ] && [ "$PROFILE" != "public" ]; then die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)" fi if [ "$DRY_RUN" -eq 1 ]; then step "Dry run — stopping here" if [ "$SKIP_PREFLIGHT" -eq 1 ]; then ok "the release resolved; the host was not checked; nothing was installed" else ok "preflight passed and the release resolved; nothing was installed" fi echo echo " would install : $IMAGE" echo " data root : $DATA_ROOT" echo " model cache : $CACHE_DIR" echo " UI port : $PORT" echo " worker channel: $MTLS_PORT" echo " inference : $INFERENCE_PORT" echo " network : ${NETWORK:-bridge}" case "$EXISTING" in manifest) echo " existing : yes — updated in place" ;; leftovers) echo " existing : traces of a previous install — taken over" ;; *) echo " existing : no" ;; esac if [ -n "$PROFILE" ]; then echo " profile : $PROFILE" else echo " profile : (asked on the terminal, or chosen during first run)" fi [ -z "$DOMAIN" ] || echo " domain : $DOMAIN" exit 0 fi # ── 3. Resolve the installation profile (INS-45) ───────────────────────────── # Only ever on a terminal, and only when --profile was not given. The answer is # relayed verbatim as --profile: the backend stays the sole applier of profile # defaults and floors. Without a terminal the historical path is untouched — # profile unset, enterprise floors, the first-run wizard requires the choice # (PRF-05). if [ -z "$PROFILE" ] && can_prompt; then step "Security profile — how will this kitchen be used?" cat > /dev/tty <<'PROFILES' 1) Home lab a trusted home network No sign-in on your home network. Nearby workers are found and join without waiting for approval. 2) Studio one control station, shared compute Control stays on this machine. Colleagues use API keys, while nearby workers wait for your approval. 3) Enterprise a managed company network Sign-in is required. The controller stays quiet on the network; known workers can request your approval. 4) Public server direct internet exposure Sign-in and hardened browser transport are required. Network discovery is off and workers join only by token. You can change this later in Settings. Stricter floors re-apply. Nothing already issued is revoked. PROFILES while :; do ask " Choose a profile [1-4, Enter = 1 (Home lab)]: " case "$REPLY" in ""|1|homelab) PROFILE="homelab" ;; 2|studio) PROFILE="studio" ;; 3|enterprise) PROFILE="enterprise" ;; 4|public) PROFILE="public" ;; *) printf '%s\n' " pick 1, 2, 3 or 4" > /dev/tty; continue ;; esac break done ok "profile: $PROFILE" fi # Under public, a domain lets gpuk write a FILLED TLS reverse-proxy example # (INS-47). Optional: Enter skips, and the banner still points at the shipped # Caddyfile.example. if [ "$PROFILE" = "public" ] && [ -z "$DOMAIN" ] && can_prompt; then while :; do ask " Domain for HTTPS access (e.g. gpu.example.com — Enter to skip): " # Accept the copy-paste reflex: strip a pasted scheme and anything after # the first slash, then insist on a bare domain rather than skipping — # a silently dropped answer would be discovered hours later, at DNS time. REPLY="${REPLY#https://}" REPLY="${REPLY#http://}" REPLY="${REPLY%%/*}" [ -n "$REPLY" ] || break case "$REPLY" in *[!A-Za-z0-9.-]*) printf '%s\n' " not a bare domain name: $REPLY. Try again, or press Enter to skip." > /dev/tty ;; *) DOMAIN="$REPLY"; break ;; esac done fi if [ -n "$DOMAIN" ] && [ "$PROFILE" != "public" ]; then die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)" fi # ── 4. The host daemon ─────────────────────────────────────────────────────── step "Host daemon — gpu-kitchen-worker" TMP=$(mktemp -d) # shellcheck disable=SC2064 # expand TMP now: it must be removed even if it changes trap "rm -rf '$TMP'" EXIT INT TERM if [ -z "$GPUK_SCRIPT" ]; then # A checkout right here beats a download (that is how contributors run it). _local="$(dirname "$0")/gpuk" if [ -f "$_local" ]; then GPUK_SCRIPT="$_local" ok "using the gpuk installer from this checkout" else [ -n "${GPUK_SCRIPT_URL:-}" ] \ || die "the release channel names no gpuk installer, and none was found locally" curl -fsSL --max-time 60 "$GPUK_SCRIPT_URL" -o "$TMP/gpuk" \ || die "cannot download the gpuk installer from $GPUK_SCRIPT_URL" chmod +x "$TMP/gpuk" GPUK_SCRIPT="$TMP/gpuk" ok "downloaded the gpuk installer" fi fi # `gpuk install` does the rest: it drops the binary, writes the systemd unit, # writes the manifest (the declarative description of the app container) and # applies it. Everything below is passed straight through to it. set -- install \ --mode controller \ --image "$IMAGE" \ --cluster "$CLUSTER" \ --data-root "$DATA_ROOT" \ --cache-dir "$CACHE_DIR" \ --http-port "$PORT" \ --mtls-port "$MTLS_PORT" \ --inference-port "$INFERENCE_PORT" [ -z "$PROFILE" ] || set -- "$@" --profile "$PROFILE" [ -z "$DOMAIN" ] || set -- "$@" --domain "$DOMAIN" [ "$RESET_MANIFEST" -eq 0 ] || set -- "$@" --reset-manifest # Inherited from the manifest or given by flag; unset on a first install, so # gpuk's own default (bridge, INS-49) applies and this script never restates it. [ -z "$NETWORK" ] || set -- "$@" --network "$NETWORK" if [ -n "$WORKER_BINARY" ]; then [ -f "$WORKER_BINARY" ] || die "no such worker binary: $WORKER_BINARY" set -- "$@" --binary "$WORKER_BINARY" ok "using a locally-built gpu-kitchen-worker" elif [ -n "${WORKER_RELEASE_BASE:-}" ]; then GPUK_RELEASE_BASE="$WORKER_RELEASE_BASE" export GPUK_RELEASE_BASE ok "gpu-kitchen-worker will be downloaded from the release" fi # else: gpuk reuses an already-installed binary, or fails with its own message. step "Installing — this pulls the image, so it can take a few minutes" # gpuk names its own failure on stderr before exiting (a refused manifest, a # failed apply, a missing binary…), so the cause is the line right above this # one. journalctl only has something to say once the daemon has started, and # gpuk points at `gpuk logs` itself in that case. sh "$GPUK_SCRIPT" "$@" || die "the install failed — gpuk reported the cause just above" # ── 5. Wait for the app, then say where it is ──────────────────────────────── step "Waiting for the controller to answer" i=0 until curl -fsS -o /dev/null --max-time 3 "http://127.0.0.1:$PORT/health" 2>/dev/null; do i=$((i + 1)) [ "$i" -lt 90 ] || die "the controller did not answer on :$PORT within 90s. See: gpuk logs" sleep 2 done ok "the controller is up" # The URL must name the REAL host: the person installing this is very often not # sitting at the machine, and "localhost" would be a lie on every box but theirs # (same reason the backend resolves its own hostname — core/host-name.ts). HOSTNAME_FQDN=$(hostname -f 2>/dev/null || hostname 2>/dev/null || echo localhost) LAN_IP=$(ip route get 1.1.1.1 2>/dev/null | sed -n 's/.* src \([0-9.]*\).*/\1/p' | head -1) echo echo "${BOLD}${GREEN}GPU Kitchen is running.${RESET}" echo echo " ${BOLD}Open:${RESET} http://${HOSTNAME_FQDN}:${PORT}" [ -n "$LAN_IP" ] && echo " http://${LAN_IP}:${PORT}" echo CLAIM_CODE_FILE="$DATA_ROOT/secrets/claim_code" CLAIM_CODE="" [ ! -s "$CLAIM_CODE_FILE" ] || CLAIM_CODE=$(cat "$CLAIM_CODE_FILE" 2>/dev/null || true) if [ -n "$CLAIM_CODE" ]; then echo " ${BOLD}Claim code:${RESET} $CLAIM_CODE" echo " ${BOLD}Read again:${RESET} $CLAIM_CODE_FILE (mode 0600; removed after claim)" else echo " ${BOLD}Claim code:${RESET} consumed (the first account already exists)" fi echo # The wizard's first step asks for a Kitchen ACCOUNT key (CPT-04). It is the # person's credential, never the machine's — this script cannot create or print # it, and the browser must not receive it in a URL (CPT-05/06). What it can do # is say so, and say where the key comes from, before the page does. echo " ${BOLD}Next:${RESET} open the URL above. Its first step asks for your GPU Kitchen account key" echo " (gpuk_…). Sign in to your GPU Kitchen account and create one under Install keys:" echo " https://gpu.kitchen/account#install-keys" echo " That key is yours, not this machine's: the installer never sees it," echo " and the page exchanges it for a revocable installation token." case "$PROFILE" in public) echo " ${BOLD}Then:${RESET} create the first administrator in the UI" echo # The public profile REQUIRES a TLS reverse proxy (OPS-13, INS-47) — the UI # port speaks plain HTTP. The product cannot verify the proxy's presence # (OPS-67), so the closest thing to enforcement is saying it here, clearly. if [ -n "$DOMAIN" ]; then echo " ${BOLD}HTTPS:${RESET} https://${DOMAIN} answers after these steps:" echo " 1. DNS: point ${DOMAIN} at this machine's public IP" echo " 2. Install Caddy: https://caddyserver.com/docs/install" echo " 3. sudo cp $DATA_ROOT/caddy/Caddyfile /etc/caddy/Caddyfile" echo " sudo systemctl reload caddy" echo " A filled example was written to $DATA_ROOT/caddy/Caddyfile." echo " Until then the UI answers in cleartext on the URLs above." else echo " ${BOLD}HTTPS:${RESET} the public profile requires a TLS reverse proxy before any" echo " public exposure. See deployments/controller/Caddyfile.example." fi echo ;; enterprise) echo " ${BOLD}Then:${RESET} create the first administrator in the UI" echo ;; homelab|studio) echo " ${BOLD}Then:${RESET} finish first-run in the UI" echo ;; "") echo " ${BOLD}Then:${RESET} choose an installation profile in the first-run assistant" echo ;; esac echo " Inference endpoint : http://${HOSTNAME_FQDN}:${INFERENCE_PORT}/v1" echo " Version : ${VERSION}" echo # Pre-existing model caches (B81). `gpuk install` prints the same hint, but that # scrolls past mid-install; this banner is where people actually look. Detection # only — cache questions belong to the first-run wizard (INS-48, REG-41), never # to the CLI, and nothing outside $CACHE_DIR is touched, read as configuration, # or modified. EXISTING_CACHE="" CHOSEN_CACHE=$(readlink -f "$CACHE_DIR" 2>/dev/null || echo "$CACHE_DIR") for d in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do [ -n "$d" ] && [ -d "$d/hub" ] || continue real=$(readlink -f "$d" 2>/dev/null || echo "$d") [ "$real" != "$CHOSEN_CACHE" ] || continue ls -d "$d"/hub/models--* >/dev/null 2>&1 || continue EXISTING_CACHE="$EXISTING_CACHE $real" done if [ -n "$EXISTING_CACHE" ]; then for d in $EXISTING_CACHE; do echo " ${BOLD}Existing model cache:${RESET} $d" done echo " GPU Kitchen will offer to reuse those models at first launch, and any" echo " time from Nodes & GPU -> Storage. A reused cache is referenced in" echo " place. Nothing is moved or deleted." echo fi echo " Update : re-run this command, or press Update in the UI, or: gpuk update" echo " Status : gpuk status Logs: gpuk logs" echo " Remove : gpuk uninstall (service only) or gpuk uninstall --purge (all but the data root)" echo