#!/bin/sh # GPU Kitchen — one-command install (specs/plateforme/installation.md). # # curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh | sudo sh # # (Interim URL — becomes https://gpu.kitchen/install.sh once the hub exists; the # channel repo is written by .gitea/workflows/release.yml, see specs/developpement/ci-cd.md.) # # What it does, and nothing more: # 0. existing detect an install already on this host (manifest, or the traces # a previous one left) and KEEP its settings — ports, data root, # profile — unless a flag says otherwise (INS-03) # 1. preflight docker, the NVIDIA driver, a REAL `--gpus all` smoke test, and # the listening ports (INS-46 — a taken port is resolved HERE, on # the terminal, not three minutes later in a health-check timeout) # 2. resolve the current release from the channel (a TAG — never a floating # `latest`: an install that silently changes version under you is # not an install, it is a surprise) # 3. ask the security profile — and, under `public`, a domain — when a # terminal is attached (INS-45); no terminal, no questions, and # the first-run wizard asks instead (PRF-05) # 4. install `gpu-kitchen-worker` (workerd), the privileged host daemon, via # the existing `gpuk` installer — on a controller node it OWNS the # app container's lifecycle # 5. hand over workerd pulls the pinned all-in-one controller image and starts it # 6. print the UI URL on the real host, plus the profile-specific next step # # Re-running is how you UPDATE: on a host that already runs GPU Kitchen, the same # command hands over to the installed `gpuk update` — the signed channel checked # against the key that install pinned, the image pinned by digest, data untouched # (it lives in the data root, not the container). A flag that changes a setting, # --reinstall or --reset-manifest reinstall over it instead (INS-03). # # This installs a CONTROLLER node (the full app + UI). A headless compute node is # `gpuk install --mode worker …` — see deployments/install/gpuk and # specs/plateforme/installation.md. # # Design note — this script starts nothing itself. workerd owns the container's # lifecycle (create, health-gate, roll back, update); the UI's update button and # `gpuk update` drive that same daemon. One updater, three front doors. # # The whole script is the body of main(), called on its very last line: `sh` # reads a piped script as it arrives, so a download cut short would otherwise # run as root up to wherever it stopped. Truncated, main is never called. set -eu main() { # ── Defaults (every one overridable by flag or env) ─────────────────────────── # The channel is the single source of truth for "what is the current release": # it is served next to this script, and the backend's update check reads the SAME # document (apps/controller/api/src/core/release-channel.ts). One file, one answer. Interim # default: the public Gitea channel repo — flips to https://gpu.kitchen/latest.json # once the hub exists (keep the three defaults in sync, see specs/developpement/ci-cd.md). CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen/channel/raw/branch/main/latest.json}" EDITION="${GPUK_EDITION:-community}" DATA_ROOT="${GPUK_DATA_ROOT:-/var/lib/gpu-kitchen}" CACHE_DIR="${GPUK_CACHE_DIR:-}" # 1337, not 8080: the single most-squatted port in existence would make the # conflict preflight fire on half the lab boxes out there (INS-01). PORT="${GPUK_PORT:-1337}" # The worker mTLS channel and the gpuk-proxy inference endpoint (defaults mirror # core/cluster-settings.ts). Movable at install time like the UI port (INS-46): # 8443 is every second appliance's HTTPS alias and 8200 is HashiCorp Vault's. MTLS_PORT="${GPUK_MTLS_PORT:-8443}" INFERENCE_PORT="${GPUK_INFERENCE_PORT:-8200}" CLUSTER="${GPUK_CLUSTER:-default}" # The app container's docker network. Empty = gpuk's default (bridge, INS-49); # an existing install keeps the mode it runs in (INS-03) — flipping the default # must never move a host-mode install to bridge on its next re-run. NETWORK="${GPUK_NETWORK:-}" # Which of those came from the operator (flag or env) — an existing install keeps # its own value for everything the operator did not ask to change (INS-03). PORT_GIVEN=0; [ -z "${GPUK_PORT:-}" ] || PORT_GIVEN=1 MTLS_GIVEN=0; [ -z "${GPUK_MTLS_PORT:-}" ] || MTLS_GIVEN=1 INFERENCE_GIVEN=0; [ -z "${GPUK_INFERENCE_PORT:-}" ] || INFERENCE_GIVEN=1 DATA_ROOT_GIVEN=0; [ -z "${GPUK_DATA_ROOT:-}" ] || DATA_ROOT_GIVEN=1 CACHE_GIVEN=0; [ -z "${GPUK_CACHE_DIR:-}" ] || CACHE_GIVEN=1 CLUSTER_GIVEN=0; [ -z "${GPUK_CLUSTER:-}" ] || CLUSTER_GIVEN=1 NETWORK_GIVEN=0; [ -z "${GPUK_NETWORK:-}" ] || NETWORK_GIVEN=1 # Where an existing install keeps its manifest. Same override as gpuk's, and for # the same reason: it is the only way to exercise the re-run path without root. ETC_DIR="${GPUK_ETC_DIR:-/etc/gpu-kitchen}" MANIFEST="$ETC_DIR/manifest.json" UNIT_DEST="${GPUK_UNIT_DEST:-/etc/systemd/system/gpu-kitchen-worker.service}" BIN_DEST="${GPUK_BIN_DEST:-/usr/local/bin/gpu-kitchen-worker}" # The gpuk CLI `gpuk install` leaves on the host, next to the daemon binary: the # one that carries the release key this install pinned, and the one a re-run # hands the update to (same override as gpuk's). GPUK_CLI="${GPUK_CLI_DEST:-$(dirname "$BIN_DEST")/gpuk}" # The release public key pinned in THIS copy of install.sh (OPS-20). The channel # copy gets the real key substituted at publish time, exactly like gpuk's; the # operator override covers a self-hosted channel with its own keypair. The # script itself arrives over HTTPS (trust on first use); everything it then # takes from the channel — latest.json, the gpuk installer, the daemon binary, # the image digest — is verified against this key before it is used. CHANNEL_PUBKEY="${GPUK_UPDATE_PUBKEY:-RWQ7BKXJqGX2jdKXu1GxeSPVAN3JDRTefpImM/mFRjtFwq4E7mhnQtA7}" EDITION_GIVEN=0; [ -z "${GPUK_EDITION:-}" ] || EDITION_GIVEN=1 PROFILE="" RESET_MANIFEST=0 REINSTALL=0 DOMAIN="" VERSION="" IMAGE="" WORKER_BINARY="" GPUK_SCRIPT="" SKIP_GPU_CHECK=0 SKIP_PREFLIGHT=0 NON_INTERACTIVE=0 DRY_RUN=0 GREEN=''; RED=''; YELLOW=''; BOLD=''; RESET='' if [ -t 1 ]; then GREEN=$(printf '\033[32m'); RED=$(printf '\033[31m') YELLOW=$(printf '\033[33m'); BOLD=$(printf '\033[1m'); RESET=$(printf '\033[0m') fi ok() { echo " ${GREEN}✓${RESET} $*"; } warn() { echo " ${YELLOW}!${RESET} $*"; } step() { echo; echo "${BOLD}$*${RESET}"; } die() { echo " ${RED}✗${RESET} $*" >&2; exit 1; } # `curl … | sudo sh` leaves stdin holding the script itself, so questions are # asked and answered on the controlling terminal — /dev/tty — when there is one # (INS-45). No terminal (CI, provisioning), --non-interactive or --dry-run keep # every historical flags-only behaviour. can_prompt() { [ "$NON_INTERACTIVE" -eq 0 ] || return 1 [ "$DRY_RUN" -eq 0 ] || return 1 (: < /dev/tty) 2>/dev/null } ask() { # $1 = prompt → $REPLY (empty on EOF) printf '%s' "$1" > /dev/tty IFS= read -r REPLY < /dev/tty || REPLY="" } usage() { cat < The release the channel names (another one: --image @sha256:…) --image Use this controller image outright (implies --version none) --edition community (default) | enterprise --port

Port the UI listens on (default 1337) --mtls-port

Port workers dial to join this controller (default 8443) --inference-port

Port of the OpenAI-compatible inference endpoint (default 8200) --data-root Where the database and secrets live (default /var/lib/gpu-kitchen) --cache-dir Model cache (default /hf) --cluster Cluster name workers join (default "default") --network bridge (default) | host. Bridged, only the three ports above touch the host; host makes every listener of the container a host-wide claim. An existing install keeps its mode. --profile

homelab | studio | enterprise | public (no flag + a terminal = the script asks; no flag + no terminal = first-run asks) --domain Name the UI is reached by through a reverse proxy, any profile: allowed as a host of the install, proxy trusted for the client address, and a filled TLS reverse-proxy example written to /caddy/Caddyfile --reinstall On a host that already runs GPU Kitchen, reinstall over it (settings kept, merged into its manifest) instead of the default re-run, which is 'gpuk update' --reset-manifest Rebuild /etc/gpu-kitchen/manifest.json from this run's settings when the daemon refuses the existing one (a field an older release wrote); the old file is archived beside it and what it carried that the rebuild does not is listed --non-interactive Never ask anything, even with a terminal attached --worker-binary

Use a locally-built gpu-kitchen-worker instead of downloading one --gpuk-script

Use a local copy of the gpuk installer --skip-gpu-check Skip the 'docker run --gpus all' smoke test --skip-preflight Skip the docker, driver, GPU and disk checks (CI: no docker, no GPU); the listening ports are still checked --dry-run Run the preflight and resolve the release, change nothing -h, --help This EOF } # The command line as given, shell-quoted: a refusal can then print the exact # command to run next instead of a flag to add by hand. ORIG_ARGS="" for _arg in "$@"; do case "$_arg" in *[!A-Za-z0-9_./:=@,+-]*|"") ORIG_ARGS="$ORIG_ARGS '$(printf '%s' "$_arg" | sed "s/'/'\\\\''/g")'" ;; *) ORIG_ARGS="$ORIG_ARGS $_arg" ;; esac done while [ $# -gt 0 ]; do case "$1" in --version) VERSION="$2"; shift 2 ;; --image) IMAGE="$2"; shift 2 ;; --edition) EDITION="$2"; EDITION_GIVEN=1; shift 2 ;; --port) PORT="$2"; PORT_GIVEN=1; shift 2 ;; --mtls-port) MTLS_PORT="$2"; MTLS_GIVEN=1; shift 2 ;; --inference-port) INFERENCE_PORT="$2"; INFERENCE_GIVEN=1; shift 2 ;; --data-root) DATA_ROOT="$2"; DATA_ROOT_GIVEN=1; shift 2 ;; --cache-dir) CACHE_DIR="$2"; CACHE_GIVEN=1; shift 2 ;; --cluster) CLUSTER="$2"; CLUSTER_GIVEN=1; shift 2 ;; --network) NETWORK="$2"; NETWORK_GIVEN=1; shift 2 ;; --profile) PROFILE="$2"; shift 2 ;; --domain) DOMAIN="$2"; shift 2 ;; --non-interactive) NON_INTERACTIVE=1; shift ;; --reset-manifest) RESET_MANIFEST=1; shift ;; --reinstall) REINSTALL=1; shift ;; --worker-binary) WORKER_BINARY="$2"; shift 2 ;; --gpuk-script) GPUK_SCRIPT="$2"; shift 2 ;; --skip-gpu-check) SKIP_GPU_CHECK=1; shift ;; --skip-preflight) SKIP_PREFLIGHT=1; shift ;; --dry-run) DRY_RUN=1; shift ;; -h|--help) usage; exit 0 ;; *) die "unknown option: $1 (try --help)" ;; esac done case "$EDITION" in community|enterprise) ;; *) die "--edition must be community or enterprise (got '$EDITION')" ;; esac case "$PROFILE" in ""|homelab|studio|enterprise|public) ;; *) die "--profile must be homelab, studio, enterprise or public (got '$PROFILE')" ;; esac case "$DOMAIN" in *[!A-Za-z0-9.-]*) die "--domain must be a bare domain name (got '$DOMAIN')" ;; esac # bridge, host, or the name of a docker network gpuk hands to `docker create # --network` — never a value that could be read as another flag or as whitespace. case "$NETWORK" in ""|bridge|host) ;; -*|*[!A-Za-z0-9_.-]*) die "--network must be bridge, host or a docker network name (got '$NETWORK')" ;; esac for _pv in "$PORT" "$MTLS_PORT" "$INFERENCE_PORT"; do case "$_pv" in ''|*[!0-9]*) die "not a port number: '$_pv'" ;; esac [ "$_pv" -ge 1 ] && [ "$_pv" -le 65535 ] || die "port out of range: $_pv" done echo echo "${BOLD}GPU Kitchen — installing the $EDITION controller${RESET}" # ── 0. An existing install (INS-03) ────────────────────────────────────────── # Re-running this script is the update path, so before checking anything it # reads what is already here — and KEEPS it. An update that silently moved the # UI to another port, re-asked the profile (Enter = homelab would downgrade a # public install) or pointed at a fresh data root beside the real one is not an # update. Read-only. The manifest is the authority (WRK-55); without it, the # traces a previous install leaves (container, unit, binary, data root) still # mean "take over", never "start beside", and the container's own ports are ours. manifest_str() { # $1 = key of a string field, anywhere in the manifest sed -n "s/.*\"$1\"[[:space:]]*:[[:space:]]*\"\([^\"]*\)\".*/\1/p" "$MANIFEST" 2>/dev/null | head -1 } manifest_binding() { # $1 = container port → the host port the ports table publishes it on sed -n "s/.*\"$1\(\/tcp\)\{0,1\}\"[[:space:]]*:[[:space:]]*\([0-9][0-9]*\).*/\2/p" "$MANIFEST" 2>/dev/null | head -1 } C_STATUS=""; C_NETMODE="" E_PORT=""; E_MTLS_PORT=""; E_LISTEN_ADDR="" E_PUBLIC_PORT=""; E_PUBLIC_MTLS_PORT=""; E_PROXY_PUBLIC_PORT="" B_8080=""; B_8443=""; B_8200="" container_facts() { # $1 = name → C_*, E_*, B_* from docker; 1 when absent command -v docker >/dev/null 2>&1 || return 1 _f=$(docker inspect -f '{{.State.Status}}|{{.HostConfig.NetworkMode}}|{{range $p, $b := .HostConfig.PortBindings}}{{range $b}}{{$p}}={{.HostPort}} {{end}}{{end}}{{"\n"}}{{range .Config.Env}}{{.}}{{"\n"}}{{end}}' "$1" 2>/dev/null) \ || return 1 [ -n "$_f" ] || return 1 _head=$(printf '%s\n' "$_f" | head -1) C_STATUS=${_head%%|*}; _r=${_head#*|}; C_NETMODE=${_r%%|*}; _bind=${_r#*|} E_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PORT=//p' | head -1) E_MTLS_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_MTLS_PORT=//p' | head -1) E_LISTEN_ADDR=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_LISTEN_ADDR=//p' | head -1) E_PUBLIC_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PUBLIC_PORT=//p' | head -1) E_PUBLIC_MTLS_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PUBLIC_MTLS_PORT=//p' | head -1) E_PROXY_PUBLIC_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PROXY_PUBLIC_PORT=//p' | head -1) B_8080=$(printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n 's/^8080\/tcp=//p' | head -1) B_8443=$(printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n 's/^8443\/tcp=//p' | head -1) B_8200=$(printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n 's/^8200\/tcp=//p' | head -1) } # The ports an install is REACHED on, with the precedence the controller itself # applies (core/published-ports.ts, INS-43): host networking moves the listeners # (GPUK_PORT, GPUK_MTLS_PORT, GPUK_LISTEN_ADDR); anything else keeps the image's # fixed listeners and publishes them (GPUK_PUBLIC_*, then the ports table). published_ports() { # $1 = network mode → OURS_UI OURS_MTLS OURS_INF if [ "$1" = "host" ]; then OURS_UI="${E_PORT:-8080}" OURS_MTLS="${E_MTLS_PORT:-8443}" OURS_INF="${E_PROXY_PUBLIC_PORT:-${E_LISTEN_ADDR##*:}}" else OURS_UI="${E_PUBLIC_PORT:-${B_8080:-8080}}" OURS_MTLS="${E_PUBLIC_MTLS_PORT:-${B_8443:-8443}}" OURS_INF="${E_PROXY_PUBLIC_PORT:-${B_8200:-8200}}" fi [ -n "$OURS_INF" ] || OURS_INF=8200 } EXISTING=""; OUR_PORTS=""; CONTAINER_NAME="gpu-kitchen"; RECONFIGURE="" reconfigure() { RECONFIGURE="${RECONFIGURE:+$RECONFIGURE, }$1"; } OURS_UI=""; OURS_MTLS=""; OURS_INF="" if [ -f "$MANIFEST" ] && [ ! -r "$MANIFEST" ]; then EXISTING="unreadable" elif [ -f "$MANIFEST" ]; then EXISTING="manifest" _cn=$(manifest_str containerName); [ -z "$_cn" ] || CONTAINER_NAME="$_cn" C_NETMODE=$(manifest_str networkMode) E_PORT=$(manifest_str GPUK_PORT); E_MTLS_PORT=$(manifest_str GPUK_MTLS_PORT) E_LISTEN_ADDR=$(manifest_str GPUK_LISTEN_ADDR) E_PUBLIC_PORT=$(manifest_str GPUK_PUBLIC_PORT) E_PUBLIC_MTLS_PORT=$(manifest_str GPUK_PUBLIC_MTLS_PORT) E_PROXY_PUBLIC_PORT=$(manifest_str GPUK_PROXY_PUBLIC_PORT) B_8080=$(manifest_binding 8080); B_8443=$(manifest_binding 8443); B_8200=$(manifest_binding 8200) published_ports "${C_NETMODE:-host}" OUR_PORTS="$OURS_UI $OURS_MTLS $OURS_INF" elif container_facts "$CONTAINER_NAME"; then EXISTING="leftovers" published_ports "$C_NETMODE" OUR_PORTS="$OURS_UI $OURS_MTLS $OURS_INF" elif [ -f "$UNIT_DEST" ] || [ -x "$BIN_DEST" ] || [ -d "$DATA_ROOT/secrets" ]; then EXISTING="leftovers" fi case "$EXISTING" in manifest) step "Existing install — $MANIFEST" ok "image $(manifest_str image)" _root=$(manifest_str dataRoot); _cache=$(manifest_str hostPath) _cluster=$(manifest_str GPUK_CLUSTER); _profile=$(manifest_str GPUK_INSTALL_PROFILE) # The edition an install runs is its image's basename (gpukitchen-controller-ee). _edition=$(manifest_str image); _edition=${_edition%@*}; _edition=${_edition##*/} case "${_edition%%:*}" in *-ee) _edition=enterprise ;; *) _edition=community ;; esac # What this run asks to CHANGE, before anything is inherited: a flag restating # the installed value changes nothing, so the same command line re-run later # is still an update. Anything else is an explicit reinstall. [ "$PORT_GIVEN" -eq 0 ] || [ "$PORT" = "$OURS_UI" ] || reconfigure "--port" [ "$MTLS_GIVEN" -eq 0 ] || [ "$MTLS_PORT" = "$OURS_MTLS" ] || reconfigure "--mtls-port" [ "$INFERENCE_GIVEN" -eq 0 ] || [ "$INFERENCE_PORT" = "$OURS_INF" ] || reconfigure "--inference-port" [ "$DATA_ROOT_GIVEN" -eq 0 ] || [ "$DATA_ROOT" = "$_root" ] || reconfigure "--data-root" [ "$CACHE_GIVEN" -eq 0 ] || [ "$CACHE_DIR" = "$_cache" ] || reconfigure "--cache-dir" [ "$CLUSTER_GIVEN" -eq 0 ] || [ "$CLUSTER" = "$_cluster" ] || reconfigure "--cluster" [ "$NETWORK_GIVEN" -eq 0 ] || [ "$NETWORK" = "${C_NETMODE:-host}" ] || reconfigure "--network" [ "$EDITION_GIVEN" -eq 0 ] || [ "$EDITION" = "$_edition" ] || reconfigure "--edition" [ -z "$PROFILE" ] || [ "$PROFILE" = "$_profile" ] || reconfigure "--profile" [ -z "$DOMAIN" ] || reconfigure "--domain" [ -z "$IMAGE" ] || reconfigure "--image" [ -z "$VERSION" ] || reconfigure "--version" [ -z "$WORKER_BINARY" ] || reconfigure "--worker-binary" [ -z "$GPUK_SCRIPT" ] || reconfigure "--gpuk-script" [ "$REINSTALL" -eq 0 ] || reconfigure "--reinstall" [ "$RESET_MANIFEST" -eq 0 ] || reconfigure "--reset-manifest" [ "$DATA_ROOT_GIVEN" -eq 1 ] || [ -z "$_root" ] || DATA_ROOT="$_root" [ "$CACHE_GIVEN" -eq 1 ] || [ -z "$_cache" ] || CACHE_DIR="$_cache" [ "$CLUSTER_GIVEN" -eq 1 ] || [ -z "$_cluster" ] || CLUSTER="$_cluster" [ "$PORT_GIVEN" -eq 1 ] || PORT="$OURS_UI" [ "$MTLS_GIVEN" -eq 1 ] || MTLS_PORT="$OURS_MTLS" [ "$INFERENCE_GIVEN" -eq 1 ] || INFERENCE_PORT="$OURS_INF" # The network mode is a setting like the ports: an install that runs on the # host network stays there when the default is bridge (INS-49), and the # reverse — only --network moves it. An old manifest without the field is # read as the daemon reads it (host). [ "$NETWORK_GIVEN" -eq 1 ] || NETWORK="${C_NETMODE:-host}" case "$_profile" in homelab|studio|enterprise|public) [ -n "$PROFILE" ] || PROFILE="$_profile" ;; esac ok "data root $DATA_ROOT" ok "ports UI $OURS_UI, worker channel $OURS_MTLS, inference $OURS_INF" ok "network ${C_NETMODE:-host}" [ -z "$_profile" ] || ok "profile $_profile" if [ -n "$RECONFIGURE" ]; then ok "reinstalling over it ($RECONFIGURE). Its other settings are kept." else ok "re-running updates it in place, through the installed gpuk. Its settings are kept." fi ;; unreadable) step "Existing install — $MANIFEST" warn "present, but not readable from here: run as root to keep its settings" ;; leftovers) step "Existing install — traces of a previous install, no manifest" if [ -n "$C_STATUS" ]; then ok "container $CONTAINER_NAME ($C_STATUS; UI $OURS_UI, worker channel $OURS_MTLS, inference $OURS_INF) — the install replaces it" fi [ ! -f "$UNIT_DEST" ] || ok "systemd unit $UNIT_DEST — rewritten" [ ! -x "$BIN_DEST" ] || ok "daemon binary $BIN_DEST — replaced" [ ! -d "$DATA_ROOT/secrets" ] || ok "data root $DATA_ROOT — reused, nothing in it is touched" warn "without $MANIFEST no setting can be kept: the flags and the defaults apply" ;; esac [ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf" # ── 0b. A plain re-run is `gpuk update` (INS-03, OPS-20) ────────────────────── # The gpuk this install left on the host carries the release key pinned when it # was installed: it verifies the signed channel against THAT key, pins the image # by digest, and lets the daemon pull, health-gate and roll back. A freshly # downloaded install.sh is trust-on-first-use again; handing the update to the # installed gpuk keeps a re-run inside the chain the first install established. # An install older than gpuk-on-the-host has nothing to hand the update to: the # only way forward is a reinstall over it. Offer it on a terminal, and always # print the complete command — never a flag to splice in by hand. if [ "$EXISTING" = "manifest" ] && [ -z "$RECONFIGURE" ] && [ "$DRY_RUN" -eq 0 ] \ && [ ! -x "$GPUK_CLI" ]; then step "Update — this install predates gpuk on the host" REINSTALL_CMD="curl -fsSL https://gpu.kitchen/install.sh | sudo sh -s --$ORIG_ARGS --reinstall" echo " It has no $GPUK_CLI to update it with. Reinstalling over it keeps its settings" echo " and leaves gpuk on the host for every later update:" echo echo " $REINSTALL_CMD" echo if can_prompt; then ask " Reinstall over it now? [Y/n] " case "$REPLY" in ""|y|Y|yes|YES|o|O|oui) REINSTALL=1; reconfigure "--reinstall" ;; *) die "nothing was changed. Run the command above when you are ready." ;; esac else die "nothing was changed. Run the command above to reinstall over it." fi fi if [ "$EXISTING" = "manifest" ] && [ -z "$RECONFIGURE" ]; then step "Update — handed to the installed gpuk" if [ "$DRY_RUN" -eq 1 ]; then step "Dry run — stopping here" echo " would run : $GPUK_CLI update" echo " existing : yes — updated in place" exit 0 fi "$GPUK_CLI" update || die "the update failed — gpuk reported the cause just above" echo echo "${BOLD}${GREEN}GPU Kitchen is up to date.${RESET}" echo echo " Status : gpuk status Logs: gpuk logs" echo " Change a setting: re-run with its flag (a reinstall), or use Settings in the UI" echo exit 0 fi if [ "$PORT" = "$MTLS_PORT" ] || [ "$PORT" = "$INFERENCE_PORT" ] || [ "$MTLS_PORT" = "$INFERENCE_PORT" ]; then die "the UI, worker channel and inference ports must differ (got $PORT, $MTLS_PORT, $INFERENCE_PORT)" fi # ── 1. Preflight ───────────────────────────────────────────────────────────── # The same checks tools/provision-feeder.sh makes, minus the compose ones: the # all-in-one image is driven by workerd through the plain docker CLI, so there is # no compose dependency to satisfy any more. if [ "$SKIP_PREFLIGHT" -eq 1 ]; then step "Preflight — skipped (--skip-preflight)" warn "the host is NOT being checked for docker, a driver or a GPU" else step "Preflight — docker, NVIDIA driver, container toolkit" # Root, or a user-owned prefix (GPUK_ETC_DIR) — the same rule as gpuk's # need_root, and the only way the full path is testable without handing root # to a test suite. [ "$DRY_RUN" -eq 1 ] || [ "$(id -u)" -eq 0 ] || [ -w "$ETC_DIR" ] \ || die "run as root: curl -fsSL … | sudo sh" command -v curl >/dev/null 2>&1 || die "curl not found. Install curl first." if command -v docker >/dev/null 2>&1; then if docker info >/dev/null 2>&1; then ok "docker daemon reachable ($(docker --version | cut -d' ' -f3 | tr -d ,))" else die "docker is installed but its daemon does not answer. Start the daemon, or run as root." fi else die "docker not found. Install Docker Engine first: https://docs.docker.com/engine/install/" fi if command -v nvidia-smi >/dev/null 2>&1; then DRIVER=$(nvidia-smi --query-gpu=driver_version --format=csv,noheader 2>/dev/null | head -1 || true) GPU_COUNT=$(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | grep -c . || true) if [ -n "$DRIVER" ] && [ "${GPU_COUNT:-0}" -gt 0 ]; then ok "NVIDIA driver $DRIVER — ${GPU_COUNT} GPU(s): $(nvidia-smi --query-gpu=name --format=csv,noheader | sort -u | paste -sd', ')" else die "nvidia-smi is present but reports no GPU" fi else die "nvidia-smi not found. Install the NVIDIA driver first." fi if [ "$SKIP_GPU_CHECK" -eq 1 ]; then warn "nvidia-container-toolkit check skipped (--skip-gpu-check)" elif docker info 2>/dev/null | grep -qiE 'runtimes:.*nvidia'; then # The runtime being REGISTERED is not the same as it working. With the toolkit # installed, --gpus injects the driver and nvidia-smi into a plain image; that # is the exact mechanism the controller container relies on, so test it rather # than infer it. if docker run --rm --gpus all ubuntu:24.04 nvidia-smi -L >/dev/null 2>&1; then ok "nvidia-container-toolkit works (a container can see the GPUs)" else die "the nvidia runtime is registered but 'docker run --gpus all' cannot see the GPUs. Reinstall nvidia-container-toolkit." fi else die "nvidia-container-toolkit is not registered with docker. Install it, then restart dockerd: https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html" fi # Model weights are large and the failure mode (a download dying at 90%) is # miserable, so say so up front. A warning, not a refusal: it is the user's disk. CACHE_PARENT="$CACHE_DIR" while [ ! -d "$CACHE_PARENT" ] && [ "$CACHE_PARENT" != "/" ]; do CACHE_PARENT=$(dirname "$CACHE_PARENT") done FREE_GB=$(df -BG --output=avail "$CACHE_PARENT" 2>/dev/null | tail -1 | tr -dc '0-9' || true) if [ "${FREE_GB:-0}" -ge 100 ]; then ok "model cache $CACHE_DIR — ${FREE_GB}G free" else warn "only ${FREE_GB:-?}G free under $CACHE_PARENT. Model weights need 100G or more." fi fi # end preflight # ── Listening ports (INS-46) ────────────────────────────────────────────────── # Deliberately OUTSIDE the preflight branch: --skip-preflight skips docker, the # driver, the GPU smoke test and the disk (things a runner or a VM cannot have), # but a taken port is exactly as fatal there, and checking it costs nothing. # Skipping it here only moved the failure to gpuk's non-interactive refusal. if [ "$SKIP_PREFLIGHT" -eq 1 ]; then step "Listening ports — checked even without the preflight" fi # A taken port must fail HERE, before anything mutates the host — today's # alternative is a 3-minute health-check timeout with zero diagnosis. Best-effort # detection (ss, then netstat); neither present is a warn, never a false red. # A listener owned by an EXISTING GPU Kitchen install is not a conflict: re-running # this script is the documented update path, and the manifest names our port. # Test hook: force the detector. The netstat fallback is unreachable on any # host that has ss (all of them, in practice), so the CI smoke pins it here to # keep its parsing honest. PORT_TOOL="${GPUK_PORT_CHECK_TOOL:-}" case "$PORT_TOOL" in ""|ss|netstat) ;; *) die "GPUK_PORT_CHECK_TOOL must be ss or netstat (got '$PORT_TOOL')" ;; esac if [ -z "$PORT_TOOL" ]; then if command -v ss >/dev/null 2>&1; then PORT_TOOL="ss" elif command -v netstat >/dev/null 2>&1; then PORT_TOOL="netstat"; fi fi port_busy() { # $1 = port → 0 iff something listens on TCP :$1 case "$PORT_TOOL" in ss) [ -n "$(ss -ltnH "sport = :$1" 2>/dev/null)" ] ;; netstat) netstat -ltn 2>/dev/null \ | awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) found=1} END{exit found?0:1}' ;; *) return 1 ;; esac } # Which container a listener belongs to, if any: the process's cgroup names the # container id (host networking — the listener IS the container's process), and a # bridged publication shows up as the container's port mapping in `docker ps` # (the host-side holder is docker-proxy, which says nothing by itself). "nginx" # is a riddle; "nginx in container gpu-kitchen-dev" is the answer. port_container() { # $1 = port, $2 = pid ("" if unknown) → container name or "" command -v docker >/dev/null 2>&1 || return 0 if [ -n "$2" ] && [ -r "/proc/$2/cgroup" ]; then _cid=$(sed -n 's#.*docker[-/]\([0-9a-f]\{64\}\).*#\1#p' "/proc/$2/cgroup" 2>/dev/null | head -1) if [ -n "$_cid" ]; then docker inspect -f '{{.Name}}' "$_cid" 2>/dev/null | sed 's|^/||' return 0 fi fi docker ps --format '{{.Names}} {{.Ports}}' 2>/dev/null \ | awk -v p=":$1->" 'index($0, p) { print $1; exit }' } port_owner() { # $1 = port → best-effort "process", "process in container NAME", or "" _proc=""; _pid="" case "$PORT_TOOL" in ss) _line=$(ss -ltnpH "sport = :$1" 2>/dev/null | head -1) _proc=$(printf '%s' "$_line" | sed -n 's/.*users:((\"\([^"]*\)\".*/\1/p') _pid=$(printf '%s' "$_line" | sed -n 's/.*pid=\([0-9]*\).*/\1/p') ;; netstat) _field=$(netstat -ltnp 2>/dev/null \ | awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) {print $NF; exit}}') case "$_field" in */*) _pid=${_field%%/*}; _proc=${_field#*/} ;; *) _proc="$_field" ;; esac ;; esac case "$_pid" in *[!0-9]*|"") _pid="" ;; esac _ctr=$(port_container "$1" "$_pid") if [ -n "$_ctr" ]; then printf '%s' "${_proc:-a process} in container $_ctr" else printf '%s' "$_proc" fi } # A port is ours when the existing install (step 0) already holds it: the # re-run replaces that container, so what it listens on is not a conflict. port_is_ours() { case " $OUR_PORTS " in *" $1 "*) return 0 ;; esac; return 1; } # Ports this run may not hand out twice: the three requested ones, plus every # alternative already accepted. Without it, a busy 8442 would be offered 8443 # and collide with the worker channel one question later. RESERVED_PORTS="$PORT $MTLS_PORT $INFERENCE_PORT" port_available() { # $1 → free on the host AND not reserved by this run case " $RESERVED_PORTS " in *" $1 "*) return 1 ;; esac port_is_ours "$1" && return 1 ! port_busy "$1" } # resolve_port

." ;; "") _want="$ALT" ;; *) case "$REPLY" in *[!0-9]*) die "not a port number: $REPLY" ;; esac port_available "$REPLY" \ || die "port $REPLY is busy, or already taken by another GPU Kitchen listener. Run the install again with $_flag

." _want="$REPLY" ;; esac RESERVED_PORTS="$RESERVED_PORTS $_want" ok "$_label port $_want is free" else die "$_label port $_want is already in use by $OWNER. Pass $_flag

to choose another port." fi else ok "$_label port $_want is free" fi RESOLVED="$_want" } if [ -z "$PORT_TOOL" ]; then warn "cannot check for port conflicts (neither ss nor netstat found)" else resolve_port "UI" "$PORT" "--port"; PORT="$RESOLVED" resolve_port "worker channel" "$MTLS_PORT" "--mtls-port"; MTLS_PORT="$RESOLVED" resolve_port "inference" "$INFERENCE_PORT" "--inference-port"; INFERENCE_PORT="$RESOLVED" fi # ── 2. Resolve the release ─────────────────────────────────────────────────── step "Release — resolving the version to install" # Everything fetched from the channel lands here, and is verified here, before use. TMP=$(mktemp -d) # shellcheck disable=SC2064 # expand TMP now: it must be removed even if it changes trap "rm -rf '$TMP'" EXIT INT TERM # The trust chain (OPS-20): latest.json is signed with the release key, and so is # every daemon binary; latest.json names the gpuk installer's sha256 and the # image digests. Fail-closed: no pinned key, no minisign, no or a bad signature # are all fatal. GPUK_CHANNEL_INSECURE=1 is the explicit, loudly reported # opt-out for a private mirror that does not sign — gpuk's own opt-out. INSECURE=0; [ "${GPUK_CHANNEL_INSECURE:-}" != "1" ] || INSECURE=1 # The minisign CLI is what verifies the release; a host without it gets it from # its own package manager — detected, non-interactive, quiet — before anything # else changes. No known manager, or an install that fails: stop, naming the # manual command. GPUK_MINISIGN names another verifier binary (test seam). MINISIGN="${GPUK_MINISIGN:-minisign}" minisign_manual_command() { if command -v apt-get >/dev/null 2>&1; then echo "apt-get install minisign" elif command -v dnf >/dev/null 2>&1; then echo "dnf install minisign (EPEL on RHEL)" elif command -v yum >/dev/null 2>&1; then echo "yum install minisign (EPEL)" elif command -v zypper >/dev/null 2>&1; then echo "zypper install minisign" elif command -v apk >/dev/null 2>&1; then echo "apk add minisign" elif command -v pacman >/dev/null 2>&1; then echo "pacman -S minisign" else echo "install minisign from https://jedisct1.github.io/minisign/" fi } ensure_minisign() { # $1 = 1 when nothing may be installed (dry run) command -v "$MINISIGN" >/dev/null 2>&1 && return 0 [ "${1:-0}" -eq 0 ] \ || die "minisign is required to verify the release and a dry run installs nothing: $(minisign_manual_command), or set GPUK_CHANNEL_INSECURE=1" _mlog=$(mktemp) for _pm in apt-get dnf yum zypper apk pacman; do command -v "$_pm" >/dev/null 2>&1 || continue ok "minisign is missing — installing it with $_pm" case "$_pm" in apt-get) { DEBIAN_FRONTEND=noninteractive apt-get update -qq \ && DEBIAN_FRONTEND=noninteractive apt-get install -y -qq minisign; } ;; dnf) dnf install -y -q minisign ;; yum) yum install -y -q minisign ;; zypper) zypper --non-interactive --quiet install minisign ;; apk) apk add --quiet minisign ;; pacman) pacman -S --noconfirm --needed --quiet minisign ;; esac >"$_mlog" 2>&1 || true if command -v "$MINISIGN" >/dev/null 2>&1; then rm -f "$_mlog"; return 0; fi done tail -5 "$_mlog" >&2 2>/dev/null || true rm -f "$_mlog" die "minisign is required to verify the release and could not be installed automatically. Nothing was changed: run '$(minisign_manual_command)' as root, then re-run — or set GPUK_CHANNEL_INSECURE=1" } require_verifier() { case "$CHANNEL_PUBKEY" in # The unstamped placeholder, matched by its prefix only: the release stamps # the key by substituting the whole placeholder wherever it appears, and a # guard spelling it in full would become one refusing the very key it pinned. ""|__GPUK_UPDATE_*) die "this install.sh carries no pinned release public key: use the channel's copy, set GPUK_UPDATE_PUBKEY (the minisign public-key line), or GPUK_CHANNEL_INSECURE=1 to skip verification" ;; esac ensure_minisign "$DRY_RUN" } verify_signature() { # $1 = file, $2 = its .minisig, $3 = what it is "$MINISIGN" -Vq -m "$1" -x "$2" -P "$CHANNEL_PUBKEY" >/dev/null 2>&1 \ || die "$3: signature verification FAILED against the pinned release key — refusing it (OPS-20)" } # One tiny JSON document, fetched over TLS, holding what the current release IS. # Parsed with sed rather than jq: `curl … | sudo sh` cannot assume jq exists, and # the document is ours and flat. json_field() { sed -n "s/.*\"$1\"[[:space:]]*:[[:space:]]*\"\([^\"]*\)\".*/\1/p" | head -1; } # Strip a tag off an image reference WITHOUT mangling a registry's host:port. # `${ref%%:*}` cuts at the FIRST colon and is WRONG: a # `registry.internal:5000/gpuk/controller:v1.2.3` (or an untagged # `registry.internal:5000/gpuk/controller`) would collapse to `registry.internal`. # A colon is a tag separator only when the last colon comes AFTER the last slash. # Same rule as deployments/install/gpuk's image_repo() and core/release-channel.ts (unit-tested). image_repo() { case "$1" in *@*) printf '%s' "${1%@*}"; return ;; # digest pin → repo@sha256:… esac _t="${1##*:}" case "$_t" in "$1") printf '%s' "$1" ;; # no colon at all → already untagged */*) printf '%s' "$1" ;; # last colon is inside a path → host:port, untagged *) printf '%s' "${1%:*}" ;; esac } if [ -n "$IMAGE" ]; then ok "using the image given on the command line: $IMAGE" [ -n "$VERSION" ] || VERSION="(pinned by --image)" else CHANNEL_DOC="$TMP/latest.json" curl -fsSL --max-time 20 "$CHANNEL_URL" -o "$CHANNEL_DOC" 2>/dev/null \ || die "cannot reach the release channel at $CHANNEL_URL If this host is offline, pass --image to install a specific image directly." if [ "$INSECURE" -eq 1 ]; then warn "GPUK_CHANNEL_INSECURE=1 — the release channel and what it names are NOT verified" else require_verifier curl -fsSL --max-time 20 "$CHANNEL_URL.minisig" -o "$CHANNEL_DOC.minisig" 2>/dev/null \ || die "no signature at $CHANNEL_URL.minisig — refusing an unsigned channel document (OPS-20)" verify_signature "$CHANNEL_DOC" "$CHANNEL_DOC.minisig" "the release channel ($CHANNEL_URL)" ok "release channel signature verified" fi CHANNEL_VERSION=$(json_field version < "$CHANNEL_DOC") [ -n "$CHANNEL_VERSION" ] || die "the release channel returned no version: $CHANNEL_URL" if [ "$EDITION" = "enterprise" ]; then IMAGE_TEMPLATE=$(json_field controllerImageEnterprise < "$CHANNEL_DOC") IMAGE_DIGEST=$(json_field controllerImageDigestEnterprise < "$CHANNEL_DOC") else IMAGE_TEMPLATE=$(json_field controllerImage < "$CHANNEL_DOC") IMAGE_DIGEST=$(json_field controllerImageDigest < "$CHANNEL_DOC") fi [ -n "$IMAGE_TEMPLATE" ] \ || die "the release channel names no $EDITION controller image: $CHANNEL_URL" # The signed document vouches for ONE release: its digest belongs to that # version and no other. Another release is installed by its full reference. if [ -n "$VERSION" ] && [ "$VERSION" != "$CHANNEL_VERSION" ]; then [ "$INSECURE" -eq 1 ] \ || die "the signed channel vouches for $CHANNEL_VERSION only, not $VERSION. To install another release, pass its full reference: --image :$VERSION@sha256:" IMAGE_DIGEST="" fi [ -n "$VERSION" ] || VERSION="$CHANNEL_VERSION" case "$IMAGE_DIGEST" in sha256:*) _hex=${IMAGE_DIGEST#sha256:} case "$_hex" in *[!0-9a-f]*) die "the release channel carries a malformed image digest: $IMAGE_DIGEST" ;; esac [ "${#_hex}" -eq 64 ] || die "the release channel carries a malformed image digest: $IMAGE_DIGEST" ;; "") [ "$INSECURE" -eq 1 ] \ || die "the release channel names no $EDITION image digest: refusing to pin a mutable tag (OPS-20)" ;; *) die "the release channel carries a malformed image digest: $IMAGE_DIGEST" ;; esac # The channel gives the repository; WE pin the tag and, from the signed # document, the content: `repo:tag@sha256:…` keeps the tag readable while # docker resolves by digest, so a registry cannot swap what the tag points at. # A floating `:latest` would make every recreate a silent, unrequested upgrade. IMAGE="$(image_repo "$IMAGE_TEMPLATE"):${VERSION}${IMAGE_DIGEST:+@$IMAGE_DIGEST}" ok "release $VERSION" ok "image $IMAGE" [ -n "$IMAGE_DIGEST" ] || warn "the image is pinned by tag only (no digest in an unverified channel)" [ -n "$WORKER_BINARY" ] || WORKER_RELEASE_BASE=$(json_field workerBase < "$CHANNEL_DOC") if [ -z "${GPUK_SCRIPT}" ]; then GPUK_SCRIPT_URL=$(json_field gpukScript < "$CHANNEL_DOC") GPUK_SCRIPT_SHA256=$(json_field gpukScriptSha256 < "$CHANNEL_DOC") fi fi # A direct --image and a channel version are held to the same immutable-image # rule. A registry host:port is not a tag separator; only the last colon after # the last slash counts. Digest pins are accepted too. case "$IMAGE" in *@sha256:*) ;; *:latest) die "refusing floating image tag '$IMAGE'. Use an explicit release tag or digest." ;; *) IMAGE_TAG="${IMAGE##*:}" case "$IMAGE_TAG" in "$IMAGE"|*/*) die "controller image must carry an explicit tag or sha256 digest (got '$IMAGE')" ;; esac ;; esac if [ "$DRY_RUN" -eq 1 ]; then step "Dry run — stopping here" if [ "$SKIP_PREFLIGHT" -eq 1 ]; then ok "the release resolved; the host was not checked; nothing was installed" else ok "preflight passed and the release resolved; nothing was installed" fi echo echo " would install : $IMAGE" echo " data root : $DATA_ROOT" echo " model cache : $CACHE_DIR" echo " UI port : $PORT" echo " worker channel: $MTLS_PORT" echo " inference : $INFERENCE_PORT" echo " network : ${NETWORK:-bridge}" case "$EXISTING" in manifest) echo " existing : yes — updated in place" ;; leftovers) echo " existing : traces of a previous install — taken over" ;; *) echo " existing : no" ;; esac if [ -n "$PROFILE" ]; then echo " profile : $PROFILE" else echo " profile : (asked on the terminal, or chosen during first run)" fi [ -z "$DOMAIN" ] || echo " domain : $DOMAIN" exit 0 fi # ── 3. Resolve the installation profile (INS-45) ───────────────────────────── # Only ever on a terminal, and only when --profile was not given. The answer is # relayed verbatim as --profile: the backend stays the sole applier of profile # defaults and floors. Without a terminal the historical path is untouched — # profile unset, enterprise floors, the first-run wizard requires the choice # (PRF-05). if [ -z "$PROFILE" ] && can_prompt; then step "Security profile — how will this kitchen be used?" cat > /dev/tty <<'PROFILES' 1) Home lab a trusted home network No sign-in on your home network. Nearby workers are found and join without waiting for approval. 2) Studio one control station, shared compute Control stays on this machine. Colleagues use API keys, while nearby workers wait for your approval. 3) Enterprise a managed company network Sign-in is required. The controller stays quiet on the network; known workers can request your approval. 4) Public server direct internet exposure Sign-in and hardened browser transport are required. Network discovery is off and workers join only by token. You can change this later in Settings. Stricter floors re-apply. Nothing already issued is revoked. PROFILES while :; do ask " Choose a profile [1-4, Enter = 1 (Home lab)]: " case "$REPLY" in ""|1|homelab) PROFILE="homelab" ;; 2|studio) PROFILE="studio" ;; 3|enterprise) PROFILE="enterprise" ;; 4|public) PROFILE="public" ;; *) printf '%s\n' " pick 1, 2, 3 or 4" > /dev/tty; continue ;; esac break done ok "profile: $PROFILE" fi # Under public, a domain lets gpuk write a FILLED TLS reverse-proxy example # (INS-47). Optional: Enter skips, and the banner still points at the shipped # Caddyfile.example. if [ "$PROFILE" = "public" ] && [ -z "$DOMAIN" ] && can_prompt; then while :; do ask " Domain for HTTPS access (e.g. gpu.example.com — Enter to skip): " # Accept the copy-paste reflex: strip a pasted scheme and anything after # the first slash, then insist on a bare domain rather than skipping — # a silently dropped answer would be discovered hours later, at DNS time. REPLY="${REPLY#https://}" REPLY="${REPLY#http://}" REPLY="${REPLY%%/*}" [ -n "$REPLY" ] || break case "$REPLY" in *[!A-Za-z0-9.-]*) printf '%s\n' " not a bare domain name: $REPLY. Try again, or press Enter to skip." > /dev/tty ;; *) DOMAIN="$REPLY"; break ;; esac done fi # ── 4. The host daemon ─────────────────────────────────────────────────────── step "Host daemon — gpu-kitchen-worker" if [ -z "$GPUK_SCRIPT" ]; then # A checkout right here beats a download (that is how contributors run it) — # but ONLY a checkout. Piped into `sh`, $0 is `sh` and dirname "$0" is the # current directory: a stray `./gpuk` in /tmp or a shared folder would run as # root, unverified. So the local copy counts only when this script was run by # its path, next to its gpuk, inside a repository checkout (dev.sh two levels # up, deployments/install/); everything else downloads (INS-03). _local="" case "$0" in install.sh|*/install.sh) _here=$(dirname "$0") [ -f "$_here/gpuk" ] && [ -f "$_here/../../dev.sh" ] && _local="$_here/gpuk" ;; esac if [ -n "$_local" ]; then GPUK_SCRIPT="$_local" ok "using the gpuk installer from this checkout ($_local)" else [ -n "${GPUK_SCRIPT_URL:-}" ] \ || die "the release channel names no gpuk installer, and none was found locally" curl -fsSL --max-time 60 "$GPUK_SCRIPT_URL" -o "$TMP/gpuk" \ || die "cannot download the gpuk installer from $GPUK_SCRIPT_URL" # The signed channel names the installer's sha256: the bytes that run as root # next are the ones the release signed for, wherever they were served from. if [ -n "${GPUK_SCRIPT_SHA256:-}" ]; then _sum=$(sha256sum "$TMP/gpuk" | cut -d' ' -f1) [ "$_sum" = "$GPUK_SCRIPT_SHA256" ] \ || die "the gpuk installer from $GPUK_SCRIPT_URL does not match the signed channel (sha256 $_sum)" ok "downloaded the gpuk installer (sha256 matches the signed channel)" else [ "$INSECURE" -eq 1 ] \ || die "the release channel names no gpukScriptSha256: refusing an unverified gpuk installer (OPS-20)" warn "downloaded the gpuk installer, NOT verified (GPUK_CHANNEL_INSECURE=1)" fi chmod +x "$TMP/gpuk" GPUK_SCRIPT="$TMP/gpuk" fi fi # `gpuk install` does the rest: it drops the binary, writes the systemd unit, # writes the manifest (the declarative description of the app container) and # applies it. Everything below is passed straight through to it. set -- install \ --mode controller \ --image "$IMAGE" \ --cluster "$CLUSTER" \ --data-root "$DATA_ROOT" \ --cache-dir "$CACHE_DIR" \ --http-port "$PORT" \ --mtls-port "$MTLS_PORT" \ --inference-port "$INFERENCE_PORT" [ -z "$PROFILE" ] || set -- "$@" --profile "$PROFILE" [ -z "$DOMAIN" ] || set -- "$@" --domain "$DOMAIN" [ "$RESET_MANIFEST" -eq 0 ] || set -- "$@" --reset-manifest # Inherited from the manifest or given by flag; unset on a first install, so # gpuk's own default (bridge, INS-49) applies and this script never restates it. [ -z "$NETWORK" ] || set -- "$@" --network "$NETWORK" if [ -n "$WORKER_BINARY" ]; then [ -f "$WORKER_BINARY" ] || die "no such worker binary: $WORKER_BINARY" set -- "$@" --binary "$WORKER_BINARY" ok "using a locally-built gpu-kitchen-worker" elif [ -n "${WORKER_RELEASE_BASE:-}" ]; then # The privileged daemon: downloaded here, its minisign signature checked # against the pinned release key, and only then handed to gpuk (OPS-20). case "$(uname -m)" in x86_64|amd64) _asset="gpu-kitchen-worker-x86_64"; _build_key="workerBuildIdX86_64" ;; aarch64|arm64) _asset="gpu-kitchen-worker-aarch64"; _build_key="workerBuildIdAarch64" ;; *) die "unsupported architecture: $(uname -m)" ;; esac ok "downloading $_asset from the release" curl -fL --progress-bar "$WORKER_RELEASE_BASE/$_asset" -o "$TMP/gpu-kitchen-worker" \ || die "cannot download $WORKER_RELEASE_BASE/$_asset" if [ "$INSECURE" -eq 1 ]; then warn "gpu-kitchen-worker NOT verified (GPUK_CHANNEL_INSECURE=1)" else curl -fsSL --max-time 60 "$WORKER_RELEASE_BASE/$_asset.minisig" -o "$TMP/gpu-kitchen-worker.minisig" \ || die "no signature at $WORKER_RELEASE_BASE/$_asset.minisig — refusing an unsigned daemon binary" verify_signature "$TMP/gpu-kitchen-worker" "$TMP/gpu-kitchen-worker.minisig" "gpu-kitchen-worker ($_asset)" # The signature binds artifact AND build (trusted comment # `gpu-kitchen-worker@`, P2-29): the build must be the one the signed # channel names for this release, so an older signed binary served in its # place is refused, and never older than the daemon already installed. _build=$(json_field "$_build_key" < "$CHANNEL_DOC") [ -n "$_build" ] \ || die "the release channel names no $_build_key: cannot tell which gpu-kitchen-worker build it vouches for (OPS-20)" _comment=$("$MINISIGN" -V -m "$TMP/gpu-kitchen-worker" -x "$TMP/gpu-kitchen-worker.minisig" -P "$CHANNEL_PUBKEY" 2>/dev/null \ | sed -n 's/^Trusted comment: //p' | head -1) [ "$_comment" = "gpu-kitchen-worker@$_build" ] \ || die "gpu-kitchen-worker ($_asset) is signed for '${_comment:-nothing}', not gpu-kitchen-worker@$_build — refusing it (OPS-20)" if [ -x "$BIN_DEST" ]; then _installed=$("$BIN_DEST" --version 2>/dev/null | sed -n 's/^gpu-kitchen-worker //p' | head -1) case "$_installed" in [0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9]-*) # buildId = YYYYMMDDHHMMSS-: the timestamp orders builds. [ "${_build%%-*}" -ge "${_installed%%-*}" ] \ || die "gpu-kitchen-worker $_build is OLDER than the installed $_installed — refusing a downgrade" ;; esac fi ok "gpu-kitchen-worker signature verified (build $_build)" fi set -- "$@" --binary "$TMP/gpu-kitchen-worker" fi # else: gpuk reuses an already-installed binary, or fails with its own message. step "Installing — this pulls the image, so it can take a few minutes" # gpuk names its own failure on stderr before exiting (a refused manifest, a # failed apply, a missing binary…), so the cause is the line right above this # one. journalctl only has something to say once the daemon has started, and # gpuk points at `gpuk logs` itself in that case. sh "$GPUK_SCRIPT" "$@" || die "the install failed — gpuk reported the cause just above" # ── 5. Wait for the app, then say where it is ──────────────────────────────── step "Waiting for the controller to answer" i=0 until curl -fsS -o /dev/null --max-time 3 "http://127.0.0.1:$PORT/health" 2>/dev/null; do i=$((i + 1)) [ "$i" -lt 90 ] || die "the controller did not answer on :$PORT within 90s. See: gpuk logs" sleep 2 done ok "the controller is up" # The URL must name the REAL host: the person installing this is very often not # sitting at the machine, and "localhost" would be a lie on every box but theirs # (same reason the backend resolves its own hostname — core/host-name.ts). HOSTNAME_FQDN=$(hostname -f 2>/dev/null || hostname 2>/dev/null || echo localhost) LAN_IP=$(ip route get 1.1.1.1 2>/dev/null | sed -n 's/.* src \([0-9.]*\).*/\1/p' | head -1) echo echo "${BOLD}${GREEN}GPU Kitchen is running.${RESET}" echo echo " ${BOLD}Open:${RESET} http://${HOSTNAME_FQDN}:${PORT}" [ -n "$LAN_IP" ] && echo " http://${LAN_IP}:${PORT}" echo # [PRF-12] [SEC-53] homelab and studio open straight into the app: the controller # closed their claim window at first boot, so there is no code to show. Every other # profile — or none yet — starts with the claim code (SEC-52). case "$PROFILE" in homelab|studio) echo " ${BOLD}Next:${RESET} open the URL above: GPU Kitchen opens straight away." ;; *) CLAIM_CODE_FILE="$DATA_ROOT/secrets/claim_code" CLAIM_CODE="" [ ! -s "$CLAIM_CODE_FILE" ] || CLAIM_CODE=$(cat "$CLAIM_CODE_FILE" 2>/dev/null || true) if [ -n "$CLAIM_CODE" ]; then echo " ${BOLD}Claim code:${RESET} $CLAIM_CODE" echo " ${BOLD}Read again:${RESET} $CLAIM_CODE_FILE (mode 0600; removed after claim)" else echo " ${BOLD}Claim code:${RESET} consumed (the first account already exists)" fi echo echo " ${BOLD}Next:${RESET} open the URL above and enter the claim code." ;; esac # [INS-50] Linking to a GPU Kitchen account is never a first-run step: a banner in # the app offers it, and the person approves on gpu.kitchen — no key is copied here, # none ever travels in a URL (CPT-05/06). echo " A banner in the app connects this installation to your GPU Kitchen account:" echo " \"Connect to my account\" or \"Create an account\" on the GPU.Kitchen hub (gpu.kitchen) — nothing to copy." case "$PROFILE" in public) echo " ${BOLD}Then:${RESET} create the first administrator in the UI" echo # The public profile REQUIRES a TLS reverse proxy (OPS-13, INS-47) — the UI # port speaks plain HTTP. The product cannot verify the proxy's presence # (OPS-67), so the closest thing to enforcement is saying it here, clearly. if [ -n "$DOMAIN" ]; then echo " ${BOLD}HTTPS:${RESET} https://${DOMAIN} answers after these steps:" echo " 1. DNS: point ${DOMAIN} at this machine's public IP" echo " 2. Install Caddy: https://caddyserver.com/docs/install" echo " 3. sudo cp $DATA_ROOT/caddy/Caddyfile /etc/caddy/Caddyfile" echo " sudo systemctl reload caddy" echo " A filled example was written to $DATA_ROOT/caddy/Caddyfile." echo " Until then the UI answers in cleartext on the URLs above." else echo " ${BOLD}HTTPS:${RESET} the public profile requires a TLS reverse proxy before any" echo " public exposure. See deployments/controller/Caddyfile.example." fi echo ;; enterprise) echo " ${BOLD}Then:${RESET} create the first administrator in the UI" echo ;; homelab|studio) echo ;; "") echo " ${BOLD}Then:${RESET} choose an installation profile in the first-run assistant" echo ;; esac # Any profile may be reached by a name through a reverse proxy (INS-47); the # public case above already printed its steps. if [ -n "$DOMAIN" ] && [ "$PROFILE" != "public" ]; then echo " ${BOLD}HTTPS:${RESET} https://${DOMAIN} once your reverse proxy serves it; a filled" echo " Caddy example was written to $DATA_ROOT/caddy/Caddyfile." echo fi echo " Inference endpoint : http://${HOSTNAME_FQDN}:${INFERENCE_PORT}/v1" echo " Version : ${VERSION}" echo # Pre-existing model caches (B81). `gpuk install` prints the same hint, but that # scrolls past mid-install; this banner is where people actually look. Detection # only — cache questions belong to the first-run wizard (INS-48, REG-41), never # to the CLI, and nothing outside $CACHE_DIR is touched, read as configuration, # or modified. EXISTING_CACHE="" CHOSEN_CACHE=$(readlink -f "$CACHE_DIR" 2>/dev/null || echo "$CACHE_DIR") for d in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do [ -n "$d" ] && [ -d "$d/hub" ] || continue real=$(readlink -f "$d" 2>/dev/null || echo "$d") [ "$real" != "$CHOSEN_CACHE" ] || continue ls -d "$d"/hub/models--* >/dev/null 2>&1 || continue EXISTING_CACHE="$EXISTING_CACHE $real" done if [ -n "$EXISTING_CACHE" ]; then for d in $EXISTING_CACHE; do echo " ${BOLD}Existing model cache:${RESET} $d" done echo " GPU Kitchen will offer to reuse those models at first launch, and any" echo " time from Nodes & GPU -> Storage. A reused cache is referenced in" echo " place. Nothing is moved or deleted." echo fi echo " Update : re-run this command, or press Update in the UI, or: gpuk update" echo " Status : gpuk status Logs: gpuk logs" echo " Remove : gpuk uninstall (service only) or gpuk uninstall --purge (all but the data root)" echo } main "$@"