From 406d457b8da3e9e46fc707e3a247a474d59c4e85 Mon Sep 17 00:00:00 2001 From: gpuk-release Date: Sun, 13 Sep 2026 21:53:54 +0000 Subject: [PATCH] release v0.1.3 --- gpuk | 493 +++++++++++++++++++++++++++++++++++++------- install.sh | 346 ++++++++++++++++++++++++++++--- latest.json | 8 +- latest.json.minisig | 4 + 4 files changed, 737 insertions(+), 114 deletions(-) create mode 100644 latest.json.minisig diff --git a/gpuk b/gpuk index 193ab38..9a43995 100755 --- a/gpuk +++ b/gpuk @@ -12,7 +12,7 @@ # # Two roles: # worker a headless compute node. Installs the binary + systemd unit + a -# manifest (cacheDisks, browseRoots) with NO app container, NO +# manifest (cacheDisks) with NO app container, NO # Postgres, NO docker app image. It ENROLLS over mTLS with a # enrollment token, or waits for LAN discovery admission, then runs. # controller the full app + UI. Installs the binary + systemd unit + an @@ -44,9 +44,10 @@ BIN_DEST="${GPUK_BIN_DEST:-/usr/local/bin/gpu-kitchen-worker}" ETC_DIR="${GPUK_ETC_DIR:-/etc/gpu-kitchen}" MANIFEST="$ETC_DIR/manifest.json" IDENTITY_DIR="${GPUK_IDENTITY_DIR:-$ETC_DIR/identity}" +MACHINE_ID_FILE="${GPUK_MACHINE_ID_FILE:-$ETC_DIR/machine-id}" WORKER_ENV="$ETC_DIR/worker.env" SERVICE_NAME="gpu-kitchen-worker" -UNIT_DEST="/etc/systemd/system/$SERVICE_NAME.service" +UNIT_DEST="${GPUK_UNIT_DEST:-/etc/systemd/system/$SERVICE_NAME.service}" # Where to download the binary from when no --binary is given. GPUK_RELEASE_BASE="${GPUK_RELEASE_BASE:-}" @@ -102,63 +103,120 @@ install_binary() { # [local-path] } gen_secret() { head -c 32 /dev/urandom | base64 | tr '+/' '-_' | tr -d '='; } -# A password a human retypes once, from a terminal. Ambiguous glyphs removed. -gen_password() { head -c 24 /dev/urandom | base64 | tr -d '=+/OIl01' | cut -c1-16; } -seed_secrets() { # DATA_ROOT MODE +seed_secrets() { # DATA_ROOT _sd="$1/secrets" mkdir -p "$_sd"; chmod 0700 "$_sd" [ -f "$_sd/encryption_key" ] || { umask 077; gen_secret > "$_sd/encryption_key"; } [ -f "$_sd/node_id" ] || { umask 077; (cat /proc/sys/kernel/random/uuid 2>/dev/null || uuidgen) > "$_sd/node_id"; } - # The controller's first-run password. Generated here — NOT left to the backend - # to print into a log nobody watches when the install is one piped command. The - # installer prints it once, and the first-run wizard makes the operator replace - # it (GPUK_BOOTSTRAP_MUST_CHANGE). - if [ "$2" = "controller" ] && [ ! -f "$_sd/bootstrap_admin_password" ]; then - umask 077; gen_password > "$_sd/bootstrap_admin_password" + # No nominal first-run password is generated. An operator may pre-provision + # this documented break-glass file; only then is it injected into the app. + [ ! -f "$_sd/bootstrap_admin_password" ] || chmod 0600 "$_sd/bootstrap_admin_password" +} + +prepare_claim_code() { + # The file is the operator's recoverable proof of machine possession. Create + # it once, preserve it across reinstalls, and let the backend unlink it after + # the atomic first-account claim. A missing file beside an existing manifest + # therefore means "consumed", never "rotate the credential". + if [ -f "$CLAIM_CODE_FILE" ]; then + chmod 0600 "$CLAIM_CODE_FILE" + elif [ ! -f "$MANIFEST" ]; then + umask 077 + gen_secret > "$CLAIM_CODE_FILE" + chmod 0600 "$CLAIM_CODE_FILE" fi } -# Write the systemd unit: prefer a sibling file, else embed. The daemon runs the -# binary with NO arguments (steady-state); role/controller URL/CA come from the -# enrolled identity (worker) and the optional EnvironmentFile. +# Write the systemd unit generated from the worker's canonical template. The +# generator injects the Rust lock-contention exit code here too, so the binary +# and systemd restart policy cannot silently drift apart. write_unit() { - _src_unit="$(dirname "$0")/../../apps/worker/install/$SERVICE_NAME.service" - [ -f "$_src_unit" ] || _src_unit="$(dirname "$0")/$SERVICE_NAME.service" - if [ -f "$_src_unit" ]; then - install -m 0644 "$_src_unit" "$UNIT_DEST" - else - cat > "$UNIT_DEST" < "$UNIT_DEST" <`. Kept aligned with +# deployments/controller/Caddyfile.example (the compose variant); this copy +# targets the all-in-one image, where nginx on the UI port is the single front +# door (INS-09) so one upstream carries pages, /api and the /ws upgrade alike. +# We write a file and NOTHING more: no package install, no service start, no +# other program's config read or touched — putting the proxy in service stays an +# operator act (OPS-13), and its presence stays unverifiable (OPS-67). +write_caddyfile() { + mkdir -p "$DATA_ROOT/caddy" + cat > "$DATA_ROOT/caddy/Caddyfile" <worker data transfers (:8300): LAN-only by contract (OPS-68). + +$DOMAIN { + encode zstd gzip + reverse_proxy localhost:$HTTP_PORT +} + +# OpenAI-compatible inference endpoint (gpuk-proxy) — uncomment when inference +# clients live beyond the trusted LAN; TLS keeps their API keys off the wire. +# +# inference.$DOMAIN { +# encode zstd gzip +# reverse_proxy localhost:8200 +# } +CADDY + chmod 0644 "$DATA_ROOT/caddy/Caddyfile" + echo "==> wrote $DATA_ROOT/caddy/Caddyfile (filled TLS reverse-proxy example for $DOMAIN)" } # Pre-existing model caches (B81): a server that installs GPU Kitchen usually already -# holds tens or hundreds of GB of weights. We print a hint and nothing more — the -# install stays NON-INTERACTIVE, no config of any other program is read or touched, -# and referencing a cache is an explicit choice made later in the UI. +# holds tens or hundreds of GB of weights. We print a hint and nothing more — cache +# questions belong to the first-run wizard (INS-48, REG-41), never to the CLI; no +# config of any other program is read or touched, and referencing a cache stays an +# explicit choice made in the UI. hint_existing_caches() { hec_chosen=$(readlink -f "$1" 2>/dev/null || echo "$1") hec_found="" @@ -176,20 +234,23 @@ hint_existing_caches() { for hec_dir in $hec_found; do echo "==> existing model cache found at $hec_dir" done - echo " Reference it from Settings -> Cache folders to reuse those models." - echo " GPU Kitchen only READS a referenced cache: nothing is moved or deleted." + echo " GPU Kitchen will offer to reuse those models at first launch, and any" + echo " time from Nodes & GPU -> Storage. A reused cache is referenced in" + echo " place. Nothing is moved or deleted." } cmd_install() { - need_root IMAGE=""; MODE="worker"; CLUSTER="default"; DATA_ROOT="/var/lib/gpu-kitchen" CACHE_DIR=""; NETWORK="host"; CONTROLLER_URL=""; BIN_SRC=""; HEALTH_PORT="8001" - BROWSE_ROOTS=""; ENROLL_TOKEN=""; HTTP_PORT="8080" + # 1337, not 8080: kept in lockstep with install.sh's default (INS-01). + ENROLL_TOKEN=""; HTTP_PORT="1337"; PROFILE=""; DOMAIN=""; DRY_RUN=0 while [ $# -gt 0 ]; do case "$1" in --image) IMAGE="$2"; shift 2 ;; --mode) MODE="$2"; shift 2 ;; --cluster) CLUSTER="$2"; shift 2 ;; + --profile) PROFILE="$2"; shift 2 ;; + --domain) DOMAIN="$2"; shift 2 ;; --data-root) DATA_ROOT="$2"; shift 2 ;; --cache-dir) CACHE_DIR="$2"; shift 2 ;; --network) NETWORK="$2"; shift 2 ;; @@ -201,7 +262,7 @@ cmd_install() { --health-port) HEALTH_PORT="$2"; shift 2 ;; --http-port) HTTP_PORT="$2"; shift 2 ;; --binary) BIN_SRC="$2"; shift 2 ;; - --browse-root) BROWSE_ROOTS="$BROWSE_ROOTS $2"; shift 2 ;; + --dry-run) DRY_RUN=1; shift ;; *) die "unknown install option: $1" ;; esac done @@ -215,20 +276,98 @@ cmd_install() { *) die "--mode must be controller or worker (got '$MODE')" ;; esac - [ "$MODE" != "controller" ] || [ -n "$IMAGE" ] \ - || die "--image registry/gpukitchen-controller: is required for a controller" + if [ "$MODE" = "worker" ] && [ -n "$PROFILE" ]; then + die "--profile applies only to --mode controller; workers do not have an installation profile" + fi + case "$PROFILE" in + ""|homelab|studio|enterprise|public) ;; + *) die "--profile must be homelab, studio, enterprise or public (got '$PROFILE')" ;; + esac + + if [ -n "$DOMAIN" ]; then + [ "$PROFILE" = "public" ] \ + || die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)" + case "$DOMAIN" in + *[!A-Za-z0-9.-]*) die "--domain must be a bare domain name (got '$DOMAIN')" ;; + esac + fi + + if [ "$MODE" = "controller" ]; then + if [ -z "$IMAGE" ] && [ "$DRY_RUN" -eq 1 ]; then + IMAGE="example.invalid/gpukitchen-controller:v0.0.0-dry-run" + fi + [ -n "$IMAGE" ] || die "--image registry/gpukitchen-controller: is required for a controller" + case "$IMAGE" in + *@sha256:*) ;; + *:latest) die "refusing floating image tag '$IMAGE' — use an explicit release tag or digest" ;; + *) + _image_tag="${IMAGE##*:}" + case "$_image_tag" in + "$IMAGE"|*/*) die "controller image must carry an explicit tag or sha256 digest (got '$IMAGE')" ;; + esac + ;; + esac + fi [ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf" + CACHE_JSON="[{\"hostPath\":\"$(json_str "$CACHE_DIR")\",\"shared\":false}]" + SELF_ENROLL_FILE="$DATA_ROOT/self-enroll-token" + BOOTSTRAP_PASSWORD_FILE="$DATA_ROOT/secrets/bootstrap_admin_password" + CLAIM_CODE_FILE="$DATA_ROOT/secrets/claim_code" + CLAIM_CODE_AVAILABLE=0 + + if [ "$DRY_RUN" -eq 1 ]; then + [ "$MODE" != "controller" ] || CLAIM_CODE_AVAILABLE=1 + echo "==> dry run: no file, service or container was changed" + if [ "$MODE" = "worker" ]; then render_worker_manifest; else render_controller_manifest; fi + [ -z "$DOMAIN" ] || echo "==> dry run: would write $DATA_ROOT/caddy/Caddyfile for $DOMAIN" + return 0 + fi + + # ── Port conflicts (INS-46) — the mutator's own guard ──────────────────────── + # install.sh's preflight already checks these, and on a terminal it can offer + # an alternative port. gpuk is the actual mutator and contributors call it + # DIRECTLY, so it re-checks and refuses, non-interactively. Same helpers as + # install.sh (both scripts ship standalone from the channel). A listener owned + # by an existing install is not a conflict: a manifest on disk means the + # re-run is the update path, and every checked port is then our own. + if [ ! -f "$MANIFEST" ]; then + _port_tool="" + if command -v ss >/dev/null 2>&1; then _port_tool="ss" + elif command -v netstat >/dev/null 2>&1; then _port_tool="netstat"; fi + if [ -z "$_port_tool" ]; then + echo "==> warning: cannot check for port conflicts (no ss or netstat)" + else + if [ "$MODE" = "controller" ]; then + set -- "$HTTP_PORT" 8443 8200 + else + # Worker data-plane ports (OPS-68) plus the local health listener. + set -- "$HEALTH_PORT" 8300 8301 8302 + fi + for _p in "$@"; do + _busy=1 + case "$_port_tool" in + ss) [ -n "$(ss -ltnH "sport = :$_p" 2>/dev/null)" ] || _busy=0 ;; + netstat) + netstat -ltn 2>/dev/null \ + | awk -v p="$_p" '{n=split($4,a,":"); if (a[n]==p) found=1} END{exit found?0:1}' \ + || _busy=0 + ;; + esac + [ "$_busy" -eq 0 ] || die "port $_p is already in use. Free it first, then run the install again." + done + fi + fi + + need_root install_binary "$BIN_SRC" mkdir -p "$ETC_DIR" "$DATA_ROOT" "$CACHE_DIR" - seed_secrets "$DATA_ROOT" "$MODE" - - # Default browse roots: common mount points + the dirs we already use. - [ -n "$BROWSE_ROOTS" ] || BROWSE_ROOTS="/mnt /data /srv $DATA_ROOT $(dirname "$CACHE_DIR")" - # shellcheck disable=SC2086 - BROWSE_JSON=$(json_array $BROWSE_ROOTS) - CACHE_JSON="[{\"hostPath\":\"$(json_str "$CACHE_DIR")\",\"shared\":false}]" + seed_secrets "$DATA_ROOT" + if [ "$MODE" = "controller" ]; then + prepare_claim_code + [ ! -s "$CLAIM_CODE_FILE" ] || CLAIM_CODE_AVAILABLE=1 + fi umask 077 if [ "$MODE" = "worker" ]; then @@ -239,6 +378,7 @@ cmd_install() { { echo "GPUK_CLUSTER=$CLUSTER" echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT" + echo "GPUK_MACHINE_ID_FILE=$MACHINE_ID_FILE" echo "NODE_DISPLAY_NAME=$(hostname)" [ -n "$CONTROLLER_URL" ] && echo "GPUK_CONTROLLER_URLS=$CONTROLLER_URL" } > "$WORKER_ENV" @@ -246,9 +386,21 @@ cmd_install() { echo "==> wrote $MANIFEST (worker: no app container)" else write_controller_manifest + # The host daemon and app container share DATA_ROOT. Only controller-mode + # workerd gets this private bootstrap channel; remote workers stay tokenless. + { + echo "GPUK_CLUSTER=$CLUSTER" + echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT" + echo "GPUK_MACHINE_ID_FILE=$MACHINE_ID_FILE" + echo "GPUK_CONTROLLER_URLS=wss://127.0.0.1:8443" + echo "GPUK_SELF_ENROLL_FILE=$SELF_ENROLL_FILE" + echo "NODE_DISPLAY_NAME=$(hostname)" + } > "$WORKER_ENV" + chmod 0600 "$WORKER_ENV" echo "==> wrote $MANIFEST (controller: app container $IMAGE)" fi chmod 0600 "$MANIFEST" + [ -z "$DOMAIN" ] || write_caddyfile write_unit systemctl daemon-reload @@ -259,12 +411,17 @@ cmd_install() { if [ -n "$ENROLL_TOKEN" ]; then [ -n "$CONTROLLER_URL" ] || die "--controller wss://: is required to enroll" echo "==> enrolling against $CONTROLLER_URL ..." - GPUK_IDENTITY_DIR="$IDENTITY_DIR" "$BIN_DEST" enroll \ + GPUK_IDENTITY_DIR="$IDENTITY_DIR" GPUK_MACHINE_ID_FILE="$MACHINE_ID_FILE" "$BIN_DEST" enroll \ --controller "$CONTROLLER_URL" --token "$ENROLL_TOKEN" \ || die "enrollment failed (bad/expired token, or controller unreachable)" else - echo "==> no token given: the daemon will discover its cluster on the LAN and wait" - echo " for automatic admission or administrator approval." + if [ -n "$CONTROLLER_URL" ]; then + echo "==> no token given: the daemon will request admission from $CONTROLLER_URL and wait" + echo " for automatic admission or administrator approval." + else + echo "==> no token or controller given: the daemon will discover its cluster on the LAN and wait" + echo " for automatic admission or administrator approval." + fi fi systemctl enable --now "$SERVICE_NAME" echo "==> $SERVICE_NAME enabled and started" @@ -282,18 +439,18 @@ cmd_install() { # No unix socket any more: workerd reconciles the app container in-process from # the on-disk manifest (there is no backend to relay through on the very first # boot). Steady-state updates go through the daemon over WS. - "$BIN_DEST" apply || die "apply failed — check: gpuk logs" + "$BIN_DEST" apply || die "apply failed. Check: gpuk logs" echo echo "Done. The worker daemon is running and the app container is up." echo " Status : gpuk status App logs: gpuk logs Daemon: journalctl -u $SERVICE_NAME" hint_existing_caches "$CACHE_DIR" } -# Worker manifest: cacheDisks + browseRoots, and an EMPTY image so +# Worker manifest: cacheDisks and an EMPTY image so # has_app_container() is false (apps/worker/src/modules/manifest.rs). No ports, no env, # no secretsRef — a worker runs no app container. -write_worker_manifest() { - cat > "$MANIFEST" < "$MANIFEST"; } + # Controller manifest: the declarative app-container description workerd applies. -write_controller_manifest() { +render_controller_manifest() { # NOTE there is deliberately no PORT here: PORT is the backend's own port, a - # loopback-only 8000 behind nginx in the all-in-one image. What the outside - # world dials is GPUK_PORT. + # loopback-only 8000 behind nginx in the all-in-one image. What the outside world + # dials is GPUK_PUBLIC_PORT (the published host port), which falls back to nginx's + # own GPUK_PORT when nothing republishes it. EXTRA_ENV="" - EXTRA_ENV="$EXTRA_ENV,\"GPUK_PORT\":\"$(json_str "$HTTP_PORT")\"" EXTRA_ENV="$EXTRA_ENV,\"GPUK_DATA_ROOT\":\"$(json_str "$DATA_ROOT")\"" - EXTRA_ENV="$EXTRA_ENV,\"GPUK_BOOTSTRAP_MUST_CHANGE\":\"1\"" + [ -z "$PROFILE" ] || EXTRA_ENV="$EXTRA_ENV,\"GPUK_INSTALL_PROFILE\":\"$(json_str "$PROFILE")\"" + if [ "$PROFILE" = "public" ]; then + EXTRA_ENV="$EXTRA_ENV,\"GPUK_HSTS\":\"true\"" + EXTRA_ENV="$EXTRA_ENV,\"GPUK_SESSION_COOKIE_SECURE\":\"true\"" + fi + BOOTSTRAP_SECRET_JSON="" + if [ -s "$BOOTSTRAP_PASSWORD_FILE" ]; then + EXTRA_ENV="$EXTRA_ENV,\"GPUK_BOOTSTRAP_MUST_CHANGE\":\"1\"" + BOOTSTRAP_SECRET_JSON=",\"GPUK_BOOTSTRAP_ADMIN_PASSWORD\":\"$(json_str "$BOOTSTRAP_PASSWORD_FILE")\"" + fi + CLAIM_SECRET_JSON="" + if [ "$CLAIM_CODE_AVAILABLE" -eq 1 ]; then + CLAIM_SECRET_JSON=",\"GPUK_CLAIM_CODE\":\"$(json_str "$CLAIM_CODE_FILE")\"" + fi [ "$NETWORK" = "host" ] || EXTRA_ENV="$EXTRA_ENV,\"BACKEND_DOCKER_NETWORK\":\"$(json_str "$NETWORK")\"" + # Published != bound (INS-43). On a bridged network the container keeps the image's + # FIXED listeners — nginx 8080, worker mTLS 8443 — and --http-port only moves the HOST + # side of the publication; Settings -> Network moves it later by patching this same + # manifest, so the container port must never become a variable. With host networking + # nothing is published and the listener itself takes the port. PORTS_JSON="{}" - if [ "$NETWORK" != "host" ]; then - PORTS_JSON="{\"$HTTP_PORT\":$HTTP_PORT,\"8200\":8200}" + if [ "$NETWORK" = "host" ]; then + EXTRA_ENV="$EXTRA_ENV,\"GPUK_PORT\":\"$(json_str "$HTTP_PORT")\"" + else + EXTRA_ENV="$EXTRA_ENV,\"GPUK_PORT\":\"8080\"" + EXTRA_ENV="$EXTRA_ENV,\"GPUK_PUBLIC_PORT\":\"$(json_str "$HTTP_PORT")\"" + PORTS_JSON="{\"8080\":$HTTP_PORT,\"8200\":8200,\"8443\":8443}" fi - cat > "$MANIFEST" < "$MANIFEST"; } + # ── control subcommands ──────────────────────────────────────────────────────── container_name() { sed -n 's/.*"containerName"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1 } +manifest_data_root() { + sed -n 's/.*"dataRoot"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1 +} + manifest_image() { sed -n 's/.*"image"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1 } @@ -374,9 +559,9 @@ health_port() { # Rule: it is a tag only if the last colon comes after the last slash. # (Same logic as apps/controller/api/src/core/release-channel.ts — see its unit tests.) image_repo() { - case "$1" in - *@*) printf '%s' "${1%@*}"; return ;; # digest pin → repo@sha256:… - esac + # A digest suffix (…@sha256:…) never carries the repo; drop it, then apply + # the tag logic — `repo:tag@sha256:…` and `repo@sha256:…` both reduce right. + set -- "${1%@*}" _t="${1##*:}" case "$_t" in "$1") printf '%s' "$1" ;; # no colon at all → untagged @@ -386,9 +571,9 @@ image_repo() { } image_tag() { - case "$1" in - *@*) return ;; # digest pin: no version to speak of - esac + # `repo:tag@sha256:…` keeps the human-readable tag next to the content pin — + # docker resolves by digest and ignores the tag. Strip the digest, then parse. + set -- "${1%@*}" _t="${1##*:}" case "$_t" in "$1") return ;; @@ -403,10 +588,138 @@ image_tag() { # https://gpu.kitchen/latest.json once the hub exists (specs/developpement/ci-cd.md). CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen/channel/raw/branch/main/latest.json}" -channel_version() { +# The release public key pinned in THIS copy of gpuk (OPS-20). The channel copy +# gets the real key substituted at publish time; the operator override +# (GPUK_UPDATE_PUBKEY) covers a self-hosted channel with its own keypair. The +# first install fetched gpuk itself over HTTPS from the channel — that moment is +# trust-on-first-use, like a worker's enrolment token pin; every later `update` +# is verified against the key pinned HERE, so whoever controls latest.json can +# no longer pick what an existing install runs. +CHANNEL_PUBKEY="${GPUK_UPDATE_PUBKEY:-RWQ7BKXJqGX2jdKXu1GxeSPVAN3JDRTefpImM/mFRjtFwq4E7mhnQtA7}" + +# Fetch latest.json AND its minisign signature, verify, and leave the verified +# document at $CHANNEL_DOC. Fail-closed: no signature, bad signature, no +# minisign CLI or no pinned key are all fatal — GPUK_CHANNEL_INSECURE=1 is the +# explicit, logged opt-out (a private mirror that does not sign). +CHANNEL_DOC="" +channel_fetch() { command -v curl >/dev/null 2>&1 || return 1 - curl -fsSL --max-time 20 "$CHANNEL_URL" 2>/dev/null \ - | sed -n 's/.*"version"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' | head -1 + CHANNEL_DOC=$(mktemp) || return 1 + curl -fsSL --max-time 20 "$CHANNEL_URL" -o "$CHANNEL_DOC" 2>/dev/null || return 1 + if [ "${GPUK_CHANNEL_INSECURE:-}" = "1" ]; then + echo "WARNING: GPUK_CHANNEL_INSECURE=1 — release channel signature NOT verified" >&2 + return 0 + fi + case "$CHANNEL_PUBKEY" in + ""|RWQ7BKXJqGX2jdKXu1GxeSPVAN3JDRTefpImM/mFRjtFwq4E7mhnQtA7*) + die "this gpuk carries no pinned release public key — set GPUK_UPDATE_PUBKEY (the minisign public-key line), or GPUK_CHANNEL_INSECURE=1 to skip verification" ;; + esac + command -v minisign >/dev/null 2>&1 \ + || die "minisign is required to verify the release channel (apt install minisign), or set GPUK_CHANNEL_INSECURE=1" + _sig=$(mktemp) + if ! curl -fsSL --max-time 20 "${CHANNEL_URL}.minisig" -o "$_sig" 2>/dev/null; then + rm -f "$_sig" + die "no signature at ${CHANNEL_URL}.minisig — refusing an unsigned channel document (OPS-20)" + fi + if ! minisign -Vq -m "$CHANNEL_DOC" -x "$_sig" -P "$CHANNEL_PUBKEY" >/dev/null 2>&1; then + rm -f "$_sig" + die "latest.json signature verification FAILED — refusing the channel document (OPS-20)" + fi + rm -f "$_sig" +} + +channel_field() { + sed -n 's/.*"'"$2"'"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$1" | head -1 +} + +channel_version() { + channel_fetch || return 1 + channel_field "$CHANNEL_DOC" version +} + +# ── backup ───────────────────────────────────────────────────────────────────── +# OPS-10: the embedded Postgres is only backed up COLD — hot-copying pgdata with +# a file tool is forbidden (torn pages). Order matters: stop the DAEMON first +# (its reconciler would immediately restart a stopped app container), then the +# container, snapshot, and restarting the service re-applies the manifest. +# An install on an external DATABASE_URL is refused here: gpuk only owns the +# embedded pgdata — back the real database up with pg_dump/backup-compose.sh. +# The finished directory still has to be copied to encrypted off-host storage, +# next to the recovery set (OPS-04/OPS-05: ENCRYPTION_KEY above all). +cmd_backup() { + need_root + _img=$(manifest_image) + [ -n "$_img" ] || die "this is a worker node — no controller data to back up here" + if grep -q '"DATABASE_URL"' "$MANIFEST" 2>/dev/null; then + die "this install uses an external DATABASE_URL — back THAT database up (pg_dump, or the compose procedures in specs/plateforme/operations.md); gpuk backup only snapshots the embedded pgdata" + fi + _root=$(manifest_data_root) + [ -n "$_root" ] || die "no dataRoot in $MANIFEST" + [ -d "$_root/pgdata" ] || die "no embedded pgdata under $_root — nothing to snapshot" + command -v sha256sum >/dev/null 2>&1 || die "sha256sum is required" + + _out="${1:-$_root/backups/$(date -u +%Y%m%dT%H%M%SZ)}" + [ -e "$_out" ] && die "refusing to overwrite existing $_out" + mkdir -p "$(dirname "$_out")" + _tmp="$_out.partial" + rm -rf "$_tmp"; mkdir -p "$_tmp" + + _cn=$(container_name) + echo "==> Stopping $SERVICE_NAME (its reconciler would restart the container mid-snapshot)..." + systemctl stop "$SERVICE_NAME" || die "could not stop $SERVICE_NAME" + _restart_daemon() { systemctl start "$SERVICE_NAME" 2>/dev/null || true; } + trap _restart_daemon EXIT + if [ -n "$_cn" ]; then + echo "==> Stopping $_cn (cold snapshot — OPS-10)..." + docker stop "$_cn" >/dev/null 2>&1 || true + _state=$(docker inspect -f '{{.State.Status}}' "$_cn" 2>/dev/null || echo absent) + case "$_state" in + running) die "container $_cn is still running — refusing a hot snapshot" ;; + esac + fi + + echo "==> Snapshotting $_root/pgdata..." + tar -C "$_root" -czf "$_tmp/pgdata.tar.gz" pgdata || die "snapshot failed" + cp "$MANIFEST" "$_tmp/host-manifest.json" 2>/dev/null || true + { + echo "{" + echo " \"created_utc\": \"$(date -u +%Y-%m-%dT%H:%M:%SZ)\"," + echo " \"image\": \"$(json_str "$_img")\"," + echo " \"data_root\": \"$(json_str "$_root")\"," + echo " \"kind\": \"cold-pgdata-snapshot\"" + echo "}" + } > "$_tmp/backup-manifest.json" + (cd "$_tmp" && sha256sum ./* > SHA256SUMS) || die "checksums failed" + chmod 0700 "$_tmp" + mv "$_tmp" "$_out" + + echo "==> Restarting $SERVICE_NAME (re-applies the manifest, container included)..." + systemctl start "$SERVICE_NAME" || die "could not restart $SERVICE_NAME — start it manually" + trap - EXIT + echo "backup: $_out" + echo "Copy it to encrypted OFF-HOST storage together with the recovery set" + echo "(ENCRYPTION_KEY above all — without it the data is unrecoverable, OPS-04)." + echo "A physical pgdata restore requires the same Postgres major and a throwaway" + echo "host rehearsal first (OPS-10)." +} + +# ── channel ──────────────────────────────────────────────────────────────────── +# Diagnostic (no root): fetch + VERIFY the channel document, print what it +# offers. Exercises exactly the trust chain `update` relies on — the CI probes +# it with a throwaway keypair, an operator uses it to debug a mirror. +cmd_channel() { + channel_fetch || die "cannot fetch $CHANNEL_URL" + echo "channel : $CHANNEL_URL" + echo "version : $(channel_field "$CHANNEL_DOC" version)" + _d=$(channel_field "$CHANNEL_DOC" controllerImageDigest) + [ -n "$_d" ] && echo "digest : $_d" + _d=$(channel_field "$CHANNEL_DOC" controllerImageDigestEnterprise) + [ -n "$_d" ] && echo "digest ee : $_d" + if [ "${GPUK_CHANNEL_INSECURE:-}" = "1" ]; then + echo "signature : SKIPPED (GPUK_CHANNEL_INSECURE=1)" + else + echo "signature : verified" + fi } # ── status ───────────────────────────────────────────────────────────────────── @@ -447,7 +760,8 @@ cmd_enroll() { done [ -n "$_url" ] || die "enroll needs --controller wss://:" [ -n "$_tok" ] || die "enroll needs --token gk_enroll_..." - GPUK_IDENTITY_DIR="$IDENTITY_DIR" "$BIN_DEST" enroll --controller "$_url" --token "$_tok" + GPUK_IDENTITY_DIR="$IDENTITY_DIR" GPUK_MACHINE_ID_FILE="$MACHINE_ID_FILE" \ + "$BIN_DEST" enroll --controller "$_url" --token "$_tok" systemctl restart "$SERVICE_NAME" 2>/dev/null || true } @@ -499,20 +813,35 @@ cmd_update() { return 0 fi + _digest="" if [ -z "$_want" ]; then - _want=$(channel_version) || true + # channel_fetch runs in THIS shell (not a $(…) subshell) so a signature + # failure is fatal here — fail-closed — and $CHANNEL_DOC survives. The + # verified signature closes the document half of OPS-20; the digest read + # from it pins CONTENT, closing the mutable-tag half. + if channel_fetch; then + _want=$(channel_field "$CHANNEL_DOC" version) + fi if [ -z "$_want" ]; then echo "==> cannot reach $CHANNEL_URL — re-applying the pinned $_current" "$BIN_DEST" apply return 0 fi + # Pick the digest matching the installed edition by image basename — the + # repo itself may be a mirror, the basename is the edition marker. + case "${_repo##*/}" in + *-ee) _digest=$(channel_field "$CHANNEL_DOC" controllerImageDigestEnterprise) ;; + *) _digest=$(channel_field "$CHANNEL_DOC" controllerImageDigest) ;; + esac fi - if [ "$_want" != "$_current" ]; then - echo "==> $_current → $_want" - # Pin the new tag into the manifest, then apply. sed edits the single "image" - # line in place (atomic tmp + move). - _new="$_repo:$_want" + _new="$_repo:$_want" + [ -n "$_digest" ] && _new="$_repo:$_want@$_digest" + _installed=$(manifest_image) + if [ "$_new" != "$_installed" ]; then + echo "==> $_current → $_want${_digest:+ (pinned by digest)}" + # Pin the new reference into the manifest, then apply. sed edits the single + # "image" line in place (atomic tmp + move). _tmp="$MANIFEST.new" sed "s#\"image\"[[:space:]]*:[[:space:]]*\"[^\"]*\"#\"image\": \"$(json_str "$_new")\"#" "$MANIFEST" > "$_tmp" \ || die "could not rewrite the image in $MANIFEST" @@ -535,16 +864,20 @@ usage() { cat < [--cluster N] [--cache-dir P] + gpuk install --mode controller --image [--profile homelab|studio|enterprise|public] + [--cluster N] [--cache-dir P] [--data-root P] [--http-port P] [--network host|bridge|] [--binary ] gpuk install --mode worker --controller wss://: --enroll-token gk_enroll_... - [--cluster N] [--cache-dir P] [--browse-root P]... [--binary ] + [--cluster N] [--cache-dir P] [--binary ] + gpuk install ... --dry-run Validate inputs and print the manifest without changing the host gpuk status Service state, enrollment, /health, app container status gpuk enroll --controller wss://: --token gk_enroll_... gpuk apply (controller) Reconcile the app container from the manifest gpuk update (controller) Move to the current release (pull + recreate, rollback) gpuk update --check (controller) Compare the installed version with the release gpuk update --version (controller) Move to a specific release + gpuk channel Fetch + VERIFY the release channel and print what it offers + gpuk backup [DIR] (controller) Cold snapshot of the embedded pgdata (stop → tar → restart) gpuk manifest Print the current manifest gpuk logs Follow the app container logs (controller) or the daemon journal gpuk uninstall Remove the systemd service @@ -562,6 +895,8 @@ case "$cmd" in enroll) cmd_enroll "$@" ;; apply) cmd_apply ;; update) cmd_update "$@" ;; + channel) cmd_channel ;; + backup) cmd_backup "${1:-}" ;; manifest) cat "$MANIFEST" ;; logs) _img=$(manifest_image) diff --git a/install.sh b/install.sh index cdd0c34..102716b 100644 --- a/install.sh +++ b/install.sh @@ -7,15 +7,20 @@ # channel repo is written by .gitea/workflows/release.yml, see specs/developpement/ci-cd.md.) # # What it does, and nothing more: -# 1. preflight docker, the NVIDIA driver, and a REAL `--gpus all` smoke test +# 1. preflight docker, the NVIDIA driver, a REAL `--gpus all` smoke test, and +# the listening ports (INS-46 — a taken port fails HERE, not three +# minutes later in a health-check timeout) # 2. resolve the current release from the channel (a TAG — never a floating # `latest`: an install that silently changes version under you is # not an install, it is a surprise) -# 3. install `gpu-kitchen-worker` (workerd), the privileged host daemon, via +# 3. ask the security profile — and, under `public`, a domain — when a +# terminal is attached (INS-45); no terminal, no questions, and +# the first-run wizard asks instead (PRF-05) +# 4. install `gpu-kitchen-worker` (workerd), the privileged host daemon, via # the existing `gpuk` installer — on a controller node it OWNS the # app container's lifecycle -# 4. hand over workerd pulls the pinned all-in-one controller image and starts it -# 5. print the UI URL on the real host, and the first-run password +# 5. hand over workerd pulls the pinned all-in-one controller image and starts it +# 6. print the UI URL on the real host, plus the profile-specific next step # # Re-running is how you UPDATE: same command, newer tag, `docker pull` + recreate, # data untouched (it lives in the data root, not the container). @@ -39,14 +44,24 @@ CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen/channel/raw EDITION="${GPUK_EDITION:-community}" DATA_ROOT="${GPUK_DATA_ROOT:-/var/lib/gpu-kitchen}" CACHE_DIR="${GPUK_CACHE_DIR:-}" -PORT="${GPUK_PORT:-8080}" +# 1337, not 8080: the single most-squatted port in existence would make the +# conflict preflight fire on half the lab boxes out there (INS-01). +PORT="${GPUK_PORT:-1337}" +# Fixed listeners: mirror of the gpuk-proxy default (core/cluster-settings.ts +# proxyPublicPort) and of the worker mTLS channel — no install-time flag moves +# them (INS-46). +INFERENCE_PORT=8200 +MTLS_PORT=8443 CLUSTER="${GPUK_CLUSTER:-default}" +PROFILE="" +DOMAIN="" VERSION="" IMAGE="" WORKER_BINARY="" GPUK_SCRIPT="" SKIP_GPU_CHECK=0 SKIP_PREFLIGHT=0 +NON_INTERACTIVE=0 DRY_RUN=0 GREEN=''; RED=''; YELLOW=''; BOLD=''; RESET='' @@ -60,6 +75,21 @@ warn() { echo " ${YELLOW}!${RESET} $*"; } step() { echo; echo "${BOLD}$*${RESET}"; } die() { echo " ${RED}✗${RESET} $*" >&2; exit 1; } +# `curl … | sudo sh` leaves stdin holding the script itself, so questions are +# asked and answered on the controlling terminal — /dev/tty — when there is one +# (INS-45). No terminal (CI, provisioning), --non-interactive or --dry-run keep +# every historical flags-only behaviour. +can_prompt() { + [ "$NON_INTERACTIVE" -eq 0 ] || return 1 + [ "$DRY_RUN" -eq 0 ] || return 1 + (: < /dev/tty) 2>/dev/null +} + +ask() { # $1 = prompt → $REPLY (empty on EOF) + printf '%s' "$1" > /dev/tty + IFS= read -r REPLY < /dev/tty || REPLY="" +} + usage() { cat < Install this release instead of the channel's current one --image Use this controller image outright (implies --version none) --edition community (default) | enterprise - --port

Port the UI listens on (default 8080) + --port

Port the UI listens on (default 1337) --data-root Where the database and secrets live (default /var/lib/gpu-kitchen) --cache-dir Model cache (default /hf) --cluster Cluster name workers join (default "default") + --profile

homelab | studio | enterprise | public (no flag + a terminal + = the script asks; no flag + no terminal = first-run asks) + --domain Domain for the public profile: writes a filled TLS + reverse-proxy example to /caddy/Caddyfile + --non-interactive Never ask anything, even with a terminal attached --worker-binary

Use a locally-built gpu-kitchen-worker instead of downloading one --gpuk-script

Use a local copy of the gpuk installer --skip-gpu-check Skip the 'docker run --gpus all' smoke test @@ -93,6 +128,9 @@ while [ $# -gt 0 ]; do --data-root) DATA_ROOT="$2"; shift 2 ;; --cache-dir) CACHE_DIR="$2"; shift 2 ;; --cluster) CLUSTER="$2"; shift 2 ;; + --profile) PROFILE="$2"; shift 2 ;; + --domain) DOMAIN="$2"; shift 2 ;; + --non-interactive) NON_INTERACTIVE=1; shift ;; --worker-binary) WORKER_BINARY="$2"; shift 2 ;; --gpuk-script) GPUK_SCRIPT="$2"; shift 2 ;; --skip-gpu-check) SKIP_GPU_CHECK=1; shift ;; @@ -110,6 +148,15 @@ case "$EDITION" in *) die "--edition must be community or enterprise (got '$EDITION')" ;; esac +case "$PROFILE" in + ""|homelab|studio|enterprise|public) ;; + *) die "--profile must be homelab, studio, enterprise or public (got '$PROFILE')" ;; +esac + +case "$DOMAIN" in + *[!A-Za-z0-9.-]*) die "--domain must be a bare domain name (got '$DOMAIN')" ;; +esac + echo echo "${BOLD}GPU Kitchen — installing the $EDITION controller${RESET}" @@ -129,16 +176,16 @@ step "Preflight — docker, NVIDIA driver, container toolkit" [ "$DRY_RUN" -eq 1 ] || [ "$(id -u)" -eq 0 ] \ || die "run as root: curl -fsSL … | sudo sh" -command -v curl >/dev/null 2>&1 || die "curl not found — install it first" +command -v curl >/dev/null 2>&1 || die "curl not found. Install curl first." if command -v docker >/dev/null 2>&1; then if docker info >/dev/null 2>&1; then ok "docker daemon reachable ($(docker --version | cut -d' ' -f3 | tr -d ,))" else - die "docker is installed but its daemon is unreachable (is it running? are you root?)" + die "docker is installed but its daemon does not answer. Start the daemon, or run as root." fi else - die "docker not found — install Docker Engine first: https://docs.docker.com/engine/install/" + die "docker not found. Install Docker Engine first: https://docs.docker.com/engine/install/" fi if command -v nvidia-smi >/dev/null 2>&1; then @@ -150,7 +197,7 @@ if command -v nvidia-smi >/dev/null 2>&1; then die "nvidia-smi is present but reports no GPU" fi else - die "nvidia-smi not found — install the NVIDIA driver first" + die "nvidia-smi not found. Install the NVIDIA driver first." fi if [ "$SKIP_GPU_CHECK" -eq 1 ]; then @@ -163,10 +210,10 @@ elif docker info 2>/dev/null | grep -qiE 'runtimes:.*nvidia'; then if docker run --rm --gpus all ubuntu:24.04 nvidia-smi -L >/dev/null 2>&1; then ok "nvidia-container-toolkit works (a container can see the GPUs)" else - die "the nvidia runtime is registered but 'docker run --gpus all' cannot see the GPUs — reinstall nvidia-container-toolkit" + die "the nvidia runtime is registered but 'docker run --gpus all' cannot see the GPUs. Reinstall nvidia-container-toolkit." fi else - die "nvidia-container-toolkit is not registered with docker — install it, then restart dockerd: + die "nvidia-container-toolkit is not registered with docker. Install it, then restart dockerd: https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html" fi @@ -180,7 +227,109 @@ FREE_GB=$(df -BG --output=avail "$CACHE_PARENT" 2>/dev/null | tail -1 | tr -dc ' if [ "${FREE_GB:-0}" -ge 100 ]; then ok "model cache $CACHE_DIR — ${FREE_GB}G free" else - warn "only ${FREE_GB:-?}G free under $CACHE_PARENT — model weights want 100G+" + warn "only ${FREE_GB:-?}G free under $CACHE_PARENT. Model weights need 100G or more." +fi + +# ── Port conflicts (INS-46) ── +# A taken port must fail HERE, before anything mutates the host — today's +# alternative is a 3-minute health-check timeout with zero diagnosis. Best-effort +# detection (ss, then netstat); neither present is a warn, never a false red. +# A listener owned by an EXISTING GPU Kitchen install is not a conflict: re-running +# this script is the documented update path, and the manifest names our port. +# Test hook: force the detector. The netstat fallback is unreachable on any +# host that has ss (all of them, in practice), so the CI smoke pins it here to +# keep its parsing honest. +PORT_TOOL="${GPUK_PORT_CHECK_TOOL:-}" +case "$PORT_TOOL" in + ""|ss|netstat) ;; + *) die "GPUK_PORT_CHECK_TOOL must be ss or netstat (got '$PORT_TOOL')" ;; +esac +if [ -z "$PORT_TOOL" ]; then + if command -v ss >/dev/null 2>&1; then PORT_TOOL="ss" + elif command -v netstat >/dev/null 2>&1; then PORT_TOOL="netstat"; fi +fi + +port_busy() { # $1 = port → 0 iff something listens on TCP :$1 + case "$PORT_TOOL" in + ss) [ -n "$(ss -ltnH "sport = :$1" 2>/dev/null)" ] ;; + netstat) netstat -ltn 2>/dev/null \ + | awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) found=1} END{exit found?0:1}' ;; + *) return 1 ;; + esac +} + +port_owner() { # $1 = port → best-effort process name (needs root for -p) + case "$PORT_TOOL" in + ss) ss -ltnpH "sport = :$1" 2>/dev/null \ + | sed -n 's/.*users:((\"\([^"]*\)\".*/\1/p' | head -1 ;; + netstat) netstat -ltnp 2>/dev/null \ + | awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) {print $NF; exit}}' \ + | sed 's|^[0-9]*/||' ;; + esac +} + +manifest_ui_port() { # the port an existing install already owns, if any + [ -f /etc/gpu-kitchen/manifest.json ] || return 0 + # Bridge publishes GPUK_PUBLIC_PORT over the fixed container 8080; host + # networking moves the listener itself (GPUK_PORT). Same precedence as gpuk. + _p=$(sed -n 's/.*"GPUK_PUBLIC_PORT"[[:space:]]*:[[:space:]]*"\([0-9]*\)".*/\1/p' \ + /etc/gpu-kitchen/manifest.json | head -1) + [ -n "$_p" ] || _p=$(sed -n 's/.*"GPUK_PORT"[[:space:]]*:[[:space:]]*"\([0-9]*\)".*/\1/p' \ + /etc/gpu-kitchen/manifest.json | head -1) + printf '%s' "$_p" +} + +if [ -z "$PORT_TOOL" ]; then + warn "cannot check for port conflicts (neither ss nor netstat found)" +else + HAVE_MANIFEST=0 + [ ! -f /etc/gpu-kitchen/manifest.json ] || HAVE_MANIFEST=1 + if [ "$HAVE_MANIFEST" -eq 1 ] && [ "$(manifest_ui_port)" = "$PORT" ]; then + ok "UI port $PORT — already ours (re-running is how you update)" + elif port_busy "$PORT"; then + OWNER=$(port_owner "$PORT") + OWNER="${OWNER:-an unknown process}" + if can_prompt; then + ALT=$((PORT + 1)) + while port_busy "$ALT"; do ALT=$((ALT + 1)); done + # Propose, never auto-pick: the URL printed at the end and the idempotent + # re-run both need the operator to KNOW which port they chose. + ask " ${YELLOW}!${RESET} port $PORT is busy ($OWNER). Use $ALT instead? [$ALT], another port, or 'q' to abort: " + case "$REPLY" in + q|Q) die "port $PORT is in use by $OWNER. Run the install again with --port

." ;; + "") PORT="$ALT" ;; + *) + case "$REPLY" in + *[!0-9]*) die "not a port number: $REPLY" ;; + esac + if port_busy "$REPLY"; then + die "port $REPLY is busy too. Run the install again with --port

." + fi + PORT="$REPLY" + ;; + esac + ok "UI port $PORT is free" + else + die "port $PORT is already in use by $OWNER. Pass --port

to choose another port." + fi + else + ok "UI port $PORT is free" + fi + # The mTLS and inference listeners have no install-time flag — assumed + # limitation (INS-46): the published mTLS port moves later via + # Settings -> Network. With a manifest present they are our own listeners. + if [ "$HAVE_MANIFEST" -eq 0 ]; then + if port_busy "$MTLS_PORT"; then + OWNER=$(port_owner "$MTLS_PORT") + die "port $MTLS_PORT (worker channel) is in use by ${OWNER:-an unknown process}. Free it first." + fi + if port_busy "$INFERENCE_PORT"; then + OWNER=$(port_owner "$INFERENCE_PORT") + die "port $INFERENCE_PORT (inference endpoint) is in use by ${OWNER:-an unknown process}. Free it first. + ($INFERENCE_PORT is also HashiCorp Vault's default port.)" + fi + ok "worker channel port $MTLS_PORT and inference port $INFERENCE_PORT are free" + fi fi fi # end preflight @@ -217,7 +366,7 @@ if [ -n "$IMAGE" ]; then else CHANNEL=$(curl -fsSL --max-time 20 "$CHANNEL_URL" 2>/dev/null) \ || die "cannot reach the release channel at $CHANNEL_URL - (offline? pass --image to install a specific image directly)" + If this host is offline, pass --image to install a specific image directly." [ -n "$VERSION" ] || VERSION=$(echo "$CHANNEL" | json_field version) [ -n "$VERSION" ] || die "the release channel returned no version: $CHANNEL_URL" @@ -241,6 +390,24 @@ else [ -n "${GPUK_SCRIPT}" ] || GPUK_SCRIPT_URL=$(echo "$CHANNEL" | json_field gpukScript) fi +# A direct --image and a channel version are held to the same immutable-image +# rule. A registry host:port is not a tag separator; only the last colon after +# the last slash counts. Digest pins are accepted too. +case "$IMAGE" in + *@sha256:*) ;; + *:latest) die "refusing floating image tag '$IMAGE'. Use an explicit release tag or digest." ;; + *) + IMAGE_TAG="${IMAGE##*:}" + case "$IMAGE_TAG" in + "$IMAGE"|*/*) die "controller image must carry an explicit tag or sha256 digest (got '$IMAGE')" ;; + esac + ;; +esac + +if [ -n "$DOMAIN" ] && [ -n "$PROFILE" ] && [ "$PROFILE" != "public" ]; then + die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)" +fi + if [ "$DRY_RUN" -eq 1 ]; then step "Dry run — stopping here" if [ "$SKIP_PREFLIGHT" -eq 1 ]; then @@ -253,10 +420,85 @@ if [ "$DRY_RUN" -eq 1 ]; then echo " data root : $DATA_ROOT" echo " model cache : $CACHE_DIR" echo " UI port : $PORT" + if [ -n "$PROFILE" ]; then + echo " profile : $PROFILE" + else + echo " profile : (asked on the terminal, or chosen during first run)" + fi + [ -z "$DOMAIN" ] || echo " domain : $DOMAIN" exit 0 fi -# ── 3. The host daemon ─────────────────────────────────────────────────────── +# ── 3. Resolve the installation profile (INS-45) ───────────────────────────── +# Only ever on a terminal, and only when --profile was not given. The answer is +# relayed verbatim as --profile: the backend stays the sole applier of profile +# defaults and floors. Without a terminal the historical path is untouched — +# profile unset, enterprise floors, the first-run wizard requires the choice +# (PRF-05). +if [ -z "$PROFILE" ] && can_prompt; then + step "Security profile — how will this kitchen be used?" + cat > /dev/tty <<'PROFILES' + 1) Home lab a trusted home network + No sign-in on your home network. Nearby workers are found and join + without waiting for approval. + + 2) Studio one control station, shared compute + Control stays on this machine. Colleagues use API keys, while nearby + workers wait for your approval. + + 3) Enterprise a managed company network + Sign-in is required. The controller stays quiet on the network; known + workers can request your approval. + + 4) Public server direct internet exposure + Sign-in and hardened browser transport are required. Network discovery + is off and workers join only by token. + + You can change this later in Settings. Stricter floors re-apply. Nothing + already issued is revoked. + +PROFILES + while :; do + ask " Choose a profile [1-4, Enter = 1 (Home lab)]: " + case "$REPLY" in + ""|1|homelab) PROFILE="homelab" ;; + 2|studio) PROFILE="studio" ;; + 3|enterprise) PROFILE="enterprise" ;; + 4|public) PROFILE="public" ;; + *) printf '%s\n' " pick 1, 2, 3 or 4" > /dev/tty; continue ;; + esac + break + done + ok "profile: $PROFILE" +fi + +# Under public, a domain lets gpuk write a FILLED TLS reverse-proxy example +# (INS-47). Optional: Enter skips, and the banner still points at the shipped +# Caddyfile.example. +if [ "$PROFILE" = "public" ] && [ -z "$DOMAIN" ] && can_prompt; then + while :; do + ask " Domain for HTTPS access (e.g. gpu.example.com — Enter to skip): " + # Accept the copy-paste reflex: strip a pasted scheme and anything after + # the first slash, then insist on a bare domain rather than skipping — + # a silently dropped answer would be discovered hours later, at DNS time. + REPLY="${REPLY#https://}" + REPLY="${REPLY#http://}" + REPLY="${REPLY%%/*}" + [ -n "$REPLY" ] || break + case "$REPLY" in + *[!A-Za-z0-9.-]*) + printf '%s\n' " not a bare domain name: $REPLY. Try again, or press Enter to skip." > /dev/tty + ;; + *) DOMAIN="$REPLY"; break ;; + esac + done +fi + +if [ -n "$DOMAIN" ] && [ "$PROFILE" != "public" ]; then + die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)" +fi + +# ── 4. The host daemon ─────────────────────────────────────────────────────── step "Host daemon — gpu-kitchen-worker" TMP=$(mktemp -d) @@ -291,6 +533,9 @@ set -- install \ --cache-dir "$CACHE_DIR" \ --http-port "$PORT" +[ -z "$PROFILE" ] || set -- "$@" --profile "$PROFILE" +[ -z "$DOMAIN" ] || set -- "$@" --domain "$DOMAIN" + if [ -n "$WORKER_BINARY" ]; then [ -f "$WORKER_BINARY" ] || die "no such worker binary: $WORKER_BINARY" set -- "$@" --binary "$WORKER_BINARY" @@ -303,15 +548,15 @@ fi # else: gpuk reuses an already-installed binary, or fails with its own message. step "Installing — this pulls the image, so it can take a few minutes" -sh "$GPUK_SCRIPT" "$@" || die "the install failed — see: journalctl -u gpu-kitchen-worker" +sh "$GPUK_SCRIPT" "$@" || die "the install failed. See: journalctl -u gpu-kitchen-worker" -# ── 4. Wait for the app, then say where it is ──────────────────────────────── +# ── 5. Wait for the app, then say where it is ──────────────────────────────── step "Waiting for the controller to answer" i=0 until curl -fsS -o /dev/null --max-time 3 "http://127.0.0.1:$PORT/health" 2>/dev/null; do i=$((i + 1)) - [ "$i" -lt 90 ] || die "the controller did not answer on :$PORT within 90s — see: gpuk logs" + [ "$i" -lt 90 ] || die "the controller did not answer on :$PORT within 90s. See: gpuk logs" sleep 2 done ok "the controller is up" @@ -322,30 +567,66 @@ ok "the controller is up" HOSTNAME_FQDN=$(hostname -f 2>/dev/null || hostname 2>/dev/null || echo localhost) LAN_IP=$(ip route get 1.1.1.1 2>/dev/null | sed -n 's/.* src \([0-9.]*\).*/\1/p' | head -1) -PW_FILE="$DATA_ROOT/secrets/bootstrap_admin_password" -ADMIN_PW="" -[ -f "$PW_FILE" ] && ADMIN_PW=$(cat "$PW_FILE" 2>/dev/null || true) - echo echo "${BOLD}${GREEN}GPU Kitchen is running.${RESET}" echo echo " ${BOLD}Open:${RESET} http://${HOSTNAME_FQDN}:${PORT}" [ -n "$LAN_IP" ] && echo " http://${LAN_IP}:${PORT}" echo -if [ -n "$ADMIN_PW" ]; then - echo " ${BOLD}Sign in:${RESET} admin@local" - echo " ${BOLD}Password:${RESET} ${ADMIN_PW}" - echo " (the first-run wizard asks you to change it)" - echo +CLAIM_CODE_FILE="$DATA_ROOT/secrets/claim_code" +CLAIM_CODE="" +[ ! -s "$CLAIM_CODE_FILE" ] || CLAIM_CODE=$(cat "$CLAIM_CODE_FILE" 2>/dev/null || true) +if [ -n "$CLAIM_CODE" ]; then + echo " ${BOLD}Claim code:${RESET} $CLAIM_CODE" + echo " ${BOLD}Read again:${RESET} $CLAIM_CODE_FILE (mode 0600; removed after claim)" +else + echo " ${BOLD}Claim code:${RESET} consumed (the first account already exists)" fi -echo " Inference endpoint : http://${HOSTNAME_FQDN}:8200/v1" +echo + +case "$PROFILE" in + public) + echo " ${BOLD}Next:${RESET} create the first administrator in the UI" + echo + # The public profile REQUIRES a TLS reverse proxy (OPS-13, INS-47) — the UI + # port speaks plain HTTP. The product cannot verify the proxy's presence + # (OPS-67), so the closest thing to enforcement is saying it here, clearly. + if [ -n "$DOMAIN" ]; then + echo " ${BOLD}HTTPS:${RESET} https://${DOMAIN} answers after these steps:" + echo " 1. DNS: point ${DOMAIN} at this machine's public IP" + echo " 2. Install Caddy: https://caddyserver.com/docs/install" + echo " 3. sudo cp $DATA_ROOT/caddy/Caddyfile /etc/caddy/Caddyfile" + echo " sudo systemctl reload caddy" + echo " A filled example was written to $DATA_ROOT/caddy/Caddyfile." + echo " Until then the UI answers in cleartext on the URLs above." + else + echo " ${BOLD}HTTPS:${RESET} the public profile requires a TLS reverse proxy before any" + echo " public exposure. See deployments/controller/Caddyfile.example." + fi + echo + ;; + enterprise) + echo " ${BOLD}Next:${RESET} create the first administrator in the UI" + echo + ;; + homelab|studio) + echo " ${BOLD}Next:${RESET} finish first-run in the UI" + echo + ;; + "") + echo " ${BOLD}Next:${RESET} choose an installation profile in the first-run assistant" + echo + ;; +esac +echo " Inference endpoint : http://${HOSTNAME_FQDN}:${INFERENCE_PORT}/v1" echo " Version : ${VERSION}" echo # Pre-existing model caches (B81). `gpuk install` prints the same hint, but that # scrolls past mid-install; this banner is where people actually look. Detection -# only — the install stays non-interactive and nothing outside $CACHE_DIR is -# touched, read as configuration, or modified. +# only — cache questions belong to the first-run wizard (INS-48, REG-41), never +# to the CLI, and nothing outside $CACHE_DIR is touched, read as configuration, +# or modified. EXISTING_CACHE="" CHOSEN_CACHE=$(readlink -f "$CACHE_DIR" 2>/dev/null || echo "$CACHE_DIR") for d in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do @@ -359,8 +640,9 @@ if [ -n "$EXISTING_CACHE" ]; then for d in $EXISTING_CACHE; do echo " ${BOLD}Existing model cache:${RESET} $d" done - echo " Reference it from Settings -> Cache folders to reuse those models" - echo " without downloading again. GPU Kitchen only READS a referenced cache." + echo " GPU Kitchen will offer to reuse those models at first launch, and any" + echo " time from Nodes & GPU -> Storage. A reused cache is referenced in" + echo " place. Nothing is moved or deleted." echo fi echo " Update : re-run this command, or press Update in the UI, or: gpuk update" diff --git a/latest.json b/latest.json index 587265d..339acb1 100644 --- a/latest.json +++ b/latest.json @@ -1,9 +1,11 @@ { - "version": "v0.1.2", - "semver": "0.1.2", + "version": "v0.1.3", + "semver": "0.1.3", "controllerImage": "repo.byterain.io/gpukitchen/gpukitchen-controller", "controllerImageEnterprise": "repo.byterain.io/gpukitchen-private/gpukitchen-controller-ee", - "workerBase": "https://repo.byterain.io/api/packages/gpukitchen/generic/gpu-kitchen-worker/v0.1.2", + "controllerImageDigest": "sha256:609bfdf5e9abcebdae56d868217b7d6400bd6fa8c44737b2275bfc3d4db6a100", + "controllerImageDigestEnterprise": "sha256:9fe855a97932cc134948ace4d03de592652e5872749c27273e9e7f6e930cd7ab", + "workerBase": "https://repo.byterain.io/api/packages/gpukitchen/generic/gpu-kitchen-worker/v0.1.3", "gpukScript": "https://repo.byterain.io/gpukitchen/channel/raw/branch/main/gpuk", "releaseNotes": "https://repo.byterain.io/gpukitchen/channel" } diff --git a/latest.json.minisig b/latest.json.minisig new file mode 100644 index 0000000..9eb37bb --- /dev/null +++ b/latest.json.minisig @@ -0,0 +1,4 @@ +untrusted comment: signature from minisign secret key +RUQ7BKXJqGX2je3pLbIHpqUv94Xn863QAziuHoQdpUFKIG0QvVOjHN2rPz7ooHA1Nh6hjttCUfpvGsuH4IOLfsVgTB9KyA0TDQo= +trusted comment: gpu-kitchen channel v0.1.3 +dcB01BhY0qvL9wGDhVtTG3q0IimAB1Kfh/6yd/lZhov3cQ38cM2sLelVOoE6DpkXyYZ2OXzTA0tJJEDkt68nAQ==