release v0.1.3

This commit is contained in:
gpuk-release
2026-09-13 21:53:54 +00:00
parent dd6b68a756
commit 406d457b8d
4 changed files with 737 additions and 114 deletions
+410 -75
View File
@@ -12,7 +12,7 @@
#
# Two roles:
# worker a headless compute node. Installs the binary + systemd unit + a
# manifest (cacheDisks, browseRoots) with NO app container, NO
# manifest (cacheDisks) with NO app container, NO
# Postgres, NO docker app image. It ENROLLS over mTLS with a
# enrollment token, or waits for LAN discovery admission, then runs.
# controller the full app + UI. Installs the binary + systemd unit + an
@@ -44,9 +44,10 @@ BIN_DEST="${GPUK_BIN_DEST:-/usr/local/bin/gpu-kitchen-worker}"
ETC_DIR="${GPUK_ETC_DIR:-/etc/gpu-kitchen}"
MANIFEST="$ETC_DIR/manifest.json"
IDENTITY_DIR="${GPUK_IDENTITY_DIR:-$ETC_DIR/identity}"
MACHINE_ID_FILE="${GPUK_MACHINE_ID_FILE:-$ETC_DIR/machine-id}"
WORKER_ENV="$ETC_DIR/worker.env"
SERVICE_NAME="gpu-kitchen-worker"
UNIT_DEST="/etc/systemd/system/$SERVICE_NAME.service"
UNIT_DEST="${GPUK_UNIT_DEST:-/etc/systemd/system/$SERVICE_NAME.service}"
# Where to download the binary from when no --binary is given.
GPUK_RELEASE_BASE="${GPUK_RELEASE_BASE:-}"
@@ -102,63 +103,120 @@ install_binary() { # [local-path]
}
gen_secret() { head -c 32 /dev/urandom | base64 | tr '+/' '-_' | tr -d '='; }
# A password a human retypes once, from a terminal. Ambiguous glyphs removed.
gen_password() { head -c 24 /dev/urandom | base64 | tr -d '=+/OIl01' | cut -c1-16; }
seed_secrets() { # DATA_ROOT MODE
seed_secrets() { # DATA_ROOT
_sd="$1/secrets"
mkdir -p "$_sd"; chmod 0700 "$_sd"
[ -f "$_sd/encryption_key" ] || { umask 077; gen_secret > "$_sd/encryption_key"; }
[ -f "$_sd/node_id" ] || { umask 077; (cat /proc/sys/kernel/random/uuid 2>/dev/null || uuidgen) > "$_sd/node_id"; }
# The controller's first-run password. Generated here — NOT left to the backend
# to print into a log nobody watches when the install is one piped command. The
# installer prints it once, and the first-run wizard makes the operator replace
# it (GPUK_BOOTSTRAP_MUST_CHANGE).
if [ "$2" = "controller" ] && [ ! -f "$_sd/bootstrap_admin_password" ]; then
umask 077; gen_password > "$_sd/bootstrap_admin_password"
# No nominal first-run password is generated. An operator may pre-provision
# this documented break-glass file; only then is it injected into the app.
[ ! -f "$_sd/bootstrap_admin_password" ] || chmod 0600 "$_sd/bootstrap_admin_password"
}
prepare_claim_code() {
# The file is the operator's recoverable proof of machine possession. Create
# it once, preserve it across reinstalls, and let the backend unlink it after
# the atomic first-account claim. A missing file beside an existing manifest
# therefore means "consumed", never "rotate the credential".
if [ -f "$CLAIM_CODE_FILE" ]; then
chmod 0600 "$CLAIM_CODE_FILE"
elif [ ! -f "$MANIFEST" ]; then
umask 077
gen_secret > "$CLAIM_CODE_FILE"
chmod 0600 "$CLAIM_CODE_FILE"
fi
}
# Write the systemd unit: prefer a sibling file, else embed. The daemon runs the
# binary with NO arguments (steady-state); role/controller URL/CA come from the
# enrolled identity (worker) and the optional EnvironmentFile.
# Write the systemd unit generated from the worker's canonical template. The
# generator injects the Rust lock-contention exit code here too, so the binary
# and systemd restart policy cannot silently drift apart.
write_unit() {
_src_unit="$(dirname "$0")/../../apps/worker/install/$SERVICE_NAME.service"
[ -f "$_src_unit" ] || _src_unit="$(dirname "$0")/$SERVICE_NAME.service"
if [ -f "$_src_unit" ]; then
install -m 0644 "$_src_unit" "$UNIT_DEST"
else
# BEGIN GENERATED WORKER SYSTEMD UNIT
cat > "$UNIT_DEST" <<UNIT
[Unit]
Description=GPU Kitchen worker daemon (gpu-kitchen-worker)
Documentation=https://repo.byterain.io/Sebastien/GPU-Manager
After=network-online.target docker.service
Wants=network-online.target docker.service
[Service]
Type=simple
# Runs as root on the host (NVML, docker.sock, mounts) — NOT in a container.
# Identity (keypair/cert/CA) lives under $IDENTITY_DIR; a worker enrolls once
# before this unit starts. Optional overrides in $WORKER_ENV.
# The worker runs as root on the host (NVML, docker.sock, mounts) — NOT in a
# container. Identity (keypair/cert/CA) lives under $IDENTITY_DIR; enroll once
# before starting this unit. Optional overrides live in $WORKER_ENV.
EnvironmentFile=-$WORKER_ENV
ExecStart=$BIN_DEST
# WRK-173: the old MainPID transfers supervision to the verified replacement
# before it exits. Notifications remain local to this service cgroup.
NotifyAccess=all
# Transient operation state only. The node lock has its own stable inode under
# /run/lock, outside this systemd-managed directory.
RuntimeDirectory=gpu-kitchen
RuntimeDirectoryMode=0750
Restart=always
RestartSec=2
# Lock contention is an operator error, not a crash: do not retry forever while
# another directly installed worker owns the WRK-167 host lock.
RestartPreventExitStatus=75
User=root
# Fail-closed renewal: on a refused cert renewal the daemon exits and systemd
# restarts it into an enroll-required state.
KillSignal=SIGTERM
TimeoutStopSec=15
[Install]
WantedBy=multi-user.target
UNIT
fi
# END GENERATED WORKER SYSTEMD UNIT
}
# TLS reverse-proxy example, FILLED with the operator's domain (INS-47) — written
# only under `--profile public --domain <d>`. Kept aligned with
# deployments/controller/Caddyfile.example (the compose variant); this copy
# targets the all-in-one image, where nginx on the UI port is the single front
# door (INS-09) so one upstream carries pages, /api and the /ws upgrade alike.
# We write a file and NOTHING more: no package install, no service start, no
# other program's config read or touched — putting the proxy in service stays an
# operator act (OPS-13), and its presence stays unverifiable (OPS-67).
write_caddyfile() {
mkdir -p "$DATA_ROOT/caddy"
cat > "$DATA_ROOT/caddy/Caddyfile" <<CADDY
# TLS in front of GPU Kitchen — generated by the installer for --profile public
# (specs/plateforme/installation.md INS-47). Caddy provisions and renews the
# certificate itself once DNS for $DOMAIN points at this machine.
#
# sudo cp $DATA_ROOT/caddy/Caddyfile /etc/caddy/Caddyfile
# sudo systemctl reload caddy
#
# The app already runs with GPUK_HSTS=true and GPUK_SESSION_COOKIE_SECURE=true
# (set by the public profile). What does NOT go through this proxy:
# - the worker mTLS channel (:8443): workers pin the controller CA and must
# reach it DIRECTLY — terminating it here would break the pin.
# - worker<->worker data transfers (:8300): LAN-only by contract (OPS-68).
$DOMAIN {
encode zstd gzip
reverse_proxy localhost:$HTTP_PORT
}
# OpenAI-compatible inference endpoint (gpuk-proxy) — uncomment when inference
# clients live beyond the trusted LAN; TLS keeps their API keys off the wire.
#
# inference.$DOMAIN {
# encode zstd gzip
# reverse_proxy localhost:8200
# }
CADDY
chmod 0644 "$DATA_ROOT/caddy/Caddyfile"
echo "==> wrote $DATA_ROOT/caddy/Caddyfile (filled TLS reverse-proxy example for $DOMAIN)"
}
# Pre-existing model caches (B81): a server that installs GPU Kitchen usually already
# holds tens or hundreds of GB of weights. We print a hint and nothing more — the
# install stays NON-INTERACTIVE, no config of any other program is read or touched,
# and referencing a cache is an explicit choice made later in the UI.
# holds tens or hundreds of GB of weights. We print a hint and nothing more — cache
# questions belong to the first-run wizard (INS-48, REG-41), never to the CLI; no
# config of any other program is read or touched, and referencing a cache stays an
# explicit choice made in the UI.
hint_existing_caches() {
hec_chosen=$(readlink -f "$1" 2>/dev/null || echo "$1")
hec_found=""
@@ -176,20 +234,23 @@ hint_existing_caches() {
for hec_dir in $hec_found; do
echo "==> existing model cache found at $hec_dir"
done
echo " Reference it from Settings -> Cache folders to reuse those models."
echo " GPU Kitchen only READS a referenced cache: nothing is moved or deleted."
echo " GPU Kitchen will offer to reuse those models at first launch, and any"
echo " time from Nodes & GPU -> Storage. A reused cache is referenced in"
echo " place. Nothing is moved or deleted."
}
cmd_install() {
need_root
IMAGE=""; MODE="worker"; CLUSTER="default"; DATA_ROOT="/var/lib/gpu-kitchen"
CACHE_DIR=""; NETWORK="host"; CONTROLLER_URL=""; BIN_SRC=""; HEALTH_PORT="8001"
BROWSE_ROOTS=""; ENROLL_TOKEN=""; HTTP_PORT="8080"
# 1337, not 8080: kept in lockstep with install.sh's default (INS-01).
ENROLL_TOKEN=""; HTTP_PORT="1337"; PROFILE=""; DOMAIN=""; DRY_RUN=0
while [ $# -gt 0 ]; do
case "$1" in
--image) IMAGE="$2"; shift 2 ;;
--mode) MODE="$2"; shift 2 ;;
--cluster) CLUSTER="$2"; shift 2 ;;
--profile) PROFILE="$2"; shift 2 ;;
--domain) DOMAIN="$2"; shift 2 ;;
--data-root) DATA_ROOT="$2"; shift 2 ;;
--cache-dir) CACHE_DIR="$2"; shift 2 ;;
--network) NETWORK="$2"; shift 2 ;;
@@ -201,7 +262,7 @@ cmd_install() {
--health-port) HEALTH_PORT="$2"; shift 2 ;;
--http-port) HTTP_PORT="$2"; shift 2 ;;
--binary) BIN_SRC="$2"; shift 2 ;;
--browse-root) BROWSE_ROOTS="$BROWSE_ROOTS $2"; shift 2 ;;
--dry-run) DRY_RUN=1; shift ;;
*) die "unknown install option: $1" ;;
esac
done
@@ -215,20 +276,98 @@ cmd_install() {
*) die "--mode must be controller or worker (got '$MODE')" ;;
esac
[ "$MODE" != "controller" ] || [ -n "$IMAGE" ] \
|| die "--image registry/gpukitchen-controller:<tag> is required for a controller"
if [ "$MODE" = "worker" ] && [ -n "$PROFILE" ]; then
die "--profile applies only to --mode controller; workers do not have an installation profile"
fi
case "$PROFILE" in
""|homelab|studio|enterprise|public) ;;
*) die "--profile must be homelab, studio, enterprise or public (got '$PROFILE')" ;;
esac
if [ -n "$DOMAIN" ]; then
[ "$PROFILE" = "public" ] \
|| die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)"
case "$DOMAIN" in
*[!A-Za-z0-9.-]*) die "--domain must be a bare domain name (got '$DOMAIN')" ;;
esac
fi
if [ "$MODE" = "controller" ]; then
if [ -z "$IMAGE" ] && [ "$DRY_RUN" -eq 1 ]; then
IMAGE="example.invalid/gpukitchen-controller:v0.0.0-dry-run"
fi
[ -n "$IMAGE" ] || die "--image registry/gpukitchen-controller:<tag> is required for a controller"
case "$IMAGE" in
*@sha256:*) ;;
*:latest) die "refusing floating image tag '$IMAGE' — use an explicit release tag or digest" ;;
*)
_image_tag="${IMAGE##*:}"
case "$_image_tag" in
"$IMAGE"|*/*) die "controller image must carry an explicit tag or sha256 digest (got '$IMAGE')" ;;
esac
;;
esac
fi
[ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf"
CACHE_JSON="[{\"hostPath\":\"$(json_str "$CACHE_DIR")\",\"shared\":false}]"
SELF_ENROLL_FILE="$DATA_ROOT/self-enroll-token"
BOOTSTRAP_PASSWORD_FILE="$DATA_ROOT/secrets/bootstrap_admin_password"
CLAIM_CODE_FILE="$DATA_ROOT/secrets/claim_code"
CLAIM_CODE_AVAILABLE=0
if [ "$DRY_RUN" -eq 1 ]; then
[ "$MODE" != "controller" ] || CLAIM_CODE_AVAILABLE=1
echo "==> dry run: no file, service or container was changed"
if [ "$MODE" = "worker" ]; then render_worker_manifest; else render_controller_manifest; fi
[ -z "$DOMAIN" ] || echo "==> dry run: would write $DATA_ROOT/caddy/Caddyfile for $DOMAIN"
return 0
fi
# ── Port conflicts (INS-46) — the mutator's own guard ────────────────────────
# install.sh's preflight already checks these, and on a terminal it can offer
# an alternative port. gpuk is the actual mutator and contributors call it
# DIRECTLY, so it re-checks and refuses, non-interactively. Same helpers as
# install.sh (both scripts ship standalone from the channel). A listener owned
# by an existing install is not a conflict: a manifest on disk means the
# re-run is the update path, and every checked port is then our own.
if [ ! -f "$MANIFEST" ]; then
_port_tool=""
if command -v ss >/dev/null 2>&1; then _port_tool="ss"
elif command -v netstat >/dev/null 2>&1; then _port_tool="netstat"; fi
if [ -z "$_port_tool" ]; then
echo "==> warning: cannot check for port conflicts (no ss or netstat)"
else
if [ "$MODE" = "controller" ]; then
set -- "$HTTP_PORT" 8443 8200
else
# Worker data-plane ports (OPS-68) plus the local health listener.
set -- "$HEALTH_PORT" 8300 8301 8302
fi
for _p in "$@"; do
_busy=1
case "$_port_tool" in
ss) [ -n "$(ss -ltnH "sport = :$_p" 2>/dev/null)" ] || _busy=0 ;;
netstat)
netstat -ltn 2>/dev/null \
| awk -v p="$_p" '{n=split($4,a,":"); if (a[n]==p) found=1} END{exit found?0:1}' \
|| _busy=0
;;
esac
[ "$_busy" -eq 0 ] || die "port $_p is already in use. Free it first, then run the install again."
done
fi
fi
need_root
install_binary "$BIN_SRC"
mkdir -p "$ETC_DIR" "$DATA_ROOT" "$CACHE_DIR"
seed_secrets "$DATA_ROOT" "$MODE"
# Default browse roots: common mount points + the dirs we already use.
[ -n "$BROWSE_ROOTS" ] || BROWSE_ROOTS="/mnt /data /srv $DATA_ROOT $(dirname "$CACHE_DIR")"
# shellcheck disable=SC2086
BROWSE_JSON=$(json_array $BROWSE_ROOTS)
CACHE_JSON="[{\"hostPath\":\"$(json_str "$CACHE_DIR")\",\"shared\":false}]"
seed_secrets "$DATA_ROOT"
if [ "$MODE" = "controller" ]; then
prepare_claim_code
[ ! -s "$CLAIM_CODE_FILE" ] || CLAIM_CODE_AVAILABLE=1
fi
umask 077
if [ "$MODE" = "worker" ]; then
@@ -239,6 +378,7 @@ cmd_install() {
{
echo "GPUK_CLUSTER=$CLUSTER"
echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT"
echo "GPUK_MACHINE_ID_FILE=$MACHINE_ID_FILE"
echo "NODE_DISPLAY_NAME=$(hostname)"
[ -n "$CONTROLLER_URL" ] && echo "GPUK_CONTROLLER_URLS=$CONTROLLER_URL"
} > "$WORKER_ENV"
@@ -246,9 +386,21 @@ cmd_install() {
echo "==> wrote $MANIFEST (worker: no app container)"
else
write_controller_manifest
# The host daemon and app container share DATA_ROOT. Only controller-mode
# workerd gets this private bootstrap channel; remote workers stay tokenless.
{
echo "GPUK_CLUSTER=$CLUSTER"
echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT"
echo "GPUK_MACHINE_ID_FILE=$MACHINE_ID_FILE"
echo "GPUK_CONTROLLER_URLS=wss://127.0.0.1:8443"
echo "GPUK_SELF_ENROLL_FILE=$SELF_ENROLL_FILE"
echo "NODE_DISPLAY_NAME=$(hostname)"
} > "$WORKER_ENV"
chmod 0600 "$WORKER_ENV"
echo "==> wrote $MANIFEST (controller: app container $IMAGE)"
fi
chmod 0600 "$MANIFEST"
[ -z "$DOMAIN" ] || write_caddyfile
write_unit
systemctl daemon-reload
@@ -259,12 +411,17 @@ cmd_install() {
if [ -n "$ENROLL_TOKEN" ]; then
[ -n "$CONTROLLER_URL" ] || die "--controller wss://<controller>:<port> is required to enroll"
echo "==> enrolling against $CONTROLLER_URL ..."
GPUK_IDENTITY_DIR="$IDENTITY_DIR" "$BIN_DEST" enroll \
GPUK_IDENTITY_DIR="$IDENTITY_DIR" GPUK_MACHINE_ID_FILE="$MACHINE_ID_FILE" "$BIN_DEST" enroll \
--controller "$CONTROLLER_URL" --token "$ENROLL_TOKEN" \
|| die "enrollment failed (bad/expired token, or controller unreachable)"
else
echo "==> no token given: the daemon will discover its cluster on the LAN and wait"
if [ -n "$CONTROLLER_URL" ]; then
echo "==> no token given: the daemon will request admission from $CONTROLLER_URL and wait"
echo " for automatic admission or administrator approval."
else
echo "==> no token or controller given: the daemon will discover its cluster on the LAN and wait"
echo " for automatic admission or administrator approval."
fi
fi
systemctl enable --now "$SERVICE_NAME"
echo "==> $SERVICE_NAME enabled and started"
@@ -282,18 +439,18 @@ cmd_install() {
# No unix socket any more: workerd reconciles the app container in-process from
# the on-disk manifest (there is no backend to relay through on the very first
# boot). Steady-state updates go through the daemon over WS.
"$BIN_DEST" apply || die "apply failed — check: gpuk logs"
"$BIN_DEST" apply || die "apply failed. Check: gpuk logs"
echo
echo "Done. The worker daemon is running and the app container is up."
echo " Status : gpuk status App logs: gpuk logs Daemon: journalctl -u $SERVICE_NAME"
hint_existing_caches "$CACHE_DIR"
}
# Worker manifest: cacheDisks + browseRoots, and an EMPTY image so
# Worker manifest: cacheDisks and an EMPTY image so
# has_app_container() is false (apps/worker/src/modules/manifest.rs). No ports, no env,
# no secretsRef — a worker runs no app container.
write_worker_manifest() {
cat > "$MANIFEST" <<JSON
render_worker_manifest() {
cat <<JSON
{
"schemaVersion": 1,
"image": "",
@@ -301,29 +458,52 @@ write_worker_manifest() {
"cluster": "$(json_str "$CLUSTER")",
"gpus": "all",
"dataRoot": "$(json_str "$DATA_ROOT")",
"cacheDisks": $CACHE_JSON,
"browseRoots": $BROWSE_JSON
"cacheDisks": $CACHE_JSON
}
JSON
}
write_worker_manifest() { render_worker_manifest > "$MANIFEST"; }
# Controller manifest: the declarative app-container description workerd applies.
write_controller_manifest() {
render_controller_manifest() {
# NOTE there is deliberately no PORT here: PORT is the backend's own port, a
# loopback-only 8000 behind nginx in the all-in-one image. What the outside
# world dials is GPUK_PORT.
# loopback-only 8000 behind nginx in the all-in-one image. What the outside world
# dials is GPUK_PUBLIC_PORT (the published host port), which falls back to nginx's
# own GPUK_PORT when nothing republishes it.
EXTRA_ENV=""
EXTRA_ENV="$EXTRA_ENV,\"GPUK_PORT\":\"$(json_str "$HTTP_PORT")\""
EXTRA_ENV="$EXTRA_ENV,\"GPUK_DATA_ROOT\":\"$(json_str "$DATA_ROOT")\""
[ -z "$PROFILE" ] || EXTRA_ENV="$EXTRA_ENV,\"GPUK_INSTALL_PROFILE\":\"$(json_str "$PROFILE")\""
if [ "$PROFILE" = "public" ]; then
EXTRA_ENV="$EXTRA_ENV,\"GPUK_HSTS\":\"true\""
EXTRA_ENV="$EXTRA_ENV,\"GPUK_SESSION_COOKIE_SECURE\":\"true\""
fi
BOOTSTRAP_SECRET_JSON=""
if [ -s "$BOOTSTRAP_PASSWORD_FILE" ]; then
EXTRA_ENV="$EXTRA_ENV,\"GPUK_BOOTSTRAP_MUST_CHANGE\":\"1\""
BOOTSTRAP_SECRET_JSON=",\"GPUK_BOOTSTRAP_ADMIN_PASSWORD\":\"$(json_str "$BOOTSTRAP_PASSWORD_FILE")\""
fi
CLAIM_SECRET_JSON=""
if [ "$CLAIM_CODE_AVAILABLE" -eq 1 ]; then
CLAIM_SECRET_JSON=",\"GPUK_CLAIM_CODE\":\"$(json_str "$CLAIM_CODE_FILE")\""
fi
[ "$NETWORK" = "host" ] || EXTRA_ENV="$EXTRA_ENV,\"BACKEND_DOCKER_NETWORK\":\"$(json_str "$NETWORK")\""
# Published != bound (INS-43). On a bridged network the container keeps the image's
# FIXED listeners — nginx 8080, worker mTLS 8443 — and --http-port only moves the HOST
# side of the publication; Settings -> Network moves it later by patching this same
# manifest, so the container port must never become a variable. With host networking
# nothing is published and the listener itself takes the port.
PORTS_JSON="{}"
if [ "$NETWORK" != "host" ]; then
PORTS_JSON="{\"$HTTP_PORT\":$HTTP_PORT,\"8200\":8200}"
if [ "$NETWORK" = "host" ]; then
EXTRA_ENV="$EXTRA_ENV,\"GPUK_PORT\":\"$(json_str "$HTTP_PORT")\""
else
EXTRA_ENV="$EXTRA_ENV,\"GPUK_PORT\":\"8080\""
EXTRA_ENV="$EXTRA_ENV,\"GPUK_PUBLIC_PORT\":\"$(json_str "$HTTP_PORT")\""
PORTS_JSON="{\"8080\":$HTTP_PORT,\"8200\":8200,\"8443\":8443}"
fi
cat > "$MANIFEST" <<JSON
cat <<JSON
{
"schemaVersion": 1,
"image": "$(json_str "$IMAGE")",
@@ -335,29 +515,34 @@ write_controller_manifest() {
"restartPolicy": "unless-stopped",
"dataRoot": "$(json_str "$DATA_ROOT")",
"cacheDisks": $CACHE_JSON,
"browseRoots": $BROWSE_JSON,
"ports": $PORTS_JSON,
"env": {
"NODE_ENV": "production",
"GPUK_MODE": "controller",
"GPUK_CLUSTER": "$(json_str "$CLUSTER")",
"GPUK_SELF_ENROLL_FILE": "$(json_str "$SELF_ENROLL_FILE")",
"NODE_DISPLAY_NAME": "$(json_str "$(hostname)")",
"HF_HOME": "$(json_str "$CACHE_DIR")"$EXTRA_ENV
},
"secretsRef": {
"ENCRYPTION_KEY": "$(json_str "$DATA_ROOT/secrets/encryption_key")",
"NODE_ID": "$(json_str "$DATA_ROOT/secrets/node_id")",
"GPUK_BOOTSTRAP_ADMIN_PASSWORD": "$(json_str "$DATA_ROOT/secrets/bootstrap_admin_password")"
"NODE_ID": "$(json_str "$DATA_ROOT/secrets/node_id")"$BOOTSTRAP_SECRET_JSON$CLAIM_SECRET_JSON
}
}
JSON
}
write_controller_manifest() { render_controller_manifest > "$MANIFEST"; }
# ── control subcommands ────────────────────────────────────────────────────────
container_name() {
sed -n 's/.*"containerName"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1
}
manifest_data_root() {
sed -n 's/.*"dataRoot"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1
}
manifest_image() {
sed -n 's/.*"image"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1
}
@@ -374,9 +559,9 @@ health_port() {
# Rule: it is a tag only if the last colon comes after the last slash.
# (Same logic as apps/controller/api/src/core/release-channel.ts — see its unit tests.)
image_repo() {
case "$1" in
*@*) printf '%s' "${1%@*}"; return ;; # digest pin → repo@sha256:…
esac
# A digest suffix (…@sha256:…) never carries the repo; drop it, then apply
# the tag logic — `repo:tag@sha256:…` and `repo@sha256:…` both reduce right.
set -- "${1%@*}"
_t="${1##*:}"
case "$_t" in
"$1") printf '%s' "$1" ;; # no colon at all → untagged
@@ -386,9 +571,9 @@ image_repo() {
}
image_tag() {
case "$1" in
*@*) return ;; # digest pin: no version to speak of
esac
# `repo:tag@sha256:…` keeps the human-readable tag next to the content pin —
# docker resolves by digest and ignores the tag. Strip the digest, then parse.
set -- "${1%@*}"
_t="${1##*:}"
case "$_t" in
"$1") return ;;
@@ -403,10 +588,138 @@ image_tag() {
# https://gpu.kitchen/latest.json once the hub exists (specs/developpement/ci-cd.md).
CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen/channel/raw/branch/main/latest.json}"
channel_version() {
# The release public key pinned in THIS copy of gpuk (OPS-20). The channel copy
# gets the real key substituted at publish time; the operator override
# (GPUK_UPDATE_PUBKEY) covers a self-hosted channel with its own keypair. The
# first install fetched gpuk itself over HTTPS from the channel — that moment is
# trust-on-first-use, like a worker's enrolment token pin; every later `update`
# is verified against the key pinned HERE, so whoever controls latest.json can
# no longer pick what an existing install runs.
CHANNEL_PUBKEY="${GPUK_UPDATE_PUBKEY:-RWQ7BKXJqGX2jdKXu1GxeSPVAN3JDRTefpImM/mFRjtFwq4E7mhnQtA7}"
# Fetch latest.json AND its minisign signature, verify, and leave the verified
# document at $CHANNEL_DOC. Fail-closed: no signature, bad signature, no
# minisign CLI or no pinned key are all fatal — GPUK_CHANNEL_INSECURE=1 is the
# explicit, logged opt-out (a private mirror that does not sign).
CHANNEL_DOC=""
channel_fetch() {
command -v curl >/dev/null 2>&1 || return 1
curl -fsSL --max-time 20 "$CHANNEL_URL" 2>/dev/null \
| sed -n 's/.*"version"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' | head -1
CHANNEL_DOC=$(mktemp) || return 1
curl -fsSL --max-time 20 "$CHANNEL_URL" -o "$CHANNEL_DOC" 2>/dev/null || return 1
if [ "${GPUK_CHANNEL_INSECURE:-}" = "1" ]; then
echo "WARNING: GPUK_CHANNEL_INSECURE=1 — release channel signature NOT verified" >&2
return 0
fi
case "$CHANNEL_PUBKEY" in
""|RWQ7BKXJqGX2jdKXu1GxeSPVAN3JDRTefpImM/mFRjtFwq4E7mhnQtA7*)
die "this gpuk carries no pinned release public key — set GPUK_UPDATE_PUBKEY (the minisign public-key line), or GPUK_CHANNEL_INSECURE=1 to skip verification" ;;
esac
command -v minisign >/dev/null 2>&1 \
|| die "minisign is required to verify the release channel (apt install minisign), or set GPUK_CHANNEL_INSECURE=1"
_sig=$(mktemp)
if ! curl -fsSL --max-time 20 "${CHANNEL_URL}.minisig" -o "$_sig" 2>/dev/null; then
rm -f "$_sig"
die "no signature at ${CHANNEL_URL}.minisig — refusing an unsigned channel document (OPS-20)"
fi
if ! minisign -Vq -m "$CHANNEL_DOC" -x "$_sig" -P "$CHANNEL_PUBKEY" >/dev/null 2>&1; then
rm -f "$_sig"
die "latest.json signature verification FAILED — refusing the channel document (OPS-20)"
fi
rm -f "$_sig"
}
channel_field() {
sed -n 's/.*"'"$2"'"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$1" | head -1
}
channel_version() {
channel_fetch || return 1
channel_field "$CHANNEL_DOC" version
}
# ── backup ─────────────────────────────────────────────────────────────────────
# OPS-10: the embedded Postgres is only backed up COLD — hot-copying pgdata with
# a file tool is forbidden (torn pages). Order matters: stop the DAEMON first
# (its reconciler would immediately restart a stopped app container), then the
# container, snapshot, and restarting the service re-applies the manifest.
# An install on an external DATABASE_URL is refused here: gpuk only owns the
# embedded pgdata — back the real database up with pg_dump/backup-compose.sh.
# The finished directory still has to be copied to encrypted off-host storage,
# next to the recovery set (OPS-04/OPS-05: ENCRYPTION_KEY above all).
cmd_backup() {
need_root
_img=$(manifest_image)
[ -n "$_img" ] || die "this is a worker node — no controller data to back up here"
if grep -q '"DATABASE_URL"' "$MANIFEST" 2>/dev/null; then
die "this install uses an external DATABASE_URL — back THAT database up (pg_dump, or the compose procedures in specs/plateforme/operations.md); gpuk backup only snapshots the embedded pgdata"
fi
_root=$(manifest_data_root)
[ -n "$_root" ] || die "no dataRoot in $MANIFEST"
[ -d "$_root/pgdata" ] || die "no embedded pgdata under $_root — nothing to snapshot"
command -v sha256sum >/dev/null 2>&1 || die "sha256sum is required"
_out="${1:-$_root/backups/$(date -u +%Y%m%dT%H%M%SZ)}"
[ -e "$_out" ] && die "refusing to overwrite existing $_out"
mkdir -p "$(dirname "$_out")"
_tmp="$_out.partial"
rm -rf "$_tmp"; mkdir -p "$_tmp"
_cn=$(container_name)
echo "==> Stopping $SERVICE_NAME (its reconciler would restart the container mid-snapshot)..."
systemctl stop "$SERVICE_NAME" || die "could not stop $SERVICE_NAME"
_restart_daemon() { systemctl start "$SERVICE_NAME" 2>/dev/null || true; }
trap _restart_daemon EXIT
if [ -n "$_cn" ]; then
echo "==> Stopping $_cn (cold snapshot — OPS-10)..."
docker stop "$_cn" >/dev/null 2>&1 || true
_state=$(docker inspect -f '{{.State.Status}}' "$_cn" 2>/dev/null || echo absent)
case "$_state" in
running) die "container $_cn is still running — refusing a hot snapshot" ;;
esac
fi
echo "==> Snapshotting $_root/pgdata..."
tar -C "$_root" -czf "$_tmp/pgdata.tar.gz" pgdata || die "snapshot failed"
cp "$MANIFEST" "$_tmp/host-manifest.json" 2>/dev/null || true
{
echo "{"
echo " \"created_utc\": \"$(date -u +%Y-%m-%dT%H:%M:%SZ)\","
echo " \"image\": \"$(json_str "$_img")\","
echo " \"data_root\": \"$(json_str "$_root")\","
echo " \"kind\": \"cold-pgdata-snapshot\""
echo "}"
} > "$_tmp/backup-manifest.json"
(cd "$_tmp" && sha256sum ./* > SHA256SUMS) || die "checksums failed"
chmod 0700 "$_tmp"
mv "$_tmp" "$_out"
echo "==> Restarting $SERVICE_NAME (re-applies the manifest, container included)..."
systemctl start "$SERVICE_NAME" || die "could not restart $SERVICE_NAME — start it manually"
trap - EXIT
echo "backup: $_out"
echo "Copy it to encrypted OFF-HOST storage together with the recovery set"
echo "(ENCRYPTION_KEY above all — without it the data is unrecoverable, OPS-04)."
echo "A physical pgdata restore requires the same Postgres major and a throwaway"
echo "host rehearsal first (OPS-10)."
}
# ── channel ────────────────────────────────────────────────────────────────────
# Diagnostic (no root): fetch + VERIFY the channel document, print what it
# offers. Exercises exactly the trust chain `update` relies on — the CI probes
# it with a throwaway keypair, an operator uses it to debug a mirror.
cmd_channel() {
channel_fetch || die "cannot fetch $CHANNEL_URL"
echo "channel : $CHANNEL_URL"
echo "version : $(channel_field "$CHANNEL_DOC" version)"
_d=$(channel_field "$CHANNEL_DOC" controllerImageDigest)
[ -n "$_d" ] && echo "digest : $_d"
_d=$(channel_field "$CHANNEL_DOC" controllerImageDigestEnterprise)
[ -n "$_d" ] && echo "digest ee : $_d"
if [ "${GPUK_CHANNEL_INSECURE:-}" = "1" ]; then
echo "signature : SKIPPED (GPUK_CHANNEL_INSECURE=1)"
else
echo "signature : verified"
fi
}
# ── status ─────────────────────────────────────────────────────────────────────
@@ -447,7 +760,8 @@ cmd_enroll() {
done
[ -n "$_url" ] || die "enroll needs --controller wss://<controller>:<port>"
[ -n "$_tok" ] || die "enroll needs --token gk_enroll_..."
GPUK_IDENTITY_DIR="$IDENTITY_DIR" "$BIN_DEST" enroll --controller "$_url" --token "$_tok"
GPUK_IDENTITY_DIR="$IDENTITY_DIR" GPUK_MACHINE_ID_FILE="$MACHINE_ID_FILE" \
"$BIN_DEST" enroll --controller "$_url" --token "$_tok"
systemctl restart "$SERVICE_NAME" 2>/dev/null || true
}
@@ -499,20 +813,35 @@ cmd_update() {
return 0
fi
_digest=""
if [ -z "$_want" ]; then
_want=$(channel_version) || true
# channel_fetch runs in THIS shell (not a $(…) subshell) so a signature
# failure is fatal here — fail-closed — and $CHANNEL_DOC survives. The
# verified signature closes the document half of OPS-20; the digest read
# from it pins CONTENT, closing the mutable-tag half.
if channel_fetch; then
_want=$(channel_field "$CHANNEL_DOC" version)
fi
if [ -z "$_want" ]; then
echo "==> cannot reach $CHANNEL_URL — re-applying the pinned $_current"
"$BIN_DEST" apply
return 0
fi
# Pick the digest matching the installed edition by image basename — the
# repo itself may be a mirror, the basename is the edition marker.
case "${_repo##*/}" in
*-ee) _digest=$(channel_field "$CHANNEL_DOC" controllerImageDigestEnterprise) ;;
*) _digest=$(channel_field "$CHANNEL_DOC" controllerImageDigest) ;;
esac
fi
if [ "$_want" != "$_current" ]; then
echo "==> $_current → $_want"
# Pin the new tag into the manifest, then apply. sed edits the single "image"
# line in place (atomic tmp + move).
_new="$_repo:$_want"
[ -n "$_digest" ] && _new="$_repo:$_want@$_digest"
_installed=$(manifest_image)
if [ "$_new" != "$_installed" ]; then
echo "==> $_current → $_want${_digest:+ (pinned by digest)}"
# Pin the new reference into the manifest, then apply. sed edits the single
# "image" line in place (atomic tmp + move).
_tmp="$MANIFEST.new"
sed "s#\"image\"[[:space:]]*:[[:space:]]*\"[^\"]*\"#\"image\": \"$(json_str "$_new")\"#" "$MANIFEST" > "$_tmp" \
|| die "could not rewrite the image in $MANIFEST"
@@ -535,16 +864,20 @@ usage() {
cat <<EOF
gpuk — GPU Kitchen host daemon (gpu-kitchen-worker)
gpuk install --mode controller --image <ref> [--cluster N] [--cache-dir P]
gpuk install --mode controller --image <ref> [--profile homelab|studio|enterprise|public]
[--cluster N] [--cache-dir P]
[--data-root P] [--http-port P] [--network host|bridge|<net>] [--binary <path>]
gpuk install --mode worker --controller wss://<host>:<port> --enroll-token gk_enroll_...
[--cluster N] [--cache-dir P] [--browse-root P]... [--binary <path>]
[--cluster N] [--cache-dir P] [--binary <path>]
gpuk install ... --dry-run Validate inputs and print the manifest without changing the host
gpuk status Service state, enrollment, /health, app container status
gpuk enroll --controller wss://<host>:<port> --token gk_enroll_...
gpuk apply (controller) Reconcile the app container from the manifest
gpuk update (controller) Move to the current release (pull + recreate, rollback)
gpuk update --check (controller) Compare the installed version with the release
gpuk update --version <tag> (controller) Move to a specific release
gpuk channel Fetch + VERIFY the release channel and print what it offers
gpuk backup [DIR] (controller) Cold snapshot of the embedded pgdata (stop → tar → restart)
gpuk manifest Print the current manifest
gpuk logs Follow the app container logs (controller) or the daemon journal
gpuk uninstall Remove the systemd service
@@ -562,6 +895,8 @@ case "$cmd" in
enroll) cmd_enroll "$@" ;;
apply) cmd_apply ;;
update) cmd_update "$@" ;;
channel) cmd_channel ;;
backup) cmd_backup "${1:-}" ;;
manifest) cat "$MANIFEST" ;;
logs)
_img=$(manifest_image)
+314 -32
View File
@@ -7,15 +7,20 @@
# channel repo is written by .gitea/workflows/release.yml, see specs/developpement/ci-cd.md.)
#
# What it does, and nothing more:
# 1. preflight docker, the NVIDIA driver, and a REAL `--gpus all` smoke test
# 1. preflight docker, the NVIDIA driver, a REAL `--gpus all` smoke test, and
# the listening ports (INS-46 — a taken port fails HERE, not three
# minutes later in a health-check timeout)
# 2. resolve the current release from the channel (a TAG — never a floating
# `latest`: an install that silently changes version under you is
# not an install, it is a surprise)
# 3. install `gpu-kitchen-worker` (workerd), the privileged host daemon, via
# 3. ask the security profile — and, under `public`, a domain — when a
# terminal is attached (INS-45); no terminal, no questions, and
# the first-run wizard asks instead (PRF-05)
# 4. install `gpu-kitchen-worker` (workerd), the privileged host daemon, via
# the existing `gpuk` installer — on a controller node it OWNS the
# app container's lifecycle
# 4. hand over workerd pulls the pinned all-in-one controller image and starts it
# 5. print the UI URL on the real host, and the first-run password
# 5. hand over workerd pulls the pinned all-in-one controller image and starts it
# 6. print the UI URL on the real host, plus the profile-specific next step
#
# Re-running is how you UPDATE: same command, newer tag, `docker pull` + recreate,
# data untouched (it lives in the data root, not the container).
@@ -39,14 +44,24 @@ CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen/channel/raw
EDITION="${GPUK_EDITION:-community}"
DATA_ROOT="${GPUK_DATA_ROOT:-/var/lib/gpu-kitchen}"
CACHE_DIR="${GPUK_CACHE_DIR:-}"
PORT="${GPUK_PORT:-8080}"
# 1337, not 8080: the single most-squatted port in existence would make the
# conflict preflight fire on half the lab boxes out there (INS-01).
PORT="${GPUK_PORT:-1337}"
# Fixed listeners: mirror of the gpuk-proxy default (core/cluster-settings.ts
# proxyPublicPort) and of the worker mTLS channel — no install-time flag moves
# them (INS-46).
INFERENCE_PORT=8200
MTLS_PORT=8443
CLUSTER="${GPUK_CLUSTER:-default}"
PROFILE=""
DOMAIN=""
VERSION=""
IMAGE=""
WORKER_BINARY=""
GPUK_SCRIPT=""
SKIP_GPU_CHECK=0
SKIP_PREFLIGHT=0
NON_INTERACTIVE=0
DRY_RUN=0
GREEN=''; RED=''; YELLOW=''; BOLD=''; RESET=''
@@ -60,6 +75,21 @@ warn() { echo " ${YELLOW}!${RESET} $*"; }
step() { echo; echo "${BOLD}$*${RESET}"; }
die() { echo " ${RED}✗${RESET} $*" >&2; exit 1; }
# `curl … | sudo sh` leaves stdin holding the script itself, so questions are
# asked and answered on the controlling terminal — /dev/tty — when there is one
# (INS-45). No terminal (CI, provisioning), --non-interactive or --dry-run keep
# every historical flags-only behaviour.
can_prompt() {
[ "$NON_INTERACTIVE" -eq 0 ] || return 1
[ "$DRY_RUN" -eq 0 ] || return 1
(: < /dev/tty) 2>/dev/null
}
ask() { # $1 = prompt → $REPLY (empty on EOF)
printf '%s' "$1" > /dev/tty
IFS= read -r REPLY < /dev/tty || REPLY=""
}
usage() {
cat <<EOF
GPU Kitchen installer
@@ -71,10 +101,15 @@ Options:
--version <tag> Install this release instead of the channel's current one
--image <ref> Use this controller image outright (implies --version none)
--edition <ed> community (default) | enterprise
--port <p> Port the UI listens on (default 8080)
--port <p> Port the UI listens on (default 1337)
--data-root <path> Where the database and secrets live (default /var/lib/gpu-kitchen)
--cache-dir <path> Model cache (default <data-root>/hf)
--cluster <name> Cluster name workers join (default "default")
--profile <p> homelab | studio | enterprise | public (no flag + a terminal
= the script asks; no flag + no terminal = first-run asks)
--domain <d> Domain for the public profile: writes a filled TLS
reverse-proxy example to <data-root>/caddy/Caddyfile
--non-interactive Never ask anything, even with a terminal attached
--worker-binary <p> Use a locally-built gpu-kitchen-worker instead of downloading one
--gpuk-script <p> Use a local copy of the gpuk installer
--skip-gpu-check Skip the 'docker run --gpus all' smoke test
@@ -93,6 +128,9 @@ while [ $# -gt 0 ]; do
--data-root) DATA_ROOT="$2"; shift 2 ;;
--cache-dir) CACHE_DIR="$2"; shift 2 ;;
--cluster) CLUSTER="$2"; shift 2 ;;
--profile) PROFILE="$2"; shift 2 ;;
--domain) DOMAIN="$2"; shift 2 ;;
--non-interactive) NON_INTERACTIVE=1; shift ;;
--worker-binary) WORKER_BINARY="$2"; shift 2 ;;
--gpuk-script) GPUK_SCRIPT="$2"; shift 2 ;;
--skip-gpu-check) SKIP_GPU_CHECK=1; shift ;;
@@ -110,6 +148,15 @@ case "$EDITION" in
*) die "--edition must be community or enterprise (got '$EDITION')" ;;
esac
case "$PROFILE" in
""|homelab|studio|enterprise|public) ;;
*) die "--profile must be homelab, studio, enterprise or public (got '$PROFILE')" ;;
esac
case "$DOMAIN" in
*[!A-Za-z0-9.-]*) die "--domain must be a bare domain name (got '$DOMAIN')" ;;
esac
echo
echo "${BOLD}GPU Kitchen — installing the $EDITION controller${RESET}"
@@ -129,16 +176,16 @@ step "Preflight — docker, NVIDIA driver, container toolkit"
[ "$DRY_RUN" -eq 1 ] || [ "$(id -u)" -eq 0 ] \
|| die "run as root: curl -fsSL … | sudo sh"
command -v curl >/dev/null 2>&1 || die "curl not found — install it first"
command -v curl >/dev/null 2>&1 || die "curl not found. Install curl first."
if command -v docker >/dev/null 2>&1; then
if docker info >/dev/null 2>&1; then
ok "docker daemon reachable ($(docker --version | cut -d' ' -f3 | tr -d ,))"
else
die "docker is installed but its daemon is unreachable (is it running? are you root?)"
die "docker is installed but its daemon does not answer. Start the daemon, or run as root."
fi
else
die "docker not found — install Docker Engine first: https://docs.docker.com/engine/install/"
die "docker not found. Install Docker Engine first: https://docs.docker.com/engine/install/"
fi
if command -v nvidia-smi >/dev/null 2>&1; then
@@ -150,7 +197,7 @@ if command -v nvidia-smi >/dev/null 2>&1; then
die "nvidia-smi is present but reports no GPU"
fi
else
die "nvidia-smi not found — install the NVIDIA driver first"
die "nvidia-smi not found. Install the NVIDIA driver first."
fi
if [ "$SKIP_GPU_CHECK" -eq 1 ]; then
@@ -163,10 +210,10 @@ elif docker info 2>/dev/null | grep -qiE 'runtimes:.*nvidia'; then
if docker run --rm --gpus all ubuntu:24.04 nvidia-smi -L >/dev/null 2>&1; then
ok "nvidia-container-toolkit works (a container can see the GPUs)"
else
die "the nvidia runtime is registered but 'docker run --gpus all' cannot see the GPUs — reinstall nvidia-container-toolkit"
die "the nvidia runtime is registered but 'docker run --gpus all' cannot see the GPUs. Reinstall nvidia-container-toolkit."
fi
else
die "nvidia-container-toolkit is not registered with docker — install it, then restart dockerd:
die "nvidia-container-toolkit is not registered with docker. Install it, then restart dockerd:
https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html"
fi
@@ -180,7 +227,109 @@ FREE_GB=$(df -BG --output=avail "$CACHE_PARENT" 2>/dev/null | tail -1 | tr -dc '
if [ "${FREE_GB:-0}" -ge 100 ]; then
ok "model cache $CACHE_DIR — ${FREE_GB}G free"
else
warn "only ${FREE_GB:-?}G free under $CACHE_PARENT — model weights want 100G+"
warn "only ${FREE_GB:-?}G free under $CACHE_PARENT. Model weights need 100G or more."
fi
# ── Port conflicts (INS-46) ──
# A taken port must fail HERE, before anything mutates the host — today's
# alternative is a 3-minute health-check timeout with zero diagnosis. Best-effort
# detection (ss, then netstat); neither present is a warn, never a false red.
# A listener owned by an EXISTING GPU Kitchen install is not a conflict: re-running
# this script is the documented update path, and the manifest names our port.
# Test hook: force the detector. The netstat fallback is unreachable on any
# host that has ss (all of them, in practice), so the CI smoke pins it here to
# keep its parsing honest.
PORT_TOOL="${GPUK_PORT_CHECK_TOOL:-}"
case "$PORT_TOOL" in
""|ss|netstat) ;;
*) die "GPUK_PORT_CHECK_TOOL must be ss or netstat (got '$PORT_TOOL')" ;;
esac
if [ -z "$PORT_TOOL" ]; then
if command -v ss >/dev/null 2>&1; then PORT_TOOL="ss"
elif command -v netstat >/dev/null 2>&1; then PORT_TOOL="netstat"; fi
fi
port_busy() { # $1 = port → 0 iff something listens on TCP :$1
case "$PORT_TOOL" in
ss) [ -n "$(ss -ltnH "sport = :$1" 2>/dev/null)" ] ;;
netstat) netstat -ltn 2>/dev/null \
| awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) found=1} END{exit found?0:1}' ;;
*) return 1 ;;
esac
}
port_owner() { # $1 = port → best-effort process name (needs root for -p)
case "$PORT_TOOL" in
ss) ss -ltnpH "sport = :$1" 2>/dev/null \
| sed -n 's/.*users:((\"\([^"]*\)\".*/\1/p' | head -1 ;;
netstat) netstat -ltnp 2>/dev/null \
| awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) {print $NF; exit}}' \
| sed 's|^[0-9]*/||' ;;
esac
}
manifest_ui_port() { # the port an existing install already owns, if any
[ -f /etc/gpu-kitchen/manifest.json ] || return 0
# Bridge publishes GPUK_PUBLIC_PORT over the fixed container 8080; host
# networking moves the listener itself (GPUK_PORT). Same precedence as gpuk.
_p=$(sed -n 's/.*"GPUK_PUBLIC_PORT"[[:space:]]*:[[:space:]]*"\([0-9]*\)".*/\1/p' \
/etc/gpu-kitchen/manifest.json | head -1)
[ -n "$_p" ] || _p=$(sed -n 's/.*"GPUK_PORT"[[:space:]]*:[[:space:]]*"\([0-9]*\)".*/\1/p' \
/etc/gpu-kitchen/manifest.json | head -1)
printf '%s' "$_p"
}
if [ -z "$PORT_TOOL" ]; then
warn "cannot check for port conflicts (neither ss nor netstat found)"
else
HAVE_MANIFEST=0
[ ! -f /etc/gpu-kitchen/manifest.json ] || HAVE_MANIFEST=1
if [ "$HAVE_MANIFEST" -eq 1 ] && [ "$(manifest_ui_port)" = "$PORT" ]; then
ok "UI port $PORT — already ours (re-running is how you update)"
elif port_busy "$PORT"; then
OWNER=$(port_owner "$PORT")
OWNER="${OWNER:-an unknown process}"
if can_prompt; then
ALT=$((PORT + 1))
while port_busy "$ALT"; do ALT=$((ALT + 1)); done
# Propose, never auto-pick: the URL printed at the end and the idempotent
# re-run both need the operator to KNOW which port they chose.
ask " ${YELLOW}!${RESET} port $PORT is busy ($OWNER). Use $ALT instead? [$ALT], another port, or 'q' to abort: "
case "$REPLY" in
q|Q) die "port $PORT is in use by $OWNER. Run the install again with --port <p>." ;;
"") PORT="$ALT" ;;
*)
case "$REPLY" in
*[!0-9]*) die "not a port number: $REPLY" ;;
esac
if port_busy "$REPLY"; then
die "port $REPLY is busy too. Run the install again with --port <p>."
fi
PORT="$REPLY"
;;
esac
ok "UI port $PORT is free"
else
die "port $PORT is already in use by $OWNER. Pass --port <p> to choose another port."
fi
else
ok "UI port $PORT is free"
fi
# The mTLS and inference listeners have no install-time flag — assumed
# limitation (INS-46): the published mTLS port moves later via
# Settings -> Network. With a manifest present they are our own listeners.
if [ "$HAVE_MANIFEST" -eq 0 ]; then
if port_busy "$MTLS_PORT"; then
OWNER=$(port_owner "$MTLS_PORT")
die "port $MTLS_PORT (worker channel) is in use by ${OWNER:-an unknown process}. Free it first."
fi
if port_busy "$INFERENCE_PORT"; then
OWNER=$(port_owner "$INFERENCE_PORT")
die "port $INFERENCE_PORT (inference endpoint) is in use by ${OWNER:-an unknown process}. Free it first.
($INFERENCE_PORT is also HashiCorp Vault's default port.)"
fi
ok "worker channel port $MTLS_PORT and inference port $INFERENCE_PORT are free"
fi
fi
fi # end preflight
@@ -217,7 +366,7 @@ if [ -n "$IMAGE" ]; then
else
CHANNEL=$(curl -fsSL --max-time 20 "$CHANNEL_URL" 2>/dev/null) \
|| die "cannot reach the release channel at $CHANNEL_URL
(offline? pass --image <ref> to install a specific image directly)"
If this host is offline, pass --image <ref> to install a specific image directly."
[ -n "$VERSION" ] || VERSION=$(echo "$CHANNEL" | json_field version)
[ -n "$VERSION" ] || die "the release channel returned no version: $CHANNEL_URL"
@@ -241,6 +390,24 @@ else
[ -n "${GPUK_SCRIPT}" ] || GPUK_SCRIPT_URL=$(echo "$CHANNEL" | json_field gpukScript)
fi
# A direct --image and a channel version are held to the same immutable-image
# rule. A registry host:port is not a tag separator; only the last colon after
# the last slash counts. Digest pins are accepted too.
case "$IMAGE" in
*@sha256:*) ;;
*:latest) die "refusing floating image tag '$IMAGE'. Use an explicit release tag or digest." ;;
*)
IMAGE_TAG="${IMAGE##*:}"
case "$IMAGE_TAG" in
"$IMAGE"|*/*) die "controller image must carry an explicit tag or sha256 digest (got '$IMAGE')" ;;
esac
;;
esac
if [ -n "$DOMAIN" ] && [ -n "$PROFILE" ] && [ "$PROFILE" != "public" ]; then
die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)"
fi
if [ "$DRY_RUN" -eq 1 ]; then
step "Dry run — stopping here"
if [ "$SKIP_PREFLIGHT" -eq 1 ]; then
@@ -253,10 +420,85 @@ if [ "$DRY_RUN" -eq 1 ]; then
echo " data root : $DATA_ROOT"
echo " model cache : $CACHE_DIR"
echo " UI port : $PORT"
if [ -n "$PROFILE" ]; then
echo " profile : $PROFILE"
else
echo " profile : (asked on the terminal, or chosen during first run)"
fi
[ -z "$DOMAIN" ] || echo " domain : $DOMAIN"
exit 0
fi
# ── 3. The host daemon ───────────────────────────────────────────────────────
# ── 3. Resolve the installation profile (INS-45) ─────────────────────────────
# Only ever on a terminal, and only when --profile was not given. The answer is
# relayed verbatim as --profile: the backend stays the sole applier of profile
# defaults and floors. Without a terminal the historical path is untouched —
# profile unset, enterprise floors, the first-run wizard requires the choice
# (PRF-05).
if [ -z "$PROFILE" ] && can_prompt; then
step "Security profile — how will this kitchen be used?"
cat > /dev/tty <<'PROFILES'
1) Home lab a trusted home network
No sign-in on your home network. Nearby workers are found and join
without waiting for approval.
2) Studio one control station, shared compute
Control stays on this machine. Colleagues use API keys, while nearby
workers wait for your approval.
3) Enterprise a managed company network
Sign-in is required. The controller stays quiet on the network; known
workers can request your approval.
4) Public server direct internet exposure
Sign-in and hardened browser transport are required. Network discovery
is off and workers join only by token.
You can change this later in Settings. Stricter floors re-apply. Nothing
already issued is revoked.
PROFILES
while :; do
ask " Choose a profile [1-4, Enter = 1 (Home lab)]: "
case "$REPLY" in
""|1|homelab) PROFILE="homelab" ;;
2|studio) PROFILE="studio" ;;
3|enterprise) PROFILE="enterprise" ;;
4|public) PROFILE="public" ;;
*) printf '%s\n' " pick 1, 2, 3 or 4" > /dev/tty; continue ;;
esac
break
done
ok "profile: $PROFILE"
fi
# Under public, a domain lets gpuk write a FILLED TLS reverse-proxy example
# (INS-47). Optional: Enter skips, and the banner still points at the shipped
# Caddyfile.example.
if [ "$PROFILE" = "public" ] && [ -z "$DOMAIN" ] && can_prompt; then
while :; do
ask " Domain for HTTPS access (e.g. gpu.example.com — Enter to skip): "
# Accept the copy-paste reflex: strip a pasted scheme and anything after
# the first slash, then insist on a bare domain rather than skipping —
# a silently dropped answer would be discovered hours later, at DNS time.
REPLY="${REPLY#https://}"
REPLY="${REPLY#http://}"
REPLY="${REPLY%%/*}"
[ -n "$REPLY" ] || break
case "$REPLY" in
*[!A-Za-z0-9.-]*)
printf '%s\n' " not a bare domain name: $REPLY. Try again, or press Enter to skip." > /dev/tty
;;
*) DOMAIN="$REPLY"; break ;;
esac
done
fi
if [ -n "$DOMAIN" ] && [ "$PROFILE" != "public" ]; then
die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)"
fi
# ── 4. The host daemon ───────────────────────────────────────────────────────
step "Host daemon — gpu-kitchen-worker"
TMP=$(mktemp -d)
@@ -291,6 +533,9 @@ set -- install \
--cache-dir "$CACHE_DIR" \
--http-port "$PORT"
[ -z "$PROFILE" ] || set -- "$@" --profile "$PROFILE"
[ -z "$DOMAIN" ] || set -- "$@" --domain "$DOMAIN"
if [ -n "$WORKER_BINARY" ]; then
[ -f "$WORKER_BINARY" ] || die "no such worker binary: $WORKER_BINARY"
set -- "$@" --binary "$WORKER_BINARY"
@@ -303,15 +548,15 @@ fi
# else: gpuk reuses an already-installed binary, or fails with its own message.
step "Installing — this pulls the image, so it can take a few minutes"
sh "$GPUK_SCRIPT" "$@" || die "the install failed — see: journalctl -u gpu-kitchen-worker"
sh "$GPUK_SCRIPT" "$@" || die "the install failed. See: journalctl -u gpu-kitchen-worker"
# ── 4. Wait for the app, then say where it is ────────────────────────────────
# ── 5. Wait for the app, then say where it is ────────────────────────────────
step "Waiting for the controller to answer"
i=0
until curl -fsS -o /dev/null --max-time 3 "http://127.0.0.1:$PORT/health" 2>/dev/null; do
i=$((i + 1))
[ "$i" -lt 90 ] || die "the controller did not answer on :$PORT within 90s — see: gpuk logs"
[ "$i" -lt 90 ] || die "the controller did not answer on :$PORT within 90s. See: gpuk logs"
sleep 2
done
ok "the controller is up"
@@ -322,30 +567,66 @@ ok "the controller is up"
HOSTNAME_FQDN=$(hostname -f 2>/dev/null || hostname 2>/dev/null || echo localhost)
LAN_IP=$(ip route get 1.1.1.1 2>/dev/null | sed -n 's/.* src \([0-9.]*\).*/\1/p' | head -1)
PW_FILE="$DATA_ROOT/secrets/bootstrap_admin_password"
ADMIN_PW=""
[ -f "$PW_FILE" ] && ADMIN_PW=$(cat "$PW_FILE" 2>/dev/null || true)
echo
echo "${BOLD}${GREEN}GPU Kitchen is running.${RESET}"
echo
echo " ${BOLD}Open:${RESET} http://${HOSTNAME_FQDN}:${PORT}"
[ -n "$LAN_IP" ] && echo " http://${LAN_IP}:${PORT}"
echo
if [ -n "$ADMIN_PW" ]; then
echo " ${BOLD}Sign in:${RESET} admin@local"
echo " ${BOLD}Password:${RESET} ${ADMIN_PW}"
echo " (the first-run wizard asks you to change it)"
echo
CLAIM_CODE_FILE="$DATA_ROOT/secrets/claim_code"
CLAIM_CODE=""
[ ! -s "$CLAIM_CODE_FILE" ] || CLAIM_CODE=$(cat "$CLAIM_CODE_FILE" 2>/dev/null || true)
if [ -n "$CLAIM_CODE" ]; then
echo " ${BOLD}Claim code:${RESET} $CLAIM_CODE"
echo " ${BOLD}Read again:${RESET} $CLAIM_CODE_FILE (mode 0600; removed after claim)"
else
echo " ${BOLD}Claim code:${RESET} consumed (the first account already exists)"
fi
echo " Inference endpoint : http://${HOSTNAME_FQDN}:8200/v1"
echo
case "$PROFILE" in
public)
echo " ${BOLD}Next:${RESET} create the first administrator in the UI"
echo
# The public profile REQUIRES a TLS reverse proxy (OPS-13, INS-47) — the UI
# port speaks plain HTTP. The product cannot verify the proxy's presence
# (OPS-67), so the closest thing to enforcement is saying it here, clearly.
if [ -n "$DOMAIN" ]; then
echo " ${BOLD}HTTPS:${RESET} https://${DOMAIN} answers after these steps:"
echo " 1. DNS: point ${DOMAIN} at this machine's public IP"
echo " 2. Install Caddy: https://caddyserver.com/docs/install"
echo " 3. sudo cp $DATA_ROOT/caddy/Caddyfile /etc/caddy/Caddyfile"
echo " sudo systemctl reload caddy"
echo " A filled example was written to $DATA_ROOT/caddy/Caddyfile."
echo " Until then the UI answers in cleartext on the URLs above."
else
echo " ${BOLD}HTTPS:${RESET} the public profile requires a TLS reverse proxy before any"
echo " public exposure. See deployments/controller/Caddyfile.example."
fi
echo
;;
enterprise)
echo " ${BOLD}Next:${RESET} create the first administrator in the UI"
echo
;;
homelab|studio)
echo " ${BOLD}Next:${RESET} finish first-run in the UI"
echo
;;
"")
echo " ${BOLD}Next:${RESET} choose an installation profile in the first-run assistant"
echo
;;
esac
echo " Inference endpoint : http://${HOSTNAME_FQDN}:${INFERENCE_PORT}/v1"
echo " Version : ${VERSION}"
echo
# Pre-existing model caches (B81). `gpuk install` prints the same hint, but that
# scrolls past mid-install; this banner is where people actually look. Detection
# only — the install stays non-interactive and nothing outside $CACHE_DIR is
# touched, read as configuration, or modified.
# only — cache questions belong to the first-run wizard (INS-48, REG-41), never
# to the CLI, and nothing outside $CACHE_DIR is touched, read as configuration,
# or modified.
EXISTING_CACHE=""
CHOSEN_CACHE=$(readlink -f "$CACHE_DIR" 2>/dev/null || echo "$CACHE_DIR")
for d in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do
@@ -359,8 +640,9 @@ if [ -n "$EXISTING_CACHE" ]; then
for d in $EXISTING_CACHE; do
echo " ${BOLD}Existing model cache:${RESET} $d"
done
echo " Reference it from Settings -> Cache folders to reuse those models"
echo " without downloading again. GPU Kitchen only READS a referenced cache."
echo " GPU Kitchen will offer to reuse those models at first launch, and any"
echo " time from Nodes & GPU -> Storage. A reused cache is referenced in"
echo " place. Nothing is moved or deleted."
echo
fi
echo " Update : re-run this command, or press Update in the UI, or: gpuk update"
+5 -3
View File
@@ -1,9 +1,11 @@
{
"version": "v0.1.2",
"semver": "0.1.2",
"version": "v0.1.3",
"semver": "0.1.3",
"controllerImage": "repo.byterain.io/gpukitchen/gpukitchen-controller",
"controllerImageEnterprise": "repo.byterain.io/gpukitchen-private/gpukitchen-controller-ee",
"workerBase": "https://repo.byterain.io/api/packages/gpukitchen/generic/gpu-kitchen-worker/v0.1.2",
"controllerImageDigest": "sha256:609bfdf5e9abcebdae56d868217b7d6400bd6fa8c44737b2275bfc3d4db6a100",
"controllerImageDigestEnterprise": "sha256:9fe855a97932cc134948ace4d03de592652e5872749c27273e9e7f6e930cd7ab",
"workerBase": "https://repo.byterain.io/api/packages/gpukitchen/generic/gpu-kitchen-worker/v0.1.3",
"gpukScript": "https://repo.byterain.io/gpukitchen/channel/raw/branch/main/gpuk",
"releaseNotes": "https://repo.byterain.io/gpukitchen/channel"
}
+4
View File
@@ -0,0 +1,4 @@
untrusted comment: signature from minisign secret key
RUQ7BKXJqGX2je3pLbIHpqUv94Xn863QAziuHoQdpUFKIG0QvVOjHN2rPz7ooHA1Nh6hjttCUfpvGsuH4IOLfsVgTB9KyA0TDQo=
trusted comment: gpu-kitchen channel v0.1.3
dcB01BhY0qvL9wGDhVtTG3q0IimAB1Kfh/6yd/lZhov3cQ38cM2sLelVOoE6DpkXyYZ2OXzTA0tJJEDkt68nAQ==