1395 lines
66 KiB
Bash
1395 lines
66 KiB
Bash
#!/bin/sh
|
|
# gpuk — installer + control CLI for the GPU Kitchen host daemon (gpu-kitchen-worker).
|
|
#
|
|
# Since C1 a compute node runs NO backend. The single privileged host component is
|
|
# the Rust `gpu-kitchen-worker` daemon (workerd), running as root DIRECTLY on the
|
|
# host (not in a container). It owns NVML clock/power locks, whitelisted host
|
|
# browsing, the model-cache staging pipeline, the signed `gpu-kitchen-bench`
|
|
# runner, and — on a controller node — the app container's lifecycle (create,
|
|
# health-gate, roll back, pull+recreate to update) via its manifest. There is no
|
|
# separate host daemon and no unix control socket: workerd is driven over the
|
|
# wss+mTLS channel by the controller, and locally by this CLI.
|
|
#
|
|
# Two roles:
|
|
# worker a headless compute node. Installs the binary + systemd unit + a
|
|
# manifest (cacheDisks) with NO app container, NO
|
|
# Postgres, NO docker app image. It ENROLLS over mTLS with a
|
|
# enrollment token, or waits for LAN discovery admission, then runs.
|
|
# controller the full app + UI. Installs the binary + systemd unit + an
|
|
# app-container manifest (image, ports, env, secretsRef) that workerd
|
|
# applies (`gpu-kitchen-worker apply`, then cmd:host_apply in steady
|
|
# state).
|
|
#
|
|
# Install (root):
|
|
# # controller:
|
|
# curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/gpuk \
|
|
# | sudo sh -s -- install --mode controller \
|
|
# --image repo.byterain.io/gpukitchen/gpukitchen-controller:vX.Y.Z
|
|
# # worker (enroll against a controller with a single-use token from its UI):
|
|
# sudo ./deployments/install/gpuk install --mode worker \
|
|
# --controller wss://controller.lan:8443 --enroll-token gk_enroll_... \
|
|
# --cache-dir /mnt/models --binary apps/worker/target/release/gpu-kitchen-worker
|
|
#
|
|
# Control:
|
|
# gpuk status | apply | update | enroll | logs | manifest | uninstall
|
|
#
|
|
set -eu
|
|
|
|
# ── Paths ────────────────────────────────────────────────────────────────────
|
|
# Overridable, so the daemon can be driven against a prefix a normal user owns —
|
|
# the only way any of this is testable without handing a test suite root on the
|
|
# host. Unset (the real install) they are exactly the systemd defaults workerd uses
|
|
# (apps/worker/src/module.rs, apps/worker/src/identity.rs).
|
|
BIN_DEST="${GPUK_BIN_DEST:-/usr/local/bin/gpu-kitchen-worker}"
|
|
# This CLI, installed next to the daemon by `gpuk install`: it carries the
|
|
# release key pinned at install time, and install.sh hands a plain re-run to its
|
|
# `update` (INS-03, OPS-20).
|
|
CLI_DEST="${GPUK_CLI_DEST:-$(dirname "$BIN_DEST")/gpuk}"
|
|
ETC_DIR="${GPUK_ETC_DIR:-/etc/gpu-kitchen}"
|
|
MANIFEST="$ETC_DIR/manifest.json"
|
|
IDENTITY_DIR="${GPUK_IDENTITY_DIR:-$ETC_DIR/identity}"
|
|
MACHINE_ID_FILE="${GPUK_MACHINE_ID_FILE:-$ETC_DIR/machine-id}"
|
|
WORKER_ENV="$ETC_DIR/worker.env"
|
|
SERVICE_NAME="gpu-kitchen-worker"
|
|
UNIT_DEST="${GPUK_UNIT_DEST:-/etc/systemd/system/$SERVICE_NAME.service}"
|
|
|
|
# Where to download the binary from when no --binary is given.
|
|
GPUK_RELEASE_BASE="${GPUK_RELEASE_BASE:-}"
|
|
|
|
die() { echo "gpuk: $*" >&2; exit 1; }
|
|
|
|
# Root, or able to do the job anyway. Fail with a clear message BEFORE touching
|
|
# /etc, /usr/local/bin or systemd. Against a user-owned prefix it simply works.
|
|
need_root() {
|
|
[ "$(id -u)" -eq 0 ] && return 0
|
|
[ -w "$ETC_DIR" ] && return 0
|
|
die "this command must run as root (use sudo)"
|
|
}
|
|
|
|
# ── JSON helpers (controlled inputs: paths + identifiers) ──────────────────────
|
|
json_str() { printf '%s' "$1" | sed 's/\\/\\\\/g; s/"/\\"/g'; }
|
|
|
|
json_array() { # args → ["a","b",...]
|
|
_out=""
|
|
for _p in "$@"; do
|
|
_e=$(json_str "$_p")
|
|
if [ -z "$_out" ]; then _out="\"$_e\""; else _out="$_out,\"$_e\""; fi
|
|
done
|
|
printf '[%s]' "$_out"
|
|
}
|
|
|
|
# ── install ────────────────────────────────────────────────────────────────────
|
|
arch_asset() {
|
|
case "$(uname -m)" in
|
|
x86_64|amd64) echo "gpu-kitchen-worker-x86_64" ;;
|
|
aarch64|arm64) echo "gpu-kitchen-worker-aarch64" ;;
|
|
*) die "unsupported architecture: $(uname -m)" ;;
|
|
esac
|
|
}
|
|
|
|
# [INS-19] A (re)install has just replaced the binary under a daemon that may be
|
|
# running: `systemctl enable --now` is a no-op on an active unit and would leave the
|
|
# OLD process in memory, speaking the old protocol to the new controller. Enable, then
|
|
# restart — which also starts a daemon that was not running.
|
|
start_daemon() {
|
|
systemctl enable "$SERVICE_NAME"
|
|
systemctl restart "$SERVICE_NAME"
|
|
}
|
|
|
|
install_binary() { # [local-path]
|
|
if [ -n "${1:-}" ]; then
|
|
[ -f "$1" ] || die "binary not found: $1"
|
|
install -m 0755 "$1" "$BIN_DEST"
|
|
echo "==> installed $BIN_DEST from $1"
|
|
elif [ -n "$GPUK_RELEASE_BASE" ]; then
|
|
command -v curl >/dev/null || die "curl is required to download the binary"
|
|
_url="$GPUK_RELEASE_BASE/$(arch_asset)"
|
|
echo "==> downloading $_url"
|
|
# Not -s: the binary is tens of MB and a silent download reads as a hang.
|
|
curl -fL --progress-bar "$_url" -o "$BIN_DEST.new" \
|
|
|| { rm -f "$BIN_DEST.new"; die "cannot download $_url"; }
|
|
# The root daemon is swapped in only once its minisign signature verifies
|
|
# against the release key pinned in this gpuk (OPS-20, INS-05).
|
|
if [ "${GPUK_CHANNEL_INSECURE:-}" = "1" ]; then
|
|
echo "WARNING: GPUK_CHANNEL_INSECURE=1 — $_url NOT verified" >&2
|
|
else
|
|
require_release_key
|
|
curl -fsSL "$_url.minisig" -o "$BIN_DEST.new.minisig" \
|
|
|| { rm -f "$BIN_DEST.new" "$BIN_DEST.new.minisig"; die "no signature at $_url.minisig — refusing an unsigned daemon binary"; }
|
|
"$MINISIGN" -Vq -m "$BIN_DEST.new" -x "$BIN_DEST.new.minisig" -P "$CHANNEL_PUBKEY" >/dev/null 2>&1 \
|
|
|| { rm -f "$BIN_DEST.new" "$BIN_DEST.new.minisig"; die "$_url: signature verification FAILED — refusing it (OPS-20)"; }
|
|
rm -f "$BIN_DEST.new.minisig"
|
|
echo "==> signature verified"
|
|
fi
|
|
chmod 0755 "$BIN_DEST.new"
|
|
mv "$BIN_DEST.new" "$BIN_DEST"
|
|
elif [ -x "$BIN_DEST" ]; then
|
|
echo "==> reusing existing $BIN_DEST"
|
|
else
|
|
die "no binary: pass --binary <path> or set GPUK_RELEASE_BASE=<url base>"
|
|
fi
|
|
}
|
|
|
|
# Leave this very gpuk on the host, next to the daemon: `gpuk status`, `gpuk
|
|
# update` and install.sh's plain re-run all use it. It is the copy that carries
|
|
# the release key this install pinned (substituted at publish time), so later
|
|
# updates are verified against the key the first install trusted.
|
|
install_cli() {
|
|
# Already this copy (gpuk re-run from its installed path, or identical bytes).
|
|
if [ -e "$CLI_DEST" ] && cmp -s "$0" "$CLI_DEST"; then return 0; fi
|
|
install -m 0755 "$0" "$CLI_DEST.new" && mv "$CLI_DEST.new" "$CLI_DEST" \
|
|
|| die "cannot install the gpuk CLI at $CLI_DEST"
|
|
echo "==> installed $CLI_DEST"
|
|
}
|
|
|
|
gen_secret() { head -c 32 /dev/urandom | base64 | tr '+/' '-_' | tr -d '='; }
|
|
|
|
# The host ports a leftover app container is reached on ("ui mtls inference"),
|
|
# read off the container itself with the precedence the controller applies
|
|
# (core/published-ports.ts): host networking moves the listeners (GPUK_PORT,
|
|
# GPUK_MTLS_PORT, GPUK_LISTEN_ADDR); anything else keeps the image's fixed
|
|
# listeners and publishes them (GPUK_PUBLIC_*, then the port bindings). Empty
|
|
# when there is no such container or no docker.
|
|
leftover_container_ports() { # $1 = container name
|
|
command -v docker >/dev/null 2>&1 || return 0
|
|
# Same format string as install.sh's container_facts — one reading, two scripts.
|
|
_f=$(docker inspect -f '{{.State.Status}}|{{.HostConfig.NetworkMode}}|{{range $p, $b := .HostConfig.PortBindings}}{{range $b}}{{$p}}={{.HostPort}} {{end}}{{end}}{{"\n"}}{{range .Config.Env}}{{.}}{{"\n"}}{{end}}' "$1" 2>/dev/null) \
|
|
|| return 0
|
|
[ -n "$_f" ] || return 0
|
|
_head=$(printf '%s\n' "$_f" | head -1)
|
|
_r=${_head#*|}; _net=${_r%%|*}; _bind=${_r#*|}
|
|
_env() { printf '%s\n' "$_f" | sed -n "s/^$1=//p" | head -1; }
|
|
_pub() { printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n "s/^$1\/tcp=//p" | head -1; }
|
|
if [ "$_net" = "host" ]; then
|
|
_ui=$(_env GPUK_PORT); _mtls=$(_env GPUK_MTLS_PORT)
|
|
_inf=$(_env GPUK_PROXY_PUBLIC_PORT)
|
|
[ -n "$_inf" ] || { _la=$(_env GPUK_LISTEN_ADDR); _inf=${_la##*:}; }
|
|
else
|
|
_ui=$(_env GPUK_PUBLIC_PORT); [ -n "$_ui" ] || _ui=$(_pub 8080)
|
|
_mtls=$(_env GPUK_PUBLIC_MTLS_PORT); [ -n "$_mtls" ] || _mtls=$(_pub 8443)
|
|
_inf=$(_env GPUK_PROXY_PUBLIC_PORT); [ -n "$_inf" ] || _inf=$(_pub 8200)
|
|
fi
|
|
printf '%s %s %s' "${_ui:-8080}" "${_mtls:-8443}" "${_inf:-8200}"
|
|
}
|
|
|
|
# Who holds a port: "process", "process in container NAME", or "". The process's
|
|
# cgroup names its container (host networking); a bridged publication is found
|
|
# as the container's port mapping in `docker ps` (the host-side holder is
|
|
# docker-proxy, which says nothing by itself). Same reading as install.sh.
|
|
port_owner() { # $1 = ss|netstat, $2 = port
|
|
_proc=""; _pid=""
|
|
case "$1" in
|
|
ss)
|
|
_line=$(ss -ltnpH "sport = :$2" 2>/dev/null | head -1)
|
|
_proc=$(printf '%s' "$_line" | sed -n 's/.*users:((\"\([^"]*\)\".*/\1/p')
|
|
_pid=$(printf '%s' "$_line" | sed -n 's/.*pid=\([0-9]*\).*/\1/p')
|
|
;;
|
|
netstat)
|
|
_field=$(netstat -ltnp 2>/dev/null \
|
|
| awk -v p="$2" '{n=split($4,a,":"); if (a[n]==p) {print $NF; exit}}')
|
|
case "$_field" in
|
|
*/*) _pid=${_field%%/*}; _proc=${_field#*/} ;;
|
|
*) _proc="$_field" ;;
|
|
esac
|
|
;;
|
|
esac
|
|
case "$_pid" in *[!0-9]*|"") _pid="" ;; esac
|
|
_ctr=""
|
|
if command -v docker >/dev/null 2>&1; then
|
|
if [ -n "$_pid" ] && [ -r "/proc/$_pid/cgroup" ]; then
|
|
_cid=$(sed -n 's#.*docker[-/]\([0-9a-f]\{64\}\).*#\1#p' "/proc/$_pid/cgroup" 2>/dev/null | head -1)
|
|
[ -z "$_cid" ] || _ctr=$(docker inspect -f '{{.Name}}' "$_cid" 2>/dev/null | sed 's|^/||')
|
|
fi
|
|
[ -n "$_ctr" ] || _ctr=$(docker ps --format '{{.Names}} {{.Ports}}' 2>/dev/null \
|
|
| awk -v p=":$2->" 'index($0, p) { print $1; exit }')
|
|
fi
|
|
if [ -n "$_ctr" ]; then
|
|
printf '%s' "${_proc:-a process} in container $_ctr"
|
|
else
|
|
printf '%s' "$_proc"
|
|
fi
|
|
}
|
|
|
|
# This host's LAN IPv4 — what a browser, a worker or a container on the docker
|
|
# bridge dials to reach the machine. The route to a public address names the
|
|
# source the default route uses (never a docker or libvirt bridge); failing that,
|
|
# the first global address on a physical-looking interface; failing that, the
|
|
# resolver's word. Empty when nothing answers — the caller says so. Same
|
|
# derivation as dev.sh best_host_lan_ipv4 and install.sh's banner.
|
|
host_lan_ipv4() {
|
|
_ip=$(ip route get 1.1.1.1 2>/dev/null | sed -n 's/.* src \([0-9.]*\).*/\1/p' | head -1)
|
|
if [ -z "$_ip" ]; then
|
|
_ip=$(ip -o -4 addr show scope global 2>/dev/null \
|
|
| awk '$2 !~ /^(virbr|docker|br-|veth|mpqemubr|tun|tap|vnet)/ {sub(/\/.*/, "", $4); print $4; exit}')
|
|
fi
|
|
[ -n "$_ip" ] || _ip=$(hostname -I 2>/dev/null | awk '{print $1}')
|
|
printf '%s' "$_ip"
|
|
}
|
|
|
|
seed_secrets() { # DATA_ROOT
|
|
_sd="$1/secrets"
|
|
mkdir -p "$_sd"; chmod 0700 "$_sd"
|
|
[ -f "$_sd/encryption_key" ] || { umask 077; gen_secret > "$_sd/encryption_key"; }
|
|
[ -f "$_sd/node_id" ] || { umask 077; (cat /proc/sys/kernel/random/uuid 2>/dev/null || uuidgen) > "$_sd/node_id"; }
|
|
# No nominal first-run password is generated. An operator may pre-provision
|
|
# this documented break-glass file; only then is it injected into the app.
|
|
[ ! -f "$_sd/bootstrap_admin_password" ] || chmod 0600 "$_sd/bootstrap_admin_password"
|
|
}
|
|
|
|
prepare_claim_code() {
|
|
# The file is the operator's recoverable proof of machine possession. Create
|
|
# it once, preserve it across reinstalls, and let the backend unlink it after
|
|
# the atomic first-account claim. A missing file beside an existing manifest
|
|
# therefore means "consumed", never "rotate the credential". homelab and studio
|
|
# have no claim window at all (PRF-12, SEC-53): no code is made for them.
|
|
case "$PROFILE" in homelab|studio) return 0 ;; esac
|
|
if [ -f "$CLAIM_CODE_FILE" ]; then
|
|
chmod 0600 "$CLAIM_CODE_FILE"
|
|
elif [ ! -f "$MANIFEST" ]; then
|
|
umask 077
|
|
gen_secret > "$CLAIM_CODE_FILE"
|
|
chmod 0600 "$CLAIM_CODE_FILE"
|
|
fi
|
|
}
|
|
|
|
# Write the systemd unit generated from the worker's canonical template. The
|
|
# generator injects the Rust lock-contention exit code here too, so the binary
|
|
# and systemd restart policy cannot silently drift apart.
|
|
write_unit() {
|
|
# BEGIN GENERATED WORKER SYSTEMD UNIT
|
|
cat > "$UNIT_DEST" <<UNIT
|
|
[Unit]
|
|
Description=GPU Kitchen worker daemon (gpu-kitchen-worker)
|
|
Documentation=https://repo.byterain.io/Sebastien/GPU-Manager
|
|
After=network-online.target docker.service
|
|
Wants=network-online.target docker.service
|
|
|
|
[Service]
|
|
Type=simple
|
|
# The worker runs as root on the host (NVML, docker.sock, mounts) — NOT in a
|
|
# container. Identity (keypair/cert/CA) lives under $IDENTITY_DIR; enroll once
|
|
# before starting this unit. Optional overrides live in $WORKER_ENV.
|
|
EnvironmentFile=-$WORKER_ENV
|
|
ExecStart=$BIN_DEST
|
|
# WRK-173: the old MainPID transfers supervision to the verified replacement
|
|
# before it exits. Notifications remain local to this service cgroup.
|
|
NotifyAccess=all
|
|
# Transient operation state only. The node lock has its own stable inode under
|
|
# /run/lock, outside this systemd-managed directory.
|
|
RuntimeDirectory=gpu-kitchen
|
|
RuntimeDirectoryMode=0750
|
|
Restart=always
|
|
RestartSec=2
|
|
# Lock contention is an operator error, not a crash: do not retry forever while
|
|
# another directly installed worker owns the WRK-167 host lock.
|
|
RestartPreventExitStatus=75
|
|
User=root
|
|
# Fail-closed renewal: on a refused cert renewal the daemon exits and systemd
|
|
# restarts it into an enroll-required state.
|
|
KillSignal=SIGTERM
|
|
TimeoutStopSec=15
|
|
|
|
[Install]
|
|
WantedBy=multi-user.target
|
|
UNIT
|
|
# END GENERATED WORKER SYSTEMD UNIT
|
|
}
|
|
|
|
# TLS reverse-proxy example, FILLED with the operator's domain (INS-47) — written
|
|
# under `--domain <d>`, whatever the profile. Kept aligned with
|
|
# deployments/controller/Caddyfile.example (the compose variant); this copy
|
|
# targets the all-in-one image, where nginx on the UI port is the single front
|
|
# door (INS-09) so one upstream carries pages, /api and the /ws upgrade alike.
|
|
# We write a file and NOTHING more: no package install, no service start, no
|
|
# other program's config read or touched — putting the proxy in service stays an
|
|
# operator act (OPS-13), and its presence stays unverifiable (OPS-67).
|
|
write_caddyfile() {
|
|
mkdir -p "$DATA_ROOT/caddy"
|
|
cat > "$DATA_ROOT/caddy/Caddyfile" <<CADDY
|
|
# TLS in front of GPU Kitchen — generated by the installer for --domain $DOMAIN
|
|
# (specs/plateforme/installation.md INS-47). Caddy provisions and renews the
|
|
# certificate itself once DNS for $DOMAIN points at this machine.
|
|
#
|
|
# sudo cp $DATA_ROOT/caddy/Caddyfile /etc/caddy/Caddyfile
|
|
# sudo systemctl reload caddy
|
|
#
|
|
# The app already knows $DOMAIN as one of its names (GPUK_ALLOWED_HOSTS) and reads
|
|
# the real client address from this proxy's X-Forwarded-For (GPUK_TRUST_PROXY) —
|
|
# without it every visitor would share Caddy's address, and one login throttle.
|
|
# Under --profile public it also runs with GPUK_HSTS=true and
|
|
# GPUK_SESSION_COOKIE_SECURE=true. What does NOT go through this proxy:
|
|
# - the worker mTLS channel (:$MTLS_PORT): workers pin the controller CA and must
|
|
# reach it DIRECTLY — terminating it here would break the pin.
|
|
# - worker<->worker data transfers (:8300): LAN-only by contract (OPS-68).
|
|
|
|
$DOMAIN {
|
|
encode zstd gzip
|
|
reverse_proxy localhost:$HTTP_PORT
|
|
}
|
|
|
|
# OpenAI-compatible inference endpoint (gpuk-proxy) — uncomment when inference
|
|
# clients live beyond the trusted LAN; TLS keeps their API keys off the wire.
|
|
#
|
|
# inference.$DOMAIN {
|
|
# encode zstd gzip
|
|
# reverse_proxy localhost:$INFERENCE_PORT
|
|
# }
|
|
CADDY
|
|
chmod 0644 "$DATA_ROOT/caddy/Caddyfile"
|
|
echo "==> wrote $DATA_ROOT/caddy/Caddyfile (filled TLS reverse-proxy example for $DOMAIN)"
|
|
}
|
|
|
|
# Pre-existing model caches (B81): a server that installs GPU Kitchen usually already
|
|
# holds tens or hundreds of GB of weights. We print a hint and nothing more — cache
|
|
# questions belong to the first-run wizard (INS-48, REG-41), never to the CLI; no
|
|
# config of any other program is read or touched, and referencing a cache stays an
|
|
# explicit choice made in the UI.
|
|
hint_existing_caches() {
|
|
hec_chosen=$(readlink -f "$1" 2>/dev/null || echo "$1")
|
|
hec_found=""
|
|
for hec_dir in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do
|
|
[ -n "$hec_dir" ] || continue
|
|
[ -d "$hec_dir/hub" ] || continue
|
|
hec_real=$(readlink -f "$hec_dir" 2>/dev/null || echo "$hec_dir")
|
|
[ "$hec_real" != "$hec_chosen" ] || continue
|
|
# Hub layout only (models--*) — matches what the scan can actually reference.
|
|
ls -d "$hec_dir"/hub/models--* >/dev/null 2>&1 || continue
|
|
hec_found="$hec_found $hec_real"
|
|
done
|
|
[ -n "$hec_found" ] || return 0
|
|
echo
|
|
for hec_dir in $hec_found; do
|
|
echo "==> existing model cache found at $hec_dir"
|
|
done
|
|
echo " GPU Kitchen will offer to reuse those models at first launch, and any"
|
|
echo " time from Nodes & GPU -> Storage. A reused cache is referenced in"
|
|
echo " place. Nothing is moved or deleted."
|
|
}
|
|
|
|
cmd_install() {
|
|
IMAGE=""; MODE="worker"; CLUSTER="default"; DATA_ROOT="/var/lib/gpu-kitchen"
|
|
# bridge, not host (INS-49): on the host network every listener the image opens
|
|
# is a host-wide claim, and a box that already runs something on 3000 or 8000
|
|
# killed Nitro. Bridged, only the three published ports touch the host; what
|
|
# that costs — the host daemon must be told its LAN address and the engines'
|
|
# docker network — is written to worker.env below. `--network host` remains.
|
|
CACHE_DIR=""; NETWORK="bridge"; CONTROLLER_URL=""; BIN_SRC=""; HEALTH_PORT="8001"
|
|
# 1337, not 8080: kept in lockstep with install.sh's default (INS-01). The
|
|
# worker channel and the inference endpoint move the same way (INS-46).
|
|
ENROLL_TOKEN=""; HTTP_PORT="1337"; MTLS_PORT="8443"; INFERENCE_PORT="8200"
|
|
# Worker data plane (OPS-68): the daemon reads GPUK_DATA_PORT,
|
|
# GPUK_MODEL_TRANSFER_PORT and GPUK_HANDOVER_DATA_PORT from worker.env and
|
|
# announces the first two to the controller, so moving them here is complete.
|
|
DATA_PORT="8300"; TRANSFER_PORT="8301"; HANDOVER_PORT="8302"
|
|
PROFILE=""; DOMAIN=""; DRY_RUN=0; RESET_MANIFEST=0
|
|
while [ $# -gt 0 ]; do
|
|
case "$1" in
|
|
--image) IMAGE="$2"; shift 2 ;;
|
|
--reset-manifest) RESET_MANIFEST=1; shift ;;
|
|
--mode) MODE="$2"; shift 2 ;;
|
|
--cluster) CLUSTER="$2"; shift 2 ;;
|
|
--profile) PROFILE="$2"; shift 2 ;;
|
|
--domain) DOMAIN="$2"; shift 2 ;;
|
|
--data-root) DATA_ROOT="$2"; shift 2 ;;
|
|
--cache-dir) CACHE_DIR="$2"; shift 2 ;;
|
|
--network) NETWORK="$2"; shift 2 ;;
|
|
--controller) CONTROLLER_URL="$2"; shift 2 ;;
|
|
# `--token` is an ENROLLMENT token (single-use or shared), NOT a bearer:
|
|
# the worker↔controller channel is cert-only since D11. `--token` is kept as
|
|
# a spelling of `--enroll-token`.
|
|
--enroll-token|--token) ENROLL_TOKEN="$2"; shift 2 ;;
|
|
--health-port) HEALTH_PORT="$2"; shift 2 ;;
|
|
--http-port) HTTP_PORT="$2"; shift 2 ;;
|
|
--mtls-port) MTLS_PORT="$2"; shift 2 ;;
|
|
--inference-port) INFERENCE_PORT="$2"; shift 2 ;;
|
|
--data-port) DATA_PORT="$2"; shift 2 ;;
|
|
--transfer-port) TRANSFER_PORT="$2"; shift 2 ;;
|
|
--handover-port) HANDOVER_PORT="$2"; shift 2 ;;
|
|
--binary) BIN_SRC="$2"; shift 2 ;;
|
|
--dry-run) DRY_RUN=1; shift ;;
|
|
*) die "unknown install option: $1" ;;
|
|
esac
|
|
done
|
|
|
|
# Roles, as the backend names them (multi-server/config.ts): "controller" (full
|
|
# app + UI, accepts workers) and "worker" (headless compute node). Historical
|
|
# spellings still work.
|
|
case "$MODE" in
|
|
controller|server|standalone|manager) MODE="controller" ;;
|
|
worker|agent) MODE="worker" ;;
|
|
*) die "--mode must be controller or worker (got '$MODE')" ;;
|
|
esac
|
|
|
|
if [ "$MODE" = "worker" ] && [ -n "$PROFILE" ]; then
|
|
die "--profile applies only to --mode controller; workers do not have an installation profile"
|
|
fi
|
|
|
|
for _pv in "$HTTP_PORT" "$MTLS_PORT" "$INFERENCE_PORT" "$HEALTH_PORT" \
|
|
"$DATA_PORT" "$TRANSFER_PORT" "$HANDOVER_PORT"; do
|
|
case "$_pv" in
|
|
''|*[!0-9]*) die "not a port number: '$_pv'" ;;
|
|
esac
|
|
[ "$_pv" -ge 1 ] && [ "$_pv" -le 65535 ] || die "port out of range: $_pv"
|
|
done
|
|
if [ "$HTTP_PORT" = "$MTLS_PORT" ] || [ "$HTTP_PORT" = "$INFERENCE_PORT" ] || [ "$MTLS_PORT" = "$INFERENCE_PORT" ]; then
|
|
die "--http-port, --mtls-port and --inference-port must differ (got $HTTP_PORT, $MTLS_PORT, $INFERENCE_PORT)"
|
|
fi
|
|
# The worker's four listeners (health + data plane) must differ too; the
|
|
# handover port is the data port's temporary twin (WRK-173), never the same.
|
|
_seen=""
|
|
for _pv in "$HEALTH_PORT" "$DATA_PORT" "$TRANSFER_PORT" "$HANDOVER_PORT"; do
|
|
case " $_seen " in
|
|
*" $_pv "*) die "--health-port, --data-port, --transfer-port and --handover-port must differ (got $HEALTH_PORT, $DATA_PORT, $TRANSFER_PORT, $HANDOVER_PORT)" ;;
|
|
esac
|
|
_seen="$_seen $_pv"
|
|
done
|
|
case "$PROFILE" in
|
|
""|homelab|studio|enterprise|public) ;;
|
|
*) die "--profile must be homelab, studio, enterprise or public (got '$PROFILE')" ;;
|
|
esac
|
|
|
|
if [ -n "$DOMAIN" ]; then
|
|
case "$DOMAIN" in
|
|
*[!A-Za-z0-9.-]*) die "--domain must be a bare domain name (got '$DOMAIN')" ;;
|
|
esac
|
|
fi
|
|
|
|
if [ "$MODE" = "controller" ]; then
|
|
if [ -z "$IMAGE" ] && [ "$DRY_RUN" -eq 1 ]; then
|
|
IMAGE="example.invalid/gpukitchen-controller:v0.0.0-dry-run"
|
|
fi
|
|
[ -n "$IMAGE" ] || die "--image registry/gpukitchen-controller:<tag> is required for a controller"
|
|
case "$IMAGE" in
|
|
*@sha256:*) ;;
|
|
*:latest) die "refusing floating image tag '$IMAGE' — use an explicit release tag or digest" ;;
|
|
*)
|
|
_image_tag="${IMAGE##*:}"
|
|
case "$_image_tag" in
|
|
"$IMAGE"|*/*) die "controller image must carry an explicit tag or sha256 digest (got '$IMAGE')" ;;
|
|
esac
|
|
;;
|
|
esac
|
|
fi
|
|
[ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf"
|
|
|
|
CACHE_JSON="[{\"hostPath\":\"$(json_str "$CACHE_DIR")\",\"shared\":false}]"
|
|
SELF_ENROLL_FILE="$DATA_ROOT/self-enroll-token"
|
|
BOOTSTRAP_PASSWORD_FILE="$DATA_ROOT/secrets/bootstrap_admin_password"
|
|
CLAIM_CODE_FILE="$DATA_ROOT/secrets/claim_code"
|
|
CLAIM_CODE_AVAILABLE=0
|
|
|
|
if [ "$DRY_RUN" -eq 1 ]; then
|
|
case "$MODE:$PROFILE" in
|
|
controller:homelab|controller:studio) ;;
|
|
controller:*) CLAIM_CODE_AVAILABLE=1 ;;
|
|
esac
|
|
echo "==> dry run: no file, service or container was changed"
|
|
[ ! -f "$MANIFEST" ] \
|
|
|| echo "==> dry run: $MANIFEST exists — this render would be MERGED into it, controller-owned settings kept"
|
|
if [ "$MODE" = "worker" ]; then render_worker_manifest; else render_controller_manifest; fi
|
|
[ -z "$DOMAIN" ] || echo "==> dry run: would write $DATA_ROOT/caddy/Caddyfile for $DOMAIN"
|
|
return 0
|
|
fi
|
|
|
|
# ── Port conflicts (INS-46) — the mutator's own guard ────────────────────────
|
|
# install.sh's preflight already checks these, and on a terminal it offers an
|
|
# alternative port. gpuk is the actual mutator and contributors call it
|
|
# DIRECTLY, so it re-checks and refuses, non-interactively, naming the flag
|
|
# that moves the port. Same helpers as install.sh (both scripts ship standalone
|
|
# from the channel). A listener owned by an existing install is not a conflict:
|
|
# a manifest on disk means the re-run is the update path, and every checked
|
|
# port is then our own; without a manifest, a leftover app container still
|
|
# holds OUR ports — apply() renames and stops it before the new one starts.
|
|
if [ ! -f "$MANIFEST" ]; then
|
|
_port_tool="${GPUK_PORT_CHECK_TOOL:-}"
|
|
case "$_port_tool" in
|
|
""|ss|netstat) ;;
|
|
*) die "GPUK_PORT_CHECK_TOOL must be ss or netstat (got '$_port_tool')" ;;
|
|
esac
|
|
if [ -z "$_port_tool" ]; then
|
|
if command -v ss >/dev/null 2>&1; then _port_tool="ss"
|
|
elif command -v netstat >/dev/null 2>&1; then _port_tool="netstat"; fi
|
|
fi
|
|
_ours=""
|
|
if [ "$MODE" = "controller" ]; then
|
|
_ours=$(leftover_container_ports gpu-kitchen)
|
|
[ -z "$_ours" ] || echo "==> existing app container gpu-kitchen found (ports $_ours): the install replaces it"
|
|
fi
|
|
if [ -z "$_port_tool" ]; then
|
|
echo "==> warning: cannot check for port conflicts (no ss or netstat)"
|
|
else
|
|
if [ "$MODE" = "controller" ]; then
|
|
set -- "$HTTP_PORT:UI:--http-port" \
|
|
"$MTLS_PORT:worker channel:--mtls-port" \
|
|
"$INFERENCE_PORT:inference endpoint:--inference-port"
|
|
else
|
|
# Worker data-plane ports (OPS-68) plus the local health listener.
|
|
set -- "$HEALTH_PORT:worker health:--health-port" \
|
|
"$DATA_PORT:worker data plane:--data-port" \
|
|
"$TRANSFER_PORT:model transfers:--transfer-port" \
|
|
"$HANDOVER_PORT:handover data plane:--handover-port"
|
|
fi
|
|
for _spec in "$@"; do
|
|
_p=${_spec%%:*}; _rest=${_spec#*:}; _label=${_rest%%:*}; _flag=${_rest#*:}
|
|
case " $_ours " in *" $_p "*) continue ;; esac
|
|
_busy=1
|
|
case "$_port_tool" in
|
|
ss) [ -n "$(ss -ltnH "sport = :$_p" 2>/dev/null)" ] || _busy=0 ;;
|
|
netstat)
|
|
netstat -ltn 2>/dev/null \
|
|
| awk -v p="$_p" '{n=split($4,a,":"); if (a[n]==p) found=1} END{exit found?0:1}' \
|
|
|| _busy=0
|
|
;;
|
|
esac
|
|
[ "$_busy" -eq 0 ] && continue
|
|
_owner=$(port_owner "$_port_tool" "$_p")
|
|
_hint="Free it first, then run the install again."
|
|
[ -z "$_flag" ] || _hint="Pass $_flag <p> to choose another port, or free it first."
|
|
die "port $_p ($_label) is already in use by ${_owner:-an unknown process}. $_hint"
|
|
done
|
|
fi
|
|
fi
|
|
|
|
need_root
|
|
install_binary "$BIN_SRC"
|
|
install_cli
|
|
|
|
mkdir -p "$ETC_DIR" "$DATA_ROOT" "$CACHE_DIR"
|
|
seed_secrets "$DATA_ROOT"
|
|
if [ "$MODE" = "controller" ]; then
|
|
prepare_claim_code
|
|
[ ! -s "$CLAIM_CODE_FILE" ] || CLAIM_CODE_AVAILABLE=1
|
|
fi
|
|
|
|
umask 077
|
|
if [ "$MODE" = "worker" ]; then
|
|
write_worker_manifest
|
|
# Optional env overrides for the daemon (cluster grouping, display name, an
|
|
# explicit controller URL). The enrolled identity carries the controller URL
|
|
# + CA too — this is belt-and-braces / pre-enroll discovery grouping.
|
|
{
|
|
echo "GPUK_CLUSTER=$CLUSTER"
|
|
echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT"
|
|
echo "GPUK_DATA_PORT=$DATA_PORT"
|
|
echo "GPUK_MODEL_TRANSFER_PORT=$TRANSFER_PORT"
|
|
echo "GPUK_HANDOVER_DATA_PORT=$HANDOVER_PORT"
|
|
echo "GPUK_MACHINE_ID_FILE=$MACHINE_ID_FILE"
|
|
echo "NODE_DISPLAY_NAME=$(hostname)"
|
|
[ -n "$CONTROLLER_URL" ] && echo "GPUK_CONTROLLER_URLS=$CONTROLLER_URL"
|
|
} > "$WORKER_ENV"
|
|
chmod 0600 "$WORKER_ENV"
|
|
echo "==> wrote $MANIFEST (worker: no app container)"
|
|
else
|
|
write_controller_manifest
|
|
# The host daemon and app container share DATA_ROOT. Only controller-mode
|
|
# workerd gets this private bootstrap channel; remote workers stay tokenless.
|
|
#
|
|
# A bridged app container (INS-49) needs two more lines, read by the daemon
|
|
# from its OWN env — the daemon runs on the host, not in the container:
|
|
# - GPUK_DATA_HOST: the daemon dials the published mTLS port through docker
|
|
# NAT, so the controller sees it arrive from the bridge gateway (172.17.0.1)
|
|
# and would persist THAT as the node's address (WRK-93). The LAN address is
|
|
# what peers, the proxy and the UI must dial instead.
|
|
# - BACKEND_DOCKER_NETWORK: the engines the daemon launches join the app
|
|
# container's docker network, so the backend and gpuk-proxy reach them by
|
|
# container IP (apps/worker/src/modules/docker.rs, staging/engine.rs).
|
|
# Exactly what dev.sh hands the dev worker; host networking needs neither.
|
|
_data_host=""
|
|
if [ "$NETWORK" != "host" ]; then
|
|
_data_host="${GPUK_DATA_HOST:-$(host_lan_ipv4)}"
|
|
[ -n "$_data_host" ] || echo "==> warning: this host's LAN IPv4 could not be determined (no ip, no hostname -I)." \
|
|
"Set GPUK_DATA_HOST=<lan ip> in $WORKER_ENV, or the controller will address its own worker through the docker bridge."
|
|
fi
|
|
{
|
|
echo "GPUK_CLUSTER=$CLUSTER"
|
|
echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT"
|
|
echo "GPUK_MACHINE_ID_FILE=$MACHINE_ID_FILE"
|
|
echo "GPUK_CONTROLLER_URLS=wss://127.0.0.1:$MTLS_PORT"
|
|
echo "GPUK_SELF_ENROLL_FILE=$SELF_ENROLL_FILE"
|
|
echo "NODE_DISPLAY_NAME=$(hostname)"
|
|
if [ "$NETWORK" != "host" ]; then
|
|
[ -z "$_data_host" ] || echo "GPUK_DATA_HOST=$_data_host"
|
|
echo "BACKEND_DOCKER_NETWORK=$NETWORK"
|
|
fi
|
|
} > "$WORKER_ENV"
|
|
chmod 0600 "$WORKER_ENV"
|
|
echo "==> wrote $MANIFEST (controller: app container $IMAGE, network $NETWORK)"
|
|
fi
|
|
chmod 0600 "$MANIFEST"
|
|
[ -z "$DOMAIN" ] || write_caddyfile
|
|
|
|
write_unit
|
|
systemctl daemon-reload
|
|
echo "==> wrote $UNIT_DEST"
|
|
|
|
# ── Worker: token enrollment before start, or unattended LAN discovery ──
|
|
if [ "$MODE" = "worker" ]; then
|
|
if [ -n "$ENROLL_TOKEN" ]; then
|
|
[ -n "$CONTROLLER_URL" ] || die "--controller wss://<controller>:<port> is required to enroll"
|
|
echo "==> enrolling against $CONTROLLER_URL ..."
|
|
GPUK_IDENTITY_DIR="$IDENTITY_DIR" GPUK_MACHINE_ID_FILE="$MACHINE_ID_FILE" "$BIN_DEST" enroll \
|
|
--controller "$CONTROLLER_URL" --token "$ENROLL_TOKEN" \
|
|
|| die "enrollment failed (bad/expired token, or controller unreachable)"
|
|
else
|
|
if [ -n "$CONTROLLER_URL" ]; then
|
|
echo "==> no token given: the daemon will request admission from $CONTROLLER_URL and wait"
|
|
echo " for automatic admission or administrator approval."
|
|
else
|
|
echo "==> no token or controller given: the daemon will discover its cluster on the LAN and wait"
|
|
echo " for automatic admission or administrator approval."
|
|
fi
|
|
fi
|
|
start_daemon
|
|
echo "==> $SERVICE_NAME enabled and started"
|
|
echo
|
|
echo "Done. The worker daemon is running${ENROLL_TOKEN:+ and enrolled}."
|
|
echo " Status : gpuk status Logs: gpuk logs"
|
|
hint_existing_caches "$CACHE_DIR"
|
|
return 0
|
|
fi
|
|
|
|
# ── Controller: start the daemon, then bring up the app container ──
|
|
start_daemon
|
|
echo "==> $SERVICE_NAME enabled and started"
|
|
# The pull is the long part of a first install (a multi-GB image) and workerd
|
|
# runs docker with its output captured, so pulling from inside `apply` is
|
|
# minutes of silence that read as a hang. Pull here, on the terminal, where
|
|
# docker's own per-layer progress is what the operator sees; `apply` then finds
|
|
# the image present and skips its pull (INS-03).
|
|
# Feedback only, never the verdict: `apply` pulls again whatever happened here
|
|
# and reports the cause itself (the CI shell gate installs a stub daemon against
|
|
# an image that does not exist anywhere — the daemon's pull is the one that counts).
|
|
if ! docker image inspect "$IMAGE" >/dev/null 2>&1; then
|
|
echo "==> pulling $IMAGE (docker shows the progress per layer) ..."
|
|
docker pull "$IMAGE" \
|
|
|| echo "==> the pull did not complete here; the daemon retries it during apply and reports the cause if it fails again"
|
|
fi
|
|
echo "==> applying manifest — starting the app container and waiting for its health check (up to 60s) ..."
|
|
# No unix socket any more: workerd reconciles the app container in-process from
|
|
# the on-disk manifest (there is no backend to relay through on the very first
|
|
# boot). Steady-state updates go through the daemon over WS.
|
|
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" apply \
|
|
|| die "the app container did not come up — the [worker] lines above say why (a [controller] FATAL line names the component and the fix). Full container logs: gpuk logs"
|
|
echo
|
|
echo "Done. The worker daemon is running and the app container is up."
|
|
echo " Status : gpuk status App logs: gpuk logs Daemon: journalctl -u $SERVICE_NAME"
|
|
hint_existing_caches "$CACHE_DIR"
|
|
}
|
|
|
|
# Worker manifest: cacheDisks and an EMPTY image so
|
|
# has_app_container() is false (apps/worker/src/modules/manifest.rs). No ports, no env,
|
|
# no secretsRef — a worker runs no app container.
|
|
render_worker_manifest() {
|
|
cat <<JSON
|
|
{
|
|
"schemaVersion": 1,
|
|
"image": "",
|
|
"mode": "worker",
|
|
"cluster": "$(json_str "$CLUSTER")",
|
|
"gpus": "all",
|
|
"dataRoot": "$(json_str "$DATA_ROOT")",
|
|
"cacheDisks": $CACHE_JSON
|
|
}
|
|
JSON
|
|
}
|
|
|
|
# Every env key gpuk may ever write into a manifest — conditional ones included.
|
|
# On a re-run the merge sets each of them to the fresh value or DELETES it when the
|
|
# fresh render no longer carries it (a profile change drops GPUK_HSTS); any other
|
|
# key was pushed by the controller and is kept. Keep this list in step with
|
|
# render_controller_manifest.
|
|
INSTALL_ENV_KEYS="NODE_ENV,GPUK_MODE,GPUK_CLUSTER,GPUK_SELF_ENROLL_FILE,NODE_DISPLAY_NAME,HF_HOME,GPUK_DATA_ROOT,GPUK_INSTALL_PROFILE,GPUK_HSTS,GPUK_SESSION_COOKIE_SECURE,GPUK_BOOTSTRAP_MUST_CHANGE,BACKEND_DOCKER_NETWORK,GPUK_PORT,GPUK_PUBLIC_PORT,GPUK_MTLS_PORT,GPUK_PUBLIC_MTLS_PORT,GPUK_LISTEN_ADDR,GPUK_PROXY_PUBLIC_PORT"
|
|
|
|
# Write the manifest — or, when one exists, MERGE into it (INS-03). Re-running the
|
|
# installer is the update path, and the manifest is not ours alone: since the first
|
|
# install the controller has patched it (extra cache disks, shared origin, eviction,
|
|
# LED binary, ports moved from Settings -> Network…). Rewriting it from flags would
|
|
# silently undo all of that, so the fresh render goes through the daemon's own
|
|
# `install-manifest`, which only replaces what the installer owns (image, mode,
|
|
# cluster, network, data root, ports, secret references, the primary cache disk,
|
|
# the env keys above) and keeps the rest. The first write is the render as-is.
|
|
write_manifest() { # $1 = render function
|
|
if [ -f "$MANIFEST" ] && [ "$RESET_MANIFEST" -eq 1 ]; then
|
|
# The operator's explicit regeneration (WRK-191): the daemon archives the
|
|
# existing document — readable or not — writes the render as-is and NAMES what
|
|
# the archive carried that the render does not. The flags install.sh inherited
|
|
# from the old file (ports, profile, data root) are already in the render.
|
|
"$1" > "$MANIFEST.new"
|
|
chmod 0600 "$MANIFEST.new"
|
|
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" install-manifest \
|
|
--from "$MANIFEST.new" --reset >/dev/null \
|
|
|| { rm -f "$MANIFEST.new"; die "could not reset $MANIFEST — the [worker] line above names the cause"; }
|
|
rm -f "$MANIFEST.new"
|
|
echo "==> $MANIFEST rebuilt from this install's settings (--reset-manifest); the previous file is archived beside it"
|
|
elif [ -f "$MANIFEST" ]; then
|
|
"$1" > "$MANIFEST.new"
|
|
chmod 0600 "$MANIFEST.new"
|
|
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" install-manifest \
|
|
--from "$MANIFEST.new" --own-env "$INSTALL_ENV_KEYS" >/dev/null \
|
|
|| {
|
|
rm -f "$MANIFEST.new"
|
|
die "could not merge the new settings into $MANIFEST — the [worker] line above names the cause.
|
|
A manifest written by an older release can carry a field this release removed. Re-run the same
|
|
command with --reset-manifest: the file is archived beside itself as manifest.json.before-reset.<time>,
|
|
rebuilt from this install's settings, and every setting the archive carried that the rebuild does
|
|
not (extra cache disks, LED binary, eviction…) is listed so you can set it again from the UI."
|
|
}
|
|
rm -f "$MANIFEST.new"
|
|
echo "==> merged into the existing $MANIFEST (controller-owned settings kept)"
|
|
else
|
|
"$1" > "$MANIFEST"
|
|
fi
|
|
}
|
|
|
|
write_worker_manifest() { write_manifest render_worker_manifest; }
|
|
|
|
# Controller manifest: the declarative app-container description workerd applies.
|
|
render_controller_manifest() {
|
|
# NOTE there is deliberately no PORT here: PORT is the backend's own port, a
|
|
# loopback-only 8000 behind nginx in the all-in-one image. What the outside world
|
|
# dials is GPUK_PUBLIC_PORT (the published host port), which falls back to nginx's
|
|
# own GPUK_PORT when nothing republishes it.
|
|
EXTRA_ENV=""
|
|
EXTRA_ENV="$EXTRA_ENV,\"GPUK_DATA_ROOT\":\"$(json_str "$DATA_ROOT")\""
|
|
[ -z "$PROFILE" ] || EXTRA_ENV="$EXTRA_ENV,\"GPUK_INSTALL_PROFILE\":\"$(json_str "$PROFILE")\""
|
|
if [ "$PROFILE" = "public" ]; then
|
|
EXTRA_ENV="$EXTRA_ENV,\"GPUK_HSTS\":\"true\""
|
|
EXTRA_ENV="$EXTRA_ENV,\"GPUK_SESSION_COOKIE_SECURE\":\"true\""
|
|
fi
|
|
# SEC-16: the DNS-rebinding Host guard stays on during the anonymous first run even
|
|
# under `enforced`, so the operator's own name must be a known host of the install
|
|
# before the wizard can load through it — whatever the profile. SEC-15: that name
|
|
# is served through a reverse proxy, whose X-Forwarded-For carries the real client.
|
|
if [ -n "$DOMAIN" ]; then
|
|
EXTRA_ENV="$EXTRA_ENV,\"GPUK_ALLOWED_HOSTS\":\"$(json_str "$DOMAIN")\""
|
|
EXTRA_ENV="$EXTRA_ENV,\"GPUK_TRUST_PROXY\":\"true\""
|
|
fi
|
|
BOOTSTRAP_SECRET_JSON=""
|
|
if [ -s "$BOOTSTRAP_PASSWORD_FILE" ]; then
|
|
EXTRA_ENV="$EXTRA_ENV,\"GPUK_BOOTSTRAP_MUST_CHANGE\":\"1\""
|
|
BOOTSTRAP_SECRET_JSON=",\"GPUK_BOOTSTRAP_ADMIN_PASSWORD\":\"$(json_str "$BOOTSTRAP_PASSWORD_FILE")\""
|
|
fi
|
|
CLAIM_SECRET_JSON=""
|
|
if [ "$CLAIM_CODE_AVAILABLE" -eq 1 ]; then
|
|
CLAIM_SECRET_JSON=",\"GPUK_CLAIM_CODE\":\"$(json_str "$CLAIM_CODE_FILE")\""
|
|
fi
|
|
[ "$NETWORK" = "host" ] || EXTRA_ENV="$EXTRA_ENV,\"BACKEND_DOCKER_NETWORK\":\"$(json_str "$NETWORK")\""
|
|
|
|
# Published != bound (INS-43). On a bridged network the container keeps the image's
|
|
# FIXED listeners — nginx 8080, worker mTLS 8443, gpuk-proxy 8200 — and the port
|
|
# flags only move the HOST side of the publication; Settings -> Network moves it
|
|
# later by patching this same manifest, so the container port must never become a
|
|
# variable. With host networking nothing is published and the listeners themselves
|
|
# take the ports (core/published-ports.ts reads this back with the same rules; the
|
|
# proxy port is GPUK_LISTEN_ADDR for the proxy, GPUK_PROXY_PUBLIC_PORT for what the
|
|
# backend shows — core/cluster-settings.ts proxyPublicPort).
|
|
PORTS_JSON="{}"
|
|
if [ "$NETWORK" = "host" ]; then
|
|
EXTRA_ENV="$EXTRA_ENV,\"GPUK_PORT\":\"$(json_str "$HTTP_PORT")\""
|
|
EXTRA_ENV="$EXTRA_ENV,\"GPUK_MTLS_PORT\":\"$(json_str "$MTLS_PORT")\""
|
|
EXTRA_ENV="$EXTRA_ENV,\"GPUK_LISTEN_ADDR\":\"0.0.0.0:$(json_str "$INFERENCE_PORT")\""
|
|
else
|
|
EXTRA_ENV="$EXTRA_ENV,\"GPUK_PORT\":\"8080\""
|
|
EXTRA_ENV="$EXTRA_ENV,\"GPUK_PUBLIC_PORT\":\"$(json_str "$HTTP_PORT")\""
|
|
EXTRA_ENV="$EXTRA_ENV,\"GPUK_PUBLIC_MTLS_PORT\":\"$(json_str "$MTLS_PORT")\""
|
|
PORTS_JSON="{\"8080\":$HTTP_PORT,\"8200\":$INFERENCE_PORT,\"8443\":$MTLS_PORT}"
|
|
fi
|
|
EXTRA_ENV="$EXTRA_ENV,\"GPUK_PROXY_PUBLIC_PORT\":\"$(json_str "$INFERENCE_PORT")\""
|
|
|
|
cat <<JSON
|
|
{
|
|
"schemaVersion": 1,
|
|
"image": "$(json_str "$IMAGE")",
|
|
"containerName": "gpu-kitchen",
|
|
"mode": "controller",
|
|
"cluster": "$(json_str "$CLUSTER")",
|
|
"networkMode": "$(json_str "$NETWORK")",
|
|
"gpus": "all",
|
|
"restartPolicy": "unless-stopped",
|
|
"dataRoot": "$(json_str "$DATA_ROOT")",
|
|
"cacheDisks": $CACHE_JSON,
|
|
"ports": $PORTS_JSON,
|
|
"env": {
|
|
"NODE_ENV": "production",
|
|
"GPUK_MODE": "controller",
|
|
"GPUK_CLUSTER": "$(json_str "$CLUSTER")",
|
|
"GPUK_SELF_ENROLL_FILE": "$(json_str "$SELF_ENROLL_FILE")",
|
|
"NODE_DISPLAY_NAME": "$(json_str "$(hostname)")",
|
|
"HF_HOME": "$(json_str "$CACHE_DIR")"$EXTRA_ENV
|
|
},
|
|
"secretsRef": {
|
|
"ENCRYPTION_KEY": "$(json_str "$DATA_ROOT/secrets/encryption_key")",
|
|
"NODE_ID": "$(json_str "$DATA_ROOT/secrets/node_id")"$BOOTSTRAP_SECRET_JSON$CLAIM_SECRET_JSON
|
|
}
|
|
}
|
|
JSON
|
|
}
|
|
|
|
write_controller_manifest() { write_manifest render_controller_manifest; }
|
|
|
|
# ── control subcommands ────────────────────────────────────────────────────────
|
|
container_name() {
|
|
sed -n 's/.*"containerName"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1
|
|
}
|
|
|
|
manifest_data_root() {
|
|
sed -n 's/.*"dataRoot"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1
|
|
}
|
|
|
|
manifest_image() {
|
|
sed -n 's/.*"image"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1
|
|
}
|
|
|
|
health_port() {
|
|
sed -n 's/.*GPUK_WORKER_HEALTH_PORT=\([0-9]*\).*/\1/p' "$WORKER_ENV" 2>/dev/null | head -1
|
|
}
|
|
|
|
# Split an image reference into repository and tag.
|
|
#
|
|
# `${img%%:*}` cuts at the FIRST colon and is WRONG:
|
|
# `registry.internal:5000/gpuk/controller:v1.2.3` would yield repo
|
|
# `registry.internal`. The colon in a registry's host:port is not a tag separator.
|
|
# Rule: it is a tag only if the last colon comes after the last slash.
|
|
# (Same logic as apps/controller/api/src/core/release-channel.ts — see its unit tests.)
|
|
image_repo() {
|
|
# A digest suffix (…@sha256:…) never carries the repo; drop it, then apply
|
|
# the tag logic — `repo:tag@sha256:…` and `repo@sha256:…` both reduce right.
|
|
set -- "${1%@*}"
|
|
_t="${1##*:}"
|
|
case "$_t" in
|
|
"$1") printf '%s' "$1" ;; # no colon at all → untagged
|
|
*/*) printf '%s' "$1" ;; # the last colon is inside a path → host:port, untagged
|
|
*) printf '%s' "${1%:*}" ;;
|
|
esac
|
|
}
|
|
|
|
image_tag() {
|
|
# `repo:tag@sha256:…` keeps the human-readable tag next to the content pin —
|
|
# docker resolves by digest and ignores the tag. Strip the digest, then parse.
|
|
set -- "${1%@*}"
|
|
_t="${1##*:}"
|
|
case "$_t" in
|
|
"$1") return ;;
|
|
*/*) return ;;
|
|
*) printf '%s' "$_t" ;;
|
|
esac
|
|
}
|
|
|
|
# The release channel: one flat JSON document served next to the installer. The
|
|
# UI's update check reads the same one (apps/controller/api/src/core/release-channel.ts) — one
|
|
# source of truth. Interim default: the public Gitea channel repo — flips to
|
|
# https://gpu.kitchen/latest.json once the hub exists (specs/developpement/ci-cd.md).
|
|
CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen/channel/raw/branch/main/latest.json}"
|
|
|
|
# The release public key pinned in THIS copy of gpuk (OPS-20). The channel copy
|
|
# gets the real key substituted at publish time; the operator override
|
|
# (GPUK_UPDATE_PUBKEY) covers a self-hosted channel with its own keypair. The
|
|
# first install fetched gpuk itself over HTTPS from the channel — that moment is
|
|
# trust-on-first-use, like a worker's enrolment token pin; every later `update`
|
|
# is verified against the key pinned HERE, so whoever controls latest.json can
|
|
# no longer pick what an existing install runs.
|
|
CHANNEL_PUBKEY="${GPUK_UPDATE_PUBKEY:-RWQ7BKXJqGX2jdKXu1GxeSPVAN3JDRTefpImM/mFRjtFwq4E7mhnQtA7}"
|
|
|
|
# The pinned key and the minisign CLI every verification needs (channel
|
|
# document, downloaded daemon binary); fail-closed.
|
|
require_release_key() {
|
|
case "$CHANNEL_PUBKEY" in
|
|
# The unstamped placeholder, matched by its prefix only: the release stamps
|
|
# the key by substituting the whole placeholder wherever it appears, and a
|
|
# guard spelling it in full would become one refusing the very key it pinned.
|
|
""|__GPUK_UPDATE_*)
|
|
die "this gpuk carries no pinned release public key — set GPUK_UPDATE_PUBKEY (the minisign public-key line), or GPUK_CHANNEL_INSECURE=1 to skip verification" ;;
|
|
esac
|
|
ensure_minisign 0
|
|
}
|
|
|
|
# The minisign CLI is what verifies the release; a host without it gets it from
|
|
# its own package manager — detected, non-interactive, quiet — before anything
|
|
# else changes. No known manager, or an install that fails: stop, naming the
|
|
# manual command. GPUK_MINISIGN names another verifier binary (test seam).
|
|
MINISIGN="${GPUK_MINISIGN:-minisign}"
|
|
minisign_manual_command() {
|
|
if command -v apt-get >/dev/null 2>&1; then echo "apt-get install minisign"
|
|
elif command -v dnf >/dev/null 2>&1; then echo "dnf install minisign (EPEL on RHEL)"
|
|
elif command -v yum >/dev/null 2>&1; then echo "yum install minisign (EPEL)"
|
|
elif command -v zypper >/dev/null 2>&1; then echo "zypper install minisign"
|
|
elif command -v apk >/dev/null 2>&1; then echo "apk add minisign"
|
|
elif command -v pacman >/dev/null 2>&1; then echo "pacman -S minisign"
|
|
else echo "install minisign from https://jedisct1.github.io/minisign/"
|
|
fi
|
|
}
|
|
ensure_minisign() { # $1 = 1 when nothing may be installed (dry run)
|
|
command -v "$MINISIGN" >/dev/null 2>&1 && return 0
|
|
[ "${1:-0}" -eq 0 ] \
|
|
|| die "minisign is required to verify the release and a dry run installs nothing: $(minisign_manual_command), or set GPUK_CHANNEL_INSECURE=1"
|
|
_mlog=$(mktemp)
|
|
for _pm in apt-get dnf yum zypper apk pacman; do
|
|
command -v "$_pm" >/dev/null 2>&1 || continue
|
|
echo "==> minisign is missing — installing it with $_pm"
|
|
case "$_pm" in
|
|
apt-get) { DEBIAN_FRONTEND=noninteractive apt-get update -qq \
|
|
&& DEBIAN_FRONTEND=noninteractive apt-get install -y -qq minisign; } ;;
|
|
dnf) dnf install -y -q minisign ;;
|
|
yum) yum install -y -q minisign ;;
|
|
zypper) zypper --non-interactive --quiet install minisign ;;
|
|
apk) apk add --quiet minisign ;;
|
|
pacman) pacman -S --noconfirm --needed --quiet minisign ;;
|
|
esac >"$_mlog" 2>&1 || true
|
|
if command -v "$MINISIGN" >/dev/null 2>&1; then rm -f "$_mlog"; return 0; fi
|
|
done
|
|
tail -5 "$_mlog" >&2 2>/dev/null || true
|
|
rm -f "$_mlog"
|
|
die "minisign is required to verify the release and could not be installed automatically. Nothing was changed: run '$(minisign_manual_command)' as root, then re-run — or set GPUK_CHANNEL_INSECURE=1"
|
|
}
|
|
|
|
# Fetch latest.json AND its minisign signature, verify, and leave the verified
|
|
# document at $CHANNEL_DOC. Fail-closed: no signature, bad signature, no
|
|
# minisign CLI or no pinned key are all fatal — GPUK_CHANNEL_INSECURE=1 is the
|
|
# explicit, logged opt-out (a private mirror that does not sign).
|
|
CHANNEL_DOC=""
|
|
channel_fetch() {
|
|
command -v curl >/dev/null 2>&1 || return 1
|
|
CHANNEL_DOC=$(mktemp) || return 1
|
|
curl -fsSL --max-time 20 "$CHANNEL_URL" -o "$CHANNEL_DOC" 2>/dev/null || return 1
|
|
if [ "${GPUK_CHANNEL_INSECURE:-}" = "1" ]; then
|
|
echo "WARNING: GPUK_CHANNEL_INSECURE=1 — release channel signature NOT verified" >&2
|
|
return 0
|
|
fi
|
|
require_release_key
|
|
_sig=$(mktemp)
|
|
if ! curl -fsSL --max-time 20 "${CHANNEL_URL}.minisig" -o "$_sig" 2>/dev/null; then
|
|
rm -f "$_sig"
|
|
die "no signature at ${CHANNEL_URL}.minisig — refusing an unsigned channel document (OPS-20)"
|
|
fi
|
|
if ! "$MINISIGN" -Vq -m "$CHANNEL_DOC" -x "$_sig" -P "$CHANNEL_PUBKEY" >/dev/null 2>&1; then
|
|
rm -f "$_sig"
|
|
die "latest.json signature verification FAILED — refusing the channel document (OPS-20)"
|
|
fi
|
|
rm -f "$_sig"
|
|
}
|
|
|
|
channel_field() {
|
|
sed -n 's/.*"'"$2"'"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$1" | head -1
|
|
}
|
|
|
|
channel_version() {
|
|
channel_fetch || return 1
|
|
channel_field "$CHANNEL_DOC" version
|
|
}
|
|
|
|
# ── backup ─────────────────────────────────────────────────────────────────────
|
|
# OPS-10: the embedded Postgres is only backed up COLD — hot-copying pgdata with
|
|
# a file tool is forbidden (torn pages). Order matters: stop the DAEMON first
|
|
# (its reconciler would immediately restart a stopped app container), then the
|
|
# container, snapshot, and restarting the service re-applies the manifest.
|
|
# An install on an external DATABASE_URL is refused here: gpuk only owns the
|
|
# embedded pgdata — back the real database up with pg_dump/backup-compose.sh.
|
|
# The finished directory still has to be copied to encrypted off-host storage,
|
|
# next to the recovery set (OPS-04/OPS-05: ENCRYPTION_KEY above all).
|
|
cmd_backup() {
|
|
need_root
|
|
_img=$(manifest_image)
|
|
[ -n "$_img" ] || die "this is a worker node — no controller data to back up here"
|
|
if grep -q '"DATABASE_URL"' "$MANIFEST" 2>/dev/null; then
|
|
die "this install uses an external DATABASE_URL — back THAT database up (pg_dump, or the compose procedures in specs/plateforme/operations.md); gpuk backup only snapshots the embedded pgdata"
|
|
fi
|
|
_root=$(manifest_data_root)
|
|
[ -n "$_root" ] || die "no dataRoot in $MANIFEST"
|
|
[ -d "$_root/pgdata" ] || die "no embedded pgdata under $_root — nothing to snapshot"
|
|
command -v sha256sum >/dev/null 2>&1 || die "sha256sum is required"
|
|
|
|
_out="${1:-$_root/backups/$(date -u +%Y%m%dT%H%M%SZ)}"
|
|
[ -e "$_out" ] && die "refusing to overwrite existing $_out"
|
|
mkdir -p "$(dirname "$_out")"
|
|
_tmp="$_out.partial"
|
|
rm -rf "$_tmp"; mkdir -p "$_tmp"
|
|
|
|
_cn=$(container_name)
|
|
echo "==> Stopping $SERVICE_NAME (its reconciler would restart the container mid-snapshot)..."
|
|
systemctl stop "$SERVICE_NAME" || die "could not stop $SERVICE_NAME"
|
|
_restart_daemon() { systemctl start "$SERVICE_NAME" 2>/dev/null || true; }
|
|
trap _restart_daemon EXIT
|
|
if [ -n "$_cn" ]; then
|
|
echo "==> Stopping $_cn (cold snapshot — OPS-10)..."
|
|
docker stop "$_cn" >/dev/null 2>&1 || true
|
|
_state=$(docker inspect -f '{{.State.Status}}' "$_cn" 2>/dev/null || echo absent)
|
|
case "$_state" in
|
|
running) die "container $_cn is still running — refusing a hot snapshot" ;;
|
|
esac
|
|
fi
|
|
|
|
echo "==> Snapshotting $_root/pgdata..."
|
|
tar -C "$_root" -czf "$_tmp/pgdata.tar.gz" pgdata || die "snapshot failed"
|
|
cp "$MANIFEST" "$_tmp/host-manifest.json" 2>/dev/null || true
|
|
{
|
|
echo "{"
|
|
echo " \"created_utc\": \"$(date -u +%Y-%m-%dT%H:%M:%SZ)\","
|
|
echo " \"image\": \"$(json_str "$_img")\","
|
|
echo " \"data_root\": \"$(json_str "$_root")\","
|
|
echo " \"kind\": \"cold-pgdata-snapshot\""
|
|
echo "}"
|
|
} > "$_tmp/backup-manifest.json"
|
|
(cd "$_tmp" && sha256sum ./* > SHA256SUMS) || die "checksums failed"
|
|
chmod 0700 "$_tmp"
|
|
mv "$_tmp" "$_out"
|
|
|
|
echo "==> Restarting $SERVICE_NAME (re-applies the manifest, container included)..."
|
|
systemctl start "$SERVICE_NAME" || die "could not restart $SERVICE_NAME — start it manually"
|
|
trap - EXIT
|
|
echo "backup: $_out"
|
|
echo "Copy it to encrypted OFF-HOST storage together with the recovery set"
|
|
echo "(ENCRYPTION_KEY above all — without it the data is unrecoverable, OPS-04)."
|
|
echo "A physical pgdata restore requires the same Postgres major and a throwaway"
|
|
echo "host rehearsal first (OPS-10)."
|
|
}
|
|
|
|
# ── channel ────────────────────────────────────────────────────────────────────
|
|
# Diagnostic (no root): fetch + VERIFY the channel document, print what it
|
|
# offers. Exercises exactly the trust chain `update` relies on — the CI probes
|
|
# it with a throwaway keypair, an operator uses it to debug a mirror.
|
|
cmd_channel() {
|
|
channel_fetch || die "cannot fetch $CHANNEL_URL"
|
|
echo "channel : $CHANNEL_URL"
|
|
echo "version : $(channel_field "$CHANNEL_DOC" version)"
|
|
_d=$(channel_field "$CHANNEL_DOC" controllerImageDigest)
|
|
[ -n "$_d" ] && echo "digest : $_d"
|
|
_d=$(channel_field "$CHANNEL_DOC" controllerImageDigestEnterprise)
|
|
[ -n "$_d" ] && echo "digest ee : $_d"
|
|
if [ "${GPUK_CHANNEL_INSECURE:-}" = "1" ]; then
|
|
echo "signature : SKIPPED (GPUK_CHANNEL_INSECURE=1)"
|
|
else
|
|
echo "signature : verified"
|
|
fi
|
|
}
|
|
|
|
# ── status ─────────────────────────────────────────────────────────────────────
|
|
# No control socket any more. Status = the systemd unit state + the daemon's own
|
|
# /health endpoint + whether a worker has enrolled (identity present).
|
|
cmd_status() {
|
|
_active=$(systemctl is-active "$SERVICE_NAME" 2>/dev/null || true)
|
|
echo "service : $_active"
|
|
if [ -f "$IDENTITY_DIR/identity.json" ]; then
|
|
echo "enrolled : yes ($IDENTITY_DIR)"
|
|
else
|
|
echo "enrolled : no (daemon waits for LAN admission; token enrollment is also available)"
|
|
fi
|
|
_hp=$(health_port); [ -n "$_hp" ] || _hp=8001
|
|
if command -v curl >/dev/null 2>&1; then
|
|
_h=$(curl -fsS --max-time 3 "http://127.0.0.1:$_hp/health" 2>/dev/null || true)
|
|
[ -n "$_h" ] && echo "health : $_h" || echo "health : (no answer on :$_hp)"
|
|
fi
|
|
_img=$(manifest_image)
|
|
if [ -n "$_img" ]; then
|
|
echo "app image : $_img"
|
|
echo "app cont. : $(docker inspect -f '{{.State.Status}}' "$(container_name)" 2>/dev/null || echo 'not running')"
|
|
else
|
|
echo "role : worker (no app container)"
|
|
fi
|
|
}
|
|
|
|
# ── enroll ─────────────────────────────────────────────────────────────────────
|
|
cmd_enroll() {
|
|
need_root
|
|
_url=""; _tok=""
|
|
while [ $# -gt 0 ]; do
|
|
case "$1" in
|
|
--controller) _url="$2"; shift 2 ;;
|
|
--enroll-token|--token) _tok="$2"; shift 2 ;;
|
|
*) die "unknown enroll option: $1" ;;
|
|
esac
|
|
done
|
|
[ -n "$_url" ] || die "enroll needs --controller wss://<controller>:<port>"
|
|
[ -n "$_tok" ] || die "enroll needs --token gk_enroll_..."
|
|
GPUK_IDENTITY_DIR="$IDENTITY_DIR" GPUK_MACHINE_ID_FILE="$MACHINE_ID_FILE" \
|
|
"$BIN_DEST" enroll --controller "$_url" --token "$_tok"
|
|
systemctl restart "$SERVICE_NAME" 2>/dev/null || true
|
|
}
|
|
|
|
# ── apply ──────────────────────────────────────────────────────────────────────
|
|
# Controller: reconcile the app container from the manifest, in-process. A worker
|
|
# has no app container — `apply` there is a no-op with a clear message.
|
|
cmd_apply() {
|
|
need_root
|
|
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" apply
|
|
}
|
|
|
|
# ── update ─────────────────────────────────────────────────────────────────────
|
|
# Controller: decide WHICH image TAG to pin, write it into the manifest, then let
|
|
# workerd pull + recreate (health-gate + rollback are the daemon's — apply()).
|
|
# Worker: the signed-binary self-update is DRIVEN FROM THE CONTROLLER (its Update
|
|
# button → POST /api/nodes/:id/host/update → cmd:host_update → verified swap).
|
|
# There is no local unverified swap path.
|
|
# [INS-03] `gpuk update` moves the WHOLE host to the release the signed channel
|
|
# vouches for: the root daemon as well as the app container. Updating only the
|
|
# container leaves the co-located worker on the previous protocol, which the new
|
|
# controller refuses — and a refused worker cannot receive the UI's Update either.
|
|
# The binary is held to the same chain as install.sh's: minisign signature by the
|
|
# pinned key, trusted comment naming the build the channel designates for this
|
|
# architecture, never older than the daemon already installed. Sets DAEMON_UPDATED.
|
|
DAEMON_UPDATED=0
|
|
update_daemon() {
|
|
_base=$(channel_field "$CHANNEL_DOC" workerBase)
|
|
case "$(uname -m)" in
|
|
x86_64|amd64) _bkey="workerBuildIdX86_64" ;;
|
|
aarch64|arm64) _bkey="workerBuildIdAarch64" ;;
|
|
*) die "unsupported architecture: $(uname -m)" ;;
|
|
esac
|
|
_build=$(channel_field "$CHANNEL_DOC" "$_bkey")
|
|
if [ -z "$_base" ] || [ -z "$_build" ]; then
|
|
die "the signed release channel names no gpu-kitchen-worker build for $(uname -m) — refusing to leave the daemon on another release (OPS-20)"
|
|
fi
|
|
_installed=$("$BIN_DEST" --version 2>/dev/null | sed -n 's/^gpu-kitchen-worker //p' | head -1)
|
|
if [ "$_installed" = "$_build" ]; then
|
|
echo "==> worker daemon already on build $_build"
|
|
return 0
|
|
fi
|
|
case "$_installed" in
|
|
[0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9]-*)
|
|
# buildId = YYYYMMDDHHMMSS-<commit>: the timestamp orders builds.
|
|
[ "${_build%%-*}" -ge "${_installed%%-*}" ] \
|
|
|| die "the channel's gpu-kitchen-worker $_build is OLDER than the installed $_installed — refusing a downgrade" ;;
|
|
esac
|
|
_url="$_base/$(arch_asset)"
|
|
echo "==> gpu-kitchen-worker ${_installed:-(unknown build)} → $_build"
|
|
command -v curl >/dev/null || die "curl is required to download the daemon"
|
|
curl -fL --progress-bar "$_url" -o "$BIN_DEST.new" \
|
|
|| { rm -f "$BIN_DEST.new"; die "cannot download $_url"; }
|
|
if [ "${GPUK_CHANNEL_INSECURE:-}" = "1" ]; then
|
|
echo "WARNING: GPUK_CHANNEL_INSECURE=1 — $_url NOT verified" >&2
|
|
else
|
|
require_release_key
|
|
curl -fsSL "$_url.minisig" -o "$BIN_DEST.new.minisig" \
|
|
|| { rm -f "$BIN_DEST.new" "$BIN_DEST.new.minisig"; die "no signature at $_url.minisig — refusing an unsigned daemon binary"; }
|
|
_comment=$("$MINISIGN" -V -m "$BIN_DEST.new" -x "$BIN_DEST.new.minisig" -P "$CHANNEL_PUBKEY" 2>/dev/null \
|
|
| sed -n 's/^Trusted comment: //p' | head -1)
|
|
rm -f "$BIN_DEST.new.minisig"
|
|
[ "$_comment" = "gpu-kitchen-worker@$_build" ] \
|
|
|| { rm -f "$BIN_DEST.new"; die "$_url is signed for '${_comment:-nothing}', not gpu-kitchen-worker@$_build — refusing it (OPS-20)"; }
|
|
echo "==> signature verified (build $_build)"
|
|
fi
|
|
chmod 0755 "$BIN_DEST.new"
|
|
mv "$BIN_DEST.new" "$BIN_DEST"
|
|
DAEMON_UPDATED=1
|
|
}
|
|
|
|
# The running daemon still executes the replaced binary until it restarts.
|
|
restart_updated_daemon() {
|
|
[ "$DAEMON_UPDATED" -eq 1 ] || return 0
|
|
systemctl restart "$SERVICE_NAME" \
|
|
|| die "the daemon binary was updated but $SERVICE_NAME did not restart — run: systemctl restart $SERVICE_NAME"
|
|
echo "==> $SERVICE_NAME restarted on the new build"
|
|
}
|
|
|
|
cmd_update() {
|
|
need_root
|
|
_want=""; _check=0
|
|
while [ $# -gt 0 ]; do
|
|
case "$1" in
|
|
--version) _want="$2"; shift 2 ;;
|
|
--check) _check=1; shift ;;
|
|
*) die "unknown update option: $1" ;;
|
|
esac
|
|
done
|
|
|
|
_image=$(manifest_image)
|
|
if [ -z "$_image" ]; then
|
|
# A worker node: the daemon is the whole install. The controller UI's Update
|
|
# button drives the same swap, but only for a worker it still accepts.
|
|
[ "$_check" -eq 0 ] || { echo "This is a worker node: 'gpuk update' moves its daemon to the channel's build."; return 0; }
|
|
channel_fetch || die "cannot read the release channel at $CHANNEL_URL"
|
|
update_daemon
|
|
restart_updated_daemon
|
|
return 0
|
|
fi
|
|
|
|
_repo=$(image_repo "$_image")
|
|
_current=$(image_tag "$_image")
|
|
[ -n "$_current" ] || _current="(untagged)"
|
|
|
|
if [ "$_check" -eq 1 ]; then
|
|
_latest=$(channel_version) || true
|
|
echo "installed : $_current"
|
|
if [ -z "$_latest" ]; then
|
|
echo "available : unknown (cannot reach $CHANNEL_URL)"
|
|
exit 1
|
|
fi
|
|
echo "available : $_latest"
|
|
if [ "$_latest" = "$_current" ]; then echo "up to date."; else echo "run 'gpuk update' to move to $_latest"; fi
|
|
return 0
|
|
fi
|
|
|
|
_digest=""
|
|
# channel_fetch runs in THIS shell (not a $(…) subshell) so a signature
|
|
# failure is fatal here — fail-closed — and $CHANNEL_DOC survives. The
|
|
# verified signature closes the document half of OPS-20; the digest read
|
|
# from it pins CONTENT, closing the mutable-tag half. A named --version goes
|
|
# through the same document: it vouches for ONE release, the one it names.
|
|
_latest=""
|
|
if channel_fetch; then
|
|
_latest=$(channel_field "$CHANNEL_DOC" version)
|
|
fi
|
|
if [ -z "$_latest" ]; then
|
|
[ -z "$_want" ] \
|
|
|| die "cannot read the release channel at $CHANNEL_URL — refusing to pin $_want unverified"
|
|
echo "==> cannot reach $CHANNEL_URL — re-applying the pinned $_current"
|
|
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" apply
|
|
return 0
|
|
fi
|
|
if [ -n "$_want" ] && [ "$_want" != "$_latest" ]; then
|
|
[ "${GPUK_CHANNEL_INSECURE:-}" = "1" ] \
|
|
|| die "the signed channel vouches for $_latest only, not $_want — refusing an unpinned tag (OPS-20)"
|
|
else
|
|
_want="$_latest"
|
|
# Pick the digest matching the installed edition by image basename — the
|
|
# repo itself may be a mirror, the basename is the edition marker.
|
|
case "${_repo##*/}" in
|
|
*-ee) _digest=$(channel_field "$CHANNEL_DOC" controllerImageDigestEnterprise) ;;
|
|
*) _digest=$(channel_field "$CHANNEL_DOC" controllerImageDigest) ;;
|
|
esac
|
|
case "$_digest" in
|
|
sha256:*) ;;
|
|
*) [ "${GPUK_CHANNEL_INSECURE:-}" = "1" ] \
|
|
|| die "the signed release channel names no image digest for $_want — refusing to pin a mutable tag (OPS-20)" ;;
|
|
esac
|
|
fi
|
|
|
|
_new="$_repo:$_want"
|
|
[ -n "$_digest" ] && _new="$_repo:$_want@$_digest"
|
|
_installed=$(manifest_image)
|
|
if [ "$_new" != "$_installed" ]; then
|
|
echo "==> $_current → $_want${_digest:+ (pinned by digest)}"
|
|
# Pin the new reference into the manifest, then apply. sed edits the single
|
|
# "image" line in place (atomic tmp + move).
|
|
_tmp="$MANIFEST.new"
|
|
sed "s#\"image\"[[:space:]]*:[[:space:]]*\"[^\"]*\"#\"image\": \"$(json_str "$_new")\"#" "$MANIFEST" > "$_tmp" \
|
|
|| die "could not rewrite the image in $MANIFEST"
|
|
chmod 0600 "$_tmp"; mv "$_tmp" "$MANIFEST"
|
|
else
|
|
echo "==> already on $_current — re-pulling and recreating"
|
|
fi
|
|
# The daemon follows the release it is pinned to — only the channel's own one
|
|
# carries a designated, signed build.
|
|
[ "$_want" != "$_latest" ] || update_daemon
|
|
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" apply \
|
|
|| die "the update did not apply — the [worker] lines above say why. A manifest refused for a field
|
|
an older release wrote is repaired by re-running the same install command with --reset-manifest:
|
|
the file is archived beside itself and rebuilt from this install's settings."
|
|
restart_updated_daemon
|
|
}
|
|
|
|
# ── uninstall ──────────────────────────────────────────────────────────────────
|
|
# Plain: stop and remove the service, touch nothing else (a pause, reversible by
|
|
# `gpuk install`). --purge: everything the installer created goes — the app
|
|
# container (and its -old / -failed twins), $ETC_DIR with the manifest and the
|
|
# enrolled identity, the binary — EXCEPT the data root: database, models and the
|
|
# secrets (ENCRYPTION_KEY above all, OPS-04) are the operator's to delete, by
|
|
# hand, knowingly. After a purge the next install is a first install (INS-03);
|
|
# without it, install.sh sees the manifest and treats the re-run as an update.
|
|
cmd_uninstall() {
|
|
need_root
|
|
_purge=0
|
|
while [ $# -gt 0 ]; do
|
|
case "$1" in
|
|
--purge) _purge=1; shift ;;
|
|
*) die "unknown uninstall option: $1 (expected: --purge)" ;;
|
|
esac
|
|
done
|
|
_cn=$(container_name)
|
|
_root=$(manifest_data_root)
|
|
systemctl disable --now "$SERVICE_NAME" 2>/dev/null || true
|
|
rm -f "$UNIT_DEST"; systemctl daemon-reload 2>/dev/null || true
|
|
echo "==> removed the systemd service"
|
|
if [ "$_purge" -eq 0 ]; then
|
|
echo "Left in place: $BIN_DEST, $ETC_DIR (manifest, identity), the data root and any"
|
|
echo "app container — 'gpuk install' brings the service back on them."
|
|
echo "For a clean slate (everything but the data root): gpuk uninstall --purge"
|
|
return 0
|
|
fi
|
|
if [ -n "$_cn" ] && command -v docker >/dev/null 2>&1; then
|
|
for _c in "$_cn" "$_cn-old" "$_cn-failed"; do
|
|
docker rm -f "$_c" >/dev/null 2>&1 && echo "==> removed container $_c"
|
|
done
|
|
fi
|
|
for _d in "$ETC_DIR" "$IDENTITY_DIR"; do
|
|
case "$_d" in ""|/|/etc|/usr|/var) die "refusing to remove $_d" ;; esac
|
|
[ ! -e "$_d" ] || { rm -rf "$_d"; echo "==> removed $_d"; }
|
|
done
|
|
[ ! -e "$MACHINE_ID_FILE" ] || rm -f "$MACHINE_ID_FILE"
|
|
[ ! -e "$BIN_DEST" ] || { rm -f "$BIN_DEST"; echo "==> removed $BIN_DEST"; }
|
|
[ ! -e "$CLI_DEST" ] || { rm -f "$CLI_DEST"; echo "==> removed $CLI_DEST"; }
|
|
echo
|
|
echo "Kept: the data root${_root:+ $_root} — database, models and secrets (ENCRYPTION_KEY)."
|
|
echo "A new install over it reuses them. To delete it too, knowingly: rm -rf ${_root:-<data-root>}"
|
|
}
|
|
|
|
usage() {
|
|
cat <<EOF
|
|
gpuk — GPU Kitchen host daemon (gpu-kitchen-worker)
|
|
|
|
gpuk install --mode controller --image <ref> [--profile homelab|studio|enterprise|public]
|
|
[--domain D] [--cluster N] [--cache-dir P] [--data-root P]
|
|
[--http-port P] [--mtls-port P] [--inference-port P]
|
|
[--network bridge|host|<net>] [--binary <path>] [--reset-manifest]
|
|
--network: bridge by default — the container publishes its three
|
|
ports and nothing else it listens on touches the host; host makes
|
|
every listener a host-wide claim (a re-run keeps the mode installed)
|
|
--domain: the name the UI is reached by through a reverse proxy, any
|
|
profile — allowed as a host, trusted proxy on, filled Caddyfile written
|
|
gpuk install --mode worker --controller wss://<host>:<port> --enroll-token gk_enroll_...
|
|
[--cluster N] [--cache-dir P] [--binary <path>]
|
|
[--health-port P] [--data-port P] [--transfer-port P] [--handover-port P]
|
|
gpuk install ... --dry-run Validate inputs and print the manifest without changing the host
|
|
gpuk status Service state, enrollment, /health, app container status
|
|
gpuk enroll --controller wss://<host>:<port> --token gk_enroll_...
|
|
gpuk apply (controller) Reconcile the app container from the manifest
|
|
gpuk update (controller) Move to the current release (pull + recreate, rollback)
|
|
gpuk update --check (controller) Compare the installed version with the release
|
|
gpuk update --version <tag> (controller) Move to a specific release
|
|
gpuk channel Fetch + VERIFY the release channel and print what it offers
|
|
gpuk backup [DIR] (controller) Cold snapshot of the embedded pgdata (stop → tar → restart)
|
|
gpuk manifest Print the current manifest
|
|
gpuk logs Follow the app container logs (controller) or the daemon journal;
|
|
after a failed start, the kept <container>-failed log
|
|
gpuk uninstall Remove the systemd service, leave everything else in place
|
|
gpuk uninstall --purge Also remove the app container, /etc/gpu-kitchen, the binary and gpuk
|
|
(never the data root: database, models, secrets)
|
|
|
|
Most people never run this directly: the channel's install.sh installs it
|
|
(https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh).
|
|
EOF
|
|
}
|
|
|
|
# ── dispatch ────────────────────────────────────────────────────────────────────
|
|
cmd="${1:-help}"; if [ $# -gt 0 ]; then shift; fi
|
|
case "$cmd" in
|
|
install) cmd_install "$@" ;;
|
|
status) cmd_status ;;
|
|
enroll) cmd_enroll "$@" ;;
|
|
apply) cmd_apply ;;
|
|
update) cmd_update "$@" ;;
|
|
channel) cmd_channel ;;
|
|
backup) cmd_backup "${1:-}" ;;
|
|
manifest) cat "$MANIFEST" ;;
|
|
logs)
|
|
_img=$(manifest_image)
|
|
if [ -n "$_img" ]; then
|
|
_cn=$(container_name)
|
|
if docker inspect "$_cn" >/dev/null 2>&1; then
|
|
exec docker logs -f "$_cn"
|
|
elif docker inspect "$_cn-failed" >/dev/null 2>&1; then
|
|
# No live app container, but the last failed start was kept (INS-15):
|
|
# its whole log is the diagnosis, not the daemon journal.
|
|
echo "==> no running app container; showing the full log of the last failed start ($_cn-failed)" >&2
|
|
exec docker logs "$_cn-failed"
|
|
else
|
|
exec journalctl -u "$SERVICE_NAME" -f
|
|
fi
|
|
else
|
|
exec journalctl -u "$SERVICE_NAME" -f
|
|
fi
|
|
;;
|
|
uninstall) cmd_uninstall "$@" ;;
|
|
help|-h|--help) usage ;;
|
|
*) usage; exit 1 ;;
|
|
esac
|