Files
channel/gpuk
T

1111 lines
50 KiB
Bash
Raw Normal View History

2026-08-03 09:02:45 +00:00
#!/bin/sh
# gpuk — installer + control CLI for the GPU Kitchen host daemon (gpu-kitchen-worker).
#
# Since C1 a compute node runs NO backend. The single privileged host component is
# the Rust `gpu-kitchen-worker` daemon (workerd), running as root DIRECTLY on the
# host (not in a container). It owns NVML clock/power locks, whitelisted host
# browsing, the model-cache staging pipeline, the signed `gpu-kitchen-bench`
# runner, and — on a controller node — the app container's lifecycle (create,
# health-gate, roll back, pull+recreate to update) via its manifest. There is no
# separate host daemon and no unix control socket: workerd is driven over the
# wss+mTLS channel by the controller, and locally by this CLI.
#
# Two roles:
# worker a headless compute node. Installs the binary + systemd unit + a
2026-09-13 21:53:54 +00:00
# manifest (cacheDisks) with NO app container, NO
2026-08-03 09:02:45 +00:00
# Postgres, NO docker app image. It ENROLLS over mTLS with a
# enrollment token, or waits for LAN discovery admission, then runs.
# controller the full app + UI. Installs the binary + systemd unit + an
# app-container manifest (image, ports, env, secretsRef) that workerd
# applies (`gpu-kitchen-worker apply`, then cmd:host_apply in steady
# state).
#
# Install (root):
# # controller:
2026-08-03 09:19:52 +00:00
# curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/gpuk \
2026-08-03 09:02:45 +00:00
# | sudo sh -s -- install --mode controller \
2026-08-03 09:19:52 +00:00
# --image repo.byterain.io/gpukitchen/gpukitchen-controller:vX.Y.Z
2026-08-03 09:02:45 +00:00
# # worker (enroll against a controller with a single-use token from its UI):
# sudo ./deployments/install/gpuk install --mode worker \
# --controller wss://controller.lan:8443 --enroll-token gk_enroll_... \
# --cache-dir /mnt/models --binary apps/worker/target/release/gpu-kitchen-worker
#
# Control:
# gpuk status | apply | update | enroll | logs | manifest | uninstall
#
set -eu
# ── Paths ────────────────────────────────────────────────────────────────────
# Overridable, so the daemon can be driven against a prefix a normal user owns —
# the only way any of this is testable without handing a test suite root on the
# host. Unset (the real install) they are exactly the systemd defaults workerd uses
# (apps/worker/src/module.rs, apps/worker/src/identity.rs).
BIN_DEST="${GPUK_BIN_DEST:-/usr/local/bin/gpu-kitchen-worker}"
ETC_DIR="${GPUK_ETC_DIR:-/etc/gpu-kitchen}"
MANIFEST="$ETC_DIR/manifest.json"
IDENTITY_DIR="${GPUK_IDENTITY_DIR:-$ETC_DIR/identity}"
2026-09-13 21:53:54 +00:00
MACHINE_ID_FILE="${GPUK_MACHINE_ID_FILE:-$ETC_DIR/machine-id}"
2026-08-03 09:02:45 +00:00
WORKER_ENV="$ETC_DIR/worker.env"
SERVICE_NAME="gpu-kitchen-worker"
2026-09-13 21:53:54 +00:00
UNIT_DEST="${GPUK_UNIT_DEST:-/etc/systemd/system/$SERVICE_NAME.service}"
2026-08-03 09:02:45 +00:00
# Where to download the binary from when no --binary is given.
GPUK_RELEASE_BASE="${GPUK_RELEASE_BASE:-}"
die() { echo "gpuk: $*" >&2; exit 1; }
# Root, or able to do the job anyway. Fail with a clear message BEFORE touching
# /etc, /usr/local/bin or systemd. Against a user-owned prefix it simply works.
need_root() {
[ "$(id -u)" -eq 0 ] && return 0
[ -w "$ETC_DIR" ] && return 0
die "this command must run as root (use sudo)"
}
# ── JSON helpers (controlled inputs: paths + identifiers) ──────────────────────
json_str() { printf '%s' "$1" | sed 's/\\/\\\\/g; s/"/\\"/g'; }
json_array() { # args → ["a","b",...]
_out=""
for _p in "$@"; do
_e=$(json_str "$_p")
if [ -z "$_out" ]; then _out="\"$_e\""; else _out="$_out,\"$_e\""; fi
done
printf '[%s]' "$_out"
}
# ── install ────────────────────────────────────────────────────────────────────
arch_asset() {
case "$(uname -m)" in
x86_64|amd64) echo "gpu-kitchen-worker-x86_64" ;;
aarch64|arm64) echo "gpu-kitchen-worker-aarch64" ;;
*) die "unsupported architecture: $(uname -m)" ;;
esac
}
install_binary() { # [local-path]
if [ -n "${1:-}" ]; then
[ -f "$1" ] || die "binary not found: $1"
install -m 0755 "$1" "$BIN_DEST"
echo "==> installed $BIN_DEST from $1"
elif [ -n "$GPUK_RELEASE_BASE" ]; then
command -v curl >/dev/null || die "curl is required to download the binary"
_url="$GPUK_RELEASE_BASE/$(arch_asset)"
echo "==> downloading $_url"
curl -fsSL "$_url" -o "$BIN_DEST.new"
chmod 0755 "$BIN_DEST.new"
mv "$BIN_DEST.new" "$BIN_DEST"
elif [ -x "$BIN_DEST" ]; then
echo "==> reusing existing $BIN_DEST"
else
die "no binary: pass --binary <path> or set GPUK_RELEASE_BASE=<url base>"
fi
}
gen_secret() { head -c 32 /dev/urandom | base64 | tr '+/' '-_' | tr -d '='; }
2026-09-17 00:02:06 +00:00
# The host ports a leftover app container is reached on ("ui mtls inference"),
# read off the container itself with the precedence the controller applies
# (core/published-ports.ts): host networking moves the listeners (GPUK_PORT,
# GPUK_MTLS_PORT, GPUK_LISTEN_ADDR); anything else keeps the image's fixed
# listeners and publishes them (GPUK_PUBLIC_*, then the port bindings). Empty
# when there is no such container or no docker.
leftover_container_ports() { # $1 = container name
command -v docker >/dev/null 2>&1 || return 0
# Same format string as install.sh's container_facts — one reading, two scripts.
_f=$(docker inspect -f '{{.State.Status}}|{{.HostConfig.NetworkMode}}|{{range $p, $b := .HostConfig.PortBindings}}{{range $b}}{{$p}}={{.HostPort}} {{end}}{{end}}{{"\n"}}{{range .Config.Env}}{{.}}{{"\n"}}{{end}}' "$1" 2>/dev/null) \
|| return 0
[ -n "$_f" ] || return 0
_head=$(printf '%s\n' "$_f" | head -1)
_r=${_head#*|}; _net=${_r%%|*}; _bind=${_r#*|}
_env() { printf '%s\n' "$_f" | sed -n "s/^$1=//p" | head -1; }
_pub() { printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n "s/^$1\/tcp=//p" | head -1; }
if [ "$_net" = "host" ]; then
_ui=$(_env GPUK_PORT); _mtls=$(_env GPUK_MTLS_PORT)
_inf=$(_env GPUK_PROXY_PUBLIC_PORT)
[ -n "$_inf" ] || { _la=$(_env GPUK_LISTEN_ADDR); _inf=${_la##*:}; }
else
_ui=$(_env GPUK_PUBLIC_PORT); [ -n "$_ui" ] || _ui=$(_pub 8080)
_mtls=$(_env GPUK_PUBLIC_MTLS_PORT); [ -n "$_mtls" ] || _mtls=$(_pub 8443)
_inf=$(_env GPUK_PROXY_PUBLIC_PORT); [ -n "$_inf" ] || _inf=$(_pub 8200)
fi
printf '%s %s %s' "${_ui:-8080}" "${_mtls:-8443}" "${_inf:-8200}"
}
# Who holds a port: "process", "process in container NAME", or "". The process's
# cgroup names its container (host networking); a bridged publication is found
# as the container's port mapping in `docker ps` (the host-side holder is
# docker-proxy, which says nothing by itself). Same reading as install.sh.
port_owner() { # $1 = ss|netstat, $2 = port
_proc=""; _pid=""
case "$1" in
ss)
_line=$(ss -ltnpH "sport = :$2" 2>/dev/null | head -1)
_proc=$(printf '%s' "$_line" | sed -n 's/.*users:((\"\([^"]*\)\".*/\1/p')
_pid=$(printf '%s' "$_line" | sed -n 's/.*pid=\([0-9]*\).*/\1/p')
;;
netstat)
_field=$(netstat -ltnp 2>/dev/null \
| awk -v p="$2" '{n=split($4,a,":"); if (a[n]==p) {print $NF; exit}}')
case "$_field" in
*/*) _pid=${_field%%/*}; _proc=${_field#*/} ;;
*) _proc="$_field" ;;
esac
;;
esac
case "$_pid" in *[!0-9]*|"") _pid="" ;; esac
_ctr=""
if command -v docker >/dev/null 2>&1; then
if [ -n "$_pid" ] && [ -r "/proc/$_pid/cgroup" ]; then
_cid=$(sed -n 's#.*docker[-/]\([0-9a-f]\{64\}\).*#\1#p' "/proc/$_pid/cgroup" 2>/dev/null | head -1)
[ -z "$_cid" ] || _ctr=$(docker inspect -f '{{.Name}}' "$_cid" 2>/dev/null | sed 's|^/||')
fi
[ -n "$_ctr" ] || _ctr=$(docker ps --format '{{.Names}} {{.Ports}}' 2>/dev/null \
| awk -v p=":$2->" 'index($0, p) { print $1; exit }')
fi
if [ -n "$_ctr" ]; then
printf '%s' "${_proc:-a process} in container $_ctr"
else
printf '%s' "$_proc"
fi
}
2026-09-13 21:53:54 +00:00
seed_secrets() { # DATA_ROOT
2026-08-03 09:02:45 +00:00
_sd="$1/secrets"
mkdir -p "$_sd"; chmod 0700 "$_sd"
[ -f "$_sd/encryption_key" ] || { umask 077; gen_secret > "$_sd/encryption_key"; }
[ -f "$_sd/node_id" ] || { umask 077; (cat /proc/sys/kernel/random/uuid 2>/dev/null || uuidgen) > "$_sd/node_id"; }
2026-09-13 21:53:54 +00:00
# No nominal first-run password is generated. An operator may pre-provision
# this documented break-glass file; only then is it injected into the app.
[ ! -f "$_sd/bootstrap_admin_password" ] || chmod 0600 "$_sd/bootstrap_admin_password"
}
prepare_claim_code() {
# The file is the operator's recoverable proof of machine possession. Create
# it once, preserve it across reinstalls, and let the backend unlink it after
# the atomic first-account claim. A missing file beside an existing manifest
# therefore means "consumed", never "rotate the credential".
if [ -f "$CLAIM_CODE_FILE" ]; then
chmod 0600 "$CLAIM_CODE_FILE"
elif [ ! -f "$MANIFEST" ]; then
umask 077
gen_secret > "$CLAIM_CODE_FILE"
chmod 0600 "$CLAIM_CODE_FILE"
2026-08-03 09:02:45 +00:00
fi
}
2026-09-13 21:53:54 +00:00
# Write the systemd unit generated from the worker's canonical template. The
# generator injects the Rust lock-contention exit code here too, so the binary
# and systemd restart policy cannot silently drift apart.
2026-08-03 09:02:45 +00:00
write_unit() {
2026-09-13 21:53:54 +00:00
# BEGIN GENERATED WORKER SYSTEMD UNIT
cat > "$UNIT_DEST" <<UNIT
2026-08-03 09:02:45 +00:00
[Unit]
Description=GPU Kitchen worker daemon (gpu-kitchen-worker)
2026-09-13 21:53:54 +00:00
Documentation=https://repo.byterain.io/Sebastien/GPU-Manager
2026-08-03 09:02:45 +00:00
After=network-online.target docker.service
Wants=network-online.target docker.service
[Service]
Type=simple
2026-09-13 21:53:54 +00:00
# The worker runs as root on the host (NVML, docker.sock, mounts) — NOT in a
# container. Identity (keypair/cert/CA) lives under $IDENTITY_DIR; enroll once
# before starting this unit. Optional overrides live in $WORKER_ENV.
2026-08-03 09:02:45 +00:00
EnvironmentFile=-$WORKER_ENV
ExecStart=$BIN_DEST
2026-09-13 21:53:54 +00:00
# WRK-173: the old MainPID transfers supervision to the verified replacement
# before it exits. Notifications remain local to this service cgroup.
NotifyAccess=all
# Transient operation state only. The node lock has its own stable inode under
# /run/lock, outside this systemd-managed directory.
2026-08-03 09:02:45 +00:00
RuntimeDirectory=gpu-kitchen
RuntimeDirectoryMode=0750
Restart=always
RestartSec=2
2026-09-13 21:53:54 +00:00
# Lock contention is an operator error, not a crash: do not retry forever while
# another directly installed worker owns the WRK-167 host lock.
RestartPreventExitStatus=75
2026-08-03 09:02:45 +00:00
User=root
2026-09-13 21:53:54 +00:00
# Fail-closed renewal: on a refused cert renewal the daemon exits and systemd
# restarts it into an enroll-required state.
2026-08-03 09:02:45 +00:00
KillSignal=SIGTERM
TimeoutStopSec=15
[Install]
WantedBy=multi-user.target
UNIT
2026-09-13 21:53:54 +00:00
# END GENERATED WORKER SYSTEMD UNIT
}
# TLS reverse-proxy example, FILLED with the operator's domain (INS-47) — written
# only under `--profile public --domain <d>`. Kept aligned with
# deployments/controller/Caddyfile.example (the compose variant); this copy
# targets the all-in-one image, where nginx on the UI port is the single front
# door (INS-09) so one upstream carries pages, /api and the /ws upgrade alike.
# We write a file and NOTHING more: no package install, no service start, no
# other program's config read or touched — putting the proxy in service stays an
# operator act (OPS-13), and its presence stays unverifiable (OPS-67).
write_caddyfile() {
mkdir -p "$DATA_ROOT/caddy"
cat > "$DATA_ROOT/caddy/Caddyfile" <<CADDY
# TLS in front of GPU Kitchen — generated by the installer for --profile public
# (specs/plateforme/installation.md INS-47). Caddy provisions and renews the
# certificate itself once DNS for $DOMAIN points at this machine.
#
# sudo cp $DATA_ROOT/caddy/Caddyfile /etc/caddy/Caddyfile
# sudo systemctl reload caddy
#
# The app already runs with GPUK_HSTS=true and GPUK_SESSION_COOKIE_SECURE=true
# (set by the public profile). What does NOT go through this proxy:
2026-09-17 00:02:06 +00:00
# - the worker mTLS channel (:$MTLS_PORT): workers pin the controller CA and must
2026-09-13 21:53:54 +00:00
# reach it DIRECTLY — terminating it here would break the pin.
# - worker<->worker data transfers (:8300): LAN-only by contract (OPS-68).
$DOMAIN {
encode zstd gzip
reverse_proxy localhost:$HTTP_PORT
}
# OpenAI-compatible inference endpoint (gpuk-proxy) — uncomment when inference
# clients live beyond the trusted LAN; TLS keeps their API keys off the wire.
#
# inference.$DOMAIN {
# encode zstd gzip
2026-09-17 00:02:06 +00:00
# reverse_proxy localhost:$INFERENCE_PORT
2026-09-13 21:53:54 +00:00
# }
CADDY
chmod 0644 "$DATA_ROOT/caddy/Caddyfile"
echo "==> wrote $DATA_ROOT/caddy/Caddyfile (filled TLS reverse-proxy example for $DOMAIN)"
2026-08-03 09:02:45 +00:00
}
# Pre-existing model caches (B81): a server that installs GPU Kitchen usually already
2026-09-13 21:53:54 +00:00
# holds tens or hundreds of GB of weights. We print a hint and nothing more — cache
# questions belong to the first-run wizard (INS-48, REG-41), never to the CLI; no
# config of any other program is read or touched, and referencing a cache stays an
# explicit choice made in the UI.
2026-08-03 09:02:45 +00:00
hint_existing_caches() {
hec_chosen=$(readlink -f "$1" 2>/dev/null || echo "$1")
hec_found=""
for hec_dir in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do
[ -n "$hec_dir" ] || continue
[ -d "$hec_dir/hub" ] || continue
hec_real=$(readlink -f "$hec_dir" 2>/dev/null || echo "$hec_dir")
[ "$hec_real" != "$hec_chosen" ] || continue
# Hub layout only (models--*) — matches what the scan can actually reference.
ls -d "$hec_dir"/hub/models--* >/dev/null 2>&1 || continue
hec_found="$hec_found $hec_real"
done
[ -n "$hec_found" ] || return 0
echo
for hec_dir in $hec_found; do
echo "==> existing model cache found at $hec_dir"
done
2026-09-13 21:53:54 +00:00
echo " GPU Kitchen will offer to reuse those models at first launch, and any"
echo " time from Nodes & GPU -> Storage. A reused cache is referenced in"
echo " place. Nothing is moved or deleted."
2026-08-03 09:02:45 +00:00
}
cmd_install() {
IMAGE=""; MODE="worker"; CLUSTER="default"; DATA_ROOT="/var/lib/gpu-kitchen"
CACHE_DIR=""; NETWORK="host"; CONTROLLER_URL=""; BIN_SRC=""; HEALTH_PORT="8001"
2026-09-17 00:02:06 +00:00
# 1337, not 8080: kept in lockstep with install.sh's default (INS-01). The
# worker channel and the inference endpoint move the same way (INS-46).
ENROLL_TOKEN=""; HTTP_PORT="1337"; MTLS_PORT="8443"; INFERENCE_PORT="8200"
# Worker data plane (OPS-68): the daemon reads GPUK_DATA_PORT,
# GPUK_MODEL_TRANSFER_PORT and GPUK_HANDOVER_DATA_PORT from worker.env and
# announces the first two to the controller, so moving them here is complete.
DATA_PORT="8300"; TRANSFER_PORT="8301"; HANDOVER_PORT="8302"
PROFILE=""; DOMAIN=""; DRY_RUN=0
2026-08-03 09:02:45 +00:00
while [ $# -gt 0 ]; do
case "$1" in
--image) IMAGE="$2"; shift 2 ;;
--mode) MODE="$2"; shift 2 ;;
--cluster) CLUSTER="$2"; shift 2 ;;
2026-09-13 21:53:54 +00:00
--profile) PROFILE="$2"; shift 2 ;;
--domain) DOMAIN="$2"; shift 2 ;;
2026-08-03 09:02:45 +00:00
--data-root) DATA_ROOT="$2"; shift 2 ;;
--cache-dir) CACHE_DIR="$2"; shift 2 ;;
--network) NETWORK="$2"; shift 2 ;;
--controller) CONTROLLER_URL="$2"; shift 2 ;;
# `--token` is an ENROLLMENT token (single-use or shared), NOT a bearer:
# the worker↔controller channel is cert-only since D11. `--token` is kept as
# a spelling of `--enroll-token`.
--enroll-token|--token) ENROLL_TOKEN="$2"; shift 2 ;;
--health-port) HEALTH_PORT="$2"; shift 2 ;;
--http-port) HTTP_PORT="$2"; shift 2 ;;
2026-09-17 00:02:06 +00:00
--mtls-port) MTLS_PORT="$2"; shift 2 ;;
--inference-port) INFERENCE_PORT="$2"; shift 2 ;;
--data-port) DATA_PORT="$2"; shift 2 ;;
--transfer-port) TRANSFER_PORT="$2"; shift 2 ;;
--handover-port) HANDOVER_PORT="$2"; shift 2 ;;
2026-08-03 09:02:45 +00:00
--binary) BIN_SRC="$2"; shift 2 ;;
2026-09-13 21:53:54 +00:00
--dry-run) DRY_RUN=1; shift ;;
2026-08-03 09:02:45 +00:00
*) die "unknown install option: $1" ;;
esac
done
# Roles, as the backend names them (multi-server/config.ts): "controller" (full
# app + UI, accepts workers) and "worker" (headless compute node). Historical
# spellings still work.
case "$MODE" in
controller|server|standalone|manager) MODE="controller" ;;
worker|agent) MODE="worker" ;;
*) die "--mode must be controller or worker (got '$MODE')" ;;
esac
2026-09-13 21:53:54 +00:00
if [ "$MODE" = "worker" ] && [ -n "$PROFILE" ]; then
die "--profile applies only to --mode controller; workers do not have an installation profile"
fi
2026-09-17 00:02:06 +00:00
for _pv in "$HTTP_PORT" "$MTLS_PORT" "$INFERENCE_PORT" "$HEALTH_PORT" \
"$DATA_PORT" "$TRANSFER_PORT" "$HANDOVER_PORT"; do
case "$_pv" in
''|*[!0-9]*) die "not a port number: '$_pv'" ;;
esac
[ "$_pv" -ge 1 ] && [ "$_pv" -le 65535 ] || die "port out of range: $_pv"
done
if [ "$HTTP_PORT" = "$MTLS_PORT" ] || [ "$HTTP_PORT" = "$INFERENCE_PORT" ] || [ "$MTLS_PORT" = "$INFERENCE_PORT" ]; then
die "--http-port, --mtls-port and --inference-port must differ (got $HTTP_PORT, $MTLS_PORT, $INFERENCE_PORT)"
fi
# The worker's four listeners (health + data plane) must differ too; the
# handover port is the data port's temporary twin (WRK-173), never the same.
_seen=""
for _pv in "$HEALTH_PORT" "$DATA_PORT" "$TRANSFER_PORT" "$HANDOVER_PORT"; do
case " $_seen " in
*" $_pv "*) die "--health-port, --data-port, --transfer-port and --handover-port must differ (got $HEALTH_PORT, $DATA_PORT, $TRANSFER_PORT, $HANDOVER_PORT)" ;;
esac
_seen="$_seen $_pv"
done
2026-09-13 21:53:54 +00:00
case "$PROFILE" in
""|homelab|studio|enterprise|public) ;;
*) die "--profile must be homelab, studio, enterprise or public (got '$PROFILE')" ;;
esac
if [ -n "$DOMAIN" ]; then
[ "$PROFILE" = "public" ] \
|| die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)"
case "$DOMAIN" in
*[!A-Za-z0-9.-]*) die "--domain must be a bare domain name (got '$DOMAIN')" ;;
esac
fi
if [ "$MODE" = "controller" ]; then
if [ -z "$IMAGE" ] && [ "$DRY_RUN" -eq 1 ]; then
IMAGE="example.invalid/gpukitchen-controller:v0.0.0-dry-run"
fi
[ -n "$IMAGE" ] || die "--image registry/gpukitchen-controller:<tag> is required for a controller"
case "$IMAGE" in
*@sha256:*) ;;
*:latest) die "refusing floating image tag '$IMAGE' — use an explicit release tag or digest" ;;
*)
_image_tag="${IMAGE##*:}"
case "$_image_tag" in
"$IMAGE"|*/*) die "controller image must carry an explicit tag or sha256 digest (got '$IMAGE')" ;;
esac
;;
esac
fi
2026-08-03 09:02:45 +00:00
[ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf"
2026-09-13 21:53:54 +00:00
CACHE_JSON="[{\"hostPath\":\"$(json_str "$CACHE_DIR")\",\"shared\":false}]"
SELF_ENROLL_FILE="$DATA_ROOT/self-enroll-token"
BOOTSTRAP_PASSWORD_FILE="$DATA_ROOT/secrets/bootstrap_admin_password"
CLAIM_CODE_FILE="$DATA_ROOT/secrets/claim_code"
CLAIM_CODE_AVAILABLE=0
if [ "$DRY_RUN" -eq 1 ]; then
[ "$MODE" != "controller" ] || CLAIM_CODE_AVAILABLE=1
echo "==> dry run: no file, service or container was changed"
2026-09-17 00:02:06 +00:00
[ ! -f "$MANIFEST" ] \
|| echo "==> dry run: $MANIFEST exists — this render would be MERGED into it, controller-owned settings kept"
2026-09-13 21:53:54 +00:00
if [ "$MODE" = "worker" ]; then render_worker_manifest; else render_controller_manifest; fi
[ -z "$DOMAIN" ] || echo "==> dry run: would write $DATA_ROOT/caddy/Caddyfile for $DOMAIN"
return 0
fi
# ── Port conflicts (INS-46) — the mutator's own guard ────────────────────────
2026-09-17 00:02:06 +00:00
# install.sh's preflight already checks these, and on a terminal it offers an
# alternative port. gpuk is the actual mutator and contributors call it
# DIRECTLY, so it re-checks and refuses, non-interactively, naming the flag
# that moves the port. Same helpers as install.sh (both scripts ship standalone
# from the channel). A listener owned by an existing install is not a conflict:
# a manifest on disk means the re-run is the update path, and every checked
# port is then our own; without a manifest, a leftover app container still
# holds OUR ports — apply() renames and stops it before the new one starts.
2026-09-13 21:53:54 +00:00
if [ ! -f "$MANIFEST" ]; then
2026-09-17 00:02:06 +00:00
_port_tool="${GPUK_PORT_CHECK_TOOL:-}"
case "$_port_tool" in
""|ss|netstat) ;;
*) die "GPUK_PORT_CHECK_TOOL must be ss or netstat (got '$_port_tool')" ;;
esac
if [ -z "$_port_tool" ]; then
if command -v ss >/dev/null 2>&1; then _port_tool="ss"
elif command -v netstat >/dev/null 2>&1; then _port_tool="netstat"; fi
fi
_ours=""
if [ "$MODE" = "controller" ]; then
_ours=$(leftover_container_ports gpu-kitchen)
[ -z "$_ours" ] || echo "==> existing app container gpu-kitchen found (ports $_ours): the install replaces it"
fi
2026-09-13 21:53:54 +00:00
if [ -z "$_port_tool" ]; then
echo "==> warning: cannot check for port conflicts (no ss or netstat)"
else
if [ "$MODE" = "controller" ]; then
2026-09-17 00:02:06 +00:00
set -- "$HTTP_PORT:UI:--http-port" \
"$MTLS_PORT:worker channel:--mtls-port" \
"$INFERENCE_PORT:inference endpoint:--inference-port"
2026-09-13 21:53:54 +00:00
else
# Worker data-plane ports (OPS-68) plus the local health listener.
2026-09-17 00:02:06 +00:00
set -- "$HEALTH_PORT:worker health:--health-port" \
"$DATA_PORT:worker data plane:--data-port" \
"$TRANSFER_PORT:model transfers:--transfer-port" \
"$HANDOVER_PORT:handover data plane:--handover-port"
2026-09-13 21:53:54 +00:00
fi
2026-09-17 00:02:06 +00:00
for _spec in "$@"; do
_p=${_spec%%:*}; _rest=${_spec#*:}; _label=${_rest%%:*}; _flag=${_rest#*:}
case " $_ours " in *" $_p "*) continue ;; esac
2026-09-13 21:53:54 +00:00
_busy=1
case "$_port_tool" in
ss) [ -n "$(ss -ltnH "sport = :$_p" 2>/dev/null)" ] || _busy=0 ;;
netstat)
netstat -ltn 2>/dev/null \
| awk -v p="$_p" '{n=split($4,a,":"); if (a[n]==p) found=1} END{exit found?0:1}' \
|| _busy=0
;;
esac
2026-09-17 00:02:06 +00:00
[ "$_busy" -eq 0 ] && continue
_owner=$(port_owner "$_port_tool" "$_p")
_hint="Free it first, then run the install again."
[ -z "$_flag" ] || _hint="Pass $_flag <p> to choose another port, or free it first."
die "port $_p ($_label) is already in use by ${_owner:-an unknown process}. $_hint"
2026-09-13 21:53:54 +00:00
done
fi
fi
need_root
2026-08-03 09:02:45 +00:00
install_binary "$BIN_SRC"
mkdir -p "$ETC_DIR" "$DATA_ROOT" "$CACHE_DIR"
2026-09-13 21:53:54 +00:00
seed_secrets "$DATA_ROOT"
if [ "$MODE" = "controller" ]; then
prepare_claim_code
[ ! -s "$CLAIM_CODE_FILE" ] || CLAIM_CODE_AVAILABLE=1
fi
2026-08-03 09:02:45 +00:00
umask 077
if [ "$MODE" = "worker" ]; then
write_worker_manifest
# Optional env overrides for the daemon (cluster grouping, display name, an
# explicit controller URL). The enrolled identity carries the controller URL
# + CA too — this is belt-and-braces / pre-enroll discovery grouping.
{
echo "GPUK_CLUSTER=$CLUSTER"
echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT"
2026-09-17 00:02:06 +00:00
echo "GPUK_DATA_PORT=$DATA_PORT"
echo "GPUK_MODEL_TRANSFER_PORT=$TRANSFER_PORT"
echo "GPUK_HANDOVER_DATA_PORT=$HANDOVER_PORT"
2026-09-13 21:53:54 +00:00
echo "GPUK_MACHINE_ID_FILE=$MACHINE_ID_FILE"
2026-08-03 09:02:45 +00:00
echo "NODE_DISPLAY_NAME=$(hostname)"
[ -n "$CONTROLLER_URL" ] && echo "GPUK_CONTROLLER_URLS=$CONTROLLER_URL"
} > "$WORKER_ENV"
chmod 0600 "$WORKER_ENV"
echo "==> wrote $MANIFEST (worker: no app container)"
else
write_controller_manifest
2026-09-13 21:53:54 +00:00
# The host daemon and app container share DATA_ROOT. Only controller-mode
# workerd gets this private bootstrap channel; remote workers stay tokenless.
{
echo "GPUK_CLUSTER=$CLUSTER"
echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT"
echo "GPUK_MACHINE_ID_FILE=$MACHINE_ID_FILE"
2026-09-17 00:02:06 +00:00
echo "GPUK_CONTROLLER_URLS=wss://127.0.0.1:$MTLS_PORT"
2026-09-13 21:53:54 +00:00
echo "GPUK_SELF_ENROLL_FILE=$SELF_ENROLL_FILE"
echo "NODE_DISPLAY_NAME=$(hostname)"
} > "$WORKER_ENV"
chmod 0600 "$WORKER_ENV"
2026-08-03 09:02:45 +00:00
echo "==> wrote $MANIFEST (controller: app container $IMAGE)"
fi
chmod 0600 "$MANIFEST"
2026-09-13 21:53:54 +00:00
[ -z "$DOMAIN" ] || write_caddyfile
2026-08-03 09:02:45 +00:00
write_unit
systemctl daemon-reload
echo "==> wrote $UNIT_DEST"
# ── Worker: token enrollment before start, or unattended LAN discovery ──
if [ "$MODE" = "worker" ]; then
if [ -n "$ENROLL_TOKEN" ]; then
[ -n "$CONTROLLER_URL" ] || die "--controller wss://<controller>:<port> is required to enroll"
echo "==> enrolling against $CONTROLLER_URL ..."
2026-09-13 21:53:54 +00:00
GPUK_IDENTITY_DIR="$IDENTITY_DIR" GPUK_MACHINE_ID_FILE="$MACHINE_ID_FILE" "$BIN_DEST" enroll \
2026-08-03 09:02:45 +00:00
--controller "$CONTROLLER_URL" --token "$ENROLL_TOKEN" \
|| die "enrollment failed (bad/expired token, or controller unreachable)"
else
2026-09-13 21:53:54 +00:00
if [ -n "$CONTROLLER_URL" ]; then
echo "==> no token given: the daemon will request admission from $CONTROLLER_URL and wait"
echo " for automatic admission or administrator approval."
else
echo "==> no token or controller given: the daemon will discover its cluster on the LAN and wait"
echo " for automatic admission or administrator approval."
fi
2026-08-03 09:02:45 +00:00
fi
systemctl enable --now "$SERVICE_NAME"
echo "==> $SERVICE_NAME enabled and started"
echo
echo "Done. The worker daemon is running${ENROLL_TOKEN:+ and enrolled}."
echo " Status : gpuk status Logs: gpuk logs"
hint_existing_caches "$CACHE_DIR"
return 0
fi
# ── Controller: start the daemon, then bring up the app container ──
systemctl enable --now "$SERVICE_NAME"
echo "==> $SERVICE_NAME enabled and started"
echo "==> applying manifest (first app-container start) ..."
# No unix socket any more: workerd reconciles the app container in-process from
# the on-disk manifest (there is no backend to relay through on the very first
# boot). Steady-state updates go through the daemon over WS.
2026-09-17 00:02:06 +00:00
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" apply || die "apply failed. Check: gpuk logs"
2026-08-03 09:02:45 +00:00
echo
echo "Done. The worker daemon is running and the app container is up."
echo " Status : gpuk status App logs: gpuk logs Daemon: journalctl -u $SERVICE_NAME"
hint_existing_caches "$CACHE_DIR"
}
2026-09-13 21:53:54 +00:00
# Worker manifest: cacheDisks and an EMPTY image so
2026-08-03 09:02:45 +00:00
# has_app_container() is false (apps/worker/src/modules/manifest.rs). No ports, no env,
# no secretsRef — a worker runs no app container.
2026-09-13 21:53:54 +00:00
render_worker_manifest() {
cat <<JSON
2026-08-03 09:02:45 +00:00
{
"schemaVersion": 1,
"image": "",
"mode": "worker",
"cluster": "$(json_str "$CLUSTER")",
"gpus": "all",
"dataRoot": "$(json_str "$DATA_ROOT")",
2026-09-13 21:53:54 +00:00
"cacheDisks": $CACHE_JSON
2026-08-03 09:02:45 +00:00
}
JSON
}
2026-09-17 00:02:06 +00:00
# Every env key gpuk may ever write into a manifest — conditional ones included.
# On a re-run the merge sets each of them to the fresh value or DELETES it when the
# fresh render no longer carries it (a profile change drops GPUK_HSTS); any other
# key was pushed by the controller and is kept. Keep this list in step with
# render_controller_manifest.
INSTALL_ENV_KEYS="NODE_ENV,GPUK_MODE,GPUK_CLUSTER,GPUK_SELF_ENROLL_FILE,NODE_DISPLAY_NAME,HF_HOME,GPUK_DATA_ROOT,GPUK_INSTALL_PROFILE,GPUK_HSTS,GPUK_SESSION_COOKIE_SECURE,GPUK_BOOTSTRAP_MUST_CHANGE,BACKEND_DOCKER_NETWORK,GPUK_PORT,GPUK_PUBLIC_PORT,GPUK_MTLS_PORT,GPUK_PUBLIC_MTLS_PORT,GPUK_LISTEN_ADDR,GPUK_PROXY_PUBLIC_PORT"
# Write the manifest — or, when one exists, MERGE into it (INS-03). Re-running the
# installer is the update path, and the manifest is not ours alone: since the first
# install the controller has patched it (extra cache disks, shared origin, eviction,
# LED binary, ports moved from Settings -> Network…). Rewriting it from flags would
# silently undo all of that, so the fresh render goes through the daemon's own
# `install-manifest`, which only replaces what the installer owns (image, mode,
# cluster, network, data root, ports, secret references, the primary cache disk,
# the env keys above) and keeps the rest. The first write is the render as-is.
write_manifest() { # $1 = render function
if [ -f "$MANIFEST" ]; then
"$1" > "$MANIFEST.new"
chmod 0600 "$MANIFEST.new"
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" install-manifest \
--from "$MANIFEST.new" --own-env "$INSTALL_ENV_KEYS" >/dev/null \
|| { rm -f "$MANIFEST.new"; die "could not merge the new settings into $MANIFEST"; }
rm -f "$MANIFEST.new"
echo "==> merged into the existing $MANIFEST (controller-owned settings kept)"
else
"$1" > "$MANIFEST"
fi
}
write_worker_manifest() { write_manifest render_worker_manifest; }
2026-09-13 21:53:54 +00:00
2026-08-03 09:02:45 +00:00
# Controller manifest: the declarative app-container description workerd applies.
2026-09-13 21:53:54 +00:00
render_controller_manifest() {
2026-08-03 09:02:45 +00:00
# NOTE there is deliberately no PORT here: PORT is the backend's own port, a
2026-09-13 21:53:54 +00:00
# loopback-only 8000 behind nginx in the all-in-one image. What the outside world
# dials is GPUK_PUBLIC_PORT (the published host port), which falls back to nginx's
# own GPUK_PORT when nothing republishes it.
2026-08-03 09:02:45 +00:00
EXTRA_ENV=""
EXTRA_ENV="$EXTRA_ENV,\"GPUK_DATA_ROOT\":\"$(json_str "$DATA_ROOT")\""
2026-09-13 21:53:54 +00:00
[ -z "$PROFILE" ] || EXTRA_ENV="$EXTRA_ENV,\"GPUK_INSTALL_PROFILE\":\"$(json_str "$PROFILE")\""
if [ "$PROFILE" = "public" ]; then
EXTRA_ENV="$EXTRA_ENV,\"GPUK_HSTS\":\"true\""
EXTRA_ENV="$EXTRA_ENV,\"GPUK_SESSION_COOKIE_SECURE\":\"true\""
fi
BOOTSTRAP_SECRET_JSON=""
if [ -s "$BOOTSTRAP_PASSWORD_FILE" ]; then
EXTRA_ENV="$EXTRA_ENV,\"GPUK_BOOTSTRAP_MUST_CHANGE\":\"1\""
BOOTSTRAP_SECRET_JSON=",\"GPUK_BOOTSTRAP_ADMIN_PASSWORD\":\"$(json_str "$BOOTSTRAP_PASSWORD_FILE")\""
fi
CLAIM_SECRET_JSON=""
if [ "$CLAIM_CODE_AVAILABLE" -eq 1 ]; then
CLAIM_SECRET_JSON=",\"GPUK_CLAIM_CODE\":\"$(json_str "$CLAIM_CODE_FILE")\""
fi
2026-08-03 09:02:45 +00:00
[ "$NETWORK" = "host" ] || EXTRA_ENV="$EXTRA_ENV,\"BACKEND_DOCKER_NETWORK\":\"$(json_str "$NETWORK")\""
2026-09-13 21:53:54 +00:00
# Published != bound (INS-43). On a bridged network the container keeps the image's
2026-09-17 00:02:06 +00:00
# FIXED listeners — nginx 8080, worker mTLS 8443, gpuk-proxy 8200 — and the port
# flags only move the HOST side of the publication; Settings -> Network moves it
# later by patching this same manifest, so the container port must never become a
# variable. With host networking nothing is published and the listeners themselves
# take the ports (core/published-ports.ts reads this back with the same rules; the
# proxy port is GPUK_LISTEN_ADDR for the proxy, GPUK_PROXY_PUBLIC_PORT for what the
# backend shows — core/cluster-settings.ts proxyPublicPort).
2026-08-03 09:02:45 +00:00
PORTS_JSON="{}"
2026-09-13 21:53:54 +00:00
if [ "$NETWORK" = "host" ]; then
EXTRA_ENV="$EXTRA_ENV,\"GPUK_PORT\":\"$(json_str "$HTTP_PORT")\""
2026-09-17 00:02:06 +00:00
EXTRA_ENV="$EXTRA_ENV,\"GPUK_MTLS_PORT\":\"$(json_str "$MTLS_PORT")\""
EXTRA_ENV="$EXTRA_ENV,\"GPUK_LISTEN_ADDR\":\"0.0.0.0:$(json_str "$INFERENCE_PORT")\""
2026-09-13 21:53:54 +00:00
else
EXTRA_ENV="$EXTRA_ENV,\"GPUK_PORT\":\"8080\""
EXTRA_ENV="$EXTRA_ENV,\"GPUK_PUBLIC_PORT\":\"$(json_str "$HTTP_PORT")\""
2026-09-17 00:02:06 +00:00
EXTRA_ENV="$EXTRA_ENV,\"GPUK_PUBLIC_MTLS_PORT\":\"$(json_str "$MTLS_PORT")\""
PORTS_JSON="{\"8080\":$HTTP_PORT,\"8200\":$INFERENCE_PORT,\"8443\":$MTLS_PORT}"
2026-08-03 09:02:45 +00:00
fi
2026-09-17 00:02:06 +00:00
EXTRA_ENV="$EXTRA_ENV,\"GPUK_PROXY_PUBLIC_PORT\":\"$(json_str "$INFERENCE_PORT")\""
2026-08-03 09:02:45 +00:00
2026-09-13 21:53:54 +00:00
cat <<JSON
2026-08-03 09:02:45 +00:00
{
"schemaVersion": 1,
"image": "$(json_str "$IMAGE")",
"containerName": "gpu-kitchen",
"mode": "controller",
"cluster": "$(json_str "$CLUSTER")",
"networkMode": "$(json_str "$NETWORK")",
"gpus": "all",
"restartPolicy": "unless-stopped",
"dataRoot": "$(json_str "$DATA_ROOT")",
"cacheDisks": $CACHE_JSON,
"ports": $PORTS_JSON,
"env": {
"NODE_ENV": "production",
"GPUK_MODE": "controller",
"GPUK_CLUSTER": "$(json_str "$CLUSTER")",
2026-09-13 21:53:54 +00:00
"GPUK_SELF_ENROLL_FILE": "$(json_str "$SELF_ENROLL_FILE")",
2026-08-03 09:02:45 +00:00
"NODE_DISPLAY_NAME": "$(json_str "$(hostname)")",
"HF_HOME": "$(json_str "$CACHE_DIR")"$EXTRA_ENV
},
"secretsRef": {
"ENCRYPTION_KEY": "$(json_str "$DATA_ROOT/secrets/encryption_key")",
2026-09-13 21:53:54 +00:00
"NODE_ID": "$(json_str "$DATA_ROOT/secrets/node_id")"$BOOTSTRAP_SECRET_JSON$CLAIM_SECRET_JSON
2026-08-03 09:02:45 +00:00
}
}
JSON
}
2026-09-17 00:02:06 +00:00
write_controller_manifest() { write_manifest render_controller_manifest; }
2026-09-13 21:53:54 +00:00
2026-08-03 09:02:45 +00:00
# ── control subcommands ────────────────────────────────────────────────────────
container_name() {
sed -n 's/.*"containerName"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1
}
2026-09-13 21:53:54 +00:00
manifest_data_root() {
sed -n 's/.*"dataRoot"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1
}
2026-08-03 09:02:45 +00:00
manifest_image() {
sed -n 's/.*"image"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1
}
health_port() {
sed -n 's/.*GPUK_WORKER_HEALTH_PORT=\([0-9]*\).*/\1/p' "$WORKER_ENV" 2>/dev/null | head -1
}
# Split an image reference into repository and tag.
#
# `${img%%:*}` cuts at the FIRST colon and is WRONG:
# `registry.internal:5000/gpuk/controller:v1.2.3` would yield repo
# `registry.internal`. The colon in a registry's host:port is not a tag separator.
# Rule: it is a tag only if the last colon comes after the last slash.
# (Same logic as apps/controller/api/src/core/release-channel.ts — see its unit tests.)
image_repo() {
2026-09-13 21:53:54 +00:00
# A digest suffix (…@sha256:…) never carries the repo; drop it, then apply
# the tag logic — `repo:tag@sha256:…` and `repo@sha256:…` both reduce right.
set -- "${1%@*}"
2026-08-03 09:02:45 +00:00
_t="${1##*:}"
case "$_t" in
"$1") printf '%s' "$1" ;; # no colon at all → untagged
*/*) printf '%s' "$1" ;; # the last colon is inside a path → host:port, untagged
*) printf '%s' "${1%:*}" ;;
esac
}
image_tag() {
2026-09-13 21:53:54 +00:00
# `repo:tag@sha256:…` keeps the human-readable tag next to the content pin —
# docker resolves by digest and ignores the tag. Strip the digest, then parse.
set -- "${1%@*}"
2026-08-03 09:02:45 +00:00
_t="${1##*:}"
case "$_t" in
"$1") return ;;
*/*) return ;;
*) printf '%s' "$_t" ;;
esac
}
# The release channel: one flat JSON document served next to the installer. The
# UI's update check reads the same one (apps/controller/api/src/core/release-channel.ts) — one
# source of truth. Interim default: the public Gitea channel repo — flips to
# https://gpu.kitchen/latest.json once the hub exists (specs/developpement/ci-cd.md).
2026-08-03 09:19:52 +00:00
CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen/channel/raw/branch/main/latest.json}"
2026-08-03 09:02:45 +00:00
2026-09-13 21:53:54 +00:00
# The release public key pinned in THIS copy of gpuk (OPS-20). The channel copy
# gets the real key substituted at publish time; the operator override
# (GPUK_UPDATE_PUBKEY) covers a self-hosted channel with its own keypair. The
# first install fetched gpuk itself over HTTPS from the channel — that moment is
# trust-on-first-use, like a worker's enrolment token pin; every later `update`
# is verified against the key pinned HERE, so whoever controls latest.json can
# no longer pick what an existing install runs.
CHANNEL_PUBKEY="${GPUK_UPDATE_PUBKEY:-RWQ7BKXJqGX2jdKXu1GxeSPVAN3JDRTefpImM/mFRjtFwq4E7mhnQtA7}"
# Fetch latest.json AND its minisign signature, verify, and leave the verified
# document at $CHANNEL_DOC. Fail-closed: no signature, bad signature, no
# minisign CLI or no pinned key are all fatal — GPUK_CHANNEL_INSECURE=1 is the
# explicit, logged opt-out (a private mirror that does not sign).
CHANNEL_DOC=""
channel_fetch() {
2026-08-03 09:02:45 +00:00
command -v curl >/dev/null 2>&1 || return 1
2026-09-13 21:53:54 +00:00
CHANNEL_DOC=$(mktemp) || return 1
curl -fsSL --max-time 20 "$CHANNEL_URL" -o "$CHANNEL_DOC" 2>/dev/null || return 1
if [ "${GPUK_CHANNEL_INSECURE:-}" = "1" ]; then
echo "WARNING: GPUK_CHANNEL_INSECURE=1 — release channel signature NOT verified" >&2
return 0
fi
case "$CHANNEL_PUBKEY" in
""|RWQ7BKXJqGX2jdKXu1GxeSPVAN3JDRTefpImM/mFRjtFwq4E7mhnQtA7*)
die "this gpuk carries no pinned release public key — set GPUK_UPDATE_PUBKEY (the minisign public-key line), or GPUK_CHANNEL_INSECURE=1 to skip verification" ;;
esac
command -v minisign >/dev/null 2>&1 \
|| die "minisign is required to verify the release channel (apt install minisign), or set GPUK_CHANNEL_INSECURE=1"
_sig=$(mktemp)
if ! curl -fsSL --max-time 20 "${CHANNEL_URL}.minisig" -o "$_sig" 2>/dev/null; then
rm -f "$_sig"
die "no signature at ${CHANNEL_URL}.minisig — refusing an unsigned channel document (OPS-20)"
fi
if ! minisign -Vq -m "$CHANNEL_DOC" -x "$_sig" -P "$CHANNEL_PUBKEY" >/dev/null 2>&1; then
rm -f "$_sig"
die "latest.json signature verification FAILED — refusing the channel document (OPS-20)"
fi
rm -f "$_sig"
}
channel_field() {
sed -n 's/.*"'"$2"'"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$1" | head -1
}
channel_version() {
channel_fetch || return 1
channel_field "$CHANNEL_DOC" version
}
# ── backup ─────────────────────────────────────────────────────────────────────
# OPS-10: the embedded Postgres is only backed up COLD — hot-copying pgdata with
# a file tool is forbidden (torn pages). Order matters: stop the DAEMON first
# (its reconciler would immediately restart a stopped app container), then the
# container, snapshot, and restarting the service re-applies the manifest.
# An install on an external DATABASE_URL is refused here: gpuk only owns the
# embedded pgdata — back the real database up with pg_dump/backup-compose.sh.
# The finished directory still has to be copied to encrypted off-host storage,
# next to the recovery set (OPS-04/OPS-05: ENCRYPTION_KEY above all).
cmd_backup() {
need_root
_img=$(manifest_image)
[ -n "$_img" ] || die "this is a worker node — no controller data to back up here"
if grep -q '"DATABASE_URL"' "$MANIFEST" 2>/dev/null; then
die "this install uses an external DATABASE_URL — back THAT database up (pg_dump, or the compose procedures in specs/plateforme/operations.md); gpuk backup only snapshots the embedded pgdata"
fi
_root=$(manifest_data_root)
[ -n "$_root" ] || die "no dataRoot in $MANIFEST"
[ -d "$_root/pgdata" ] || die "no embedded pgdata under $_root — nothing to snapshot"
command -v sha256sum >/dev/null 2>&1 || die "sha256sum is required"
_out="${1:-$_root/backups/$(date -u +%Y%m%dT%H%M%SZ)}"
[ -e "$_out" ] && die "refusing to overwrite existing $_out"
mkdir -p "$(dirname "$_out")"
_tmp="$_out.partial"
rm -rf "$_tmp"; mkdir -p "$_tmp"
_cn=$(container_name)
echo "==> Stopping $SERVICE_NAME (its reconciler would restart the container mid-snapshot)..."
systemctl stop "$SERVICE_NAME" || die "could not stop $SERVICE_NAME"
_restart_daemon() { systemctl start "$SERVICE_NAME" 2>/dev/null || true; }
trap _restart_daemon EXIT
if [ -n "$_cn" ]; then
echo "==> Stopping $_cn (cold snapshot — OPS-10)..."
docker stop "$_cn" >/dev/null 2>&1 || true
_state=$(docker inspect -f '{{.State.Status}}' "$_cn" 2>/dev/null || echo absent)
case "$_state" in
running) die "container $_cn is still running — refusing a hot snapshot" ;;
esac
fi
echo "==> Snapshotting $_root/pgdata..."
tar -C "$_root" -czf "$_tmp/pgdata.tar.gz" pgdata || die "snapshot failed"
cp "$MANIFEST" "$_tmp/host-manifest.json" 2>/dev/null || true
{
echo "{"
echo " \"created_utc\": \"$(date -u +%Y-%m-%dT%H:%M:%SZ)\","
echo " \"image\": \"$(json_str "$_img")\","
echo " \"data_root\": \"$(json_str "$_root")\","
echo " \"kind\": \"cold-pgdata-snapshot\""
echo "}"
} > "$_tmp/backup-manifest.json"
(cd "$_tmp" && sha256sum ./* > SHA256SUMS) || die "checksums failed"
chmod 0700 "$_tmp"
mv "$_tmp" "$_out"
echo "==> Restarting $SERVICE_NAME (re-applies the manifest, container included)..."
systemctl start "$SERVICE_NAME" || die "could not restart $SERVICE_NAME — start it manually"
trap - EXIT
echo "backup: $_out"
echo "Copy it to encrypted OFF-HOST storage together with the recovery set"
echo "(ENCRYPTION_KEY above all — without it the data is unrecoverable, OPS-04)."
echo "A physical pgdata restore requires the same Postgres major and a throwaway"
echo "host rehearsal first (OPS-10)."
}
# ── channel ────────────────────────────────────────────────────────────────────
# Diagnostic (no root): fetch + VERIFY the channel document, print what it
# offers. Exercises exactly the trust chain `update` relies on — the CI probes
# it with a throwaway keypair, an operator uses it to debug a mirror.
cmd_channel() {
channel_fetch || die "cannot fetch $CHANNEL_URL"
echo "channel : $CHANNEL_URL"
echo "version : $(channel_field "$CHANNEL_DOC" version)"
_d=$(channel_field "$CHANNEL_DOC" controllerImageDigest)
[ -n "$_d" ] && echo "digest : $_d"
_d=$(channel_field "$CHANNEL_DOC" controllerImageDigestEnterprise)
[ -n "$_d" ] && echo "digest ee : $_d"
if [ "${GPUK_CHANNEL_INSECURE:-}" = "1" ]; then
echo "signature : SKIPPED (GPUK_CHANNEL_INSECURE=1)"
else
echo "signature : verified"
fi
2026-08-03 09:02:45 +00:00
}
# ── status ─────────────────────────────────────────────────────────────────────
# No control socket any more. Status = the systemd unit state + the daemon's own
# /health endpoint + whether a worker has enrolled (identity present).
cmd_status() {
_active=$(systemctl is-active "$SERVICE_NAME" 2>/dev/null || true)
echo "service : $_active"
if [ -f "$IDENTITY_DIR/identity.json" ]; then
echo "enrolled : yes ($IDENTITY_DIR)"
else
echo "enrolled : no (daemon waits for LAN admission; token enrollment is also available)"
fi
_hp=$(health_port); [ -n "$_hp" ] || _hp=8001
if command -v curl >/dev/null 2>&1; then
_h=$(curl -fsS --max-time 3 "http://127.0.0.1:$_hp/health" 2>/dev/null || true)
[ -n "$_h" ] && echo "health : $_h" || echo "health : (no answer on :$_hp)"
fi
_img=$(manifest_image)
if [ -n "$_img" ]; then
echo "app image : $_img"
echo "app cont. : $(docker inspect -f '{{.State.Status}}' "$(container_name)" 2>/dev/null || echo 'not running')"
else
echo "role : worker (no app container)"
fi
}
# ── enroll ─────────────────────────────────────────────────────────────────────
cmd_enroll() {
need_root
_url=""; _tok=""
while [ $# -gt 0 ]; do
case "$1" in
--controller) _url="$2"; shift 2 ;;
--enroll-token|--token) _tok="$2"; shift 2 ;;
*) die "unknown enroll option: $1" ;;
esac
done
[ -n "$_url" ] || die "enroll needs --controller wss://<controller>:<port>"
[ -n "$_tok" ] || die "enroll needs --token gk_enroll_..."
2026-09-13 21:53:54 +00:00
GPUK_IDENTITY_DIR="$IDENTITY_DIR" GPUK_MACHINE_ID_FILE="$MACHINE_ID_FILE" \
"$BIN_DEST" enroll --controller "$_url" --token "$_tok"
2026-08-03 09:02:45 +00:00
systemctl restart "$SERVICE_NAME" 2>/dev/null || true
}
# ── apply ──────────────────────────────────────────────────────────────────────
# Controller: reconcile the app container from the manifest, in-process. A worker
# has no app container — `apply` there is a no-op with a clear message.
cmd_apply() {
need_root
2026-09-17 00:02:06 +00:00
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" apply
2026-08-03 09:02:45 +00:00
}
# ── update ─────────────────────────────────────────────────────────────────────
# Controller: decide WHICH image TAG to pin, write it into the manifest, then let
# workerd pull + recreate (health-gate + rollback are the daemon's — apply()).
# Worker: the signed-binary self-update is DRIVEN FROM THE CONTROLLER (its Update
# button → POST /api/nodes/:id/host/update → cmd:host_update → verified swap).
# There is no local unverified swap path.
cmd_update() {
need_root
_want=""; _check=0
while [ $# -gt 0 ]; do
case "$1" in
--version) _want="$2"; shift 2 ;;
--check) _check=1; shift ;;
*) die "unknown update option: $1" ;;
esac
done
_image=$(manifest_image)
if [ -z "$_image" ]; then
echo "This is a worker node. Worker self-update is driven from the controller UI"
echo "(the node's Update button), which pushes a minisign-verified binary swap."
return 0
fi
_repo=$(image_repo "$_image")
_current=$(image_tag "$_image")
[ -n "$_current" ] || _current="(untagged)"
if [ "$_check" -eq 1 ]; then
_latest=$(channel_version) || true
echo "installed : $_current"
if [ -z "$_latest" ]; then
echo "available : unknown (cannot reach $CHANNEL_URL)"
exit 1
fi
echo "available : $_latest"
if [ "$_latest" = "$_current" ]; then echo "up to date."; else echo "run 'gpuk update' to move to $_latest"; fi
return 0
fi
2026-09-13 21:53:54 +00:00
_digest=""
2026-08-03 09:02:45 +00:00
if [ -z "$_want" ]; then
2026-09-13 21:53:54 +00:00
# channel_fetch runs in THIS shell (not a $(…) subshell) so a signature
# failure is fatal here — fail-closed — and $CHANNEL_DOC survives. The
# verified signature closes the document half of OPS-20; the digest read
# from it pins CONTENT, closing the mutable-tag half.
if channel_fetch; then
_want=$(channel_field "$CHANNEL_DOC" version)
fi
2026-08-03 09:02:45 +00:00
if [ -z "$_want" ]; then
echo "==> cannot reach $CHANNEL_URL — re-applying the pinned $_current"
2026-09-17 00:02:06 +00:00
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" apply
2026-08-03 09:02:45 +00:00
return 0
fi
2026-09-13 21:53:54 +00:00
# Pick the digest matching the installed edition by image basename — the
# repo itself may be a mirror, the basename is the edition marker.
case "${_repo##*/}" in
*-ee) _digest=$(channel_field "$CHANNEL_DOC" controllerImageDigestEnterprise) ;;
*) _digest=$(channel_field "$CHANNEL_DOC" controllerImageDigest) ;;
esac
2026-08-03 09:02:45 +00:00
fi
2026-09-13 21:53:54 +00:00
_new="$_repo:$_want"
[ -n "$_digest" ] && _new="$_repo:$_want@$_digest"
_installed=$(manifest_image)
if [ "$_new" != "$_installed" ]; then
echo "==> $_current → $_want${_digest:+ (pinned by digest)}"
# Pin the new reference into the manifest, then apply. sed edits the single
# "image" line in place (atomic tmp + move).
2026-08-03 09:02:45 +00:00
_tmp="$MANIFEST.new"
sed "s#\"image\"[[:space:]]*:[[:space:]]*\"[^\"]*\"#\"image\": \"$(json_str "$_new")\"#" "$MANIFEST" > "$_tmp" \
|| die "could not rewrite the image in $MANIFEST"
chmod 0600 "$_tmp"; mv "$_tmp" "$MANIFEST"
else
echo "==> already on $_current — re-pulling and recreating"
fi
2026-09-17 00:02:06 +00:00
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" apply
2026-08-03 09:02:45 +00:00
}
2026-09-17 00:02:06 +00:00
# ── uninstall ──────────────────────────────────────────────────────────────────
# Plain: stop and remove the service, touch nothing else (a pause, reversible by
# `gpuk install`). --purge: everything the installer created goes — the app
# container (and a leftover -old twin), $ETC_DIR with the manifest and the
# enrolled identity, the binary — EXCEPT the data root: database, models and the
# secrets (ENCRYPTION_KEY above all, OPS-04) are the operator's to delete, by
# hand, knowingly. After a purge the next install is a first install (INS-03);
# without it, install.sh sees the manifest and treats the re-run as an update.
2026-08-03 09:02:45 +00:00
cmd_uninstall() {
need_root
2026-09-17 00:02:06 +00:00
_purge=0
while [ $# -gt 0 ]; do
case "$1" in
--purge) _purge=1; shift ;;
*) die "unknown uninstall option: $1 (expected: --purge)" ;;
esac
done
_cn=$(container_name)
_root=$(manifest_data_root)
2026-08-03 09:02:45 +00:00
systemctl disable --now "$SERVICE_NAME" 2>/dev/null || true
rm -f "$UNIT_DEST"; systemctl daemon-reload 2>/dev/null || true
2026-09-17 00:02:06 +00:00
echo "==> removed the systemd service"
if [ "$_purge" -eq 0 ]; then
echo "Left in place: $BIN_DEST, $ETC_DIR (manifest, identity), the data root and any"
echo "app container — 'gpuk install' brings the service back on them."
echo "For a clean slate (everything but the data root): gpuk uninstall --purge"
return 0
fi
if [ -n "$_cn" ] && command -v docker >/dev/null 2>&1; then
for _c in "$_cn" "$_cn-old"; do
docker rm -f "$_c" >/dev/null 2>&1 && echo "==> removed container $_c"
done
fi
for _d in "$ETC_DIR" "$IDENTITY_DIR"; do
case "$_d" in ""|/|/etc|/usr|/var) die "refusing to remove $_d" ;; esac
[ ! -e "$_d" ] || { rm -rf "$_d"; echo "==> removed $_d"; }
done
[ ! -e "$MACHINE_ID_FILE" ] || rm -f "$MACHINE_ID_FILE"
[ ! -e "$BIN_DEST" ] || { rm -f "$BIN_DEST"; echo "==> removed $BIN_DEST"; }
echo
echo "Kept: the data root${_root:+ $_root} — database, models and secrets (ENCRYPTION_KEY)."
echo "A new install over it reuses them. To delete it too, knowingly: rm -rf ${_root:-<data-root>}"
2026-08-03 09:02:45 +00:00
}
usage() {
cat <<EOF
gpuk — GPU Kitchen host daemon (gpu-kitchen-worker)
2026-09-13 21:53:54 +00:00
gpuk install --mode controller --image <ref> [--profile homelab|studio|enterprise|public]
2026-09-17 00:02:06 +00:00
[--cluster N] [--cache-dir P] [--data-root P]
[--http-port P] [--mtls-port P] [--inference-port P]
[--network host|bridge|<net>] [--binary <path>]
2026-08-03 09:02:45 +00:00
gpuk install --mode worker --controller wss://<host>:<port> --enroll-token gk_enroll_...
2026-09-13 21:53:54 +00:00
[--cluster N] [--cache-dir P] [--binary <path>]
2026-09-17 00:02:06 +00:00
[--health-port P] [--data-port P] [--transfer-port P] [--handover-port P]
2026-09-13 21:53:54 +00:00
gpuk install ... --dry-run Validate inputs and print the manifest without changing the host
2026-08-03 09:02:45 +00:00
gpuk status Service state, enrollment, /health, app container status
gpuk enroll --controller wss://<host>:<port> --token gk_enroll_...
gpuk apply (controller) Reconcile the app container from the manifest
gpuk update (controller) Move to the current release (pull + recreate, rollback)
gpuk update --check (controller) Compare the installed version with the release
gpuk update --version <tag> (controller) Move to a specific release
2026-09-13 21:53:54 +00:00
gpuk channel Fetch + VERIFY the release channel and print what it offers
gpuk backup [DIR] (controller) Cold snapshot of the embedded pgdata (stop → tar → restart)
2026-08-03 09:02:45 +00:00
gpuk manifest Print the current manifest
gpuk logs Follow the app container logs (controller) or the daemon journal
2026-09-17 00:02:06 +00:00
gpuk uninstall Remove the systemd service, leave everything else in place
gpuk uninstall --purge Also remove the app container, /etc/gpu-kitchen and the binary
(never the data root: database, models, secrets)
2026-08-03 09:02:45 +00:00
Most people never run this directly: the channel's install.sh installs it
2026-08-03 09:19:52 +00:00
(https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh).
2026-08-03 09:02:45 +00:00
EOF
}
# ── dispatch ────────────────────────────────────────────────────────────────────
cmd="${1:-help}"; if [ $# -gt 0 ]; then shift; fi
case "$cmd" in
install) cmd_install "$@" ;;
status) cmd_status ;;
enroll) cmd_enroll "$@" ;;
apply) cmd_apply ;;
update) cmd_update "$@" ;;
2026-09-13 21:53:54 +00:00
channel) cmd_channel ;;
backup) cmd_backup "${1:-}" ;;
2026-08-03 09:02:45 +00:00
manifest) cat "$MANIFEST" ;;
logs)
_img=$(manifest_image)
if [ -n "$_img" ]; then exec docker logs -f "$(container_name)"; else exec journalctl -u "$SERVICE_NAME" -f; fi
;;
2026-09-17 00:02:06 +00:00
uninstall) cmd_uninstall "$@" ;;
2026-08-03 09:02:45 +00:00
help|-h|--help) usage ;;
*) usage; exit 1 ;;
esac