2026-08-03 09:02:45 +00:00
#!/bin/sh
# gpuk — installer + control CLI for the GPU Kitchen host daemon (gpu-kitchen-worker).
#
# Since C1 a compute node runs NO backend. The single privileged host component is
# the Rust `gpu-kitchen-worker` daemon (workerd), running as root DIRECTLY on the
# host (not in a container). It owns NVML clock/power locks, whitelisted host
# browsing, the model-cache staging pipeline, the signed `gpu-kitchen-bench`
# runner, and — on a controller node — the app container's lifecycle (create,
# health-gate, roll back, pull+recreate to update) via its manifest. There is no
# separate host daemon and no unix control socket: workerd is driven over the
# wss+mTLS channel by the controller, and locally by this CLI.
#
# Two roles:
# worker a headless compute node. Installs the binary + systemd unit + a
2026-09-13 21:53:54 +00:00
# manifest (cacheDisks) with NO app container, NO
2026-08-03 09:02:45 +00:00
# Postgres, NO docker app image. It ENROLLS over mTLS with a
# enrollment token, or waits for LAN discovery admission, then runs.
# controller the full app + UI. Installs the binary + systemd unit + an
# app-container manifest (image, ports, env, secretsRef) that workerd
# applies (`gpu-kitchen-worker apply`, then cmd:host_apply in steady
# state).
#
# Install (root):
# # controller:
2026-08-03 09:19:52 +00:00
# curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/gpuk \
2026-08-03 09:02:45 +00:00
# | sudo sh -s -- install --mode controller \
2026-08-03 09:19:52 +00:00
# --image repo.byterain.io/gpukitchen/gpukitchen-controller:vX.Y.Z
2026-08-03 09:02:45 +00:00
# # worker (enroll against a controller with a single-use token from its UI):
# sudo ./deployments/install/gpuk install --mode worker \
# --controller wss://controller.lan:8443 --enroll-token gk_enroll_... \
# --cache-dir /mnt/models --binary apps/worker/target/release/gpu-kitchen-worker
#
# Control:
# gpuk status | apply | update | enroll | logs | manifest | uninstall
#
set -eu
# ── Paths ────────────────────────────────────────────────────────────────────
# Overridable, so the daemon can be driven against a prefix a normal user owns —
# the only way any of this is testable without handing a test suite root on the
# host. Unset (the real install) they are exactly the systemd defaults workerd uses
# (apps/worker/src/module.rs, apps/worker/src/identity.rs).
BIN_DEST = " ${ GPUK_BIN_DEST :- /usr/local/bin/gpu-kitchen-worker } "
ETC_DIR = " ${ GPUK_ETC_DIR :- /etc/gpu-kitchen } "
MANIFEST = " $ETC_DIR /manifest.json"
IDENTITY_DIR = " ${ GPUK_IDENTITY_DIR :- $ETC_DIR /identity } "
2026-09-13 21:53:54 +00:00
MACHINE_ID_FILE = " ${ GPUK_MACHINE_ID_FILE :- $ETC_DIR /machine-id } "
2026-08-03 09:02:45 +00:00
WORKER_ENV = " $ETC_DIR /worker.env"
SERVICE_NAME = "gpu-kitchen-worker"
2026-09-13 21:53:54 +00:00
UNIT_DEST = " ${ GPUK_UNIT_DEST :- /etc/systemd/system/ $SERVICE_NAME .service } "
2026-08-03 09:02:45 +00:00
# Where to download the binary from when no --binary is given.
GPUK_RELEASE_BASE = " ${ GPUK_RELEASE_BASE :- } "
die() { echo "gpuk: $* " >& 2; exit 1; }
# Root, or able to do the job anyway. Fail with a clear message BEFORE touching
# /etc, /usr/local/bin or systemd. Against a user-owned prefix it simply works.
need_root() {
[ " $( id -u) " -eq 0 ] && return 0
[ -w " $ETC_DIR " ] && return 0
die "this command must run as root (use sudo)"
}
# ── JSON helpers (controlled inputs: paths + identifiers) ──────────────────────
json_str() { printf '%s' " $1 " | sed 's/\\/\\\\/g; s/"/\\"/g' ; }
json_array() { # args → ["a","b",...]
_out = ""
for _p in " $@ " ; do
_e = $( json_str " $_p " )
if [ -z " $_out " ] ; then _out = "\" $_e \"" ; else _out = " $_out ,\" $_e \"" ; fi
done
printf '[%s]' " $_out "
}
# ── install ────────────────────────────────────────────────────────────────────
arch_asset() {
case " $( uname -m) " in
x86_64| amd64) echo "gpu-kitchen-worker-x86_64" ;;
aarch64| arm64) echo "gpu-kitchen-worker-aarch64" ;;
*) die "unsupported architecture: $( uname -m) " ;;
esac
}
install_binary() { # [local-path]
if [ -n " ${ 1 :- } " ] ; then
[ -f " $1 " ] || die "binary not found: $1 "
install -m 0755 " $1 " " $BIN_DEST "
echo "==> installed $BIN_DEST from $1 "
elif [ -n " $GPUK_RELEASE_BASE " ] ; then
command -v curl >/dev/null || die "curl is required to download the binary"
_url = " $GPUK_RELEASE_BASE / $( arch_asset) "
echo "==> downloading $_url "
curl -fsSL " $_url " -o " $BIN_DEST .new"
chmod 0755 " $BIN_DEST .new"
mv " $BIN_DEST .new" " $BIN_DEST "
elif [ -x " $BIN_DEST " ] ; then
echo "==> reusing existing $BIN_DEST "
else
die "no binary: pass --binary <path> or set GPUK_RELEASE_BASE=<url base>"
fi
}
gen_secret() { head -c 32 /dev/urandom | base64 | tr '+/' '-_' | tr -d '=' ; }
2026-09-17 00:02:06 +00:00
# The host ports a leftover app container is reached on ("ui mtls inference"),
# read off the container itself with the precedence the controller applies
# (core/published-ports.ts): host networking moves the listeners (GPUK_PORT,
# GPUK_MTLS_PORT, GPUK_LISTEN_ADDR); anything else keeps the image's fixed
# listeners and publishes them (GPUK_PUBLIC_*, then the port bindings). Empty
# when there is no such container or no docker.
leftover_container_ports() { # $1 = container name
command -v docker >/dev/null 2>& 1 || return 0
# Same format string as install.sh's container_facts — one reading, two scripts.
_f = $( docker inspect -f '{{.State.Status}}|{{.HostConfig.NetworkMode}}|{{range $p, $b := .HostConfig.PortBindings}}{{range $b}}{{$p}}={{.HostPort}} {{end}}{{end}}{{"\n"}}{{range .Config.Env}}{{.}}{{"\n"}}{{end}}' " $1 " 2>/dev/null) \
|| return 0
[ -n " $_f " ] || return 0
_head = $( printf '%s\n' " $_f " | head -1)
_r = ${ _head #*| } ; _net = ${ _r %%|* } ; _bind = ${ _r #*| }
_env() { printf '%s\n' " $_f " | sed -n "s/^ $1 =//p" | head -1; }
_pub() { printf '%s\n' " $_bind " | tr ' ' '\n' | sed -n "s/^ $1 \/tcp=//p" | head -1; }
if [ " $_net " = "host" ] ; then
_ui = $( _env GPUK_PORT) ; _mtls = $( _env GPUK_MTLS_PORT)
_inf = $( _env GPUK_PROXY_PUBLIC_PORT)
[ -n " $_inf " ] || { _la = $( _env GPUK_LISTEN_ADDR) ; _inf = ${ _la ##*: } ; }
else
_ui = $( _env GPUK_PUBLIC_PORT) ; [ -n " $_ui " ] || _ui = $( _pub 8080)
_mtls = $( _env GPUK_PUBLIC_MTLS_PORT) ; [ -n " $_mtls " ] || _mtls = $( _pub 8443)
_inf = $( _env GPUK_PROXY_PUBLIC_PORT) ; [ -n " $_inf " ] || _inf = $( _pub 8200)
fi
printf '%s %s %s' " ${ _ui :- 8080 } " " ${ _mtls :- 8443 } " " ${ _inf :- 8200 } "
}
# Who holds a port: "process", "process in container NAME", or "". The process's
# cgroup names its container (host networking); a bridged publication is found
# as the container's port mapping in `docker ps` (the host-side holder is
# docker-proxy, which says nothing by itself). Same reading as install.sh.
port_owner() { # $1 = ss|netstat, $2 = port
_proc = "" ; _pid = ""
case " $1 " in
ss)
_line = $( ss -ltnpH "sport = : $2 " 2>/dev/null | head -1)
_proc = $( printf '%s' " $_line " | sed -n 's/.*users:((\"\([^"]*\)\".*/\1/p' )
_pid = $( printf '%s' " $_line " | sed -n 's/.*pid=\([0-9]*\).*/\1/p' )
;;
netstat)
_field = $( netstat -ltnp 2>/dev/null \
| awk -v p = " $2 " '{n=split($4,a,":"); if (a[n]==p) {print $NF; exit}}' )
case " $_field " in
*/*) _pid = ${ _field %%/* } ; _proc = ${ _field #*/ } ;;
*) _proc = " $_field " ;;
esac
;;
esac
case " $_pid " in *[ !0-9] *| "" ) _pid = "" ;; esac
_ctr = ""
if command -v docker >/dev/null 2>& 1; then
if [ -n " $_pid " ] && [ -r "/proc/ $_pid /cgroup" ] ; then
_cid = $( sed -n 's#.*docker[-/]\([0-9a-f]\{64\}\).*#\1#p' "/proc/ $_pid /cgroup" 2>/dev/null | head -1)
[ -z " $_cid " ] || _ctr = $( docker inspect -f '{{.Name}}' " $_cid " 2>/dev/null | sed 's|^/||' )
fi
[ -n " $_ctr " ] || _ctr = $( docker ps --format '{{.Names}} {{.Ports}}' 2>/dev/null \
| awk -v p = ": $2 ->" 'index($0, p) { print $1; exit }' )
fi
if [ -n " $_ctr " ] ; then
printf '%s' " ${ _proc :- a process } in container $_ctr "
else
printf '%s' " $_proc "
fi
}
2026-09-13 21:53:54 +00:00
seed_secrets() { # DATA_ROOT
2026-08-03 09:02:45 +00:00
_sd = " $1 /secrets"
mkdir -p " $_sd " ; chmod 0700 " $_sd "
[ -f " $_sd /encryption_key" ] || { umask 077; gen_secret > " $_sd /encryption_key" ; }
[ -f " $_sd /node_id" ] || { umask 077; ( cat /proc/sys/kernel/random/uuid 2>/dev/null || uuidgen) > " $_sd /node_id" ; }
2026-09-13 21:53:54 +00:00
# No nominal first-run password is generated. An operator may pre-provision
# this documented break-glass file; only then is it injected into the app.
[ ! -f " $_sd /bootstrap_admin_password" ] || chmod 0600 " $_sd /bootstrap_admin_password"
}
prepare_claim_code() {
# The file is the operator's recoverable proof of machine possession. Create
# it once, preserve it across reinstalls, and let the backend unlink it after
# the atomic first-account claim. A missing file beside an existing manifest
# therefore means "consumed", never "rotate the credential".
if [ -f " $CLAIM_CODE_FILE " ] ; then
chmod 0600 " $CLAIM_CODE_FILE "
elif [ ! -f " $MANIFEST " ] ; then
umask 077
gen_secret > " $CLAIM_CODE_FILE "
chmod 0600 " $CLAIM_CODE_FILE "
2026-08-03 09:02:45 +00:00
fi
}
2026-09-13 21:53:54 +00:00
# Write the systemd unit generated from the worker's canonical template. The
# generator injects the Rust lock-contention exit code here too, so the binary
# and systemd restart policy cannot silently drift apart.
2026-08-03 09:02:45 +00:00
write_unit() {
2026-09-13 21:53:54 +00:00
# BEGIN GENERATED WORKER SYSTEMD UNIT
cat > " $UNIT_DEST " <<UNIT
2026-08-03 09:02:45 +00:00
[Unit]
Description=GPU Kitchen worker daemon (gpu-kitchen-worker)
2026-09-13 21:53:54 +00:00
Documentation=https://repo.byterain.io/Sebastien/GPU-Manager
2026-08-03 09:02:45 +00:00
After=network-online.target docker.service
Wants=network-online.target docker.service
[Service]
Type=simple
2026-09-13 21:53:54 +00:00
# The worker runs as root on the host (NVML, docker.sock, mounts) — NOT in a
# container. Identity (keypair/cert/CA) lives under $IDENTITY_DIR; enroll once
# before starting this unit. Optional overrides live in $WORKER_ENV.
2026-08-03 09:02:45 +00:00
EnvironmentFile=-$WORKER_ENV
ExecStart=$BIN_DEST
2026-09-13 21:53:54 +00:00
# WRK-173: the old MainPID transfers supervision to the verified replacement
# before it exits. Notifications remain local to this service cgroup.
NotifyAccess=all
# Transient operation state only. The node lock has its own stable inode under
# /run/lock, outside this systemd-managed directory.
2026-08-03 09:02:45 +00:00
RuntimeDirectory=gpu-kitchen
RuntimeDirectoryMode=0750
Restart=always
RestartSec=2
2026-09-13 21:53:54 +00:00
# Lock contention is an operator error, not a crash: do not retry forever while
# another directly installed worker owns the WRK-167 host lock.
RestartPreventExitStatus=75
2026-08-03 09:02:45 +00:00
User=root
2026-09-13 21:53:54 +00:00
# Fail-closed renewal: on a refused cert renewal the daemon exits and systemd
# restarts it into an enroll-required state.
2026-08-03 09:02:45 +00:00
KillSignal=SIGTERM
TimeoutStopSec=15
[Install]
WantedBy=multi-user.target
UNIT
2026-09-13 21:53:54 +00:00
# END GENERATED WORKER SYSTEMD UNIT
}
# TLS reverse-proxy example, FILLED with the operator's domain (INS-47) — written
# only under `--profile public --domain <d>`. Kept aligned with
# deployments/controller/Caddyfile.example (the compose variant); this copy
# targets the all-in-one image, where nginx on the UI port is the single front
# door (INS-09) so one upstream carries pages, /api and the /ws upgrade alike.
# We write a file and NOTHING more: no package install, no service start, no
# other program's config read or touched — putting the proxy in service stays an
# operator act (OPS-13), and its presence stays unverifiable (OPS-67).
write_caddyfile() {
mkdir -p " $DATA_ROOT /caddy"
cat > " $DATA_ROOT /caddy/Caddyfile" <<CADDY
# TLS in front of GPU Kitchen — generated by the installer for --profile public
# (specs/plateforme/installation.md INS-47). Caddy provisions and renews the
# certificate itself once DNS for $DOMAIN points at this machine.
#
# sudo cp $DATA_ROOT/caddy/Caddyfile /etc/caddy/Caddyfile
# sudo systemctl reload caddy
#
# The app already runs with GPUK_HSTS=true and GPUK_SESSION_COOKIE_SECURE=true
# (set by the public profile). What does NOT go through this proxy:
2026-09-17 00:02:06 +00:00
# - the worker mTLS channel (:$MTLS_PORT): workers pin the controller CA and must
2026-09-13 21:53:54 +00:00
# reach it DIRECTLY — terminating it here would break the pin.
# - worker<->worker data transfers (:8300): LAN-only by contract (OPS-68).
$DOMAIN {
encode zstd gzip
reverse_proxy localhost:$HTTP_PORT
}
# OpenAI-compatible inference endpoint (gpuk-proxy) — uncomment when inference
# clients live beyond the trusted LAN; TLS keeps their API keys off the wire.
#
# inference.$DOMAIN {
# encode zstd gzip
2026-09-17 00:02:06 +00:00
# reverse_proxy localhost:$INFERENCE_PORT
2026-09-13 21:53:54 +00:00
# }
CADDY
chmod 0644 " $DATA_ROOT /caddy/Caddyfile"
echo "==> wrote $DATA_ROOT /caddy/Caddyfile (filled TLS reverse-proxy example for $DOMAIN )"
2026-08-03 09:02:45 +00:00
}
# Pre-existing model caches (B81): a server that installs GPU Kitchen usually already
2026-09-13 21:53:54 +00:00
# holds tens or hundreds of GB of weights. We print a hint and nothing more — cache
# questions belong to the first-run wizard (INS-48, REG-41), never to the CLI; no
# config of any other program is read or touched, and referencing a cache stays an
# explicit choice made in the UI.
2026-08-03 09:02:45 +00:00
hint_existing_caches() {
hec_chosen = $( readlink -f " $1 " 2>/dev/null || echo " $1 " )
hec_found = ""
for hec_dir in /root/.cache/huggingface /home/*/.cache/huggingface " ${ HF_HOME :- } " ; do
[ -n " $hec_dir " ] || continue
[ -d " $hec_dir /hub" ] || continue
hec_real = $( readlink -f " $hec_dir " 2>/dev/null || echo " $hec_dir " )
[ " $hec_real " != " $hec_chosen " ] || continue
# Hub layout only (models--*) — matches what the scan can actually reference.
ls -d " $hec_dir " /hub/models--* >/dev/null 2>& 1 || continue
hec_found = " $hec_found $hec_real "
done
[ -n " $hec_found " ] || return 0
echo
for hec_dir in $hec_found ; do
echo "==> existing model cache found at $hec_dir "
done
2026-09-13 21:53:54 +00:00
echo " GPU Kitchen will offer to reuse those models at first launch, and any"
echo " time from Nodes & GPU -> Storage. A reused cache is referenced in"
echo " place. Nothing is moved or deleted."
2026-08-03 09:02:45 +00:00
}
cmd_install() {
IMAGE = "" ; MODE = "worker" ; CLUSTER = "default" ; DATA_ROOT = "/var/lib/gpu-kitchen"
CACHE_DIR = "" ; NETWORK = "host" ; CONTROLLER_URL = "" ; BIN_SRC = "" ; HEALTH_PORT = "8001"
2026-09-17 00:02:06 +00:00
# 1337, not 8080: kept in lockstep with install.sh's default (INS-01). The
# worker channel and the inference endpoint move the same way (INS-46).
ENROLL_TOKEN = "" ; HTTP_PORT = "1337" ; MTLS_PORT = "8443" ; INFERENCE_PORT = "8200"
# Worker data plane (OPS-68): the daemon reads GPUK_DATA_PORT,
# GPUK_MODEL_TRANSFER_PORT and GPUK_HANDOVER_DATA_PORT from worker.env and
# announces the first two to the controller, so moving them here is complete.
DATA_PORT = "8300" ; TRANSFER_PORT = "8301" ; HANDOVER_PORT = "8302"
PROFILE = "" ; DOMAIN = "" ; DRY_RUN = 0
2026-08-03 09:02:45 +00:00
while [ $# -gt 0 ] ; do
case " $1 " in
--image) IMAGE = " $2 " ; shift 2 ;;
--mode) MODE = " $2 " ; shift 2 ;;
--cluster) CLUSTER = " $2 " ; shift 2 ;;
2026-09-13 21:53:54 +00:00
--profile) PROFILE = " $2 " ; shift 2 ;;
--domain) DOMAIN = " $2 " ; shift 2 ;;
2026-08-03 09:02:45 +00:00
--data-root) DATA_ROOT = " $2 " ; shift 2 ;;
--cache-dir) CACHE_DIR = " $2 " ; shift 2 ;;
--network) NETWORK = " $2 " ; shift 2 ;;
--controller) CONTROLLER_URL = " $2 " ; shift 2 ;;
# `--token` is an ENROLLMENT token (single-use or shared), NOT a bearer:
# the worker↔controller channel is cert-only since D11. `--token` is kept as
# a spelling of `--enroll-token`.
--enroll-token| --token) ENROLL_TOKEN = " $2 " ; shift 2 ;;
--health-port) HEALTH_PORT = " $2 " ; shift 2 ;;
--http-port) HTTP_PORT = " $2 " ; shift 2 ;;
2026-09-17 00:02:06 +00:00
--mtls-port) MTLS_PORT = " $2 " ; shift 2 ;;
--inference-port) INFERENCE_PORT = " $2 " ; shift 2 ;;
--data-port) DATA_PORT = " $2 " ; shift 2 ;;
--transfer-port) TRANSFER_PORT = " $2 " ; shift 2 ;;
--handover-port) HANDOVER_PORT = " $2 " ; shift 2 ;;
2026-08-03 09:02:45 +00:00
--binary) BIN_SRC = " $2 " ; shift 2 ;;
2026-09-13 21:53:54 +00:00
--dry-run) DRY_RUN = 1; shift ;;
2026-08-03 09:02:45 +00:00
*) die "unknown install option: $1 " ;;
esac
done
# Roles, as the backend names them (multi-server/config.ts): "controller" (full
# app + UI, accepts workers) and "worker" (headless compute node). Historical
# spellings still work.
case " $MODE " in
controller| server| standalone| manager) MODE = "controller" ;;
worker| agent) MODE = "worker" ;;
*) die "--mode must be controller or worker (got ' $MODE ')" ;;
esac
2026-09-13 21:53:54 +00:00
if [ " $MODE " = "worker" ] && [ -n " $PROFILE " ] ; then
die "--profile applies only to --mode controller; workers do not have an installation profile"
fi
2026-09-17 00:02:06 +00:00
for _pv in " $HTTP_PORT " " $MTLS_PORT " " $INFERENCE_PORT " " $HEALTH_PORT " \
" $DATA_PORT " " $TRANSFER_PORT " " $HANDOVER_PORT " ; do
case " $_pv " in
'' | *[ !0-9] *) die "not a port number: ' $_pv '" ;;
esac
[ " $_pv " -ge 1 ] && [ " $_pv " -le 65535 ] || die "port out of range: $_pv "
done
if [ " $HTTP_PORT " = " $MTLS_PORT " ] || [ " $HTTP_PORT " = " $INFERENCE_PORT " ] || [ " $MTLS_PORT " = " $INFERENCE_PORT " ] ; then
die "--http-port, --mtls-port and --inference-port must differ (got $HTTP_PORT , $MTLS_PORT , $INFERENCE_PORT )"
fi
# The worker's four listeners (health + data plane) must differ too; the
# handover port is the data port's temporary twin (WRK-173), never the same.
_seen = ""
for _pv in " $HEALTH_PORT " " $DATA_PORT " " $TRANSFER_PORT " " $HANDOVER_PORT " ; do
case " $_seen " in
*" $_pv " *) die "--health-port, --data-port, --transfer-port and --handover-port must differ (got $HEALTH_PORT , $DATA_PORT , $TRANSFER_PORT , $HANDOVER_PORT )" ;;
esac
_seen = " $_seen $_pv "
done
2026-09-13 21:53:54 +00:00
case " $PROFILE " in
"" | homelab| studio| enterprise| public) ;;
*) die "--profile must be homelab, studio, enterprise or public (got ' $PROFILE ')" ;;
esac
if [ -n " $DOMAIN " ] ; then
[ " $PROFILE " = "public" ] \
|| die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)"
case " $DOMAIN " in
*[ !A-Za-z0-9.-] *) die "--domain must be a bare domain name (got ' $DOMAIN ')" ;;
esac
fi
if [ " $MODE " = "controller" ] ; then
if [ -z " $IMAGE " ] && [ " $DRY_RUN " -eq 1 ] ; then
IMAGE = "example.invalid/gpukitchen-controller:v0.0.0-dry-run"
fi
[ -n " $IMAGE " ] || die "--image registry/gpukitchen-controller:<tag> is required for a controller"
case " $IMAGE " in
*@sha256:*) ;;
*:latest) die "refusing floating image tag ' $IMAGE ' — use an explicit release tag or digest" ;;
*)
_image_tag = " ${ IMAGE ##*: } "
case " $_image_tag " in
" $IMAGE " | */*) die "controller image must carry an explicit tag or sha256 digest (got ' $IMAGE ')" ;;
esac
;;
esac
fi
2026-08-03 09:02:45 +00:00
[ -n " $CACHE_DIR " ] || CACHE_DIR = " $DATA_ROOT /hf"
2026-09-13 21:53:54 +00:00
CACHE_JSON = "[{\"hostPath\":\" $( json_str " $CACHE_DIR " ) \",\"shared\":false}]"
SELF_ENROLL_FILE = " $DATA_ROOT /self-enroll-token"
BOOTSTRAP_PASSWORD_FILE = " $DATA_ROOT /secrets/bootstrap_admin_password"
CLAIM_CODE_FILE = " $DATA_ROOT /secrets/claim_code"
CLAIM_CODE_AVAILABLE = 0
if [ " $DRY_RUN " -eq 1 ] ; then
[ " $MODE " != "controller" ] || CLAIM_CODE_AVAILABLE = 1
echo "==> dry run: no file, service or container was changed"
2026-09-17 00:02:06 +00:00
[ ! -f " $MANIFEST " ] \
|| echo "==> dry run: $MANIFEST exists — this render would be MERGED into it, controller-owned settings kept"
2026-09-13 21:53:54 +00:00
if [ " $MODE " = "worker" ] ; then render_worker_manifest; else render_controller_manifest; fi
[ -z " $DOMAIN " ] || echo "==> dry run: would write $DATA_ROOT /caddy/Caddyfile for $DOMAIN "
return 0
fi
# ── Port conflicts (INS-46) — the mutator's own guard ────────────────────────
2026-09-17 00:02:06 +00:00
# install.sh's preflight already checks these, and on a terminal it offers an
# alternative port. gpuk is the actual mutator and contributors call it
# DIRECTLY, so it re-checks and refuses, non-interactively, naming the flag
# that moves the port. Same helpers as install.sh (both scripts ship standalone
# from the channel). A listener owned by an existing install is not a conflict:
# a manifest on disk means the re-run is the update path, and every checked
# port is then our own; without a manifest, a leftover app container still
# holds OUR ports — apply() renames and stops it before the new one starts.
2026-09-13 21:53:54 +00:00
if [ ! -f " $MANIFEST " ] ; then
2026-09-17 00:02:06 +00:00
_port_tool = " ${ GPUK_PORT_CHECK_TOOL :- } "
case " $_port_tool " in
"" | ss| netstat) ;;
*) die "GPUK_PORT_CHECK_TOOL must be ss or netstat (got ' $_port_tool ')" ;;
esac
if [ -z " $_port_tool " ] ; then
if command -v ss >/dev/null 2>& 1; then _port_tool = "ss"
elif command -v netstat >/dev/null 2>& 1; then _port_tool = "netstat" ; fi
fi
_ours = ""
if [ " $MODE " = "controller" ] ; then
_ours = $( leftover_container_ports gpu-kitchen)
[ -z " $_ours " ] || echo "==> existing app container gpu-kitchen found (ports $_ours ): the install replaces it"
fi
2026-09-13 21:53:54 +00:00
if [ -z " $_port_tool " ] ; then
echo "==> warning: cannot check for port conflicts (no ss or netstat)"
else
if [ " $MODE " = "controller" ] ; then
2026-09-17 00:02:06 +00:00
set -- " $HTTP_PORT :UI:--http-port" \
" $MTLS_PORT :worker channel:--mtls-port" \
" $INFERENCE_PORT :inference endpoint:--inference-port"
2026-09-13 21:53:54 +00:00
else
# Worker data-plane ports (OPS-68) plus the local health listener.
2026-09-17 00:02:06 +00:00
set -- " $HEALTH_PORT :worker health:--health-port" \
" $DATA_PORT :worker data plane:--data-port" \
" $TRANSFER_PORT :model transfers:--transfer-port" \
" $HANDOVER_PORT :handover data plane:--handover-port"
2026-09-13 21:53:54 +00:00
fi
2026-09-17 00:02:06 +00:00
for _spec in " $@ " ; do
_p = ${ _spec %%:* } ; _rest = ${ _spec #*: } ; _label = ${ _rest %%:* } ; _flag = ${ _rest #*: }
case " $_ours " in *" $_p " *) continue ;; esac
2026-09-13 21:53:54 +00:00
_busy = 1
case " $_port_tool " in
ss) [ -n " $( ss -ltnH "sport = : $_p " 2>/dev/null) " ] || _busy = 0 ;;
netstat)
netstat -ltn 2>/dev/null \
| awk -v p = " $_p " '{n=split($4,a,":"); if (a[n]==p) found=1} END{exit found?0:1}' \
|| _busy = 0
;;
esac
2026-09-17 00:02:06 +00:00
[ " $_busy " -eq 0 ] && continue
_owner = $( port_owner " $_port_tool " " $_p " )
_hint = "Free it first, then run the install again."
[ -z " $_flag " ] || _hint = "Pass $_flag <p> to choose another port, or free it first."
die "port $_p ( $_label ) is already in use by ${ _owner :- an unknown process } . $_hint "
2026-09-13 21:53:54 +00:00
done
fi
fi
need_root
2026-08-03 09:02:45 +00:00
install_binary " $BIN_SRC "
mkdir -p " $ETC_DIR " " $DATA_ROOT " " $CACHE_DIR "
2026-09-13 21:53:54 +00:00
seed_secrets " $DATA_ROOT "
if [ " $MODE " = "controller" ] ; then
prepare_claim_code
[ ! -s " $CLAIM_CODE_FILE " ] || CLAIM_CODE_AVAILABLE = 1
fi
2026-08-03 09:02:45 +00:00
umask 077
if [ " $MODE " = "worker" ] ; then
write_worker_manifest
# Optional env overrides for the daemon (cluster grouping, display name, an
# explicit controller URL). The enrolled identity carries the controller URL
# + CA too — this is belt-and-braces / pre-enroll discovery grouping.
{
echo "GPUK_CLUSTER= $CLUSTER "
echo "GPUK_WORKER_HEALTH_PORT= $HEALTH_PORT "
2026-09-17 00:02:06 +00:00
echo "GPUK_DATA_PORT= $DATA_PORT "
echo "GPUK_MODEL_TRANSFER_PORT= $TRANSFER_PORT "
echo "GPUK_HANDOVER_DATA_PORT= $HANDOVER_PORT "
2026-09-13 21:53:54 +00:00
echo "GPUK_MACHINE_ID_FILE= $MACHINE_ID_FILE "
2026-08-03 09:02:45 +00:00
echo "NODE_DISPLAY_NAME= $( hostname) "
[ -n " $CONTROLLER_URL " ] && echo "GPUK_CONTROLLER_URLS= $CONTROLLER_URL "
} > " $WORKER_ENV "
chmod 0600 " $WORKER_ENV "
echo "==> wrote $MANIFEST (worker: no app container)"
else
write_controller_manifest
2026-09-13 21:53:54 +00:00
# The host daemon and app container share DATA_ROOT. Only controller-mode
# workerd gets this private bootstrap channel; remote workers stay tokenless.
{
echo "GPUK_CLUSTER= $CLUSTER "
echo "GPUK_WORKER_HEALTH_PORT= $HEALTH_PORT "
echo "GPUK_MACHINE_ID_FILE= $MACHINE_ID_FILE "
2026-09-17 00:02:06 +00:00
echo "GPUK_CONTROLLER_URLS=wss://127.0.0.1: $MTLS_PORT "
2026-09-13 21:53:54 +00:00
echo "GPUK_SELF_ENROLL_FILE= $SELF_ENROLL_FILE "
echo "NODE_DISPLAY_NAME= $( hostname) "
} > " $WORKER_ENV "
chmod 0600 " $WORKER_ENV "
2026-08-03 09:02:45 +00:00
echo "==> wrote $MANIFEST (controller: app container $IMAGE )"
fi
chmod 0600 " $MANIFEST "
2026-09-13 21:53:54 +00:00
[ -z " $DOMAIN " ] || write_caddyfile
2026-08-03 09:02:45 +00:00
write_unit
systemctl daemon-reload
echo "==> wrote $UNIT_DEST "
# ── Worker: token enrollment before start, or unattended LAN discovery ──
if [ " $MODE " = "worker" ] ; then
if [ -n " $ENROLL_TOKEN " ] ; then
[ -n " $CONTROLLER_URL " ] || die "--controller wss://<controller>:<port> is required to enroll"
echo "==> enrolling against $CONTROLLER_URL ..."
2026-09-13 21:53:54 +00:00
GPUK_IDENTITY_DIR = " $IDENTITY_DIR " GPUK_MACHINE_ID_FILE = " $MACHINE_ID_FILE " " $BIN_DEST " enroll \
2026-08-03 09:02:45 +00:00
--controller " $CONTROLLER_URL " --token " $ENROLL_TOKEN " \
|| die "enrollment failed (bad/expired token, or controller unreachable)"
else
2026-09-13 21:53:54 +00:00
if [ -n " $CONTROLLER_URL " ] ; then
echo "==> no token given: the daemon will request admission from $CONTROLLER_URL and wait"
echo " for automatic admission or administrator approval."
else
echo "==> no token or controller given: the daemon will discover its cluster on the LAN and wait"
echo " for automatic admission or administrator approval."
fi
2026-08-03 09:02:45 +00:00
fi
systemctl enable --now " $SERVICE_NAME "
echo "==> $SERVICE_NAME enabled and started"
echo
echo "Done. The worker daemon is running ${ ENROLL_TOKEN :+ and enrolled } ."
echo " Status : gpuk status Logs: gpuk logs"
hint_existing_caches " $CACHE_DIR "
return 0
fi
# ── Controller: start the daemon, then bring up the app container ──
systemctl enable --now " $SERVICE_NAME "
echo "==> $SERVICE_NAME enabled and started"
echo "==> applying manifest (first app-container start) ..."
# No unix socket any more: workerd reconciles the app container in-process from
# the on-disk manifest (there is no backend to relay through on the very first
# boot). Steady-state updates go through the daemon over WS.
2026-09-17 00:02:06 +00:00
GPUK_MANIFEST_PATH = " $MANIFEST " " $BIN_DEST " apply || die "apply failed. Check: gpuk logs"
2026-08-03 09:02:45 +00:00
echo
echo "Done. The worker daemon is running and the app container is up."
echo " Status : gpuk status App logs: gpuk logs Daemon: journalctl -u $SERVICE_NAME "
hint_existing_caches " $CACHE_DIR "
}
2026-09-13 21:53:54 +00:00
# Worker manifest: cacheDisks and an EMPTY image so
2026-08-03 09:02:45 +00:00
# has_app_container() is false (apps/worker/src/modules/manifest.rs). No ports, no env,
# no secretsRef — a worker runs no app container.
2026-09-13 21:53:54 +00:00
render_worker_manifest() {
cat <<JSON
2026-08-03 09:02:45 +00:00
{
"schemaVersion": 1,
"image": "",
"mode": "worker",
"cluster": "$(json_str "$CLUSTER")",
"gpus": "all",
"dataRoot": "$(json_str "$DATA_ROOT")",
2026-09-13 21:53:54 +00:00
"cacheDisks": $CACHE_JSON
2026-08-03 09:02:45 +00:00
}
JSON
}
2026-09-17 00:02:06 +00:00
# Every env key gpuk may ever write into a manifest — conditional ones included.
# On a re-run the merge sets each of them to the fresh value or DELETES it when the
# fresh render no longer carries it (a profile change drops GPUK_HSTS); any other
# key was pushed by the controller and is kept. Keep this list in step with
# render_controller_manifest.
INSTALL_ENV_KEYS = "NODE_ENV,GPUK_MODE,GPUK_CLUSTER,GPUK_SELF_ENROLL_FILE,NODE_DISPLAY_NAME,HF_HOME,GPUK_DATA_ROOT,GPUK_INSTALL_PROFILE,GPUK_HSTS,GPUK_SESSION_COOKIE_SECURE,GPUK_BOOTSTRAP_MUST_CHANGE,BACKEND_DOCKER_NETWORK,GPUK_PORT,GPUK_PUBLIC_PORT,GPUK_MTLS_PORT,GPUK_PUBLIC_MTLS_PORT,GPUK_LISTEN_ADDR,GPUK_PROXY_PUBLIC_PORT"
# Write the manifest — or, when one exists, MERGE into it (INS-03). Re-running the
# installer is the update path, and the manifest is not ours alone: since the first
# install the controller has patched it (extra cache disks, shared origin, eviction,
# LED binary, ports moved from Settings -> Network…). Rewriting it from flags would
# silently undo all of that, so the fresh render goes through the daemon's own
# `install-manifest`, which only replaces what the installer owns (image, mode,
# cluster, network, data root, ports, secret references, the primary cache disk,
# the env keys above) and keeps the rest. The first write is the render as-is.
write_manifest() { # $1 = render function
if [ -f " $MANIFEST " ] ; then
" $1 " > " $MANIFEST .new"
chmod 0600 " $MANIFEST .new"
GPUK_MANIFEST_PATH = " $MANIFEST " " $BIN_DEST " install-manifest \
--from " $MANIFEST .new" --own-env " $INSTALL_ENV_KEYS " >/dev/null \
|| { rm -f " $MANIFEST .new" ; die "could not merge the new settings into $MANIFEST " ; }
rm -f " $MANIFEST .new"
echo "==> merged into the existing $MANIFEST (controller-owned settings kept)"
else
" $1 " > " $MANIFEST "
fi
}
write_worker_manifest() { write_manifest render_worker_manifest; }
2026-09-13 21:53:54 +00:00
2026-08-03 09:02:45 +00:00
# Controller manifest: the declarative app-container description workerd applies.
2026-09-13 21:53:54 +00:00
render_controller_manifest() {
2026-08-03 09:02:45 +00:00
# NOTE there is deliberately no PORT here: PORT is the backend's own port, a
2026-09-13 21:53:54 +00:00
# loopback-only 8000 behind nginx in the all-in-one image. What the outside world
# dials is GPUK_PUBLIC_PORT (the published host port), which falls back to nginx's
# own GPUK_PORT when nothing republishes it.
2026-08-03 09:02:45 +00:00
EXTRA_ENV = ""
EXTRA_ENV = " $EXTRA_ENV ,\"GPUK_DATA_ROOT\":\" $( json_str " $DATA_ROOT " ) \""
2026-09-13 21:53:54 +00:00
[ -z " $PROFILE " ] || EXTRA_ENV = " $EXTRA_ENV ,\"GPUK_INSTALL_PROFILE\":\" $( json_str " $PROFILE " ) \""
if [ " $PROFILE " = "public" ] ; then
EXTRA_ENV = " $EXTRA_ENV ,\"GPUK_HSTS\":\"true\""
EXTRA_ENV = " $EXTRA_ENV ,\"GPUK_SESSION_COOKIE_SECURE\":\"true\""
fi
BOOTSTRAP_SECRET_JSON = ""
if [ -s " $BOOTSTRAP_PASSWORD_FILE " ] ; then
EXTRA_ENV = " $EXTRA_ENV ,\"GPUK_BOOTSTRAP_MUST_CHANGE\":\"1\""
BOOTSTRAP_SECRET_JSON = ",\"GPUK_BOOTSTRAP_ADMIN_PASSWORD\":\" $( json_str " $BOOTSTRAP_PASSWORD_FILE " ) \""
fi
CLAIM_SECRET_JSON = ""
if [ " $CLAIM_CODE_AVAILABLE " -eq 1 ] ; then
CLAIM_SECRET_JSON = ",\"GPUK_CLAIM_CODE\":\" $( json_str " $CLAIM_CODE_FILE " ) \""
fi
2026-08-03 09:02:45 +00:00
[ " $NETWORK " = "host" ] || EXTRA_ENV = " $EXTRA_ENV ,\"BACKEND_DOCKER_NETWORK\":\" $( json_str " $NETWORK " ) \""
2026-09-13 21:53:54 +00:00
# Published != bound (INS-43). On a bridged network the container keeps the image's
2026-09-17 00:02:06 +00:00
# FIXED listeners — nginx 8080, worker mTLS 8443, gpuk-proxy 8200 — and the port
# flags only move the HOST side of the publication; Settings -> Network moves it
# later by patching this same manifest, so the container port must never become a
# variable. With host networking nothing is published and the listeners themselves
# take the ports (core/published-ports.ts reads this back with the same rules; the
# proxy port is GPUK_LISTEN_ADDR for the proxy, GPUK_PROXY_PUBLIC_PORT for what the
# backend shows — core/cluster-settings.ts proxyPublicPort).
2026-08-03 09:02:45 +00:00
PORTS_JSON = "{}"
2026-09-13 21:53:54 +00:00
if [ " $NETWORK " = "host" ] ; then
EXTRA_ENV = " $EXTRA_ENV ,\"GPUK_PORT\":\" $( json_str " $HTTP_PORT " ) \""
2026-09-17 00:02:06 +00:00
EXTRA_ENV = " $EXTRA_ENV ,\"GPUK_MTLS_PORT\":\" $( json_str " $MTLS_PORT " ) \""
EXTRA_ENV = " $EXTRA_ENV ,\"GPUK_LISTEN_ADDR\":\"0.0.0.0: $( json_str " $INFERENCE_PORT " ) \""
2026-09-13 21:53:54 +00:00
else
EXTRA_ENV = " $EXTRA_ENV ,\"GPUK_PORT\":\"8080\""
EXTRA_ENV = " $EXTRA_ENV ,\"GPUK_PUBLIC_PORT\":\" $( json_str " $HTTP_PORT " ) \""
2026-09-17 00:02:06 +00:00
EXTRA_ENV = " $EXTRA_ENV ,\"GPUK_PUBLIC_MTLS_PORT\":\" $( json_str " $MTLS_PORT " ) \""
PORTS_JSON = "{\"8080\": $HTTP_PORT ,\"8200\": $INFERENCE_PORT ,\"8443\": $MTLS_PORT }"
2026-08-03 09:02:45 +00:00
fi
2026-09-17 00:02:06 +00:00
EXTRA_ENV = " $EXTRA_ENV ,\"GPUK_PROXY_PUBLIC_PORT\":\" $( json_str " $INFERENCE_PORT " ) \""
2026-08-03 09:02:45 +00:00
2026-09-13 21:53:54 +00:00
cat <<JSON
2026-08-03 09:02:45 +00:00
{
"schemaVersion": 1,
"image": "$(json_str "$IMAGE")",
"containerName": "gpu-kitchen",
"mode": "controller",
"cluster": "$(json_str "$CLUSTER")",
"networkMode": "$(json_str "$NETWORK")",
"gpus": "all",
"restartPolicy": "unless-stopped",
"dataRoot": "$(json_str "$DATA_ROOT")",
"cacheDisks": $CACHE_JSON ,
"ports" : $PORTS_JSON ,
"env" : {
"NODE_ENV" : "production" ,
"GPUK_MODE" : "controller" ,
"GPUK_CLUSTER" : " $( json_str " $CLUSTER " ) " ,
2026-09-13 21:53:54 +00:00
"GPUK_SELF_ENROLL_FILE" : " $( json_str " $SELF_ENROLL_FILE " ) " ,
2026-08-03 09:02:45 +00:00
"NODE_DISPLAY_NAME" : " $( json_str " $( hostname) " ) " ,
"HF_HOME" : " $( json_str " $CACHE_DIR " ) " $EXTRA_ENV
} ,
"secretsRef" : {
"ENCRYPTION_KEY" : " $( json_str " $DATA_ROOT /secrets/encryption_key" ) " ,
2026-09-13 21:53:54 +00:00
"NODE_ID" : " $( json_str " $DATA_ROOT /secrets/node_id" ) " $BOOTSTRAP_SECRET_JSON$CLAIM_SECRET_JSON
2026-08-03 09:02:45 +00:00
}
}
JSON
}
2026-09-17 00:02:06 +00:00
write_controller_manifest() { write_manifest render_controller_manifest; }
2026-09-13 21:53:54 +00:00
2026-08-03 09:02:45 +00:00
# ── control subcommands ────────────────────────────────────────────────────────
container_name() {
sed -n 's/.*"containerName"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' " $MANIFEST " 2>/dev/null | head -1
}
2026-09-13 21:53:54 +00:00
manifest_data_root() {
sed -n 's/.*"dataRoot"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' " $MANIFEST " 2>/dev/null | head -1
}
2026-08-03 09:02:45 +00:00
manifest_image() {
sed -n 's/.*"image"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' " $MANIFEST " 2>/dev/null | head -1
}
health_port() {
sed -n 's/.*GPUK_WORKER_HEALTH_PORT=\([0-9]*\).*/\1/p' " $WORKER_ENV " 2>/dev/null | head -1
}
# Split an image reference into repository and tag.
#
# `${img%%:*}` cuts at the FIRST colon and is WRONG:
# `registry.internal:5000/gpuk/controller:v1.2.3` would yield repo
# `registry.internal`. The colon in a registry's host:port is not a tag separator.
# Rule: it is a tag only if the last colon comes after the last slash.
# (Same logic as apps/controller/api/src/core/release-channel.ts — see its unit tests.)
image_repo() {
2026-09-13 21:53:54 +00:00
# A digest suffix (…@sha256:…) never carries the repo; drop it, then apply
# the tag logic — `repo:tag@sha256:…` and `repo@sha256:…` both reduce right.
set -- " ${ 1 %@* } "
2026-08-03 09:02:45 +00:00
_t = " ${ 1 ##*: } "
case " $_t " in
" $1 " ) printf '%s' " $1 " ;; # no colon at all → untagged
*/*) printf '%s' " $1 " ;; # the last colon is inside a path → host:port, untagged
*) printf '%s' " ${ 1 %:* } " ;;
esac
}
image_tag() {
2026-09-13 21:53:54 +00:00
# `repo:tag@sha256:…` keeps the human-readable tag next to the content pin —
# docker resolves by digest and ignores the tag. Strip the digest, then parse.
set -- " ${ 1 %@* } "
2026-08-03 09:02:45 +00:00
_t = " ${ 1 ##*: } "
case " $_t " in
" $1 " ) return ;;
*/*) return ;;
*) printf '%s' " $_t " ;;
esac
}
# The release channel: one flat JSON document served next to the installer. The
# UI's update check reads the same one (apps/controller/api/src/core/release-channel.ts) — one
# source of truth. Interim default: the public Gitea channel repo — flips to
# https://gpu.kitchen/latest.json once the hub exists (specs/developpement/ci-cd.md).
2026-08-03 09:19:52 +00:00
CHANNEL_URL = " ${ GPUK_CHANNEL_URL :- https ://repo.byterain.io/gpukitchen/channel/raw/branch/main/latest.json } "
2026-08-03 09:02:45 +00:00
2026-09-13 21:53:54 +00:00
# The release public key pinned in THIS copy of gpuk (OPS-20). The channel copy
# gets the real key substituted at publish time; the operator override
# (GPUK_UPDATE_PUBKEY) covers a self-hosted channel with its own keypair. The
# first install fetched gpuk itself over HTTPS from the channel — that moment is
# trust-on-first-use, like a worker's enrolment token pin; every later `update`
# is verified against the key pinned HERE, so whoever controls latest.json can
# no longer pick what an existing install runs.
CHANNEL_PUBKEY = " ${ GPUK_UPDATE_PUBKEY :- RWQ7BKXJqGX2jdKXu1GxeSPVAN3JDRTefpImM /mFRjtFwq4E7mhnQtA7 } "
# Fetch latest.json AND its minisign signature, verify, and leave the verified
# document at $CHANNEL_DOC. Fail-closed: no signature, bad signature, no
# minisign CLI or no pinned key are all fatal — GPUK_CHANNEL_INSECURE=1 is the
# explicit, logged opt-out (a private mirror that does not sign).
CHANNEL_DOC = ""
channel_fetch() {
2026-08-03 09:02:45 +00:00
command -v curl >/dev/null 2>& 1 || return 1
2026-09-13 21:53:54 +00:00
CHANNEL_DOC = $( mktemp) || return 1
curl -fsSL --max-time 20 " $CHANNEL_URL " -o " $CHANNEL_DOC " 2>/dev/null || return 1
if [ " ${ GPUK_CHANNEL_INSECURE :- } " = "1" ] ; then
echo "WARNING: GPUK_CHANNEL_INSECURE=1 — release channel signature NOT verified" >& 2
return 0
fi
case " $CHANNEL_PUBKEY " in
"" | RWQ7BKXJqGX2jdKXu1GxeSPVAN3JDRTefpImM/mFRjtFwq4E7mhnQtA7*)
die "this gpuk carries no pinned release public key — set GPUK_UPDATE_PUBKEY (the minisign public-key line), or GPUK_CHANNEL_INSECURE=1 to skip verification" ;;
esac
command -v minisign >/dev/null 2>& 1 \
|| die "minisign is required to verify the release channel (apt install minisign), or set GPUK_CHANNEL_INSECURE=1"
_sig = $( mktemp)
if ! curl -fsSL --max-time 20 " ${ CHANNEL_URL } .minisig" -o " $_sig " 2>/dev/null; then
rm -f " $_sig "
die "no signature at ${ CHANNEL_URL } .minisig — refusing an unsigned channel document (OPS-20)"
fi
if ! minisign -Vq -m " $CHANNEL_DOC " -x " $_sig " -P " $CHANNEL_PUBKEY " >/dev/null 2>& 1; then
rm -f " $_sig "
die "latest.json signature verification FAILED — refusing the channel document (OPS-20)"
fi
rm -f " $_sig "
}
channel_field() {
sed -n 's/.*"' " $2 " '"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' " $1 " | head -1
}
channel_version() {
channel_fetch || return 1
channel_field " $CHANNEL_DOC " version
}
# ── backup ─────────────────────────────────────────────────────────────────────
# OPS-10: the embedded Postgres is only backed up COLD — hot-copying pgdata with
# a file tool is forbidden (torn pages). Order matters: stop the DAEMON first
# (its reconciler would immediately restart a stopped app container), then the
# container, snapshot, and restarting the service re-applies the manifest.
# An install on an external DATABASE_URL is refused here: gpuk only owns the
# embedded pgdata — back the real database up with pg_dump/backup-compose.sh.
# The finished directory still has to be copied to encrypted off-host storage,
# next to the recovery set (OPS-04/OPS-05: ENCRYPTION_KEY above all).
cmd_backup() {
need_root
_img = $( manifest_image)
[ -n " $_img " ] || die "this is a worker node — no controller data to back up here"
if grep -q '"DATABASE_URL"' " $MANIFEST " 2>/dev/null; then
die "this install uses an external DATABASE_URL — back THAT database up (pg_dump, or the compose procedures in specs/plateforme/operations.md); gpuk backup only snapshots the embedded pgdata"
fi
_root = $( manifest_data_root)
[ -n " $_root " ] || die "no dataRoot in $MANIFEST "
[ -d " $_root /pgdata" ] || die "no embedded pgdata under $_root — nothing to snapshot"
command -v sha256sum >/dev/null 2>& 1 || die "sha256sum is required"
_out = " ${ 1 :- $_root /backups/ $( date -u +%Y%m%dT%H%M%SZ) } "
[ -e " $_out " ] && die "refusing to overwrite existing $_out "
mkdir -p " $( dirname " $_out " ) "
_tmp = " $_out .partial"
rm -rf " $_tmp " ; mkdir -p " $_tmp "
_cn = $( container_name)
echo "==> Stopping $SERVICE_NAME (its reconciler would restart the container mid-snapshot)..."
systemctl stop " $SERVICE_NAME " || die "could not stop $SERVICE_NAME "
_restart_daemon() { systemctl start " $SERVICE_NAME " 2>/dev/null || true; }
trap _restart_daemon EXIT
if [ -n " $_cn " ] ; then
echo "==> Stopping $_cn (cold snapshot — OPS-10)..."
docker stop " $_cn " >/dev/null 2>& 1 || true
_state = $( docker inspect -f '{{.State.Status}}' " $_cn " 2>/dev/null || echo absent)
case " $_state " in
running) die "container $_cn is still running — refusing a hot snapshot" ;;
esac
fi
echo "==> Snapshotting $_root /pgdata..."
tar -C " $_root " -czf " $_tmp /pgdata.tar.gz" pgdata || die "snapshot failed"
cp " $MANIFEST " " $_tmp /host-manifest.json" 2>/dev/null || true
{
echo "{"
echo " \"created_utc\": \" $( date -u +%Y-%m-%dT%H:%M:%SZ) \","
echo " \"image\": \" $( json_str " $_img " ) \","
echo " \"data_root\": \" $( json_str " $_root " ) \","
echo " \"kind\": \"cold-pgdata-snapshot\""
echo "}"
} > " $_tmp /backup-manifest.json"
( cd " $_tmp " && sha256sum ./* > SHA256SUMS) || die "checksums failed"
chmod 0700 " $_tmp "
mv " $_tmp " " $_out "
echo "==> Restarting $SERVICE_NAME (re-applies the manifest, container included)..."
systemctl start " $SERVICE_NAME " || die "could not restart $SERVICE_NAME — start it manually"
trap - EXIT
echo "backup: $_out "
echo "Copy it to encrypted OFF-HOST storage together with the recovery set"
echo "(ENCRYPTION_KEY above all — without it the data is unrecoverable, OPS-04)."
echo "A physical pgdata restore requires the same Postgres major and a throwaway"
echo "host rehearsal first (OPS-10)."
}
# ── channel ────────────────────────────────────────────────────────────────────
# Diagnostic (no root): fetch + VERIFY the channel document, print what it
# offers. Exercises exactly the trust chain `update` relies on — the CI probes
# it with a throwaway keypair, an operator uses it to debug a mirror.
cmd_channel() {
channel_fetch || die "cannot fetch $CHANNEL_URL "
echo "channel : $CHANNEL_URL "
echo "version : $( channel_field " $CHANNEL_DOC " version) "
_d = $( channel_field " $CHANNEL_DOC " controllerImageDigest)
[ -n " $_d " ] && echo "digest : $_d "
_d = $( channel_field " $CHANNEL_DOC " controllerImageDigestEnterprise)
[ -n " $_d " ] && echo "digest ee : $_d "
if [ " ${ GPUK_CHANNEL_INSECURE :- } " = "1" ] ; then
echo "signature : SKIPPED (GPUK_CHANNEL_INSECURE=1)"
else
echo "signature : verified"
fi
2026-08-03 09:02:45 +00:00
}
# ── status ─────────────────────────────────────────────────────────────────────
# No control socket any more. Status = the systemd unit state + the daemon's own
# /health endpoint + whether a worker has enrolled (identity present).
cmd_status() {
_active = $( systemctl is-active " $SERVICE_NAME " 2>/dev/null || true )
echo "service : $_active "
if [ -f " $IDENTITY_DIR /identity.json" ] ; then
echo "enrolled : yes ( $IDENTITY_DIR )"
else
echo "enrolled : no (daemon waits for LAN admission; token enrollment is also available)"
fi
_hp = $( health_port) ; [ -n " $_hp " ] || _hp = 8001
if command -v curl >/dev/null 2>& 1; then
_h = $( curl -fsS --max-time 3 "http://127.0.0.1: $_hp /health" 2>/dev/null || true )
[ -n " $_h " ] && echo "health : $_h " || echo "health : (no answer on : $_hp )"
fi
_img = $( manifest_image)
if [ -n " $_img " ] ; then
echo "app image : $_img "
echo "app cont. : $( docker inspect -f '{{.State.Status}}' " $( container_name) " 2>/dev/null || echo 'not running' ) "
else
echo "role : worker (no app container)"
fi
}
# ── enroll ─────────────────────────────────────────────────────────────────────
cmd_enroll() {
need_root
_url = "" ; _tok = ""
while [ $# -gt 0 ] ; do
case " $1 " in
--controller) _url = " $2 " ; shift 2 ;;
--enroll-token| --token) _tok = " $2 " ; shift 2 ;;
*) die "unknown enroll option: $1 " ;;
esac
done
[ -n " $_url " ] || die "enroll needs --controller wss://<controller>:<port>"
[ -n " $_tok " ] || die "enroll needs --token gk_enroll_..."
2026-09-13 21:53:54 +00:00
GPUK_IDENTITY_DIR = " $IDENTITY_DIR " GPUK_MACHINE_ID_FILE = " $MACHINE_ID_FILE " \
" $BIN_DEST " enroll --controller " $_url " --token " $_tok "
2026-08-03 09:02:45 +00:00
systemctl restart " $SERVICE_NAME " 2>/dev/null || true
}
# ── apply ──────────────────────────────────────────────────────────────────────
# Controller: reconcile the app container from the manifest, in-process. A worker
# has no app container — `apply` there is a no-op with a clear message.
cmd_apply() {
need_root
2026-09-17 00:02:06 +00:00
GPUK_MANIFEST_PATH = " $MANIFEST " " $BIN_DEST " apply
2026-08-03 09:02:45 +00:00
}
# ── update ─────────────────────────────────────────────────────────────────────
# Controller: decide WHICH image TAG to pin, write it into the manifest, then let
# workerd pull + recreate (health-gate + rollback are the daemon's — apply()).
# Worker: the signed-binary self-update is DRIVEN FROM THE CONTROLLER (its Update
# button → POST /api/nodes/:id/host/update → cmd:host_update → verified swap).
# There is no local unverified swap path.
cmd_update() {
need_root
_want = "" ; _check = 0
while [ $# -gt 0 ] ; do
case " $1 " in
--version) _want = " $2 " ; shift 2 ;;
--check) _check = 1; shift ;;
*) die "unknown update option: $1 " ;;
esac
done
_image = $( manifest_image)
if [ -z " $_image " ] ; then
echo "This is a worker node. Worker self-update is driven from the controller UI"
echo "(the node's Update button), which pushes a minisign-verified binary swap."
return 0
fi
_repo = $( image_repo " $_image " )
_current = $( image_tag " $_image " )
[ -n " $_current " ] || _current = "(untagged)"
if [ " $_check " -eq 1 ] ; then
_latest = $( channel_version) || true
echo "installed : $_current "
if [ -z " $_latest " ] ; then
echo "available : unknown (cannot reach $CHANNEL_URL )"
exit 1
fi
echo "available : $_latest "
if [ " $_latest " = " $_current " ] ; then echo "up to date." ; else echo "run 'gpuk update' to move to $_latest " ; fi
return 0
fi
2026-09-13 21:53:54 +00:00
_digest = ""
2026-08-03 09:02:45 +00:00
if [ -z " $_want " ] ; then
2026-09-13 21:53:54 +00:00
# channel_fetch runs in THIS shell (not a $(…) subshell) so a signature
# failure is fatal here — fail-closed — and $CHANNEL_DOC survives. The
# verified signature closes the document half of OPS-20; the digest read
# from it pins CONTENT, closing the mutable-tag half.
if channel_fetch; then
_want = $( channel_field " $CHANNEL_DOC " version)
fi
2026-08-03 09:02:45 +00:00
if [ -z " $_want " ] ; then
echo "==> cannot reach $CHANNEL_URL — re-applying the pinned $_current "
2026-09-17 00:02:06 +00:00
GPUK_MANIFEST_PATH = " $MANIFEST " " $BIN_DEST " apply
2026-08-03 09:02:45 +00:00
return 0
fi
2026-09-13 21:53:54 +00:00
# Pick the digest matching the installed edition by image basename — the
# repo itself may be a mirror, the basename is the edition marker.
case " ${ _repo ##*/ } " in
*-ee) _digest = $( channel_field " $CHANNEL_DOC " controllerImageDigestEnterprise) ;;
*) _digest = $( channel_field " $CHANNEL_DOC " controllerImageDigest) ;;
esac
2026-08-03 09:02:45 +00:00
fi
2026-09-13 21:53:54 +00:00
_new = " $_repo : $_want "
[ -n " $_digest " ] && _new = " $_repo : $_want @ $_digest "
_installed = $( manifest_image)
if [ " $_new " != " $_installed " ] ; then
echo "==> $_current → $_want ${ _digest :+ (pinned by digest) } "
# Pin the new reference into the manifest, then apply. sed edits the single
# "image" line in place (atomic tmp + move).
2026-08-03 09:02:45 +00:00
_tmp = " $MANIFEST .new"
sed "s#\"image\"[[:space:]]*:[[:space:]]*\"[^\"]*\"#\"image\": \" $( json_str " $_new " ) \"#" " $MANIFEST " > " $_tmp " \
|| die "could not rewrite the image in $MANIFEST "
chmod 0600 " $_tmp " ; mv " $_tmp " " $MANIFEST "
else
echo "==> already on $_current — re-pulling and recreating"
fi
2026-09-17 00:02:06 +00:00
GPUK_MANIFEST_PATH = " $MANIFEST " " $BIN_DEST " apply
2026-08-03 09:02:45 +00:00
}
2026-09-17 00:02:06 +00:00
# ── uninstall ──────────────────────────────────────────────────────────────────
# Plain: stop and remove the service, touch nothing else (a pause, reversible by
# `gpuk install`). --purge: everything the installer created goes — the app
# container (and a leftover -old twin), $ETC_DIR with the manifest and the
# enrolled identity, the binary — EXCEPT the data root: database, models and the
# secrets (ENCRYPTION_KEY above all, OPS-04) are the operator's to delete, by
# hand, knowingly. After a purge the next install is a first install (INS-03);
# without it, install.sh sees the manifest and treats the re-run as an update.
2026-08-03 09:02:45 +00:00
cmd_uninstall() {
need_root
2026-09-17 00:02:06 +00:00
_purge = 0
while [ $# -gt 0 ] ; do
case " $1 " in
--purge) _purge = 1; shift ;;
*) die "unknown uninstall option: $1 (expected: --purge)" ;;
esac
done
_cn = $( container_name)
_root = $( manifest_data_root)
2026-08-03 09:02:45 +00:00
systemctl disable --now " $SERVICE_NAME " 2>/dev/null || true
rm -f " $UNIT_DEST " ; systemctl daemon-reload 2>/dev/null || true
2026-09-17 00:02:06 +00:00
echo "==> removed the systemd service"
if [ " $_purge " -eq 0 ] ; then
echo "Left in place: $BIN_DEST , $ETC_DIR (manifest, identity), the data root and any"
echo "app container — 'gpuk install' brings the service back on them."
echo "For a clean slate (everything but the data root): gpuk uninstall --purge"
return 0
fi
if [ -n " $_cn " ] && command -v docker >/dev/null 2>& 1; then
for _c in " $_cn " " $_cn -old" ; do
docker rm -f " $_c " >/dev/null 2>& 1 && echo "==> removed container $_c "
done
fi
for _d in " $ETC_DIR " " $IDENTITY_DIR " ; do
case " $_d " in "" | /| /etc| /usr| /var) die "refusing to remove $_d " ;; esac
[ ! -e " $_d " ] || { rm -rf " $_d " ; echo "==> removed $_d " ; }
done
[ ! -e " $MACHINE_ID_FILE " ] || rm -f " $MACHINE_ID_FILE "
[ ! -e " $BIN_DEST " ] || { rm -f " $BIN_DEST " ; echo "==> removed $BIN_DEST " ; }
echo
echo "Kept: the data root ${ _root :+ $_root } — database, models and secrets (ENCRYPTION_KEY)."
echo "A new install over it reuses them. To delete it too, knowingly: rm -rf ${ _root :- <data-root> } "
2026-08-03 09:02:45 +00:00
}
usage() {
cat <<EOF
gpuk — GPU Kitchen host daemon (gpu-kitchen-worker)
2026-09-13 21:53:54 +00:00
gpuk install --mode controller --image <ref> [--profile homelab|studio|enterprise|public]
2026-09-17 00:02:06 +00:00
[--cluster N] [--cache-dir P] [--data-root P]
[--http-port P] [--mtls-port P] [--inference-port P]
[--network host|bridge|<net>] [--binary <path>]
2026-08-03 09:02:45 +00:00
gpuk install --mode worker --controller wss://<host>:<port> --enroll-token gk_enroll_...
2026-09-13 21:53:54 +00:00
[--cluster N] [--cache-dir P] [--binary <path>]
2026-09-17 00:02:06 +00:00
[--health-port P] [--data-port P] [--transfer-port P] [--handover-port P]
2026-09-13 21:53:54 +00:00
gpuk install ... --dry-run Validate inputs and print the manifest without changing the host
2026-08-03 09:02:45 +00:00
gpuk status Service state, enrollment, /health, app container status
gpuk enroll --controller wss://<host>:<port> --token gk_enroll_...
gpuk apply (controller) Reconcile the app container from the manifest
gpuk update (controller) Move to the current release (pull + recreate, rollback)
gpuk update --check (controller) Compare the installed version with the release
gpuk update --version <tag> (controller) Move to a specific release
2026-09-13 21:53:54 +00:00
gpuk channel Fetch + VERIFY the release channel and print what it offers
gpuk backup [DIR] (controller) Cold snapshot of the embedded pgdata (stop → tar → restart)
2026-08-03 09:02:45 +00:00
gpuk manifest Print the current manifest
gpuk logs Follow the app container logs (controller) or the daemon journal
2026-09-17 00:02:06 +00:00
gpuk uninstall Remove the systemd service, leave everything else in place
gpuk uninstall --purge Also remove the app container, /etc/gpu-kitchen and the binary
(never the data root: database, models, secrets)
2026-08-03 09:02:45 +00:00
Most people never run this directly: the channel's install.sh installs it
2026-08-03 09:19:52 +00:00
(https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh).
2026-08-03 09:02:45 +00:00
EOF
}
# ── dispatch ────────────────────────────────────────────────────────────────────
cmd = " ${ 1 :- help } " ; if [ $# -gt 0 ] ; then shift; fi
case " $cmd " in
install) cmd_install " $@ " ;;
status) cmd_status ;;
enroll) cmd_enroll " $@ " ;;
apply) cmd_apply ;;
update) cmd_update " $@ " ;;
2026-09-13 21:53:54 +00:00
channel) cmd_channel ;;
backup) cmd_backup " ${ 1 :- } " ;;
2026-08-03 09:02:45 +00:00
manifest) cat " $MANIFEST " ;;
logs)
_img = $( manifest_image)
if [ -n " $_img " ] ; then exec docker logs -f " $( container_name) " ; else exec journalctl -u " $SERVICE_NAME " -f; fi
;;
2026-09-17 00:02:06 +00:00
uninstall) cmd_uninstall " $@ " ;;
2026-08-03 09:02:45 +00:00
help| -h| --help) usage ;;
*) usage; exit 1 ;;
esac