release v0.1.8

This commit is contained in:
gpuk-release
2026-09-18 00:11:58 +00:00
parent 983733003b
commit a9df4fbc1f
4 changed files with 168 additions and 26 deletions
+109 -13
View File
@@ -92,7 +92,9 @@ install_binary() { # [local-path]
command -v curl >/dev/null || die "curl is required to download the binary"
_url="$GPUK_RELEASE_BASE/$(arch_asset)"
echo "==> downloading $_url"
curl -fsSL "$_url" -o "$BIN_DEST.new"
# Not -s: the binary is tens of MB and a silent download reads as a hang.
curl -fL --progress-bar "$_url" -o "$BIN_DEST.new" \
|| { rm -f "$BIN_DEST.new"; die "cannot download $_url"; }
chmod 0755 "$BIN_DEST.new"
mv "$BIN_DEST.new" "$BIN_DEST"
elif [ -x "$BIN_DEST" ]; then
@@ -170,6 +172,22 @@ port_owner() { # $1 = ss|netstat, $2 = port
fi
}
# This host's LAN IPv4 — what a browser, a worker or a container on the docker
# bridge dials to reach the machine. The route to a public address names the
# source the default route uses (never a docker or libvirt bridge); failing that,
# the first global address on a physical-looking interface; failing that, the
# resolver's word. Empty when nothing answers — the caller says so. Same
# derivation as dev.sh best_host_lan_ipv4 and install.sh's banner.
host_lan_ipv4() {
_ip=$(ip route get 1.1.1.1 2>/dev/null | sed -n 's/.* src \([0-9.]*\).*/\1/p' | head -1)
if [ -z "$_ip" ]; then
_ip=$(ip -o -4 addr show scope global 2>/dev/null \
| awk '$2 !~ /^(virbr|docker|br-|veth|mpqemubr|tun|tap|vnet)/ {sub(/\/.*/, "", $4); print $4; exit}')
fi
[ -n "$_ip" ] || _ip=$(hostname -I 2>/dev/null | awk '{print $1}')
printf '%s' "$_ip"
}
seed_secrets() { # DATA_ROOT
_sd="$1/secrets"
mkdir -p "$_sd"; chmod 0700 "$_sd"
@@ -307,7 +325,12 @@ hint_existing_caches() {
cmd_install() {
IMAGE=""; MODE="worker"; CLUSTER="default"; DATA_ROOT="/var/lib/gpu-kitchen"
CACHE_DIR=""; NETWORK="host"; CONTROLLER_URL=""; BIN_SRC=""; HEALTH_PORT="8001"
# bridge, not host (INS-49): on the host network every listener the image opens
# is a host-wide claim, and a box that already runs something on 3000 or 8000
# killed Nitro. Bridged, only the three published ports touch the host; what
# that costs — the host daemon must be told its LAN address and the engines'
# docker network — is written to worker.env below. `--network host` remains.
CACHE_DIR=""; NETWORK="bridge"; CONTROLLER_URL=""; BIN_SRC=""; HEALTH_PORT="8001"
# 1337, not 8080: kept in lockstep with install.sh's default (INS-01). The
# worker channel and the inference endpoint move the same way (INS-46).
ENROLL_TOKEN=""; HTTP_PORT="1337"; MTLS_PORT="8443"; INFERENCE_PORT="8200"
@@ -315,10 +338,11 @@ cmd_install() {
# GPUK_MODEL_TRANSFER_PORT and GPUK_HANDOVER_DATA_PORT from worker.env and
# announces the first two to the controller, so moving them here is complete.
DATA_PORT="8300"; TRANSFER_PORT="8301"; HANDOVER_PORT="8302"
PROFILE=""; DOMAIN=""; DRY_RUN=0
PROFILE=""; DOMAIN=""; DRY_RUN=0; RESET_MANIFEST=0
while [ $# -gt 0 ]; do
case "$1" in
--image) IMAGE="$2"; shift 2 ;;
--reset-manifest) RESET_MANIFEST=1; shift ;;
--mode) MODE="$2"; shift 2 ;;
--cluster) CLUSTER="$2"; shift 2 ;;
--profile) PROFILE="$2"; shift 2 ;;
@@ -514,6 +538,23 @@ cmd_install() {
write_controller_manifest
# The host daemon and app container share DATA_ROOT. Only controller-mode
# workerd gets this private bootstrap channel; remote workers stay tokenless.
#
# A bridged app container (INS-49) needs two more lines, read by the daemon
# from its OWN env — the daemon runs on the host, not in the container:
# - GPUK_DATA_HOST: the daemon dials the published mTLS port through docker
# NAT, so the controller sees it arrive from the bridge gateway (172.17.0.1)
# and would persist THAT as the node's address (WRK-93). The LAN address is
# what peers, the proxy and the UI must dial instead.
# - BACKEND_DOCKER_NETWORK: the engines the daemon launches join the app
# container's docker network, so the backend and gpuk-proxy reach them by
# container IP (apps/worker/src/modules/docker.rs, staging/engine.rs).
# Exactly what dev.sh hands the dev worker; host networking needs neither.
_data_host=""
if [ "$NETWORK" != "host" ]; then
_data_host="${GPUK_DATA_HOST:-$(host_lan_ipv4)}"
[ -n "$_data_host" ] || echo "==> warning: this host's LAN IPv4 could not be determined (no ip, no hostname -I)." \
"Set GPUK_DATA_HOST=<lan ip> in $WORKER_ENV, or the controller will address its own worker through the docker bridge."
fi
{
echo "GPUK_CLUSTER=$CLUSTER"
echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT"
@@ -521,9 +562,13 @@ cmd_install() {
echo "GPUK_CONTROLLER_URLS=wss://127.0.0.1:$MTLS_PORT"
echo "GPUK_SELF_ENROLL_FILE=$SELF_ENROLL_FILE"
echo "NODE_DISPLAY_NAME=$(hostname)"
if [ "$NETWORK" != "host" ]; then
[ -z "$_data_host" ] || echo "GPUK_DATA_HOST=$_data_host"
echo "BACKEND_DOCKER_NETWORK=$NETWORK"
fi
} > "$WORKER_ENV"
chmod 0600 "$WORKER_ENV"
echo "==> wrote $MANIFEST (controller: app container $IMAGE)"
echo "==> wrote $MANIFEST (controller: app container $IMAGE, network $NETWORK)"
fi
chmod 0600 "$MANIFEST"
[ -z "$DOMAIN" ] || write_caddyfile
@@ -561,11 +606,25 @@ cmd_install() {
# ── Controller: start the daemon, then bring up the app container ──
systemctl enable --now "$SERVICE_NAME"
echo "==> $SERVICE_NAME enabled and started"
echo "==> applying manifest (first app-container start) ..."
# The pull is the long part of a first install (a multi-GB image) and workerd
# runs docker with its output captured, so pulling from inside `apply` is
# minutes of silence that read as a hang. Pull here, on the terminal, where
# docker's own per-layer progress is what the operator sees; `apply` then finds
# the image present and skips its pull (INS-03).
# Feedback only, never the verdict: `apply` pulls again whatever happened here
# and reports the cause itself (the CI shell gate installs a stub daemon against
# an image that does not exist anywhere — the daemon's pull is the one that counts).
if ! docker image inspect "$IMAGE" >/dev/null 2>&1; then
echo "==> pulling $IMAGE (docker shows the progress per layer) ..."
docker pull "$IMAGE" \
|| echo "==> the pull did not complete here; the daemon retries it during apply and reports the cause if it fails again"
fi
echo "==> applying manifest — starting the app container and waiting for its health check (up to 60s) ..."
# No unix socket any more: workerd reconciles the app container in-process from
# the on-disk manifest (there is no backend to relay through on the very first
# boot). Steady-state updates go through the daemon over WS.
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" apply || die "apply failed. Check: gpuk logs"
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" apply \
|| die "the app container did not come up — the [worker] lines above say why (a [controller] FATAL line names the component and the fix). Full container logs: gpuk logs"
echo
echo "Done. The worker daemon is running and the app container is up."
echo " Status : gpuk status App logs: gpuk logs Daemon: journalctl -u $SERVICE_NAME"
@@ -605,12 +664,31 @@ INSTALL_ENV_KEYS="NODE_ENV,GPUK_MODE,GPUK_CLUSTER,GPUK_SELF_ENROLL_FILE,NODE_DIS
# cluster, network, data root, ports, secret references, the primary cache disk,
# the env keys above) and keeps the rest. The first write is the render as-is.
write_manifest() { # $1 = render function
if [ -f "$MANIFEST" ]; then
if [ -f "$MANIFEST" ] && [ "$RESET_MANIFEST" -eq 1 ]; then
# The operator's explicit regeneration (WRK-191): the daemon archives the
# existing document — readable or not — writes the render as-is and NAMES what
# the archive carried that the render does not. The flags install.sh inherited
# from the old file (ports, profile, data root) are already in the render.
"$1" > "$MANIFEST.new"
chmod 0600 "$MANIFEST.new"
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" install-manifest \
--from "$MANIFEST.new" --reset >/dev/null \
|| { rm -f "$MANIFEST.new"; die "could not reset $MANIFEST — the [worker] line above names the cause"; }
rm -f "$MANIFEST.new"
echo "==> $MANIFEST rebuilt from this install's settings (--reset-manifest); the previous file is archived beside it"
elif [ -f "$MANIFEST" ]; then
"$1" > "$MANIFEST.new"
chmod 0600 "$MANIFEST.new"
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" install-manifest \
--from "$MANIFEST.new" --own-env "$INSTALL_ENV_KEYS" >/dev/null \
|| { rm -f "$MANIFEST.new"; die "could not merge the new settings into $MANIFEST"; }
|| {
rm -f "$MANIFEST.new"
die "could not merge the new settings into $MANIFEST — the [worker] line above names the cause.
A manifest written by an older release can carry a field this release removed. Re-run the same
command with --reset-manifest: the file is archived beside itself as manifest.json.before-reset.<time>,
rebuilt from this install's settings, and every setting the archive carried that the rebuild does
not (extra cache disks, LED binary, eviction…) is listed so you can set it again from the UI."
}
rm -f "$MANIFEST.new"
echo "==> merged into the existing $MANIFEST (controller-owned settings kept)"
else
@@ -1017,7 +1095,7 @@ cmd_update() {
# ── uninstall ──────────────────────────────────────────────────────────────────
# Plain: stop and remove the service, touch nothing else (a pause, reversible by
# `gpuk install`). --purge: everything the installer created goes — the app
# container (and a leftover -old twin), $ETC_DIR with the manifest and the
# container (and its -old / -failed twins), $ETC_DIR with the manifest and the
# enrolled identity, the binary — EXCEPT the data root: database, models and the
# secrets (ENCRYPTION_KEY above all, OPS-04) are the operator's to delete, by
# hand, knowingly. After a purge the next install is a first install (INS-03);
@@ -1043,7 +1121,7 @@ cmd_uninstall() {
return 0
fi
if [ -n "$_cn" ] && command -v docker >/dev/null 2>&1; then
for _c in "$_cn" "$_cn-old"; do
for _c in "$_cn" "$_cn-old" "$_cn-failed"; do
docker rm -f "$_c" >/dev/null 2>&1 && echo "==> removed container $_c"
done
fi
@@ -1065,7 +1143,10 @@ gpuk — GPU Kitchen host daemon (gpu-kitchen-worker)
gpuk install --mode controller --image <ref> [--profile homelab|studio|enterprise|public]
[--cluster N] [--cache-dir P] [--data-root P]
[--http-port P] [--mtls-port P] [--inference-port P]
[--network host|bridge|<net>] [--binary <path>]
[--network bridge|host|<net>] [--binary <path>] [--reset-manifest]
--network: bridge by default — the container publishes its three
ports and nothing else it listens on touches the host; host makes
every listener a host-wide claim (a re-run keeps the mode installed)
gpuk install --mode worker --controller wss://<host>:<port> --enroll-token gk_enroll_...
[--cluster N] [--cache-dir P] [--binary <path>]
[--health-port P] [--data-port P] [--transfer-port P] [--handover-port P]
@@ -1079,7 +1160,8 @@ gpuk — GPU Kitchen host daemon (gpu-kitchen-worker)
gpuk channel Fetch + VERIFY the release channel and print what it offers
gpuk backup [DIR] (controller) Cold snapshot of the embedded pgdata (stop → tar → restart)
gpuk manifest Print the current manifest
gpuk logs Follow the app container logs (controller) or the daemon journal
gpuk logs Follow the app container logs (controller) or the daemon journal;
after a failed start, the kept <container>-failed log
gpuk uninstall Remove the systemd service, leave everything else in place
gpuk uninstall --purge Also remove the app container, /etc/gpu-kitchen and the binary
(never the data root: database, models, secrets)
@@ -1102,7 +1184,21 @@ case "$cmd" in
manifest) cat "$MANIFEST" ;;
logs)
_img=$(manifest_image)
if [ -n "$_img" ]; then exec docker logs -f "$(container_name)"; else exec journalctl -u "$SERVICE_NAME" -f; fi
if [ -n "$_img" ]; then
_cn=$(container_name)
if docker inspect "$_cn" >/dev/null 2>&1; then
exec docker logs -f "$_cn"
elif docker inspect "$_cn-failed" >/dev/null 2>&1; then
# No live app container, but the last failed start was kept (INS-15):
# its whole log is the diagnosis, not the daemon journal.
echo "==> no running app container; showing the full log of the last failed start ($_cn-failed)" >&2
exec docker logs "$_cn-failed"
else
exec journalctl -u "$SERVICE_NAME" -f
fi
else
exec journalctl -u "$SERVICE_NAME" -f
fi
;;
uninstall) cmd_uninstall "$@" ;;
help|-h|--help) usage ;;