release v0.1.8
This commit is contained in:
@@ -92,7 +92,9 @@ install_binary() { # [local-path]
|
||||
command -v curl >/dev/null || die "curl is required to download the binary"
|
||||
_url="$GPUK_RELEASE_BASE/$(arch_asset)"
|
||||
echo "==> downloading $_url"
|
||||
curl -fsSL "$_url" -o "$BIN_DEST.new"
|
||||
# Not -s: the binary is tens of MB and a silent download reads as a hang.
|
||||
curl -fL --progress-bar "$_url" -o "$BIN_DEST.new" \
|
||||
|| { rm -f "$BIN_DEST.new"; die "cannot download $_url"; }
|
||||
chmod 0755 "$BIN_DEST.new"
|
||||
mv "$BIN_DEST.new" "$BIN_DEST"
|
||||
elif [ -x "$BIN_DEST" ]; then
|
||||
@@ -170,6 +172,22 @@ port_owner() { # $1 = ss|netstat, $2 = port
|
||||
fi
|
||||
}
|
||||
|
||||
# This host's LAN IPv4 — what a browser, a worker or a container on the docker
|
||||
# bridge dials to reach the machine. The route to a public address names the
|
||||
# source the default route uses (never a docker or libvirt bridge); failing that,
|
||||
# the first global address on a physical-looking interface; failing that, the
|
||||
# resolver's word. Empty when nothing answers — the caller says so. Same
|
||||
# derivation as dev.sh best_host_lan_ipv4 and install.sh's banner.
|
||||
host_lan_ipv4() {
|
||||
_ip=$(ip route get 1.1.1.1 2>/dev/null | sed -n 's/.* src \([0-9.]*\).*/\1/p' | head -1)
|
||||
if [ -z "$_ip" ]; then
|
||||
_ip=$(ip -o -4 addr show scope global 2>/dev/null \
|
||||
| awk '$2 !~ /^(virbr|docker|br-|veth|mpqemubr|tun|tap|vnet)/ {sub(/\/.*/, "", $4); print $4; exit}')
|
||||
fi
|
||||
[ -n "$_ip" ] || _ip=$(hostname -I 2>/dev/null | awk '{print $1}')
|
||||
printf '%s' "$_ip"
|
||||
}
|
||||
|
||||
seed_secrets() { # DATA_ROOT
|
||||
_sd="$1/secrets"
|
||||
mkdir -p "$_sd"; chmod 0700 "$_sd"
|
||||
@@ -307,7 +325,12 @@ hint_existing_caches() {
|
||||
|
||||
cmd_install() {
|
||||
IMAGE=""; MODE="worker"; CLUSTER="default"; DATA_ROOT="/var/lib/gpu-kitchen"
|
||||
CACHE_DIR=""; NETWORK="host"; CONTROLLER_URL=""; BIN_SRC=""; HEALTH_PORT="8001"
|
||||
# bridge, not host (INS-49): on the host network every listener the image opens
|
||||
# is a host-wide claim, and a box that already runs something on 3000 or 8000
|
||||
# killed Nitro. Bridged, only the three published ports touch the host; what
|
||||
# that costs — the host daemon must be told its LAN address and the engines'
|
||||
# docker network — is written to worker.env below. `--network host` remains.
|
||||
CACHE_DIR=""; NETWORK="bridge"; CONTROLLER_URL=""; BIN_SRC=""; HEALTH_PORT="8001"
|
||||
# 1337, not 8080: kept in lockstep with install.sh's default (INS-01). The
|
||||
# worker channel and the inference endpoint move the same way (INS-46).
|
||||
ENROLL_TOKEN=""; HTTP_PORT="1337"; MTLS_PORT="8443"; INFERENCE_PORT="8200"
|
||||
@@ -315,10 +338,11 @@ cmd_install() {
|
||||
# GPUK_MODEL_TRANSFER_PORT and GPUK_HANDOVER_DATA_PORT from worker.env and
|
||||
# announces the first two to the controller, so moving them here is complete.
|
||||
DATA_PORT="8300"; TRANSFER_PORT="8301"; HANDOVER_PORT="8302"
|
||||
PROFILE=""; DOMAIN=""; DRY_RUN=0
|
||||
PROFILE=""; DOMAIN=""; DRY_RUN=0; RESET_MANIFEST=0
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--image) IMAGE="$2"; shift 2 ;;
|
||||
--reset-manifest) RESET_MANIFEST=1; shift ;;
|
||||
--mode) MODE="$2"; shift 2 ;;
|
||||
--cluster) CLUSTER="$2"; shift 2 ;;
|
||||
--profile) PROFILE="$2"; shift 2 ;;
|
||||
@@ -514,6 +538,23 @@ cmd_install() {
|
||||
write_controller_manifest
|
||||
# The host daemon and app container share DATA_ROOT. Only controller-mode
|
||||
# workerd gets this private bootstrap channel; remote workers stay tokenless.
|
||||
#
|
||||
# A bridged app container (INS-49) needs two more lines, read by the daemon
|
||||
# from its OWN env — the daemon runs on the host, not in the container:
|
||||
# - GPUK_DATA_HOST: the daemon dials the published mTLS port through docker
|
||||
# NAT, so the controller sees it arrive from the bridge gateway (172.17.0.1)
|
||||
# and would persist THAT as the node's address (WRK-93). The LAN address is
|
||||
# what peers, the proxy and the UI must dial instead.
|
||||
# - BACKEND_DOCKER_NETWORK: the engines the daemon launches join the app
|
||||
# container's docker network, so the backend and gpuk-proxy reach them by
|
||||
# container IP (apps/worker/src/modules/docker.rs, staging/engine.rs).
|
||||
# Exactly what dev.sh hands the dev worker; host networking needs neither.
|
||||
_data_host=""
|
||||
if [ "$NETWORK" != "host" ]; then
|
||||
_data_host="${GPUK_DATA_HOST:-$(host_lan_ipv4)}"
|
||||
[ -n "$_data_host" ] || echo "==> warning: this host's LAN IPv4 could not be determined (no ip, no hostname -I)." \
|
||||
"Set GPUK_DATA_HOST=<lan ip> in $WORKER_ENV, or the controller will address its own worker through the docker bridge."
|
||||
fi
|
||||
{
|
||||
echo "GPUK_CLUSTER=$CLUSTER"
|
||||
echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT"
|
||||
@@ -521,9 +562,13 @@ cmd_install() {
|
||||
echo "GPUK_CONTROLLER_URLS=wss://127.0.0.1:$MTLS_PORT"
|
||||
echo "GPUK_SELF_ENROLL_FILE=$SELF_ENROLL_FILE"
|
||||
echo "NODE_DISPLAY_NAME=$(hostname)"
|
||||
if [ "$NETWORK" != "host" ]; then
|
||||
[ -z "$_data_host" ] || echo "GPUK_DATA_HOST=$_data_host"
|
||||
echo "BACKEND_DOCKER_NETWORK=$NETWORK"
|
||||
fi
|
||||
} > "$WORKER_ENV"
|
||||
chmod 0600 "$WORKER_ENV"
|
||||
echo "==> wrote $MANIFEST (controller: app container $IMAGE)"
|
||||
echo "==> wrote $MANIFEST (controller: app container $IMAGE, network $NETWORK)"
|
||||
fi
|
||||
chmod 0600 "$MANIFEST"
|
||||
[ -z "$DOMAIN" ] || write_caddyfile
|
||||
@@ -561,11 +606,25 @@ cmd_install() {
|
||||
# ── Controller: start the daemon, then bring up the app container ──
|
||||
systemctl enable --now "$SERVICE_NAME"
|
||||
echo "==> $SERVICE_NAME enabled and started"
|
||||
echo "==> applying manifest (first app-container start) ..."
|
||||
# The pull is the long part of a first install (a multi-GB image) and workerd
|
||||
# runs docker with its output captured, so pulling from inside `apply` is
|
||||
# minutes of silence that read as a hang. Pull here, on the terminal, where
|
||||
# docker's own per-layer progress is what the operator sees; `apply` then finds
|
||||
# the image present and skips its pull (INS-03).
|
||||
# Feedback only, never the verdict: `apply` pulls again whatever happened here
|
||||
# and reports the cause itself (the CI shell gate installs a stub daemon against
|
||||
# an image that does not exist anywhere — the daemon's pull is the one that counts).
|
||||
if ! docker image inspect "$IMAGE" >/dev/null 2>&1; then
|
||||
echo "==> pulling $IMAGE (docker shows the progress per layer) ..."
|
||||
docker pull "$IMAGE" \
|
||||
|| echo "==> the pull did not complete here; the daemon retries it during apply and reports the cause if it fails again"
|
||||
fi
|
||||
echo "==> applying manifest — starting the app container and waiting for its health check (up to 60s) ..."
|
||||
# No unix socket any more: workerd reconciles the app container in-process from
|
||||
# the on-disk manifest (there is no backend to relay through on the very first
|
||||
# boot). Steady-state updates go through the daemon over WS.
|
||||
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" apply || die "apply failed. Check: gpuk logs"
|
||||
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" apply \
|
||||
|| die "the app container did not come up — the [worker] lines above say why (a [controller] FATAL line names the component and the fix). Full container logs: gpuk logs"
|
||||
echo
|
||||
echo "Done. The worker daemon is running and the app container is up."
|
||||
echo " Status : gpuk status App logs: gpuk logs Daemon: journalctl -u $SERVICE_NAME"
|
||||
@@ -605,12 +664,31 @@ INSTALL_ENV_KEYS="NODE_ENV,GPUK_MODE,GPUK_CLUSTER,GPUK_SELF_ENROLL_FILE,NODE_DIS
|
||||
# cluster, network, data root, ports, secret references, the primary cache disk,
|
||||
# the env keys above) and keeps the rest. The first write is the render as-is.
|
||||
write_manifest() { # $1 = render function
|
||||
if [ -f "$MANIFEST" ]; then
|
||||
if [ -f "$MANIFEST" ] && [ "$RESET_MANIFEST" -eq 1 ]; then
|
||||
# The operator's explicit regeneration (WRK-191): the daemon archives the
|
||||
# existing document — readable or not — writes the render as-is and NAMES what
|
||||
# the archive carried that the render does not. The flags install.sh inherited
|
||||
# from the old file (ports, profile, data root) are already in the render.
|
||||
"$1" > "$MANIFEST.new"
|
||||
chmod 0600 "$MANIFEST.new"
|
||||
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" install-manifest \
|
||||
--from "$MANIFEST.new" --reset >/dev/null \
|
||||
|| { rm -f "$MANIFEST.new"; die "could not reset $MANIFEST — the [worker] line above names the cause"; }
|
||||
rm -f "$MANIFEST.new"
|
||||
echo "==> $MANIFEST rebuilt from this install's settings (--reset-manifest); the previous file is archived beside it"
|
||||
elif [ -f "$MANIFEST" ]; then
|
||||
"$1" > "$MANIFEST.new"
|
||||
chmod 0600 "$MANIFEST.new"
|
||||
GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" install-manifest \
|
||||
--from "$MANIFEST.new" --own-env "$INSTALL_ENV_KEYS" >/dev/null \
|
||||
|| { rm -f "$MANIFEST.new"; die "could not merge the new settings into $MANIFEST"; }
|
||||
|| {
|
||||
rm -f "$MANIFEST.new"
|
||||
die "could not merge the new settings into $MANIFEST — the [worker] line above names the cause.
|
||||
A manifest written by an older release can carry a field this release removed. Re-run the same
|
||||
command with --reset-manifest: the file is archived beside itself as manifest.json.before-reset.<time>,
|
||||
rebuilt from this install's settings, and every setting the archive carried that the rebuild does
|
||||
not (extra cache disks, LED binary, eviction…) is listed so you can set it again from the UI."
|
||||
}
|
||||
rm -f "$MANIFEST.new"
|
||||
echo "==> merged into the existing $MANIFEST (controller-owned settings kept)"
|
||||
else
|
||||
@@ -1017,7 +1095,7 @@ cmd_update() {
|
||||
# ── uninstall ──────────────────────────────────────────────────────────────────
|
||||
# Plain: stop and remove the service, touch nothing else (a pause, reversible by
|
||||
# `gpuk install`). --purge: everything the installer created goes — the app
|
||||
# container (and a leftover -old twin), $ETC_DIR with the manifest and the
|
||||
# container (and its -old / -failed twins), $ETC_DIR with the manifest and the
|
||||
# enrolled identity, the binary — EXCEPT the data root: database, models and the
|
||||
# secrets (ENCRYPTION_KEY above all, OPS-04) are the operator's to delete, by
|
||||
# hand, knowingly. After a purge the next install is a first install (INS-03);
|
||||
@@ -1043,7 +1121,7 @@ cmd_uninstall() {
|
||||
return 0
|
||||
fi
|
||||
if [ -n "$_cn" ] && command -v docker >/dev/null 2>&1; then
|
||||
for _c in "$_cn" "$_cn-old"; do
|
||||
for _c in "$_cn" "$_cn-old" "$_cn-failed"; do
|
||||
docker rm -f "$_c" >/dev/null 2>&1 && echo "==> removed container $_c"
|
||||
done
|
||||
fi
|
||||
@@ -1065,7 +1143,10 @@ gpuk — GPU Kitchen host daemon (gpu-kitchen-worker)
|
||||
gpuk install --mode controller --image <ref> [--profile homelab|studio|enterprise|public]
|
||||
[--cluster N] [--cache-dir P] [--data-root P]
|
||||
[--http-port P] [--mtls-port P] [--inference-port P]
|
||||
[--network host|bridge|<net>] [--binary <path>]
|
||||
[--network bridge|host|<net>] [--binary <path>] [--reset-manifest]
|
||||
--network: bridge by default — the container publishes its three
|
||||
ports and nothing else it listens on touches the host; host makes
|
||||
every listener a host-wide claim (a re-run keeps the mode installed)
|
||||
gpuk install --mode worker --controller wss://<host>:<port> --enroll-token gk_enroll_...
|
||||
[--cluster N] [--cache-dir P] [--binary <path>]
|
||||
[--health-port P] [--data-port P] [--transfer-port P] [--handover-port P]
|
||||
@@ -1079,7 +1160,8 @@ gpuk — GPU Kitchen host daemon (gpu-kitchen-worker)
|
||||
gpuk channel Fetch + VERIFY the release channel and print what it offers
|
||||
gpuk backup [DIR] (controller) Cold snapshot of the embedded pgdata (stop → tar → restart)
|
||||
gpuk manifest Print the current manifest
|
||||
gpuk logs Follow the app container logs (controller) or the daemon journal
|
||||
gpuk logs Follow the app container logs (controller) or the daemon journal;
|
||||
after a failed start, the kept <container>-failed log
|
||||
gpuk uninstall Remove the systemd service, leave everything else in place
|
||||
gpuk uninstall --purge Also remove the app container, /etc/gpu-kitchen and the binary
|
||||
(never the data root: database, models, secrets)
|
||||
@@ -1102,7 +1184,21 @@ case "$cmd" in
|
||||
manifest) cat "$MANIFEST" ;;
|
||||
logs)
|
||||
_img=$(manifest_image)
|
||||
if [ -n "$_img" ]; then exec docker logs -f "$(container_name)"; else exec journalctl -u "$SERVICE_NAME" -f; fi
|
||||
if [ -n "$_img" ]; then
|
||||
_cn=$(container_name)
|
||||
if docker inspect "$_cn" >/dev/null 2>&1; then
|
||||
exec docker logs -f "$_cn"
|
||||
elif docker inspect "$_cn-failed" >/dev/null 2>&1; then
|
||||
# No live app container, but the last failed start was kept (INS-15):
|
||||
# its whole log is the diagnosis, not the daemon journal.
|
||||
echo "==> no running app container; showing the full log of the last failed start ($_cn-failed)" >&2
|
||||
exec docker logs "$_cn-failed"
|
||||
else
|
||||
exec journalctl -u "$SERVICE_NAME" -f
|
||||
fi
|
||||
else
|
||||
exec journalctl -u "$SERVICE_NAME" -f
|
||||
fi
|
||||
;;
|
||||
uninstall) cmd_uninstall "$@" ;;
|
||||
help|-h|--help) usage ;;
|
||||
|
||||
Reference in New Issue
Block a user