#!/bin/sh # gpuk — installer + control CLI for the GPU Kitchen host daemon (gpu-kitchen-worker). # # Since C1 a compute node runs NO backend. The single privileged host component is # the Rust `gpu-kitchen-worker` daemon (workerd), running as root DIRECTLY on the # host (not in a container). It owns NVML clock/power locks, whitelisted host # browsing, the model-cache staging pipeline, the signed `gpu-kitchen-bench` # runner, and — on a controller node — the app container's lifecycle (create, # health-gate, roll back, pull+recreate to update) via its manifest. There is no # separate host daemon and no unix control socket: workerd is driven over the # wss+mTLS channel by the controller, and locally by this CLI. # # Two roles: # worker a headless compute node. Installs the binary + systemd unit + a # manifest (cacheDisks) with NO app container, NO # Postgres, NO docker app image. It ENROLLS over mTLS with a # enrollment token, or waits for LAN discovery admission, then runs. # controller the full app + UI. Installs the binary + systemd unit + an # app-container manifest (image, ports, env, secretsRef) that workerd # applies (`gpu-kitchen-worker apply`, then cmd:host_apply in steady # state). # # Install (root): # # controller: # curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/gpuk \ # | sudo sh -s -- install --mode controller \ # --image repo.byterain.io/gpukitchen/gpukitchen-controller:vX.Y.Z # # worker (enroll against a controller with a single-use token from its UI): # sudo ./deployments/install/gpuk install --mode worker \ # --controller wss://controller.lan:8443 --enroll-token gk_enroll_... \ # --cache-dir /mnt/models --binary apps/worker/target/release/gpu-kitchen-worker # # Control: # gpuk status | apply | update | enroll | logs | manifest | uninstall # set -eu # ── Paths ──────────────────────────────────────────────────────────────────── # Overridable, so the daemon can be driven against a prefix a normal user owns — # the only way any of this is testable without handing a test suite root on the # host. Unset (the real install) they are exactly the systemd defaults workerd uses # (apps/worker/src/module.rs, apps/worker/src/identity.rs). BIN_DEST="${GPUK_BIN_DEST:-/usr/local/bin/gpu-kitchen-worker}" # This CLI, installed next to the daemon by `gpuk install`: it carries the # release key pinned at install time, and install.sh hands a plain re-run to its # `update` (INS-03, OPS-20). CLI_DEST="${GPUK_CLI_DEST:-$(dirname "$BIN_DEST")/gpuk}" ETC_DIR="${GPUK_ETC_DIR:-/etc/gpu-kitchen}" MANIFEST="$ETC_DIR/manifest.json" IDENTITY_DIR="${GPUK_IDENTITY_DIR:-$ETC_DIR/identity}" MACHINE_ID_FILE="${GPUK_MACHINE_ID_FILE:-$ETC_DIR/machine-id}" WORKER_ENV="$ETC_DIR/worker.env" SERVICE_NAME="gpu-kitchen-worker" UNIT_DEST="${GPUK_UNIT_DEST:-/etc/systemd/system/$SERVICE_NAME.service}" # Where to download the binary from when no --binary is given. GPUK_RELEASE_BASE="${GPUK_RELEASE_BASE:-}" die() { echo "gpuk: $*" >&2; exit 1; } # Root, or able to do the job anyway. Fail with a clear message BEFORE touching # /etc, /usr/local/bin or systemd. Against a user-owned prefix it simply works. need_root() { [ "$(id -u)" -eq 0 ] && return 0 [ -w "$ETC_DIR" ] && return 0 die "this command must run as root (use sudo)" } # ── JSON helpers (controlled inputs: paths + identifiers) ────────────────────── json_str() { printf '%s' "$1" | sed 's/\\/\\\\/g; s/"/\\"/g'; } json_array() { # args → ["a","b",...] _out="" for _p in "$@"; do _e=$(json_str "$_p") if [ -z "$_out" ]; then _out="\"$_e\""; else _out="$_out,\"$_e\""; fi done printf '[%s]' "$_out" } # ── install ──────────────────────────────────────────────────────────────────── arch_asset() { case "$(uname -m)" in x86_64|amd64) echo "gpu-kitchen-worker-x86_64" ;; aarch64|arm64) echo "gpu-kitchen-worker-aarch64" ;; *) die "unsupported architecture: $(uname -m)" ;; esac } # [INS-19] A (re)install has just replaced the binary under a daemon that may be # running: `systemctl enable --now` is a no-op on an active unit and would leave the # OLD process in memory, speaking the old protocol to the new controller. Enable, then # restart — which also starts a daemon that was not running. start_daemon() { systemctl enable "$SERVICE_NAME" systemctl restart "$SERVICE_NAME" } install_binary() { # [local-path] if [ -n "${1:-}" ]; then [ -f "$1" ] || die "binary not found: $1" install -m 0755 "$1" "$BIN_DEST" echo "==> installed $BIN_DEST from $1" elif [ -n "$GPUK_RELEASE_BASE" ]; then command -v curl >/dev/null || die "curl is required to download the binary" _url="$GPUK_RELEASE_BASE/$(arch_asset)" echo "==> downloading $_url" # Not -s: the binary is tens of MB and a silent download reads as a hang. curl -fL --progress-bar "$_url" -o "$BIN_DEST.new" \ || { rm -f "$BIN_DEST.new"; die "cannot download $_url"; } # The root daemon is swapped in only once its minisign signature verifies # against the release key pinned in this gpuk (OPS-20, INS-05). if [ "${GPUK_CHANNEL_INSECURE:-}" = "1" ]; then echo "WARNING: GPUK_CHANNEL_INSECURE=1 — $_url NOT verified" >&2 else require_release_key curl -fsSL "$_url.minisig" -o "$BIN_DEST.new.minisig" \ || { rm -f "$BIN_DEST.new" "$BIN_DEST.new.minisig"; die "no signature at $_url.minisig — refusing an unsigned daemon binary"; } "$MINISIGN" -Vq -m "$BIN_DEST.new" -x "$BIN_DEST.new.minisig" -P "$CHANNEL_PUBKEY" >/dev/null 2>&1 \ || { rm -f "$BIN_DEST.new" "$BIN_DEST.new.minisig"; die "$_url: signature verification FAILED — refusing it (OPS-20)"; } rm -f "$BIN_DEST.new.minisig" echo "==> signature verified" fi chmod 0755 "$BIN_DEST.new" mv "$BIN_DEST.new" "$BIN_DEST" elif [ -x "$BIN_DEST" ]; then echo "==> reusing existing $BIN_DEST" else die "no binary: pass --binary or set GPUK_RELEASE_BASE=" fi } # Leave this very gpuk on the host, next to the daemon: `gpuk status`, `gpuk # update` and install.sh's plain re-run all use it. It is the copy that carries # the release key this install pinned (substituted at publish time), so later # updates are verified against the key the first install trusted. install_cli() { # Already this copy (gpuk re-run from its installed path, or identical bytes). if [ -e "$CLI_DEST" ] && cmp -s "$0" "$CLI_DEST"; then return 0; fi install -m 0755 "$0" "$CLI_DEST.new" && mv "$CLI_DEST.new" "$CLI_DEST" \ || die "cannot install the gpuk CLI at $CLI_DEST" echo "==> installed $CLI_DEST" } gen_secret() { head -c 32 /dev/urandom | base64 | tr '+/' '-_' | tr -d '='; } # The host ports a leftover app container is reached on ("ui mtls inference"), # read off the container itself with the precedence the controller applies # (core/published-ports.ts): host networking moves the listeners (GPUK_PORT, # GPUK_MTLS_PORT, GPUK_LISTEN_ADDR); anything else keeps the image's fixed # listeners and publishes them (GPUK_PUBLIC_*, then the port bindings). Empty # when there is no such container or no docker. leftover_container_ports() { # $1 = container name command -v docker >/dev/null 2>&1 || return 0 # Same format string as install.sh's container_facts — one reading, two scripts. _f=$(docker inspect -f '{{.State.Status}}|{{.HostConfig.NetworkMode}}|{{range $p, $b := .HostConfig.PortBindings}}{{range $b}}{{$p}}={{.HostPort}} {{end}}{{end}}{{"\n"}}{{range .Config.Env}}{{.}}{{"\n"}}{{end}}' "$1" 2>/dev/null) \ || return 0 [ -n "$_f" ] || return 0 _head=$(printf '%s\n' "$_f" | head -1) _r=${_head#*|}; _net=${_r%%|*}; _bind=${_r#*|} _env() { printf '%s\n' "$_f" | sed -n "s/^$1=//p" | head -1; } _pub() { printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n "s/^$1\/tcp=//p" | head -1; } if [ "$_net" = "host" ]; then _ui=$(_env GPUK_PORT); _mtls=$(_env GPUK_MTLS_PORT) _inf=$(_env GPUK_PROXY_PUBLIC_PORT) [ -n "$_inf" ] || { _la=$(_env GPUK_LISTEN_ADDR); _inf=${_la##*:}; } else _ui=$(_env GPUK_PUBLIC_PORT); [ -n "$_ui" ] || _ui=$(_pub 8080) _mtls=$(_env GPUK_PUBLIC_MTLS_PORT); [ -n "$_mtls" ] || _mtls=$(_pub 8443) _inf=$(_env GPUK_PROXY_PUBLIC_PORT); [ -n "$_inf" ] || _inf=$(_pub 8200) fi printf '%s %s %s' "${_ui:-8080}" "${_mtls:-8443}" "${_inf:-8200}" } # Who holds a port: "process", "process in container NAME", or "". The process's # cgroup names its container (host networking); a bridged publication is found # as the container's port mapping in `docker ps` (the host-side holder is # docker-proxy, which says nothing by itself). Same reading as install.sh. port_owner() { # $1 = ss|netstat, $2 = port _proc=""; _pid="" case "$1" in ss) _line=$(ss -ltnpH "sport = :$2" 2>/dev/null | head -1) _proc=$(printf '%s' "$_line" | sed -n 's/.*users:((\"\([^"]*\)\".*/\1/p') _pid=$(printf '%s' "$_line" | sed -n 's/.*pid=\([0-9]*\).*/\1/p') ;; netstat) _field=$(netstat -ltnp 2>/dev/null \ | awk -v p="$2" '{n=split($4,a,":"); if (a[n]==p) {print $NF; exit}}') case "$_field" in */*) _pid=${_field%%/*}; _proc=${_field#*/} ;; *) _proc="$_field" ;; esac ;; esac case "$_pid" in *[!0-9]*|"") _pid="" ;; esac _ctr="" if command -v docker >/dev/null 2>&1; then if [ -n "$_pid" ] && [ -r "/proc/$_pid/cgroup" ]; then _cid=$(sed -n 's#.*docker[-/]\([0-9a-f]\{64\}\).*#\1#p' "/proc/$_pid/cgroup" 2>/dev/null | head -1) [ -z "$_cid" ] || _ctr=$(docker inspect -f '{{.Name}}' "$_cid" 2>/dev/null | sed 's|^/||') fi [ -n "$_ctr" ] || _ctr=$(docker ps --format '{{.Names}} {{.Ports}}' 2>/dev/null \ | awk -v p=":$2->" 'index($0, p) { print $1; exit }') fi if [ -n "$_ctr" ]; then printf '%s' "${_proc:-a process} in container $_ctr" else printf '%s' "$_proc" fi } # This host's LAN IPv4 — what a browser, a worker or a container on the docker # bridge dials to reach the machine. The route to a public address names the # source the default route uses (never a docker or libvirt bridge); failing that, # the first global address on a physical-looking interface; failing that, the # resolver's word. Empty when nothing answers — the caller says so. Same # derivation as dev.sh best_host_lan_ipv4 and install.sh's banner. host_lan_ipv4() { _ip=$(ip route get 1.1.1.1 2>/dev/null | sed -n 's/.* src \([0-9.]*\).*/\1/p' | head -1) if [ -z "$_ip" ]; then _ip=$(ip -o -4 addr show scope global 2>/dev/null \ | awk '$2 !~ /^(virbr|docker|br-|veth|mpqemubr|tun|tap|vnet)/ {sub(/\/.*/, "", $4); print $4; exit}') fi [ -n "$_ip" ] || _ip=$(hostname -I 2>/dev/null | awk '{print $1}') printf '%s' "$_ip" } seed_secrets() { # DATA_ROOT _sd="$1/secrets" mkdir -p "$_sd"; chmod 0700 "$_sd" [ -f "$_sd/encryption_key" ] || { umask 077; gen_secret > "$_sd/encryption_key"; } [ -f "$_sd/node_id" ] || { umask 077; (cat /proc/sys/kernel/random/uuid 2>/dev/null || uuidgen) > "$_sd/node_id"; } # No nominal first-run password is generated. An operator may pre-provision # this documented break-glass file; only then is it injected into the app. [ ! -f "$_sd/bootstrap_admin_password" ] || chmod 0600 "$_sd/bootstrap_admin_password" } prepare_claim_code() { # The file is the operator's recoverable proof of machine possession. Create # it once, preserve it across reinstalls, and let the backend unlink it after # the atomic first-account claim. A missing file beside an existing manifest # therefore means "consumed", never "rotate the credential". homelab and studio # have no claim window at all (PRF-12, SEC-53): no code is made for them. case "$PROFILE" in homelab|studio) return 0 ;; esac if [ -f "$CLAIM_CODE_FILE" ]; then chmod 0600 "$CLAIM_CODE_FILE" elif [ ! -f "$MANIFEST" ]; then umask 077 gen_secret > "$CLAIM_CODE_FILE" chmod 0600 "$CLAIM_CODE_FILE" fi } # Write the systemd unit generated from the worker's canonical template. The # generator injects the Rust lock-contention exit code here too, so the binary # and systemd restart policy cannot silently drift apart. write_unit() { # BEGIN GENERATED WORKER SYSTEMD UNIT cat > "$UNIT_DEST" <`, whatever the profile. Kept aligned with # deployments/controller/Caddyfile.example (the compose variant); this copy # targets the all-in-one image, where nginx on the UI port is the single front # door (INS-09) so one upstream carries pages, /api and the /ws upgrade alike. # We write a file and NOTHING more: no package install, no service start, no # other program's config read or touched — putting the proxy in service stays an # operator act (OPS-13), and its presence stays unverifiable (OPS-67). write_caddyfile() { mkdir -p "$DATA_ROOT/caddy" cat > "$DATA_ROOT/caddy/Caddyfile" <worker data transfers (:8300): LAN-only by contract (OPS-68). $DOMAIN { encode zstd gzip reverse_proxy localhost:$HTTP_PORT } # OpenAI-compatible inference endpoint (gpuk-proxy) — uncomment when inference # clients live beyond the trusted LAN; TLS keeps their API keys off the wire. # # inference.$DOMAIN { # encode zstd gzip # reverse_proxy localhost:$INFERENCE_PORT # } CADDY chmod 0644 "$DATA_ROOT/caddy/Caddyfile" echo "==> wrote $DATA_ROOT/caddy/Caddyfile (filled TLS reverse-proxy example for $DOMAIN)" } # Pre-existing model caches (B81): a server that installs GPU Kitchen usually already # holds tens or hundreds of GB of weights. We print a hint and nothing more — cache # questions belong to the first-run wizard (INS-48, REG-41), never to the CLI; no # config of any other program is read or touched, and referencing a cache stays an # explicit choice made in the UI. hint_existing_caches() { hec_chosen=$(readlink -f "$1" 2>/dev/null || echo "$1") hec_found="" for hec_dir in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do [ -n "$hec_dir" ] || continue [ -d "$hec_dir/hub" ] || continue hec_real=$(readlink -f "$hec_dir" 2>/dev/null || echo "$hec_dir") [ "$hec_real" != "$hec_chosen" ] || continue # Hub layout only (models--*) — matches what the scan can actually reference. ls -d "$hec_dir"/hub/models--* >/dev/null 2>&1 || continue hec_found="$hec_found $hec_real" done [ -n "$hec_found" ] || return 0 echo for hec_dir in $hec_found; do echo "==> existing model cache found at $hec_dir" done echo " GPU Kitchen will offer to reuse those models at first launch, and any" echo " time from Nodes & GPU -> Storage. A reused cache is referenced in" echo " place. Nothing is moved or deleted." } cmd_install() { IMAGE=""; MODE="worker"; CLUSTER="default"; DATA_ROOT="/var/lib/gpu-kitchen" # bridge, not host (INS-49): on the host network every listener the image opens # is a host-wide claim, and a box that already runs something on 3000 or 8000 # killed Nitro. Bridged, only the three published ports touch the host; what # that costs — the host daemon must be told its LAN address and the engines' # docker network — is written to worker.env below. `--network host` remains. CACHE_DIR=""; NETWORK="bridge"; CONTROLLER_URL=""; BIN_SRC=""; HEALTH_PORT="8001" # 1337, not 8080: kept in lockstep with install.sh's default (INS-01). The # worker channel and the inference endpoint move the same way (INS-46). ENROLL_TOKEN=""; HTTP_PORT="1337"; MTLS_PORT="8443"; INFERENCE_PORT="8200" # Worker data plane (OPS-68): the daemon reads GPUK_DATA_PORT, # GPUK_MODEL_TRANSFER_PORT and GPUK_HANDOVER_DATA_PORT from worker.env and # announces the first two to the controller, so moving them here is complete. DATA_PORT="8300"; TRANSFER_PORT="8301"; HANDOVER_PORT="8302" PROFILE=""; DOMAIN=""; DRY_RUN=0; RESET_MANIFEST=0 while [ $# -gt 0 ]; do case "$1" in --image) IMAGE="$2"; shift 2 ;; --reset-manifest) RESET_MANIFEST=1; shift ;; --mode) MODE="$2"; shift 2 ;; --cluster) CLUSTER="$2"; shift 2 ;; --profile) PROFILE="$2"; shift 2 ;; --domain) DOMAIN="$2"; shift 2 ;; --data-root) DATA_ROOT="$2"; shift 2 ;; --cache-dir) CACHE_DIR="$2"; shift 2 ;; --network) NETWORK="$2"; shift 2 ;; --controller) CONTROLLER_URL="$2"; shift 2 ;; # `--token` is an ENROLLMENT token (single-use or shared), NOT a bearer: # the worker↔controller channel is cert-only since D11. `--token` is kept as # a spelling of `--enroll-token`. --enroll-token|--token) ENROLL_TOKEN="$2"; shift 2 ;; --health-port) HEALTH_PORT="$2"; shift 2 ;; --http-port) HTTP_PORT="$2"; shift 2 ;; --mtls-port) MTLS_PORT="$2"; shift 2 ;; --inference-port) INFERENCE_PORT="$2"; shift 2 ;; --data-port) DATA_PORT="$2"; shift 2 ;; --transfer-port) TRANSFER_PORT="$2"; shift 2 ;; --handover-port) HANDOVER_PORT="$2"; shift 2 ;; --binary) BIN_SRC="$2"; shift 2 ;; --dry-run) DRY_RUN=1; shift ;; *) die "unknown install option: $1" ;; esac done # Roles, as the backend names them (multi-server/config.ts): "controller" (full # app + UI, accepts workers) and "worker" (headless compute node). Historical # spellings still work. case "$MODE" in controller|server|standalone|manager) MODE="controller" ;; worker|agent) MODE="worker" ;; *) die "--mode must be controller or worker (got '$MODE')" ;; esac if [ "$MODE" = "worker" ] && [ -n "$PROFILE" ]; then die "--profile applies only to --mode controller; workers do not have an installation profile" fi for _pv in "$HTTP_PORT" "$MTLS_PORT" "$INFERENCE_PORT" "$HEALTH_PORT" \ "$DATA_PORT" "$TRANSFER_PORT" "$HANDOVER_PORT"; do case "$_pv" in ''|*[!0-9]*) die "not a port number: '$_pv'" ;; esac [ "$_pv" -ge 1 ] && [ "$_pv" -le 65535 ] || die "port out of range: $_pv" done if [ "$HTTP_PORT" = "$MTLS_PORT" ] || [ "$HTTP_PORT" = "$INFERENCE_PORT" ] || [ "$MTLS_PORT" = "$INFERENCE_PORT" ]; then die "--http-port, --mtls-port and --inference-port must differ (got $HTTP_PORT, $MTLS_PORT, $INFERENCE_PORT)" fi # The worker's four listeners (health + data plane) must differ too; the # handover port is the data port's temporary twin (WRK-173), never the same. _seen="" for _pv in "$HEALTH_PORT" "$DATA_PORT" "$TRANSFER_PORT" "$HANDOVER_PORT"; do case " $_seen " in *" $_pv "*) die "--health-port, --data-port, --transfer-port and --handover-port must differ (got $HEALTH_PORT, $DATA_PORT, $TRANSFER_PORT, $HANDOVER_PORT)" ;; esac _seen="$_seen $_pv" done case "$PROFILE" in ""|homelab|studio|enterprise|public) ;; *) die "--profile must be homelab, studio, enterprise or public (got '$PROFILE')" ;; esac if [ -n "$DOMAIN" ]; then case "$DOMAIN" in *[!A-Za-z0-9.-]*) die "--domain must be a bare domain name (got '$DOMAIN')" ;; esac fi if [ "$MODE" = "controller" ]; then if [ -z "$IMAGE" ] && [ "$DRY_RUN" -eq 1 ]; then IMAGE="example.invalid/gpukitchen-controller:v0.0.0-dry-run" fi [ -n "$IMAGE" ] || die "--image registry/gpukitchen-controller: is required for a controller" case "$IMAGE" in *@sha256:*) ;; *:latest) die "refusing floating image tag '$IMAGE' — use an explicit release tag or digest" ;; *) _image_tag="${IMAGE##*:}" case "$_image_tag" in "$IMAGE"|*/*) die "controller image must carry an explicit tag or sha256 digest (got '$IMAGE')" ;; esac ;; esac fi [ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf" CACHE_JSON="[{\"hostPath\":\"$(json_str "$CACHE_DIR")\",\"shared\":false}]" SELF_ENROLL_FILE="$DATA_ROOT/self-enroll-token" BOOTSTRAP_PASSWORD_FILE="$DATA_ROOT/secrets/bootstrap_admin_password" CLAIM_CODE_FILE="$DATA_ROOT/secrets/claim_code" CLAIM_CODE_AVAILABLE=0 if [ "$DRY_RUN" -eq 1 ]; then case "$MODE:$PROFILE" in controller:homelab|controller:studio) ;; controller:*) CLAIM_CODE_AVAILABLE=1 ;; esac echo "==> dry run: no file, service or container was changed" [ ! -f "$MANIFEST" ] \ || echo "==> dry run: $MANIFEST exists — this render would be MERGED into it, controller-owned settings kept" if [ "$MODE" = "worker" ]; then render_worker_manifest; else render_controller_manifest; fi [ -z "$DOMAIN" ] || echo "==> dry run: would write $DATA_ROOT/caddy/Caddyfile for $DOMAIN" return 0 fi # ── Port conflicts (INS-46) — the mutator's own guard ──────────────────────── # install.sh's preflight already checks these, and on a terminal it offers an # alternative port. gpuk is the actual mutator and contributors call it # DIRECTLY, so it re-checks and refuses, non-interactively, naming the flag # that moves the port. Same helpers as install.sh (both scripts ship standalone # from the channel). A listener owned by an existing install is not a conflict: # a manifest on disk means the re-run is the update path, and every checked # port is then our own; without a manifest, a leftover app container still # holds OUR ports — apply() renames and stops it before the new one starts. if [ ! -f "$MANIFEST" ]; then _port_tool="${GPUK_PORT_CHECK_TOOL:-}" case "$_port_tool" in ""|ss|netstat) ;; *) die "GPUK_PORT_CHECK_TOOL must be ss or netstat (got '$_port_tool')" ;; esac if [ -z "$_port_tool" ]; then if command -v ss >/dev/null 2>&1; then _port_tool="ss" elif command -v netstat >/dev/null 2>&1; then _port_tool="netstat"; fi fi _ours="" if [ "$MODE" = "controller" ]; then _ours=$(leftover_container_ports gpu-kitchen) [ -z "$_ours" ] || echo "==> existing app container gpu-kitchen found (ports $_ours): the install replaces it" fi if [ -z "$_port_tool" ]; then echo "==> warning: cannot check for port conflicts (no ss or netstat)" else if [ "$MODE" = "controller" ]; then set -- "$HTTP_PORT:UI:--http-port" \ "$MTLS_PORT:worker channel:--mtls-port" \ "$INFERENCE_PORT:inference endpoint:--inference-port" else # Worker data-plane ports (OPS-68) plus the local health listener. set -- "$HEALTH_PORT:worker health:--health-port" \ "$DATA_PORT:worker data plane:--data-port" \ "$TRANSFER_PORT:model transfers:--transfer-port" \ "$HANDOVER_PORT:handover data plane:--handover-port" fi for _spec in "$@"; do _p=${_spec%%:*}; _rest=${_spec#*:}; _label=${_rest%%:*}; _flag=${_rest#*:} case " $_ours " in *" $_p "*) continue ;; esac _busy=1 case "$_port_tool" in ss) [ -n "$(ss -ltnH "sport = :$_p" 2>/dev/null)" ] || _busy=0 ;; netstat) netstat -ltn 2>/dev/null \ | awk -v p="$_p" '{n=split($4,a,":"); if (a[n]==p) found=1} END{exit found?0:1}' \ || _busy=0 ;; esac [ "$_busy" -eq 0 ] && continue _owner=$(port_owner "$_port_tool" "$_p") _hint="Free it first, then run the install again." [ -z "$_flag" ] || _hint="Pass $_flag

to choose another port, or free it first." die "port $_p ($_label) is already in use by ${_owner:-an unknown process}. $_hint" done fi fi need_root install_binary "$BIN_SRC" install_cli mkdir -p "$ETC_DIR" "$DATA_ROOT" "$CACHE_DIR" seed_secrets "$DATA_ROOT" if [ "$MODE" = "controller" ]; then prepare_claim_code [ ! -s "$CLAIM_CODE_FILE" ] || CLAIM_CODE_AVAILABLE=1 fi umask 077 if [ "$MODE" = "worker" ]; then write_worker_manifest # Optional env overrides for the daemon (cluster grouping, display name, an # explicit controller URL). The enrolled identity carries the controller URL # + CA too — this is belt-and-braces / pre-enroll discovery grouping. { echo "GPUK_CLUSTER=$CLUSTER" echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT" echo "GPUK_DATA_PORT=$DATA_PORT" echo "GPUK_MODEL_TRANSFER_PORT=$TRANSFER_PORT" echo "GPUK_HANDOVER_DATA_PORT=$HANDOVER_PORT" echo "GPUK_MACHINE_ID_FILE=$MACHINE_ID_FILE" echo "NODE_DISPLAY_NAME=$(hostname)" [ -n "$CONTROLLER_URL" ] && echo "GPUK_CONTROLLER_URLS=$CONTROLLER_URL" } > "$WORKER_ENV" chmod 0600 "$WORKER_ENV" echo "==> wrote $MANIFEST (worker: no app container)" else write_controller_manifest # The host daemon and app container share DATA_ROOT. Only controller-mode # workerd gets this private bootstrap channel; remote workers stay tokenless. # # A bridged app container (INS-49) needs two more lines, read by the daemon # from its OWN env — the daemon runs on the host, not in the container: # - GPUK_DATA_HOST: the daemon dials the published mTLS port through docker # NAT, so the controller sees it arrive from the bridge gateway (172.17.0.1) # and would persist THAT as the node's address (WRK-93). The LAN address is # what peers, the proxy and the UI must dial instead. # - BACKEND_DOCKER_NETWORK: the engines the daemon launches join the app # container's docker network, so the backend and gpuk-proxy reach them by # container IP (apps/worker/src/modules/docker.rs, staging/engine.rs). # Exactly what dev.sh hands the dev worker; host networking needs neither. _data_host="" if [ "$NETWORK" != "host" ]; then _data_host="${GPUK_DATA_HOST:-$(host_lan_ipv4)}" [ -n "$_data_host" ] || echo "==> warning: this host's LAN IPv4 could not be determined (no ip, no hostname -I)." \ "Set GPUK_DATA_HOST= in $WORKER_ENV, or the controller will address its own worker through the docker bridge." fi { echo "GPUK_CLUSTER=$CLUSTER" echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT" echo "GPUK_MACHINE_ID_FILE=$MACHINE_ID_FILE" echo "GPUK_CONTROLLER_URLS=wss://127.0.0.1:$MTLS_PORT" echo "GPUK_SELF_ENROLL_FILE=$SELF_ENROLL_FILE" echo "NODE_DISPLAY_NAME=$(hostname)" if [ "$NETWORK" != "host" ]; then [ -z "$_data_host" ] || echo "GPUK_DATA_HOST=$_data_host" echo "BACKEND_DOCKER_NETWORK=$NETWORK" fi } > "$WORKER_ENV" chmod 0600 "$WORKER_ENV" echo "==> wrote $MANIFEST (controller: app container $IMAGE, network $NETWORK)" fi chmod 0600 "$MANIFEST" [ -z "$DOMAIN" ] || write_caddyfile write_unit systemctl daemon-reload echo "==> wrote $UNIT_DEST" # ── Worker: token enrollment before start, or unattended LAN discovery ── if [ "$MODE" = "worker" ]; then if [ -n "$ENROLL_TOKEN" ]; then [ -n "$CONTROLLER_URL" ] || die "--controller wss://: is required to enroll" echo "==> enrolling against $CONTROLLER_URL ..." GPUK_IDENTITY_DIR="$IDENTITY_DIR" GPUK_MACHINE_ID_FILE="$MACHINE_ID_FILE" "$BIN_DEST" enroll \ --controller "$CONTROLLER_URL" --token "$ENROLL_TOKEN" \ || die "enrollment failed (bad/expired token, or controller unreachable)" else if [ -n "$CONTROLLER_URL" ]; then echo "==> no token given: the daemon will request admission from $CONTROLLER_URL and wait" echo " for automatic admission or administrator approval." else echo "==> no token or controller given: the daemon will discover its cluster on the LAN and wait" echo " for automatic admission or administrator approval." fi fi start_daemon echo "==> $SERVICE_NAME enabled and started" echo echo "Done. The worker daemon is running${ENROLL_TOKEN:+ and enrolled}." echo " Status : gpuk status Logs: gpuk logs" hint_existing_caches "$CACHE_DIR" return 0 fi # ── Controller: start the daemon, then bring up the app container ── start_daemon echo "==> $SERVICE_NAME enabled and started" # The pull is the long part of a first install (a multi-GB image) and workerd # runs docker with its output captured, so pulling from inside `apply` is # minutes of silence that read as a hang. Pull here, on the terminal, where # docker's own per-layer progress is what the operator sees; `apply` then finds # the image present and skips its pull (INS-03). # Feedback only, never the verdict: `apply` pulls again whatever happened here # and reports the cause itself (the CI shell gate installs a stub daemon against # an image that does not exist anywhere — the daemon's pull is the one that counts). if ! docker image inspect "$IMAGE" >/dev/null 2>&1; then echo "==> pulling $IMAGE (docker shows the progress per layer) ..." docker pull "$IMAGE" \ || echo "==> the pull did not complete here; the daemon retries it during apply and reports the cause if it fails again" fi echo "==> applying manifest — starting the app container and waiting for its health check (up to 60s) ..." # No unix socket any more: workerd reconciles the app container in-process from # the on-disk manifest (there is no backend to relay through on the very first # boot). Steady-state updates go through the daemon over WS. GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" apply \ || die "the app container did not come up — the [worker] lines above say why (a [controller] FATAL line names the component and the fix). Full container logs: gpuk logs" echo echo "Done. The worker daemon is running and the app container is up." echo " Status : gpuk status App logs: gpuk logs Daemon: journalctl -u $SERVICE_NAME" hint_existing_caches "$CACHE_DIR" } # Worker manifest: cacheDisks and an EMPTY image so # has_app_container() is false (apps/worker/src/modules/manifest.rs). No ports, no env, # no secretsRef — a worker runs no app container. render_worker_manifest() { cat < Network…). Rewriting it from flags would # silently undo all of that, so the fresh render goes through the daemon's own # `install-manifest`, which only replaces what the installer owns (image, mode, # cluster, network, data root, ports, secret references, the primary cache disk, # the env keys above) and keeps the rest. The first write is the render as-is. write_manifest() { # $1 = render function if [ -f "$MANIFEST" ] && [ "$RESET_MANIFEST" -eq 1 ]; then # The operator's explicit regeneration (WRK-191): the daemon archives the # existing document — readable or not — writes the render as-is and NAMES what # the archive carried that the render does not. The flags install.sh inherited # from the old file (ports, profile, data root) are already in the render. "$1" > "$MANIFEST.new" chmod 0600 "$MANIFEST.new" GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" install-manifest \ --from "$MANIFEST.new" --reset >/dev/null \ || { rm -f "$MANIFEST.new"; die "could not reset $MANIFEST — the [worker] line above names the cause"; } rm -f "$MANIFEST.new" echo "==> $MANIFEST rebuilt from this install's settings (--reset-manifest); the previous file is archived beside it" elif [ -f "$MANIFEST" ]; then "$1" > "$MANIFEST.new" chmod 0600 "$MANIFEST.new" GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" install-manifest \ --from "$MANIFEST.new" --own-env "$INSTALL_ENV_KEYS" >/dev/null \ || { rm -f "$MANIFEST.new" die "could not merge the new settings into $MANIFEST — the [worker] line above names the cause. A manifest written by an older release can carry a field this release removed. Re-run the same command with --reset-manifest: the file is archived beside itself as manifest.json.before-reset.