diff --git a/gpuk b/gpuk index 9a43995..bde0736 100755 --- a/gpuk +++ b/gpuk @@ -104,6 +104,72 @@ install_binary() { # [local-path] gen_secret() { head -c 32 /dev/urandom | base64 | tr '+/' '-_' | tr -d '='; } +# The host ports a leftover app container is reached on ("ui mtls inference"), +# read off the container itself with the precedence the controller applies +# (core/published-ports.ts): host networking moves the listeners (GPUK_PORT, +# GPUK_MTLS_PORT, GPUK_LISTEN_ADDR); anything else keeps the image's fixed +# listeners and publishes them (GPUK_PUBLIC_*, then the port bindings). Empty +# when there is no such container or no docker. +leftover_container_ports() { # $1 = container name + command -v docker >/dev/null 2>&1 || return 0 + # Same format string as install.sh's container_facts — one reading, two scripts. + _f=$(docker inspect -f '{{.State.Status}}|{{.HostConfig.NetworkMode}}|{{range $p, $b := .HostConfig.PortBindings}}{{range $b}}{{$p}}={{.HostPort}} {{end}}{{end}}{{"\n"}}{{range .Config.Env}}{{.}}{{"\n"}}{{end}}' "$1" 2>/dev/null) \ + || return 0 + [ -n "$_f" ] || return 0 + _head=$(printf '%s\n' "$_f" | head -1) + _r=${_head#*|}; _net=${_r%%|*}; _bind=${_r#*|} + _env() { printf '%s\n' "$_f" | sed -n "s/^$1=//p" | head -1; } + _pub() { printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n "s/^$1\/tcp=//p" | head -1; } + if [ "$_net" = "host" ]; then + _ui=$(_env GPUK_PORT); _mtls=$(_env GPUK_MTLS_PORT) + _inf=$(_env GPUK_PROXY_PUBLIC_PORT) + [ -n "$_inf" ] || { _la=$(_env GPUK_LISTEN_ADDR); _inf=${_la##*:}; } + else + _ui=$(_env GPUK_PUBLIC_PORT); [ -n "$_ui" ] || _ui=$(_pub 8080) + _mtls=$(_env GPUK_PUBLIC_MTLS_PORT); [ -n "$_mtls" ] || _mtls=$(_pub 8443) + _inf=$(_env GPUK_PROXY_PUBLIC_PORT); [ -n "$_inf" ] || _inf=$(_pub 8200) + fi + printf '%s %s %s' "${_ui:-8080}" "${_mtls:-8443}" "${_inf:-8200}" +} + +# Who holds a port: "process", "process in container NAME", or "". The process's +# cgroup names its container (host networking); a bridged publication is found +# as the container's port mapping in `docker ps` (the host-side holder is +# docker-proxy, which says nothing by itself). Same reading as install.sh. +port_owner() { # $1 = ss|netstat, $2 = port + _proc=""; _pid="" + case "$1" in + ss) + _line=$(ss -ltnpH "sport = :$2" 2>/dev/null | head -1) + _proc=$(printf '%s' "$_line" | sed -n 's/.*users:((\"\([^"]*\)\".*/\1/p') + _pid=$(printf '%s' "$_line" | sed -n 's/.*pid=\([0-9]*\).*/\1/p') + ;; + netstat) + _field=$(netstat -ltnp 2>/dev/null \ + | awk -v p="$2" '{n=split($4,a,":"); if (a[n]==p) {print $NF; exit}}') + case "$_field" in + */*) _pid=${_field%%/*}; _proc=${_field#*/} ;; + *) _proc="$_field" ;; + esac + ;; + esac + case "$_pid" in *[!0-9]*|"") _pid="" ;; esac + _ctr="" + if command -v docker >/dev/null 2>&1; then + if [ -n "$_pid" ] && [ -r "/proc/$_pid/cgroup" ]; then + _cid=$(sed -n 's#.*docker[-/]\([0-9a-f]\{64\}\).*#\1#p' "/proc/$_pid/cgroup" 2>/dev/null | head -1) + [ -z "$_cid" ] || _ctr=$(docker inspect -f '{{.Name}}' "$_cid" 2>/dev/null | sed 's|^/||') + fi + [ -n "$_ctr" ] || _ctr=$(docker ps --format '{{.Names}} {{.Ports}}' 2>/dev/null \ + | awk -v p=":$2->" 'index($0, p) { print $1; exit }') + fi + if [ -n "$_ctr" ]; then + printf '%s' "${_proc:-a process} in container $_ctr" + else + printf '%s' "$_proc" + fi +} + seed_secrets() { # DATA_ROOT _sd="$1/secrets" mkdir -p "$_sd"; chmod 0700 "$_sd" @@ -191,7 +257,7 @@ write_caddyfile() { # # The app already runs with GPUK_HSTS=true and GPUK_SESSION_COOKIE_SECURE=true # (set by the public profile). What does NOT go through this proxy: -# - the worker mTLS channel (:8443): workers pin the controller CA and must +# - the worker mTLS channel (:$MTLS_PORT): workers pin the controller CA and must # reach it DIRECTLY — terminating it here would break the pin. # - worker<->worker data transfers (:8300): LAN-only by contract (OPS-68). @@ -205,7 +271,7 @@ $DOMAIN { # # inference.$DOMAIN { # encode zstd gzip -# reverse_proxy localhost:8200 +# reverse_proxy localhost:$INFERENCE_PORT # } CADDY chmod 0644 "$DATA_ROOT/caddy/Caddyfile" @@ -242,8 +308,14 @@ hint_existing_caches() { cmd_install() { IMAGE=""; MODE="worker"; CLUSTER="default"; DATA_ROOT="/var/lib/gpu-kitchen" CACHE_DIR=""; NETWORK="host"; CONTROLLER_URL=""; BIN_SRC=""; HEALTH_PORT="8001" - # 1337, not 8080: kept in lockstep with install.sh's default (INS-01). - ENROLL_TOKEN=""; HTTP_PORT="1337"; PROFILE=""; DOMAIN=""; DRY_RUN=0 + # 1337, not 8080: kept in lockstep with install.sh's default (INS-01). The + # worker channel and the inference endpoint move the same way (INS-46). + ENROLL_TOKEN=""; HTTP_PORT="1337"; MTLS_PORT="8443"; INFERENCE_PORT="8200" + # Worker data plane (OPS-68): the daemon reads GPUK_DATA_PORT, + # GPUK_MODEL_TRANSFER_PORT and GPUK_HANDOVER_DATA_PORT from worker.env and + # announces the first two to the controller, so moving them here is complete. + DATA_PORT="8300"; TRANSFER_PORT="8301"; HANDOVER_PORT="8302" + PROFILE=""; DOMAIN=""; DRY_RUN=0 while [ $# -gt 0 ]; do case "$1" in --image) IMAGE="$2"; shift 2 ;; @@ -261,6 +333,11 @@ cmd_install() { --enroll-token|--token) ENROLL_TOKEN="$2"; shift 2 ;; --health-port) HEALTH_PORT="$2"; shift 2 ;; --http-port) HTTP_PORT="$2"; shift 2 ;; + --mtls-port) MTLS_PORT="$2"; shift 2 ;; + --inference-port) INFERENCE_PORT="$2"; shift 2 ;; + --data-port) DATA_PORT="$2"; shift 2 ;; + --transfer-port) TRANSFER_PORT="$2"; shift 2 ;; + --handover-port) HANDOVER_PORT="$2"; shift 2 ;; --binary) BIN_SRC="$2"; shift 2 ;; --dry-run) DRY_RUN=1; shift ;; *) die "unknown install option: $1" ;; @@ -279,6 +356,26 @@ cmd_install() { if [ "$MODE" = "worker" ] && [ -n "$PROFILE" ]; then die "--profile applies only to --mode controller; workers do not have an installation profile" fi + + for _pv in "$HTTP_PORT" "$MTLS_PORT" "$INFERENCE_PORT" "$HEALTH_PORT" \ + "$DATA_PORT" "$TRANSFER_PORT" "$HANDOVER_PORT"; do + case "$_pv" in + ''|*[!0-9]*) die "not a port number: '$_pv'" ;; + esac + [ "$_pv" -ge 1 ] && [ "$_pv" -le 65535 ] || die "port out of range: $_pv" + done + if [ "$HTTP_PORT" = "$MTLS_PORT" ] || [ "$HTTP_PORT" = "$INFERENCE_PORT" ] || [ "$MTLS_PORT" = "$INFERENCE_PORT" ]; then + die "--http-port, --mtls-port and --inference-port must differ (got $HTTP_PORT, $MTLS_PORT, $INFERENCE_PORT)" + fi + # The worker's four listeners (health + data plane) must differ too; the + # handover port is the data port's temporary twin (WRK-173), never the same. + _seen="" + for _pv in "$HEALTH_PORT" "$DATA_PORT" "$TRANSFER_PORT" "$HANDOVER_PORT"; do + case " $_seen " in + *" $_pv "*) die "--health-port, --data-port, --transfer-port and --handover-port must differ (got $HEALTH_PORT, $DATA_PORT, $TRANSFER_PORT, $HANDOVER_PORT)" ;; + esac + _seen="$_seen $_pv" + done case "$PROFILE" in ""|homelab|studio|enterprise|public) ;; *) die "--profile must be homelab, studio, enterprise or public (got '$PROFILE')" ;; @@ -319,32 +416,54 @@ cmd_install() { if [ "$DRY_RUN" -eq 1 ]; then [ "$MODE" != "controller" ] || CLAIM_CODE_AVAILABLE=1 echo "==> dry run: no file, service or container was changed" + [ ! -f "$MANIFEST" ] \ + || echo "==> dry run: $MANIFEST exists — this render would be MERGED into it, controller-owned settings kept" if [ "$MODE" = "worker" ]; then render_worker_manifest; else render_controller_manifest; fi [ -z "$DOMAIN" ] || echo "==> dry run: would write $DATA_ROOT/caddy/Caddyfile for $DOMAIN" return 0 fi # ── Port conflicts (INS-46) — the mutator's own guard ──────────────────────── - # install.sh's preflight already checks these, and on a terminal it can offer - # an alternative port. gpuk is the actual mutator and contributors call it - # DIRECTLY, so it re-checks and refuses, non-interactively. Same helpers as - # install.sh (both scripts ship standalone from the channel). A listener owned - # by an existing install is not a conflict: a manifest on disk means the - # re-run is the update path, and every checked port is then our own. + # install.sh's preflight already checks these, and on a terminal it offers an + # alternative port. gpuk is the actual mutator and contributors call it + # DIRECTLY, so it re-checks and refuses, non-interactively, naming the flag + # that moves the port. Same helpers as install.sh (both scripts ship standalone + # from the channel). A listener owned by an existing install is not a conflict: + # a manifest on disk means the re-run is the update path, and every checked + # port is then our own; without a manifest, a leftover app container still + # holds OUR ports — apply() renames and stops it before the new one starts. if [ ! -f "$MANIFEST" ]; then - _port_tool="" - if command -v ss >/dev/null 2>&1; then _port_tool="ss" - elif command -v netstat >/dev/null 2>&1; then _port_tool="netstat"; fi + _port_tool="${GPUK_PORT_CHECK_TOOL:-}" + case "$_port_tool" in + ""|ss|netstat) ;; + *) die "GPUK_PORT_CHECK_TOOL must be ss or netstat (got '$_port_tool')" ;; + esac + if [ -z "$_port_tool" ]; then + if command -v ss >/dev/null 2>&1; then _port_tool="ss" + elif command -v netstat >/dev/null 2>&1; then _port_tool="netstat"; fi + fi + _ours="" + if [ "$MODE" = "controller" ]; then + _ours=$(leftover_container_ports gpu-kitchen) + [ -z "$_ours" ] || echo "==> existing app container gpu-kitchen found (ports $_ours): the install replaces it" + fi if [ -z "$_port_tool" ]; then echo "==> warning: cannot check for port conflicts (no ss or netstat)" else if [ "$MODE" = "controller" ]; then - set -- "$HTTP_PORT" 8443 8200 + set -- "$HTTP_PORT:UI:--http-port" \ + "$MTLS_PORT:worker channel:--mtls-port" \ + "$INFERENCE_PORT:inference endpoint:--inference-port" else # Worker data-plane ports (OPS-68) plus the local health listener. - set -- "$HEALTH_PORT" 8300 8301 8302 + set -- "$HEALTH_PORT:worker health:--health-port" \ + "$DATA_PORT:worker data plane:--data-port" \ + "$TRANSFER_PORT:model transfers:--transfer-port" \ + "$HANDOVER_PORT:handover data plane:--handover-port" fi - for _p in "$@"; do + for _spec in "$@"; do + _p=${_spec%%:*}; _rest=${_spec#*:}; _label=${_rest%%:*}; _flag=${_rest#*:} + case " $_ours " in *" $_p "*) continue ;; esac _busy=1 case "$_port_tool" in ss) [ -n "$(ss -ltnH "sport = :$_p" 2>/dev/null)" ] || _busy=0 ;; @@ -354,7 +473,11 @@ cmd_install() { || _busy=0 ;; esac - [ "$_busy" -eq 0 ] || die "port $_p is already in use. Free it first, then run the install again." + [ "$_busy" -eq 0 ] && continue + _owner=$(port_owner "$_port_tool" "$_p") + _hint="Free it first, then run the install again." + [ -z "$_flag" ] || _hint="Pass $_flag
to choose another port, or free it first."
+ die "port $_p ($_label) is already in use by ${_owner:-an unknown process}. $_hint"
done
fi
fi
@@ -378,6 +501,9 @@ cmd_install() {
{
echo "GPUK_CLUSTER=$CLUSTER"
echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT"
+ echo "GPUK_DATA_PORT=$DATA_PORT"
+ echo "GPUK_MODEL_TRANSFER_PORT=$TRANSFER_PORT"
+ echo "GPUK_HANDOVER_DATA_PORT=$HANDOVER_PORT"
echo "GPUK_MACHINE_ID_FILE=$MACHINE_ID_FILE"
echo "NODE_DISPLAY_NAME=$(hostname)"
[ -n "$CONTROLLER_URL" ] && echo "GPUK_CONTROLLER_URLS=$CONTROLLER_URL"
@@ -392,7 +518,7 @@ cmd_install() {
echo "GPUK_CLUSTER=$CLUSTER"
echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT"
echo "GPUK_MACHINE_ID_FILE=$MACHINE_ID_FILE"
- echo "GPUK_CONTROLLER_URLS=wss://127.0.0.1:8443"
+ echo "GPUK_CONTROLLER_URLS=wss://127.0.0.1:$MTLS_PORT"
echo "GPUK_SELF_ENROLL_FILE=$SELF_ENROLL_FILE"
echo "NODE_DISPLAY_NAME=$(hostname)"
} > "$WORKER_ENV"
@@ -439,7 +565,7 @@ cmd_install() {
# No unix socket any more: workerd reconciles the app container in-process from
# the on-disk manifest (there is no backend to relay through on the very first
# boot). Steady-state updates go through the daemon over WS.
- "$BIN_DEST" apply || die "apply failed. Check: gpuk logs"
+ GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" apply || die "apply failed. Check: gpuk logs"
echo
echo "Done. The worker daemon is running and the app container is up."
echo " Status : gpuk status App logs: gpuk logs Daemon: journalctl -u $SERVICE_NAME"
@@ -463,7 +589,36 @@ render_worker_manifest() {
JSON
}
-write_worker_manifest() { render_worker_manifest > "$MANIFEST"; }
+# Every env key gpuk may ever write into a manifest — conditional ones included.
+# On a re-run the merge sets each of them to the fresh value or DELETES it when the
+# fresh render no longer carries it (a profile change drops GPUK_HSTS); any other
+# key was pushed by the controller and is kept. Keep this list in step with
+# render_controller_manifest.
+INSTALL_ENV_KEYS="NODE_ENV,GPUK_MODE,GPUK_CLUSTER,GPUK_SELF_ENROLL_FILE,NODE_DISPLAY_NAME,HF_HOME,GPUK_DATA_ROOT,GPUK_INSTALL_PROFILE,GPUK_HSTS,GPUK_SESSION_COOKIE_SECURE,GPUK_BOOTSTRAP_MUST_CHANGE,BACKEND_DOCKER_NETWORK,GPUK_PORT,GPUK_PUBLIC_PORT,GPUK_MTLS_PORT,GPUK_PUBLIC_MTLS_PORT,GPUK_LISTEN_ADDR,GPUK_PROXY_PUBLIC_PORT"
+
+# Write the manifest — or, when one exists, MERGE into it (INS-03). Re-running the
+# installer is the update path, and the manifest is not ours alone: since the first
+# install the controller has patched it (extra cache disks, shared origin, eviction,
+# LED binary, ports moved from Settings -> Network…). Rewriting it from flags would
+# silently undo all of that, so the fresh render goes through the daemon's own
+# `install-manifest`, which only replaces what the installer owns (image, mode,
+# cluster, network, data root, ports, secret references, the primary cache disk,
+# the env keys above) and keeps the rest. The first write is the render as-is.
+write_manifest() { # $1 = render function
+ if [ -f "$MANIFEST" ]; then
+ "$1" > "$MANIFEST.new"
+ chmod 0600 "$MANIFEST.new"
+ GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" install-manifest \
+ --from "$MANIFEST.new" --own-env "$INSTALL_ENV_KEYS" >/dev/null \
+ || { rm -f "$MANIFEST.new"; die "could not merge the new settings into $MANIFEST"; }
+ rm -f "$MANIFEST.new"
+ echo "==> merged into the existing $MANIFEST (controller-owned settings kept)"
+ else
+ "$1" > "$MANIFEST"
+ fi
+}
+
+write_worker_manifest() { write_manifest render_worker_manifest; }
# Controller manifest: the declarative app-container description workerd applies.
render_controller_manifest() {
@@ -490,18 +645,25 @@ render_controller_manifest() {
[ "$NETWORK" = "host" ] || EXTRA_ENV="$EXTRA_ENV,\"BACKEND_DOCKER_NETWORK\":\"$(json_str "$NETWORK")\""
# Published != bound (INS-43). On a bridged network the container keeps the image's
- # FIXED listeners — nginx 8080, worker mTLS 8443 — and --http-port only moves the HOST
- # side of the publication; Settings -> Network moves it later by patching this same
- # manifest, so the container port must never become a variable. With host networking
- # nothing is published and the listener itself takes the port.
+ # FIXED listeners — nginx 8080, worker mTLS 8443, gpuk-proxy 8200 — and the port
+ # flags only move the HOST side of the publication; Settings -> Network moves it
+ # later by patching this same manifest, so the container port must never become a
+ # variable. With host networking nothing is published and the listeners themselves
+ # take the ports (core/published-ports.ts reads this back with the same rules; the
+ # proxy port is GPUK_LISTEN_ADDR for the proxy, GPUK_PROXY_PUBLIC_PORT for what the
+ # backend shows — core/cluster-settings.ts proxyPublicPort).
PORTS_JSON="{}"
if [ "$NETWORK" = "host" ]; then
EXTRA_ENV="$EXTRA_ENV,\"GPUK_PORT\":\"$(json_str "$HTTP_PORT")\""
+ EXTRA_ENV="$EXTRA_ENV,\"GPUK_MTLS_PORT\":\"$(json_str "$MTLS_PORT")\""
+ EXTRA_ENV="$EXTRA_ENV,\"GPUK_LISTEN_ADDR\":\"0.0.0.0:$(json_str "$INFERENCE_PORT")\""
else
EXTRA_ENV="$EXTRA_ENV,\"GPUK_PORT\":\"8080\""
EXTRA_ENV="$EXTRA_ENV,\"GPUK_PUBLIC_PORT\":\"$(json_str "$HTTP_PORT")\""
- PORTS_JSON="{\"8080\":$HTTP_PORT,\"8200\":8200,\"8443\":8443}"
+ EXTRA_ENV="$EXTRA_ENV,\"GPUK_PUBLIC_MTLS_PORT\":\"$(json_str "$MTLS_PORT")\""
+ PORTS_JSON="{\"8080\":$HTTP_PORT,\"8200\":$INFERENCE_PORT,\"8443\":$MTLS_PORT}"
fi
+ EXTRA_ENV="$EXTRA_ENV,\"GPUK_PROXY_PUBLIC_PORT\":\"$(json_str "$INFERENCE_PORT")\""
cat < Port the UI listens on (default 1337)
+ --mtls-port Port workers dial to join this controller (default 8443)
+ --inference-port Port of the OpenAI-compatible inference endpoint (default 8200)
--data-root Use a locally-built gpu-kitchen-worker instead of downloading one
--gpuk-script Use a local copy of the gpuk installer
--skip-gpu-check Skip the 'docker run --gpus all' smoke test
- --skip-preflight Skip the host checks entirely (CI: no docker, no GPU)
+ --skip-preflight Skip the docker, driver, GPU and disk checks (CI: no docker,
+ no GPU); the listening ports are still checked
--dry-run Run the preflight and resolve the release, change nothing
-h, --help This
EOF
@@ -124,10 +144,12 @@ while [ $# -gt 0 ]; do
--version) VERSION="$2"; shift 2 ;;
--image) IMAGE="$2"; shift 2 ;;
--edition) EDITION="$2"; shift 2 ;;
- --port) PORT="$2"; shift 2 ;;
- --data-root) DATA_ROOT="$2"; shift 2 ;;
- --cache-dir) CACHE_DIR="$2"; shift 2 ;;
- --cluster) CLUSTER="$2"; shift 2 ;;
+ --port) PORT="$2"; PORT_GIVEN=1; shift 2 ;;
+ --mtls-port) MTLS_PORT="$2"; MTLS_GIVEN=1; shift 2 ;;
+ --inference-port) INFERENCE_PORT="$2"; INFERENCE_GIVEN=1; shift 2 ;;
+ --data-root) DATA_ROOT="$2"; DATA_ROOT_GIVEN=1; shift 2 ;;
+ --cache-dir) CACHE_DIR="$2"; CACHE_GIVEN=1; shift 2 ;;
+ --cluster) CLUSTER="$2"; CLUSTER_GIVEN=1; shift 2 ;;
--profile) PROFILE="$2"; shift 2 ;;
--domain) DOMAIN="$2"; shift 2 ;;
--non-interactive) NON_INTERACTIVE=1; shift ;;
@@ -141,8 +163,6 @@ while [ $# -gt 0 ]; do
esac
done
-[ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf"
-
case "$EDITION" in
community|enterprise) ;;
*) die "--edition must be community or enterprise (got '$EDITION')" ;;
@@ -157,9 +177,132 @@ case "$DOMAIN" in
*[!A-Za-z0-9.-]*) die "--domain must be a bare domain name (got '$DOMAIN')" ;;
esac
+for _pv in "$PORT" "$MTLS_PORT" "$INFERENCE_PORT"; do
+ case "$_pv" in
+ ''|*[!0-9]*) die "not a port number: '$_pv'" ;;
+ esac
+ [ "$_pv" -ge 1 ] && [ "$_pv" -le 65535 ] || die "port out of range: $_pv"
+done
+
echo
echo "${BOLD}GPU Kitchen — installing the $EDITION controller${RESET}"
+# ── 0. An existing install (INS-03) ──────────────────────────────────────────
+# Re-running this script is the update path, so before checking anything it
+# reads what is already here — and KEEPS it. An update that silently moved the
+# UI to another port, re-asked the profile (Enter = homelab would downgrade a
+# public install) or pointed at a fresh data root beside the real one is not an
+# update. Read-only. The manifest is the authority (WRK-55); without it, the
+# traces a previous install leaves (container, unit, binary, data root) still
+# mean "take over", never "start beside", and the container's own ports are ours.
+manifest_str() { # $1 = key of a string field, anywhere in the manifest
+ sed -n "s/.*\"$1\"[[:space:]]*:[[:space:]]*\"\([^\"]*\)\".*/\1/p" "$MANIFEST" 2>/dev/null | head -1
+}
+manifest_binding() { # $1 = container port → the host port the ports table publishes it on
+ sed -n "s/.*\"$1\(\/tcp\)\{0,1\}\"[[:space:]]*:[[:space:]]*\([0-9][0-9]*\).*/\2/p" "$MANIFEST" 2>/dev/null | head -1
+}
+C_STATUS=""; C_NETMODE=""
+E_PORT=""; E_MTLS_PORT=""; E_LISTEN_ADDR=""
+E_PUBLIC_PORT=""; E_PUBLIC_MTLS_PORT=""; E_PROXY_PUBLIC_PORT=""
+B_8080=""; B_8443=""; B_8200=""
+container_facts() { # $1 = name → C_*, E_*, B_* from docker; 1 when absent
+ command -v docker >/dev/null 2>&1 || return 1
+ _f=$(docker inspect -f '{{.State.Status}}|{{.HostConfig.NetworkMode}}|{{range $p, $b := .HostConfig.PortBindings}}{{range $b}}{{$p}}={{.HostPort}} {{end}}{{end}}{{"\n"}}{{range .Config.Env}}{{.}}{{"\n"}}{{end}}' "$1" 2>/dev/null) \
+ || return 1
+ [ -n "$_f" ] || return 1
+ _head=$(printf '%s\n' "$_f" | head -1)
+ C_STATUS=${_head%%|*}; _r=${_head#*|}; C_NETMODE=${_r%%|*}; _bind=${_r#*|}
+ E_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PORT=//p' | head -1)
+ E_MTLS_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_MTLS_PORT=//p' | head -1)
+ E_LISTEN_ADDR=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_LISTEN_ADDR=//p' | head -1)
+ E_PUBLIC_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PUBLIC_PORT=//p' | head -1)
+ E_PUBLIC_MTLS_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PUBLIC_MTLS_PORT=//p' | head -1)
+ E_PROXY_PUBLIC_PORT=$(printf '%s\n' "$_f" | sed -n 's/^GPUK_PROXY_PUBLIC_PORT=//p' | head -1)
+ B_8080=$(printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n 's/^8080\/tcp=//p' | head -1)
+ B_8443=$(printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n 's/^8443\/tcp=//p' | head -1)
+ B_8200=$(printf '%s\n' "$_bind" | tr ' ' '\n' | sed -n 's/^8200\/tcp=//p' | head -1)
+}
+# The ports an install is REACHED on, with the precedence the controller itself
+# applies (core/published-ports.ts, INS-43): host networking moves the listeners
+# (GPUK_PORT, GPUK_MTLS_PORT, GPUK_LISTEN_ADDR); anything else keeps the image's
+# fixed listeners and publishes them (GPUK_PUBLIC_*, then the ports table).
+published_ports() { # $1 = network mode → OURS_UI OURS_MTLS OURS_INF
+ if [ "$1" = "host" ]; then
+ OURS_UI="${E_PORT:-8080}"
+ OURS_MTLS="${E_MTLS_PORT:-8443}"
+ OURS_INF="${E_PROXY_PUBLIC_PORT:-${E_LISTEN_ADDR##*:}}"
+ else
+ OURS_UI="${E_PUBLIC_PORT:-${B_8080:-8080}}"
+ OURS_MTLS="${E_PUBLIC_MTLS_PORT:-${B_8443:-8443}}"
+ OURS_INF="${E_PROXY_PUBLIC_PORT:-${B_8200:-8200}}"
+ fi
+ [ -n "$OURS_INF" ] || OURS_INF=8200
+}
+EXISTING=""; OUR_PORTS=""; CONTAINER_NAME="gpu-kitchen"
+OURS_UI=""; OURS_MTLS=""; OURS_INF=""
+if [ -f "$MANIFEST" ] && [ ! -r "$MANIFEST" ]; then
+ EXISTING="unreadable"
+elif [ -f "$MANIFEST" ]; then
+ EXISTING="manifest"
+ _cn=$(manifest_str containerName); [ -z "$_cn" ] || CONTAINER_NAME="$_cn"
+ C_NETMODE=$(manifest_str networkMode)
+ E_PORT=$(manifest_str GPUK_PORT); E_MTLS_PORT=$(manifest_str GPUK_MTLS_PORT)
+ E_LISTEN_ADDR=$(manifest_str GPUK_LISTEN_ADDR)
+ E_PUBLIC_PORT=$(manifest_str GPUK_PUBLIC_PORT)
+ E_PUBLIC_MTLS_PORT=$(manifest_str GPUK_PUBLIC_MTLS_PORT)
+ E_PROXY_PUBLIC_PORT=$(manifest_str GPUK_PROXY_PUBLIC_PORT)
+ B_8080=$(manifest_binding 8080); B_8443=$(manifest_binding 8443); B_8200=$(manifest_binding 8200)
+ published_ports "${C_NETMODE:-host}"
+ OUR_PORTS="$OURS_UI $OURS_MTLS $OURS_INF"
+elif container_facts "$CONTAINER_NAME"; then
+ EXISTING="leftovers"
+ published_ports "$C_NETMODE"
+ OUR_PORTS="$OURS_UI $OURS_MTLS $OURS_INF"
+elif [ -f "$UNIT_DEST" ] || [ -x "$BIN_DEST" ] || [ -d "$DATA_ROOT/secrets" ]; then
+ EXISTING="leftovers"
+fi
+
+case "$EXISTING" in
+ manifest)
+ step "Existing install — $MANIFEST"
+ ok "image $(manifest_str image)"
+ _root=$(manifest_str dataRoot); _cache=$(manifest_str hostPath)
+ _cluster=$(manifest_str GPUK_CLUSTER); _profile=$(manifest_str GPUK_INSTALL_PROFILE)
+ [ "$DATA_ROOT_GIVEN" -eq 1 ] || [ -z "$_root" ] || DATA_ROOT="$_root"
+ [ "$CACHE_GIVEN" -eq 1 ] || [ -z "$_cache" ] || CACHE_DIR="$_cache"
+ [ "$CLUSTER_GIVEN" -eq 1 ] || [ -z "$_cluster" ] || CLUSTER="$_cluster"
+ [ "$PORT_GIVEN" -eq 1 ] || PORT="$OURS_UI"
+ [ "$MTLS_GIVEN" -eq 1 ] || MTLS_PORT="$OURS_MTLS"
+ [ "$INFERENCE_GIVEN" -eq 1 ] || INFERENCE_PORT="$OURS_INF"
+ case "$_profile" in
+ homelab|studio|enterprise|public) [ -n "$PROFILE" ] || PROFILE="$_profile" ;;
+ esac
+ ok "data root $DATA_ROOT"
+ ok "ports UI $OURS_UI, worker channel $OURS_MTLS, inference $OURS_INF"
+ [ -z "$_profile" ] || ok "profile $_profile"
+ ok "re-running updates it in place. Its settings are kept unless a flag says otherwise."
+ ;;
+ unreadable)
+ step "Existing install — $MANIFEST"
+ warn "present, but not readable from here: run as root to keep its settings"
+ ;;
+ leftovers)
+ step "Existing install — traces of a previous install, no manifest"
+ if [ -n "$C_STATUS" ]; then
+ ok "container $CONTAINER_NAME ($C_STATUS; UI $OURS_UI, worker channel $OURS_MTLS, inference $OURS_INF) — the install replaces it"
+ fi
+ [ ! -f "$UNIT_DEST" ] || ok "systemd unit $UNIT_DEST — rewritten"
+ [ ! -x "$BIN_DEST" ] || ok "daemon binary $BIN_DEST — replaced"
+ [ ! -d "$DATA_ROOT/secrets" ] || ok "data root $DATA_ROOT — reused, nothing in it is touched"
+ warn "without $MANIFEST no setting can be kept: the flags and the defaults apply"
+ ;;
+esac
+[ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf"
+
+if [ "$PORT" = "$MTLS_PORT" ] || [ "$PORT" = "$INFERENCE_PORT" ] || [ "$MTLS_PORT" = "$INFERENCE_PORT" ]; then
+ die "the UI, worker channel and inference ports must differ (got $PORT, $MTLS_PORT, $INFERENCE_PORT)"
+fi
+
# ── 1. Preflight ─────────────────────────────────────────────────────────────
# The same checks tools/provision-feeder.sh makes, minus the compose ones: the
# all-in-one image is driven by workerd through the plain docker CLI, so there is
@@ -173,7 +316,10 @@ else
step "Preflight — docker, NVIDIA driver, container toolkit"
-[ "$DRY_RUN" -eq 1 ] || [ "$(id -u)" -eq 0 ] \
+# Root, or a user-owned prefix (GPUK_ETC_DIR) — the same rule as gpuk's
+# need_root, and the only way the full path is testable without handing root
+# to a test suite.
+[ "$DRY_RUN" -eq 1 ] || [ "$(id -u)" -eq 0 ] || [ -w "$ETC_DIR" ] \
|| die "run as root: curl -fsSL … | sudo sh"
command -v curl >/dev/null 2>&1 || die "curl not found. Install curl first."
@@ -230,7 +376,16 @@ else
warn "only ${FREE_GB:-?}G free under $CACHE_PARENT. Model weights need 100G or more."
fi
-# ── Port conflicts (INS-46) ──
+fi # end preflight
+
+# ── Listening ports (INS-46) ──────────────────────────────────────────────────
+# Deliberately OUTSIDE the preflight branch: --skip-preflight skips docker, the
+# driver, the GPU smoke test and the disk (things a runner or a VM cannot have),
+# but a taken port is exactly as fatal there, and checking it costs nothing.
+# Skipping it here only moved the failure to gpuk's non-interactive refusal.
+if [ "$SKIP_PREFLIGHT" -eq 1 ]; then
+ step "Listening ports — checked even without the preflight"
+fi
# A taken port must fail HERE, before anything mutates the host — today's
# alternative is a 3-minute health-check timeout with zero diagnosis. Best-effort
# detection (ss, then netstat); neither present is a warn, never a false red.
@@ -258,82 +413,111 @@ port_busy() { # $1 = port → 0 iff something listens on TCP :$1
esac
}
-port_owner() { # $1 = port → best-effort process name (needs root for -p)
- case "$PORT_TOOL" in
- ss) ss -ltnpH "sport = :$1" 2>/dev/null \
- | sed -n 's/.*users:((\"\([^"]*\)\".*/\1/p' | head -1 ;;
- netstat) netstat -ltnp 2>/dev/null \
- | awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) {print $NF; exit}}' \
- | sed 's|^[0-9]*/||' ;;
- esac
+# Which container a listener belongs to, if any: the process's cgroup names the
+# container id (host networking — the listener IS the container's process), and a
+# bridged publication shows up as the container's port mapping in `docker ps`
+# (the host-side holder is docker-proxy, which says nothing by itself). "nginx"
+# is a riddle; "nginx in container gpu-kitchen-dev" is the answer.
+port_container() { # $1 = port, $2 = pid ("" if unknown) → container name or ""
+ command -v docker >/dev/null 2>&1 || return 0
+ if [ -n "$2" ] && [ -r "/proc/$2/cgroup" ]; then
+ _cid=$(sed -n 's#.*docker[-/]\([0-9a-f]\{64\}\).*#\1#p' "/proc/$2/cgroup" 2>/dev/null | head -1)
+ if [ -n "$_cid" ]; then
+ docker inspect -f '{{.Name}}' "$_cid" 2>/dev/null | sed 's|^/||'
+ return 0
+ fi
+ fi
+ docker ps --format '{{.Names}} {{.Ports}}' 2>/dev/null \
+ | awk -v p=":$1->" 'index($0, p) { print $1; exit }'
}
-manifest_ui_port() { # the port an existing install already owns, if any
- [ -f /etc/gpu-kitchen/manifest.json ] || return 0
- # Bridge publishes GPUK_PUBLIC_PORT over the fixed container 8080; host
- # networking moves the listener itself (GPUK_PORT). Same precedence as gpuk.
- _p=$(sed -n 's/.*"GPUK_PUBLIC_PORT"[[:space:]]*:[[:space:]]*"\([0-9]*\)".*/\1/p' \
- /etc/gpu-kitchen/manifest.json | head -1)
- [ -n "$_p" ] || _p=$(sed -n 's/.*"GPUK_PORT"[[:space:]]*:[[:space:]]*"\([0-9]*\)".*/\1/p' \
- /etc/gpu-kitchen/manifest.json | head -1)
- printf '%s' "$_p"
+port_owner() { # $1 = port → best-effort "process", "process in container NAME", or ""
+ _proc=""; _pid=""
+ case "$PORT_TOOL" in
+ ss)
+ _line=$(ss -ltnpH "sport = :$1" 2>/dev/null | head -1)
+ _proc=$(printf '%s' "$_line" | sed -n 's/.*users:((\"\([^"]*\)\".*/\1/p')
+ _pid=$(printf '%s' "$_line" | sed -n 's/.*pid=\([0-9]*\).*/\1/p')
+ ;;
+ netstat)
+ _field=$(netstat -ltnp 2>/dev/null \
+ | awk -v p="$1" '{n=split($4,a,":"); if (a[n]==p) {print $NF; exit}}')
+ case "$_field" in
+ */*) _pid=${_field%%/*}; _proc=${_field#*/} ;;
+ *) _proc="$_field" ;;
+ esac
+ ;;
+ esac
+ case "$_pid" in *[!0-9]*|"") _pid="" ;; esac
+ _ctr=$(port_container "$1" "$_pid")
+ if [ -n "$_ctr" ]; then
+ printf '%s' "${_proc:-a process} in container $_ctr"
+ else
+ printf '%s' "$_proc"
+ fi
+}
+
+# A port is ours when the existing install (step 0) already holds it: the
+# re-run replaces that container, so what it listens on is not a conflict.
+port_is_ours() { case " $OUR_PORTS " in *" $1 "*) return 0 ;; esac; return 1; }
+
+# Ports this run may not hand out twice: the three requested ones, plus every
+# alternative already accepted. Without it, a busy 8442 would be offered 8443
+# and collide with the worker channel one question later.
+RESERVED_PORTS="$PORT $MTLS_PORT $INFERENCE_PORT"
+port_available() { # $1 → free on the host AND not reserved by this run
+ case " $RESERVED_PORTS " in *" $1 "*) return 1 ;; esac
+ port_is_ours "$1" && return 1
+ ! port_busy "$1"
+}
+
+# resolve_port