diff --git a/gpuk b/gpuk index bde0736..dbc1d58 100755 --- a/gpuk +++ b/gpuk @@ -92,7 +92,9 @@ install_binary() { # [local-path] command -v curl >/dev/null || die "curl is required to download the binary" _url="$GPUK_RELEASE_BASE/$(arch_asset)" echo "==> downloading $_url" - curl -fsSL "$_url" -o "$BIN_DEST.new" + # Not -s: the binary is tens of MB and a silent download reads as a hang. + curl -fL --progress-bar "$_url" -o "$BIN_DEST.new" \ + || { rm -f "$BIN_DEST.new"; die "cannot download $_url"; } chmod 0755 "$BIN_DEST.new" mv "$BIN_DEST.new" "$BIN_DEST" elif [ -x "$BIN_DEST" ]; then @@ -170,6 +172,22 @@ port_owner() { # $1 = ss|netstat, $2 = port fi } +# This host's LAN IPv4 — what a browser, a worker or a container on the docker +# bridge dials to reach the machine. The route to a public address names the +# source the default route uses (never a docker or libvirt bridge); failing that, +# the first global address on a physical-looking interface; failing that, the +# resolver's word. Empty when nothing answers — the caller says so. Same +# derivation as dev.sh best_host_lan_ipv4 and install.sh's banner. +host_lan_ipv4() { + _ip=$(ip route get 1.1.1.1 2>/dev/null | sed -n 's/.* src \([0-9.]*\).*/\1/p' | head -1) + if [ -z "$_ip" ]; then + _ip=$(ip -o -4 addr show scope global 2>/dev/null \ + | awk '$2 !~ /^(virbr|docker|br-|veth|mpqemubr|tun|tap|vnet)/ {sub(/\/.*/, "", $4); print $4; exit}') + fi + [ -n "$_ip" ] || _ip=$(hostname -I 2>/dev/null | awk '{print $1}') + printf '%s' "$_ip" +} + seed_secrets() { # DATA_ROOT _sd="$1/secrets" mkdir -p "$_sd"; chmod 0700 "$_sd" @@ -307,7 +325,12 @@ hint_existing_caches() { cmd_install() { IMAGE=""; MODE="worker"; CLUSTER="default"; DATA_ROOT="/var/lib/gpu-kitchen" - CACHE_DIR=""; NETWORK="host"; CONTROLLER_URL=""; BIN_SRC=""; HEALTH_PORT="8001" + # bridge, not host (INS-49): on the host network every listener the image opens + # is a host-wide claim, and a box that already runs something on 3000 or 8000 + # killed Nitro. Bridged, only the three published ports touch the host; what + # that costs — the host daemon must be told its LAN address and the engines' + # docker network — is written to worker.env below. `--network host` remains. + CACHE_DIR=""; NETWORK="bridge"; CONTROLLER_URL=""; BIN_SRC=""; HEALTH_PORT="8001" # 1337, not 8080: kept in lockstep with install.sh's default (INS-01). The # worker channel and the inference endpoint move the same way (INS-46). ENROLL_TOKEN=""; HTTP_PORT="1337"; MTLS_PORT="8443"; INFERENCE_PORT="8200" @@ -315,10 +338,11 @@ cmd_install() { # GPUK_MODEL_TRANSFER_PORT and GPUK_HANDOVER_DATA_PORT from worker.env and # announces the first two to the controller, so moving them here is complete. DATA_PORT="8300"; TRANSFER_PORT="8301"; HANDOVER_PORT="8302" - PROFILE=""; DOMAIN=""; DRY_RUN=0 + PROFILE=""; DOMAIN=""; DRY_RUN=0; RESET_MANIFEST=0 while [ $# -gt 0 ]; do case "$1" in --image) IMAGE="$2"; shift 2 ;; + --reset-manifest) RESET_MANIFEST=1; shift ;; --mode) MODE="$2"; shift 2 ;; --cluster) CLUSTER="$2"; shift 2 ;; --profile) PROFILE="$2"; shift 2 ;; @@ -514,6 +538,23 @@ cmd_install() { write_controller_manifest # The host daemon and app container share DATA_ROOT. Only controller-mode # workerd gets this private bootstrap channel; remote workers stay tokenless. + # + # A bridged app container (INS-49) needs two more lines, read by the daemon + # from its OWN env — the daemon runs on the host, not in the container: + # - GPUK_DATA_HOST: the daemon dials the published mTLS port through docker + # NAT, so the controller sees it arrive from the bridge gateway (172.17.0.1) + # and would persist THAT as the node's address (WRK-93). The LAN address is + # what peers, the proxy and the UI must dial instead. + # - BACKEND_DOCKER_NETWORK: the engines the daemon launches join the app + # container's docker network, so the backend and gpuk-proxy reach them by + # container IP (apps/worker/src/modules/docker.rs, staging/engine.rs). + # Exactly what dev.sh hands the dev worker; host networking needs neither. + _data_host="" + if [ "$NETWORK" != "host" ]; then + _data_host="${GPUK_DATA_HOST:-$(host_lan_ipv4)}" + [ -n "$_data_host" ] || echo "==> warning: this host's LAN IPv4 could not be determined (no ip, no hostname -I)." \ + "Set GPUK_DATA_HOST= in $WORKER_ENV, or the controller will address its own worker through the docker bridge." + fi { echo "GPUK_CLUSTER=$CLUSTER" echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT" @@ -521,9 +562,13 @@ cmd_install() { echo "GPUK_CONTROLLER_URLS=wss://127.0.0.1:$MTLS_PORT" echo "GPUK_SELF_ENROLL_FILE=$SELF_ENROLL_FILE" echo "NODE_DISPLAY_NAME=$(hostname)" + if [ "$NETWORK" != "host" ]; then + [ -z "$_data_host" ] || echo "GPUK_DATA_HOST=$_data_host" + echo "BACKEND_DOCKER_NETWORK=$NETWORK" + fi } > "$WORKER_ENV" chmod 0600 "$WORKER_ENV" - echo "==> wrote $MANIFEST (controller: app container $IMAGE)" + echo "==> wrote $MANIFEST (controller: app container $IMAGE, network $NETWORK)" fi chmod 0600 "$MANIFEST" [ -z "$DOMAIN" ] || write_caddyfile @@ -561,11 +606,25 @@ cmd_install() { # ── Controller: start the daemon, then bring up the app container ── systemctl enable --now "$SERVICE_NAME" echo "==> $SERVICE_NAME enabled and started" - echo "==> applying manifest (first app-container start) ..." + # The pull is the long part of a first install (a multi-GB image) and workerd + # runs docker with its output captured, so pulling from inside `apply` is + # minutes of silence that read as a hang. Pull here, on the terminal, where + # docker's own per-layer progress is what the operator sees; `apply` then finds + # the image present and skips its pull (INS-03). + # Feedback only, never the verdict: `apply` pulls again whatever happened here + # and reports the cause itself (the CI shell gate installs a stub daemon against + # an image that does not exist anywhere — the daemon's pull is the one that counts). + if ! docker image inspect "$IMAGE" >/dev/null 2>&1; then + echo "==> pulling $IMAGE (docker shows the progress per layer) ..." + docker pull "$IMAGE" \ + || echo "==> the pull did not complete here; the daemon retries it during apply and reports the cause if it fails again" + fi + echo "==> applying manifest — starting the app container and waiting for its health check (up to 60s) ..." # No unix socket any more: workerd reconciles the app container in-process from # the on-disk manifest (there is no backend to relay through on the very first # boot). Steady-state updates go through the daemon over WS. - GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" apply || die "apply failed. Check: gpuk logs" + GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" apply \ + || die "the app container did not come up — the [worker] lines above say why (a [controller] FATAL line names the component and the fix). Full container logs: gpuk logs" echo echo "Done. The worker daemon is running and the app container is up." echo " Status : gpuk status App logs: gpuk logs Daemon: journalctl -u $SERVICE_NAME" @@ -605,12 +664,31 @@ INSTALL_ENV_KEYS="NODE_ENV,GPUK_MODE,GPUK_CLUSTER,GPUK_SELF_ENROLL_FILE,NODE_DIS # cluster, network, data root, ports, secret references, the primary cache disk, # the env keys above) and keeps the rest. The first write is the render as-is. write_manifest() { # $1 = render function - if [ -f "$MANIFEST" ]; then + if [ -f "$MANIFEST" ] && [ "$RESET_MANIFEST" -eq 1 ]; then + # The operator's explicit regeneration (WRK-191): the daemon archives the + # existing document — readable or not — writes the render as-is and NAMES what + # the archive carried that the render does not. The flags install.sh inherited + # from the old file (ports, profile, data root) are already in the render. + "$1" > "$MANIFEST.new" + chmod 0600 "$MANIFEST.new" + GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" install-manifest \ + --from "$MANIFEST.new" --reset >/dev/null \ + || { rm -f "$MANIFEST.new"; die "could not reset $MANIFEST — the [worker] line above names the cause"; } + rm -f "$MANIFEST.new" + echo "==> $MANIFEST rebuilt from this install's settings (--reset-manifest); the previous file is archived beside it" + elif [ -f "$MANIFEST" ]; then "$1" > "$MANIFEST.new" chmod 0600 "$MANIFEST.new" GPUK_MANIFEST_PATH="$MANIFEST" "$BIN_DEST" install-manifest \ --from "$MANIFEST.new" --own-env "$INSTALL_ENV_KEYS" >/dev/null \ - || { rm -f "$MANIFEST.new"; die "could not merge the new settings into $MANIFEST"; } + || { + rm -f "$MANIFEST.new" + die "could not merge the new settings into $MANIFEST — the [worker] line above names the cause. + A manifest written by an older release can carry a field this release removed. Re-run the same + command with --reset-manifest: the file is archived beside itself as manifest.json.before-reset.