#!/bin/sh # gpuk — installer + control CLI for the GPU Kitchen host daemon (gpu-kitchen-worker). # # Since C1 a compute node runs NO backend. The single privileged host component is # the Rust `gpu-kitchen-worker` daemon (workerd), running as root DIRECTLY on the # host (not in a container). It owns NVML clock/power locks, whitelisted host # browsing, the model-cache staging pipeline, the signed `gpu-kitchen-bench` # runner, and — on a controller node — the app container's lifecycle (create, # health-gate, roll back, pull+recreate to update) via its manifest. There is no # separate host daemon and no unix control socket: workerd is driven over the # wss+mTLS channel by the controller, and locally by this CLI. # # Two roles: # worker a headless compute node. Installs the binary + systemd unit + a # manifest (cacheDisks) with NO app container, NO # Postgres, NO docker app image. It ENROLLS over mTLS with a # enrollment token, or waits for LAN discovery admission, then runs. # controller the full app + UI. Installs the binary + systemd unit + an # app-container manifest (image, ports, env, secretsRef) that workerd # applies (`gpu-kitchen-worker apply`, then cmd:host_apply in steady # state). # # Install (root): # # controller: # curl -fsSL https://repo.byterain.io/gpukitchen/channel/raw/branch/main/gpuk \ # | sudo sh -s -- install --mode controller \ # --image repo.byterain.io/gpukitchen/gpukitchen-controller:vX.Y.Z # # worker (enroll against a controller with a single-use token from its UI): # sudo ./deployments/install/gpuk install --mode worker \ # --controller wss://controller.lan:8443 --enroll-token gk_enroll_... \ # --cache-dir /mnt/models --binary apps/worker/target/release/gpu-kitchen-worker # # Control: # gpuk status | apply | update | enroll | logs | manifest | uninstall # set -eu # ── Paths ──────────────────────────────────────────────────────────────────── # Overridable, so the daemon can be driven against a prefix a normal user owns — # the only way any of this is testable without handing a test suite root on the # host. Unset (the real install) they are exactly the systemd defaults workerd uses # (apps/worker/src/module.rs, apps/worker/src/identity.rs). BIN_DEST="${GPUK_BIN_DEST:-/usr/local/bin/gpu-kitchen-worker}" ETC_DIR="${GPUK_ETC_DIR:-/etc/gpu-kitchen}" MANIFEST="$ETC_DIR/manifest.json" IDENTITY_DIR="${GPUK_IDENTITY_DIR:-$ETC_DIR/identity}" MACHINE_ID_FILE="${GPUK_MACHINE_ID_FILE:-$ETC_DIR/machine-id}" WORKER_ENV="$ETC_DIR/worker.env" SERVICE_NAME="gpu-kitchen-worker" UNIT_DEST="${GPUK_UNIT_DEST:-/etc/systemd/system/$SERVICE_NAME.service}" # Where to download the binary from when no --binary is given. GPUK_RELEASE_BASE="${GPUK_RELEASE_BASE:-}" die() { echo "gpuk: $*" >&2; exit 1; } # Root, or able to do the job anyway. Fail with a clear message BEFORE touching # /etc, /usr/local/bin or systemd. Against a user-owned prefix it simply works. need_root() { [ "$(id -u)" -eq 0 ] && return 0 [ -w "$ETC_DIR" ] && return 0 die "this command must run as root (use sudo)" } # ── JSON helpers (controlled inputs: paths + identifiers) ────────────────────── json_str() { printf '%s' "$1" | sed 's/\\/\\\\/g; s/"/\\"/g'; } json_array() { # args → ["a","b",...] _out="" for _p in "$@"; do _e=$(json_str "$_p") if [ -z "$_out" ]; then _out="\"$_e\""; else _out="$_out,\"$_e\""; fi done printf '[%s]' "$_out" } # ── install ──────────────────────────────────────────────────────────────────── arch_asset() { case "$(uname -m)" in x86_64|amd64) echo "gpu-kitchen-worker-x86_64" ;; aarch64|arm64) echo "gpu-kitchen-worker-aarch64" ;; *) die "unsupported architecture: $(uname -m)" ;; esac } install_binary() { # [local-path] if [ -n "${1:-}" ]; then [ -f "$1" ] || die "binary not found: $1" install -m 0755 "$1" "$BIN_DEST" echo "==> installed $BIN_DEST from $1" elif [ -n "$GPUK_RELEASE_BASE" ]; then command -v curl >/dev/null || die "curl is required to download the binary" _url="$GPUK_RELEASE_BASE/$(arch_asset)" echo "==> downloading $_url" curl -fsSL "$_url" -o "$BIN_DEST.new" chmod 0755 "$BIN_DEST.new" mv "$BIN_DEST.new" "$BIN_DEST" elif [ -x "$BIN_DEST" ]; then echo "==> reusing existing $BIN_DEST" else die "no binary: pass --binary or set GPUK_RELEASE_BASE=" fi } gen_secret() { head -c 32 /dev/urandom | base64 | tr '+/' '-_' | tr -d '='; } seed_secrets() { # DATA_ROOT _sd="$1/secrets" mkdir -p "$_sd"; chmod 0700 "$_sd" [ -f "$_sd/encryption_key" ] || { umask 077; gen_secret > "$_sd/encryption_key"; } [ -f "$_sd/node_id" ] || { umask 077; (cat /proc/sys/kernel/random/uuid 2>/dev/null || uuidgen) > "$_sd/node_id"; } # No nominal first-run password is generated. An operator may pre-provision # this documented break-glass file; only then is it injected into the app. [ ! -f "$_sd/bootstrap_admin_password" ] || chmod 0600 "$_sd/bootstrap_admin_password" } prepare_claim_code() { # The file is the operator's recoverable proof of machine possession. Create # it once, preserve it across reinstalls, and let the backend unlink it after # the atomic first-account claim. A missing file beside an existing manifest # therefore means "consumed", never "rotate the credential". if [ -f "$CLAIM_CODE_FILE" ]; then chmod 0600 "$CLAIM_CODE_FILE" elif [ ! -f "$MANIFEST" ]; then umask 077 gen_secret > "$CLAIM_CODE_FILE" chmod 0600 "$CLAIM_CODE_FILE" fi } # Write the systemd unit generated from the worker's canonical template. The # generator injects the Rust lock-contention exit code here too, so the binary # and systemd restart policy cannot silently drift apart. write_unit() { # BEGIN GENERATED WORKER SYSTEMD UNIT cat > "$UNIT_DEST" <`. Kept aligned with # deployments/controller/Caddyfile.example (the compose variant); this copy # targets the all-in-one image, where nginx on the UI port is the single front # door (INS-09) so one upstream carries pages, /api and the /ws upgrade alike. # We write a file and NOTHING more: no package install, no service start, no # other program's config read or touched — putting the proxy in service stays an # operator act (OPS-13), and its presence stays unverifiable (OPS-67). write_caddyfile() { mkdir -p "$DATA_ROOT/caddy" cat > "$DATA_ROOT/caddy/Caddyfile" <worker data transfers (:8300): LAN-only by contract (OPS-68). $DOMAIN { encode zstd gzip reverse_proxy localhost:$HTTP_PORT } # OpenAI-compatible inference endpoint (gpuk-proxy) — uncomment when inference # clients live beyond the trusted LAN; TLS keeps their API keys off the wire. # # inference.$DOMAIN { # encode zstd gzip # reverse_proxy localhost:8200 # } CADDY chmod 0644 "$DATA_ROOT/caddy/Caddyfile" echo "==> wrote $DATA_ROOT/caddy/Caddyfile (filled TLS reverse-proxy example for $DOMAIN)" } # Pre-existing model caches (B81): a server that installs GPU Kitchen usually already # holds tens or hundreds of GB of weights. We print a hint and nothing more — cache # questions belong to the first-run wizard (INS-48, REG-41), never to the CLI; no # config of any other program is read or touched, and referencing a cache stays an # explicit choice made in the UI. hint_existing_caches() { hec_chosen=$(readlink -f "$1" 2>/dev/null || echo "$1") hec_found="" for hec_dir in /root/.cache/huggingface /home/*/.cache/huggingface "${HF_HOME:-}"; do [ -n "$hec_dir" ] || continue [ -d "$hec_dir/hub" ] || continue hec_real=$(readlink -f "$hec_dir" 2>/dev/null || echo "$hec_dir") [ "$hec_real" != "$hec_chosen" ] || continue # Hub layout only (models--*) — matches what the scan can actually reference. ls -d "$hec_dir"/hub/models--* >/dev/null 2>&1 || continue hec_found="$hec_found $hec_real" done [ -n "$hec_found" ] || return 0 echo for hec_dir in $hec_found; do echo "==> existing model cache found at $hec_dir" done echo " GPU Kitchen will offer to reuse those models at first launch, and any" echo " time from Nodes & GPU -> Storage. A reused cache is referenced in" echo " place. Nothing is moved or deleted." } cmd_install() { IMAGE=""; MODE="worker"; CLUSTER="default"; DATA_ROOT="/var/lib/gpu-kitchen" CACHE_DIR=""; NETWORK="host"; CONTROLLER_URL=""; BIN_SRC=""; HEALTH_PORT="8001" # 1337, not 8080: kept in lockstep with install.sh's default (INS-01). ENROLL_TOKEN=""; HTTP_PORT="1337"; PROFILE=""; DOMAIN=""; DRY_RUN=0 while [ $# -gt 0 ]; do case "$1" in --image) IMAGE="$2"; shift 2 ;; --mode) MODE="$2"; shift 2 ;; --cluster) CLUSTER="$2"; shift 2 ;; --profile) PROFILE="$2"; shift 2 ;; --domain) DOMAIN="$2"; shift 2 ;; --data-root) DATA_ROOT="$2"; shift 2 ;; --cache-dir) CACHE_DIR="$2"; shift 2 ;; --network) NETWORK="$2"; shift 2 ;; --controller) CONTROLLER_URL="$2"; shift 2 ;; # `--token` is an ENROLLMENT token (single-use or shared), NOT a bearer: # the worker↔controller channel is cert-only since D11. `--token` is kept as # a spelling of `--enroll-token`. --enroll-token|--token) ENROLL_TOKEN="$2"; shift 2 ;; --health-port) HEALTH_PORT="$2"; shift 2 ;; --http-port) HTTP_PORT="$2"; shift 2 ;; --binary) BIN_SRC="$2"; shift 2 ;; --dry-run) DRY_RUN=1; shift ;; *) die "unknown install option: $1" ;; esac done # Roles, as the backend names them (multi-server/config.ts): "controller" (full # app + UI, accepts workers) and "worker" (headless compute node). Historical # spellings still work. case "$MODE" in controller|server|standalone|manager) MODE="controller" ;; worker|agent) MODE="worker" ;; *) die "--mode must be controller or worker (got '$MODE')" ;; esac if [ "$MODE" = "worker" ] && [ -n "$PROFILE" ]; then die "--profile applies only to --mode controller; workers do not have an installation profile" fi case "$PROFILE" in ""|homelab|studio|enterprise|public) ;; *) die "--profile must be homelab, studio, enterprise or public (got '$PROFILE')" ;; esac if [ -n "$DOMAIN" ]; then [ "$PROFILE" = "public" ] \ || die "--domain applies only to --profile public (it fills the TLS reverse-proxy example)" case "$DOMAIN" in *[!A-Za-z0-9.-]*) die "--domain must be a bare domain name (got '$DOMAIN')" ;; esac fi if [ "$MODE" = "controller" ]; then if [ -z "$IMAGE" ] && [ "$DRY_RUN" -eq 1 ]; then IMAGE="example.invalid/gpukitchen-controller:v0.0.0-dry-run" fi [ -n "$IMAGE" ] || die "--image registry/gpukitchen-controller: is required for a controller" case "$IMAGE" in *@sha256:*) ;; *:latest) die "refusing floating image tag '$IMAGE' — use an explicit release tag or digest" ;; *) _image_tag="${IMAGE##*:}" case "$_image_tag" in "$IMAGE"|*/*) die "controller image must carry an explicit tag or sha256 digest (got '$IMAGE')" ;; esac ;; esac fi [ -n "$CACHE_DIR" ] || CACHE_DIR="$DATA_ROOT/hf" CACHE_JSON="[{\"hostPath\":\"$(json_str "$CACHE_DIR")\",\"shared\":false}]" SELF_ENROLL_FILE="$DATA_ROOT/self-enroll-token" BOOTSTRAP_PASSWORD_FILE="$DATA_ROOT/secrets/bootstrap_admin_password" CLAIM_CODE_FILE="$DATA_ROOT/secrets/claim_code" CLAIM_CODE_AVAILABLE=0 if [ "$DRY_RUN" -eq 1 ]; then [ "$MODE" != "controller" ] || CLAIM_CODE_AVAILABLE=1 echo "==> dry run: no file, service or container was changed" if [ "$MODE" = "worker" ]; then render_worker_manifest; else render_controller_manifest; fi [ -z "$DOMAIN" ] || echo "==> dry run: would write $DATA_ROOT/caddy/Caddyfile for $DOMAIN" return 0 fi # ── Port conflicts (INS-46) — the mutator's own guard ──────────────────────── # install.sh's preflight already checks these, and on a terminal it can offer # an alternative port. gpuk is the actual mutator and contributors call it # DIRECTLY, so it re-checks and refuses, non-interactively. Same helpers as # install.sh (both scripts ship standalone from the channel). A listener owned # by an existing install is not a conflict: a manifest on disk means the # re-run is the update path, and every checked port is then our own. if [ ! -f "$MANIFEST" ]; then _port_tool="" if command -v ss >/dev/null 2>&1; then _port_tool="ss" elif command -v netstat >/dev/null 2>&1; then _port_tool="netstat"; fi if [ -z "$_port_tool" ]; then echo "==> warning: cannot check for port conflicts (no ss or netstat)" else if [ "$MODE" = "controller" ]; then set -- "$HTTP_PORT" 8443 8200 else # Worker data-plane ports (OPS-68) plus the local health listener. set -- "$HEALTH_PORT" 8300 8301 8302 fi for _p in "$@"; do _busy=1 case "$_port_tool" in ss) [ -n "$(ss -ltnH "sport = :$_p" 2>/dev/null)" ] || _busy=0 ;; netstat) netstat -ltn 2>/dev/null \ | awk -v p="$_p" '{n=split($4,a,":"); if (a[n]==p) found=1} END{exit found?0:1}' \ || _busy=0 ;; esac [ "$_busy" -eq 0 ] || die "port $_p is already in use. Free it first, then run the install again." done fi fi need_root install_binary "$BIN_SRC" mkdir -p "$ETC_DIR" "$DATA_ROOT" "$CACHE_DIR" seed_secrets "$DATA_ROOT" if [ "$MODE" = "controller" ]; then prepare_claim_code [ ! -s "$CLAIM_CODE_FILE" ] || CLAIM_CODE_AVAILABLE=1 fi umask 077 if [ "$MODE" = "worker" ]; then write_worker_manifest # Optional env overrides for the daemon (cluster grouping, display name, an # explicit controller URL). The enrolled identity carries the controller URL # + CA too — this is belt-and-braces / pre-enroll discovery grouping. { echo "GPUK_CLUSTER=$CLUSTER" echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT" echo "GPUK_MACHINE_ID_FILE=$MACHINE_ID_FILE" echo "NODE_DISPLAY_NAME=$(hostname)" [ -n "$CONTROLLER_URL" ] && echo "GPUK_CONTROLLER_URLS=$CONTROLLER_URL" } > "$WORKER_ENV" chmod 0600 "$WORKER_ENV" echo "==> wrote $MANIFEST (worker: no app container)" else write_controller_manifest # The host daemon and app container share DATA_ROOT. Only controller-mode # workerd gets this private bootstrap channel; remote workers stay tokenless. { echo "GPUK_CLUSTER=$CLUSTER" echo "GPUK_WORKER_HEALTH_PORT=$HEALTH_PORT" echo "GPUK_MACHINE_ID_FILE=$MACHINE_ID_FILE" echo "GPUK_CONTROLLER_URLS=wss://127.0.0.1:8443" echo "GPUK_SELF_ENROLL_FILE=$SELF_ENROLL_FILE" echo "NODE_DISPLAY_NAME=$(hostname)" } > "$WORKER_ENV" chmod 0600 "$WORKER_ENV" echo "==> wrote $MANIFEST (controller: app container $IMAGE)" fi chmod 0600 "$MANIFEST" [ -z "$DOMAIN" ] || write_caddyfile write_unit systemctl daemon-reload echo "==> wrote $UNIT_DEST" # ── Worker: token enrollment before start, or unattended LAN discovery ── if [ "$MODE" = "worker" ]; then if [ -n "$ENROLL_TOKEN" ]; then [ -n "$CONTROLLER_URL" ] || die "--controller wss://: is required to enroll" echo "==> enrolling against $CONTROLLER_URL ..." GPUK_IDENTITY_DIR="$IDENTITY_DIR" GPUK_MACHINE_ID_FILE="$MACHINE_ID_FILE" "$BIN_DEST" enroll \ --controller "$CONTROLLER_URL" --token "$ENROLL_TOKEN" \ || die "enrollment failed (bad/expired token, or controller unreachable)" else if [ -n "$CONTROLLER_URL" ]; then echo "==> no token given: the daemon will request admission from $CONTROLLER_URL and wait" echo " for automatic admission or administrator approval." else echo "==> no token or controller given: the daemon will discover its cluster on the LAN and wait" echo " for automatic admission or administrator approval." fi fi systemctl enable --now "$SERVICE_NAME" echo "==> $SERVICE_NAME enabled and started" echo echo "Done. The worker daemon is running${ENROLL_TOKEN:+ and enrolled}." echo " Status : gpuk status Logs: gpuk logs" hint_existing_caches "$CACHE_DIR" return 0 fi # ── Controller: start the daemon, then bring up the app container ── systemctl enable --now "$SERVICE_NAME" echo "==> $SERVICE_NAME enabled and started" echo "==> applying manifest (first app-container start) ..." # No unix socket any more: workerd reconciles the app container in-process from # the on-disk manifest (there is no backend to relay through on the very first # boot). Steady-state updates go through the daemon over WS. "$BIN_DEST" apply || die "apply failed. Check: gpuk logs" echo echo "Done. The worker daemon is running and the app container is up." echo " Status : gpuk status App logs: gpuk logs Daemon: journalctl -u $SERVICE_NAME" hint_existing_caches "$CACHE_DIR" } # Worker manifest: cacheDisks and an EMPTY image so # has_app_container() is false (apps/worker/src/modules/manifest.rs). No ports, no env, # no secretsRef — a worker runs no app container. render_worker_manifest() { cat < "$MANIFEST"; } # Controller manifest: the declarative app-container description workerd applies. render_controller_manifest() { # NOTE there is deliberately no PORT here: PORT is the backend's own port, a # loopback-only 8000 behind nginx in the all-in-one image. What the outside world # dials is GPUK_PUBLIC_PORT (the published host port), which falls back to nginx's # own GPUK_PORT when nothing republishes it. EXTRA_ENV="" EXTRA_ENV="$EXTRA_ENV,\"GPUK_DATA_ROOT\":\"$(json_str "$DATA_ROOT")\"" [ -z "$PROFILE" ] || EXTRA_ENV="$EXTRA_ENV,\"GPUK_INSTALL_PROFILE\":\"$(json_str "$PROFILE")\"" if [ "$PROFILE" = "public" ]; then EXTRA_ENV="$EXTRA_ENV,\"GPUK_HSTS\":\"true\"" EXTRA_ENV="$EXTRA_ENV,\"GPUK_SESSION_COOKIE_SECURE\":\"true\"" fi BOOTSTRAP_SECRET_JSON="" if [ -s "$BOOTSTRAP_PASSWORD_FILE" ]; then EXTRA_ENV="$EXTRA_ENV,\"GPUK_BOOTSTRAP_MUST_CHANGE\":\"1\"" BOOTSTRAP_SECRET_JSON=",\"GPUK_BOOTSTRAP_ADMIN_PASSWORD\":\"$(json_str "$BOOTSTRAP_PASSWORD_FILE")\"" fi CLAIM_SECRET_JSON="" if [ "$CLAIM_CODE_AVAILABLE" -eq 1 ]; then CLAIM_SECRET_JSON=",\"GPUK_CLAIM_CODE\":\"$(json_str "$CLAIM_CODE_FILE")\"" fi [ "$NETWORK" = "host" ] || EXTRA_ENV="$EXTRA_ENV,\"BACKEND_DOCKER_NETWORK\":\"$(json_str "$NETWORK")\"" # Published != bound (INS-43). On a bridged network the container keeps the image's # FIXED listeners — nginx 8080, worker mTLS 8443 — and --http-port only moves the HOST # side of the publication; Settings -> Network moves it later by patching this same # manifest, so the container port must never become a variable. With host networking # nothing is published and the listener itself takes the port. PORTS_JSON="{}" if [ "$NETWORK" = "host" ]; then EXTRA_ENV="$EXTRA_ENV,\"GPUK_PORT\":\"$(json_str "$HTTP_PORT")\"" else EXTRA_ENV="$EXTRA_ENV,\"GPUK_PORT\":\"8080\"" EXTRA_ENV="$EXTRA_ENV,\"GPUK_PUBLIC_PORT\":\"$(json_str "$HTTP_PORT")\"" PORTS_JSON="{\"8080\":$HTTP_PORT,\"8200\":8200,\"8443\":8443}" fi cat < "$MANIFEST"; } # ── control subcommands ──────────────────────────────────────────────────────── container_name() { sed -n 's/.*"containerName"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1 } manifest_data_root() { sed -n 's/.*"dataRoot"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1 } manifest_image() { sed -n 's/.*"image"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$MANIFEST" 2>/dev/null | head -1 } health_port() { sed -n 's/.*GPUK_WORKER_HEALTH_PORT=\([0-9]*\).*/\1/p' "$WORKER_ENV" 2>/dev/null | head -1 } # Split an image reference into repository and tag. # # `${img%%:*}` cuts at the FIRST colon and is WRONG: # `registry.internal:5000/gpuk/controller:v1.2.3` would yield repo # `registry.internal`. The colon in a registry's host:port is not a tag separator. # Rule: it is a tag only if the last colon comes after the last slash. # (Same logic as apps/controller/api/src/core/release-channel.ts — see its unit tests.) image_repo() { # A digest suffix (…@sha256:…) never carries the repo; drop it, then apply # the tag logic — `repo:tag@sha256:…` and `repo@sha256:…` both reduce right. set -- "${1%@*}" _t="${1##*:}" case "$_t" in "$1") printf '%s' "$1" ;; # no colon at all → untagged */*) printf '%s' "$1" ;; # the last colon is inside a path → host:port, untagged *) printf '%s' "${1%:*}" ;; esac } image_tag() { # `repo:tag@sha256:…` keeps the human-readable tag next to the content pin — # docker resolves by digest and ignores the tag. Strip the digest, then parse. set -- "${1%@*}" _t="${1##*:}" case "$_t" in "$1") return ;; */*) return ;; *) printf '%s' "$_t" ;; esac } # The release channel: one flat JSON document served next to the installer. The # UI's update check reads the same one (apps/controller/api/src/core/release-channel.ts) — one # source of truth. Interim default: the public Gitea channel repo — flips to # https://gpu.kitchen/latest.json once the hub exists (specs/developpement/ci-cd.md). CHANNEL_URL="${GPUK_CHANNEL_URL:-https://repo.byterain.io/gpukitchen/channel/raw/branch/main/latest.json}" # The release public key pinned in THIS copy of gpuk (OPS-20). The channel copy # gets the real key substituted at publish time; the operator override # (GPUK_UPDATE_PUBKEY) covers a self-hosted channel with its own keypair. The # first install fetched gpuk itself over HTTPS from the channel — that moment is # trust-on-first-use, like a worker's enrolment token pin; every later `update` # is verified against the key pinned HERE, so whoever controls latest.json can # no longer pick what an existing install runs. CHANNEL_PUBKEY="${GPUK_UPDATE_PUBKEY:-RWQ7BKXJqGX2jdKXu1GxeSPVAN3JDRTefpImM/mFRjtFwq4E7mhnQtA7}" # Fetch latest.json AND its minisign signature, verify, and leave the verified # document at $CHANNEL_DOC. Fail-closed: no signature, bad signature, no # minisign CLI or no pinned key are all fatal — GPUK_CHANNEL_INSECURE=1 is the # explicit, logged opt-out (a private mirror that does not sign). CHANNEL_DOC="" channel_fetch() { command -v curl >/dev/null 2>&1 || return 1 CHANNEL_DOC=$(mktemp) || return 1 curl -fsSL --max-time 20 "$CHANNEL_URL" -o "$CHANNEL_DOC" 2>/dev/null || return 1 if [ "${GPUK_CHANNEL_INSECURE:-}" = "1" ]; then echo "WARNING: GPUK_CHANNEL_INSECURE=1 — release channel signature NOT verified" >&2 return 0 fi case "$CHANNEL_PUBKEY" in ""|RWQ7BKXJqGX2jdKXu1GxeSPVAN3JDRTefpImM/mFRjtFwq4E7mhnQtA7*) die "this gpuk carries no pinned release public key — set GPUK_UPDATE_PUBKEY (the minisign public-key line), or GPUK_CHANNEL_INSECURE=1 to skip verification" ;; esac command -v minisign >/dev/null 2>&1 \ || die "minisign is required to verify the release channel (apt install minisign), or set GPUK_CHANNEL_INSECURE=1" _sig=$(mktemp) if ! curl -fsSL --max-time 20 "${CHANNEL_URL}.minisig" -o "$_sig" 2>/dev/null; then rm -f "$_sig" die "no signature at ${CHANNEL_URL}.minisig — refusing an unsigned channel document (OPS-20)" fi if ! minisign -Vq -m "$CHANNEL_DOC" -x "$_sig" -P "$CHANNEL_PUBKEY" >/dev/null 2>&1; then rm -f "$_sig" die "latest.json signature verification FAILED — refusing the channel document (OPS-20)" fi rm -f "$_sig" } channel_field() { sed -n 's/.*"'"$2"'"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$1" | head -1 } channel_version() { channel_fetch || return 1 channel_field "$CHANNEL_DOC" version } # ── backup ───────────────────────────────────────────────────────────────────── # OPS-10: the embedded Postgres is only backed up COLD — hot-copying pgdata with # a file tool is forbidden (torn pages). Order matters: stop the DAEMON first # (its reconciler would immediately restart a stopped app container), then the # container, snapshot, and restarting the service re-applies the manifest. # An install on an external DATABASE_URL is refused here: gpuk only owns the # embedded pgdata — back the real database up with pg_dump/backup-compose.sh. # The finished directory still has to be copied to encrypted off-host storage, # next to the recovery set (OPS-04/OPS-05: ENCRYPTION_KEY above all). cmd_backup() { need_root _img=$(manifest_image) [ -n "$_img" ] || die "this is a worker node — no controller data to back up here" if grep -q '"DATABASE_URL"' "$MANIFEST" 2>/dev/null; then die "this install uses an external DATABASE_URL — back THAT database up (pg_dump, or the compose procedures in specs/plateforme/operations.md); gpuk backup only snapshots the embedded pgdata" fi _root=$(manifest_data_root) [ -n "$_root" ] || die "no dataRoot in $MANIFEST" [ -d "$_root/pgdata" ] || die "no embedded pgdata under $_root — nothing to snapshot" command -v sha256sum >/dev/null 2>&1 || die "sha256sum is required" _out="${1:-$_root/backups/$(date -u +%Y%m%dT%H%M%SZ)}" [ -e "$_out" ] && die "refusing to overwrite existing $_out" mkdir -p "$(dirname "$_out")" _tmp="$_out.partial" rm -rf "$_tmp"; mkdir -p "$_tmp" _cn=$(container_name) echo "==> Stopping $SERVICE_NAME (its reconciler would restart the container mid-snapshot)..." systemctl stop "$SERVICE_NAME" || die "could not stop $SERVICE_NAME" _restart_daemon() { systemctl start "$SERVICE_NAME" 2>/dev/null || true; } trap _restart_daemon EXIT if [ -n "$_cn" ]; then echo "==> Stopping $_cn (cold snapshot — OPS-10)..." docker stop "$_cn" >/dev/null 2>&1 || true _state=$(docker inspect -f '{{.State.Status}}' "$_cn" 2>/dev/null || echo absent) case "$_state" in running) die "container $_cn is still running — refusing a hot snapshot" ;; esac fi echo "==> Snapshotting $_root/pgdata..." tar -C "$_root" -czf "$_tmp/pgdata.tar.gz" pgdata || die "snapshot failed" cp "$MANIFEST" "$_tmp/host-manifest.json" 2>/dev/null || true { echo "{" echo " \"created_utc\": \"$(date -u +%Y-%m-%dT%H:%M:%SZ)\"," echo " \"image\": \"$(json_str "$_img")\"," echo " \"data_root\": \"$(json_str "$_root")\"," echo " \"kind\": \"cold-pgdata-snapshot\"" echo "}" } > "$_tmp/backup-manifest.json" (cd "$_tmp" && sha256sum ./* > SHA256SUMS) || die "checksums failed" chmod 0700 "$_tmp" mv "$_tmp" "$_out" echo "==> Restarting $SERVICE_NAME (re-applies the manifest, container included)..." systemctl start "$SERVICE_NAME" || die "could not restart $SERVICE_NAME — start it manually" trap - EXIT echo "backup: $_out" echo "Copy it to encrypted OFF-HOST storage together with the recovery set" echo "(ENCRYPTION_KEY above all — without it the data is unrecoverable, OPS-04)." echo "A physical pgdata restore requires the same Postgres major and a throwaway" echo "host rehearsal first (OPS-10)." } # ── channel ──────────────────────────────────────────────────────────────────── # Diagnostic (no root): fetch + VERIFY the channel document, print what it # offers. Exercises exactly the trust chain `update` relies on — the CI probes # it with a throwaway keypair, an operator uses it to debug a mirror. cmd_channel() { channel_fetch || die "cannot fetch $CHANNEL_URL" echo "channel : $CHANNEL_URL" echo "version : $(channel_field "$CHANNEL_DOC" version)" _d=$(channel_field "$CHANNEL_DOC" controllerImageDigest) [ -n "$_d" ] && echo "digest : $_d" _d=$(channel_field "$CHANNEL_DOC" controllerImageDigestEnterprise) [ -n "$_d" ] && echo "digest ee : $_d" if [ "${GPUK_CHANNEL_INSECURE:-}" = "1" ]; then echo "signature : SKIPPED (GPUK_CHANNEL_INSECURE=1)" else echo "signature : verified" fi } # ── status ───────────────────────────────────────────────────────────────────── # No control socket any more. Status = the systemd unit state + the daemon's own # /health endpoint + whether a worker has enrolled (identity present). cmd_status() { _active=$(systemctl is-active "$SERVICE_NAME" 2>/dev/null || true) echo "service : $_active" if [ -f "$IDENTITY_DIR/identity.json" ]; then echo "enrolled : yes ($IDENTITY_DIR)" else echo "enrolled : no (daemon waits for LAN admission; token enrollment is also available)" fi _hp=$(health_port); [ -n "$_hp" ] || _hp=8001 if command -v curl >/dev/null 2>&1; then _h=$(curl -fsS --max-time 3 "http://127.0.0.1:$_hp/health" 2>/dev/null || true) [ -n "$_h" ] && echo "health : $_h" || echo "health : (no answer on :$_hp)" fi _img=$(manifest_image) if [ -n "$_img" ]; then echo "app image : $_img" echo "app cont. : $(docker inspect -f '{{.State.Status}}' "$(container_name)" 2>/dev/null || echo 'not running')" else echo "role : worker (no app container)" fi } # ── enroll ───────────────────────────────────────────────────────────────────── cmd_enroll() { need_root _url=""; _tok="" while [ $# -gt 0 ]; do case "$1" in --controller) _url="$2"; shift 2 ;; --enroll-token|--token) _tok="$2"; shift 2 ;; *) die "unknown enroll option: $1" ;; esac done [ -n "$_url" ] || die "enroll needs --controller wss://:" [ -n "$_tok" ] || die "enroll needs --token gk_enroll_..." GPUK_IDENTITY_DIR="$IDENTITY_DIR" GPUK_MACHINE_ID_FILE="$MACHINE_ID_FILE" \ "$BIN_DEST" enroll --controller "$_url" --token "$_tok" systemctl restart "$SERVICE_NAME" 2>/dev/null || true } # ── apply ────────────────────────────────────────────────────────────────────── # Controller: reconcile the app container from the manifest, in-process. A worker # has no app container — `apply` there is a no-op with a clear message. cmd_apply() { need_root "$BIN_DEST" apply } # ── update ───────────────────────────────────────────────────────────────────── # Controller: decide WHICH image TAG to pin, write it into the manifest, then let # workerd pull + recreate (health-gate + rollback are the daemon's — apply()). # Worker: the signed-binary self-update is DRIVEN FROM THE CONTROLLER (its Update # button → POST /api/nodes/:id/host/update → cmd:host_update → verified swap). # There is no local unverified swap path. cmd_update() { need_root _want=""; _check=0 while [ $# -gt 0 ]; do case "$1" in --version) _want="$2"; shift 2 ;; --check) _check=1; shift ;; *) die "unknown update option: $1" ;; esac done _image=$(manifest_image) if [ -z "$_image" ]; then echo "This is a worker node. Worker self-update is driven from the controller UI" echo "(the node's Update button), which pushes a minisign-verified binary swap." return 0 fi _repo=$(image_repo "$_image") _current=$(image_tag "$_image") [ -n "$_current" ] || _current="(untagged)" if [ "$_check" -eq 1 ]; then _latest=$(channel_version) || true echo "installed : $_current" if [ -z "$_latest" ]; then echo "available : unknown (cannot reach $CHANNEL_URL)" exit 1 fi echo "available : $_latest" if [ "$_latest" = "$_current" ]; then echo "up to date."; else echo "run 'gpuk update' to move to $_latest"; fi return 0 fi _digest="" if [ -z "$_want" ]; then # channel_fetch runs in THIS shell (not a $(…) subshell) so a signature # failure is fatal here — fail-closed — and $CHANNEL_DOC survives. The # verified signature closes the document half of OPS-20; the digest read # from it pins CONTENT, closing the mutable-tag half. if channel_fetch; then _want=$(channel_field "$CHANNEL_DOC" version) fi if [ -z "$_want" ]; then echo "==> cannot reach $CHANNEL_URL — re-applying the pinned $_current" "$BIN_DEST" apply return 0 fi # Pick the digest matching the installed edition by image basename — the # repo itself may be a mirror, the basename is the edition marker. case "${_repo##*/}" in *-ee) _digest=$(channel_field "$CHANNEL_DOC" controllerImageDigestEnterprise) ;; *) _digest=$(channel_field "$CHANNEL_DOC" controllerImageDigest) ;; esac fi _new="$_repo:$_want" [ -n "$_digest" ] && _new="$_repo:$_want@$_digest" _installed=$(manifest_image) if [ "$_new" != "$_installed" ]; then echo "==> $_current → $_want${_digest:+ (pinned by digest)}" # Pin the new reference into the manifest, then apply. sed edits the single # "image" line in place (atomic tmp + move). _tmp="$MANIFEST.new" sed "s#\"image\"[[:space:]]*:[[:space:]]*\"[^\"]*\"#\"image\": \"$(json_str "$_new")\"#" "$MANIFEST" > "$_tmp" \ || die "could not rewrite the image in $MANIFEST" chmod 0600 "$_tmp"; mv "$_tmp" "$MANIFEST" else echo "==> already on $_current — re-pulling and recreating" fi "$BIN_DEST" apply } cmd_uninstall() { need_root systemctl disable --now "$SERVICE_NAME" 2>/dev/null || true rm -f "$UNIT_DEST"; systemctl daemon-reload 2>/dev/null || true echo "Removed the systemd service. Left in place: $BIN_DEST, $ETC_DIR (incl. identity)," echo "the data root and any app container. Delete them manually for a full cleanup." } usage() { cat < [--profile homelab|studio|enterprise|public] [--cluster N] [--cache-dir P] [--data-root P] [--http-port P] [--network host|bridge|] [--binary ] gpuk install --mode worker --controller wss://: --enroll-token gk_enroll_... [--cluster N] [--cache-dir P] [--binary ] gpuk install ... --dry-run Validate inputs and print the manifest without changing the host gpuk status Service state, enrollment, /health, app container status gpuk enroll --controller wss://: --token gk_enroll_... gpuk apply (controller) Reconcile the app container from the manifest gpuk update (controller) Move to the current release (pull + recreate, rollback) gpuk update --check (controller) Compare the installed version with the release gpuk update --version (controller) Move to a specific release gpuk channel Fetch + VERIFY the release channel and print what it offers gpuk backup [DIR] (controller) Cold snapshot of the embedded pgdata (stop → tar → restart) gpuk manifest Print the current manifest gpuk logs Follow the app container logs (controller) or the daemon journal gpuk uninstall Remove the systemd service Most people never run this directly: the channel's install.sh installs it (https://repo.byterain.io/gpukitchen/channel/raw/branch/main/install.sh). EOF } # ── dispatch ──────────────────────────────────────────────────────────────────── cmd="${1:-help}"; if [ $# -gt 0 ]; then shift; fi case "$cmd" in install) cmd_install "$@" ;; status) cmd_status ;; enroll) cmd_enroll "$@" ;; apply) cmd_apply ;; update) cmd_update "$@" ;; channel) cmd_channel ;; backup) cmd_backup "${1:-}" ;; manifest) cat "$MANIFEST" ;; logs) _img=$(manifest_image) if [ -n "$_img" ]; then exec docker logs -f "$(container_name)"; else exec journalctl -u "$SERVICE_NAME" -f; fi ;; uninstall) cmd_uninstall ;; help|-h|--help) usage ;; *) usage; exit 1 ;; esac