#!/usr/bin/env bash
set -u
umask 077

export PATH="$HOME/.local/bin:/opt/homebrew/bin:/usr/local/bin:/usr/bin:/bin:/usr/sbin:/sbin"

SCRIPT_PATH="$(/usr/bin/python3 -c 'import os,sys; print(os.path.realpath(sys.argv[1]))' "$0")"
SCRIPT_DIR="$(CDPATH= cd -- "$(dirname -- "$SCRIPT_PATH")" && pwd)"
STUDIO_ROOT="${HIVEMIND_STUDIO_ROOT:-$(CDPATH= cd -- "$SCRIPT_DIR/.." && pwd)}"
MEDIA_STATE_ROOT="${HIVEMIND_MEDIA_STATE_DIR:-$HOME/.hivemindos/media-studio}"
# The interpreter is part of the artifact. `python3` on PATH is whatever
# Homebrew last upgraded to (3.14 removed `cgi` and took the gateway down on
# 2026-07-26); the project venv is the one pyproject pins and the one that has
# cryptography and Pillow. Falls back only so a checkout without a venv still
# starts the parts that do not need it.
STUDIO_PYTHON="$STUDIO_ROOT/.venv/bin/python"
[ -x "$STUDIO_PYTHON" ] || STUDIO_PYTHON="python3"

LABEL="com.liam.zimage-stack"
OLD_CF_LABEL="com.liam.zimage-cloudflared"
OLD_OPEN_GEN_LABEL="com.liam.open-generative-ai-hosted"
DOMAIN="gui/$(id -u)"
PRIV="${COMFY_PRIVATE_ROOT:-$HOME/.comfy-private.noindex}"
COMFY="${COMFY_DIR:-$HOME/comfy/ComfyUI}"
APP="$STUDIO_ROOT/packages/media-gateway"
MOBILE="$STUDIO_ROOT/packages/comfyui-mobile"
KREA2_IDENTITY_NODE="$STUDIO_ROOT/packages/comfyui-custom-nodes/hivemind-krea2-identity"
SEEDVR2_TRT_NODE="$STUDIO_ROOT/packages/comfyui-custom-nodes/hivemind-seedvr2-trt"
PRIVATE_MEDIA_NODE="$STUDIO_ROOT/packages/comfyui-custom-nodes/hivemind-private-media"
PROGRESS_NODE="$STUDIO_ROOT/packages/comfyui-custom-nodes/hivemind-progress"
AUDIO_SPLIT_NODE="$STUDIO_ROOT/packages/comfyui-custom-nodes/hivemind-audio-split"
OPEN_GEN="$STUDIO_ROOT/packages/open-generative-ai"
CONTENT_STUDIO="$STUDIO_ROOT"
TAILSCALE_CLI="/Applications/Tailscale.app/Contents/MacOS/tailscale"
# What `tailscale serve` publishes on when the studio's remote-access switch is
# turned on and nothing publishes the studio yet. Must match
# DEFAULT_TAILNET_HTTPS_PORT in remote_access.py. A door that already exists on
# another HTTPS port (8789 — the retired proxy's number — is the URL the owner's
# other devices kept) is found by what it proxies and used as it is; nothing
# here keys on this number.
TAILNET_HTTPS_PORT="${CONTENT_STUDIO_TAILNET_PORT:-8765}"
# Parse before touching runtime state. launchd needs explicit environment
# forwarding; it does not inherit the invoking shell's flags.
STACK_ACTION="${1:-status}"
[ "$#" -eq 0 ] || shift
REMOTE_ACCESS="${CONTENT_STUDIO_REMOTE_ACCESS:-0}"
REMOTE_ACCESS_FLAG=0
while [ "$#" -gt 0 ]; do
  case "$1" in
    --remote-access) REMOTE_ACCESS=1; REMOTE_ACCESS_FLAG=1; shift ;;
    --tailnet-port)
      [ "$#" -ge 2 ] || { echo "--tailnet-port needs a port" >&2; exit 2; }
      TAILNET_HTTPS_PORT="$2"; shift 2 ;;
    *) echo "Unknown option: $1" >&2; exit 2 ;;
  esac
done
case "$TAILNET_HTTPS_PORT" in
  ''|*[!0-9]*) echo "Invalid tailnet port" >&2; exit 2 ;;
esac
if [ "$TAILNET_HTTPS_PORT" -lt 1 ] || [ "$TAILNET_HTTPS_PORT" -gt 65535 ]; then
  echo "Tailnet port must be between 1 and 65535" >&2; exit 2
fi
if [ "$REMOTE_ACCESS_FLAG" = 1 ]; then
  case "$STACK_ACTION" in
    start|restart|supervise) ;;
    *) echo "--remote-access requires start, restart, or supervise" >&2; exit 2 ;;
  esac
fi
export CONTENT_STUDIO_REMOTE_ACCESS="$REMOTE_ACCESS"
export CONTENT_STUDIO_TAILNET_PORT="$TAILNET_HTTPS_PORT"
MEDIA_STUDIO_MCP_PORT="${MEDIA_STUDIO_MCP_PORT:-8796}"
FLUX2_SERVER_DIR="$STUDIO_ROOT/engines/flux-2-swift-mlx"
FLUX2_SERVER_BIN="$FLUX2_SERVER_DIR/.build/arm64-apple-macosx/release/Flux2Server"
FLUX2_SERVER_PORT="8791"
COMFY_DEFAULT_PORT="8188"
COMFY_ANIMA_PORT="8198"
COMFY_LTX_PORT="8199"
CONTENT_STUDIO_PORT="8765"
OPEN_GEN_PORT="8794"
# The one Node child. It mounts the Canvas surface, the local-inference bridge
# and the agent MCP on this port, and keeps 8788, $OPEN_GEN_PORT and
# $MEDIA_STUDIO_MCP_PORT answering as compatibility ports.
NODE_SERVICES_PORT="${HIVEMIND_NODE_SERVICES_PORT:-8793}"
# 90 seconds, not the 45 this gate carried through the 2026-09-04 boot loop.
# The Node child has to reach a Next.js PRODUCTION build before /healthz can
# answer, and it starts AFTER up to three ComfyUI lanes have loaded — each of
# which pins several GB and saturates the disk on its way in. 45s is plenty on
# a warm machine and is not obviously enough on a cold or busy one. Being wrong
# the tight way costs a boot loop; being wrong the loose way costs 45 extra
# seconds on a boot that was going to fail anyway, and the log now says which
# of the two it was. Override with HIVEMIND_NODE_SERVICES_TIMEOUT.
NODE_SERVICES_TIMEOUT="${HIVEMIND_NODE_SERVICES_TIMEOUT:-90}"
ZIMG_TOKEN_FILE="${ZIMG_TOKEN_FILE:-$MEDIA_STATE_ROOT/secure/zimg-token}"

# Start a service with the machine's shared credentials loaded.
#
# PassBook owns the store, so `passbook run` is the direct route and it names
# the asking service, which is the difference that matters: without --app every
# read in the access ledger is attributed to the wrapper, and five services
# collapse into one row that says nothing about who wanted the key.
#
# `hive-env-run` is the older shim over the same store and stays as the
# fallback, because a machine can have HivemindOS without having run
# `passbook install` yet. Running bare is last: the services still start, they
# just see only what this shell already exported.
# The keys this stack actually needs, from the access ledger rather than from
# memory. Named so each service is handed ten credentials instead of the three
# hundred in the store: `passbook run` without --only passes the lot, and every
# one of those is a key some dependency could read, log, or ship in a crash
# report. It also means the machine can seal reads and stop trusting this app
# by name — a grant is proof of how the process started, which a name is not.
# CLOUDFLARE_*: gpu_rentals.py presigns the weights manifest on R2 itself
# (_r2_credentials -> _env), and a plain os.environ read never appears in
# PassBook's access ledger — which is what this list was built from. Without
# them renting dies at manifest publish with "CLOUDFLARE_API_TOKEN is not
# configured in the environment", after the prices have already loaded, which
# reads as a billing fault rather than a missing key. The marketplace keys are
# deliberately NOT here: with a HivemindOS account connected those calls go
# through the hosted worker and this machine holds no marketplace key at all.
# HUGGINGFACE_READ_WRITE_KEY: gpu_rentals.py presigns the GATED Hugging Face
# files (stock Flux.2 Klein 9B, the LTX Ingredients LoRA) for a rental's add-on
# packs on this Mac, so the box never holds the token. It is the only HF token
# in PassBook today (owner's call, 2026-09-24); swap it for a read-only one here
# and in _HF_TOKEN_ENV_KEYS if one is ever added.
STUDIO_KEYS="CIVITAI_API_KEY CLOUDFLARE_ACCOUNT_ID CLOUDFLARE_API_TOKEN
GOOGLE_AI_STUDIO_API_KEY HIVEMINDOS_DASHBOARD_DEVICE_TOKEN HUGGINGFACE_READ_WRITE_KEY
OPENAI_API_KEY OPENAI_OAUTH_ACCESS_TOKEN OPENAI_OAUTH_ACCOUNT_ID
OPENAI_OAUTH_EXPIRES_AT OPENAI_OAUTH_REFRESH_TOKEN OPENROUTER_API_KEY
VENICE_API_KEY XAI_API_KEY"

with_credentials() {
  local app="$1"
  shift
  if command -v passbook >/dev/null 2>&1; then
    local only=()
    for key in $STUDIO_KEYS; do only+=(--only "$key"); done
    passbook run --app "$app" "${only[@]}" -- "$@"
  elif command -v hive-env-run >/dev/null 2>&1; then
    hive-env-run -- "$@"
  else
    log "no passbook and no hive-env-run; $app starts without the shared store"
    "$@"
  fi
}

# PassBook's vault can be locked when the stack starts: it is after every
# reboot until someone signs in. `passbook run` then hands over nothing for a
# sealed key, and every credentialed feature fails as if it were unconfigured.
# On 2026-09-24 renting said "CLOUDFLARE_API_TOKEN is not configured" for ~90
# min after a reboot, and signing in changed nothing until a manual restart.
# The supervisor now says so at start and restarts itself once the vault opens
# (see supervise), waiting for the render queues to empty first.
studio_keys_locked() {
  command -v passbook >/dev/null 2>&1 || return 1
  # shellcheck disable=SC2086 # the list is split on purpose
  passbook check --app hivemind-content-studio $STUDIO_KEYS 2>/dev/null | grep -q ': locked'
}

renders_idle() {
  local port body
  for port in "$COMFY_DEFAULT_PORT" "$COMFY_ANIMA_PORT" "$COMFY_LTX_PORT"; do
    body="$(curl -s --max-time 3 "http://127.0.0.1:$port/queue" 2>/dev/null)" || continue
    [ -n "$body" ] || continue
    printf '%s' "$body" | python3 -c 'import json,sys; d=json.load(sys.stdin); sys.exit(1 if d.get("queue_running") or d.get("queue_pending") else 0)' \
      || return 1
  done
  return 0
}

load_hardware_profile() {
  if [ -f "$APP/hardware_profile.py" ]; then
    eval "$("$STUDIO_PYTHON" "$APP/hardware_profile.py" --shell 2>/dev/null)" || true
  fi
  ZIMG_ACCELERATOR_PROFILE="${ZIMG_ACCELERATOR_PROFILE:-cpu}"
  ZIMG_IS_APPLE_SILICON="${ZIMG_IS_APPLE_SILICON:-0}"
  ZIMG_ENABLE_APPLE_SILICON_OPTIMIZATIONS="${ZIMG_ENABLE_APPLE_SILICON_OPTIMIZATIONS:-0}"
  ZIMG_DEFAULT_ENABLE_FLUX2_SERVER="${ZIMG_DEFAULT_ENABLE_FLUX2_SERVER:-0}"
  ZIMG_DEFAULT_ASFP8_INT8_EXT="${ZIMG_DEFAULT_ASFP8_INT8_EXT:-0}"
  ZIMG_DEFAULT_ASFP8_FP8_EXT="${ZIMG_DEFAULT_ASFP8_FP8_EXT:-0}"
  ZIMG_DEFAULT_ASFP8_TRACE_OPS="${ZIMG_DEFAULT_ASFP8_TRACE_OPS:-0}"
  ZIMG_DEFAULT_ASFP8_PROFILE="${ZIMG_DEFAULT_ASFP8_PROFILE:-0}"
  ZIMG_DEFAULT_COMFY_ATTENTION="${ZIMG_DEFAULT_COMFY_ATTENTION:-}"
}

load_hardware_profile
ZIMG_ENABLE_FLUX2_SERVER="${ZIMG_ENABLE_FLUX2_SERVER:-$ZIMG_DEFAULT_ENABLE_FLUX2_SERVER}"
COMFY_DEFAULT_ATTENTION="${COMFY_DEFAULT_ATTENTION:-$ZIMG_DEFAULT_COMFY_ATTENTION}"
COMFY_ANIMA_ATTENTION="${COMFY_ANIMA_ATTENTION:-$ZIMG_DEFAULT_COMFY_ATTENTION}"
COMFY_LTX_ATTENTION="${COMFY_LTX_ATTENTION:-$ZIMG_DEFAULT_COMFY_ATTENTION}"
COMFY_ENABLE_LTX_LANE="${COMFY_ENABLE_LTX_LANE:-$ZIMG_ENABLE_APPLE_SILICON_OPTIMIZATIONS}"
ZIMG_LTX_PRIORITY_MODE="${ZIMG_LTX_PRIORITY_MODE:-$COMFY_ENABLE_LTX_LANE}"
if [ "${ZIMG_LTX_PRIORITY_MODE:-0}" = "1" ] && [ "${ZIMG_FORCE_FLUX2_SERVER:-0}" != "1" ]; then
  # LTXAV keeps a very large MPS working set. Leaving the unrelated Flux2 MLX
  # server resident costs measurable LTX throughput, so the dedicated LTX lane
  # gets priority unless Flux2 is explicitly forced on.
  ZIMG_ENABLE_FLUX2_SERVER=0
fi
COMFY_DEFAULT_ASFP8_INT8_EXT="${COMFY_DEFAULT_ASFP8_INT8_EXT:-$ZIMG_DEFAULT_ASFP8_INT8_EXT}"
COMFY_DEFAULT_ASFP8_FP8_EXT="${COMFY_DEFAULT_ASFP8_FP8_EXT:-$ZIMG_DEFAULT_ASFP8_FP8_EXT}"
COMFY_DEFAULT_ASFP8_TRACE_OPS="${COMFY_DEFAULT_ASFP8_TRACE_OPS:-$ZIMG_DEFAULT_ASFP8_TRACE_OPS}"
COMFY_DEFAULT_ASFP8_PROFILE="${COMFY_DEFAULT_ASFP8_PROFILE:-$ZIMG_DEFAULT_ASFP8_PROFILE}"
LOG_DIR="$HOME/Library/Logs"
SUP_LOG="$LOG_DIR/zimage-stack-supervisor.log"
LOCK_DIR="$PRIV/zimage-stack-supervise.lock"

mkdir -p "$PRIV/output" "$PRIV/z_image_outputs" "$PRIV/debug_outputs" "$PRIV/input" "$PRIV/temp" "$PRIV/temp-anima" "$PRIV/temp-ltx" "$LOG_DIR" "$MEDIA_STATE_ROOT/state/media-gateway" "$MEDIA_STATE_ROOT/secure"
touch "$PRIV/.metadata_never_index" "$PRIV/output/.metadata_never_index" "$PRIV/z_image_outputs/.metadata_never_index" "$PRIV/debug_outputs/.metadata_never_index"
chmod -R go-rwx "$PRIV" 2>/dev/null || true
chflags hidden "$PRIV" 2>/dev/null || true

log() { printf '%s %s\n' "$(date '+%Y-%m-%d %H:%M:%S')" "$*" | tee -a "$SUP_LOG"; }

# Keep the last run's log instead of erasing it, and keep the pair bounded.
# Only ever called before the children are started, so no writer holds an fd on
# the file being renamed.
LOG_ROLL_MAX_BYTES="${LOG_ROLL_MAX_BYTES:-33554432}"
roll_log() {
  local path="$1" size=0
  [ -f "$path" ] || return 0
  size=$(wc -c < "$path" 2>/dev/null | tr -d ' ') || size=0
  if [ "${size:-0}" -gt "$LOG_ROLL_MAX_BYTES" ]; then
    mv -f "$path" "$path.1" 2>/dev/null || true
  fi
}

kill_port() {
  local port="$1"
  local pids
  pids=$(lsof -tiTCP:"$port" -sTCP:LISTEN 2>/dev/null || true)
  if [ -n "$pids" ]; then
    log "stopping listener(s) on :$port: $pids"
    kill $pids 2>/dev/null || true
  fi
}

# Is there a ComfyUI checkout for THIS stack to run?
#
# ComfyUI is an optional engine, not a prerequisite: the studio boots without
# one and the owner attaches theirs (or a cloud/rented lane) afterwards. This
# predicate is also what keeps the stack off a ComfyUI it did not start — the
# custom-node symlinks and every kill_port on a lane port are behind it, so a
# machine where the only ComfyUI is somebody's Desktop install on :8188 keeps
# it when this stack starts, restarts or stops.
comfy_available() {
  [ -f "$COMFY/main.py" ] && [ -x "$COMFY/.venv/bin/python" ]
}

# ── Health gates that say what actually failed ───────────────────────────────
#
# On 2026-09-04 this stack boot-looped for an hour and the log said only
# "Z-Image frontend did not become healthy in time". The frontend was fine: it
# bound its port and logged "listening" on all 29 attempts, and a hand-run
# `curl -fsS --max-time 5 http://127.0.0.1:8788/healthz` answered 200 in 0.02s
# within about four seconds of each child starting. The old message was
# byte-identical whether the server never bound, bound and answered, or the
# supervisor's own curl could not be executed — so an hour went into telling
# those apart by hand.
#
# wait_http now records WHY the last probe failed and the gate beside it logs
# that. These are globals because a bash function returns a status, not a
# record, and printing a line to parse would be worse.
WAIT_HTTP_URL=""
WAIT_HTTP_CURL_EXIT=""
WAIT_HTTP_STATUS=""
WAIT_HTTP_ELAPSED=0
WAIT_HTTP_PROBE_TIMEOUT=5
# How long ONE probe may take by default. Separate from the gate's overall
# budget, and overridable per call as wait_http's third argument.
WAIT_HTTP_DEFAULT_PROBE_TIMEOUT="${WAIT_HTTP_DEFAULT_PROBE_TIMEOUT:-5}"

# wait_http <url> [budget_seconds] [probe_seconds]
# A budget of 0 makes this a single recorded probe, which is how the liveness
# check gets a reason to log without duplicating any of this.
wait_http() {
  local url="$1" timeout="${2:-120}" status="" rc=0 started=$SECONDS
  WAIT_HTTP_URL="$url"
  WAIT_HTTP_CURL_EXIT=""
  WAIT_HTTP_STATUS=""
  WAIT_HTTP_ELAPSED=0
  WAIT_HTTP_PROBE_TIMEOUT="${3:-$WAIT_HTTP_DEFAULT_PROBE_TIMEOUT}"
  while :; do
    # No -f. `-f` collapses every 4xx and 5xx into exit 22 and throws the
    # status away, so "the app is up and returning 503 while it warms" and
    # "the app returns 404 because the route moved" were the same failure.
    # Ask for the status and judge it here: both halves survive.
    status="$(curl -s -o /dev/null -w '%{http_code}' --max-time "$WAIT_HTTP_PROBE_TIMEOUT" "$url" 2>/dev/null)"
    rc=$?
    WAIT_HTTP_CURL_EXIT="$rc"
    WAIT_HTTP_STATUS="$status"
    WAIT_HTTP_ELAPSED=$((SECONDS - started))
    # Healthy is what -f called healthy: a 2xx or 3xx curl actually received.
    case "$rc:$status" in
      0:2??|0:3??) return 0 ;;
    esac
    # 127 is not a transient condition — curl is not on this PATH, and no
    # amount of waiting changes that. Sitting in the loop would only delay the
    # report by the whole budget, and if curl is missing `sleep` may be too, in
    # which case the loop below spins hot. Give up on the first one.
    [ "$rc" -eq 127 ] && return 1
    [ "$WAIT_HTTP_ELAPSED" -ge "$timeout" ] && return 1
    sleep 2
  done
}

# The last probe's verdict in the words of a bug report. The four that matter:
# 7 is nothing listening, 28 is something listening that will not answer, 127
# is curl itself failing to run (a PATH problem in this supervisor, not a child
# problem at all), and exit 0 with a status the old -f would have rejected.
wait_http_reason() {
  case "${WAIT_HTTP_CURL_EXIT:-}" in
    '') printf 'no probe was made' ;;
    0) printf 'HTTP %s' "${WAIT_HTTP_STATUS:-000}" ;;
    6) printf 'could not resolve host' ;;
    7) printf 'connection refused' ;;
    22) printf 'HTTP %s' "${WAIT_HTTP_STATUS:-000}" ;;
    28) printf 'timed out after %ss with no reply' "$WAIT_HTTP_PROBE_TIMEOUT" ;;
    35|60) printf 'TLS handshake failed' ;;
    52) printf 'empty reply from server' ;;
    56) printf 'connection reset' ;;
    127) printf 'curl exited 127 (curl could not be executed; check PATH)' ;;
    *) printf 'curl exited %s' "${WAIT_HTTP_CURL_EXIT}" ;;
  esac
}

# Whoever holds a port right now, as a space-separated pid list (empty if free).
port_listeners() {
  local pids=""
  pids="$(lsof -tiTCP:"$1" -sTCP:LISTEN 2>/dev/null)" || pids=""
  # Unquoted on purpose: word splitting is what collapses lsof's newlines.
  # shellcheck disable=SC2086
  echo $pids
}

# The port out of a URL, using only shell builtins. A diagnosis that needs sed
# to run is one more thing that can be missing at the moment it is wanted —
# which is exactly the class of failure this whole block exists to name.
url_port() {
  local rest="${1#*://}"
  rest="${rest%%/*}"
  local port="${rest##*:}"
  case "$port" in
    ''|*[!0-9]*) printf '' ;;
    *) printf '%s' "$port" ;;
  esac
}

# The suffix every "did not become healthy in time" line now carries: the last
# probe's verdict, whether anything holds the port at the moment we gave up,
# and whether the child we started is still alive. A child that died before it
# ever bound is the common case, and the old log never said so.
wait_http_why() {
  local pid="${1:-}" port="" listeners="" out=""
  port="$(url_port "$WAIT_HTTP_URL")"
  out="last probe after ${WAIT_HTTP_ELAPSED}s: $(wait_http_reason)"
  if [ -n "$port" ]; then
    if ! command -v lsof >/dev/null 2>&1; then
      out="$out; cannot tell who holds :$port (no lsof)"
    else
      listeners="$(port_listeners "$port")"
      if [ -n "$listeners" ]; then
        out="$out; :$port held by pid(s) $listeners"
      else
        out="$out; nothing listening on :$port"
      fi
    fi
  fi
  if [ -n "$pid" ]; then
    if kill -0 "$pid" 2>/dev/null; then
      out="$out; child pid=$pid still alive"
    else
      out="$out; child pid=$pid is gone"
    fi
  fi
  printf '%s' "$out"
}

# tailscale_ip() lived here. Nothing binds a proxy to the tailnet address any
# more, so the only question left is the MagicDNS name a served URL is built on.
tailscale_dns() {
  if [ -x "$TAILSCALE_CLI" ]; then
    "$TAILSCALE_CLI" status --json 2>/dev/null | python3 -c 'import json,sys; d=json.load(sys.stdin); print((d.get("Self") or {}).get("DNSName","").rstrip("."))' 2>/dev/null || true
  fi
}

# The published tailnet URL, or empty when nothing publishes the studio.
# Read from `tailscale serve`, which is the only thing that publishes anything
# now — so this reports what is actually true rather than what a boot-time
# proxy would have made. The door is found by what it proxies, on whatever
# HTTPS port it is on: an entry on 8789 (the retired proxy's number, kept as
# the URL the owner's other devices open) counts the same as one on the
# default. TAILNET_HTTPS_PORT only breaks a tie between two such doors.
tailnet_studio_url() {
  local ts_dns
  [ -x "$TAILSCALE_CLI" ] || return 0
  ts_dns="$(tailscale_dns || true)"
  [ -n "$ts_dns" ] || return 0
  "$TAILSCALE_CLI" serve status --json 2>/dev/null | TS_DNS="$ts_dns" TS_PORT="$TAILNET_HTTPS_PORT" TS_TARGET="http://127.0.0.1:$CONTENT_STUDIO_PORT" python3 -c '
import json, os, sys
try:
    served = json.load(sys.stdin)
except Exception:
    sys.exit(0)
host = os.environ["TS_DNS"]
doors = []
for key, entry in (served.get("Web") or {}).items():
    entry_host, _, port = str(key).rpartition(":")
    if entry_host.lower() != host.lower() or not port.isdigit():
        continue
    root = ((entry or {}).get("Handlers") or {}).get("/") or {}
    if root.get("Proxy") == os.environ["TS_TARGET"]:
        doors.append(int(port))
if not doors:
    sys.exit(0)
wanted = int(os.environ["TS_PORT"])
port = wanted if wanted in doors else min(doors)
print("https://" + host + ("" if port == 443 else ":%d" % port) + "/")
' 2>/dev/null || true
}

# The tailnet URL is not made here any more.
#
# This used to bind a hand-rolled Node HTTPS proxy to the Tailscale address at
# every boot, in front of the Canvas port (8788), which authenticated nothing —
# so every device on the tailnet could queue ComfyUI graphs and read the
# library. When `tailscale cert` was unavailable it generated a SELF-SIGNED
# certificate with openssl, which is a full-screen browser warning the person
# reading it cannot fix. Remote access is now an explicit switch in the studio
# (Rented GPUs > Open on my other devices) that runs `tailscale serve` — a real
# certificate, no proxy process — and publishes ONLY this API's port. See
# src/hivemind_content_studio/remote_access.py.

children=()
start_children() {
  children=()
  log "starting Hivemind Content Studio as one managed app"
  # ComfyUI is OPTIONAL. This supervisor used to refuse to bring anything up
  # until an external checkout answered on :8188 — it symlinked custom nodes
  # into $COMFY, minted its secrets with $COMFY/.venv, waited 150s per lane and
  # returned 1 on the first one that did not answer, which the loop below
  # retried every 10s forever. A machine without a hand-built ComfyUI therefore
  # never saw the studio at all. The control API and the gateway now come up on
  # their own and lanes are engines you attach afterwards (the studio's Connect
  # ComfyUI card, /api/comfy/connect).
  comfy_available || log "no ComfyUI at $COMFY — bringing the studio up without local lanes; connect one from the studio (cloud and rented models need none)"
  log "hardware profile=$ZIMG_ACCELERATOR_PROFILE apple_silicon=$ZIMG_IS_APPLE_SILICON flux2_server=$ZIMG_ENABLE_FLUX2_SERVER asfp8_int8=$COMFY_DEFAULT_ASFP8_INT8_EXT asfp8_fp8=$COMFY_DEFAULT_ASFP8_FP8_EXT"
  # These were truncated with ': >' here, on every start. The supervisor loop
  # below restarts the whole stack automatically after any child exit or failed
  # health check, so the truncation ran right after a crash — the evidence of a
  # failure was deleted by the recovery from it, and there was nothing a person
  # could attach to a report. Rolled instead: the previous run is kept as .1,
  # and each file is bounded so nothing grows without end.
  # Every child appends with '>>', so nothing here shares a file descriptor
  # with the control API's own rotating log, which lives elsewhere entirely
  # (~/Library/Logs/Hivemind Content Studio) and is never touched from a shell.
  roll_log "$PRIV/comfy.log"
  roll_log "$PRIV/comfy-ltx.log"
  roll_log "$PRIV/z-image-api.log"
  roll_log "$PRIV/z-image-frontend.log"
  roll_log "$PRIV/hivemind-content-studio.log"
  roll_log "$PRIV/media-studio-mcp.log"
  # The tailnet HTTPS proxy no longer starts here — remote access is an opt-in
  # toggle that publishes the control API through `tailscale serve` — so there
  # is no proxy log to roll. The kill_port calls below stay, to clear a stale
  # one left by an older build.
  # Custom nodes only ever go into the checkout this stack runs. Nothing here
  # touches a ComfyUI the app did not start.
  if comfy_available; then
  if [ -L "$COMFY/custom_nodes/comfyui-mobile-frontend" ] || [ ! -e "$COMFY/custom_nodes/comfyui-mobile-frontend" ]; then
    ln -sfn "$MOBILE" "$COMFY/custom_nodes/comfyui-mobile-frontend"
  else
    log "cannot install embedded ComfyUI Mobile: custom node path is not a symlink"
    return 1
  fi
  if [ -L "$COMFY/custom_nodes/hivemind-krea2-identity" ] || [ ! -e "$COMFY/custom_nodes/hivemind-krea2-identity" ]; then
    ln -sfn "$KREA2_IDENTITY_NODE" "$COMFY/custom_nodes/hivemind-krea2-identity"
  else
    log "cannot install Krea2 identity adapter: custom node path is not a symlink"
    return 1
  fi
  # SeedVR2 TensorRT VAE. Inert on Apple silicon (it reports "no CUDA device"
  # and the decode runs on PyTorch), and installed anyway for two reasons: a
  # local NVIDIA machine gets the same acceleration the rented boxes get, and a
  # lane that HAS the node can say why it is not using it instead of looking
  # like a box that is merely missing something.
  if [ -L "$COMFY/custom_nodes/hivemind-seedvr2-trt" ] || [ ! -e "$COMFY/custom_nodes/hivemind-seedvr2-trt" ]; then
    ln -sfn "$SEEDVR2_TRT_NODE" "$COMFY/custom_nodes/hivemind-seedvr2-trt"
  else
    log "cannot install SeedVR2 TensorRT adapter: custom node path is not a symlink"
    return 1
  fi
  # Loads the owner's references straight out of the gateway's memory, so a
  # decrypted reference never becomes a file in ComfyUI's input directory for
  # anything on this machine to list and copy. The gateway checks each lane for
  # this node before it rewrites a graph, so a lane without it keeps the
  # filename it has always had and nothing breaks either way.
  if [ -L "$COMFY/custom_nodes/hivemind-private-media" ] || [ ! -e "$COMFY/custom_nodes/hivemind-private-media" ]; then
    ln -sfn "$PRIVATE_MEDIA_NODE" "$COMFY/custom_nodes/hivemind-private-media"
  else
    log "cannot install private media loader: custom node path is not a symlink"
    return 1
  fi
  # Serves this lane's sampler counters at /hivemind/progress, the same path a
  # rented lane answers. Without it a local generation's bar is a time estimate
  # and nothing more — ComfyUI publishes its step counts over the websocket
  # only, to the client that submitted, and our runner never holds that socket.
  if [ -L "$COMFY/custom_nodes/hivemind-progress" ] || [ ! -e "$COMFY/custom_nodes/hivemind-progress" ]; then
    ln -sfn "$PROGRESS_NODE" "$COMFY/custom_nodes/hivemind-progress"
  else
    log "cannot install the progress readout: custom node path is not a symlink"
    return 1
  fi
  # Splits a finished clip's sound into dialogue, effects, music and a track per
  # voice (the Video stage's download menu). The gateway links this itself on a
  # stack that started before the pack existed, but it can only restart an idle
  # lane to load it - linking here means a fresh start already has it.
  if [ -L "$COMFY/custom_nodes/hivemind-audio-split" ] || [ ! -e "$COMFY/custom_nodes/hivemind-audio-split" ]; then
    ln -sfn "$AUDIO_SPLIT_NODE" "$COMFY/custom_nodes/hivemind-audio-split"
  else
    log "cannot install the sound splitter: custom node path is not a symlink"
    return 1
  fi
  fi
  # Minted with the STUDIO's interpreter, not ComfyUI's: both secrets belong to
  # this app and a machine with no ComfyUI still has to boot with them.
  COMFY_PRIVATE_VIEW_TOKEN="$("$STUDIO_PYTHON" - <<'PY'
import secrets
print(secrets.token_urlsafe(32))
PY
  )"
  # What the tailnet HTTPS proxy shows the control API to prove it is the
  # proxy. The studio believes x-forwarded-proto/host/for only from a request
  # carrying this — those headers decide the session cookie's `secure` flag,
  # the WebAuthn relying-party id and which bucket a failed password counts
  # against, and any caller can write them. Minted per run and given to both
  # ends here, so the two can never drift apart.
  CONTENT_STUDIO_PROXY_SECRET="$("$STUDIO_PYTHON" - <<'PY'
import secrets
print(secrets.token_urlsafe(32))
PY
  )"

  # Only the lanes this stack runs itself. With no checkout of ours, whatever
  # answers on 8188 is the user's own ComfyUI and is not ours to reap.
  if comfy_available; then
	  kill_port "$COMFY_DEFAULT_PORT"
	  kill_port "$COMFY_LTX_PORT"
	  kill_port "$COMFY_ANIMA_PORT"
  fi
  kill_port 8787
  kill_port 8788
  kill_port "$NODE_SERVICES_PORT"
  kill_port "$MEDIA_STUDIO_MCP_PORT"
  kill_port "$FLUX2_SERVER_PORT"
  kill_port "$CONTENT_STUDIO_PORT"
  # One process holds all four Node ports now, so a stale listener on any of
  # them costs a whole surface rather than one child. `stop` already reaped
  # this one; the start path did not.
  kill_port "$OPEN_GEN_PORT"
  sleep 2

  # FLUX2_CLEAR_CACHE_EVERY_N_STEPS is NOT cosmetic: the server's own default is
  # 0 (never clear), and a process that lives across many edits then accumulates
  # an MLX buffer cache that costs more than the model reload it saves. Measured
  # 2026-08-14, Klein 9B @ 4 steps, same reference: at 1.57 MP the never-clear
  # server ran 66.5s mean (n=5) — SLOWER than the 55.8s cold CLI (n=3) it was
  # meant to beat — while clearing every 2 steps landed 52.6s (n=3). At 0.61 MP:
  # 20.0s vs 35.7s cold on the same hot machine.
  if [ "${ZIMG_ENABLE_FLUX2_SERVER:-0}" = "1" ]; then
    cd "$FLUX2_SERVER_DIR" || return 1
      FLUX2_MODEL="${FLUX2_MODEL:-klein9B}" \
    FLUX2_TRANSFORMER_PATH="${FLUX2_TRANSFORMER_PATH:-$COMFY/models/diffusion_models/BigLoveKlein3_mxfp8_swift_mapped_mlx.safetensors}" \
    FLUX2_TRANSFORMER_QUANT="${FLUX2_TRANSFORMER_QUANT:-bf16}" \
    FLUX2_CLEAR_CACHE_EVERY_N_STEPS="${FLUX2_CLEAR_CACHE_EVERY_N_STEPS:-2}" \
	    FLUX2_SERVER_PORT="$FLUX2_SERVER_PORT" MLX_METAL_PATH="$HOME/comfy/Flux2CLI-v2.1.0/mlx.metallib" "$FLUX2_SERVER_BIN" \
	      >> "$PRIV/z-image-flux2-server.log" 2>&1 &
	    local pid="$!"
	    log "Flux2 persistent server child pid=$pid"
	    if [ "${ZIMG_SUPERVISE_FLUX2_SERVER:-0}" = "1" ]; then
	      children+=("$pid")
	    else
	      log "Flux2 persistent server is optional; not restarting whole stack if it exits"
	    fi
    if ! wait_http "http://127.0.0.1:$FLUX2_SERVER_PORT/health" 45; then
      log "Flux2 persistent server did not become healthy in time; $(wait_http_why "$pid")"
      return 1
    fi
    local warm_input="$PRIV/input/mlx_test_selfie.jpg"
    if [ "${ZIMG_ENABLE_FLUX2_WARMUP:-1}" = "1" ] && [ -f "$warm_input" ]; then
      log "warming Flux2 persistent server in background"
      (
      /usr/bin/python3 - <<PY >> "$PRIV/z-image-flux2-server.log" 2>&1 || log "Flux2 warmup failed; continuing with cold first request"
import json, urllib.request, time
payload = {
  "prompt": "Replace the hat with a simple red Santa hat. Preserve the face, pose, lighting, and background.",
  "imagePath": "$warm_input",
  "outputPath": "$PRIV/z_image_outputs/.flux2_warmup.png",
  "width": 448,
  "height": 672,
  "steps": 1,
  "guidance": 1.0,
  "seed": 12345,
}
t0=time.time()
req=urllib.request.Request("http://127.0.0.1:$FLUX2_SERVER_PORT/generate", data=json.dumps(payload).encode(), headers={"Content-Type":"application/json"}, method="POST")
print(urllib.request.urlopen(req, timeout=180).read().decode())
print("warmup_wall", round(time.time()-t0, 2))
PY
      ) &
    elif [ "${ZIMG_ENABLE_FLUX2_WARMUP:-1}" = "1" ]; then
      log "Flux2 warmup skipped; no $warm_input"
    else
      log "Flux2 warmup disabled; persistent server stays ready without startup generation"
    fi
  fi

	  local comfy_default_attention_args=()
	  local comfy_anima_attention_args=()
	  local comfy_ltx_attention_args=()
	  local comfy_lanes="default=http://127.0.0.1:$COMFY_DEFAULT_PORT,anima=http://127.0.0.1:$COMFY_ANIMA_PORT"
	  local comfy_lane_rules="anima=anima,qwen35,qwen3.5"
	  if [ -n "$COMFY_DEFAULT_ATTENTION" ]; then
	    comfy_default_attention_args+=("$COMFY_DEFAULT_ATTENTION")
	  fi
	  if [ -n "$COMFY_ANIMA_ATTENTION" ]; then
	    comfy_anima_attention_args+=("$COMFY_ANIMA_ATTENTION")
	  fi
	  if [ -n "$COMFY_LTX_ATTENTION" ]; then
	    comfy_ltx_attention_args+=("$COMFY_LTX_ATTENTION")
	  fi
	  if [ "${COMFY_ENABLE_LTX_LANE:-0}" = "1" ]; then
	    comfy_lanes="$comfy_lanes,ltx=http://127.0.0.1:$COMFY_LTX_PORT"
	    comfy_lane_rules="ltx=ltx,ltxv,ltxav,ltx23,eros,10eros;$comfy_lane_rules"
	  fi

	  # Attached GPU rentals overlay (written by the studio's /api/gpu-rentals
	  # attach flow; see src/hivemind_content_studio/gpu_rentals.py). Rental
	  # lane rules are PREPENDED so an attached machine wins over local lanes
	  # for the models it serves. Tunnel URLs are loopback (SSH is the auth).
	  RENTAL_LANES_ENV="$MEDIA_STATE_ROOT/rental-lanes.env"
	  RENTAL_COMFY_LANES=""; RENTAL_COMFY_LANE_RULES=""; RENTAL_COMFY_REMOTE_LANES=""
	  if [ -f "$RENTAL_LANES_ENV" ]; then
	    . "$RENTAL_LANES_ENV"
	    if [ -n "${RENTAL_COMFY_LANES:-}" ]; then comfy_lanes="$comfy_lanes,$RENTAL_COMFY_LANES"; fi
	    if [ -n "${RENTAL_COMFY_LANE_RULES:-}" ]; then comfy_lane_rules="$RENTAL_COMFY_LANE_RULES;$comfy_lane_rules"; fi
	    log "rental lanes overlay active: ${RENTAL_COMFY_REMOTE_LANES:-none}"
	  fi

  # ---- Local ComfyUI lanes: optional from here down --------------------------
  # The lane URLs stay in $comfy_lanes either way. The gateway needs a populated
  # lane map (~30 read sites assume a `default` lane exists) and reports each
  # lane's liveness on /health, so "no ComfyUI" is a lane that does not answer —
  # never a missing lane, and never a reason to hold the studio back.
  if comfy_available; then
  cd "$COMFY" || return 1
  ZIMG_ACCELERATOR_PROFILE="$ZIMG_ACCELERATOR_PROFILE" \
  HIVEMIND_STUDIO_ROOT="$STUDIO_ROOT" \
  HIVEMIND_MEDIA_STATE_DIR="$MEDIA_STATE_ROOT" \
  ZIMG_TOKEN_FILE="$ZIMG_TOKEN_FILE" \
  SWIFT_FLUX2_BIN="$FLUX2_SERVER_DIR/.build/arm64-apple-macosx/release/Flux2CLI" \
  ZIMG_ENABLE_APPLE_SILICON_OPTIMIZATIONS="$ZIMG_ENABLE_APPLE_SILICON_OPTIMIZATIONS" \
  ASFP8_INT8_EXT="$COMFY_DEFAULT_ASFP8_INT8_EXT" \
  ASFP8_FP8_EXT="$COMFY_DEFAULT_ASFP8_FP8_EXT" \
  ASFP8_TRACE_OPS="$COMFY_DEFAULT_ASFP8_TRACE_OPS" \
  ASFP8_PROFILE="$COMFY_DEFAULT_ASFP8_PROFILE" \
  COMFY_PRIVATE_HISTORY_PROMPTS="${COMFY_PRIVATE_HISTORY_PROMPTS:-1}" \
  COMFYUI_PRIVATE_HISTORY_PROMPTS="${COMFYUI_PRIVATE_HISTORY_PROMPTS:-1}" \
  COMFY_PRIVATE_VIEW_TOKEN="$COMFY_PRIVATE_VIEW_TOKEN" \
  "$COMFY/.venv/bin/python" main.py \
    --listen 127.0.0.1 \
    --port "$COMFY_DEFAULT_PORT" \
    ${COMFY_DEFAULT_ATTENTION:+$COMFY_DEFAULT_ATTENTION} \
    --output-directory "$PRIV/output" \
    --input-directory "$PRIV/input" \
    --temp-directory "$PRIV/temp" \
    --disable-metadata \
    >> "$PRIV/comfy.log" 2>&1 &
  local pid="$!"
  children+=("$pid")
  log "ComfyUI default lane child pid=$pid port=$COMFY_DEFAULT_PORT"

  if ! wait_http "http://127.0.0.1:$COMFY_DEFAULT_PORT/system_stats" 150; then
    # Not fatal any more. A lane that will not start is one engine down, and
    # holding the control API back over it is what made a machine without
    # ComfyUI unable to open the app at all. The studio comes up, reports the
    # lane as unreachable, and cloud and rented work carries on.
    log "ComfyUI default lane did not become healthy in time; $(wait_http_why "$pid"); continuing without it"
  fi

  local comfy_warm_input="$PRIV/input/mlx_test_selfie.jpg"
  if [ "${ZIMG_ENABLE_COMFY_WARMUP:-0}" = "1" ] && [ -f "$comfy_warm_input" ]; then
    log "warming Comfy BigLoveKlein3 BF16 fast edit route in background"
    (
    /usr/bin/python3 - <<PY >> "$PRIV/comfy.log" 2>&1 || log "Comfy BigLoveKlein3 warmup failed; continuing cold"
import json, time, urllib.request
prompt = {
  '1': {'class_type':'UNETLoader','inputs':{'unet_name':'BigLoveKlein3_bf16.safetensors','weight_dtype':'default'}},
  '2': {'class_type':'CLIPLoader','inputs':{'clip_name':'qwen_3_8b_fp8mixed.safetensors','type':'flux2','device':'default'}},
  '3': {'class_type':'VAELoader','inputs':{'vae_name':'flux2-vae.safetensors'}},
  '4': {'class_type':'LoadImage','inputs':{'image':'mlx_test_selfie.jpg'}},
  '4b': {'class_type':'ImageScale','inputs':{'image':['4',0],'upscale_method':'lanczos','width':384,'height':576,'crop':'disabled'}},
  '5': {'class_type':'CLIPTextEncode','inputs':{'clip':['2',0],'text':'warmup portrait edit'}},
  '6': {'class_type':'CLIPTextEncode','inputs':{'clip':['2',0],'text':'low quality, distorted'}},
  '7': {'class_type':'VAEEncode','inputs':{'pixels':['4b',0], 'vae':['3',0]}},
  '8': {'class_type':'KSampler','inputs':{'model':['1',0],'positive':['5',0],'negative':['6',0],'latent_image':['7',0],'seed':12345,'steps':1,'cfg':1.0,'sampler_name':'euler','scheduler':'beta','denoise':0.45}},
  '9': {'class_type':'VAEDecode','inputs':{'samples':['8',0], 'vae':['3',0]}},
  '10': {'class_type':'SaveImage','inputs':{'images':['9',0], 'filename_prefix':'.biglove_bf16_warmup'}},
}
t0=time.time()
req=urllib.request.Request('http://127.0.0.1:$COMFY_DEFAULT_PORT/prompt', data=json.dumps({'prompt':prompt,'client_id':'zimage-startup-warmup'}).encode(), headers={'Content-Type':'application/json'})
pid=json.loads(urllib.request.urlopen(req, timeout=30).read().decode())['prompt_id']
while time.time()-t0 < 180:
    time.sleep(0.5)
    hist=json.loads(urllib.request.urlopen('http://127.0.0.1:$COMFY_DEFAULT_PORT/history/'+pid, timeout=10).read().decode() or '{}')
    if pid in hist:
        print('comfy_biglove_bf16_warmup_wall', round(time.time()-t0, 2), hist[pid].get('status'))
        break
PY
    ) &
	  elif [ "${ZIMG_ENABLE_COMFY_WARMUP:-0}" = "1" ]; then
	    log "Comfy BigLoveKlein3 warmup skipped; no $comfy_warm_input"
	  else
	    log "Comfy BigLoveKlein3 warmup disabled; keeping startup queue free"
	  fi
	
	  if [ "${COMFY_ENABLE_LTX_LANE:-0}" = "1" ]; then
	    ZIMG_ACCELERATOR_PROFILE="$ZIMG_ACCELERATOR_PROFILE" \
	    ZIMG_ENABLE_APPLE_SILICON_OPTIMIZATIONS="$ZIMG_ENABLE_APPLE_SILICON_OPTIMIZATIONS" \
	    ASFP8_INT8_EXT="$COMFY_DEFAULT_ASFP8_INT8_EXT" \
	    ASFP8_FP8_EXT="$COMFY_DEFAULT_ASFP8_FP8_EXT" \
	    ASFP8_TRACE_OPS="$COMFY_DEFAULT_ASFP8_TRACE_OPS" \
	    ASFP8_PROFILE="$COMFY_DEFAULT_ASFP8_PROFILE" \
	    COMFY_PRIVATE_HISTORY_PROMPTS="${COMFY_PRIVATE_HISTORY_PROMPTS:-1}" \
	    COMFYUI_PRIVATE_HISTORY_PROMPTS="${COMFYUI_PRIVATE_HISTORY_PROMPTS:-1}" \
	    COMFY_PRIVATE_VIEW_TOKEN="$COMFY_PRIVATE_VIEW_TOKEN" \
	    "$COMFY/.venv/bin/python" main.py \
	      --listen 127.0.0.1 \
	      --port "$COMFY_LTX_PORT" \
	      ${COMFY_LTX_ATTENTION:+$COMFY_LTX_ATTENTION} \
	      --gpu-only \
	      ${COMFY_LTX_EXTRA_ARGS:-} \
	      --output-directory "$PRIV/output" \
	      --input-directory "$PRIV/input" \
	      --temp-directory "$PRIV/temp-ltx" \
	      --database-url "sqlite:///$PRIV/comfy-ltx.db" \
	      --disable-metadata \
	      >> "$PRIV/comfy-ltx.log" 2>&1 &
	    pid="$!"
	    children+=("$pid")
	    log "ComfyUI LTX lane child pid=$pid port=$COMFY_LTX_PORT gpu_only=1 lora_bypass_mps=1"
	
	    if ! wait_http "http://127.0.0.1:$COMFY_LTX_PORT/system_stats" 150; then
	      log "ComfyUI LTX lane did not become healthy in time; $(wait_http_why "$pid"); continuing without it"
	    fi
	  else
	    log "ComfyUI LTX lane disabled; non-Apple-Silicon path uses default Comfy lane"
	  fi
	
	  ZIMG_ACCELERATOR_PROFILE="$ZIMG_ACCELERATOR_PROFILE" \
  ZIMG_ENABLE_APPLE_SILICON_OPTIMIZATIONS="$ZIMG_ENABLE_APPLE_SILICON_OPTIMIZATIONS" \
  ASFP8_INT8_EXT="$COMFY_DEFAULT_ASFP8_INT8_EXT" \
  ASFP8_FP8_EXT="$COMFY_DEFAULT_ASFP8_FP8_EXT" \
  ASFP8_TRACE_OPS="$COMFY_DEFAULT_ASFP8_TRACE_OPS" \
  ASFP8_PROFILE="$COMFY_DEFAULT_ASFP8_PROFILE" \
  COMFY_PRIVATE_HISTORY_PROMPTS="${COMFY_PRIVATE_HISTORY_PROMPTS:-1}" \
  COMFYUI_PRIVATE_HISTORY_PROMPTS="${COMFYUI_PRIVATE_HISTORY_PROMPTS:-1}" \
  COMFY_PRIVATE_VIEW_TOKEN="$COMFY_PRIVATE_VIEW_TOKEN" \
  "$COMFY/.venv/bin/python" main.py \
    --listen 127.0.0.1 \
    --port "$COMFY_ANIMA_PORT" \
    ${COMFY_ANIMA_ATTENTION:+$COMFY_ANIMA_ATTENTION} \
    --output-directory "$PRIV/output" \
    --input-directory "$PRIV/input" \
    --temp-directory "$PRIV/temp-anima" \
    --database-url "sqlite:///$PRIV/comfy-anima.db" \
    --disable-metadata \
    >> "$PRIV/comfy-anima.log" 2>&1 &
  pid="$!"
  children+=("$pid")
  log "ComfyUI Anima lane child pid=$pid port=$COMFY_ANIMA_PORT"

  if ! wait_http "http://127.0.0.1:$COMFY_ANIMA_PORT/system_stats" 150; then
    log "ComfyUI Anima lane did not become healthy in time; $(wait_http_why "$pid"); continuing without it"
  fi
  fi
  # ---- End of the optional local lanes --------------------------------------

  cd "$APP" || return 1
  ZIMG_ACCELERATOR_PROFILE="$ZIMG_ACCELERATOR_PROFILE" \
  HIVEMIND_STUDIO_ROOT="$STUDIO_ROOT" \
  HIVEMIND_MEDIA_STATE_DIR="$MEDIA_STATE_ROOT" \
  ZIMG_TOKEN_FILE="$ZIMG_TOKEN_FILE" \
  SWIFT_FLUX2_BIN="$FLUX2_SERVER_DIR/.build/arm64-apple-macosx/release/Flux2CLI" \
  ZIMG_OUTPUT_DIR="$PRIV/z_image_outputs" \
  ZIMG_DEBUG_OUTPUT_DIR="$PRIV/debug_outputs" \
  ZIMG_OUTPUT_ENCRYPTION="${ZIMG_OUTPUT_ENCRYPTION:-1}" \
  ZIMG_E2E_MEDIA="${ZIMG_E2E_MEDIA:-1}" \
  ZIMG_AGENT_DUAL_SEAL="${ZIMG_AGENT_DUAL_SEAL:-1}" \
  ZIMG_USE_FLUX2_SERVER="${ZIMG_ENABLE_FLUX2_SERVER:-0}" \
  ZIMG_NATIVE_MXFP8_PROMPT_INTERCEPT="${ZIMG_NATIVE_MXFP8_PROMPT_INTERCEPT:-$ZIMG_ENABLE_APPLE_SILICON_OPTIMIZATIONS}" \
  ZIMG_ALLOW_MXFP8_COMFY_FALLBACK="${ZIMG_ALLOW_MXFP8_COMFY_FALLBACK:-0}" \
  SWIFT_FLUX2_SERVER_URL="http://127.0.0.1:$FLUX2_SERVER_PORT" \
	  COMFY_LANES="$comfy_lanes" \
	  COMFY_LANE_RULES="$comfy_lane_rules" \
	  COMFY_REMOTE_LANES="${RENTAL_COMFY_REMOTE_LANES:-}" \
  COMFY_HTTP_DEFAULT="http://127.0.0.1:$COMFY_DEFAULT_PORT" \
  COMFY_HTTP="http://127.0.0.1:$COMFY_DEFAULT_PORT" \
  with_credentials z-image-api "$STUDIO_PYTHON" app.py \
    >> "$PRIV/z-image-api.log" 2>&1 &
  pid="$!"
  children+=("$pid")
  log "Z-Image API child pid=$pid"

  if ! wait_http http://127.0.0.1:8787/health 45; then
    log "Z-Image API did not become healthy in time; $(wait_http_why "$pid"); see $PRIV/z-image-api.log"
    return 1
  fi

  # ── The three Node servers are one child now ────────────────────────────
  #
  # packages/media-gateway/node-services.mjs mounts all three Node surfaces on
  # one port ($NODE_SERVICES_PORT) — /canvas, /bridge and /agent — and keeps the
  # three old numbers answering as compatibility ports, so nothing that
  # addresses 8788, 8794 or 8796 by number has to change yet. One process, one
  # health endpoint, one thing to wait on.
  #
  # This replaced three separate children. Exactly what switched:
  #
  #   cd "$APP"
  #   COMFY_PRIVATE_VIEW_TOKEN=… HIVEMIND_STUDIO_ROOT=… HIVEMIND_MEDIA_STATE_DIR=… \
  #   ZIMG_TOKEN_FILE=… COMFY_MOBILE_DIST="$MOBILE/dist" COMFY_LANES=… \
  #   COMFY_HTTP_DEFAULT=… COMFY_HTTP=… HIVEMIND_STUDIO_TARGET=… \
  #   with_credentials z-image-frontend npm run start        # Canvas, :8788
  #   wait_http http://127.0.0.1:8788/healthz 45
  #
  #   cd "$OPEN_GEN"
  #   OGA_HOST=127.0.0.1 OGA_PORT="$OPEN_GEN_PORT" HIVEMIND_MEDIA_STATE_DIR=… \
  #   ZIMAGE_TOKEN_FILE=… \
  #   with_credentials open-generative-ai node hosted-server.js   # bridge, :8794
  #   wait_http "http://127.0.0.1:$OPEN_GEN_PORT/health" 30
  #
  #   cd "$APP"                                              # after the control API
  #   MEDIA_STUDIO_MCP_HOST=127.0.0.1 MEDIA_STUDIO_MCP_PORT="$MEDIA_STUDIO_MCP_PORT" \
  #   MEDIA_STUDIO_MCP_BACKEND_URL=http://127.0.0.1:8787 \
  #   MEDIA_STUDIO_MCP_STUDIO_URL=http://127.0.0.1:8788 \
  #   MEDIA_STUDIO_MCP_PUBLIC_STUDIO_URL=http://127.0.0.1:8788 \
  #   HIVEMIND_MEDIA_STATE_DIR=… MEDIA_STUDIO_TOKEN_FILE=… MEDIA_STUDIO_E2E_PUB_FILE=… \
  #   with_credentials media-studio-mcp node "$APP/bin/media-studio-mcp.mjs" --http  # :8796
  #   curl loop against http://127.0.0.1:$MEDIA_STUDIO_MCP_PORT/mcp
  #
  # MEDIA_STUDIO_MCP_PUBLIC_STUDIO_URL is loopback, always. It used to be the
  # 8789 proxy's address, which published the Canvas port to the whole tailnet;
  # that proxy is gone. The tailnet URL the remote-access switch makes belongs
  # to the CONTROL API, and the studio there resolves media through its own
  # account-gated routes.
  cd "$APP" || return 1
  HIVEMIND_NODE_SERVICES_HOST="127.0.0.1" \
  HIVEMIND_NODE_SERVICES_PORT="$NODE_SERVICES_PORT" \
  NODE_ENV="production" \
  HOST="127.0.0.1" \
  PORT="8788" \
  OGA_HOST="127.0.0.1" \
  OGA_PORT="$OPEN_GEN_PORT" \
  MEDIA_STUDIO_MCP_HOST="127.0.0.1" \
  MEDIA_STUDIO_MCP_PORT="$MEDIA_STUDIO_MCP_PORT" \
  COMFY_PRIVATE_VIEW_TOKEN="$COMFY_PRIVATE_VIEW_TOKEN" \
  HIVEMIND_STUDIO_ROOT="$STUDIO_ROOT" \
  HIVEMIND_MEDIA_STATE_DIR="$MEDIA_STATE_ROOT" \
  ZIMG_TOKEN_FILE="$ZIMG_TOKEN_FILE" \
  ZIMAGE_TOKEN_FILE="$ZIMG_TOKEN_FILE" \
  MEDIA_STUDIO_TOKEN_FILE="$ZIMG_TOKEN_FILE" \
  MEDIA_STUDIO_E2E_PUB_FILE="${MEDIA_STUDIO_E2E_PUB_FILE:-$MEDIA_STATE_ROOT/secure/agent-e2e-pub}" \
  MEDIA_STUDIO_MCP_BACKEND_URL="http://127.0.0.1:8787" \
  MEDIA_STUDIO_MCP_STUDIO_URL="http://127.0.0.1:8788" \
  MEDIA_STUDIO_MCP_PUBLIC_STUDIO_URL="http://127.0.0.1:8788" \
  COMFY_MOBILE_DIST="$MOBILE/dist" \
	  COMFY_LANES="$comfy_lanes" \
  COMFY_HTTP_DEFAULT="http://127.0.0.1:$COMFY_DEFAULT_PORT" \
  COMFY_HTTP="http://127.0.0.1:$COMFY_DEFAULT_PORT" \
  HIVEMIND_STUDIO_TARGET="http://127.0.0.1:$CONTENT_STUDIO_PORT" \
  with_credentials hivemind-node-services node node-services.mjs >> "$PRIV/hivemind-node-services.log" 2>&1 &
  pid="$!"
  children+=("$pid")
  log "Node services child pid=$pid url=http://127.0.0.1:$NODE_SERVICES_PORT (canvas :8788, bridge :$OPEN_GEN_PORT, mcp :$MEDIA_STUDIO_MCP_PORT)"

  # One wait, not three. /healthz on the shared port is 200 only when all three
  # surfaces came up; it names the one that did not when they did not.
  if ! wait_http "http://127.0.0.1:$NODE_SERVICES_PORT/healthz" "$NODE_SERVICES_TIMEOUT"; then
    log "Node services did not become healthy in time; $(wait_http_why "$pid"); see $PRIV/hivemind-node-services.log"
    return 1
  fi

  cd "$CONTENT_STUDIO" || return 1
  # The studio's own rotating log, beside the supervisor's, where Console.app
  # lists it and a person can find it — not in $PRIV, which is chflags hidden.
  # A DIFFERENT file from the redirect below: a shell '>>' and a rotating
  # handler on one path means the shell keeps writing to the rotated-away file.
  #
  # ZIMG_TOKEN_FILE is passed explicitly now: the control API presents that
  # token when it proxies /local-ai to the gated bridge, so if this process and
  # the Node child ever resolved the path differently, every local model call
  # would 401. Both are handed the same file, by name.
  HIVEMIND_STUDIO_ROOT="$STUDIO_ROOT" \
  CONTENT_STUDIO_LOG_DIR="$LOG_DIR/Hivemind Content Studio" \
  HIVEMIND_MEDIA_STATE_DIR="$MEDIA_STATE_ROOT" \
  ZIMG_TOKEN_FILE="$ZIMG_TOKEN_FILE" \
	  COMFY_LANES="$comfy_lanes" \
  COMFY_HTTP_DEFAULT="http://127.0.0.1:$COMFY_DEFAULT_PORT" \
  CONTENT_STUDIO_PROXY_SECRET="$CONTENT_STUDIO_PROXY_SECRET" \
  with_credentials hivemind-content-studio uv run content-studio-api \
    >> "$PRIV/hivemind-content-studio.log" 2>&1 &
  pid="$!"
  children+=("$pid")
  log "Hivemind Content Studio child pid=$pid url=http://127.0.0.1:$CONTENT_STUDIO_PORT"

  # /readyz, not /api/runtime: the latter probes three engines, so "the API is
  # up" and "the engines are up" used to be one answer. /readyz turns true only
  # once the accounts bootstrap and the catalog warm have both run.
  if ! wait_http "http://127.0.0.1:$CONTENT_STUDIO_PORT/readyz" 60; then
    log "Hivemind Content Studio did not become healthy in time; $(wait_http_why "$pid"); see $PRIV/hivemind-content-studio.log"
    return 1
  fi

  # The MCP endpoint used to start here as its own child on 8796. It is one of
  # the three surfaces the collapsed Node service brings up above, and its old
  # port still answers; the block it replaced is quoted there.

  # The API startup hook publishes only when explicitly requested. Serve
  # persists independently of the children, including across stack restarts.
  local tailnet_url
  tailnet_url="$(tailnet_studio_url || true)"
  log "tailnet URL: ${tailnet_url:-not published}"
}

# Recorded pids, never a name pattern.
#
# The other half of the 2026-09-04 incident: an agent cleaning up its own test
# servers ran `pkill -f "hosted-server.js"` and the pattern matched the owner's
# running child, because hosted-server.js is one of the three surfaces this
# supervisor's Node child loads. Nothing in this repo may reap by name — every
# stop goes through a pid this script wrote down, or through kill_port, which
# asks lsof who actually holds one of OUR ports. test_supervisor_diagnosis.py
# fails the build if a `pkill -f` or `killall` reappears anywhere in the tree.
stop_children() {
  log "stopping studio stack children"
  for pid in "${children[@]:-}"; do
    kill "$pid" 2>/dev/null || true
  done
  sleep 2
  for pid in "${children[@]:-}"; do
    kill -9 "$pid" 2>/dev/null || true
  done
  children=()
  # Only the lanes this stack runs itself. With no checkout of ours, whatever
  # answers on 8188 is the user's own ComfyUI and is not ours to reap.
  if comfy_available; then
	  kill_port "$COMFY_DEFAULT_PORT"
	  kill_port "$COMFY_LTX_PORT"
	  kill_port "$COMFY_ANIMA_PORT"
  fi
  kill_port 8787
  kill_port 8788
  kill_port "$NODE_SERVICES_PORT"
  kill_port "$MEDIA_STUDIO_MCP_PORT"
  kill_port "$FLUX2_SERVER_PORT"
  kill_port "$CONTENT_STUDIO_PORT"
  kill_port "$OPEN_GEN_PORT"
}

healthy() {
  # Do not use ComfyUI's HTTP endpoint as a watchdog during long Apple Silicon
  # generations. Heavy text encoders/samplers can legitimately make the prompt
  # server miss a short /system_stats deadline; the child-pid check above still
  # catches a real ComfyUI crash without killing active renders.
  curl -fsS --max-time 5 http://127.0.0.1:8787/health >/dev/null || return 1
  curl -fsS --max-time 5 "http://127.0.0.1:$CONTENT_STUDIO_PORT/api/runtime" >/dev/null || return 1
  # One probe for all three Node surfaces. It stays a SOFT fail for the same
  # reason the Canvas probe always was: a heavy Apple Silicon render can make
  # this child miss a short deadline, and killing the stack mid-generation is
  # worse than a slow health answer. Child-pid supervision still catches a real
  # crash.
  # A budget of 0 is one recorded probe, so the soft-fail line below can name
  # the cause the same way the boot gates do rather than only that it failed.
  if wait_http "http://127.0.0.1:$NODE_SERVICES_PORT/healthz" 0 10; then
    ZIMG_FRONTEND_HEALTH_SOFT_FAILS=0
  else
    ZIMG_FRONTEND_HEALTH_SOFT_FAILS=$(( ${ZIMG_FRONTEND_HEALTH_SOFT_FAILS:-0} + 1 ))
    log "node services health soft-failed ($ZIMG_FRONTEND_HEALTH_SOFT_FAILS/6); $(wait_http_why); keeping Comfy lanes alive"
    if [ "$ZIMG_FRONTEND_HEALTH_SOFT_FAILS" -eq 6 ]; then
      log "node services remained unresponsive during a possible accelerator-heavy render; child-pid supervision stays active and the generation stack will not be restarted"
    fi
  fi
  return 0
}

# The app's own settings document, exported for the children.
#
# The gateway, the bridge and ComfyUI read environment variables and nothing
# else, so a person who moves their models folder on the Settings page needs
# that choice to reach five processes that will never open a JSON file. It is
# the same document the control API reads (settings.py), it emits only the keys
# it actually sets, and the exporter shell-quotes every value — so what is
# eval'd here is a list of `export NAME=value` lines, not arbitrary shell.
load_studio_settings_env() {
  [ -x "$STUDIO_PYTHON" ] || return 0
  local exports=""
  exports="$("$STUDIO_PYTHON" -m hivemind_content_studio.settings --env 2>/dev/null)" || return 0
  [ -n "$exports" ] || return 0
  eval "$exports"
  log "settings document applied ($(printf '%s\n' "$exports" | grep -c '^export ') key(s))"
}

# Local stack overrides, dot-sourced the same way as the rental lanes overlay.
# The supervisor runs under launchd with a FIXED inherited environment, so an
# exported shell var never reaches the children — `FOO=1 zimage-stack restart`
# silently keeps the old value. A file is the only handle that survives a
# restart. Machine-local on purpose: never put secrets here (that is the shared
# hive env's job) and never anything that should replicate across the fleet.
load_stack_local_env() {
  # Settings first, the hand-written overlay second: stack-local.env stays the
  # developer escape hatch, so it has to win over the app's own document.
  load_studio_settings_env
  local overlay="$MEDIA_STATE_ROOT/stack-local.env"
  if [ -f "$overlay" ]; then
    # shellcheck disable=SC1090
    . "$overlay"
    log "local stack overlay loaded: $overlay"
  fi
}

supervise() {
  load_stack_local_env
  if ! mkdir "$LOCK_DIR" 2>/dev/null; then
    local old_pid=""
    [ -f "$LOCK_DIR/pid" ] && old_pid="$(cat "$LOCK_DIR/pid" 2>/dev/null || true)"
    if [ -n "$old_pid" ] && kill -0 "$old_pid" 2>/dev/null; then
      log "another supervisor is already running pid=$old_pid; exiting"
      exit 0
    fi
    rm -rf "$LOCK_DIR"
    mkdir "$LOCK_DIR" 2>/dev/null || { log "could not acquire supervisor lock; exiting"; exit 0; }
  fi
  printf '%s\n' "$$" > "$LOCK_DIR/pid"
  trap 'stop_children; rm -rf "$LOCK_DIR"; exit 0' TERM INT HUP EXIT
  /usr/bin/caffeinate -i -m -w $$ >> "$SUP_LOG" 2>&1 &
  children+=("$!")
  while true; do
    stop_children
    if ! start_children; then
      log "startup failed; retrying in 10s"
      stop_children
      sleep 10
      continue
    fi
    local keys_locked=0 ticks=0
    if studio_keys_locked; then
      keys_locked=1
      log "PassBook's vault is locked, so the studio started without its keys (renting, Civitai, cloud models and gated downloads will fail). Run 'passbook signin'; the stack restarts itself once the vault opens."
    fi
    while true; do
      sleep 20
      ticks=$((ticks + 1))
      if [ "$keys_locked" = 1 ] && [ $((ticks % 3)) -eq 0 ] && ! studio_keys_locked; then
        if renders_idle; then
          log "PassBook's vault is open now; restarting the stack so the studio gets its keys"
          break
        fi
      fi
      for pid in "${children[@]:-}"; do
        if ! kill -0 "$pid" 2>/dev/null; then
          log "child pid $pid exited; restarting whole stack"
          break 2
        fi
      done
      if ! healthy; then
        log "health check failed; restarting whole stack"
        break
      fi
    done
  done
}

launch_start() {
  launchctl setenv CONTENT_STUDIO_REMOTE_ACCESS "$REMOTE_ACCESS" || return 1
  launchctl setenv CONTENT_STUDIO_TAILNET_PORT "$TAILNET_HTTPS_PORT" || return 1
  launchctl bootout "$DOMAIN/$OLD_CF_LABEL" 2>/dev/null || true
  launchctl bootout "$DOMAIN/$OLD_OPEN_GEN_LABEL" 2>/dev/null || true
  launchctl disable "$DOMAIN/$OLD_OPEN_GEN_LABEL" 2>/dev/null || true
  launchctl bootout "$DOMAIN/$LABEL" 2>/dev/null || true
  launchctl setenv ZIMG_ACCELERATOR_PROFILE "$ZIMG_ACCELERATOR_PROFILE" 2>/dev/null || true
	  if [ "${ZIMG_ENABLE_FLUX2_SERVER:-0}" = "1" ]; then
	    launchctl setenv ZIMG_ENABLE_FLUX2_SERVER 1 2>/dev/null || true
	  else
	    launchctl unsetenv ZIMG_ENABLE_FLUX2_SERVER 2>/dev/null || true
	  fi
	  if [ "${COMFY_ENABLE_LTX_LANE:-0}" = "1" ]; then
	    launchctl setenv COMFY_ENABLE_LTX_LANE 1 2>/dev/null || true
	  else
	    launchctl unsetenv COMFY_ENABLE_LTX_LANE 2>/dev/null || true
	  fi
  # The `launchctl bootout` above is ASYNCHRONOUS, so the first bootstrap here
  # usually races it and fails with EIO ("Bootstrap failed: 5: Input/output
  # error", plus launchd's own advice to re-run as root). That is a normal race
  # this loop exists to absorb — it is not something the person restarting the
  # stack has to do anything about, and telling them to try again as root sends
  # them somewhere there is no problem. So the attempts are quiet, and only a
  # real failure of all five speaks, carrying what launchd actually said.
  local loaded=0 boot_error=''
  for _ in 1 2 3 4 5; do
    if boot_error="$(launchctl bootstrap "$DOMAIN" "$HOME/Library/LaunchAgents/$LABEL.plist" 2>&1)"; then
      loaded=1
      break
    fi
    sleep 1
  done
  if [ "$loaded" -ne 1 ]; then
    echo "Could not hand $LABEL to launchd after five attempts." >&2
    [ -n "$boot_error" ] && echo "launchd said: ${boot_error//$'\n'/ }" >&2
    echo "The old job may still be shutting down — wait a moment and run 'zimage-stack restart' again." >&2
    exit 1
  fi
  launchctl kickstart -k "$DOMAIN/$LABEL" 2>/dev/null || true
  echo "Hivemind Content Studio starting as one managed app."
  echo "Local URL: http://127.0.0.1:$CONTENT_STUDIO_PORT"
  local tailnet_url
  tailnet_url="$(tailnet_studio_url || true)"
  if [ -n "$tailnet_url" ]; then
    echo "Tailnet URL: $tailnet_url (published; turn it off in the studio to stop)"
  else
    echo "Tailnet URL: not published. Turn on remote access in the studio to open it on your other devices."
  fi
}

launch_stop() {
  launchctl bootout "$DOMAIN/$OLD_CF_LABEL" 2>/dev/null || true
  launchctl bootout "$DOMAIN/$LABEL" 2>/dev/null || true
  # Only the lanes this stack runs itself. With no checkout of ours, whatever
  # answers on 8188 is the user's own ComfyUI and is not ours to reap.
  if comfy_available; then
	  kill_port "$COMFY_DEFAULT_PORT"
	  kill_port "$COMFY_LTX_PORT"
	  kill_port "$COMFY_ANIMA_PORT"
  fi
  kill_port 8787
  kill_port 8788
  kill_port "$NODE_SERVICES_PORT"
  kill_port "$MEDIA_STUDIO_MCP_PORT"
  kill_port "$FLUX2_SERVER_PORT"
  kill_port "$CONTENT_STUDIO_PORT"
  kill_port "$OPEN_GEN_PORT"
  echo "Hivemind Content Studio stopped."
}

status() {
  # Read the same overlay `supervise` runs with, or every flag below reports the
  # pre-overlay default: with the Flux2 server forced on in stack-local.env this
  # printed "server default: 0" next to a live listener on 8791.
  load_stack_local_env >/dev/null 2>&1
  echo "LaunchAgent:"
  launchctl print "$DOMAIN/$LABEL" 2>/dev/null | sed -n '1,20p' || echo "  not loaded"
  echo
  echo "Hardware profile: $ZIMG_ACCELERATOR_PROFILE (Apple Silicon optimizations: $ZIMG_ENABLE_APPLE_SILICON_OPTIMIZATIONS)"
	  echo "Default lane ASFP8: INT8=$COMFY_DEFAULT_ASFP8_INT8_EXT FP8=$COMFY_DEFAULT_ASFP8_FP8_EXT"
	  echo "Comfy attention flag: ${COMFY_DEFAULT_ATTENTION:-none}"
	  echo "Dedicated LTX lane: $COMFY_ENABLE_LTX_LANE (port $COMFY_LTX_PORT, Apple Silicon gpu-only + MPS LoRA bypass)"
	  echo "LTX priority mode: $ZIMG_LTX_PRIORITY_MODE (Flux2 forced: ${ZIMG_FORCE_FLUX2_SERVER:-0})"
	  echo "Flux2 Swift/MLX server default: $ZIMG_ENABLE_FLUX2_SERVER"
  echo
  echo "Listeners:"
	  for port in "$CONTENT_STUDIO_PORT" "$NODE_SERVICES_PORT" "$OPEN_GEN_PORT" "$COMFY_DEFAULT_PORT" "$COMFY_LTX_PORT" "$COMFY_ANIMA_PORT" 8787 8788 "$MEDIA_STUDIO_MCP_PORT" "$FLUX2_SERVER_PORT"; do
    pids=$(lsof -tiTCP:"$port" -sTCP:LISTEN 2>/dev/null || true)
    if [ -n "$pids" ]; then echo "  :$port pid(s): $pids"; else echo "  :$port not listening"; fi
  done
  echo
  local tailnet_url
  echo "Local URL: http://127.0.0.1:$CONTENT_STUDIO_PORT"
  tailnet_url="$(tailnet_studio_url || true)"
  if [ -n "$tailnet_url" ]; then
    echo "Tailnet URL: $tailnet_url"
    echo "  Reachable by every device signed in to this tailnet. The Canvas port is NOT published."
  else
    echo "Tailnet URL: not published (no tailscale serve entry proxies :$CONTENT_STUDIO_PORT). Turn on remote access in the studio to open it on your other devices."
  fi
}

case "$STACK_ACTION" in
  supervise) supervise ;;
  start) launch_start ;;
  stop) launch_stop ;;
  restart) launch_stop; sleep 2; launch_start ;;
  status) status ;;
  url)
    tailnet_url="$(tailnet_studio_url || true)"
    if [ -n "$tailnet_url" ]; then echo "$tailnet_url"; else echo "http://127.0.0.1:$CONTENT_STUDIO_PORT"; fi
    ;;
  *) echo "Usage: zimage-stack {start|stop|restart|status|url|supervise} [--remote-access] [--tailnet-port PORT]" >&2; exit 2 ;;
esac
