#!/usr/bin/env zsh
# ============================================================================
# GENERATED FILE — DO NOT EDIT. Assembled from lib/ferry-*.zsh by build.zsh.
# Edit the modules under lib/ and run ./build.zsh to regenerate this file.
# ============================================================================
# ferry — Unified LAN AI Relay and Proxy Manager (macOS + Linux).
# Consolidates local GPU model execution (macOS / Apple Silicon only), cloud API
# proxies, and LAN sharing. On Linux the cloud/route/client/dash/share/transfer
# features all work; local GPU serving is macOS-only (use --route/--cloud/--model).
# Supports both Host Mode and Client Mode (on connecting laptops).
#
# Usage:
#   ferry install                # Provision dependencies, models, and global link
#   ferry up [options]           # [Host] Start local GPU server or cloud proxy (Gemini)
#   ferry down                   # [Host] Stop all relay servers, proxies, and shares
#   ferry status                 # [Dual] View active status, LAN IPs, and test connections
#   ferry share                  # [Host] Expose client-bootstrap.sh over LAN
#   ferry msg <text>             # [Client] Send a direct text message to host's ~/.config/ferry/client_logs.txt
#   ferry log                    # [Client] Pipe stdin log stream directly back to host
#   ferry offer <path>...        # [Host] Offer files/dirs for clients to fetch over the LAN
#   ferry pull <model-id>        # [Client] Pull a model from the host cache (http|hf|nc transports)
#   ferry get <name>             # [Client] Fetch an offered file/dir from the host
#   ferry send <path> <client>   # [Host] Push a file/dir to a listening client via netcat
#   ferry receive                # [Client] Listen for and receive a netcat tar stream
#   ferry serve-hf               # [Host] Start an EXPERIMENTAL HuggingFace pass-through proxy
#   ferry serve-proxy            # [Host] Start a general HTTP(S) forward proxy for client downloads
#   ferry env                    # [Client] Emit shell exports so this laptop downloads via the host proxy
#   ferry opencode               # [Client] Auto-wire opencode to route through the host (detects served models)

set -eu

APP_DIR="$(dirname "${0:A}")"
# The script's own resolved path, captured HERE at load time: inside a function
# `$0` is the function's name, so a command that needs to re-invoke ferry (the
# relay backgrounding itself) cannot compute this for itself.
FERRY_BIN_PATH="${0:A}"

# ---- OS detection & portable host helpers (macOS + Linux) ----
case "$(uname -s)" in
  Darwin) IS_MAC=1 ;;
  *)      IS_MAC=0 ;;
esac

# detect_lan_ip — echo one primary, non-loopback IPv4 address for this host.
#   macOS: parse ifconfig directly (bypasses macOS ipconfig getifaddr quirks),
#          preferring RFC 1918 private subnets, then any non-Tailscale IP, then
#          the first address found.
#   Linux: ifconfig is often absent on Ubuntu — use iproute2 (`ip`), then
#          `hostname -I`, then ifconfig as a last resort.
detect_lan_ip() {
  if (( IS_MAC )); then
    # Parse all active non-loopback IPv4 addresses directly from ifconfig.
    local ip_list=($(ifconfig 2>/dev/null | grep "inet " | grep -v "127.0.0.1" | awk '{print $2}'))
    if [[ ${#ip_list[@]} -gt 0 ]]; then
      # 1. Prefer standard RFC 1918 private subnets (192.168.x.x, 10.x.x.x, 172.16-31.x)
      for ip in "${ip_list[@]}"; do
        if [[ "$ip" == 192.168.* || "$ip" == 10.* || "$ip" == 172.1[6-9].* || "$ip" == 172.2[0-9].* || "$ip" == 172.3[0-1].* ]]; then
          echo "$ip"
          return
        fi
      done
      # 2. Next, prefer any IP that isn't Tailscale (starts with 100.)
      for ip in "${ip_list[@]}"; do
        if [[ "$ip" != 100.* ]]; then
          echo "$ip"
          return
        fi
      done
      # 3. Fall back to first IP found (which could be Tailscale)
      echo "${ip_list[1]}"
      return
    fi
    echo "Unknown-IP"
  else
    # Linux: iproute2 first, then hostname -I, then ifconfig.
    local ip
    ip=$(ip -4 -o addr show scope global 2>/dev/null | awk '{print $4}' | cut -d/ -f1 | head -1)
    [[ -z "$ip" ]] && ip=$(hostname -I 2>/dev/null | awk '{print $1}')
    [[ -z "$ip" ]] && ip=$(ifconfig 2>/dev/null | grep "inet " | grep -v "127.0.0.1" | awk '{print $2}' | head -1)
    [[ -n "$ip" ]] && echo "$ip" || echo "Unknown-IP"
  fi
}

# detect_mdns_name — this host's advertised .local name, lowercased.
#   macOS: scutil --get LocalHostName + ".local".
#   Linux: <hostname -s> + ".local" (avahi advertises <hostname>.local).
#   Falls back to "localhost.local" if the short name can't be resolved.
detect_mdns_name() {
  local base=""
  if (( IS_MAC )); then
    base="$(scutil --get LocalHostName 2>/dev/null | tr 'A-Z' 'a-z')"
  else
    base="$(hostname -s 2>/dev/null | tr 'A-Z' 'a-z')"
  fi
  [[ -z "$base" ]] && base="localhost"
  echo "${base}.local"
}

# ---- Ports ----
# In STACK mode (plain `ferry up`) PORT is the ONE door clients use: litellm sits
# there and fans out to the cloud lanes plus the two MLX backends below, which
# listen on their own ports and are NOT meant to be addressed directly by clients.
PORT="8090"               # litellm front door — the single LAN endpoint
SHARE_PORT="8095"
HF_PORT="8096"
PROXY_PORT="8097"
RELAY_PORT="8098"         # reverse-expose control port — clients dial IN to publish OUT
LOCAL_ORCH_PORT="8092"    # MLX backend for the `local-orch` lane
LOCAL_SUB_PORT="8093"     # MLX backend for the `local-sub` lane
# Dedicated door for the schematron extraction lane (`ferry up --schematron`):
# a scraper workload targets :8094 while the stack keeps :8090, so neither
# door's restarts can disturb the other. 8094 is the one gap left in the
# 8090-8099 block (only scripts/bench-spec-ab.py ever borrows it, ad hoc).
SCHEMATRON_PORT="${FERRY_SCHEMATRON_PORT:-8094}"
# MLX backend for the `local-schematron` lane (v1.36.0). 8100, not a gap in the
# 8090-8099 block: that block is FULL — 8090 front, 8091 dash, 8092/8093 the two
# older MLX lanes, 8094 the schematron door, 8095 share, 8096 HF, 8097 proxy,
# 8098 relay, 8099 VNC. So the ferry range continues upward at 8100 rather than
# reaching below the front door.
LOCAL_SCHEMATRON_PORT="${FERRY_LOCAL_SCHEMATRON_PORT:-8100}"
# NOTE: 8091 is deliberately skipped — `ferry dash` binds it. The stack and the
# dashboard are meant to run together, so the lanes start above it.

# ---- The five served lanes ----
# Lane names are the STABLE contract clients bind to; the model behind a lane is
# swappable without touching a single client config.
#
#   heavy        -> GPT-5.6 Sol via the ChatGPT subscription, no fallback chain
#                   (legacy names `orch`/`orchestrator` still resolve to it)
#   flash        -> ~google/gemini-flash-latest via OpenRouter (currently
#                   Gemini 3.8 Flash), xhigh reasoning, Terra fallback
#   super-flash  -> Gemini Flash Latest via OpenRouter, minimal reasoning for
#                   compaction/title/summary; Gemini-only, no model fallback
#   local-orch   -> Qwen3.8-27B-nvfp4 on the host GPU (+ MTP speculative draft)
#   local-sub    -> NVIDIA Nemotron 3 Nano 30B A3B NVFP4 on the host GPU
#   schematron   -> Schematron-8B 8-bit MLX on the host GPU (HTML→JSON)
#
# The cloud lanes live in the litellm route config; the three local lanes are the
# MLX servers this script launches, wired into that same config as
# openai-compatible backends on 127.0.0.1.

# Local ORCHESTRATOR lane — the "smart" local model: dense-ish 27B at nvfp4.
# MTP speculative decoding ON with UNQUANTIZED KV @64k (see the per-lane note
# below for the crash that rules out draft+quantized-KV).
LOCAL_MODEL_ORCH="mlx-community/Qwen3.8-27B-nvfp4"
LOCAL_DRAFT_ORCH="mlx-community/Qwen3.8-27B-MTP-8bit"

# Local SUBAGENT lane — nemotron_h hybrid MoE (6/52 full-attention layers, 2 KV
# heads) => ~6KB KV/token and ~3B active params, so several concurrent subagents
# stay fast and cheap on memory. No MTP draft is published for it -> no
# --draft-model on this lane.
LOCAL_MODEL_SUB="mlx-community/NVIDIA-Nemotron-3-Nano-30B-A3B-NVFP4"
LOCAL_DRAFT_SUB=""

# Local EXTRACTION lane (v1.36.0) — the `schematron` lane runs ON-MACHINE.
# Schematron-8B (a llama-arch fine-tune of Llama-3.1-8B specialised for
# HTML→JSON structured extraction) at 8-bit MLX quant: ~8.5GB of weights, 32
# layers, GQA with 8 KV heads x 128 head dim, 128k context. No MTP draft is
# published for it -> no --draft-model on this lane.
#
# The PROMPT CONTRACT is the CLIENT's job, not ferry's. Schematron expects the
# JSON schema INSIDE the user message (system "You are a helpful assistant";
# user = "You are going to be given a JSON schema ... The schema is as
# follows:\n\n<schema>\n\nHere is the HTML page:\n\n<html>\n\nMAKE SURE ITS
# VALID JSON."). ferry fronts the model verbatim and never rewrites prompts —
# cdp-toolkit's extract_page is the caller that builds that shape.
LOCAL_MODEL_SCHEMATRON="pchamart/schematron8B-mlx-8bit"
LOCAL_DRAFT_SCHEMATRON=""

# Back-compat: the single-lane `--local` flag predates the stack and still means
# "serve the local orchestrator model on its own".
LOCAL_MODEL="$LOCAL_MODEL_ORCH"
LOCAL_DRAFT="$LOCAL_DRAFT_ORCH"

# KV-cache memory governor (measured 2026-08-25, 128GB M5 Max): the stock launch
# kept the full fp16 KV of EVERY request in the APC prefix cache — one 121k-token
# opencode session peaked the server at 97GB phys_footprint (GPU wired ceiling is
# ~90-100GB on a 128GB Mac => the "caps out around 30k tokens" wall under
# concurrent agents; `ps` RSS is blind to this, use `footprint <pid>`). With
# 4-bit KV + a bounded APC pool + a concurrent-seq cap, the SAME session peaked
# 56GB, idled at 35GB, and decoded ~60% faster (32 vs 20 tok/s at 64k context).
# Set any of these to "" to drop the flag from the launch line.
#
# STACK MODE runs ALL THREE local lanes at these same generous settings (~15GB +
# ~18GB + ~8.5GB of resident weights before any KV — v1.36.0 added the third).
# That is a deliberate choice for best single-lane latency; the tradeoff is that
# simultaneously-busy deep-context lanes CAN approach the wired ceiling. Watch it
# with `ferry status`, and shrink a lane by exporting the per-lane overrides below.
LOCAL_KV_BITS="4"         # --kv-bits: 4-bit KV cache quant (weights stay nvfp4)
LOCAL_MAX_KV="131072"     # --max-kv-size: prompt+max_tokens over this => clean 400, not OOM
LOCAL_MAX_SEQS="4"        # --max-num-seqs: max concurrent sequences (subagent fan-out)
LOCAL_APC_BLOCKS="512"    # APC_NUM_BLOCKS: retained prefix-pool size (x16 tokens)

# Per-lane overrides — export any of these to govern ONE lane without touching the
# other (e.g. LOCAL_SUB_MAX_KV=65536 to shrink the subagent lane's context budget).
#
# local-orch runs MTP speculative decoding + UNQUANTIZED KV @64k (measured
# 2026-08-25): the draft-verify path crashes on any quantized cache (tuple keys,
# AttributeError — turn 2 of every conversation 500s), but draft + full KV is
# stable (3/3 cache-hit requests 200) and decodes ~53% faster (37.8 vs 24.8
# tok/s). Full KV @64k = 16GB (256KB/token, 64 layers x 4 kv heads x 256 dim).
# To revert to the quantized no-draft lane: LOCAL_DRAFT_ORCH="" + kv-bits 4.
LOCAL_ORCH_KV_BITS=""
LOCAL_ORCH_MAX_KV="65536"
LOCAL_ORCH_MAX_SEQS="${LOCAL_ORCH_MAX_SEQS:-$LOCAL_MAX_SEQS}"
LOCAL_ORCH_APC_BLOCKS="${LOCAL_ORCH_APC_BLOCKS:-$LOCAL_APC_BLOCKS}"
LOCAL_SUB_KV_BITS="${LOCAL_SUB_KV_BITS:-$LOCAL_KV_BITS}"
LOCAL_SUB_MAX_KV="${LOCAL_SUB_MAX_KV:-$LOCAL_MAX_KV}"
LOCAL_SUB_MAX_SEQS="${LOCAL_SUB_MAX_SEQS:-$LOCAL_MAX_SEQS}"
LOCAL_SUB_APC_BLOCKS="${LOCAL_SUB_APC_BLOCKS:-$LOCAL_APC_BLOCKS}"
# local-schematron runs UNQUANTIZED KV at the model's full 128k context. The
# arithmetic is why that is affordable: 32 layers x 8 kv heads x 128 head dim x
# 2 (K and V) x 2 bytes = 131 KB/token, so a full 128k stream is ~17GB of KV in
# the worst case and a realistic single-page extraction is a small fraction of
# that. Quantizing the cache would buy little and cost extraction fidelity on a
# lane whose entire job is verbatim copying out of the prompt. max-num-seqs is
# 2, not 4: this lane's prompts are whole HTML pages (deep prefill, shallow
# decode), so admitting many concurrent extractions grows KV far faster than it
# grows throughput.
LOCAL_SCHEMATRON_KV_BITS=""
LOCAL_SCHEMATRON_MAX_KV="${LOCAL_SCHEMATRON_MAX_KV:-131072}"
LOCAL_SCHEMATRON_MAX_SEQS="${LOCAL_SCHEMATRON_MAX_SEQS:-2}"
LOCAL_SCHEMATRON_APC_BLOCKS="${LOCAL_SCHEMATRON_APC_BLOCKS:-$LOCAL_APC_BLOCKS}"
MDNS_NAME="$(detect_mdns_name)"

# Default cloud model for `ferry serve --cloud`.
#
# This one HAS to be a real provider/model id, not a lane name: --cloud runs a
# bare `litellm --model "$DEFAULT_CLOUD_MODEL"` with no route config at all, so
# there is no lane table to resolve against. It rides OpenRouter deliberately —
# one env var (OPENROUTER_API_KEY) and the stock api_base.
#
# Was `gemini/gemini-3.7-flash` (as DEFAULT_GEMINI) until v1.8.4. v1.8.0 retired
# the multi-project Gemini key pool, so `--cloud` had been defaulting to a key
# the host no longer holds and failing its own preflight check.
#
# Was `openrouter/z-ai/glm-5.3-flash` from v1.8.4 until 2026-09-04, when the
# route config was simplified and the `flash` lane it mirrors moved to
# OpenRouter Gemini 3.8 Flash.
DEFAULT_CLOUD_MODEL="openrouter/google/gemini-3.8-flash"

# Route mode: serve multiple models (orchestrator + failover workers) from a litellm config
DEFAULT_ROUTE_CONFIG="$HOME/.config/ferry/litellm.yaml"
ROUTE_TEMPLATE="$APP_DIR/litellm-route-example.yaml"

# Robust local IP discovery (OS-aware; see detect_lan_ip near the top of this
# script). macOS keeps the RFC-1918-prioritizing ifconfig parse; Linux uses the
# iproute2 path since ifconfig is often absent on Ubuntu.
get_lan_ip() {
  detect_lan_ip
}
LAN_IP=$(get_lan_ip)

# Dual-Mode Configuration: Detect if running on a Client laptop
CLIENT_MODE=0
CLIENT_HOST=""
CLIENT_PORT="8090"
CLIENT_SHARE_PORT="8095"
# v1.22.0: optional litellm master_key (LITELLM_MASTER_KEY) on the front door.
# Absent => the generators bake the legacy 'local' bearer, so a keyless LAN
# setup is unchanged. Held in a variable, never echoed.
CLIENT_MASTER_KEY=""
# v1.26.0: CLIENT_NAME identifies this caller to the front door's fleet
# resolver (X-Ferry-Client). On the host it is the literal 'host', matching
# the loopback identity the front door already assigns; on a client it comes
# from client.json's 'name', falling back to the short hostname (lower-cased)
# so a profile bootstrapped before this field existed still resolves an
# identity.
CLIENT_NAME="host"

CLIENT_CONF="$HOME/.config/ferry/client.json"
if [[ -f "$CLIENT_CONF" ]]; then
  CLIENT_MODE=1
  CLIENT_HOST=$(python3 -c "import json, os; print(json.load(open(os.path.expanduser('$CLIENT_CONF'))).get('host', ''))" 2>/dev/null || echo "")
  CLIENT_PORT=$(python3 -c "import json, os; print(json.load(open(os.path.expanduser('$CLIENT_CONF'))).get('port', '8090'))" 2>/dev/null || echo "8090")
  CLIENT_SHARE_PORT=$(python3 -c "import json, os; print(json.load(open(os.path.expanduser('$CLIENT_CONF'))).get('share_port', '8095'))" 2>/dev/null || echo "8095")
  CLIENT_MASTER_KEY=$(python3 -c "import json, os; print(json.load(open(os.path.expanduser('$CLIENT_CONF'))).get('master_key') or '')" 2>/dev/null || echo "")
  CLIENT_NAME=$(python3 -c "import json, os; print(json.load(open(os.path.expanduser('$CLIENT_CONF'))).get('name') or '')" 2>/dev/null || echo "")
  [[ -z "$CLIENT_NAME" ]] && CLIENT_NAME=$(hostname -s 2>/dev/null | tr 'A-Z' 'a-z')
fi

# Logging locations (Host Mode only)
LOG_DIR="${TMPDIR:-/tmp}/ferry-logs"
mkdir -p "$LOG_DIR"
LOCAL_LOG="$LOG_DIR/local-gpu-$PORT.log"
CLOUD_LOG="$LOG_DIR/cloud-proxy-$PORT.log"
# Stack mode gives each MLX lane its own log so a crash is attributable to a lane.
LOCAL_ORCH_LOG="$LOG_DIR/local-orch-$LOCAL_ORCH_PORT.log"
LOCAL_SUB_LOG="$LOG_DIR/local-sub-$LOCAL_SUB_PORT.log"
LOCAL_SCHEMATRON_LOG="$LOG_DIR/local-schematron-$LOCAL_SCHEMATRON_PORT.log"
SHARE_LOG="$LOG_DIR/share-$SHARE_PORT.log"

# Client telemetry (`ferry msg` / `ferry log` -> the share server's /hq) lands here.
# NOT under $LOG_DIR: this one must outlive the checkout the share server was
# launched from and the temp dir the lane logs live in, so `ferry inbox` can still
# read it after a worktree is removed (v1.8.10). The share server's embedded handler
# carries this same path as a literal, because it runs as its own process — a test
# pins the two spellings together.
HQ_LOG="$HOME/.config/ferry/client_logs.txt"

# Reverse-expose state. The token is the only thing separating "a client of yours"
# from "anything that can reach the LAN", so it is 0600 and never logged. The
# published-ports file is what `ferry status` reads, written atomically by the
# relay so a status read never catches it half-written.
RELAY_TOKEN_FILE="$HOME/.config/ferry/relay-token"
RELAY_STATE_FILE="$HOME/.config/ferry/relay-published.json"
RELAY_LOG="$LOG_DIR/relay-$RELAY_PORT.log"

# Browser VNC viewer (`ferry serve-vnc`): noVNC served from the host plus a
# WebSocket->TCP bridge onto ports the relay has published as kind=vnc.
VNC_PORT="8099"
VNC_LOG="$LOG_DIR/vnc-$VNC_PORT.log"
NOVNC_VERSION="1.7.0"
NOVNC_URL="${FERRY_NOVNC_URL:-https://github.com/novnc/noVNC/archive/refs/tags/v$NOVNC_VERSION.tar.gz}"
NOVNC_SHA256="${FERRY_NOVNC_SHA256:-b1003a11b6e6e8d8f7f5e5586daae7f8ca651d8aee0aa155ff9ac841c48f52c6}"
NOVNC_DIR="$HOME/.config/ferry/novnc"

# Load local secrets if present (e.g. GEMINI_API_KEY). Export the variable in your
# shell, or drop it in ~/.config/ferry/secrets.env — never commit real API keys.
if [[ -f "$HOME/.config/ferry/secrets.env" ]]; then
  source "$HOME/.config/ferry/secrets.env" >/dev/null 2>&1 || true
fi

# Banner Help
usage() {
  cat <<EOF
LLM-Ferry CLI (ferry) — Decoupled Local AI LAN sharing & Cloud proxying.

Usage:
  ferry <command> [options]

Commands:
  install            Install uv, litellm, and link globally (+ mlx-vlm & models on macOS)
  up                 [Host] Start local GPU server or cloud API proxy (boots Catalog by default)
  down               [Host] Stop all running servers (local, cloud, sharing)
                        ferry down [--port P]   # --port P: stop ONLY the ferry proxy on :P
  reload             [Host] Restart ONLY the litellm front door (re-reads the
                       route config); the GPU lanes stay warm. The fast path for
                       editing ~/.config/ferry/litellm.yaml.
  status             [Dual] Show server listeners, active models, and client test commands
  update             [Dual] Catch this machine up. Detects host vs client from
                       ~/.config/ferry/client.json and runs that role's reset:
                       a host rebuilds from its checkout, a client re-pulls
                       the CLI from its host
                       ferry update [--full] [--host|--client] [--dry-run]
  migrate            [Client] Promote THIS machine from a client into a host:
                       ensure a repo checkout (clone if needed), carry the master
                       key forward, seed the route config, provision host deps,
                       archive the client profile, and bring the endpoint up local
                       ferry migrate [--dir PATH] [--repo URL] [--full] [--pull] [--dry-run] [--yes]
  dash               [Host] Live dashboard for the route proxy
                       ferry dash [--open] [--port P] [--ferry URL]   # lightweight stdlib page (localhost:8091)
                       ferry dash --grafana [--open]                  # full Grafana+VictoriaMetrics stack (localhost:3001)
                       ferry dash --grafana --down                    # stop the Grafana stack
  share              [Host] Expose client-bootstrap.sh over LAN for other laptops to curl
  msg <text>         [Client] Send a direct text message to host's ~/.config/ferry/client_logs.txt
  log                [Client] Pipe stdin log stream directly back to host
  inbox              [Host] Read what clients sent — dated and attributed where the
                       share log still has the receipt
                       ferry inbox [-n N] [-f] [--all] [--path]
  relay              [Host] Accept reverse-expose registrations so a client can
                       publish one of ITS local ports through this host
                       ferry relay [--port P] [--bind ADDR] [--token] [--foreground]
  expose <port>      [Client] Publish 127.0.0.1:<port> from the host, dialling only
                       outbound — for a laptop that cannot accept inbound at all
                       ferry expose <port> [--as PUBLIC] [--host H] [--token T]
  expose-vnc         [Client] Publish this machine's VNC server (5900) through the host,
                       tagged so 'ferry serve-vnc' can show it in a browser
                       ferry expose-vnc [--local PORT] [--as PUBLIC] [--host H] [--port P] [--token T]
  env                [Client] Emit shell exports so downloads route via the host proxy
                       eval "\$(ferry env --host H)"  [--proxy-port P] [--hf-port P2] [--write]
  opencode           [Client] Auto-wire opencode to route through the host (detects served models)
                        ferry opencode [--host H] [--port P] [--config PATH] [--model M] [--small-model SM] [--super] [--no-default]
                        ferry opencode [--tui-config PATH | --no-tui-config] [--keep-cache] [--no-install]
  claude             [Dual] Wire Claude Code to the ferry endpoint: installs the
                       claude-ferry / claude-ferry-local wrappers (cloud: heavy/flash,
                       local: local-orch/local-sub) and records the lane map
                        ferry claude [--host H] [--port P] [--wrappers]
  fleet              [Dual] Read or switch which fleet (routing set) bare lane
                       names resolve to
                        ferry fleet ls | show | use <fleet> [--default] | use --clear

Ferrying models & files across the LAN:
  offer <path>...    [Host] Record files/dirs in ~/.config/ferry/offered.json for clients to fetch
  pull <model-id>    [Client] Pull a model from the host's local HF cache
                       [--host H] [--port P] [--transport http|hf|nc] [--to DIR]
                       http (default): stream+untar from the share server
                       hf:  download THROUGH the host's 'ferry serve-hf' proxy (EXPERIMENTAL)
                       nc:  listen for a netcat push (then run 'ferry send' on the host)
  get <name>         [Client] Fetch an offered file/dir  [--host H] [--port P] [--to DIR]
  receive            [Client] Listen for a netcat tar stream  [--port P] [--to DIR]
  send <path> <cli>  [Host] Push a file/dir to a listening client  [--port P]
  serve-hf           [Host] Start EXPERIMENTAL HuggingFace pass-through proxy [--port P] (default $HF_PORT)
  serve-proxy        [Host] Start a general HTTP(S) forward proxy for client downloads [--port P] (default $PROXY_PORT)
  serve-vnc          [Host] Serve the browser VNC viewer for ports published with
                       'ferry expose-vnc' [--port P] [--bind ADDR] [--foreground] [--fetch] (default $VNC_PORT)

Encrypted transfer over an UNTRUSTED channel (no LAN required):
  drop <file>|-      [Dual] Encrypt to a self-contained .ferrydrop blob + print a fresh
                       passphrase. Move the blob however you like; it is safe on a
                       channel ferry does not trust. Send the passphrase separately.
                       ferry drop <file> [--to PATH]
                       ferry drop --msg "text" [--to PATH]
  pickup <blob>      [Dual] Verify and decrypt a .ferrydrop blob
                       ferry pickup <blob> [--to PATH] [--pass-file FILE]

Options for 'up':
  (no flags)         THE STACK — all eight lanes on one endpoint (:$PORT):
                       heavy        cloud  GPT-6 Astra (ChatGPT subscription), Sol fallback
                       medium       cloud  GPT-5.6 Terra (ChatGPT subscription), OpenRouter Terra fallback
                       flash        cloud  GPT-5.6 Luna (OpenRouter), Gemini/Terra fallbacks
                       super-flash  cloud  Gemini Flash Latest (OpenRouter), Gemini-only; no model fallback
                       schematron   GPU    HTML→JSON extraction, ON-MACHINE ($LOCAL_MODEL_SCHEMATRON);
                                             temperature 0, no fallback
                       schematron-cloud  cloud  the same job off-box (OpenRouter
                                             schematron-v2-turbo), BY NAME ONLY
                       local-orch   GPU    $LOCAL_MODEL_ORCH
                       local-sub    GPU    $LOCAL_MODEL_SUB
                     The GPU lanes run on internal ports
                     $LOCAL_ORCH_PORT/$LOCAL_SUB_PORT/$LOCAL_SCHEMATRON_PORT; clients only ever
                     address :$PORT and pick a lane by name. The cloud extractor
                     is still reachable, as its own lane 'schematron-cloud' —
                     nothing falls back to it from 'schematron'.
  -a, --all, --stack Same as no flags (explicit form)
  -l, --local, --local-orch
                     Launch ONLY the local orchestrator lane ($LOCAL_MODEL_ORCH)
                       [macOS / Apple Silicon only]
  -s, --sub, --local-sub
                     Launch ONLY the local subagent lane ($LOCAL_MODEL_SUB)
                       [macOS / Apple Silicon only]
  --local-schematron Launch ONLY the local extraction lane ($LOCAL_MODEL_SCHEMATRON),
                       raw on the target port with NO litellm in front — address it
                       by its HuggingFace id, not by the 'schematron' lane name.
                       For the lane NAME, use --schematron (a litellm door) or the
                       full stack.  [macOS / Apple Silicon only]
  -o, --orch         Alias of --local-orch. NOTE: the orchestrator lane is now Qwen;
                       Nemotron moved to --local-sub.
  -c, --cloud        Proxy to default cloud model ($DEFAULT_CLOUD_MODEL)
  -m, --model <id>   Proxy directly to any specific LiteLLM cloud model string
  -r, --route        Serve only the CLOUD lanes (orch + flash) from the litellm config
                        Uses ~/.config/ferry/litellm.yaml (seeded from template on first run)
   --schematron      Serve ONLY the schematron extraction lane, on its OWN door
                        (default :$SCHEMATRON_PORT): a filtered copy of the route
                        config with just the schematron deployment, so a scraper
                        workload runs alongside the main stack without touching :$PORT.
                        Since v1.36.0 it also starts the lane's local MLX backend on
                        :$LOCAL_SCHEMATRON_PORT — or REUSES it untouched if the main
                        stack already has it warm.
   -i, --interactive  Force launch the interactive lane/model selection catalog
  -p, --port <port>  Override listening port [default: $PORT]

Examples:
  ferry up             # The full stack: orch + flash + local-orch + local-sub + schematron on :$PORT
  ferry up --route     # Cloud lanes only (no GPU weights resident)
  ferry up --schematron # The extraction lane alone, on its own door (:$SCHEMATRON_PORT)
  ferry up --local-sub # Just the Nemotron subagent lane, alone on :$PORT
  ferry up --local-schematron  # Just the Schematron-8B extraction lane, alone on :$PORT
  ferry up -i          # Interactive catalog (query Gemini's live model list)
  ferry dash --open    # Open the live route-proxy dashboard in your browser
  ferry status         # Per-lane health, memory, and served lane names
  ferry msg "hello"    # Sends telemetry message back to the host Mac
  cat err.log | ferry log  # Stream errors back to host Mac
EOF
  exit 0
}

# ----------------- COMMANDS -----------------

# _ferry_patch_nemotron_batching — make the `local-sub` lane actually answer.
#
# mlx-vlm's continuous-batching engine (generate/ar.py) passes BOTH `input_ids`
# and `inputs_embeds` on every request, but the `nemotron_h` backbone requires
# exactly one and raises `ValueError: Provide exactly one of inputs or
# inputs_embeds`. Unpatched, EVERY request to the Nemotron lane fails — the lane
# starts, reports healthy, and 500s on first use.
#
# The fix is two lines in LanguageModel.__call__: prefer inputs_embeds when it is
# present. This runs on every `ferry install` because `uv tool install --force`
# rewrites site-packages and discards the previous patch.
#
# Idempotent and fail-soft by design: it no-ops when already patched, no-ops when
# the upstream shape changes (i.e. when the bug is fixed), and never fails the
# install — a missing patch costs one lane, a broken install costs all of them.
_ferry_patch_nemotron_batching() {
  echo ">>> Patching mlx-vlm nemotron_h for continuous batching (local-sub lane)..."
  python3 - <<'PYEOF' || echo "    (patch step skipped; local-sub may 500 on every request)"
import glob, os, sys

CANDIDATES = []
# The uv-managed tool venv is the path `ferry install` creates.
CANDIDATES += glob.glob(os.path.expanduser(
    "~/.local/share/uv/tools/mlx-vlm/lib/python*/site-packages/mlx_vlm/models/nemotron_h/language.py"))
# Fall back to any importable mlx_vlm (pip/conda installs).
try:
    import mlx_vlm  # noqa
    CANDIDATES.append(os.path.join(os.path.dirname(mlx_vlm.__file__),
                                   "models", "nemotron_h", "language.py"))
except Exception:
    pass

path = next((c for c in CANDIDATES if os.path.exists(c)), None)
if not path:
    print("    nemotron_h/language.py not found - nothing to patch.")
    sys.exit(0)

src = open(path).read()
ANCHOR = "        out = self.backbone(inputs, cache=cache, inputs_embeds=inputs_embeds)"
GUARD = "        if inputs_embeds is not None:\n            inputs = None\n"

if GUARD in src:
    print(f"    Already patched: {path}")
    sys.exit(0)
if ANCHOR not in src:
    # Upstream changed this call site - most likely the bug is fixed. Do not
    # guess at a new insertion point; leave the file alone.
    print(f"    Upstream shape changed (anchor absent) - leaving {path} untouched.")
    sys.exit(0)

patched = src.replace(ANCHOR,
    "        # Continuous batching (generate/ar.py) always passes BOTH input_ids and\n"
    "        # inputs_embeds (it runs on embeddings); the backbone requires exactly\n"
    "        # one, so defer to inputs_embeds whenever it is present.\n"
    + GUARD + ANCHOR, 1)
open(path, "w").write(patched)
print(f"    Patched: {path}")
PYEOF
}

cmd_install() {
  echo "================================================================="
  echo "               PROVISIONING LLM-FERRY SYSTEM HOST"
  echo "================================================================="
  
  # Ensure uv is installed
  if ! command -v uv >/dev/null 2>&1; then
    echo ">>> Installing 'uv'..."
    curl -LsSf https://astral.sh/uv/install.sh | sh
    export PATH="$HOME/.local/bin:$PATH"
  else
    echo ">>> 'uv' is already installed."
  fi

  # Install mlx-vlm (macOS / Apple Silicon only — this is the local GPU serving path).
  if (( IS_MAC )); then
    echo ">>> Installing 'mlx-vlm' via uv..."
    uv tool install mlx-vlm --with jinja2 --force
  fi

  # Install litellm for cloud proxying.
  # Pin litellm to the version the stack is verified against. 1.99.0 restamps
  # the response `model` field to the client-requested lane name (the lane-name
  # abstraction clients rely on), uses fastapi's current get_flat_params API
  # (fastapi is deliberately NOT pinned — 1.99.0 runs with 0.141+), and needs
  # two non-default deps the proxy imports at runtime even in a no-DB setup:
  #   prisma            — the auth-error handler imports it (auth_exception_handler
  #                       -> db/exception_handler); without it an authed request
  #                       500s instead of 401ing. Required since v1.22.0.
  #   prometheus_client — litellm's native /metrics; without it, setting
  #                       `callbacks: ["prometheus"]` in litellm.yaml crashes
  #                       startup. Lets the observ Grafana stack chart per-model
  #                       usage, failures, and fallback events.
  # The [proxy] extra is what pulls fastapi/uvicorn in the first place.
  echo ">>> Installing 'litellm' via uv..."
  uv tool install 'litellm[proxy]==1.99.0' --with 'prisma' --with 'prometheus_client' --force

  # Download default local models (macOS only — Linux has no local MLX serving).
  if (( IS_MAC )); then
    echo ">>> Downloading default local models..."
    download_model() {
      local model_id=$1
      if command -v hf >/dev/null 2>&1; then
        echo "    [hf] Downloading $model_id..."
        hf download "$model_id"
      else
        echo "    [huggingface-cli] Downloading $model_id..."
        uv run huggingface-cli download "$model_id"
      fi
    }
    echo ">>> Fetching Model 1: Qwen 3.8-27B nvfp4 (the local-orch lane)"
    download_model "$LOCAL_MODEL"
    echo ">>> Fetching Model 2: Qwen 3.8-27B MTP speculative drafter, 8-bit (local-orch)"
    download_model "$LOCAL_DRAFT"
    echo ">>> Fetching Model 3: NVIDIA Nemotron 3 Nano 30B A3B NVFP4 (the local-sub lane)"
    download_model "$LOCAL_MODEL_SUB"

    # The local-sub lane is unusable without this patch — see the function's
    # comment. It MUST run after `uv tool install mlx-vlm --force` above, which
    # replaces site-packages and therefore discards any previous patch.
    _ferry_patch_nemotron_batching
  else
    echo ">>> Linux detected: skipping mlx-vlm and local model downloads."
    echo "    Local GPU serving is macOS / Apple Silicon only. On Linux, serve via"
    echo "    'ferry up --route' / '--cloud' / '--model <id>' against a cloud endpoint."
    if ! command -v zsh >/dev/null 2>&1; then
      echo ">>> NOTE: 'zsh' is not installed (ferry is a zsh script). Install it with:"
      echo "      sudo apt install zsh"
    fi
    echo ">>> Recommended: 'avahi-daemon' so '.local' mDNS names resolve across the LAN:"
    echo "      sudo apt install avahi-daemon"
    echo "    'iproute2' provides the 'ip' command used for LAN IP detection:"
    echo "      sudo apt install iproute2"
  fi

  # The local-lane opencode guardrails. Clients get these from
  # client-bootstrap.sh; before this the host got them from nowhere at all.
  _ferry_install_opencode_guardrails

  # Same story one layer out: the opencode-cloud / opencode-local shell
  # wrappers were also client-bootstrap-only, so the host had the profile files
  # and no way to select between them.
  _ferry_install_host_wrappers

  # The claude-ferry / claude-ferry-local shell wrappers, same deal: clients
  # got them via client-bootstrap.sh, so the host needed its own copy pointed
  # at the local front door.
  if (( $+functions[_ferry_install_claude_wrappers] )); then
    _ferry_install_claude_wrappers 127.0.0.1 "$PORT"
  fi

  # Link globally to ~/.local/bin/ferry (+ the ferry-dash companion)
  echo ">>> Creating global symlinks in ~/.local/bin (ferry, ferry-dash)..."
  mkdir -p "$HOME/.local/bin"
  ln -sfn "${0:A}" "$HOME/.local/bin/ferry"
  [[ -f "$APP_DIR/ferry-dash" ]] && ln -sfn "$APP_DIR/ferry-dash" "$HOME/.local/bin/ferry-dash"

  echo "================================================================="
  if (( IS_MAC )); then
    echo ">>> SUCCESS! 'ferry' CLI has been linked globally (macOS / Apple Silicon)."
    echo "    Local GPU serving, cloud proxy, route, dash, and LAN share are all available."
  else
    echo ">>> SUCCESS! 'ferry' CLI has been linked globally (Linux)."
    echo "    Available: cloud proxy, route, dash, client wiring, LAN share/transfer."
    echo "    Local GPU serving (--local) is macOS-only; use --route / --cloud / --model."
  fi
  echo "    Ensure ~/.local/bin is in your PATH. Run: ferry --help"
  echo "================================================================="
}

# Dynamic interactive catalog selector querying Gemini API live list
select_model_from_catalog() {
  echo "================================================================="
  echo "               FETCHING LIVE GEMINI MODELS LIST..."
  echo "================================================================="

  # Ensure API key is present
  if [[ -z "${GEMINI_API_KEY:-}" ]]; then
    echo "Error: GEMINI_API_KEY is not set in your environment or ~/.config/ferry/secrets.env."
    echo "Please set GEMINI_API_KEY to access cloud models dynamically."
    echo "Fallback: launching the local-orch GPU lane instead."
    LAUNCH_MODE="local-orch"
    return
  fi

  # Call Gemini REST endpoint to grab the active model list
  local raw_models
  if ! raw_models=$(curl -fsS -m 5 "https://generativelanguage.googleapis.com/v1beta/models?key=${GEMINI_API_KEY}" 2>/dev/null); then
    echo "WARNING: Could not connect to Gemini's server to retrieve live list."
    echo "Fallback: launching the local-orch GPU lane instead."
    LAUNCH_MODE="local-orch"
    return
  fi

  # Run Python script to parse, filter for generateContent, sort newest models first, and display a menu
  local chosen_model
  chosen_model=$(python3 - "$raw_models" <<'PYEOF'
import json, sys, re

# Read API response
try:
    data = json.loads(sys.argv[1])
except Exception:
    print("__ERROR:Failed to parse response JSON__")
    sys.exit(0)

models_list = data.get("models", [])

# Filter for text/generation models
chat_models = []
for m in models_list:
    name = m.get("name", "")
    # Remove prefix "models/"
    short_name = name.split("/")[-1] if "/" in name else name
    
    # Filter out embedding, semantic, translation, or legacy models
    methods = m.get("supportedGenerationMethods", [])
    if "generateContent" in methods and "embedContent" not in name and "text-embedding" not in name:
        # Match version strings for sorting
        # Prioritize 3.7, 2.0, 1.5, in descending order
        version_weight = 0.0
        if "3.7" in short_name:
            version_weight += 300.0
        elif "2.0" in short_name:
            version_weight += 200.0
        elif "1.5" in short_name:
            version_weight += 100.0
            
        # Give higher priority to flash/pro over experimental/older
        if "pro" in short_name:
            version_weight += 10.0
        elif "flash" in short_name:
            version_weight += 5.0
            
        # Push preview/thinking/experimental variants slightly down relative to stable versions
        if "thinking" in short_name or "preview" in short_name or "experimental" in short_name:
            version_weight -= 2.0
            
        chat_models.append((version_weight, short_name, m.get("displayName", short_name)))

# Sort descending (newest versions first)
chat_models.sort(key=lambda x: x[0], reverse=True)

# Generate list of options
options = []
# Option 1 is ALWAYS the full stack: every lane on one endpoint. Options 2-4 are the
# single GPU lanes for when you want ONE model on :8090 and nothing else resident.
options.append(("stack", "FULL STACK - orch + flash (cloud) + local-orch + local-sub + schematron (GPU), one endpoint"))
options.append(("local-orch", "Local GPU Qwen 3.8-27B nvfp4 only (local-orch lane, APC + speculative MTP)"))
options.append(("local-sub", "Local GPU NVIDIA Nemotron 3 Nano 30B A3B NVFP4 only (local-sub lane)"))
options.append(("local-schematron", "Local GPU Schematron-8B 8-bit only (local-schematron lane, HTML->JSON)"))

for _, m_id, m_desc in chat_models:
    options.append((f"gemini/{m_id}", f"[Cloud] {m_desc} (gemini/{m_id})"))

# Prompt the user via stderr to keep stdout clean for capturing the output model selection
sys.stderr.write("=================================================================\n")
sys.stderr.write("             LLM-FERRY ACTIVE MODEL CATALOG (NEWEST FIRST)\n")
sys.stderr.write("=================================================================\n")
for idx, (m_id, label) in enumerate(options, 1):
    sys.stderr.write(f"  {idx}) {label}\n")
sys.stderr.write("=================================================================\n")
sys.stderr.write(f"Select a lane to launch (1-{len(options)}) [Default: 1 = full stack]: ")
sys.stderr.flush()

try:
    # Read response directly from /dev/tty
    with open("/dev/tty", "r") as tty:
        choice_str = tty.readline().strip()
    choice = int(choice_str) if choice_str else 1
except Exception:
    choice = 1

if choice < 1 or choice > len(options):
    choice = 1

selected_id = options[choice - 1][0]
print(selected_id)
PYEOF
)

  if [[ "$chosen_model" == "stack" || "$chosen_model" == "local" \
     || "$chosen_model" == "local-orch" || "$chosen_model" == "local-sub" \
     || "$chosen_model" == "local-schematron" ]]; then
    LAUNCH_MODE="$chosen_model"
  elif [[ "$chosen_model" == "__ERROR:"* ]]; then
    echo "Error parsing live models. Falling back to the full stack."
    LAUNCH_MODE="stack"
  else
    LAUNCH_MODE="cloud"
    CLOUD_MODEL="$chosen_model"
    CLOUD_PROVIDER="$(_ferry_cloud_provider "$CLOUD_MODEL")"
  fi
}

# --- Single-model passthrough: which provider, and which key does it need? ---
# `--cloud` / `--model` run a BARE litellm on one model id, so the provider is
# whatever prefixes that id — there is no route config to look it up in. These
# used to be hardcoded to gemini, which meant `ferry serve --model zai/...`
# still demanded GEMINI_API_KEY and refused to start without it.
_ferry_cloud_provider() {
  print -r -- "${1%%/*}"
}

# Empty result = no preflight check (a provider we don't know the key name for;
# litellm will say so itself rather than us guessing wrong and blocking a start).
_ferry_cloud_key_var() {
  case "$1" in
    gemini)       print -r -- "GEMINI_API_KEY" ;;
    openrouter)   print -r -- "OPENROUTER_API_KEY" ;;
    zai)          print -r -- "GLM_API_KEY" ;;
    fireworks_ai) print -r -- "FIREWORKS_API_KEY" ;;
    anthropic)    print -r -- "ANTHROPIC_API_KEY" ;;
    openai)       print -r -- "OPENAI_API_KEY" ;;
    *)            print -r -- "" ;;
  esac
}

# ---- Stack helpers ---------------------------------------------------------
# `ferry up` (no args) runs the STACK: litellm on $PORT is the ONE door clients
# use, and it fans out to the cloud lanes plus two MLX servers on internal ports.
# These helpers exist so the stack and the single-lane flags launch MLX the same
# way — one launch line, one governor, no drift between them.

# _ferry_free_port <port> — stop whatever is holding <port> so a lane can bind it.
# POLLS for the port to actually clear rather than sleeping a fixed second: a
# server that is shutting down releases its listener before the process exits,
# and a fixed sleep either wastes time or races (see _ferry_reset_log).
_ferry_free_port() {
  local p="$1" waited=0
  if lsof -nP -iTCP:"$p" -sTCP:LISTEN >/dev/null 2>&1; then
    echo ">>> Port $p is already in use. Stopping conflicting server..."
    lsof -ti tcp:"$p" | xargs kill -9 2>/dev/null || true
    while (( waited < 10 )); do
      lsof -nP -iTCP:"$p" -sTCP:LISTEN >/dev/null 2>&1 || return 0
      sleep 1
      waited=$(( waited + 1 ))
    done
    echo ">>> WARNING: port $p still held after ${waited}s — the bind below may fail."
  fi
}

# _ferry_reset_log <path> — hand the next lane a FRESH log inode.
#
# Why unlink instead of truncating with `>`: a server that is shutting down keeps
# writing (uvicorn's "Application shutdown complete", "Finished server process")
# from an fd whose offset is already tens of KB in. If the launch truncates that
# same inode to zero first, those straggler writes land at the OLD offset and
# punch a SPARSE HOLE of NUL bytes, restoring the file's size — while the newly
# launched process, holding its own offset-0 fd, writes underneath the hole and
# never reaches EOF. The log's mtime then ticks on every request while its SIZE
# never moves, and ferry-log-shipper — which attaches at EOF — sits forever past
# everything being written and ships nothing. Observability goes dark with every
# process still healthy. (Hit 2026-08-26 restarting the route proxy: a 91,956-byte
# log pinned at exactly 91,956 bytes for 20 minutes of live traffic.)
#
# Unlinking severs the two writers completely: the straggler keeps its now-nameless
# inode (reclaimed when it closes), and the launch below creates a brand-new one.
# ferry-log-shipper already treats a new inode as a rotation and re-reads from zero.
_ferry_reset_log() {
  local log="$1"
  [[ -n "$log" ]] || return 0
  mkdir -p "${log:h}" 2>/dev/null || true
  rm -f "$log" 2>/dev/null || true
}

# _ferry_stop_litellm [port] — stop the litellm proxy and WAIT for it to exit.
# With a port, only the proxy serving THAT port is reaped, so `ferry up --port 8099`
# never takes down a lane someone else is using on :8090. With no argument (cmd_down)
# every litellm this CLI launches is reaped.
#
# The wait is the point. litellm drops its listener seconds before its last log
# write, so a port check alone reports "free" while the process is still writing —
# and the launch that follows then truncates that log underneath it (see
# _ferry_reset_log). Callers run this BEFORE touching the log path.
# ── The lane-catalogue front ─────────────────────────────────────────────────
# litellm advertises EVERY deployment on /v1/models, fallback hops included, and
# offers no way to hide one (`hidden` is honoured for model_group_alias only —
# see front/ferry_front.py). A client that picks a hop out of that list gets a
# single provider with no failover behind it.
#
# So the config-driven launches serve litellm's own app through a thin ASGI
# wrapper that trims the listing to the lanes marked `model_info: {public: true}`.
# It is NOT a second process and NOT a reverse proxy: same interpreter, same app,
# and every request that is not the model listing is handed to litellm untouched,
# so nothing sits between a client and a streamed token.
#
# The `--model` launch has no config to read markers from and stays on the CLI.

# The interpreter that owns litellm. `litellm` is installed as a uv tool, so its
# venv python is the only one that can import litellm AND uvicorn.
_ferry_front_python() {
  local bin real py
  bin="$(command -v litellm 2>/dev/null)" || return 1
  [[ -n "$bin" ]] || return 1
  real="$(python3 -c 'import os,sys; print(os.path.realpath(sys.argv[1]))' "$bin" 2>/dev/null)" || return 1
  py="${real:h}/python"
  [[ -x "$py" ]] || return 1
  print -r -- "$py"
}

# Start the filtered front. Returns non-zero WITHOUT starting anything if the
# front is unusable, so the caller can fall back to the plain CLI: a visible
# fallback hop is a wart, a dead endpoint is an outage.
#
# Workers: litellm's benchmark guidance is one uvicorn worker per CPU (2 -> 4
# instances halved median latency, P95 630ms -> 150ms), but a LAN host serving
# a handful of agentic clients needs only a small pool — default 4, override
# with FERRY_WORKERS=N (1 for the old single-process shape). ferry_front.py
# sets PROMETHEUS_MULTIPROC_DIR when workers > 1 so /metrics still aggregates
# (litellm 1.97.0 mounts a MultiProcessCollector when the env var is set);
# without it each worker would answer a scrape with only its own counters and
# the litellm_* dashboards would undercount by ~1/N.
FERRY_FRONT_WORKERS="${FERRY_WORKERS:-4}"
_ferry_launch_front() {
  local config="$1" port="$2" log="$3" workers="${4:-$FERRY_FRONT_WORKERS}" py front
  front="$APP_DIR/front/ferry_front.py"
  [[ -f "$front" ]] || return 1
  py="$(_ferry_front_python)" || return 1
  "$py" -c 'import uvicorn, yaml, litellm' >/dev/null 2>&1 || return 1
  # FERRY_EVENTS arms the front door's event tap (front/ferry_front.py
  # tap_enabled()), which is what ferry dash's LIVE TRAFFIC panel reads.
  # Nothing else in the launch path set it, so the panel could show "tap
  # armed" with an empty file forever. It is fail-open and the writes ride a
  # bounded queue off the response path, so on-by-default costs nothing.
  export FERRY_EVENTS="${FERRY_EVENTS:-1}"
  # Exposure control is the master key, not the bind; Tailscale serve fronts this port.
  nohup "$py" "$front" \
    --config "$config" \
    --port "$port" \
    --host 0.0.0.0 \
    --workers "$workers" >> "$log" 2>&1 & disown
  return 0
}

# litellm's ChatGPT provider PREPENDS its own "you are Codex in the Codex CLI"
# prompt to every request's instructions, ahead of the client's system prompt.
# CHATGPT_DEFAULT_INSTRUCTIONS replaces it (read per request).
#
# ferry_front.py resolves and exports this itself — see
# front/ferry_front.py resolve_chatgpt_instructions() for the full rationale and
# precedence. This mirror exists ONLY for the plain `litellm` CLI launches
# below, which never load that module. Deliberately NOT called before
# _ferry_launch_front: the front reports where its prompt came from, and a
# pre-export here would make every launch log say "operator env".
_ferry_export_chatgpt_instructions() {
  # `#` as a repetition operator (the ends-only trims below) is an extended_glob
  # feature and SILENTLY no-ops without it; local_options restores on return.
  setopt local_options extended_glob
  local f text sentinel override operator
  # Trimmed at the ENDS only, exactly like the Python resolver's str.strip():
  # interior whitespace belongs to a path, and "o f f" is not the opt-out.
  override="${${${FERRY_CHATGPT_INSTRUCTIONS:-}##[[:space:]]#}%%[[:space:]]#}"
  sentinel="${(L)override}"
  [[ "$sentinel" == "off" ]] && return 0
  # litellm reads `getenv(...) or DEFAULT`, so an all-whitespace value is NOT
  # the operator having decided anything — same .strip() test as the resolver,
  # or the model would get three spaces as its entire preamble.
  operator="${CHATGPT_DEFAULT_INSTRUCTIONS:-}"
  [[ -n "${operator//[[:space:]]/}" ]] && return 0
  local -a candidates=()
  # A leading ~ in a VALUE is not expanded by the shell (and `${~var}` only
  # arms globbing), so do the one case expanduser() does on the Python side.
  [[ "$override" == "~" ]] && override="$HOME"
  [[ "$override" == "~/"* ]] && override="$HOME/${override#\~/}"
  [[ -n "$override" ]] && candidates+=("$override")
  candidates+=("$HOME/.config/ferry/chatgpt-instructions.txt")
  candidates+=("$APP_DIR/front/chatgpt-instructions.txt")
  for f in "${candidates[@]}"; do
    # -r alone is TRUE for a directory, and `$(<dir)` then spills
    # "error when reading ...: is a directory" onto ferry's stderr; the Python
    # side swallows the same case (IsADirectoryError is an OSError).
    [[ -f "$f" && -r "$f" ]] || continue
    text="$(<"$f")"
    # A blank file must never become an empty override: litellm treats "" as
    # unset and falls straight back to the Codex prompt.
    [[ -n "${text//[[:space:]]/}" ]] || continue
    export CHATGPT_DEFAULT_INSTRUCTIONS="$text"
    return 0
  done
  return 0
}

_ferry_stop_litellm() {
  local port="${1:-}" waited=0 pat label
  if [[ -n "$port" ]]; then
    pat="(litellm|ferry_front\.py) .*--port ${port}( |$)"
    label="the litellm proxy on :$port"
  else
    pat="(litellm --(config|model) |ferry_front\.py --config )"
    label="every litellm proxy"
  fi

  pgrep -f "$pat" >/dev/null 2>&1 || return 0
  echo ">>> Stopping $label and waiting for it to exit..."
  pkill -f "$pat" 2>/dev/null || true
  while (( waited < 15 )); do
    pgrep -f "$pat" >/dev/null 2>&1 || return 0
    sleep 1
    waited=$(( waited + 1 ))
  done
  echo ">>> WARNING: $label still running after ${waited}s — forcing."
  pkill -9 -f "$pat" 2>/dev/null || true
  sleep 1
}

# _ferry_launch_mlx <label> <model> <draft|""> <port> <log> <kv_bits> <max_kv> <max_seqs> <apc_blocks>
# Launch ONE mlx_vlm.server lane in the background under the KV/memory governor.
# Every governor flag is conditional, so passing "" for one drops it from the
# launch line rather than sending an empty value.
_ferry_launch_mlx() {
  local label="$1" model="$2" draft="$3" port="$4" log="$5"
  local kv_bits="$6" max_kv="$7" max_seqs="$8" apc_blocks="$9"

  # APC is env-driven, not a flag. Exporting immediately before each nohup means
  # each lane inherits ITS OWN pool size (the child snapshots the env at fork).
  export APC_ENABLED=1
  [[ -n "$apc_blocks" ]] && export APC_NUM_BLOCKS="$apc_blocks"

  # Internal backends behind the litellm front door: the front door reaches
  # them on loopback (api_base: http://127.0.0.1:...), so loopback bind is all
  # they need. Binding 0.0.0.0 exposed unauthenticated inference to the LAN.
  local mlargs=(
    --model "$model"
    --host 127.0.0.1
    --port "$port"
  )
  [[ -n "$draft" ]]    && mlargs+=(--draft-model "$draft")
  [[ -n "$kv_bits" ]]  && mlargs+=(--kv-bits "$kv_bits")
  [[ -n "$max_kv" ]]   && mlargs+=(--max-kv-size "$max_kv")
  [[ -n "$max_seqs" ]] && mlargs+=(--max-num-seqs "$max_seqs")

  echo ">>> [$label] $model"
  echo "        port   :$port   draft=${draft:-none}"
  echo "        KV gov kv-bits=${kv_bits:-off} max-kv=${max_kv:-off} seqs=${max_seqs:-off} apc-blocks=${apc_blocks:-off}"
  echo "        log    $log"
  _ferry_reset_log "$log"
  nohup mlx_vlm.server "${mlargs[@]}" >> "$log" 2>&1 & disown
}

# _ferry_wait_http <url> <label> [timeout-seconds] [mode]
# Poll until <url> answers, so `ferry up` reports REAL readiness instead of
# "launched". mlx_vlm preloads inside its FastAPI lifespan, so the port does not
# accept connections until the weights are resident — a 200 here means the lane
# is genuinely warm, never "listening but still loading". A cold 15-18GB lane
# legitimately takes tens of seconds. Returns 1 on timeout; callers tolerate
# that (the lane keeps loading in the background).
#
# mode (default "content"):
#   content   — only a 2xx counts. The MLX lanes' /v1/models probes stay in
#               this mode: their 200 IS the weights-resident signal, and those
#               backends are loopback-only and never authenticated.
#   readiness — process-up is the question, so a 401 also counts as READY
#               ("up, but gated by auth"): v1.22.0 can put the front door
#               behind general_settings.master_key, and a readiness wait must
#               not then report a healthy proxy as never-ready. The front
#               door's own wait uses the public /health/liveliness route and
#               readiness mode, so it holds either way.
# When LITELLM_MASTER_KEY is set (host shell or secrets.env via ferry-core) it
# is presented on every probe; unset means no header at all, so keyless LAN
# installs probe exactly as they did before.
_ferry_wait_http() {
  local url="$1" label="$2" timeout="${3:-600}" mode="${4:-content}" waited=0 code
  local -a hdr=()
  [[ -n "${LITELLM_MASTER_KEY:-}" ]] && hdr=(-H "Authorization: Bearer $LITELLM_MASTER_KEY")
  while (( waited < timeout )); do
    code="$(curl -sS -m 3 -o /dev/null -w '%{http_code}' "${hdr[@]}" "$url" 2>/dev/null || true)"
    if [[ "$code" == 2* ]]; then
      echo ">>> [$label] \033[1;32mREADY\033[0m (${waited}s)"
      return 0
    fi
    if [[ "$mode" == "readiness" && "$code" == "401" ]]; then
      echo ">>> [$label] \033[1;32mREADY\033[0m (${waited}s) — up; answering 401 (master_key auth on)"
      return 0
    fi
    sleep 3
    waited=$(( waited + 3 ))
    (( waited % 15 == 0 )) && echo "    [$label] loading... ${waited}s"
  done
  echo ">>> [$label] \033[1;33mNOT READY\033[0m after ${timeout}s - still loading, or check its log."
  return 1
}

# _ferry_require_route_config — set FERRY_ROUTE_CONFIG, seeding it from the
# shipped template on first run (then exiting so keys can be filled in first).
# NOT a command-substitution helper on purpose: `exit` has to end `ferry up`,
# which it cannot do from inside a $(...) subshell.
_ferry_require_route_config() {
  FERRY_ROUTE_CONFIG="$DEFAULT_ROUTE_CONFIG"
  if [[ ! -f "$FERRY_ROUTE_CONFIG" ]]; then
    mkdir -p "$(dirname "$FERRY_ROUTE_CONFIG")"
    if [[ -f "$ROUTE_TEMPLATE" ]]; then
      cp "$ROUTE_TEMPLATE" "$FERRY_ROUTE_CONFIG"
      echo ">>> Seeded route config from template:"
      echo "    $FERRY_ROUTE_CONFIG"
      echo "    Edit it (set your model ids), export the keys it references"
      echo "    (e.g. OPENROUTER_API_KEY), then re-run."
      exit 0
    fi
    echo "Error: No route config at $FERRY_ROUTE_CONFIG and no template at $ROUTE_TEMPLATE."
    exit 1
  fi
}

# _ferry_warn_missing_keys — warn (never hard-fail) about unset keys the route
# config IN USE ($FERRY_ROUTE_CONFIG) actually references. A lane whose key is
# missing 401s; the others are fine. DERIVED from the config, not hardcoded:
# every LIVE `api_key: os.environ/VAR` line is collected via one grep pass,
# paired with the `model_name:` of the deployment block it falls under, so a
# lane rename or a new/retired lane never leaves this warning naming a stale
# var. Comment lines are stripped FIRST (`^\s*#`): operators keep commented-out
# example blocks (e.g. a `# - model_name: heavy-fallback` template leftover)
# around, and those reference vars no live lane needs and would mislabel the
# next real lane if left in the grep. If the config can't be read, print
# nothing (there's nothing to derive from).
_ferry_warn_missing_keys() {
  [[ -n "${FERRY_ROUTE_CONFIG:-}" && -f "$FERRY_ROUTE_CONFIG" ]] || return 0

  local -A lanes_for_var
  local lane="" line var
  while IFS= read -r line; do
    if [[ "$line" == *model_name:* ]]; then
      lane="${line#*model_name: }"
      lane="${lane%%[[:space:]]*}"
    elif [[ "$line" == *api_key:*os.environ/* ]]; then
      var="${line#*os.environ/}"
      var="${var%%[[:space:]]*}"
      if [[ -n "$lane" ]]; then
        lanes_for_var[$var]="${lanes_for_var[$var]:+${lanes_for_var[$var]}, }$lane"
      fi
    fi
  done < <(grep -vE '^[[:space:]]*#' "$FERRY_ROUTE_CONFIG" \
           | grep -E 'model_name:|api_key:.*os\.environ/')

  local missing=() var2
  for var2 in "${(ok)lanes_for_var[@]}"; do
    if [[ -z "${(P)var2:-}" ]]; then
      missing+=("$var2 (lanes: ${lanes_for_var[$var2]})")
    fi
  done
  if (( ${#missing[@]} > 0 )); then
    echo ">>> WARNING: these env vars are unset; lanes that need them will 401:"
    for m in "${missing[@]}"; do echo "      - $m"; done
    echo "    Export them in your shell or ~/.config/ferry/secrets.env."
  fi
}

cmd_up() {
  if (( CLIENT_MODE )); then
    echo "Error: Command 'ferry up' is only available on the LLM-Ferry Host Mac."
    exit 1
  fi

  local LAUNCH_MODE="stack" # Default if arguments parsed override it
  local CLOUD_PROVIDER=""
  local CLOUD_MODEL=""
  local target_port="$PORT"
  local port_given=0
  local skip_catalog=0

  # No arguments = the FULL STACK (all four lanes on one endpoint). The
  # interactive catalog, which used to be the no-arg default, now lives behind -i.
  if [[ $# -eq 0 ]]; then
    LAUNCH_MODE="stack"
    skip_catalog=1
  fi
  
  if (( ! skip_catalog )); then
    while [[ $# -gt 0 ]]; do
      case "$1" in
        -a|--all|--stack)
          LAUNCH_MODE="stack"
          skip_catalog=1
          shift
          ;;
        -l|--local|--local-orch)
          # The local ORCHESTRATOR lane alone on the target port.
          LAUNCH_MODE="local-orch"
          skip_catalog=1
          shift
          ;;
        -s|--sub|--local-sub)
          # The local SUBAGENT lane alone on the target port.
          LAUNCH_MODE="local-sub"
          skip_catalog=1
          shift
          ;;
        --local-schematron)
          # The local EXTRACTION lane alone on the target port: raw mlx_vlm,
          # no litellm in front. Distinct from `--schematron`, which serves the
          # litellm DOOR (and now launches this lane behind it).
          LAUNCH_MODE="local-schematron"
          skip_catalog=1
          shift
          ;;
        -o|--orch)
          # `--orch` predates the lane split, when the local orchestrator WAS
          # Nemotron. The orchestrator lane is now Qwen; Nemotron is the subagent
          # lane. Point --orch at whatever "local orchestrator" currently means
          # and say so, so a muscle-memory invocation is not silently redefined.
          echo ">>> Note: the local orchestrator lane is now $LOCAL_MODEL_ORCH."
          echo "    Nemotron moved to the subagent lane - use 'ferry up --local-sub' for it."
          LAUNCH_MODE="local-orch"
          skip_catalog=1
          shift
          ;;
        -i|--interactive)
          select_model_from_catalog
          skip_catalog=1
          shift
          ;;
        -c|--cloud)
          LAUNCH_MODE="cloud"
          CLOUD_MODEL="$DEFAULT_CLOUD_MODEL"
          CLOUD_PROVIDER="$(_ferry_cloud_provider "$CLOUD_MODEL")"
          skip_catalog=1
          shift
          ;;
        -r|--route)
          LAUNCH_MODE="route"
          skip_catalog=1
          shift
          ;;
        --schematron)
          # The extraction lane ALONE on its OWN door (default :$SCHEMATRON_PORT):
          # a scraper workload runs alongside — or instead of — the main stack,
          # and :$PORT is never touched. The default port applies only when the
          # operator has not already passed -p, so both flag orders behave.
          LAUNCH_MODE="schematron"
          (( port_given )) || target_port="$SCHEMATRON_PORT"
          skip_catalog=1
          shift
          ;;
        -m|--model)
          LAUNCH_MODE="cloud"
          CLOUD_MODEL="$2"
          skip_catalog=1
          CLOUD_PROVIDER="$(_ferry_cloud_provider "$CLOUD_MODEL")"
          shift 2
          ;;
        -p|--port)
          target_port="$2"
          port_given=1
          shift 2
          ;;
        *)
          echo "Unknown option: $1"
          usage
          ;;
      esac
    done
  fi

  # Stop the previous proxy and any conflicting server on the target port, and WAIT
  # for both to be gone. Order matters: reap litellm by name FIRST (it releases the
  # listener before it finishes writing its log), then clear whatever else holds the
  # port. Doing only the port check lets a still-shutting-down litellm survive into
  # the launch below and corrupt the fresh log — see _ferry_reset_log.
  _ferry_stop_litellm "$target_port"
  if [[ "$LAUNCH_MODE" == "schematron" ]]; then
    # A COMPANION door: its port is nobody's by default, so anything still
    # holding it after the by-name reap above is NOT ferry's — refuse rather
    # than kill -9 it the way the main-door modes do (_ferry_free_port).
    # Silently reaping an unknown listener on a port the operator never
    # promised ferry is a different risk class than recycling :$PORT.
    if lsof -nP -iTCP:"$target_port" -sTCP:LISTEN >/dev/null 2>&1; then
      echo "Error: port $target_port is held by a process that is not a ferry proxy."
      echo "       'ferry up --schematron' runs beside the main stack and refuses to"
      echo "       kill unknown listeners. Free the port, or pick another:"
      echo "         ferry up --schematron -p <port>"
      exit 1
    fi
  else
    _ferry_free_port "$target_port"
  fi

  # Port-accurate cloud/route log name (the global CLOUD_LOG is pinned to the default port).
  local cloud_log="$LOG_DIR/cloud-proxy-$target_port.log"

  # ── Shared guard for every lane that needs the GPU ───────────────────────
  # Local GPU serving is Apple MLX — macOS / Apple Silicon only. In STACK mode a
  # non-Mac degrades to the cloud lanes rather than failing outright, because the
  # cloud half of the stack is perfectly servable on Linux.
  # v1.36.0 adds "schematron" to this guard: that door's backend is now a local
  # MLX lane, not a cloud model, so it needs the same Apple-Silicon prerequisite
  # as the other GPU modes. It does NOT get the stack's degrade-to-cloud path —
  # a door whose one lane is local has nothing left to serve without the GPU.
  if [[ "$LAUNCH_MODE" == "stack" || "$LAUNCH_MODE" == local-* || "$LAUNCH_MODE" == "schematron" ]]; then
    if (( ! IS_MAC )); then
      if [[ "$LAUNCH_MODE" == "stack" ]]; then
        echo ">>> Local GPU lanes need Apple MLX (macOS / Apple Silicon only)."
        echo "    Serving the CLOUD lanes only (orch + flash) — same endpoint, two lanes."
        LAUNCH_MODE="route"
      else
        echo "Error: local GPU serving uses Apple MLX (macOS / Apple Silicon only)."
        echo "       On Linux, serve a cloud / OpenAI-compatible endpoint instead:"
        echo "         ferry up --route        # multiple models from litellm.yaml"
        echo "         ferry up --cloud        # default Gemini model"
        echo "         ferry up --model <id>   # any LiteLLM model string"
        exit 1
      fi
    elif ! command -v mlx_vlm.server >/dev/null 2>&1; then
      echo "Error: 'mlx_vlm.server' is missing. Run: ferry install"
      exit 1
    fi
  fi

  if [[ "$LAUNCH_MODE" == "stack" ]]; then
    # ── THE STACK: one door, eight lanes ────────────────────────────────────
    #   litellm on $target_port  ->  heavy        (cloud: GPT-6 Astra, ChatGPT subscription, Sol fallback)
    #                            ->  medium       (cloud: GPT-5.6 Terra, ChatGPT subscription, OpenRouter Terra fallback)
    #                            ->  flash        (cloud: GPT-5.6 Luna via OpenRouter, Gemini/Terra fallbacks)
    #                            ->  super-flash  (cloud: Gemini Flash Latest via OpenRouter, Gemini-only; no model fallback)
    #                            ->  schematron   (MLX on :$LOCAL_SCHEMATRON_PORT, HTML→JSON extraction; no fallback)
    #                            ->  schematron-cloud (cloud: OpenRouter schematron-v2-turbo; a SEPARATE lane, never a fallback)
    #                            ->  local-orch   (MLX on :$LOCAL_ORCH_PORT)
    #                            ->  local-sub    (MLX on :$LOCAL_SUB_PORT)
    # The three MLX ports are INTERNAL plumbing — clients only ever talk to
    # $target_port, and the lane names there are the contract they bind to.
    if ! command -v litellm >/dev/null 2>&1; then
      echo "Error: 'litellm' is missing. Run: ferry install"
      exit 1
    fi
    _ferry_require_route_config
    _ferry_warn_missing_keys

    echo "================================================================="
    echo "   FERRY STACK — eight lanes, one endpoint"
    echo "================================================================="
    echo "   heavy        cloud   GPT-6 Astra (ChatGPT subscription), Sol fallback"
    echo "   medium       cloud   GPT-5.6 Terra (ChatGPT subscription), OpenRouter Terra fallback"
    echo "   flash        cloud   GPT-5.6 Luna (OpenRouter), Gemini/Terra fallbacks"
    echo "   super-flash  cloud   Gemini Flash Latest (OpenRouter), Gemini-only; no model fallback"
    echo "   schematron   GPU     $LOCAL_MODEL_SCHEMATRON (HTML→JSON); no fallback"
    echo "   schematron-cloud  cloud  OpenRouter schematron-v2-turbo — BY NAME ONLY, never a fallback"
    echo "   local-orch   GPU     $LOCAL_MODEL_ORCH"
    echo "   local-sub    GPU     $LOCAL_MODEL_SUB"
    echo "================================================================="

    _ferry_free_port "$LOCAL_ORCH_PORT"
    _ferry_free_port "$LOCAL_SUB_PORT"
    _ferry_free_port "$LOCAL_SCHEMATRON_PORT"

    # All three MLX lanes start first and load CONCURRENTLY: the loads are
    # dominated by streaming ~42GB out of the HF cache, so overlapping them is
    # markedly faster than serialising, and none blocks another's warm-up.
    _ferry_launch_mlx "local-orch" "$LOCAL_MODEL_ORCH" "$LOCAL_DRAFT_ORCH" \
      "$LOCAL_ORCH_PORT" "$LOCAL_ORCH_LOG" \
      "$LOCAL_ORCH_KV_BITS" "$LOCAL_ORCH_MAX_KV" "$LOCAL_ORCH_MAX_SEQS" "$LOCAL_ORCH_APC_BLOCKS"
    _ferry_launch_mlx "local-sub" "$LOCAL_MODEL_SUB" "$LOCAL_DRAFT_SUB" \
      "$LOCAL_SUB_PORT" "$LOCAL_SUB_LOG" \
      "$LOCAL_SUB_KV_BITS" "$LOCAL_SUB_MAX_KV" "$LOCAL_SUB_MAX_SEQS" "$LOCAL_SUB_APC_BLOCKS"
    _ferry_launch_mlx "local-schematron" "$LOCAL_MODEL_SCHEMATRON" "$LOCAL_DRAFT_SCHEMATRON" \
      "$LOCAL_SCHEMATRON_PORT" "$LOCAL_SCHEMATRON_LOG" \
      "$LOCAL_SCHEMATRON_KV_BITS" "$LOCAL_SCHEMATRON_MAX_KV" "$LOCAL_SCHEMATRON_MAX_SEQS" "$LOCAL_SCHEMATRON_APC_BLOCKS"

    # litellm does NOT probe its backends at boot, so the front door can come up
    # in parallel with the GPU lanes. A call that arrives before a lane is warm
    # fails on THAT lane only — the cloud lanes are servable immediately.
    echo ">>> [front] litellm --config $FERRY_ROUTE_CONFIG"
    echo "        port   :$target_port"
    echo "        workers:$FERRY_FRONT_WORKERS"
    echo "        log    $cloud_log"
    _ferry_reset_log "$cloud_log"
    if ! _ferry_launch_front "$FERRY_ROUTE_CONFIG" "$target_port" "$cloud_log"; then
      echo "        note: catalogue filter unavailable — serving litellm directly"
      echo "              (/v1/models will advertise the fallback hops too)"
      _ferry_export_chatgpt_instructions
      nohup litellm \
        --config "$FERRY_ROUTE_CONFIG" \
        --port "$target_port" \
        --num_workers "$FERRY_FRONT_WORKERS" \
        --host 0.0.0.0 >> "$cloud_log" 2>&1 & disown
    fi

    echo ">>> Waiting for lanes (MLX loads ~33GB of weights — that is the slow part)..."
    # Front door: pure readiness, so probe the public liveliness route (it stays
    # public under master_key) and let readiness mode accept a 401 as "up" for
    # any litellm that gates it anyway. The MLX lanes keep their /v1/models
    # probes: no litellm in front, no auth ever, and the 200 is the
    # weights-resident signal the liveliness route cannot give.
    _ferry_wait_http "http://127.0.0.1:$target_port/health/liveliness" "front"      120 readiness || true
    _ferry_wait_http "http://127.0.0.1:$LOCAL_ORCH_PORT/v1/models"   "local-orch" 900 || true
    _ferry_wait_http "http://127.0.0.1:$LOCAL_SUB_PORT/v1/models"    "local-sub"  900 || true
    _ferry_wait_http "http://127.0.0.1:$LOCAL_SCHEMATRON_PORT/v1/models" "local-schematron" 900 || true

    echo "================================================================="
    echo ">>> Stack up. Lanes served on http://$MDNS_NAME:$target_port/v1 :"
    # Catalogue content, so it needs the bearer when the front door is keyed.
    local -a banner_auth=()
    [[ -n "${LITELLM_MASTER_KEY:-}" ]] && banner_auth=(-H "Authorization: Bearer $LITELLM_MASTER_KEY")
    curl -fsS -m 5 "${banner_auth[@]}" "http://127.0.0.1:$target_port/v1/models" 2>/dev/null \
      | python3 -c "import json,sys; [print('       ' + m['id']) for m in json.load(sys.stdin).get('data', [])]" 2>/dev/null \
      || echo "       (front door not answering yet — check $cloud_log)"
    echo "-----------------------------------------------------------------"
    echo "    Onboard clients:  ferry share"
    echo "    Live dashboard:   ferry dash"
    echo "    Stop everything:  ferry down"
    echo "================================================================="

  elif [[ "$LAUNCH_MODE" == "local-orch" || "$LAUNCH_MODE" == "local" ]]; then
    # ONE lane on the target port: the local orchestrator model, no litellm in
    # front. Clients address it by its HuggingFace id, not by a lane name.
    echo ">>> Launching the local-orch lane alone (no route proxy)."
    _ferry_launch_mlx "local-orch" "$LOCAL_MODEL_ORCH" "$LOCAL_DRAFT_ORCH" \
      "$target_port" "$LOCAL_LOG" \
      "$LOCAL_ORCH_KV_BITS" "$LOCAL_ORCH_MAX_KV" "$LOCAL_ORCH_MAX_SEQS" "$LOCAL_ORCH_APC_BLOCKS"
    _ferry_wait_http "http://127.0.0.1:$target_port/v1/models" "local-orch" 900 || true

  elif [[ "$LAUNCH_MODE" == "local-sub" ]]; then
    # ONE lane on the target port: the local subagent model (nemotron_h hybrid
    # MoE). Raise LOCAL_SUB_MAX_SEQS to admit more concurrent agents.
    echo ">>> Launching the local-sub lane alone (no route proxy)."
    echo "    (subagent fan-out: raise LOCAL_SUB_MAX_SEQS to admit more concurrent agents)"
    _ferry_launch_mlx "local-sub" "$LOCAL_MODEL_SUB" "$LOCAL_DRAFT_SUB" \
      "$target_port" "$LOCAL_LOG" \
      "$LOCAL_SUB_KV_BITS" "$LOCAL_SUB_MAX_KV" "$LOCAL_SUB_MAX_SEQS" "$LOCAL_SUB_APC_BLOCKS"
    _ferry_wait_http "http://127.0.0.1:$target_port/v1/models" "local-sub" 900 || true

  elif [[ "$LAUNCH_MODE" == "local-schematron" ]]; then
    # ONE lane on the target port: the local HTML→JSON extraction model, raw,
    # with no litellm in front. Clients address it by its HuggingFace id
    # ($LOCAL_MODEL_SCHEMATRON), not by the `schematron` lane name — that name
    # only exists on a litellm door (`ferry up` or `ferry up --schematron`).
    echo ">>> Launching the local-schematron lane alone (no route proxy)."
    echo "    (the JSON schema goes INSIDE the user message — see the Schematron"
    echo "     prompt contract in lib/ferry-core.zsh; ferry never rewrites prompts)"
    _ferry_launch_mlx "local-schematron" "$LOCAL_MODEL_SCHEMATRON" "$LOCAL_DRAFT_SCHEMATRON" \
      "$target_port" "$LOCAL_LOG" \
      "$LOCAL_SCHEMATRON_KV_BITS" "$LOCAL_SCHEMATRON_MAX_KV" "$LOCAL_SCHEMATRON_MAX_SEQS" "$LOCAL_SCHEMATRON_APC_BLOCKS"
    _ferry_wait_http "http://127.0.0.1:$target_port/v1/models" "local-schematron" 900 || true

  elif [[ "$LAUNCH_MODE" == "cloud" ]]; then
    # Start cloud API proxy
    if ! command -v litellm >/dev/null 2>&1; then
      echo "Error: 'litellm' is missing. Run: ferry install"
      exit 1
    fi

    # Verify the key this MODEL needs is present — not, as it used to be,
    # whatever key Gemini needed regardless of which model was asked for.
    local cloud_key_var
    cloud_key_var="$(_ferry_cloud_key_var "$CLOUD_PROVIDER")"
    if [[ -n "$cloud_key_var" && -z "${(P)cloud_key_var:-}" ]]; then
      echo "Error: $cloud_key_var is not set in your environment or ~/.config/ferry/secrets.env."
      echo "       ($CLOUD_MODEL is a '$CLOUD_PROVIDER' model.)"
      exit 1
    fi

    echo ">>> Proxying to Cloud Model: \033[1;32m$CLOUD_MODEL\033[0m"
    echo "    Port: $target_port"
    
    # Launch LiteLLM proxy
    _ferry_reset_log "$cloud_log"
    _ferry_export_chatgpt_instructions
    nohup litellm \
      --model "$CLOUD_MODEL" \
      --port "$target_port" \
      --host 0.0.0.0 >> "$cloud_log" 2>&1 & disown

    echo ">>> Cloud proxy running in background. Log: $cloud_log"
  elif [[ "$LAUNCH_MODE" == "route" ]]; then
    # The CLOUD half of the stack: the litellm route config without the local GPU
    # lanes. Same config file, same lane names — `local-orch`/`local-sub` are still
    # listed by /v1/models but have no backend running, so calls to them fail.
    if ! command -v litellm >/dev/null 2>&1; then
      echo "Error: 'litellm' is missing. Run: ferry install"
      exit 1
    fi

    _ferry_require_route_config
    _ferry_warn_missing_keys

    echo ">>> Serving the CLOUD lanes via litellm route config:"
    echo "    Config: $FERRY_ROUTE_CONFIG"
    echo "    Port:   $target_port"

    _ferry_reset_log "$cloud_log"
    if ! _ferry_launch_front "$FERRY_ROUTE_CONFIG" "$target_port" "$cloud_log"; then
      echo "    note: catalogue filter unavailable — serving litellm directly"
      echo "          (/v1/models will advertise the fallback hops too)"
      _ferry_export_chatgpt_instructions
      nohup litellm \
        --config "$FERRY_ROUTE_CONFIG" \
        --port "$target_port" \
        --num_workers "$FERRY_FRONT_WORKERS" \
        --host 0.0.0.0 >> "$cloud_log" 2>&1 & disown
    fi

    echo ">>> Route proxy running in background. Log: $cloud_log"
    # The printed command is meant to be pasted, so it references the variable
    # (resolved by the pasting shell) rather than echoing the key itself.
    if [[ -n "${LITELLM_MASTER_KEY:-}" ]]; then
      echo "    Served lanes:  curl -s -H \"Authorization: Bearer \$LITELLM_MASTER_KEY\" http://127.0.0.1:$target_port/v1/models"
    else
      echo "    Served lanes:  curl -s http://127.0.0.1:$target_port/v1/models"
    fi
  elif [[ "$LAUNCH_MODE" == "schematron" ]]; then
    # ONE lane on its OWN door: a filtered copy of the route config carrying
    # ONLY the `schematron` deployment, so a scraper workload gets a dedicated
    # endpoint that runs alongside (or instead of) the main stack. This door
    # never touches :$PORT, and :$PORT never touches it back: a different
    # config file, a different log, and the port refusal above instead of a
    # silent reap.
    if ! command -v litellm >/dev/null 2>&1; then
      echo "Error: 'litellm' is missing. Run: ferry install"
      exit 1
    fi
    _ferry_require_route_config
    _ferry_warn_missing_keys

    # Filter with a real YAML parser, not text slicing: operators keep
    # commented-out example blocks in litellm.yaml, and a hand-rolled block
    # extractor trips on exactly those. PyYAML is not stdlib, so run the
    # extractor on the first interpreter that has it — the host python3 when
    # it does, else the litellm tool venv's own python (guaranteed there: the
    # catalogue front itself imports yaml from that interpreter).
    # Deterministic destination, regenerated on every launch, written
    # atomically so a litellm still reading the old copy never sees a
    # half-written file.
    local filt_py=""
    if python3 -c "import yaml" >/dev/null 2>&1; then
      filt_py=python3
    else
      local cand; cand="$(_ferry_front_python 2>/dev/null)" || cand=""
      if [[ -n "$cand" ]] && "$cand" -c "import yaml" >/dev/null 2>&1; then
        filt_py="$cand"
      fi
    fi
    if [[ -z "$filt_py" ]]; then
      echo "Error: no python with PyYAML to filter the route config."
      echo "       Run: ferry install"
      exit 1
    fi
    local schematron_config="$HOME/.config/ferry/litellm-schematron.yaml"
    local upstream
    upstream="$("$filt_py" - "$FERRY_ROUTE_CONFIG" "$schematron_config" schematron <<'SCHEMFILTER_EOF'
import os
import sys

import yaml

src, dst, lane = sys.argv[1], sys.argv[2], sys.argv[3]
with open(src) as fh:
    cfg = yaml.safe_load(fh) or {}

# Keep ONLY this lane's deployment(s): fallback hops of other lanes would
# 401-or-500 without their primaries, and this door's catalogue is one name.
deployments = [m for m in (cfg.get("model_list") or [])
               if isinstance(m, dict) and m.get("model_name") == lane]
if not deployments:
    sys.exit("Error: no '%s' deployment in %s — add one (see litellm-route-example.yaml)."
             % (lane, src))

out = {"model_list": deployments}

# master_key rides in general_settings; dropped and this door would be the one
# unauthenticated listener on 0.0.0.0. Kept verbatim, like everything below:
# trim to what one lane needs, but never rewrite values.
if isinstance(cfg.get("general_settings"), dict):
    out["general_settings"] = cfg["general_settings"]

# drop_params / num_retries / request_timeout / callbacks travel as a unit —
# the prometheus callback mounts /metrics on THIS app (no second port), so it
# cannot collide with the main door's own /metrics.
if isinstance(cfg.get("litellm_settings"), dict):
    out["litellm_settings"] = cfg["litellm_settings"]

router_settings = dict(cfg.get("router_settings") or {})
fallbacks = router_settings.get("fallbacks")
if isinstance(fallbacks, list):
    # Other lanes' entries name model groups this file no longer carries.
    router_settings["fallbacks"] = [e for e in fallbacks
                                    if isinstance(e, dict) and lane in e]
out["router_settings"] = router_settings

if os.path.dirname(dst):
    os.makedirs(os.path.dirname(dst), exist_ok=True)
tmp = dst + ".tmp"
with open(tmp, "w") as fh:
    yaml.safe_dump(out, fh, sort_keys=False, default_flow_style=False)
os.replace(tmp, dst)

# The upstream model string, for the launch banner.
print(deployments[0].get("litellm_params", {}).get("model", lane))
SCHEMFILTER_EOF
)" || exit 1

    local schematron_log="$LOG_DIR/schematron-$target_port.log"
    echo ">>> Serving ONLY the schematron lane (filtered from $FERRY_ROUTE_CONFIG):"
    echo "    Config: $schematron_config"
    echo "    Port:   $target_port"

    # v1.36.0: the `schematron` deployment is now a LOCAL MLX backend, so this
    # door has a backend to start — it used to front a cloud model and launch
    # nothing. REUSE, DO NOT REAP: the main stack launches the very same lane on
    # :$LOCAL_SCHEMATRON_PORT, so if that port already answers /v1/models this
    # door simply fronts the warm process. Killing it would take extraction out
    # from under a running stack. Only a cold port gets a fresh launch.
    local schem_lane_warm=0
    if [[ "$(curl -sS -m 3 -o /dev/null -w '%{http_code}' \
             "http://127.0.0.1:$LOCAL_SCHEMATRON_PORT/v1/models" 2>/dev/null || true)" == "200" ]]; then
      schem_lane_warm=1
      echo ">>> [local-schematron] already warm on :$LOCAL_SCHEMATRON_PORT — reusing it (not reaped)."
    else
      _ferry_launch_mlx "local-schematron" "$LOCAL_MODEL_SCHEMATRON" "$LOCAL_DRAFT_SCHEMATRON" \
        "$LOCAL_SCHEMATRON_PORT" "$LOCAL_SCHEMATRON_LOG" \
        "$LOCAL_SCHEMATRON_KV_BITS" "$LOCAL_SCHEMATRON_MAX_KV" "$LOCAL_SCHEMATRON_MAX_SEQS" "$LOCAL_SCHEMATRON_APC_BLOCKS"
    fi

    _ferry_reset_log "$schematron_log"
    # _ferry_launch_front is trivially reusable here: the filtered config
    # carries the lane's own `public: true`, so the catalogue filter trims
    # /v1/models to exactly this lane — no fallback hops exist to hide. One
    # worker: this door serves one extraction lane, not the multi-lane pool.
    if ! _ferry_launch_front "$schematron_config" "$target_port" "$schematron_log" 1; then
      echo "    note: catalogue filter unavailable — serving litellm directly"
      _ferry_export_chatgpt_instructions
      nohup litellm \
        --config "$schematron_config" \
        --port "$target_port" \
        --num_workers 1 \
        --host 0.0.0.0 >> "$schematron_log" 2>&1 & disown
    fi
    _ferry_wait_http "http://127.0.0.1:$target_port/health/liveliness" "schematron" 120 readiness || true
    _ferry_wait_http "http://127.0.0.1:$LOCAL_SCHEMATRON_PORT/v1/models" "local-schematron" 900 || true

    echo "================================================================="
    echo "   FERRY SCHEMATRON — extraction lane"
    echo "================================================================="
    echo "   Endpoint:  http://$MDNS_NAME:$target_port/v1"
    echo "   Model:     schematron  (upstream: $upstream)"
    if (( schem_lane_warm )); then
      echo "   Backend:   MLX on :$LOCAL_SCHEMATRON_PORT (was ALREADY warm — reused, not restarted)"
    else
      echo "   Backend:   MLX on :$LOCAL_SCHEMATRON_PORT (launched by this door)"
      echo "   Log (MLX): $LOCAL_SCHEMATRON_LOG"
    fi
    echo "   Main door :$PORT is untouched — this door is independent of it."
    echo "-----------------------------------------------------------------"
    echo "   The JSON schema goes INSIDE the user message (Schematron's prompt"
    echo "   contract) — ferry fronts the model verbatim and rewrites nothing."
    echo "   cdp-toolkit: export CDP_EXTRACT_BASE_URL=http://127.0.0.1:$target_port/v1"
    echo "   Stop this door only: ferry down --port $target_port"
    echo "   Log: $schematron_log"
    echo "================================================================="
  fi
}

cmd_down() {
  if (( CLIENT_MODE )); then
    echo "Error: Command 'ferry down' is only available on the LLM-Ferry Host Mac."
    exit 1
  fi

  # `ferry down --port P` stops ONLY the ferry proxy on :P — the way to take
  # down a companion door (e.g. the schematron extractor on :$SCHEMATRON_PORT)
  # without disturbing the main endpoint or the GPU lanes.
  local stop_port=""
  while [[ $# -gt 0 ]]; do
    case "$1" in
      -p|--port)
        stop_port="$2"
        shift 2
        ;;
      *)
        echo "Unknown option: $1"
        usage
        ;;
    esac
  done

  if [[ -n "$stop_port" ]]; then
    echo ">>> Stopping ONLY the ferry proxy on :$stop_port (everything else stays up)..."
    _ferry_stop_litellm "$stop_port"
    _ferry_free_port "$stop_port"

    # v1.36.0: the schematron door now has a LOCAL MLX backend behind it, and a
    # door taken down without its backend leaves ~8.5GB of weights resident
    # forever. So stopping THAT door also stops the lane — but ONLY when the
    # main stack is down.
    #
    # The rule, and why it is conditional: `ferry up` (stack) wires its own
    # `schematron` deployment at the SAME :$LOCAL_SCHEMATRON_PORT. If :$PORT is
    # listening, the stack is serving extraction through that very process, and
    # reaping it here would silently break a lane the operator never asked to
    # touch. The companion door is the borrower in that case, never the owner.
    #
    # SCOPE: this keys on the CONVENTIONAL door port ($SCHEMATRON_PORT, itself
    # FERRY_SCHEMATRON_PORT-overridable), not on "any door serving the filtered
    # config". A door started ad hoc on some other port with `-p` is not
    # recognised as the lane's owner, so its `down --port` leaves the backend
    # resident — stop that one with `ferry down` or by freeing :8100 directly.
    # Deliberate: guessing ownership from a port the operator chose per-run is
    # how you end up reaping a lane that something else is using.
    if [[ "$stop_port" == "$SCHEMATRON_PORT" ]]; then
      if lsof -nP -iTCP:"$PORT" -sTCP:LISTEN >/dev/null 2>&1; then
        echo ">>> Leaving the local-schematron lane on :$LOCAL_SCHEMATRON_PORT up:"
        echo "    the main stack on :$PORT is running and serves the schematron lane through it."
      elif lsof -nP -iTCP:"$LOCAL_SCHEMATRON_PORT" -sTCP:LISTEN >/dev/null 2>&1; then
        echo ">>> Also stopping the local-schematron MLX lane on :$LOCAL_SCHEMATRON_PORT"
        echo "    (no main stack on :$PORT, so nothing else is using it)."
        _ferry_free_port "$LOCAL_SCHEMATRON_PORT"
      fi
    fi

    echo ">>> Success: :$stop_port cleared."
    return 0
  fi

  echo ">>> Stopping all LLM-Ferry servers, cloud proxies, and share servers..."

  # Terminate local model servers. One pkill covers BOTH stack lanes (they are the
  # same binary on different ports) as well as any single-lane server.
  pkill -f mlx_vlm.server || true

  # Belt-and-braces: free the stack's internal lane ports even if something OTHER
  # than mlx_vlm.server ended up holding one (a wedged uvicorn child, a stale
  # process from a killed run). Without this a later `ferry up` finds the port
  # taken and the lane silently never binds.
  for _p in "$LOCAL_ORCH_PORT" "$LOCAL_SUB_PORT" "$LOCAL_SCHEMATRON_PORT"; do
    if lsof -nP -iTCP:"$_p" -sTCP:LISTEN >/dev/null 2>&1; then
      lsof -ti tcp:"$_p" | xargs kill -9 2>/dev/null || true
    fi
  done
  
  # Terminate LiteLLM proxies. Both matchers plus the WAIT live in
  # _ferry_stop_litellm: `ferry down && ferry up` used to race, because litellm
  # keeps writing its log for a few seconds after `down` returns, and the next
  # launch would then truncate that same inode underneath it (see _ferry_reset_log).
  _ferry_stop_litellm

  # Terminate sharing Python servers. They run as `python3 - <port> <dir> ferry-share-marker`
  # with the script fed on stdin via heredoc, so "DynamicHandler" lives in stdin and never
  # appears in argv — `pkill -f DynamicHandler` could never match them and leaked a server
  # every session. We tag them with a stable sentinel arg and match that instead.
  pkill -f "ferry-share-marker" || true
  pkill -f "host-share.sh" || true

  # Terminate the experimental HuggingFace pass-through proxy (tagged with its sentinel arg).
  pkill -f "ferry-hf-marker" || true

  # Terminate the general HTTP(S) download forward proxy (tagged with its sentinel arg).
  pkill -f "ferry-proxy-marker" || true

  # Terminate the browser VNC viewer (its bridges die with it; the published
  # ports themselves belong to the relay, killed just below).
  pkill -f "ferry-vnc-marker" || true

  # Terminate the reverse-expose relay. Killing it drops every control connection,
  # which is what makes each client's published port close with it — an exposure
  # must not survive the thing that was supposed to be publishing it.
  if pkill -f "ferry-relay-marker" 2>/dev/null; then
    rm -f "$RELAY_STATE_FILE"
  fi

  echo ">>> Success: All servers stopped."
}

# cmd_reload — restart ONLY the litellm front door, leaving the GPU lanes
# (mlx_vlm.server on :$LOCAL_ORCH_PORT/: $LOCAL_SUB_PORT) warm. Route-config edits
# (a new fallback hop, a retuned chain, a new key) only need the proxy to re-read
# ~/.config/ferry/litellm.yaml; a full `ferry down && ferry up` would also reload
# ~33GB of MLX weights for no reason. This is the fast path for config changes.
cmd_reload() {
  if (( CLIENT_MODE )); then
    echo "Error: Command 'ferry reload' is only available on the LLM-Ferry Host Mac."
    exit 1
  fi
  if ! command -v litellm >/dev/null 2>&1; then
    echo "Error: 'litellm' is missing. Run: ferry install"
    exit 1
  fi
  _ferry_require_route_config
  _ferry_warn_missing_keys

  local target_port="$PORT" cloud_log="$CLOUD_LOG"
  echo ">>> Reloading the front door (config re-read; GPU lanes stay warm)..."
  _ferry_stop_litellm "$target_port"
  _ferry_free_port "$target_port"

  echo ">>> [front] litellm --config $FERRY_ROUTE_CONFIG"
  echo "        port   :$target_port"
  echo "        workers:$FERRY_FRONT_WORKERS"
  echo "        log    $cloud_log"
  if ! _ferry_launch_front "$FERRY_ROUTE_CONFIG" "$target_port" "$cloud_log"; then
    echo "        note: catalogue filter unavailable — serving litellm directly"
    _ferry_export_chatgpt_instructions
    nohup litellm \
      --config "$FERRY_ROUTE_CONFIG" \
      --port "$target_port" \
      --num_workers "$FERRY_FRONT_WORKERS" \
      --host 0.0.0.0 >> "$cloud_log" 2>&1 & disown
  fi

  _ferry_wait_http "http://127.0.0.1:$target_port/health/liveliness" "front" 120 readiness || true

  local -a banner_auth=()
  [[ -n "${LITELLM_MASTER_KEY:-}" ]] && banner_auth=(-H "Authorization: Bearer $LITELLM_MASTER_KEY")
  echo "================================================================="
  echo ">>> Front door reloaded on http://$MDNS_NAME:$target_port/v1 :"
  curl -fsS -m 5 "${banner_auth[@]}" "http://127.0.0.1:$target_port/v1/models" 2>/dev/null \
    | python3 -c "import json,sys; [print('       ' + m['id']) for m in json.load(sys.stdin).get('data', [])]" 2>/dev/null \
    || echo "       (not answering yet — check $cloud_log)"
  echo "    GPU lanes untouched. Full stack restart: ferry down && ferry up"
  echo "================================================================="
}

cmd_status() {
  if (( CLIENT_MODE )); then
    echo "================================================================="
    echo "                 LLM-FERRY CLIENT DIAGNOSTICS"
    echo "================================================================="
    echo "Config Profile:      $CLIENT_CONF"
    echo "Host Target Server:  http://$CLIENT_HOST:$CLIENT_PORT"
    echo "Host Telemetry Port: http://$CLIENT_HOST:$CLIENT_SHARE_PORT"
    echo "================================================================="
    
    echo ">>> Probing network connectivity to Host Mac..."
    # Connectivity is readiness, not catalogue content: under master_key
    # (v1.22.0) the catalogue answers 401 to a keyless probe, and that still
    # proves the host is reachable and serving, so count it ONLINE. Kept on
    # /v1/models rather than /health/liveliness because the target may be a
    # bare MLX lane (ferry up --local), which serves no liveliness route.
    local probe_code="$(curl -sS -m 3 -o /dev/null -w '%{http_code}' "http://$CLIENT_HOST:$CLIENT_PORT/v1/models" 2>/dev/null || true)"
    if [[ "$probe_code" == 2* ]] || [[ "$probe_code" == "401" ]]; then
      echo ">>> Connection Health: \033[1;32mONLINE\033[0m"
      [[ "$probe_code" == "401" ]] && echo "    (front door is behind master_key auth — export LITELLM_MASTER_KEY to read the lane list)"

      # Query active model. Catalogue content: send the bearer when this box
      # has the host's key; without one this degrades to skipping the line.
      local -a status_auth=()
      [[ -n "${LITELLM_MASTER_KEY:-}" ]] && status_auth=(-H "Authorization: Bearer $LITELLM_MASTER_KEY")
      local models=$(curl -fsS -m 2 "${status_auth[@]}" "http://$CLIENT_HOST:$CLIENT_PORT/v1/models" 2>/dev/null || echo "")
      if [[ -n "$models" ]]; then
        local active=$(echo "$models" | python3 -c "import json,sys; d=json.load(sys.stdin).get('data',[]); print(d[0]['id'] if d else 'None')")
        echo "    Currently active model on Host: \033[1;32m$active\033[0m"
      fi
    else
      echo ">>> Connection Health: \033[1;31mOFFLINE\033[0m (Check Wi-Fi/cable or network status)"
    fi
    echo "================================================================="
    return
  fi

  # Host status
  echo "================================================================="
  echo "                 LLM-FERRY SYSTEM ACTIVE LISTENERS"
  echo "================================================================="
  echo "Host mDNS Domain:    http://$MDNS_NAME"
  echo "Host active LAN IP:  http://$LAN_IP"
  echo "================================================================="

  # Check active ports. $PORT is the client-facing door; the three lane ports are
  # INTERNAL backends (only populated in stack mode) and are labelled as such so
  # an OFFLINE lane port is not mistaken for the endpoint being down.
  local _label
  for p in "$PORT" "$LOCAL_ORCH_PORT" "$LOCAL_SUB_PORT" "$LOCAL_SCHEMATRON_PORT" "$SHARE_PORT"; do
    case "$p" in
      "$PORT")            _label="endpoint" ;;
      "$LOCAL_ORCH_PORT") _label="local-orch lane (internal)" ;;
      "$LOCAL_SUB_PORT")  _label="local-sub lane (internal)" ;;
      "$LOCAL_SCHEMATRON_PORT") _label="local-schematron lane (internal)" ;;
      "$SHARE_PORT")      _label="client share" ;;
      *)                  _label="" ;;
    esac
    if lsof -nP -iTCP:"$p" -sTCP:LISTEN >/dev/null 2>&1; then
      local pid=$(lsof -t -iTCP:"$p" -sTCP:LISTEN | head -1)
      local cmd=$(ps -p "$pid" -o comm= 2>/dev/null | xargs basename 2>/dev/null || echo "unknown")
      echo ">>> Port $p ($_label) is \033[1;32mONLINE\033[0m (PID: $pid, Command: $cmd)"

      # phys_footprint is the number that matters for an MLX lane — RSS is blind
      # to wired GPU memory, so `ps` cheerfully under-reports a 50GB model server.
      if [[ "$p" == "$LOCAL_ORCH_PORT" || "$p" == "$LOCAL_SUB_PORT" || "$p" == "$LOCAL_SCHEMATRON_PORT" ]] && command -v footprint >/dev/null 2>&1; then
        local fp=$(footprint "$pid" 2>/dev/null | grep -iE "phys_footprint" | head -1 | tr -s ' ')
        [[ -n "$fp" ]] && echo "    Memory: $fp"
      fi

      if [[ "$p" == "$LOCAL_ORCH_PORT" || "$p" == "$LOCAL_SUB_PORT" || "$p" == "$LOCAL_SCHEMATRON_PORT" ]]; then
        # An MLX lane's /v1/models lists the whole HuggingFace CACHE, not what is
        # loaded — printing it would advertise eight models this lane cannot serve
        # without a reload. The launch line is the truth, so read --model from argv.
        local loaded=$(ps -p "$pid" -o args= 2>/dev/null | sed -E 's/.*--model[= ]+([^ ]+).*/\1/')
        [[ -n "$loaded" ]] && echo "    Model loaded: \033[1;32m$loaded\033[0m"
        echo "    (internal backend — address this lane as a lane name on :$PORT, not here)"
      elif [[ "$p" != "$SHARE_PORT" ]]; then
        # The front door: list every lane it serves, not just the first one.
        # Catalogue content, so under master_key (v1.22.0) the bearer is needed;
        # without a key this degrades to no listing, exactly as before.
        local -a models_auth=()
        [[ -n "${LITELLM_MASTER_KEY:-}" ]] && models_auth=(-H "Authorization: Bearer $LITELLM_MASTER_KEY")
        local models=$(curl -fsS -m 3 "${models_auth[@]}" "http://127.0.0.1:$p/v1/models" 2>/dev/null || echo "")
        if [[ -n "$models" ]]; then
          local served=$(echo "$models" | python3 -c "import json,sys; print(' '.join(m['id'] for m in json.load(sys.stdin).get('data',[])))" 2>/dev/null || echo "")
          if [[ -n "$served" ]]; then
            echo "    Lanes served: \033[1;32m$served\033[0m"
            local first=${served%% *}
            # max_tokens 256, not 10. Every lane on this endpoint is now a
            # REASONING model, and reasoning tokens are drawn from the same
            # budget as the answer: at 10 the whole allowance is spent thinking
            # and the response comes back finish_reason=length with
            # content=null. The one command status hands you to prove the stack
            # is alive then reads as a dead lane. Measured on the orch lane
            # 2026-08-26: reasoning_tokens 7, text_tokens 3, content None.
            # Under master_key the pasted command would 401, so the auth hint
            # references the VARIABLE — the pasting shell resolves it, and the
            # key value itself is never printed.
            local auth_hint=""
            [[ -n "${LITELLM_MASTER_KEY:-}" ]] && auth_hint=" -H \"Authorization: Bearer \$LITELLM_MASTER_KEY\""
            echo "    Test with: curl -fsS${auth_hint} -H \"Content-Type: application/json\" -d '{\"model\":\"$first\",\"messages\":[{\"role\":\"user\",\"content\":\"Say Hello!\"}],\"max_tokens\":256}' http://$MDNS_NAME:$p/v1/chat/completions"
          fi
        fi
      fi
    else
      echo ">>> Port $p ($_label) is \033[1;31mOFFLINE\033[0m"
    fi
  done

  # Experimental HuggingFace pass-through proxy (only reported when running).
  if lsof -nP -iTCP:"$HF_PORT" -sTCP:LISTEN >/dev/null 2>&1; then
    local hf_pid=$(lsof -t -iTCP:"$HF_PORT" -sTCP:LISTEN)
    echo ">>> Port $HF_PORT is \033[1;32mONLINE\033[0m (HF pass-through proxy, PID: $hf_pid)"
  fi

  # General HTTP(S) download forward proxy (only reported when running).
  if lsof -nP -iTCP:"$PROXY_PORT" -sTCP:LISTEN >/dev/null 2>&1; then
    echo ">>> HTTP download proxy is \033[1;32mONLINE\033[0m (port $PROXY_PORT)"
  fi

  # Reverse-expose relay, and — the part that matters — every port a client has
  # published through it. A port opened on this machine on someone else's behalf
  # must be visible here, or nobody can answer "what is this host serving?".
  if lsof -nP -iTCP:"$RELAY_PORT" -sTCP:LISTEN >/dev/null 2>&1; then
    echo ">>> Reverse relay is \033[1;32mONLINE\033[0m (control port $RELAY_PORT)"
    if [[ -f "$RELAY_STATE_FILE" ]]; then
      python3 - "$RELAY_STATE_FILE" <<'PYEOF' || true
import json, sys
try:
    pub = json.load(open(sys.argv[1]))
    if not isinstance(pub, dict):
        pub = {}
    pub = {p: i for p, i in pub.items() if str(p).isascii() and str(p).isdigit() and isinstance(i, dict)}
    if not pub:
        print("    No ports published right now.")
    for port, info in sorted(pub.items(), key=lambda kv: int(kv[0])):
        label = f" ({info['label']})" if info.get("label") else ""
        kind = info.get("kind", "tcp")
        print(f"    Published {info.get('bind', '?')}:{port} [{kind}] for {info.get('client', '?')}{label}"
              f"  since {info.get('since', '?')}")
except Exception:
    sys.exit(0)
PYEOF
    fi
  fi

  # Browser VNC viewer: one URL per screen a client has published with expose-vnc.
  if lsof -nP -iTCP:"$VNC_PORT" -sTCP:LISTEN >/dev/null 2>&1; then
    echo ">>> VNC viewer is \033[1;32mONLINE\033[0m (http://$MDNS_NAME:$VNC_PORT)"
    if [[ -f "$RELAY_STATE_FILE" ]]; then
      python3 - "$RELAY_STATE_FILE" "$MDNS_NAME" "$VNC_PORT" <<'PYEOF' || true
import json, sys
try:
    pub = json.load(open(sys.argv[1]))
    if not isinstance(pub, dict):
        pub = {}
    screens = {p: i for p, i in pub.items()
               if str(p).isascii() and str(p).isdigit() and isinstance(i, dict) and i.get("kind") == "vnc"}
    if not screens:
        print("    No screens published right now (client: ferry expose-vnc).")
    for port, info in sorted(screens.items(), key=lambda kv: int(kv[0])):
        print(f"    Screen {info.get('label') or info.get('client', '?')}: http://{sys.argv[2]}:{sys.argv[3]}/vnc/{port}")
except Exception:
    sys.exit(0)
PYEOF
    fi
  fi

  # Offered files manifest (from `ferry offer`).
  local offered_file="$HOME/.config/ferry/offered.json"
  if [[ -f "$offered_file" ]]; then
    local offered_count=$(python3 -c "import json; print(len(json.load(open('$offered_file'))))" 2>/dev/null || echo "?")
    echo ">>> Offered files: $offered_count  ($offered_file)"
  fi
  echo "================================================================="
}

cmd_share() {
  if (( CLIENT_MODE )); then
    echo "Error: Command 'ferry share' is only available on the LLM-Ferry Host Mac."
    exit 1
  fi

  local target_port="$SHARE_PORT"
  
  # Scan upwards to find an available port
  while lsof -nP -iTCP:"$target_port" -sTCP:LISTEN >/dev/null 2>&1; do
    echo ">>> Share port $target_port is already in use. Checking next port..."
    target_port=$((target_port + 1))
  done

  echo "================================================================="
  echo "                SHARING LLM-FERRY CLIENT BOOTSTRAPPER"
  echo "================================================================="
  echo "Serving directory: $APP_DIR"
  echo "Port bound:        $target_port"
  echo "================================================================="
  echo ">>> FIRST-TIME SETUP on any client laptop on the same LAN:"
  echo "    \033[1;32mcurl -fsSL http://$MDNS_NAME:$target_port/client-bootstrap.sh | zsh\033[0m"
  echo "    (or: curl -fsSL http://$LAN_IP:$target_port/client-bootstrap.sh | zsh)"
  echo "    Narrow the opencode scope on a laptop that already has its own setup:"
  echo "      ... /client-bootstrap.sh | zsh -s -- --profiles-only   (ferry's own profiles only)"
  echo "      ... /client-bootstrap.sh | zsh -s -- --no-opencode     (the CLI and nothing else)"
  echo ""
  echo ">>> CATCH UP an already-bootstrapped client (re-pull the CLI, re-apply"
  echo "    the opencode takeover; leaves ~/.zshrc alone):"
  echo "    \033[1;32mcurl -fsSL http://$MDNS_NAME:$target_port/client-reset.sh | zsh\033[0m"
  echo "    (or: curl -fsSL http://$LAN_IP:$target_port/client-reset.sh | zsh)"
  echo ""
  echo ">>> REMOVE ferry from a client (CLI, profile, wrappers, guardrails;"
  echo "    keeps opencode's own session history unless --full --yes):"
  echo "    \033[1;32mcurl -fsSL http://$MDNS_NAME:$target_port/client-cleanup.sh | zsh -s -- --dry-run\033[0m"
  echo "    (drop --dry-run to apply)"
  echo "================================================================="
  echo ">>> Starting Dynamic Python HTTP share server in background..."

  # Launch the dynamic share server in background (log to share log)
  # The trailing "ferry-share-marker" arg is a stable kill sentinel for `ferry down` (see cmd_down);
  # the heredoc script only reads argv[1]/argv[2], so the extra arg is ignored at runtime.
  nohup python3 - "$target_port" "$APP_DIR" "ferry-share-marker" <<'PYEOF' > "$SHARE_LOG" 2>&1 & disown
import sys, os, socket, subprocess, json, tarfile, glob
from http.server import SimpleHTTPRequestHandler, ThreadingHTTPServer

port = int(sys.argv[1])
directory = sys.argv[2]

# macOS advertises scutil's LocalHostName; other OSes (Linux/avahi) advertise the
# short hostname as <hostname>.local. Try scutil first, then fall back to the hostname.
mdns_name = subprocess.getoutput("scutil --get LocalHostName 2>/dev/null").strip().lower()
if mdns_name:
    mdns_name += ".local"
else:
    mdns_name = socket.gethostname().split(".")[0].lower() + ".local"

# Ferry transfer locations: the host's HuggingFace cache and the offered-files manifest.
HF_HUB = os.path.expanduser("~/.cache/huggingface/hub")
OFFERED = os.path.expanduser("~/.config/ferry/offered.json")
# Client telemetry (`ferry msg` / `ferry log`) lands here, NOT under the serving
# directory. `directory` is whatever tree the server was launched from, captured
# once at startup: a checkout that later moves or is deleted — a git worktree
# removed after the share server was started from it — turns every /hq POST into
# an unhandled exception and a bare 500. The client sees a failed send, the host
# sees nothing, and the message is gone. Observed 2026-08-26, two messages lost.
# Same stable-path treatment as OFFERED above, so telemetry outlives any checkout.
CLIENT_LOG = os.path.expanduser("~/.config/ferry/client_logs.txt")

class DynamicHandler(SimpleHTTPRequestHandler):
    def _tar_stream(self, root, arcname):
        # Stream a tar of `root` back to the client. dereference=True resolves the
        # HuggingFace cache's snapshot symlinks (which point into ../../blobs) into
        # real file content, so the pulled model is self-contained on the client.
        self.send_response(200)
        self.send_header("Content-Type", "application/x-tar")
        self.end_headers()
        with tarfile.open(fileobj=self.wfile, mode="w|", dereference=True) as tar:
            tar.add(root, arcname=arcname)

    def do_GET(self):
        # Client-facing scripts get the host's live identity injected. Add a name
        # here and it is served the same way; serve it as a plain static file and
        # its placeholders reach the client verbatim, where they resolve to a
        # bogus `your-host.local` and the script fails at its first request.
        INJECTED = ("client-bootstrap.sh", "client-reset.sh")
        requested = self.path.rsplit("/", 1)[-1]
        if requested in INJECTED:
            file_path = os.path.join(self.directory, requested)
            if os.path.exists(file_path):
                try:
                    with open(file_path, "r", encoding="utf-8") as f:
                        content = f.read()
                    content = content.replace("HOST_MDNS_PLACEHOLDER", mdns_name)
                    content = content.replace("SHARE_PORT_PLACEHOLDER", str(port))
                    # v1.22.0: master_key deliberately does NOT ride this channel — the share server is unauthenticated, so injecting the key here would publish it to the whole LAN (clients bring it via FERRY_MASTER_KEY / --key, or it travels in client.json after they supply it).
                    # Also rewrite the script's own your-host.local fallback, so
                    # even the no-injection code path lands on the real host.
                    content = content.replace("your-host.local", mdns_name)
                    content = content.replace('"your-host.local"', f'"{mdns_name}"')
                    body = content.encode("utf-8")
                    self.send_response(200)
                    self.send_header("Content-Type", "application/x-sh")
                    # Content-Length MUST count BYTES, not characters: the script
                    # contains multi-byte UTF-8 (em-dashes), and a char-count header
                    # silently truncates the tail (an unterminated `echo "` at EOF,
                    # which the client's zsh reports as `unmatched "`).
                    self.send_header("Content-Length", str(len(body)))
                    self.end_headers()
                    self.wfile.write(body)
                    return
                except Exception as e:
                    print(f"Error performing dynamic replacement: {e}")

        # /pull/<model-id> — tar-stream a model from the host's local HuggingFace cache.
        if self.path.startswith("/pull/"):
            model_id = self.path[len("/pull/"):].strip("/")
            base = os.path.join(HF_HUB, "models--" + model_id.replace("/", "--"))
            snaps = sorted(glob.glob(os.path.join(base, "snapshots", "*")))
            if not snaps:
                self.send_response(404); self.end_headers()
                self.wfile.write(b"model not in host cache\n"); return
            self._tar_stream(snaps[-1], model_id.split("/")[-1]); return

        # /file/<name> — tar-stream a previously offered file/dir (see `ferry offer`).
        if self.path.startswith("/file/"):
            name = self.path[len("/file/"):].strip("/")
            offered = json.load(open(OFFERED)) if os.path.exists(OFFERED) else {}
            p = offered.get(name)
            if not p or not os.path.exists(p):
                self.send_response(404); self.end_headers()
                self.wfile.write(b"not offered\n"); return
            self._tar_stream(p, os.path.basename(p.rstrip("/"))); return

        # /manifest — list the models in the host cache and the offered files.
        if self.path == "/manifest" or self.path.endswith("/manifest"):
            models = [os.path.basename(d)[len("models--"):].replace("--", "/")
                      for d in glob.glob(os.path.join(HF_HUB, "models--*"))]
            offered = json.load(open(OFFERED)) if os.path.exists(OFFERED) else {}
            body = json.dumps({"models": sorted(models), "files": sorted(offered.keys())}, indent=2).encode()
            self.send_response(200); self.send_header("Content-Type", "application/json")
            self.send_header("Content-Length", str(len(body))); self.end_headers()
            self.wfile.write(body); return

        super().do_GET()

    def do_POST(self):
        if self.path == "/hq" or self.path.endswith("/hq"):
            try:
                content_length = int(self.headers.get('Content-Length', 0))
                post_data = self.rfile.read(content_length).decode('utf-8')
                
                # Append client telemetry to the stable host-side log.
                os.makedirs(os.path.dirname(CLIENT_LOG), exist_ok=True)
                with open(CLIENT_LOG, "a", encoding="utf-8") as lf:
                    lf.write(f"=== CLIENT LOG ENTRY ===\n{post_data}\n\n")
                
                self.send_response(200)
                self.send_header("Content-Type", "text/plain")
                self.end_headers()
                self.wfile.write(b"Logged successfully at HQ!\n")
                return
            except Exception as e:
                print(f"Error handling HQ log: {e}")
                self.send_response(500)
                self.end_headers()
                return
        self.send_response(444)
        self.end_headers()

handler = lambda *a, **kw: DynamicHandler(*a, directory=directory, **kw)
ThreadingHTTPServer(('0.0.0.0', port), handler).serve_forever()
PYEOF

  echo ">>> Sharing server running in background. Log: $SHARE_LOG"
}

cmd_msg() {
  if (( ! CLIENT_MODE )); then
    echo "Error: Command 'ferry msg' is only available in Client Mode."
    exit 1
  fi
  if [[ $# -lt 1 ]]; then
    echo "Usage: ferry msg <your text message here>"
    exit 1
  fi
  local text="$*"
  echo ">>> Streaming direct text telemetry to Host HQ..."
  curl -sS -X POST --data-binary "$text" "http://$CLIENT_HOST:$CLIENT_SHARE_PORT/hq"
}

cmd_log() {
  if (( ! CLIENT_MODE )); then
    echo "Error: Command 'ferry log' is only available in Client Mode."
    exit 1
  fi
  echo ">>> Streaming stdin log stream directly back to Host HQ..."
  curl -sS -X POST --data-binary @- "http://$CLIENT_HOST:$CLIENT_SHARE_PORT/hq"
}


# cmd_inbox — read the telemetry clients POSTed to this host.
#
# `ferry msg` and `ferry log` POST a raw body to the share server's /hq endpoint,
# which appends it to $HQ_LOG. Nothing on the host read it back until this command:
# the answer lives in TWO files and neither one holds all of it.
#
#   $HQ_LOG                       every entry, verbatim, append-only. NO timestamp
#                                 and NO client IP — the handler writes a delimiter
#                                 and the body, nothing else.
#   $LOG_DIR/share-<port>.log     the share server's access log: timestamp, client
#                                 IP and status per POST — but TRUNCATED on every
#                                 `ferry share` relaunch.
#
# So a date is recovered by aligning the two from the END: the k receipts still in
# the access log belong to the k most recent entries. Older entries are real and
# undated, and this prints them as such rather than guessing.
#
# Only status-200 receipts are aligned. A 500 means the handler raised and NO entry
# was written, so counting it would shift every date by one.
cmd_inbox() {
  if (( CLIENT_MODE )); then
    echo "Error: Command 'ferry inbox' is only available on the LLM-Ferry Host Mac."
    echo "       A client SENDS telemetry ('ferry msg' / 'ferry log'); the host is"
    echo "       where it lands, so the inbox only exists there."
    exit 1
  fi

  local last=0 follow=0 show_all=0
  while [[ $# -gt 0 ]]; do
    case "$1" in
      -n|--last)   last="$2"; shift 2 ;;
      -f|--follow) follow=1; shift ;;
      -a|--all)    show_all=1; shift ;;
      --path)
        echo "$HQ_LOG"
        echo "$LOG_DIR/share-$SHARE_PORT.log"
        return 0 ;;
      -h|--help)
        echo "Usage: ferry inbox [-n N] [-f] [--all] [--path]"
        echo "  (no flags)   index the 20 most recent entries, dated where possible"
        echo "  -n N         print the last N entries IN FULL"
        echo "  -f           follow new entries as they land (tail -f)"
        echo "  --all        index every entry, not just the last 20"
        echo "  --path       print the content and receipt log paths"
        return 0 ;;
      *) echo "Unknown option for 'ferry inbox': $1"; exit 1 ;;
    esac
  done

  if (( follow )); then
    if [[ ! -f "$HQ_LOG" ]]; then
      echo ">>> No telemetry yet ($HQ_LOG does not exist). Waiting for the first entry..."
      mkdir -p "$(dirname "$HQ_LOG")"
      : >> "$HQ_LOG"
    fi
    echo ">>> Following $HQ_LOG (Ctrl-C to stop)"
    tail -f "$HQ_LOG"
    return 0
  fi

  python3 - "$HQ_LOG" "$LOG_DIR" "$last" "$show_all" "$APP_DIR/client_logs.txt" <<'PYEOF'
import glob, os, re, sys, time

hq_log, log_dir, last, show_all, legacy = sys.argv[1:6]
last, show_all = int(last), show_all == "1"
DELIM = "=== CLIENT LOG ENTRY ==="

if not os.path.exists(hq_log):
    print(f">>> No client telemetry yet: {hq_log} does not exist.")
    print("    Clients POST to the share server, so nothing lands while `ferry share` is down.")
    sys.exit(0)

with open(hq_log, encoding="utf-8", errors="replace") as f:
    raw = f.read()
# The first chunk is whatever preceded the first delimiter — normally empty.
entries = [e.strip("\n") for e in raw.split(DELIM)[1:]]

# --- Receipts: timestamp + client, newest last, across every share-<port>.log ---
# The port can differ from $SHARE_PORT: `ferry share` scans upward when the port is
# taken, so the log for a live server is not always the one named by the default.
LINE = re.compile(r'^(\S+) - - \[([^\]]+)\] "POST [^"]*/hq[^"]*" (\d{3})')
receipts, failed = [], 0
for path in glob.glob(os.path.join(log_dir, "share-*.log")):
    try:
        with open(path, encoding="utf-8", errors="replace") as f:
            for line in f:
                m = LINE.match(line)
                if not m:
                    continue
                ip, stamp, code = m.groups()
                # BaseHTTPRequestHandler stamps "28/Aug/2026 17:50:40" — a SPACE, not the
                # Apache colon. Parsing the Apache form first silently dated NOTHING (every
                # line raised ValueError and was skipped), and the listing still rendered
                # fine, just with an empty date column. Accept both, in that order.
                for fmt in ("%d/%b/%Y %H:%M:%S", "%d/%b/%Y:%H:%M:%S"):
                    try:
                        t = time.strptime(stamp, fmt)
                        break
                    except ValueError:
                        t = None
                if t is None:
                    continue
                if code == "200":
                    receipts.append((t, ip))
                else:
                    failed += 1
    except OSError:
        continue
receipts.sort(key=lambda r: r[0])

# Align from the END: receipt[-1] is entry[-1]. Anything older than the current
# access log is undated, which is a fact about the log, not a failure.
dated = {}
for i, r in enumerate(receipts[-len(entries):] if entries else []):
    dated[len(entries) - min(len(receipts), len(entries)) + i] = r

mtime = time.strftime("%d %b %H:%M", time.localtime(os.path.getmtime(hq_log)))
print(f">>> {hq_log}")
print(f"    {len(entries)} entr{'y' if len(entries) == 1 else 'ies'}, last arrival {mtime}"
      f"  ({len(dated)} dated by the current share log)")
if failed:
    print(f"    WARNING: {failed} POST(s) to /hq returned an error — those bodies were never written.")
if os.path.exists(legacy):
    print(f"    NOTE: a legacy {os.path.basename(legacy)} also exists beside the checkout")
    print(f"          ({legacy}). It is frozen pre-1.8.10 history, not live telemetry.")
print("")

def first_line(text):
    for line in text.split("\n"):
        if line.strip():
            return line.strip()
    return "(empty body)"

if last > 0:
    for i in range(max(0, len(entries) - last), len(entries)):
        t, ip = dated.get(i, (None, None))
        when = time.strftime("%d/%b %H:%M", t) if t else "undated"
        print(f"===== entry {i + 1}  [{when}  {ip or 'unknown client'}] =====")
        print(entries[i])
        print("")
else:
    shown = entries if show_all else entries[-20:]
    offset = len(entries) - len(shown)
    for i, e in enumerate(shown):
        t, ip = dated.get(offset + i, (None, None))
        when = time.strftime("%d/%b %H:%M", t) if t else "     —     "
        print(f"{offset + i + 1:3d}  {when}  {(ip or ''):15.15s}  {first_line(e):.60s}")
    if offset:
        print(f"\n    ({offset} older entr{'y' if offset == 1 else 'ies'} not shown — pass --all)")
    if entries:
        print(f"\n    Full text of the newest:  ferry inbox -n 1")
PYEOF
}

# Reverse expose — the symmetric half of ferry.
#
# Everything else in ferry pushes HOST -> CLIENT: inference, files, the forward
# proxy. This is the other direction, and it exists because of a topology ferry
# already assumes: a trusted host, and clients that can only reach it OUTBOUND.
# A locked-down laptop (a managed firewall that resets inbound connections, a
# corporate proxy that kills every tunnel service) can still publish its own local
# service — an opencode server, a dev server, a notebook — to a phone or another
# machine on the LAN, by dialling the host and letting the host do the listening.
#
#   host:    ferry relay                      # accept registrations, publish ports
#   client:  ferry expose 4290 --as 4290      # "serve my 127.0.0.1:4290 from the host"
#
# HOW THE BYTES MOVE. Two kinds of connection, BOTH dialled by the client, so
# nothing ever connects INTO the client:
#
#   control  client -> relay, held open. Carries {"op":"register"}, then one
#            {"op":"open","id":N} from the relay per inbound public connection.
#   data     client -> relay, one per public connection. The client dials its own
#            127.0.0.1:<local-port> and pumps bytes between the two sockets.
#
# The relay parks each accepted public socket until its matching data connection
# arrives, then splices them. When the control connection drops, the public
# listener and everything parked behind it are closed — an exposure cannot outlive
# the client that asked for it.
#
# WHAT THIS IS NOT: an auth layer for the service being exposed. The token proves
# that whoever REGISTERED is your client; it says nothing about whoever connects
# to the published port. Expose something that has its own authentication.
cmd_relay() {
  if (( CLIENT_MODE )); then
    echo "Error: Command 'ferry relay' is only available on the LLM-Ferry Host Mac."
    echo "       The client side is 'ferry expose <port>'."
    exit 1
  fi

  local relay_port="$RELAY_PORT" bind_addr="0.0.0.0" foreground=0 show_token=0
  # Default, not empty: a hand-run `ferry relay --foreground` should be just as
  # killable by `ferry down` as a backgrounded one.
  local marker="ferry-relay-marker"
  while [[ $# -gt 0 ]]; do
    case "$1" in
      --port)       relay_port="$2"; shift 2 ;;
      --bind)       bind_addr="$2"; shift 2 ;;
      --foreground) foreground=1; shift ;;
      --token)      show_token=1; shift ;;
      # Sentinel carried into the SERVER's argv so `ferry down` can find it; the
      # same trick cmd_share uses. It must ride on the process that actually holds
      # the port, which is why the python is exec'd below rather than spawned.
      --marker)     marker="$2"; shift 2 ;;
      -h|--help)
        echo "Usage: ferry relay [--port P] [--bind ADDR] [--foreground] [--token]"
        echo "  --port P      control port clients dial [default: $RELAY_PORT]"
        echo "  --bind ADDR   what published ports bind to [default: 0.0.0.0, i.e. the LAN]"
        echo "                  --bind 127.0.0.1 keeps an exposure on this machine only"
        echo "  --foreground  run in this terminal instead of the background"
        echo "  --token       print the shared token and exit"
        echo "Stop it with 'ferry down'."
        return 0 ;;
      *) echo "Unknown option for 'ferry relay': $1"; exit 1 ;;
    esac
  done

  # The token is what separates "a client of yours" from "anything on the LAN".
  # Generated once, 0600, and never written to a log — only to this terminal,
  # because the human is the transport that carries it to the client.
  mkdir -p "$HOME/.config/ferry"
  if [[ ! -f "$RELAY_TOKEN_FILE" ]]; then
    python3 -c "import secrets; print(secrets.token_urlsafe(24))" > "$RELAY_TOKEN_FILE"
    chmod 600 "$RELAY_TOKEN_FILE"
  fi
  local token; token="$(cat "$RELAY_TOKEN_FILE")"

  if (( show_token )); then
    echo "$token"
    return 0
  fi

  # Ports ferry itself owns are refused as publish targets, so an exposure can
  # never quietly shadow the inference endpoint or the share server. Built from
  # the same constants those services use — 8091 is the dashboard, which is a
  # literal there too.
  local reserved="$PORT,8091,$SHARE_PORT,$HF_PORT,$PROXY_PORT,$LOCAL_ORCH_PORT,$LOCAL_SUB_PORT,$LOCAL_SCHEMATRON_PORT,$SCHEMATRON_PORT,$relay_port,$VNC_PORT"

  if (( ! foreground )); then
    if lsof -nP -iTCP:"$relay_port" -sTCP:LISTEN >/dev/null 2>&1; then
      echo "Error: port $relay_port is already in use — a relay may already be running."
      echo "       'ferry down' stops it, or pass --port to use another."
      exit 1
    fi

    echo "================================================================="
    echo "                    LLM-FERRY REVERSE RELAY"
    echo "================================================================="
    echo "Control port:   $relay_port   (clients dial this, outbound)"
    echo "Published ports bind to: $bind_addr"
    echo "Log:            $RELAY_LOG"
    echo "================================================================="
    echo ">>> On the client, publish a local service through this host:"
    echo "    \033[1;32mferry expose <local-port> --as <public-port> --token $token\033[0m"
    echo "    (the token is saved on the client after the first successful run)"
    echo ""
    echo ">>> Whatever you expose keeps its OWN auth. The relay authenticates the"
    echo "    client that registers, never the visitors who reach the port."
    echo "================================================================="

    _ferry_reset_log "$RELAY_LOG"
    # Re-invoke this same script in --foreground rather than duplicating the
    # server: one implementation, one heredoc, and the sentinel lands in argv.
    nohup "$FERRY_BIN_PATH" relay --foreground --port "$relay_port" --bind "$bind_addr" \
      --marker ferry-relay-marker > "$RELAY_LOG" 2>&1 & disown
    sleep 1
    if lsof -nP -iTCP:"$relay_port" -sTCP:LISTEN >/dev/null 2>&1; then
      echo ">>> Relay running in the background. 'ferry status' lists published ports;"
      echo "    'ferry down' stops it."
    else
      echo "    WARNING: the relay is not listening. See $RELAY_LOG"
      exit 1
    fi
    return 0
  fi

  # exec, and the sentinel as the last argv: the process that ends up holding the
  # port must be the one `ferry down` can match. Spawning the python as a child of
  # this shell put the sentinel on the PARENT — `pkill -f ferry-relay-marker` then
  # killed the wrapper, reported success, and left the relay listening. The python
  # reads argv[1:6] and ignores the rest, exactly as the share server does.
  exec python3 - "$relay_port" "$bind_addr" "$RELAY_TOKEN_FILE" "$RELAY_STATE_FILE" "$reserved" "$marker" <<'PYEOF'
import hmac, json, os, socket, sys, threading, time

port, bind_addr, token_file, state_file, reserved_csv = sys.argv[1:6]
port = int(port)
reserved = {int(p) for p in reserved_csv.split(",") if p.strip()}

def log(msg):
    print(f"[{time.strftime('%Y-%m-%d %H:%M:%S')}] {msg}", flush=True)

def token():
    with open(token_file) as f:
        return f.read().strip()

def read_line(sock, limit=4096):
    """Read one \\n-terminated line ONE BYTE AT A TIME.

    Deliberately not sock.makefile(): a buffered reader can swallow the first
    bytes of the stream that follows the handshake into its readahead, and on a
    data connection whose service greets first (SSH, SMTP, anything chatty) those
    bytes are then lost with no error anywhere.
    """
    buf = bytearray()
    while len(buf) < limit:
        b = sock.recv(1)
        if not b:
            return None
        if b == b"\n":
            return bytes(buf)
        buf += b
    return None

def send_json(sock, obj):
    sock.sendall((json.dumps(obj) + "\n").encode())

# --- shared state -----------------------------------------------------------
pending = {}                       # conn id -> public socket awaiting its data conn
pending_lock = threading.Lock()
published = {}                     # public port -> what `ferry status` reports
published_lock = threading.Lock()

def write_state():
    with published_lock:
        snapshot = {str(k): v for k, v in published.items()}
    tmp = state_file + ".tmp"
    with open(tmp, "w") as f:
        json.dump(snapshot, f, indent=2)
    os.replace(tmp, state_file)     # atomic: `ferry status` never reads a half-file

def close_quietly(sock):
    try:
        sock.close()
    except OSError:
        pass

def pump(src, dst):
    try:
        while True:
            chunk = src.recv(65536)
            if not chunk:
                break
            dst.sendall(chunk)
    except OSError:
        pass
    finally:
        try:
            dst.shutdown(socket.SHUT_WR)
        except OSError:
            pass

def splice(a, b):
    """Pump both directions, and do not return until both are done."""
    t = threading.Thread(target=pump, args=(a, b), daemon=True)
    t.start()
    pump(b, a)
    t.join(timeout=5)
    close_quietly(a)
    close_quietly(b)

# --- one registered client --------------------------------------------------
def serve_registration(ctrl, addr, req):
    public_port = int(req.get("public_port", 0))
    label = str(req.get("label", ""))[:120]
    kind = "vnc" if req.get("kind") == "vnc" else "tcp"
    if public_port < 1024 or public_port > 65535:
        send_json(ctrl, {"ok": False, "error": f"public port {public_port} out of range (1024-65535)"})
        return
    if public_port in reserved:
        send_json(ctrl, {"ok": False,
                         "error": f"port {public_port} belongs to ferry itself; pick another"})
        return

    listener = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
    listener.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
    try:
        listener.bind((bind_addr, public_port))
        listener.listen(64)
    except OSError as e:
        send_json(ctrl, {"ok": False, "error": f"cannot bind {bind_addr}:{public_port} ({e})"})
        close_quietly(listener)
        return

    send_json(ctrl, {"ok": True, "public_port": public_port, "bind": bind_addr})
    with published_lock:
        published[public_port] = {"client": addr[0], "label": label, "kind": kind,
                                  "since": time.strftime("%Y-%m-%d %H:%M:%S"), "bind": bind_addr}
    write_state()
    log(f"published {bind_addr}:{public_port} for {addr[0]} {('(' + label + ')') if label else ''}")

    ctrl_lock = threading.Lock()
    stop = threading.Event()
    counter = [0]

    def accept_loop():
        while not stop.is_set():
            try:
                pub, who = listener.accept()
            except OSError:
                break
            counter[0] += 1
            cid = counter[0]
            with pending_lock:
                pending[cid] = pub
            try:
                with ctrl_lock:
                    send_json(ctrl, {"op": "open", "id": cid})
            except OSError:
                with pending_lock:
                    pending.pop(cid, None)
                close_quietly(pub)
                break
            # A client that never dials back must not leak the parked socket.
            threading.Timer(30.0, reap, args=(cid,)).start()

    def reap(cid):
        with pending_lock:
            sock = pending.pop(cid, None)
        if sock is not None:
            log(f"conn {cid} was never claimed by the client — dropping it")
            close_quietly(sock)

    acceptor = threading.Thread(target=accept_loop, daemon=True)
    acceptor.start()

    # The control connection carries nothing else from the client, so a read that
    # returns empty IS the disconnect. That is the teardown signal.
    try:
        while True:
            if not ctrl.recv(1):
                break
    except OSError:
        pass
    finally:
        stop.set()
        close_quietly(listener)
        with published_lock:
            published.pop(public_port, None)
        write_state()
        with pending_lock:
            orphans = list(pending.values())
            pending.clear()
        for sock in orphans:
            close_quietly(sock)
        log(f"unpublished {bind_addr}:{public_port} (client {addr[0]} disconnected)")

def handle(sock, addr):
    sock.settimeout(20)
    line = read_line(sock)
    if line is None:
        close_quietly(sock)
        return
    try:
        req = json.loads(line.decode())
    except ValueError:
        close_quietly(sock)
        return
    if not hmac.compare_digest(str(req.get("token", "")), token()):
        log(f"rejected {req.get('op')} from {addr[0]}: bad token")
        try:
            send_json(sock, {"ok": False, "error": "bad token"})
        except OSError:
            pass
        close_quietly(sock)
        return

    op = req.get("op")
    if op == "register":
        sock.settimeout(None)
        # Keepalive on the control connection, because teardown is driven by
        # noticing that it closed. A client that vanishes WITHOUT closing — a lid
        # shut, Wi-Fi dropped, a laptop carried out of the building — leaves a
        # bare TCP socket that never reports anything, and the host would keep a
        # port published for an absent machine indefinitely.
        sock.setsockopt(socket.SOL_SOCKET, socket.SO_KEEPALIVE, 1)
        for opt, value in (("TCP_KEEPIDLE", 60), ("TCP_KEEPALIVE", 60),
                           ("TCP_KEEPINTVL", 15), ("TCP_KEEPCNT", 4)):
            if hasattr(socket, opt):
                try:
                    sock.setsockopt(socket.IPPROTO_TCP, getattr(socket, opt), value)
                except OSError:
                    pass
        serve_registration(sock, addr, req)
        close_quietly(sock)
    elif op == "data":
        cid = int(req.get("id", 0))
        with pending_lock:
            pub = pending.pop(cid, None)
        if pub is None:
            close_quietly(sock)
            return
        sock.settimeout(None)
        pub.settimeout(None)
        splice(pub, sock)
    else:
        close_quietly(sock)

server = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
server.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
server.bind(("0.0.0.0", port))
server.listen(64)
write_state()
log(f"relay control listening on 0.0.0.0:{port}; published ports bind {bind_addr}")
try:
    while True:
        conn, addr = server.accept()
        threading.Thread(target=handle, args=(conn, addr), daemon=True).start()
except KeyboardInterrupt:
    pass
finally:
    with published_lock:
        published.clear()
    write_state()
PYEOF
}

# cmd_expose — the client half. Foreground on purpose, like `ssh -N -R`: the
# exposure lasts exactly as long as you can see it running, so a laptop that
# closes its lid stops publishing instead of leaving a port open on the host.
cmd_expose() {
  local local_port="" public_port="" host="${CLIENT_HOST:-}" relay_port="$RELAY_PORT" token=""
  while [[ $# -gt 0 ]]; do
    case "$1" in
      --as)    public_port="$2"; shift 2 ;;
      --host)  host="$2"; shift 2 ;;
      --port)  relay_port="$2"; shift 2 ;;
      --token) token="$2"; shift 2 ;;
      -h|--help)
        echo "Usage: ferry expose <local-port> [--as PUBLIC] [--host H] [--port P] [--token T]"
        echo "  <local-port>  the port on THIS machine to publish (dialled as 127.0.0.1)"
        echo "  --as PUBLIC   the port it is served from on the host [default: same number]"
        echo "  --host H      the ferry host [default: the client profile's host]"
        echo "  --port P      the host's relay control port [default: $RELAY_PORT]"
        echo "  --token T     the relay token ('ferry relay --token' on the host); saved to"
        echo "                  ~/.config/ferry/relay-token after a first successful run"
        echo "Runs in the foreground — Ctrl-C stops publishing."
        return 0 ;;
      -*) echo "Unknown option for 'ferry expose': $1"; exit 1 ;;
      *)  if [[ -z "$local_port" ]]; then local_port="$1"; shift
          else echo "Unexpected argument: $1"; exit 1; fi ;;
    esac
  done

  if [[ -z "$local_port" ]]; then
    echo "Usage: ferry expose <local-port> [--as PUBLIC] [--host H] [--port P] [--token T]"
    exit 1
  fi
  public_port="${public_port:-$local_port}"

  if [[ -z "$host" ]]; then
    echo "Error: no host. Pass --host <mdns-or-ip>, or bootstrap this machine first"
    echo "       (a client profile at ~/.config/ferry/client.json supplies one)."
    exit 1
  fi

  # Token precedence: the flag, the environment, then whatever a previous run saved.
  [[ -z "$token" ]] && token="${FERRY_RELAY_TOKEN:-}"
  if [[ -z "$token" && -f "$RELAY_TOKEN_FILE" ]]; then
    token="$(cat "$RELAY_TOKEN_FILE")"
  fi
  if [[ -z "$token" ]]; then
    echo "Error: no relay token. Run 'ferry relay --token' on the host, then:"
    echo "       ferry expose $local_port --as $public_port --token <token>"
    exit 1
  fi

  echo ">>> Publishing 127.0.0.1:$local_port  ->  $host:$public_port"
  echo "    (through the relay control port $host:$relay_port — this machine only dials OUT)"
  echo "    Ctrl-C to stop."
  if [[ "${kind:-tcp}" == "vnc" ]]; then
    echo "    Native client: vnc://$host:$public_port"
    echo "    Browser:       http://$host:${VNC_PORT:-}/vnc/$public_port   (host runs 'ferry serve-vnc')"
  fi
  # exec, not a child: the tunnel's lifetime IS this process's lifetime. Run the
  # python as a child and `kill <the pid you started>` kills only the zsh wrapper,
  # leaving the control connection open and the host still publishing a port whose
  # client is gone. Interactive Ctrl-C signals the whole process group and papers
  # over that; anything supervising ferry by pid does not.
  exec python3 - "$host" "$relay_port" "$local_port" "$public_port" "$token" "$RELAY_TOKEN_FILE" "$(hostname -s 2>/dev/null || echo client)" "${kind:-tcp}" <<'PYEOF'
import json, os, socket, sys, threading, time

host, relay_port, local_port, public_port, token, token_file, label, kind = sys.argv[1:9]
relay_port, local_port, public_port = int(relay_port), int(local_port), int(public_port)

def read_line(sock, limit=4096):
    """One byte at a time — see the note in the relay half."""
    buf = bytearray()
    while len(buf) < limit:
        b = sock.recv(1)
        if not b:
            return None
        if b == b"\n":
            return bytes(buf)
        buf += b
    return None

def send_json(sock, obj):
    sock.sendall((json.dumps(obj) + "\n").encode())

def pump(src, dst):
    try:
        while True:
            chunk = src.recv(65536)
            if not chunk:
                break
            dst.sendall(chunk)
    except OSError:
        pass
    finally:
        try:
            dst.shutdown(socket.SHUT_WR)
        except OSError:
            pass

def splice(a, b):
    t = threading.Thread(target=pump, args=(a, b), daemon=True)
    t.start()
    pump(b, a)
    t.join(timeout=5)
    for s in (a, b):
        try:
            s.close()
        except OSError:
            pass

def serve_one(cid):
    """One inbound public connection: dial our own service, dial the relay, splice."""
    try:
        local = socket.create_connection(("127.0.0.1", local_port), timeout=5)
    except OSError as e:
        print(f"    local service refused ({e}) — dropping connection {cid}", flush=True)
        # Still claim the parked socket so the relay stops holding it open.
        try:
            data = socket.create_connection((host, relay_port), timeout=10)
            send_json(data, {"op": "data", "token": token, "id": cid})
            data.close()
        except OSError:
            pass
        return
    try:
        data = socket.create_connection((host, relay_port), timeout=10)
        send_json(data, {"op": "data", "token": token, "id": cid})
    except OSError as e:
        print(f"    could not open a data channel ({e})", flush=True)
        local.close()
        return
    splice(local, data)

try:
    ctrl = socket.create_connection((host, relay_port), timeout=10)
except OSError as e:
    print(f"Error: cannot reach the relay at {host}:{relay_port} ({e})")
    print("       Is 'ferry relay' running on the host?")
    sys.exit(1)

send_json(ctrl, {"op": "register", "token": token, "public_port": public_port, "label": label, "kind": kind})
reply = read_line(ctrl)
if reply is None:
    print("Error: the relay closed the connection during registration.")
    sys.exit(1)
try:
    resp = json.loads(reply.decode())
except ValueError:
    print("Error: unreadable reply from the relay.")
    sys.exit(1)
if not resp.get("ok"):
    print(f"Error: the relay refused the registration: {resp.get('error', 'unknown reason')}")
    sys.exit(1)

# Only now is the token known-good, so only now is it worth keeping.
try:
    os.makedirs(os.path.dirname(token_file), exist_ok=True)
    if not os.path.exists(token_file):
        with open(token_file, "w") as f:
            f.write(token + "\n")
        os.chmod(token_file, 0o600)
except OSError:
    pass

print(f"    Published. Visitors reach it at {resp.get('bind')}:{resp['public_port']} on the host.",
      flush=True)

ctrl.settimeout(None)
try:
    while True:
        line = read_line(ctrl)
        if line is None:
            print("\n>>> The relay closed the tunnel (host stopped, or 'ferry down').")
            sys.exit(1)
        try:
            msg = json.loads(line.decode())
        except ValueError:
            continue
        if msg.get("op") == "open":
            threading.Thread(target=serve_one, args=(int(msg["id"]),), daemon=True).start()
except KeyboardInterrupt:
    print("\n>>> Stopped publishing.")
finally:
    try:
        ctrl.close()
    except OSError:
        pass
PYEOF
}

# rfb_preflight <port> — refuse to publish a port that is not speaking RFB.
# A VNC server greets FIRST ("RFB 003.008\n"), so one read settles it.
rfb_preflight() {
  python3 - "$1" <<'PYEOF'
import socket, sys
port = int(sys.argv[1])
# The connect and the greeting-read are two different failures: a refused/
# unreachable connect means nothing is listening at all, while a connect that
# succeeds but never sends "RFB " (including a read that times out or hits
# EOF) means something IS listening there, just not a VNC server.
try:
    s = socket.create_connection(("127.0.0.1", port), timeout=3)
except OSError as e:
    print(f"Error: nothing is listening on 127.0.0.1:{port} ({e}).")
    print("       Turn on Screen Sharing (macOS: System Settings > General > Sharing)")
    print("       or start your VNC server, then run this again.")
    sys.exit(1)
with s:
    s.settimeout(3)
    try:
        greeting = s.recv(12)
    except OSError:
        greeting = b""
    if not greeting.startswith(b"RFB "):
        print(f"Error: 127.0.0.1:{port} is not a VNC server (no RFB greeting; got {greeting[:12]!r}).")
        sys.exit(1)
    print(f"    VNC server on 127.0.0.1:{port} greets {greeting.strip().decode(errors='replace')}")
PYEOF
}

# cmd_expose_vnc — `ferry expose 5900` with a preflight and a kind tag, so the
# host can list it as a screen and serve the browser viewer for it.
cmd_expose_vnc() {
  local local_port="5900" passthrough=()
  while [[ $# -gt 0 ]]; do
    case "$1" in
      --local) local_port="$2"; shift 2 ;;
      -h|--help)
        echo "Usage: ferry expose-vnc [--local PORT] [--as PUBLIC] [--host H] [--port P] [--token T]"
        echo "  --local PORT  the VNC server on THIS machine [default: 5900]"
        echo "  (every other flag is passed to 'ferry expose'; see 'ferry expose --help')"
        echo "Runs in the foreground — Ctrl-C stops publishing."
        return 0 ;;
      *) passthrough+=("$1"); shift ;;
    esac
  done
  if [[ -n "${local_port//[0-9]/}" || -z "$local_port" ]]; then
    echo "Error: --local must be a port number (got '$local_port')"
    exit 1
  fi
  rfb_preflight "$local_port" || exit 1
  local kind="vnc"
  cmd_expose "$local_port" "${passthrough[@]}"
}
# ferry-vnc.zsh — the browser half of VNC-through-ferry.
#
# `ferry expose-vnc` (client) publishes a laptop's VNC server through the relay
# as an ordinary TCP port tagged kind=vnc. This module gives that port a URL:
# `ferry serve-vnc` (host) serves noVNC and bridges WebSocket <-> TCP onto
# 127.0.0.1:<published port>. It bridges ONLY ports the relay state file lists
# as kind=vnc at the moment of the request, so it is not a general ws->tcp hole.
#
# What this is NOT: auth. The VNC server's own password gates the screen; this
# is plain HTTP on the LAN, exactly like the relay's published ports.

cmd_serve_vnc() {
  if (( CLIENT_MODE )); then
    echo "Error: Command 'ferry serve-vnc' is only available on the LLM-Ferry Host Mac."
    echo "       The client side is 'ferry expose-vnc'."
    exit 1
  fi
  local vnc_port="$VNC_PORT" bind_addr="0.0.0.0" foreground=0 fetch_only=0 marker="ferry-vnc-marker"
  while [[ $# -gt 0 ]]; do
    case "$1" in
      --port)       vnc_port="$2"; shift 2 ;;
      --bind)       bind_addr="$2"; shift 2 ;;
      --foreground) foreground=1; shift ;;
      --fetch)      fetch_only=1; shift ;;
      --marker)     marker="$2"; shift 2 ;;
      -h|--help)
        echo "Usage: ferry serve-vnc [--port P] [--bind ADDR] [--foreground] [--fetch]"
        echo "  --port P      the viewer port [default: $VNC_PORT]"
        echo "  --bind ADDR   0.0.0.0 (the LAN, default) or 127.0.0.1 (this host only)"
        echo "  --fetch       download noVNC $NOVNC_VERSION into $NOVNC_DIR and exit"
        return 0 ;;
      *) echo "Unknown option for 'ferry serve-vnc': $1"; exit 1 ;;
    esac
  done

  if (( fetch_only )); then
    novnc_fetch || exit 1
    return 0
  fi
  if [[ ! -f "$NOVNC_DIR/VERSION" || "$(cat "$NOVNC_DIR/VERSION")" != "$NOVNC_VERSION" || ! -f "$NOVNC_DIR/vnc.html" ]]; then
    echo "Error: noVNC $NOVNC_VERSION is not installed under $NOVNC_DIR."
    echo "       Run: ferry serve-vnc --fetch   (downloads it once from GitHub)"
    exit 1
  fi

  if (( ! foreground )); then
    if lsof -nP -iTCP:"$vnc_port" -sTCP:LISTEN >/dev/null 2>&1; then
      echo "Error: port $vnc_port is already in use — a viewer may already be running."
      exit 1
    fi
    echo ">>> Starting the browser VNC viewer on port $vnc_port (bind $bind_addr)"
    echo "    Open: http://$MDNS_NAME:$vnc_port  (or http://$LAN_IP:$vnc_port)"
    echo "    Only ports published with 'ferry expose-vnc' are bridged. Stop with: ferry down"
    nohup "$FERRY_BIN_PATH" serve-vnc --foreground --port "$vnc_port" --bind "$bind_addr" \
      --marker ferry-vnc-marker > "$VNC_LOG" 2>&1 & disown
    sleep 1
    if lsof -nP -iTCP:"$vnc_port" -sTCP:LISTEN >/dev/null 2>&1; then
      echo ">>> Viewer is listening on port $vnc_port."
    else
      echo "    WARNING: the viewer is not listening. See $VNC_LOG"
    fi
    return 0
  fi

  # exec, not a child: `pkill -f ferry-vnc-marker` must hit the python that holds
  # the port, not a zsh wrapper in front of it (the relay learned this the hard way).
  exec python3 - "$vnc_port" "$bind_addr" "$RELAY_STATE_FILE" "$NOVNC_DIR" "$marker" <<'PYEOF'
import base64, hashlib, html, json, os, socket, struct, sys, threading, time
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from urllib.parse import unquote

PORT, BIND, STATE_FILE, NOVNC_DIR = int(sys.argv[1]), sys.argv[2], sys.argv[3], os.path.realpath(sys.argv[4])
GUID = "258EAFA5-E914-47DA-95CA-C5AB0DC85B11"
MAX_FRAME = 1 << 20   # a client-declared frame length is untrusted input; RFB never needs more
TYPES = {".html": "text/html; charset=utf-8", ".js": "text/javascript", ".css": "text/css",
         ".json": "application/json", ".svg": "image/svg+xml", ".png": "image/png",
         ".ico": "image/x-icon", ".mp3": "audio/mpeg", ".oga": "audio/ogg",
         ".woff": "font/woff", ".woff2": "font/woff2", ".ttf": "font/ttf"}

def log(msg):
    print(f"[{time.strftime('%Y-%m-%d %H:%M:%S')}] {msg}", flush=True)

def published_vnc():
    """{port: info} for every relay entry tagged vnc, read fresh on every call."""
    try:
        with open(STATE_FILE) as f:
            pub = json.load(f)
        return {int(p): i for p, i in pub.items() if isinstance(i, dict) and i.get("kind") == "vnc"}
    except (OSError, ValueError, AttributeError, TypeError):
        return {}

def index_html():
    rows = []
    for port, info in sorted(published_vnc().items()):
        rows.append(f'<li><a href="/vnc/{port}">{html.escape(str(info.get("label") or "screen"))}</a>'
                    f' &mdash; {html.escape(str(info.get("client", "?")))}:{port},'
                    f' since {html.escape(str(info.get("since", "?")))}</li>')
    body = "<ul>" + "".join(rows) + "</ul>" if rows else "<p>Nothing published. On the laptop: <code>ferry expose-vnc</code></p>"
    return ("<!doctype html><meta charset=utf-8><title>ferry screens</title>"
            "<h1>Screens published through this host</h1>" + body).encode()

def accept_key(key):
    return base64.b64encode(hashlib.sha1((key + GUID).encode()).digest()).decode()

def frame(payload, opcode=0x2):
    n = len(payload)
    head = bytes([0x80 | opcode])
    if n < 126:
        head += bytes([n])
    elif n < 65536:
        head += bytes([126]) + struct.pack("!H", n)
    else:
        head += bytes([127]) + struct.pack("!Q", n)
    return head + payload

def unmask(data, mask):
    if not data:
        return data
    m = (mask * (len(data) // 4 + 1))[:len(data)]
    return (int.from_bytes(data, "big") ^ int.from_bytes(m, "big")).to_bytes(len(data), "big")

def read_frame(rfile):
    """(opcode, payload, masked), or None at EOF / on a short read / on an oversized
    frame. Reads through the handler's buffered rfile so bytes that arrived with the
    request headers are not lost. EVERY read is length-checked: a client that vanishes
    mid-header would otherwise hand struct.unpack a short buffer, and struct.error is
    not an OSError, so it would escape the handler and dump a traceback into the log on
    a routine abrupt disconnect."""
    head = rfile.read(2)
    if len(head) < 2:
        return None
    opcode, n = head[0] & 0x0F, head[1] & 0x7F
    masked = bool(head[1] & 0x80)
    if n == 126:
        ext = rfile.read(2)
        if len(ext) < 2:
            return None
        n = struct.unpack("!H", ext)[0]
    elif n == 127:
        ext = rfile.read(8)
        if len(ext) < 8:
            return None
        n = struct.unpack("!Q", ext)[0]
    if n > MAX_FRAME:
        return None
    mask = None
    if masked:
        mask = rfile.read(4)
        if len(mask) < 4:
            return None
    data = rfile.read(n) if n else b""
    if len(data) < n:
        return None
    return opcode, (unmask(data, mask) if mask else data), masked

def pump_tcp_to_ws(upstream, ws, closing, lock):
    """TCP -> WS until upstream EOF. `closing` is set by the handler thread once it has
    already sent a close frame, so the shutdown it performs to wake this recv does not
    put a second, spurious close frame on a socket the client is watching for EOF.
    `lock` serialises sendall with the handler thread's pongs and close echo — without
    it a pong can splice into the middle of a large frame's payload."""
    try:
        while True:
            chunk = upstream.recv(65536)
            if not chunk:
                break
            with lock:
                ws.sendall(frame(chunk))
        if not closing.is_set():
            with lock:
                ws.sendall(frame(struct.pack("!H", 1000), 0x8))
    except OSError:
        pass

class Handler(BaseHTTPRequestHandler):
    protocol_version = "HTTP/1.1"
    timeout = 30   # a client that connects and sends nothing must not pin a thread forever;
                   # BaseHTTPRequestHandler closes the request on this timeout by itself.

    def log_message(self, fmt, *args):
        log(f"{self.address_string()} {fmt % args}")

    def reply(self, status, body=b"", ctype="text/html; charset=utf-8", extra=()):
        self.send_response(status)
        self.send_header("Content-Type", ctype)
        self.send_header("Content-Length", str(len(body)))
        for k, v in extra:
            self.send_header(k, v)
        self.end_headers()
        if body:
            self.wfile.write(body)

    def do_GET(self):
        path = self.path.split("?", 1)[0]
        if path == "/":
            return self.reply(200, index_html())
        if path.startswith("/vnc/"):
            port = self.port_of(path[5:])
            if port not in published_vnc():
                return self.reply(404, b"no such screen")
            return self.reply(302, extra=[("Location", f"/novnc/vnc.html?autoconnect=1&resize=scale&path=ws/{port}")])
        if path.startswith("/novnc/"):
            return self.serve_static(unquote(path[7:]))
        if path.startswith("/ws/"):
            return self.serve_ws(self.port_of(path[4:]))
        self.reply(404, b"not found")

    @staticmethod
    def port_of(text):
        # isascii() too: str.isdigit() accepts glyphs like U+00B2 that int() rejects.
        return int(text) if (text.isascii() and text.isdigit()) else -1

    def serve_static(self, rel):
        full = os.path.realpath(os.path.join(NOVNC_DIR, rel))
        if not full.startswith(NOVNC_DIR + os.sep) or not os.path.isfile(full):
            return self.reply(404, b"not found")
        with open(full, "rb") as f:
            data = f.read()
        self.reply(200, data, TYPES.get(os.path.splitext(full)[1], "application/octet-stream"))

    def serve_ws(self, port):
        if port not in published_vnc():
            return self.reply(403, b"that port is not published as a VNC screen")
        origin = self.headers.get("Origin")
        if origin:
            origin_host = origin.split("//", 1)[-1]
            if origin_host != self.headers.get("Host", ""):
                return self.reply(403, b"cross-origin WebSocket refused")
        key = self.headers.get("Sec-WebSocket-Key")
        if self.headers.get("Upgrade", "").lower() != "websocket" or not key:
            return self.reply(400, b"expected a WebSocket upgrade")
        try:
            upstream = socket.create_connection(("127.0.0.1", port), timeout=10)
        except OSError as e:
            log(f"bridge to 127.0.0.1:{port} failed: {e}")
            return self.reply(502, b"the published port did not answer")
        self.send_response(101, "Switching Protocols")
        self.send_header("Upgrade", "websocket")
        self.send_header("Connection", "Upgrade")
        self.send_header("Sec-WebSocket-Accept", accept_key(key))
        if "binary" in [p.strip() for p in self.headers.get("Sec-WebSocket-Protocol", "").split(",")]:
            self.send_header("Sec-WebSocket-Protocol", "binary")
        self.end_headers()
        self.wfile.flush()
        self.connection.settimeout(None)   # a live RFB session must not be killed by
                                            # the header-phase idle timeout above.
        self.close_connection = True
        ws = self.connection
        upstream.settimeout(None)
        closing = threading.Event()
        lock = threading.Lock()   # both threads write to ws; every sendall holds this
        t = threading.Thread(target=pump_tcp_to_ws, args=(upstream, ws, closing, lock), daemon=True)
        t.start()
        log(f"bridge open {self.address_string()} -> 127.0.0.1:{port}")
        try:
            while True:
                got = read_frame(self.rfile)
                if got is None:
                    break
                opcode, data, masked = got
                if not masked:
                    # RFC 6455 6.1: a client frame must be masked; fail the connection.
                    closing.set()
                    try:
                        with lock:
                            ws.sendall(frame(struct.pack("!H", 1002), 0x8))
                    except OSError:
                        pass
                    break
                # FIN/fragmentation is deliberately ignored: this is a byte stream onto RFB.
                if opcode in (0x0, 0x1, 0x2):
                    upstream.sendall(data)
                elif opcode == 0x9:
                    with lock:
                        ws.sendall(frame(data, 0xA))
                elif opcode == 0x8:
                    closing.set()
                    try:
                        with lock:
                            ws.sendall(frame(data[:2], 0x8))
                    except OSError:
                        pass
                    break
                else:
                    log(f"ignoring unknown WebSocket opcode 0x{opcode:x} from {self.address_string()}")
        except OSError:
            pass
        finally:
            closing.set()
            try:
                upstream.shutdown(socket.SHUT_RDWR)
            except OSError:
                pass
            t.join(timeout=5)   # join BEFORE close: closing an fd another thread is
            upstream.close()    # blocked in recv() on is an fd-reuse hazard.
            try:
                ws.close()
            except OSError:
                pass
            log(f"bridge closed 127.0.0.1:{port}")

class Server(ThreadingHTTPServer):
    daemon_threads = True
    allow_reuse_address = True

srv = Server((BIND, PORT), Handler)
log(f"vnc viewer listening on {BIND}:{PORT}; noVNC from {NOVNC_DIR}; state {STATE_FILE}")
try:
    srv.serve_forever()
except KeyboardInterrupt:
    pass
PYEOF
}

# novnc_fetch — one pinned tarball, checksum-verified, only the viewer tree kept.
# Downloaded rather than vendored: ~250 files of someone else's JS do not belong
# in a repo whose clients fetch `ferry` as one script. Re-run to change versions.
novnc_fetch() {
  echo ">>> Fetching noVNC $NOVNC_VERSION -> $NOVNC_DIR"
  python3 - "$NOVNC_URL" "$NOVNC_SHA256" "$NOVNC_DIR" "$NOVNC_VERSION" <<'PYEOF'
import hashlib, os, shutil, sys, tarfile, tempfile, urllib.request
url, want, dest, version = sys.argv[1:5]
prefix = f"noVNC-{version}/"
dirs = ("app", "core", "vendor")


def safe_member(m):
    """True if m's prefix-stripped .name is safe to extract: no absolute path, no
    '..' traversal segment, and either exactly "vnc.html" or under one of the
    app/core/vendor directories. This is the ONLY thing that protects extraction
    on pre-3.12 Python (no `filter=` kwarg, so it falls back below); on 3.12+
    tf.extractall(..., filter="data") is an additional backstop, not the sole
    line of defense — safety must not depend on the interpreter version."""
    rel = m.name
    if os.path.isabs(rel):
        return False
    norm = os.path.normpath(rel)
    if norm == ".." or norm.startswith(".." + os.sep) or any(seg == ".." for seg in norm.split(os.sep)):
        return False
    if not (m.isfile() or m.isdir()):
        return False
    top = norm.split(os.sep, 1)[0]
    return norm == "vnc.html" or top in dirs


tmp = tempfile.mkdtemp(prefix="ferry-novnc-")
try:
    tgz = os.path.join(tmp, "novnc.tar.gz")
    h = hashlib.sha256()
    with urllib.request.urlopen(url, timeout=120) as r, open(tgz, "wb") as f:
        while True:
            chunk = r.read(1 << 20)
            if not chunk:
                break
            h.update(chunk)
            f.write(chunk)
    if h.hexdigest() != want:
        print(f"Error: checksum mismatch for {url}\n       got  {h.hexdigest()}\n       want {want}")
        sys.exit(1)
    out = os.path.join(tmp, "tree")
    os.makedirs(out)
    with tarfile.open(tgz, "r:gz") as tf:
        members = []
        for m in tf.getmembers():
            if not m.name.startswith(prefix):
                continue
            m.name = m.name[len(prefix):]
            if not safe_member(m):
                continue
            members.append(m)
        try:
            tf.extractall(out, members=members, filter="data")
        except TypeError:                      # python < 3.12 has no filter kwarg
            tf.extractall(out, members=members)
    if not os.path.isfile(os.path.join(out, "vnc.html")):
        print("Error: the tarball has no vnc.html under " + prefix)
        sys.exit(1)
    with open(os.path.join(out, "VERSION"), "w") as f:
        f.write(version + "\n")
    os.makedirs(os.path.dirname(dest), exist_ok=True)
    old = dest + ".old"
    if os.path.isdir(old):                     # a stale leftover from a crash mid-swap
        shutil.rmtree(old)
    if os.path.isdir(dest):
        os.rename(dest, old)                    # keep the previous good install until
    shutil.move(out, dest)                      # the new one is fully in place
    if os.path.isdir(old):
        shutil.rmtree(old)
    print(f"    noVNC {version} installed ({sum(len(fs) for _, _, fs in os.walk(dest))} files).")
except (OSError, ValueError, tarfile.TarError) as e:
    print(f"Error: fetching noVNC failed: {e}")
    sys.exit(1)
finally:
    shutil.rmtree(tmp, ignore_errors=True)
PYEOF
}
# ----------------- FERRY TRANSFER COMMANDS -----------------

# Resolve the LAN host to talk to for client-side pull/get:
#   --host flag (already captured into $1) > $CLIENT_HOST (client mode) > error.
resolve_ferry_host() {
  local h="$1"
  if [[ -n "$h" ]]; then echo "$h"; return 0; fi
  if [[ -n "${CLIENT_HOST:-}" ]]; then echo "$CLIENT_HOST"; return 0; fi
  return 1
}

# Resolve the share-server port for client-side pull/get:
#   --port flag ($1) > $CLIENT_SHARE_PORT > $SHARE_PORT > 8095.
resolve_ferry_port() {
  local p="$1"
  if [[ -n "$p" ]]; then echo "$p"; return; fi
  if [[ -n "${CLIENT_SHARE_PORT:-}" ]]; then echo "$CLIENT_SHARE_PORT"; return; fi
  if [[ -n "${SHARE_PORT:-}" ]]; then echo "$SHARE_PORT"; return; fi
  echo "8095"
}

cmd_offer() {
  if (( CLIENT_MODE )); then
    echo "Error: Command 'ferry offer' is only available on the LLM-Ferry Host Mac."
    exit 1
  fi
  if [[ $# -lt 1 ]]; then
    echo "Usage: ferry offer <path>..."
    exit 1
  fi
  local cfg_dir="$HOME/.config/ferry"
  local offered="$cfg_dir/offered.json"
  mkdir -p "$cfg_dir"

  # Merge the given paths (basename -> absolute path) into offered.json via python for safe JSON.
  python3 - "$offered" "$@" <<'PYEOF'
import json, os, sys
offered_path = sys.argv[1]
paths = sys.argv[2:]
data = {}
if os.path.exists(offered_path):
    try:
        data = json.load(open(offered_path))
    except Exception:
        data = {}
for p in paths:
    ap = os.path.abspath(os.path.expanduser(p))
    if not os.path.exists(ap):
        print(f"    WARNING: path not found, skipping: {p}")
        continue
    name = os.path.basename(ap.rstrip("/"))
    data[name] = ap
    print(f"    Offered: {name}  ->  {ap}")
json.dump(data, open(offered_path, "w"), indent=2)
print(f">>> Offered manifest saved: {offered_path}")
PYEOF
}

cmd_pull() {
  if [[ $# -lt 1 || "$1" == --* ]]; then
    echo "Usage: ferry pull <model-id> [--host H] [--port P] [--transport http|hf|nc] [--to DIR]"
    exit 1
  fi
  local model_id="$1"; shift
  local host="" port="" transport="http" to=""
  while [[ $# -gt 0 ]]; do
    case "$1" in
      --host)      host="$2"; shift 2 ;;
      --port)      port="$2"; shift 2 ;;
      --transport) transport="$2"; shift 2 ;;
      --to)        to="$2"; shift 2 ;;
      *)           echo "Unknown option: $1"; exit 1 ;;
    esac
  done

  case "$transport" in
    http)
      local h; h=$(resolve_ferry_host "$host") || {
        echo "Error: no host resolved. Pass --host <hostname-or-ip> (or run on a configured client)."; exit 1; }
      local p; p=$(resolve_ferry_port "$port")
      local dest="${to:-$HOME/.cache/ferry/models/$model_id}"
      mkdir -p "$dest"
      echo ">>> Pulling model '$model_id' from http://$h:$p over HTTP (tar stream)..."
      if curl -fsS "http://$h:$p/pull/$model_id" | tar -x -C "$dest"; then
        echo ">>> Model landed at: $dest"
      else
        echo "Error: pull failed. Is the model in the host's HF cache, and is 'ferry share' running?"
        exit 1
      fi
      ;;
    hf)
      local h; h=$(resolve_ferry_host "$host") || {
        echo "Error: no host resolved. Pass --host <hostname-or-ip>."; exit 1; }
      local hfp="${port:-$HF_PORT}"
      echo ">>> [EXPERIMENTAL] Pulling '$model_id' THROUGH host HF proxy at http://$h:$hfp ..."
      echo "    (Requires the host to be running: ferry serve-hf)"
      if command -v hf >/dev/null 2>&1; then
        HF_ENDPOINT="http://$h:$hfp" hf download "$model_id"
      else
        echo "    'hf' not found; falling back to 'uv run huggingface-cli'..."
        HF_ENDPOINT="http://$h:$hfp" uv run huggingface-cli download "$model_id"
      fi
      ;;
    nc)
      echo ">>> Pull via netcat: this laptop will LISTEN and receive a tar stream."
      echo "    On the HOST, run:  ferry send <path-to-model-dir> <this-laptop-host-or-ip> --port ${port:-9099}"
      local recv_args=()
      [[ -n "$port" ]] && recv_args+=(--port "$port")
      [[ -n "$to" ]]   && recv_args+=(--to "$to")
      cmd_receive "${recv_args[@]}"
      ;;
    *)
      echo "Unknown transport: $transport (use http|hf|nc)"
      exit 1
      ;;
  esac
}

cmd_get() {
  if [[ $# -lt 1 || "$1" == --* ]]; then
    echo "Usage: ferry get <name> [--host H] [--port P] [--to DIR]"
    exit 1
  fi
  local name="$1"; shift
  local host="" port="" to=""
  while [[ $# -gt 0 ]]; do
    case "$1" in
      --host) host="$2"; shift 2 ;;
      --port) port="$2"; shift 2 ;;
      --to)   to="$2"; shift 2 ;;
      *)      echo "Unknown option: $1"; exit 1 ;;
    esac
  done
  local h; h=$(resolve_ferry_host "$host") || {
    echo "Error: no host resolved. Pass --host <hostname-or-ip>."; exit 1; }
  local p; p=$(resolve_ferry_port "$port")
  local dest="${to:-.}"
  mkdir -p "$dest"
  echo ">>> Fetching offered file '$name' from http://$h:$p into $dest ..."
  if curl -fsS "http://$h:$p/file/$name" | tar -x -C "$dest"; then
    echo ">>> Landed under: $dest"
    ls -la "$dest"
  else
    echo "Error: fetch failed. Is '$name' offered on the host (see 'ferry offer')?"
    exit 1
  fi
}

cmd_receive() {
  local port="" to=""
  while [[ $# -gt 0 ]]; do
    case "$1" in
      --port) port="$2"; shift 2 ;;
      --to)   to="$2"; shift 2 ;;
      *)      echo "Unknown option: $1"; exit 1 ;;
    esac
  done
  local rcv_port="${port:-9099}"
  local dest="${to:-.}"
  mkdir -p "$dest"
  echo ">>> Receiving: listening on port $rcv_port; extracting into $dest"
  echo "    On the HOST run:  ferry send <file|dir> <this-laptop-host-or-ip> --port $rcv_port"
  # BSD/macOS netcat: `nc -l PORT` listens; the stream is piped straight into tar.
  # `-d` stops nc from reading stdin — without it, a backgrounded listener whose
  # stdin has already reached EOF tears the connection down before the tar payload
  # finishes transferring (Apple nc reads stdin and closes the socket on its EOF).
  # openbsd-nc (the Ubuntu default) has no `-d` flag; plain `nc -l` is correct there,
  # and ncat (nmap) is preferred when present for cross-platform consistency.
  if (( IS_MAC )); then
    nc -d -l "$rcv_port" | tar -x -C "$dest"
  elif command -v ncat >/dev/null 2>&1; then
    ncat -l "$rcv_port" | tar -x -C "$dest"
  else
    nc -l "$rcv_port" | tar -x -C "$dest"
  fi
  echo ">>> Receive complete. Files are in: $dest"
}

cmd_send() {
  if (( CLIENT_MODE )); then
    echo "Error: Command 'ferry send' is only available on the LLM-Ferry Host Mac."
    exit 1
  fi
  if [[ $# -lt 2 || "$1" == --* || "$2" == --* ]]; then
    echo "Usage: ferry send <file|dir> <client-host> [--port P]"
    exit 1
  fi
  local src="$1"; local client_host="$2"; shift 2
  local port=""
  while [[ $# -gt 0 ]]; do
    case "$1" in
      --port) port="$2"; shift 2 ;;
      *)      echo "Unknown option: $1"; exit 1 ;;
    esac
  done
  local send_port="${port:-9099}"
  if [[ ! -e "$src" ]]; then
    echo "Error: path not found: $src"
    exit 1
  fi
  # zsh modifiers: :A = absolute path, :h = head (dirname), :t = tail (basename).
  # Using them avoids forking dirname/basename and works for relative paths too.
  # NB: do NOT name this var 'path' — in zsh that is the array tied to $PATH.
  local parent base
  parent="${src:A:h}"
  base="${src:t}"
  echo ">>> Sending '$base' to $client_host:$send_port via netcat (tar stream)..."
  echo "    (The client must already be running: ferry receive --port $send_port)"
  # macOS/BSD nc closes on its own once stdin (the tar) ends. openbsd-nc (Ubuntu)
  # keeps the socket half-open on stdin EOF, so add `-N` to half-close and let the
  # receiver's tar finish.
  if (( IS_MAC )); then
    tar -c -C "$parent" "$base" | nc "$client_host" "$send_port"
  else
    tar -c -C "$parent" "$base" | nc -N "$client_host" "$send_port"
  fi
  echo ">>> Sent: $base"
}


# ----------------- ENCRYPTED OFF-LAN DROP -----------------
#
# `ferry drop` / `ferry pickup` — the only ferry transport that survives an
# UNTRUSTED channel.
#
# Every other transport in this CLI assumes the private LAN and says so
# (README "Client<->host traffic is plain HTTP on your private network").
# That posture is deliberate and unchanged. This pair covers the case the
# posture cannot: getting a file to a machine that is not on the LAN at all.
#
# The model is client-side encryption plus a DUMB carrier: `drop` writes a
# self-contained blob, you move it by whatever channel already exists (email,
# chat, a gist, object storage, a USB stick), and `pickup` decrypts it. Ferry
# supplies confidentiality, never delivery — which is why there is no account,
# no credential file, and no network code here.
#
# WHY openssl, in a CLI whose badge says "zsh + python3 stdlib": python's
# standard library has PBKDF2 (hashlib) but NO AES, in any module. The choice
# was openssl or hand-rolling a cipher. ferry already shells out to `curl` and
# `nc` for exactly this class of OS-provided binary. Verified 2026-08-30 that
# stock macOS LibreSSL 3.3.6 and Homebrew OpenSSL 3.6.3 produce mutually
# decryptable blobs, and that LibreSSL genuinely honours `-iter` rather than
# accepting and ignoring it (decrypting a 600k blob with `-iter 1` exits 1).
# That second check is the one that matters: LibreSSL historically had no
# `-pbkdf2` at all, and a silently-ignored `-iter` would have produced blobs
# that fail across machines for no visible reason.

FERRYDROP_MAGIC="FERRYDROP/1"
FERRYDROP_ITER="600000"
FERRYDROP_EXT=".ferrydrop"

# Exit codes are distinct so a script can tell "wrong passphrase" from
# "someone modified the blob" — those mean very different things.
FERRYDROP_RC_USAGE=2
FERRYDROP_RC_TAMPER=3
FERRYDROP_RC_BADPASS=4
FERRYDROP_RC_NOOPENSSL=5

_ferry_drop_need_openssl() {
  if ! command -v openssl >/dev/null 2>&1; then
    echo "Error: 'openssl' not found on PATH." >&2
    echo "  ferry drop/pickup need it for AES-256-CBC; python3's standard library has no cipher." >&2
    echo "  macOS ships one at /usr/bin/openssl; on Debian/Ubuntu: apt install openssl" >&2
    exit $FERRYDROP_RC_NOOPENSSL
  fi
}

# A fresh passphrase per drop, ~100+ bits from openssl's CSPRNG.
#
# Hyphen-grouped for legibility, and the hyphens are PART OF THE SECRET — the
# string printed is exactly the string to type back, with nothing to reassemble.
# A five-word diceware form was considered and rejected: it needs a wordlist
# embedded in a single-file CLI that clients fetch over the wire, and buys ~50
# bits where this buys twice that.
_ferry_drop_genpass() {
  openssl rand -base64 24 | tr -d '/+=\n' | cut -c1-24 | sed 's/..../&-/g; s/-$//'
}

cmd_drop() {
  local src="" msg="" out="" have_msg=0 passfile_in=""
  while [[ $# -gt 0 ]]; do
    case "$1" in
      --msg)   msg="$2"; have_msg=1; shift 2 ;;
      --to)    out="$2"; shift 2 ;;
      # Supply your own passphrase instead of a generated one. For scripted
      # drops and for the test suite; a human should prefer the generated one,
      # which is drawn from openssl's CSPRNG rather than from a human's idea of
      # what looks random.
      --pass-file) passfile_in="$2"; shift 2 ;;
      -h|--help)
        echo "Usage: ferry drop <file>|-  [--to PATH] [--pass-file FILE]"
        echo "       ferry drop --msg \"text\" [--to PATH]"
        return 0 ;;
      -)       src="-"; shift ;;
      -*)      echo "Unknown option for 'ferry drop': $1" >&2; exit $FERRYDROP_RC_USAGE ;;
      *)       src="$1"; shift ;;
    esac
  done

  if (( have_msg )) && [[ -n "$src" ]]; then
    echo "Error: give either a file or --msg, not both." >&2; exit $FERRYDROP_RC_USAGE
  fi
  if (( ! have_msg )) && [[ -z "$src" ]]; then
    echo "Usage: ferry drop <file>|-  |  ferry drop --msg \"text\"   (see --help)" >&2
    exit $FERRYDROP_RC_USAGE
  fi
  if [[ -n "$src" && "$src" != "-" && ! -f "$src" ]]; then
    echo "Error: no such file: $src" >&2; exit $FERRYDROP_RC_USAGE
  fi

  _ferry_drop_need_openssl

  local work; work="$(mktemp -d "${TMPDIR:-/tmp}/ferry-drop.XXXXXX")"
  # The passphrase lives in this directory. Remove it on ANY exit path.
  trap "rm -rf '$work'" EXIT INT TERM

  # --- materialise the plaintext, and decide what to call it ---
  local kind name
  if (( have_msg )); then
    kind="msg"; name="message.txt"
    printf '%s' "$msg" > "$work/pt"
  elif [[ "$src" == "-" ]]; then
    kind="msg"; name="stdin.txt"
    cat > "$work/pt"
  else
    kind="file"; name="${src:t}"
    cp "$src" "$work/pt"
  fi

  # --- passphrase to a 0600 file, never to argv ---
  #
  # `-pass pass:<secret>` puts the secret in the process table, where any user
  # on the box can read it with ps. `file:` is the only safe form here.
  local passfile="$work/pass" generated=1
  if [[ -n "$passfile_in" ]]; then
    [[ -f "$passfile_in" ]] || { echo "Error: no such pass file: $passfile_in" >&2; exit $FERRYDROP_RC_USAGE; }
    generated=0
    ( umask 077; head -1 "$passfile_in" | tr -d '\n' > "$passfile" )
    [[ -s "$passfile" ]] || { echo "Error: pass file is empty: $passfile_in" >&2; exit $FERRYDROP_RC_USAGE; }
  else
    ( umask 077; _ferry_drop_genpass > "$passfile" )
  fi

  if ! openssl enc -aes-256-cbc -pbkdf2 -iter "$FERRYDROP_ITER" -salt -a \
        -in "$work/pt" -out "$work/ct.b64" -pass "file:$passfile" 2>"$work/err"; then
    echo "Error: encryption failed." >&2; sed 's/^/    /' "$work/err" >&2
    exit 1
  fi

  [[ -z "$out" ]] && out="${name}${FERRYDROP_EXT}"

  python3 - "$work/ct.b64" "$passfile" "$out" "$kind" "$name" "$FERRYDROP_MAGIC" "$FERRYDROP_ITER" <<'PYEOF'
import sys
sys.path.insert(0, "")
ct_path, pass_path, out_path, kind, name, magic, iters = sys.argv[1:8]

import hashlib, hmac

with open(ct_path) as f:
    ct = f.read()
with open(pass_path, "rb") as f:
    passphrase = f.read().strip()

# The MAC key is derived with a DIFFERENT salt from the one openssl used for the
# encryption key, so the two keys are independent even though one passphrase
# produced both.
mac_key = hashlib.pbkdf2_hmac("sha256", passphrase, b"ferrydrop-mac-v1", int(iters), dklen=32)

# The MAC covers the HEADER as well as the ciphertext. Authenticating the
# ciphertext alone would leave `name:` attacker-controlled, and `name` picks the
# output path on the receiving side — `name: ../../../.ssh/authorized_keys`
# would be a write-anywhere primitive on a blob the recipient can decrypt.
header = [
    magic,
    "cipher: aes-256-cbc",
    "kdf: pbkdf2",
    f"iter: {iters}",
    f"kind: {kind}",
    f"name: {name}",
]
signed = ("\n".join(header) + "\n--\n" + ct).encode()
mac = hmac.new(mac_key, signed, hashlib.sha256).hexdigest()

with open(out_path, "w") as f:
    f.write("\n".join(header) + f"\nmac: {mac}\n--\n" + ct)
PYEOF
  local rc=$?
  if (( rc != 0 )); then echo "Error: could not write the blob." >&2; exit 1; fi

  local size; size="$(wc -c < "$out" | tr -d ' ')"

  # Colour only for a human. The passphrase is a VALUE the operator has to carry
  # to another machine, so when stdout is redirected it must come out as plain
  # text a script can read — wrapping a secret in escape codes makes
  # `ferry drop f | grep passphrase` return something subtly wrong rather than
  # something that obviously failed.
  local g="" y="" z=""
  if [[ -t 1 ]]; then g=$'\033[1;32m'; y=$'\033[1;33m'; z=$'\033[0m'; fi

  echo ""
  echo "  ${g}${out}${z}  (${size} bytes, aes-256-cbc + pbkdf2@${FERRYDROP_ITER}, hmac-sha256)"
  echo ""
  if (( generated )); then
    echo "  passphrase: ${y}$(cat "$passfile")${z}"
  else
    echo "  passphrase: (the one you supplied — not echoed)"
  fi
  echo ""
  echo "  Send the BLOB and the PASSPHRASE by different channels — the blob is safe"
  echo "  on an untrusted one, the passphrase is the entire security boundary."
  echo "  Pick it up with:  ferry pickup $out"
  echo ""
}

cmd_pickup() {
  local blob="" out="" passfile_in=""
  while [[ $# -gt 0 ]]; do
    case "$1" in
      --to)        out="$2"; shift 2 ;;
      --pass-file) passfile_in="$2"; shift 2 ;;
      -h|--help)
        echo "Usage: ferry pickup <blob> [--to PATH] [--pass-file FILE]"
        return 0 ;;
      -*)          echo "Unknown option for 'ferry pickup': $1" >&2; exit $FERRYDROP_RC_USAGE ;;
      *)           blob="$1"; shift ;;
    esac
  done

  if [[ -z "$blob" ]]; then
    echo "Usage: ferry pickup <blob> [--to PATH] [--pass-file FILE]" >&2
    exit $FERRYDROP_RC_USAGE
  fi
  if [[ ! -f "$blob" ]]; then
    echo "Error: no such blob: $blob" >&2; exit $FERRYDROP_RC_USAGE
  fi

  _ferry_drop_need_openssl

  local work; work="$(mktemp -d "${TMPDIR:-/tmp}/ferry-pickup.XXXXXX")"
  trap "rm -rf '$work'" EXIT INT TERM

  local passfile="$work/pass"
  if [[ -n "$passfile_in" ]]; then
    [[ -f "$passfile_in" ]] || { echo "Error: no such pass file: $passfile_in" >&2; exit $FERRYDROP_RC_USAGE; }
    ( umask 077; head -1 "$passfile_in" | tr -d '\n' > "$passfile" )
  else
    local typed
    printf "  passphrase: " >&2
    read -rs typed
    printf "\n" >&2
    ( umask 077; printf '%s' "$typed" > "$passfile" )
    unset typed
  fi

  # --- verify BEFORE decrypting ---
  #
  # A tampered blob must fail closed without ever reaching openssl. The MAC is
  # checked here, in python, and the ciphertext is only written out for the
  # decrypt step once that check has passed.
  python3 - "$blob" "$passfile" "$work" "$FERRYDROP_MAGIC" <<'PYEOF'
import sys
blob_path, pass_path, work, magic = sys.argv[1:5]

import hashlib, hmac, os

RC_TAMPER = 3

with open(blob_path) as f:
    raw = f.read()

if "\n--\n" not in raw:
    sys.stderr.write("Error: not a ferry drop blob (no header separator).\n")
    sys.exit(RC_TAMPER)

head_text, ct = raw.split("\n--\n", 1)
lines = head_text.split("\n")

if not lines or lines[0].strip() != magic:
    got = lines[0].strip() if lines else "<empty>"
    sys.stderr.write(f"Error: unsupported blob format {got!r}; this ferry understands {magic}.\n")
    sys.stderr.write("  A newer ferry wrote it — update this machine with `ferry update`.\n")
    sys.exit(RC_TAMPER)

fields, order = {}, []
for ln in lines[1:]:
    if not ln.strip():
        continue
    if ": " not in ln:
        sys.stderr.write("Error: malformed header line in blob.\n")
        sys.exit(RC_TAMPER)
    k, v = ln.split(": ", 1)
    fields[k] = v
    order.append(k)

for required in ("cipher", "kdf", "iter", "kind", "name", "mac"):
    if required not in fields:
        sys.stderr.write(f"Error: blob header is missing '{required}'.\n")
        sys.exit(RC_TAMPER)

try:
    iters = int(fields["iter"])
except ValueError:
    sys.stderr.write("Error: blob header has a non-numeric 'iter'.\n")
    sys.exit(RC_TAMPER)
# A hostile blob could name an absurd iteration count purely to hang the
# recipient's machine. The MAC cannot help here: it is verified with this very
# number, so the cost is paid before the check can reject anything.
if not (1000 <= iters <= 5_000_000):
    sys.stderr.write(f"Error: refusing an implausible iteration count ({iters}).\n")
    sys.exit(RC_TAMPER)

with open(pass_path, "rb") as f:
    passphrase = f.read().strip()

mac_key = hashlib.pbkdf2_hmac("sha256", passphrase, b"ferrydrop-mac-v1", iters, dklen=32)
signed = ("\n".join([magic] + [f"{k}: {fields[k]}" for k in order if k != "mac"])
          + "\n--\n" + ct).encode()
expect = hmac.new(mac_key, signed, hashlib.sha256).hexdigest()

if not hmac.compare_digest(expect, fields["mac"]):
    sys.stderr.write("Error: authentication failed.\n")
    sys.stderr.write("  Either the passphrase is wrong, or this blob was modified in transit.\n")
    sys.stderr.write("  Nothing was decrypted.\n")
    sys.exit(RC_TAMPER)

# Path safety does NOT ride on the crypto being right. Reduce to a basename
# unconditionally, so even a correctly-signed blob from a sender who has turned
# hostile cannot escape the destination directory.
name = os.path.basename(fields["name"].strip().replace("\\", "/").rstrip("/"))
if name in ("", ".", ".."):
    name = "ferrydrop.out"

with open(os.path.join(work, "ct.b64"), "w") as f:
    f.write(ct)
with open(os.path.join(work, "meta"), "w") as f:
    f.write(f"{fields['kind']}\n{name}\n{iters}\n")
PYEOF
  local rc=$?
  if (( rc != 0 )); then exit $rc; fi

  local kind name iters
  kind="$(sed -n 1p "$work/meta")"
  name="$(sed -n 2p "$work/meta")"
  iters="$(sed -n 3p "$work/meta")"

  if ! openssl enc -d -aes-256-cbc -pbkdf2 -iter "$iters" -a \
        -in "$work/ct.b64" -out "$work/pt" -pass "file:$passfile" 2>"$work/err"; then
    # The MAC already passed, so the passphrase was right and the bytes are
    # intact. Reaching here means something stranger — a cipher mismatch, or a
    # blob written by a build whose parameters differ.
    echo "Error: decryption failed after the MAC verified." >&2
    sed 's/^/    /' "$work/err" >&2
    exit $FERRYDROP_RC_BADPASS
  fi

  if [[ "$kind" == "msg" && -z "$out" ]]; then
    cat "$work/pt"
    return 0
  fi

  local dest="${out:-$name}"
  # Never write THROUGH a symlink planted at the destination.
  if [[ -L "$dest" ]]; then
    echo "Error: $dest is a symlink; refusing to write through it." >&2
    exit $FERRYDROP_RC_USAGE
  fi
  [[ -d "$dest" ]] && dest="$dest/$name"
  cp "$work/pt" "$dest"
  local g="" z=""
  if [[ -t 1 ]]; then g=$'\033[1;32m'; z=$'\033[0m'; fi
  echo "  ${g}${dest}${z}  ($(wc -c < "$dest" | tr -d ' ') bytes, verified)"
}
cmd_serve_hf() {
  if (( CLIENT_MODE )); then
    echo "Error: Command 'ferry serve-hf' is only available on the LLM-Ferry Host Mac."
    exit 1
  fi
  local port=""
  while [[ $# -gt 0 ]]; do
    case "$1" in
      --port) port="$2"; shift 2 ;;
      *)      echo "Unknown option: $1"; exit 1 ;;
    esac
  done
  local hf_port="${port:-$HF_PORT}"
  local hf_log="$LOG_DIR/ferry-hf-$hf_port.log"

  # Free the port if something is already bound there.
  if lsof -nP -iTCP:"$hf_port" -sTCP:LISTEN >/dev/null 2>&1; then
    echo ">>> Port $hf_port already in use. Stopping conflicting listener..."
    lsof -ti tcp:"$hf_port" | xargs kill -9 2>/dev/null || true
    sleep 1
  fi

  echo "================================================================="
  echo "     STARTING HUGGINGFACE PASS-THROUGH PROXY (EXPERIMENTAL)"
  echo "================================================================="
  echo "Proxy port:  $hf_port"
  echo "Clients set: HF_ENDPOINT=http://$MDNS_NAME:$hf_port  (or http://$LAN_IP:$hf_port)"
  echo "Log:         $hf_log"
  echo "================================================================="

  # The trailing "ferry-hf-marker" arg is a stable kill sentinel for `ferry down`;
  # the heredoc only reads argv[1] (the port), so the extra arg is ignored at runtime.
  nohup python3 - "$hf_port" "ferry-hf-marker" <<'PYEOF' > "$hf_log" 2>&1 & disown
import sys, urllib.request, urllib.error
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer

port = int(sys.argv[1])
UPSTREAM = "https://huggingface.co"

class HFProxyHandler(BaseHTTPRequestHandler):
    def _proxy(self, method):
        url = UPSTREAM + self.path
        try:
            req = urllib.request.Request(url, method=method)
            # Forward the headers that matter for auth and ranged/LFS downloads.
            for h in ("Authorization", "User-Agent", "Range", "Accept"):
                v = self.headers.get(h)
                if v:
                    req.add_header(h, v)
            # urllib follows redirects by default — HF redirects LFS blobs to a CDN.
            with urllib.request.urlopen(req, timeout=60) as resp:
                self.send_response(resp.status)
                for k, v in resp.headers.items():
                    if k.lower() in ("transfer-encoding", "connection"):
                        continue
                    self.send_header(k, v)
                self.end_headers()
                if method != "HEAD":
                    while True:
                        chunk = resp.read(65536)
                        if not chunk:
                            break
                        self.wfile.write(chunk)
        except urllib.error.HTTPError as e:
            self.send_response(e.code)
            self.end_headers()
            try:
                if method != "HEAD":
                    self.wfile.write(e.read())
            except Exception:
                pass
        except Exception as e:
            self.send_response(502)
            self.end_headers()
            try:
                self.wfile.write(("proxy error: %s\n" % e).encode())
            except Exception:
                pass

    def do_GET(self):
        self._proxy("GET")

    def do_HEAD(self):
        self._proxy("HEAD")

    def log_message(self, fmt, *args):
        sys.stderr.write("[ferry-hf] " + (fmt % args) + "\n")

ThreadingHTTPServer(('0.0.0.0', port), HFProxyHandler).serve_forever()
PYEOF

  sleep 1
  if lsof -nP -iTCP:"$hf_port" -sTCP:LISTEN >/dev/null 2>&1; then
    echo ">>> HF proxy is listening on port $hf_port. Stop it with: ferry down"
  else
    echo "WARNING: HF proxy did not come up; check the log: $hf_log"
  fi
}

cmd_serve_proxy() {
  if (( CLIENT_MODE )); then
    echo "Error: Command 'ferry serve-proxy' is only available on the LLM-Ferry Host Mac."
    exit 1
  fi
  local port=""
  while [[ $# -gt 0 ]]; do
    case "$1" in
      --port) port="$2"; shift 2 ;;
      *)      echo "Unknown option: $1"; exit 1 ;;
    esac
  done
  local proxy_port="${port:-$PROXY_PORT}"
  local proxy_log="$LOG_DIR/ferry-proxy-$proxy_port.log"

  # Free the port if something is already bound there.
  if lsof -nP -iTCP:"$proxy_port" -sTCP:LISTEN >/dev/null 2>&1; then
    echo ">>> Port $proxy_port already in use. Stopping conflicting listener..."
    lsof -ti tcp:"$proxy_port" | xargs kill -9 2>/dev/null || true
    sleep 1
  fi

  echo "================================================================="
  echo "           STARTING HTTP(S) DOWNLOAD FORWARD PROXY"
  echo "================================================================="
  echo "Proxy port:  $proxy_port"
  echo "Routes:      uv/PyPI, huggingface_hub, sherpa-onnx/GitHub, git, curl — anything honoring proxy env vars"
  echo "Log:         $proxy_log"
  echo "================================================================="

  # A general forward proxy: CONNECT tunneling for HTTPS, plain-HTTP forwarding for GET.
  # The trailing "ferry-proxy-marker" arg is a stable kill sentinel for `ferry down`;
  # the heredoc only reads argv[1] (the port), so the extra arg is ignored at runtime.
  nohup python3 - "$proxy_port" "ferry-proxy-marker" <<'PYEOF' > "$proxy_log" 2>&1 & disown
import sys, select, socket
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer

PORT = int(sys.argv[1])   # sys.argv[2] == "ferry-proxy-marker" (kill sentinel, unused)


class Proxy(BaseHTTPRequestHandler):
    protocol_version = "HTTP/1.1"

    def do_CONNECT(self):
        # Establish a raw TCP tunnel to the requested host:port, then pump bytes
        # both ways until either side closes. This is what HTTPS clients use.
        try:
            host, _, port = self.path.partition(":")
            upstream = socket.create_connection((host, int(port or 443)), timeout=30)
        except Exception:
            self.send_error(502)
            return
        self.send_response(200, "Connection established")
        self.end_headers()
        client = self.connection
        # Both sockets stay BLOCKING. select() tells us which side has bytes to
        # read; sendall() then blocks until the other side has taken them. That
        # blocking is the backpressure: a fast upstream (a CDN pushing a
        # multi-GB model file) cannot outrun a slower LAN client. With
        # non-blocking sockets sendall() raises BlockingIOError (EAGAIN) the
        # moment the client's receive buffer fills, and the tunnel used to
        # swallow that and close, so the client saw "peer closed connection
        # without sending complete message body" after ~1 MB.
        client.setblocking(True)
        upstream.settimeout(None)
        peers = {client: upstream, upstream: client}
        open_reads = [client, upstream]
        try:
            while open_reads:
                r, _, _ = select.select(open_reads, [], [], 300)
                if not r:
                    break
                for s in r:
                    data = s.recv(65536)
                    if not data:
                        # EOF on this side: half-close the other side so it
                        # sees the end, but keep relaying the reverse direction
                        # until it also finishes.
                        open_reads.remove(s)
                        try:
                            peers[s].shutdown(socket.SHUT_WR)
                        except OSError:
                            pass
                        continue
                    peers[s].sendall(data)
        except Exception:
            pass
        finally:
            try:
                upstream.close()
            except Exception:
                pass
            self.close_connection = True

    def do_GET(self):
        # Plain-HTTP forwarding: fetch the absolute URL and relay the response.
        import urllib.request
        try:
            req = urllib.request.Request(self.path, headers=dict(self.headers))
            with urllib.request.urlopen(req, timeout=30) as resp:
                self.send_response(resp.status)
                for k, v in resp.headers.items():
                    if k.lower() not in ("transfer-encoding", "connection", "content-length"):
                        self.send_header(k, v)
                body = resp.read()
                self.send_header("Content-Length", str(len(body)))
                self.end_headers()
                self.wfile.write(body)
        except Exception:
            self.send_error(502)

    def log_message(self, *a):
        pass


ThreadingHTTPServer(("0.0.0.0", PORT), Proxy).serve_forever()
PYEOF

  sleep 1
  if lsof -nP -iTCP:"$proxy_port" -sTCP:LISTEN >/dev/null 2>&1; then
    echo ">>> HTTP download proxy is listening on port $proxy_port. Stop it with: ferry down"
    echo ">>> On a client run:  eval \"\$(ferry env --host $MDNS_NAME)\""
  else
    echo "WARNING: HTTP proxy did not come up; check the log: $proxy_log"
  fi
}

# _ferry_install_opencode_guardrails — put the local-lane guardrails where
# opencode will actually read them: the /fan-out command and the
# spawning-subagents skill.
#
# These shipped ONLY in client-bootstrap.sh, so every CLIENT got them and the
# HOST never did — even though `ferry opencode` deliberately wires the host to
# its own endpoint, so the host drives local lanes exactly like a client does.
# On this host that meant the documented mitigation for malformed/looping `task`
# calls had never been installed on the machine reporting the problem.
#
# opencode documents these as `~/.config/opencode/command(s)/<name>.md` and
# `~/.config/opencode/skill(s)/<name>/SKILL.md` — both spellings are accepted,
# and both are GLOBAL paths, independent of $OPENCODE_CONFIG. So this installs
# to the stock location even on a host whose config lives elsewhere.
#
# Source of truth is the repo (opencode/command, opencode/skills). On a client
# there is no checkout, so this no-ops and client-bootstrap.sh's embedded copies
# remain the client path.
_ferry_install_opencode_guardrails() {
  local src_cmd="$APP_DIR/opencode/command/fan-out.md"
  local src_skill="$APP_DIR/opencode/skills/spawning-subagents/SKILL.md"
  if [[ ! -f "$src_cmd" || ! -f "$src_skill" ]]; then
    return 0   # no checkout here (a client) — nothing to install from
  fi
  local dst_cmd="$HOME/.config/opencode/command"
  local dst_skill="$HOME/.config/opencode/skill/spawning-subagents"
  mkdir -p "$dst_cmd" "$dst_skill"
  cp "$src_cmd"   "$dst_cmd/fan-out.md"
  cp "$src_skill" "$dst_skill/SKILL.md"
  echo ">>> opencode guardrails installed:"
  echo "    $dst_cmd/fan-out.md"
  echo "    $dst_skill/SKILL.md"
  echo "    (the recipe must ride in the USER message — that is what /fan-out does;"
  echo "     putting it in system instructions measured WORSE.)"
  # The goal-plugin skill rides along: `ferry install` reaches THIS function and
  # never runs cmd_opencode (they are sibling top-level dispatch entries), so
  # without this call a freshly installed host has the fan-out guardrails and no
  # goal doctrine until someone happens to run `ferry opencode`.
  _ferry_install_goal_skill
}

# _ferry_install_goal_skill — put the goal-plugin USAGE skill where opencode will
# actually read it, next to the guardrails above.
#
# The plugin ships the loop MECHANICS and nothing else: it injects the
# <goal_continuation> block, exposes the goal_* tools and draws the sidebar. It
# never teaches how big a plan step should be, what counts as evidence for one,
# or what the [goal:evidence] / [goal:complete] / [goal:blocked] markers and the
# budget actually mean. A model handed the tools without that doctrine writes a
# forty-step plan, marks a step done because the code "looks right", and never
# emits [goal:blocked] at all. So the skill rides WITH the plugin: every host
# whose config ferry wires /goal into gets the doctrine in the same run.
#
# Source of truth is opencode/skills/using-the-goal-plugin/SKILL.md in this
# checkout — edit it there, not at the destination, which is overwritten on
# every run. A client has no checkout, so this says so and no-ops;
# client-bootstrap.sh ships the client's copy from its own heredoc — in its
# DEFAULT scope only, which is why the no-checkout line names that scope
# instead of promising the file outright: under --profiles-only the bootstrap
# deliberately ships no skill, and a promise here would be contradicted by the
# bootstrap's own report five lines later.
#
# The destination is a GLOBAL opencode path, independent of $OPENCODE_CONFIG,
# and singular `skill/` like the guardrails installer — so it lands where
# opencode looks even on a host whose config lives in a dotfiles directory.
typeset -g _FERRY_GOAL_SKILL_DONE=0
_ferry_install_goal_skill() {
  # Both entry points can fire in one process (`ferry install` reaches the
  # guardrails installer, which calls this). A second copy is harmless; a second
  # report line is noise, so the first call wins.
  (( _FERRY_GOAL_SKILL_DONE )) && return 0
  _FERRY_GOAL_SKILL_DONE=1
  # ...but the guard is per-PROCESS, and the scripts that drive the takeover
  # (host-reset.sh, client-bootstrap.sh, client-reset.sh) run `ferry opencode`
  # once per config target — three or four fresh processes that would each
  # print this line while reporting the same single install. Those scripts
  # report the skill themselves, so they set FERRY_GOAL_SKILL_QUIET=1 to say
  # "I already told the operator". It silences the REPORT LINE ONLY: the copy
  # below still happens, on every target, exactly as it would otherwise.
  local quiet="${FERRY_GOAL_SKILL_QUIET:-}"
  local src="$APP_DIR/opencode/skills/using-the-goal-plugin/SKILL.md"
  if [[ ! -f "$src" ]]; then
    [[ -n "$quiet" ]] || \
      echo "    Skill:   using-the-goal-plugin not installed here (no checkout); the client copy ships in client-bootstrap.sh's default scope"
    return 0
  fi
  local dst="$HOME/.config/opencode/skill/using-the-goal-plugin"
  mkdir -p "$dst"
  cp "$src" "$dst/SKILL.md"
  [[ -n "$quiet" ]] || echo "    Skill:   ~/.config/opencode/skill/using-the-goal-plugin/SKILL.md"
}

# _ferry_install_host_wrappers — put the `opencode-cloud` / `opencode-local` /
# `opencode-super` shell functions in the HOST's ~/.zshrc.
#
# The same gap as the guardrails above, one layer out: the wrappers were written
# ONLY by client-bootstrap.sh, so every client got them and the host never did.
# Confirmed by absence rather than assumed — host-bootstrap.sh contains no
# occurrence of "opencode" at all, and host-reset.sh writes the profile JSONs
# but never touches ~/.zshrc. So the host ended up with the FILES the wrappers
# select between and no way to select between them.
#
# Marker discipline is the whole point. client-bootstrap.sh strips a previous
# block by EXACT string compare on "# >>> ferry opencode profiles >>>", and
# client-cleanup.sh compares the same way. Writing a host block under any other
# marker means neither tool can see it, and the next client bootstrap appends a
# SECOND block defining the same functions (the later definition wins, so the
# duplicate is invisible until the two disagree). This writes the canonical
# marker for exactly that reason, and additionally absorbs the "(host)" variant
# that hand-wiring produced before this function existed.
#
# Named wrappers only, no bare `opencode()`. This matches the client's
# --profiles-only scope: a host that exports OPENCODE_CONFIG has chosen its
# default deliberately, and wrapping bare `opencode` would fight that choice.
FERRY_OC_MARK_START="# >>> ferry opencode profiles >>>"
FERRY_OC_MARK_END="# <<< ferry opencode profiles <<<"

_ferry_install_host_wrappers() {
  local rc="$HOME/.zshrc"
  touch "$rc"

  # Strip the canonical block AND the legacy "(host)" variant, then re-add. Both
  # spellings go, or absorbing the legacy one would just leave two again.
  python3 - "$rc" "$FERRY_OC_MARK_START" "$FERRY_OC_MARK_END" <<'PYEOF'
import sys
rc, start, end = sys.argv[1], sys.argv[2], sys.argv[3]

# The hand-wired spelling this function exists to absorb. Neither
# client-bootstrap.sh nor client-cleanup.sh can match it, because both compare
# marker lines for exact equality.
legacy_start = "# >>> ferry opencode profiles (host) >>>"
legacy_end = "# <<< ferry opencode profiles (host) <<<"

with open(rc) as f:
    lines = f.readlines()

out, skip = [], False
for ln in lines:
    s = ln.rstrip("\n")
    if s in (start, legacy_start):
        skip = True
        continue
    if s in (end, legacy_end):
        skip = False
        continue
    if not skip:
        out.append(ln)

# A stray `alias opencode-cloud=` / `alias opencode-local=` / `alias
# opencode-super=` ABOVE a function of the same name makes zsh expand the alias
# inside `name() {`, which is a parse error on every subsequent
# `source ~/.zshrc`. client-bootstrap.sh strips these for the same reason.
def is_legacy_alias(l):
    t = l.lstrip()
    return (t.startswith("alias opencode-cloud=")
            or t.startswith("alias opencode-local=")
            or t.startswith("alias opencode-super="))

out = [l for l in out if not is_legacy_alias(l)]

while out and out[-1].strip() == "":
    out.pop()
with open(rc, "w") as f:
    f.writelines(out)
    if out:
        f.write("\n")
PYEOF
  if (( $? != 0 )); then
    echo "    WARNING: could not rewrite $rc; leaving the shell wrappers alone." >&2
    return 1
  fi

  # QUOTED heredoc: written verbatim, so the $HOME and $@ inside the functions
  # survive into the file instead of being expanded now.
  cat <<'EOF' >> "$rc"
# >>> ferry opencode profiles >>>
# Installed by `ferry update` / host-reset.sh on the HOST. The profile FILES
# these select between are written by `ferry opencode` in the same pass.
#
# There is deliberately no bare `opencode` function: an explicit OPENCODE_CONFIG
# is the host's own choice and ferry does not override it.
unalias opencode-cloud opencode-local opencode-super 2>/dev/null

# opencode-cloud: heavy drives (build/plan); flash handles `light` (tasks rated
# 0-50) and explore; medium handles `standard` (51-100) when advertised;
# super-flash handles compaction and title/summary. The built-in `general`
# subagent is DISABLED. Older or unreachable hosts put standard on flash too.
opencode-cloud() {
  OPENCODE_CONFIG="$HOME/.config/ferry/opencode-cloud.json" command opencode "$@"
}

# opencode-local: the GPU pair — local-orch drives, local-sub runs the fan-out.
# Nothing leaves this machine.
opencode-local() {
  OPENCODE_CONFIG="$HOME/.config/ferry/opencode-local.json" command opencode "$@"
}

# opencode-super: heavy drives; super-flash runs the fan-out AND the
# housekeeping. The cheapest cloud profile.
opencode-super() {
  OPENCODE_CONFIG="$HOME/.config/ferry/opencode-super.json" command opencode "$@"
}
# <<< ferry opencode profiles <<<
EOF

  echo ">>> opencode shell wrappers installed in $rc:"
  echo "    opencode-cloud   -> cloud lanes: heavy drives; flash light+explore; medium standard; super-flash compaction/title/summary; general disabled"
  echo "    opencode-super   -> cloud pair: heavy drives, super-flash fans out"
  echo "    opencode-local   -> GPU pair:   local-orch drives, local-sub fans out"
  echo "    (bare 'opencode' is untouched — run: source $rc)"
}

cmd_env() {
  # Emit shell 'export' lines so a client routes its downloads through the host's
  # forward proxy (see 'ferry serve-proxy'). Designed for:  eval "$(ferry env ...)".
  # stdout stays PURELY eval-able; any human hint goes to stderr.
  local host="" proxy_port="" hf_port="" do_write=0
  while [[ $# -gt 0 ]]; do
    case "$1" in
      --host)       host="$2"; shift 2 ;;
      --proxy-port) proxy_port="$2"; shift 2 ;;
      --hf-port)    hf_port="$2"; shift 2 ;;
      --write)      do_write=1; shift ;;
      *)            echo "Unknown option: $1" >&2; exit 1 ;;
    esac
  done

  # Resolve the host: --host wins, else the client profile's CLIENT_HOST.
  local H="${host:-$CLIENT_HOST}"
  if [[ -z "$H" ]]; then
    echo "Error: no host given. Pass --host H, or configure ~/.config/ferry/client.json first." >&2
    exit 1
  fi
  local PP="${proxy_port:-$PROXY_PORT}"
  local HFP="${hf_port:-$HF_PORT}"

  # The export block. Built with an expanding heredoc so $H/$PP/$HFP interpolate.
  local block
  block=$(cat <<EOF
export HTTP_PROXY="http://$H:$PP"
export HTTPS_PROXY="http://$H:$PP"
export http_proxy="http://$H:$PP"
export https_proxy="http://$H:$PP"
export ALL_PROXY="http://$H:$PP"
export HF_ENDPOINT="http://$H:$HFP"
export NO_PROXY="localhost,127.0.0.1,::1,$H"
export no_proxy="localhost,127.0.0.1,::1,$H"
EOF
)

  if (( do_write )); then
    local rc="$HOME/.zshrc"
    local start="# >>> ferry env >>>"
    local end="# <<< ferry env <<<"
    touch "$rc"
    # Strip any existing ferry env block (inclusive of its markers) before re-adding.
    python3 - "$rc" "$start" "$end" <<'PYEOF'
import sys
rc, start, end = sys.argv[1], sys.argv[2], sys.argv[3]
with open(rc) as f:
    lines = f.readlines()
out, skip = [], False
for ln in lines:
    s = ln.rstrip("\n")
    if s == start:
        skip = True
        continue
    if s == end:
        skip = False
        continue
    if not skip:
        out.append(ln)
while out and out[-1].strip() == "":
    out.pop()
with open(rc, "w") as f:
    f.writelines(out)
    if out:
        f.write("\n")
PYEOF
    {
      echo "$start"
      echo "$block"
      echo "$end"
    } >> "$rc"
    echo ">>> Appended ferry env block to $rc (host $H). Run: source $rc"
  else
    print -r -- "$block"
    echo "# eval \"\$(ferry env)\" then run your uv/hf tool (downloads route via $H)" >&2
  fi
}

# _ferry_preinstall_goal_plugin — install the goal plugin NOW, through the
# production loader, so a broken spec surfaces at `ferry opencode` time instead
# of never.
#
# `opencode plugin '<spec>' --global` is the ONLY entry point that prints the
# real install error. Inside a normal opencode start a plugin failure is
# published as a Session event and never logged (packages/opencode/src/plugin/
# index.ts:198-201 -> :139-141, with no-op start/missing reporters at :191-192),
# the entry is dropped (loader.ts:234) and npm-source plugins are never retried
# (loader.ts:178) — which is exactly how v1.29.4's unloadable spec survived a
# release. "The cache directory has files in it" is not evidence of anything:
# the whole failure mode is a complete package on disk that opencode discards.
#
# The command also PATCHES a config to add the plugin, which we do not want —
# ferry has already written the config — so it runs against a throwaway
# XDG_CONFIG_HOME/XDG_DATA_HOME/XDG_STATE_HOME with OPENCODE_CONFIG unset, and
# its config patch lands in the sandbox. XDG_CACHE_HOME is deliberately NOT
# overridden: the package cache is the shared real one, and populating it is the
# entire point of doing this early.
#
# 120 s wall clock, without `timeout` (macOS ships none) and without `kill -0`
# (a finished background child stays a zombie until `wait`, so kill -0 reports
# it alive for the whole budget). A completion sentinel file is the observable.
#
# NEVER fails the caller: the config was written correctly either way, and an
# offline laptop must not turn a good bootstrap into a red one.
_ferry_preinstall_goal_plugin() {
  local spec="$1" ref="$2" pkg="$3"
  local sandbox log rcfile pid rc=124 waited=0
  sandbox="$(mktemp -d -t ferry-goal-oc)" || return 0
  log="$sandbox/install.log"
  rcfile="$sandbox/rc"
  echo "    Plugin:  pre-installing $spec"
  # ferry runs under `set -eu`. A FAILING install is the whole point of this
  # pass, so every step that can legitimately return non-zero — the install
  # itself, the kill after a timeout, and the wait on a child that exited 1 —
  # has to be shielded, or errexit tears the command down before it can report.
  (
    local child_rc=0
    unset OPENCODE_CONFIG
    export XDG_CONFIG_HOME="$sandbox/config"
    export XDG_DATA_HOME="$sandbox/data"
    export XDG_STATE_HOME="$sandbox/state"
    mkdir -p "$XDG_CONFIG_HOME" "$XDG_DATA_HOME" "$XDG_STATE_HOME"
    opencode plugin "$spec" --global >"$log" 2>&1 || child_rc=$?
    print -r -- "$child_rc" >"$rcfile"
  ) &
  pid=$!
  while (( waited < 120 )); do
    if [[ -f "$rcfile" ]]; then
      break
    fi
    sleep 1
    waited=$(( waited + 1 ))
  done
  if [[ -f "$rcfile" ]]; then
    rc="$(<"$rcfile")"
  else
    pkill -P "$pid" >/dev/null 2>&1 || true
    kill -TERM "$pid" >/dev/null 2>&1 || true
  fi
  wait "$pid" >/dev/null 2>&1 || true

  # Verify against the CACHE, not against the exit code: a zero exit with an
  # empty node_modules is the interrupted-install state opencode never heals.
  python3 - "$spec" "$ref" "$pkg" "$rc" "$log" <<'PYEOF'
import json, os, sys

spec, ref, pkg, rc_raw, log = sys.argv[1:6]
try:
    rc = int(rc_raw)
except ValueError:
    rc = 1

base = os.environ.get("XDG_CACHE_HOME") or os.path.join(os.path.expanduser("~"), ".cache")
# Same formula opencode uses: path.join(cache, "packages", <raw spec>), with the
# spec's slashes acting as path separators and "//" collapsed by the join.
directory = os.path.normpath(os.path.join(base, "opencode", "packages", spec))
root = os.path.join(directory, "node_modules", pkg)
manifest = os.path.join(root, "package.json")
want = ref[1:] if ref.startswith("v") else ref

problems, version = [], None
if rc == 124:
    problems.append("`opencode plugin` did not finish within 120s")
elif rc != 0:
    problems.append(f"`opencode plugin` exited {rc}")
if os.path.exists(manifest):
    try:
        version = json.load(open(manifest)).get("version")
    except Exception as e:
        problems.append(f"{manifest} is unreadable ({e})")
else:
    problems.append(f"missing {manifest}")
if version and version != want:
    problems.append(f"cache holds version {version}, expected {want}")
# Both halves: the server plugin AND the tui sidebar module. A package missing
# dist/goal-tui.js loads the /goal command and no sidebar, which is exactly the
# kind of half-success that reads as working.
for half in ("dist/goal-plugin.js", "dist/goal-tui.js"):
    if not os.path.exists(os.path.join(root, half)):
        problems.append(f"missing {half}")

if not problems:
    print(f"    Plugin:  installed {pkg} {version} (ready for next opencode start)")
else:
    print(f"    WARNING: the goal plugin is NOT installed: {'; '.join(problems)}")
    try:
        tail = [l.rstrip() for l in open(log).read().splitlines() if l.strip()][-6:]
    except OSError:
        tail = []
    for line in tail:
        print(f"             {line}")
    print("             opencode will retry on next start.")
PYEOF

  rm -rf "$sandbox"
  return 0
}

# _ferry_sync_goal_tui_copy — refresh ferry's colon-free COPY of the installed
# goal plugin and point tui.json at it.
#
# The TUI half of a plugin cannot be loaded out of opencode's package cache: the
# canonical spec's cache directory contains the component
# `opencode-goal-plugin@https:`, Bun's runtime plugin runner splits a module
# path at the FIRST colon into `namespace:path`, and a file under a colon path
# therefore never reaches opentui's host-module shim
# (packages/opencode/src/plugin/tui/runtime.ts:47) that supplies solid-js. The
# server half loads from that same spec without complaint, which is what made
# the breakage invisible. Full trace, and the one-factor-varied path table, in
# the GOAL_TUI_MARKER comment block in the python above.
#
# So: copy <cache>/node_modules/<pkg> to $XDG_DATA_HOME/ferry/<pkg> (no colon,
# no '#'), stamp it with a marker file that proves the copy is ferry's, and
# rewrite the tui.json entry from the spec to a file:// URL for the copy.
#
# Runs AFTER the pre-install, and re-verifies the cache itself rather than
# trusting it - the pre-install may have warned and still left a usable tree,
# or left nothing at all. NEVER fails the caller (ferry runs under `set -eu`,
# lib/ferry-core.zsh:25): a laptop that cannot copy still has a correct config
# and a working /goal command.
_ferry_sync_goal_tui_copy() {
  local spec="$1" ref="$2" pkg="$3" tui_file="$4" tui_dir="$5" cache_root="$6"
  python3 - "$spec" "$ref" "$pkg" "$tui_file" "$tui_dir" "$cache_root" <<'PYEOF' || true
import json, os, pathlib, shutil, sys

spec, ref, pkg, tui_file, tui_dir, cache_root = sys.argv[1:7]
want = ref[1:] if ref.startswith("v") else ref
MARKER = ".ferry-goal-plugin"


def bail(reason):
    print(f"    TUI plugin: not copied ({reason}); tui.json keeps the spec")
    sys.exit(0)


# 1. The SOURCE has to be a complete install of the version we pinned. A
#    half-written cache copied into place is a broken plugin with ferry's name
#    on it.
manifest = os.path.join(cache_root, "package.json")
if not os.path.exists(manifest):
    bail(f"missing {manifest}")
try:
    version = json.load(open(manifest)).get("version")
except Exception as e:
    bail(f"{manifest} is unreadable ({e})")
if version != want:
    bail(f"the cache holds version {version}, expected {want}")
if not os.path.exists(os.path.join(cache_root, "dist", "goal-tui.js")):
    bail(f"missing dist/goal-tui.js under {cache_root}")

# 2. Refuse the two paths that would reproduce the very bug this dodges, and
#    refuse to touch a directory ferry did not create. cmd_opencode already
#    skips the call in the ':'/'#' case (it reported it when it wrote the
#    config); the guard stands so the function is safe called on its own.
if ":" in tui_dir or "#" in tui_dir:
    bail(f"{tui_dir} contains ':' or '#'")
marker = os.path.join(tui_dir, MARKER)
if os.path.exists(tui_dir) and not os.path.isfile(marker):
    print(f"    WARNING: {tui_dir} exists but carries no {MARKER};")
    print("             it is not ferry's, so it is left alone and tui.json keeps the")
    print("             spec. Move it aside to let ferry manage the TUI copy.")
    sys.exit(0)

# 3. Replace wholesale — a merge over an older version leaves stale files.
try:
    if os.path.exists(tui_dir):
        shutil.rmtree(tui_dir)
    os.makedirs(os.path.dirname(tui_dir) or ".", exist_ok=True)
    shutil.copytree(cache_root, tui_dir, symlinks=False)
    with open(marker, "w") as f:
        f.write(f"{spec}\n{ref}\n{pkg}\n")
except Exception as e:
    print(f"    WARNING: could not copy the goal plugin to {tui_dir} ({e});")
    print("             tui.json keeps the spec and the sidebar half will not load.")
    sys.exit(0)

# Verify the COPY, not the source: rmtree+copytree onto a full disk is exactly
# the failure that would otherwise be reported as a success.
try:
    copied = json.load(open(os.path.join(tui_dir, "package.json"))).get("version")
except Exception as e:
    copied = None
if copied != want or not os.path.exists(os.path.join(tui_dir, "dist", "goal-tui.js")):
    print(f"    WARNING: the copy at {tui_dir} is incomplete (version {copied});")
    print("             tui.json keeps the spec.")
    sys.exit(0)

uri = pathlib.Path(tui_dir).as_uri()

# 4. Point tui.json at the copy. No snapshot: the main block snapshotted this
#    exact file moments ago in this same run, and it wrote it as plain JSON.
rewritten = False
if tui_file and os.path.exists(tui_file):
    try:
        with open(tui_file) as f:
            data = json.load(f)
    except Exception as e:
        print(f"    WARNING: {tui_file} is unreadable ({e}); left as it was.")
        data = None
    if isinstance(data, dict) and isinstance(data.get("plugin"), list):
        out = []
        for p in data["plugin"]:
            if isinstance(p, list) and p and p[0] == spec:
                out.append([uri] + list(p[1:]))      # options survive
            elif p == spec:
                out.append(uri)
            else:
                out.append(p)
        seen, deduped = set(), []
        for p in out:
            key = json.dumps(p, sort_keys=True)
            if key in seen:
                continue
            seen.add(key)
            deduped.append(p)
        data["plugin"] = deduped
        try:
            with open(tui_file, "w") as f:
                json.dump(data, f, indent=2)
                f.write("\n")
            rewritten = True
        except OSError as e:
            print(f"    WARNING: could not write {tui_file} ({e}).")

if rewritten:
    print(f"    TUI plugin: {tui_dir} ({copied}) -> {tui_file}")
else:
    print(f"    TUI plugin: {tui_dir} ({copied})")
PYEOF
  return 0
}

cmd_opencode() {
  # [Client] Take this machine's opencode config over so EVERY agent routes
  # through the host's ferry endpoint, addressed by LANE NAME only.
  #
  # A ferry client knows agent lanes, never a real model:
  #
  #   driver       build / plan                      heavy       local-orch
  #   light        light   (complexity 0-50)         flash       local-sub
  #   standard     standard (complexity 51-100)      medium*     local-sub
  #   explore      explore                           flash       local-sub
  #   compaction   compaction                        super-flash local-sub
  #   housekeeper  title / summary                   super-flash local-sub
  #
  # opencode's built-in `general` subagent is DISABLED — the fan-out worker is
  # split into two custom subagents banded by complexity, because opencode's
  # task tool has no model parameter: the driver picks an AGENT NAME, and each
  # agent is pinned to exactly one lane here.
  #
  # * `medium` is used only when the host catalogue advertises it. Older or
  # unreachable hosts retain flash for standard rather than receiving a broken
  # new lane reference.
  #
  # Compaction fires on its own schedule and carries the ENTIRE transcript, so
  # cloud uses the super-flash housekeeping lane. On the GPU pair there is no
  # third lane, so all non-driver agents share local-sub.
  #
  # A real model id must NEVER reach a client config. The host re-points a lane
  # whenever the economics change; a client that named the model would keep
  # asking for something the catalogue no longer advertises. Which model sits
  # behind a lane is the host's business and is not discoverable from here.
  #
  # TAKEOVER, not merge. Four keys are ferry's and get replaced outright:
  #   permission  -> "allow"
  #   model       -> ferry/<driver>
  #   small_model -> ferry/<housekeeper>
  #   agent       -> six built-ins pinned, `general` disabled, and the two
  #                  custom `light`/`standard` subagents declared (see the
  #                  AGENTS lists below)
  # provider.ferry.options.headers is ours too, rewritten every run alongside
  # baseURL/apiKey (see the prov["ferry"] block below) - it carries this
  # machine's identity and a one-shot fleet override, never a real model id.
  # `plugin` gets the goal plugin appended only when no entry already IS that
  # plugin - which includes a LOCAL PATH to a fork of it, since opencode accepts
  # a filesystem path and a private fork can only be named that way. The spec is
  # the NAME-PREFIXED TARBALL form
  # `opencode-goal-plugin@https://github.com/.../vX.Y.Z.tar.gz`, because on
  # opencode 1.18.29 a bare `github:` spec installs to disk and is then silently
  # discarded, and any git spec dies in pacote's prepare step - see the
  # GOAL_PLUGIN comment block for the file:line trace. Every earlier spelling
  # ferry ever wrote is rewritten to it on every run, and the ref is pinned
  # because the spec string IS opencode's cache key.
  # The plugin is ALSO listed in ~/.config/opencode/tui.json, the only place
  # opencode reads TUI-half plugins from - opencode.json's `plugin` array feeds
  # the server loader alone, so the plugin's sidebar panel never appears without
  # it. tui.json does NOT get the spec, though: the TUI loader cannot load a
  # module out of the package cache, whose directory name contains
  # `opencode-goal-plugin@https:`, because Bun splits a module path at the first
  # colon (see the GOAL_TUI_MARKER block). It gets a file:// URL for a
  # colon-free COPY ferry keeps in $XDG_DATA_HOME/ferry/opencode-goal-plugin and
  # refreshes from the cache after the pre-install. Only the full-takeover
  # target gets a tui.json:
  # --config paths outside ~/.config/opencode (the ferry lane profiles) never do.
  # `command.goal` is MERGED in (never taken over) so the plugin's /goal slash
  # command exists; a user's own `goal` and every other command are left alone.
  # Every OTHER key in the
  # file is left exactly as it was, and the whole original is snapshotted to
  # <name>.<UTC>.jsonc first, so a takeover is always reversible.
  # Afterwards the orphaned per-spec cache directories are removed and the plugin
  # is pre-installed through `opencode plugin`, so a broken spec is visible here
  # rather than silently absent at runtime (--keep-cache / --no-install opt out).
  local oc_host="${CLIENT_HOST:-}" oc_port="${CLIENT_PORT:-8090}"
  # v1.22.0: the front door can run litellm behind a master_key. The bearer
  # baked into the generated configs (and sent on the catalogue check) resolves
  # --key > client.json's master_key (boot-loaded as CLIENT_MASTER_KEY) > unset
  # (the legacy 'local' token, so keyless LAN setups are unchanged). The key is
  # only written into files / request headers, never printed.
  local oc_key=""
  # opencode resolves its config from $OPENCODE_CONFIG when that is set, so
  # honour it here too. Writing the hardcoded default on a machine that sets
  # OPENCODE_CONFIG edits a file opencode never reads: the command reports
  # success, and nothing changes.
  local oc_config="${OPENCODE_CONFIG:-$HOME/.config/opencode/opencode.json}"
  local force_model="" force_small="" force_house="" set_default=1 prefer_local=0 force_write=0 keep_snaps=10
  # tui.json: the TUI half of the goal plugin is loaded from tui.json ONLY -
  # opencode.json's `plugin` array is never read by the TUI plugin loader
  # (packages/opencode/src/config/tui.ts:157-210). "auto" means "mirror the
  # entry when the config we are writing IS the global takeover target".
  local tui_mode="auto" tui_path="${OPENCODE_TUI_CONFIG:-}"
  local keep_cache=0 do_install=1

  while [[ $# -gt 0 ]]; do
    case "$1" in
      --host)         oc_host="$2"; shift 2 ;;
      --port)         oc_port="$2"; shift 2 ;;
      --config)       oc_config="$2"; shift 2 ;;
      --key)          oc_key="$2"; shift 2 ;;
      --model)        force_model="$2"; shift 2 ;;
      --small-model)  force_small="$2"; shift 2 ;;
      --housekeeper)  force_house="$2"; shift 2 ;;
      # --super: the cheapest cloud profile — heavy still drives, super-flash
      # takes every non-driver agent. Set here at parse time so a later
      # --small-model / --housekeeper overrides its respective agent group.
      --super)        force_small="super-flash"; force_house="super-flash"; shift ;;
      --keep)         keep_snaps="$2"; shift 2 ;;
      # Where the TUI half of the goal plugin gets listed. An explicit path is
      # always honoured; --no-tui-config suppresses the mirror entirely.
      --tui-config)   tui_mode="explicit"; tui_path="$2"; shift 2 ;;
      --no-tui-config) tui_mode="off"; tui_path=""; shift ;;
      # Leave ~/.cache/opencode/packages/ alone (see the purge block below).
      --keep-cache)   keep_cache=1; shift ;;
      # Skip the `opencode plugin` pre-install/verify pass after the write.
      --no-install)   do_install=0; shift ;;
      --no-default)   set_default=0; shift ;;
      --local)        prefer_local=1; shift ;;
      # Retained for compatibility: --force used to bypass a refusal to rewrite
      # a commented (JSONC) config. That refusal is gone — the snapshot keeps the
      # original verbatim, comments included — so the flag is now a no-op.
      --force)        force_write=1; shift ;;
      --cloud)        prefer_local=0; shift ;;
      # Install the ~/.zshrc wrappers and nothing else. host-reset.sh calls this
      # once, after writing the profile files the wrappers select between —
      # doing it inside the normal path would re-run it once per config.
      --wrappers)     _ferry_install_host_wrappers; return $? ;;
      *) echo "Unknown option for 'ferry opencode': $1"; exit 1 ;;
    esac
  done

  if [[ -z "$oc_host" ]]; then
    if (( CLIENT_MODE )); then
      # A bootstrapped client whose profile exists but carries no host.
      echo "Error: ~/.config/ferry/client.json has no 'host'. Re-run the client"
      echo "bootstrap, or pass --host <mdns-or-ip> explicitly."
      exit 1
    fi
    # HOST: the proxy is on this very machine, so loopback is the right default.
    # Wiring the host to its own endpoint is the point of running one — every
    # local tool then shares the lanes, the fallback chain, and the observability,
    # and no tool on this box needs its own copy of a provider key.
    oc_host="127.0.0.1"
    echo ">>> No --host and no client profile: this is the HOST, so wiring it to its"
    echo "    own proxy at http://127.0.0.1:$oc_port/v1."
  fi

  # Flag wins over the boot-loaded profile key; both unset keeps the legacy token.
  [[ -z "$oc_key" ]] && oc_key="${CLIENT_MASTER_KEY:-}"

  # The canonical plugin spec lives in ONE place (the Python block below). It is
  # handed back through this scratch file so the pre-install pass cannot drift
  # from the string that was actually written into the config.
  local oc_specfile; oc_specfile="$(mktemp -t ferry-goal-spec)"

  python3 - "$oc_host" "$oc_port" "$oc_config" "$force_model" "$force_small" "$set_default" "$prefer_local" "$force_write" "$keep_snaps" "$force_house" "$oc_key" "$CLIENT_NAME" "$keep_cache" "$tui_mode" "$tui_path" "$oc_specfile" "$do_install" <<'PYEOF'
import datetime, json, os, pathlib, re, sys, shutil, urllib.parse, urllib.request

host, port, cfg_path, force_model, force_small = sys.argv[1:6]
set_default  = sys.argv[6] == "1"
prefer_local = sys.argv[7] == "1"
force_write  = sys.argv[8] == "1"   # no-op; see the --force note above
keep_snaps   = int(sys.argv[9])
force_house  = sys.argv[10]
oc_key       = sys.argv[11]
client_name  = sys.argv[12]
keep_cache   = sys.argv[13] == "1"
tui_mode     = sys.argv[14]         # auto | explicit | off
tui_path     = sys.argv[15]
spec_out     = sys.argv[16]
# Only so the stale-copy note can say whether this run will refresh the copy.
do_install   = sys.argv[17] == "1"
cfg_path = os.path.expanduser(cfg_path)
base = f"http://{host}:{port}/v1"

SCHEMA = "https://opencode.ai/config.json"
TUI_SCHEMA = "https://opencode.ai/tui.json"

# --- The goal plugin spec, and why it is spelled EXACTLY like this. ---
#
# opencode 1.18.29 has TWO independent defects that make the obvious spellings
# install-to-disk-but-never-load, with nothing in any log:
#
#   1. NO NPM NAME. For a bare `github:owner/repo[#tag]` or a bare tarball URL,
#      npm-package-arg returns no name, so `npa(pkg).name ?? pkg`
#      (packages/core/src/npm.ts:119) yields the WHOLE RAW SPEC as the "name".
#      After a perfectly successful reify, `tree.edgesOut` is empty, so
#      npm.ts:130-134 falls back to resolveEntryPoint(<raw spec>), which cannot
#      resolve, and throws NpmInstallFailedError - with the package fully
#      written to disk. The failure is published as a Session event and never
#      logged (plugin/index.ts:198-201 -> :139-141; the start/missing reporters
#      at :191-192 are no-ops), the entry is dropped (loader.ts:234) and npm
#      plugins are never retried (loader.ts:178). v1.29.4's pin
#      `github:sblattj/OpenCode-goal-plugin#v0.9.1` was dead on arrival, and so
#      was the unpinned spelling before it: the cache directory filled with a
#      complete package that opencode then threw away on every single start.
#   2. GIT PREPARE. Any GIT spec whose package.json declares any of
#      postinstall/build/preinstall/install/prepack/prepare makes pacote's
#      GitFetcher run a "prepare" step (pacote/lib/git.js:160-197) - arborist's
#      `ignoreScripts: true` does NOT suppress it. opencode derives npmBin from
#      `fileURLToPath(new URL("..", import.meta.url))`
#      (packages/core/src/npm-config.ts:10), which inside the bun single-file
#      binary is `/$bunfs/bin/npm-cli.js`; because that ends in `.js`, pacote
#      spawns process.execPath - i.e. opencode ITSELF - with it as argv[1]
#      (pacote/lib/util/npm.js:5-7). The child prints opencode's own help, exits
#      1, and the install dies with "git dep preparation failed". The plugin has
#      declared `build` + `prepack` since v0.9.1, so every git form is poisoned.
#
# The one form that installs AND loads is NAME-PREFIXED + REMOTE TARBALL. The
# `name@` prefix gives npa a real name, which fixes both the entrypoint
# resolution and the cache short-circuit (defect 1); a remote tarball is fetched
# by pacote's RemoteFetcher, which never enters the git prepare path at all, so
# it is IMMUNE to defect 2 rather than merely dodging its trigger. Verified end
# to end against opencode 1.18.29.
GOAL_PLUGIN_PKG = "opencode-goal-plugin"
GOAL_PLUGIN_REPO = "sblattj/OpenCode-goal-plugin"
# PIN THE REF, and bump it on every release that ships a new plugin version.
# opencode installs a plugin into ~/.cache/opencode/packages/<the spec string>/
# and, if node_modules/<pkg> already exists in that directory, returns
# IMMEDIATELY without refetching - no TTL, no version compare, no eviction
# (packages/core/src/npm.ts:79,125-127). An UNPINNED spec therefore freezes a
# machine at whatever it fetched first: hosts sat on plugin 0.9.0 for weeks
# while the fork's HEAD was 0.9.1. The spec string IS the cache key, so a NEW
# ref means a NEW directory and a guaranteed fresh install. That is why the ref
# is pinned here and why cutting a plugin release means bumping GOAL_PLUGIN_REF.
GOAL_PLUGIN_REF = "v0.11.0"
GOAL_PLUGIN_URL = (f"https://github.com/{GOAL_PLUGIN_REPO}"
                   f"/archive/refs/tags/{GOAL_PLUGIN_REF}.tar.gz")
# Current spec, spelled out for grep:
# opencode-goal-plugin@https://github.com/sblattj/OpenCode-goal-plugin/archive/refs/tags/v0.11.0.tar.gz
GOAL_PLUGIN = f"{GOAL_PLUGIN_PKG}@{GOAL_PLUGIN_URL}"
# Kept for MATCHING configs written by older ferries; never written any more.
GOAL_PLUGIN_BASE = f"github:{GOAL_PLUGIN_REPO}"
LEGACY_GOAL_PLUGINS = {
    "@prevalentware/opencode-goal-plugin",
    "opencode-goal-plugin",
    "willytop8/opencode-goal-plugin",
    "github:willytop8/opencode-goal-plugin",
    "sblattj/opencode-goal-plugin",
}
# Substrings that identify OUR plugin inside any spec shape a previous ferry (or
# a hand edit) could have written: `github:owner/repo`, `owner/repo#ref`,
# `git+https://github.com/owner/repo.git`, an `archive/refs/tags/*.tar.gz` URL.
# Matched case-insensitively because the repo is CamelCase and half the historic
# spellings are not. Never applied to a local filesystem path - see is_goal_plugin.
GOAL_REPO_MARKERS = ("sblattj/opencode-goal-plugin", "willytop8/opencode-goal-plugin")
# The package's own directory name, used to recognise a LOCAL PATH pointing at
# the same plugin. opencode accepts a filesystem path as a plugin entry, and Bun
# cannot resolve a PRIVATE repo over `github:` - so a hard fork of this plugin
# can only be named by path. A path never equals the npm name, so a presence
# check on the name alone re-appended upstream on EVERY run, leaving opencode
# loading both the fork and the very package the fork exists to replace.
GOAL_PLUGIN_DIR = GOAL_PLUGIN_REPO.rsplit("/", 1)[-1].lower()

# --- Why tui.json CANNOT carry the spec, and points at a copy instead. ---
#
# opencode installs a package at
# `~/.cache/opencode/packages/<the spec, verbatim>/node_modules/<pkg>`
# (packages/core/src/npm.ts:43-47,79; sanitize() is a no-op off Windows), so the
# canonical tarball spec's directory literally contains the path component
# `opencode-goal-plugin@https:`. Bun's runtime plugin runner splits any module
# path at the FIRST colon into `namespace:path`, so a module living under a
# colon-bearing directory never reaches opentui's host-module shim
# (`ensureRuntimePluginSupport`, packages/opencode/src/plugin/tui/runtime.ts:47)
# - the `file`-namespace onLoad hook that shares the host's solid-js/@opentui
# with plugins by rewriting the bundle's bare `import ... from "solid-js"` into
# `opentui:runtime-module:solid-js`. Without that rewrite Bun's native resolver
# takes over and fails: the TUI console prints
# `[tui.plugin] failed to load tui plugin ... Cannot find package 'solid-js'
# from '<cache path>/dist/goal-tui.js'` and the sidebar silently never appears.
# The SERVER half loads from the very same spec without complaint, which is why
# v1.30.1's mirror looked correct and shipped a half-dead plugin.
#
# One factor varied — the SAME bundle, copied byte for byte, listed in tui.json
# as a `file://` URL (bun 1.3.14):
#     /tmp/x/opencode-goal-plugin               loads
#     /tmp/x/pkg@v1/x/opencode-goal-plugin      loads
#     /tmp/x/node_modules/opencode-goal-plugin  loads
#     /tmp/x/https:/x/opencode-goal-plugin      FAILS
#     /tmp/x/a:b/opencode-goal-plugin           FAILS
# A `#` is fatal for the same reason (the shim slices a path at the first `?`
# or `#`), so the pre-v1.30.1 `github:...#ref` cache directories were doubly
# broken. Only a registry spec (`name@1.2.3`) gets a colon-free cache dir.
#
# Hence the split: opencode.json keeps the tarball SPEC (the server half is
# resolved out of the cache, next to the `zod` sibling installed with it), and
# tui.json gets a `file://` URL pointing at a colon-free COPY that ferry owns
# under $XDG_DATA_HOME/ferry/ and refreshes from the cache after the
# pre-install (_ferry_sync_goal_tui_copy).
GOAL_TUI_MARKER = ".ferry-goal-plugin"     # ownership marker inside the copy


def goal_tui_dir():
    """Where ferry keeps its colon-free copy of the installed plugin."""
    base = os.environ.get("XDG_DATA_HOME") or os.path.join(
        os.path.expanduser("~"), ".local", "share")
    return os.path.abspath(os.path.join(base, "ferry", "opencode-goal-plugin"))


def goal_tui_spec():
    """The tui.json entry for that copy: file:///Users/.../opencode-goal-plugin."""
    return pathlib.Path(goal_tui_dir()).as_uri()


def tui_dir_unusable():
    """A ':' or '#' in OUR path would hit the very Bun bug we are dodging."""
    d = goal_tui_dir()
    return (":" in d) or ("#" in d)


def local_path_of(raw):
    """`raw` as an absolute filesystem path, or None when it names no path."""
    if not isinstance(raw, str):
        return None
    low = raw.lower()
    if low.startswith("file://"):
        p = raw[len("file://"):]
    elif low.startswith("file:"):
        p = raw[len("file:"):]
    elif raw.startswith(("/", ".", "~")):
        p = raw
    else:
        return None
    return os.path.normpath(os.path.expanduser(urllib.parse.unquote(p)))


def is_managed_tui_entry(entry):
    """Does this entry name FERRY'S managed copy (file:// URL or plain path)?"""
    raw = entry[0] if isinstance(entry, list) and entry else entry
    p = local_path_of(raw)
    return p is not None and p == goal_tui_dir()


def tui_copy_state():
    """None when there is no managed copy; else what the one on disk holds.

    The marker file is BOTH the ownership proof (ferry never deletes a
    directory it did not write) and the record of which spec/ref produced it.
    """
    d = goal_tui_dir()
    marker = os.path.join(d, GOAL_TUI_MARKER)
    if not os.path.isfile(marker):
        return None
    try:
        lines = open(marker).read().splitlines()
    except OSError:
        lines = []
    try:
        version = json.load(open(os.path.join(d, "package.json"))).get("version")
    except Exception:
        version = None
    return {"version": version,
            "spec": lines[0] if lines else "",
            "ok": os.path.exists(os.path.join(d, "dist", "goal-tui.js"))}


# opencode's own `command` config key (top-level), NOT the
# ~/.config/opencode/command/*.md files ferry installs for /fan-out. The goal
# plugin's README requires this entry or its /goal slash command never appears.
GOAL_COMMAND = {
    "description": "Set a session-scoped goal and auto-continue until complete.",
    "template": "$ARGUMENTS",
    "agent": "build",
}

# Every remote spelling of our plugin whose NAME half identifies it outright.
GOAL_NAME_MATCHES = {x.lower() for x in LEGACY_GOAL_PLUGINS}
GOAL_NAME_MATCHES.add(GOAL_PLUGIN_BASE.lower())
GOAL_NAME_MATCHES.add(GOAL_PLUGIN_PKG.lower())


def pkg_name(entry):
    """The npm NAME half of a plugin entry (the raw string when it has none)."""
    if isinstance(entry, list) and entry:      # the ["pkg", {opts}] form
        entry = entry[0]
    if not isinstance(entry, str):
        return None
    # Strip trailing #ref
    if "#" in entry:
        entry = entry.rsplit("#", 1)[0]
    # Split at the FIRST separating "@", never the last: the canonical spec is
    # `opencode-goal-plugin@https://...`, and rsplit() would hand back the whole
    # `name@https://github.com/...` string the moment a URL carried an "@" of its
    # own. A leading "@" is an npm SCOPE, not a separator.
    if entry.startswith("@"):
        rest = entry[1:]
        return "@" + rest.split("@", 1)[0] if "@" in rest else entry
    if "@" in entry:
        return entry.split("@", 1)[0]
    return entry


def is_path_entry(entry):
    """A local filesystem path (or file:// URL): somebody's fork, never rewritten."""
    raw = entry[0] if isinstance(entry, list) and entry else entry
    if not isinstance(raw, str):
        return False
    return raw.startswith(("/", ".", "~")) or raw.lower().startswith("file:")


def is_goal_spec(entry):
    """Any REMOTE spelling of our plugin, in every shape ferry ever wrote.

    Covers the bare npm name, the scoped upstream name, `github:owner/repo`
    with or without a `#ref`, the same repo behind a `name@` prefix, a
    `git+https://` URL and an `archive/refs/tags/*.tar.gz` URL. A LOCAL PATH is
    deliberately excluded - opencode accepts a filesystem path, Bun cannot
    resolve a private repo over `github:`, and a hard fork can only be named
    that way, so a path keeps its counts-as-present behaviour untouched.

    The ONE path that is ours anyway is ferry's managed copy (goal_tui_dir()),
    checked BEFORE that exclusion: ferry wrote it, so ferry rewrites it. Every
    other path entry is still somebody's fork and is left alone.
    """
    raw = entry[0] if isinstance(entry, list) and entry else entry
    if not isinstance(raw, str):
        return False
    if is_managed_tui_entry(raw):
        return True
    if is_path_entry(raw):
        return False
    name = pkg_name(raw)
    if isinstance(name, str) and name.lower() in GOAL_NAME_MATCHES:
        return True
    low = raw.lower()
    return any(m in low for m in GOAL_REPO_MARKERS)


def is_goal_plugin(entry):
    """Does this entry already SATISFY the requirement (upstream or a fork)?"""
    if is_managed_tui_entry(entry):        # ferry's own colon-free copy
        return True
    name = pkg_name(entry)
    if not isinstance(name, str):
        return False
    # The canonical spec's name half, plus the bare `github:` form an older
    # ferry wrote (pkg_name() has already stripped any trailing #ref).
    if name.lower() in (GOAL_PLUGIN_PKG.lower(), GOAL_PLUGIN_BASE.lower()):
        return True
    # Match the package's directory name as a whole PATH SEGMENT, with or
    # without a file extension, so ".../opencode-goal-plugin/dist/server.js"
    # and ".../opencode-goal-plugin.js" both count as present while a
    # neighbouring ".../opencode-goal-plugin-extras/..." does not.
    if name.startswith(("/", ".", "~")):
        segs = [s.lower() for s in name.split("/") if s]
        return GOAL_PLUGIN_DIR in segs or GOAL_PLUGIN_DIR in (
            os.path.splitext(s)[0].lower() for s in segs)
    return False


def ensure_goal_plugin(plugins, want=None):
    """Migrate every earlier spelling to `want`, dedupe, guarantee one entry.

    Ferry OWNS this entry: any other remote spelling is drift, and rewriting it
    is the only way a working spec (or a new plugin version) ever reaches a
    machine that already has one. Options on a ["pkg", {opts}] tuple survive.

    `want` is the canonical entry FOR THIS FILE, and the two files differ:
    opencode.json always gets GOAL_PLUGIN (the server half resolves out of the
    package cache, beside the `zod` installed with it, and a managed entry found
    there is migrated back to the spec), while tui.json gets the colon-free
    managed copy - see the GOAL_TUI_MARKER block for why.

    Returns (plugins, goal_entry, migrated_from); migrated_from lists the raw
    spec strings this run replaced - i.e. the cache directories nothing
    references any more. GOAL_PLUGIN is never listed even when tui.json moves
    off it (opencode.json still points at that cache directory), and neither is
    the managed copy, which is not a cache directory at all.
    """
    if want is None:
        want = GOAL_PLUGIN
    if not isinstance(plugins, list):
        plugins = []
    migrated_from, out = [], []
    for p in plugins:
        if not is_goal_spec(p):
            out.append(p)
            continue
        old = p[0] if isinstance(p, list) and p else p
        if isinstance(p, list) and len(p) > 1:
            out.append([want, p[1]])
        else:
            out.append(want)
        if old not in (want, GOAL_PLUGIN) and not is_managed_tui_entry(old):
            migrated_from.append(old)

    # Deduplicate by package name while preserving order.
    seen, deduped = set(), []
    for p in out:
        name = pkg_name(p)
        key = name.lower() if isinstance(name, str) else None
        if key and key in seen:
            continue
        if key:
            seen.add(key)
        deduped.append(p)

    goal_entry = next((e for e in deduped if is_goal_plugin(e)), None)
    if goal_entry is None:
        deduped.append(want)
        goal_entry = want
    return deduped, goal_entry, migrated_from


def load_jsonc(path, label):
    """Read a JSON/JSONC config, tolerating comments and trailing commas."""
    if not os.path.exists(path):
        return {}
    raw = open(path).read()
    try:
        return json.loads(raw)
    except json.JSONDecodeError:
        stripped = re.sub(r'("(?:\\.|[^"\\])*")|//[^\n]*|/\*.*?\*/',
                          lambda m: m.group(1) or '', raw, flags=re.S)
        stripped = re.sub(r',\s*([}\]])', r'\1', stripped)
        try:
            return json.loads(stripped)
        except json.JSONDecodeError as e:
            print(f"    WARNING: {label} is unparseable ({e}); starting from a fresh one.")
            print("    The original is preserved verbatim in the snapshot below.")
            return {}


# --- tui.json: the ONLY place the TUI half of a plugin is read from. ---
# opencode.json's `plugin` array feeds the SERVER plugin loader. A module loaded
# with kind:"tui" (the goal plugin's sidebar panel) comes from tui.json /
# tui.jsonc in the global config dir, $OPENCODE_TUI_CONFIG, project tui files,
# or a .opencode directory - packages/opencode/src/config/tui.ts:157-210. Listing
# the spec in opencode.json alone loads the server half and silently nothing else.
def global_opencode_dir():
    base = os.environ.get("XDG_CONFIG_HOME") or os.path.join(os.path.expanduser("~"), ".config")
    return os.path.abspath(os.path.join(base, "opencode"))


def tui_target(cfg_path, mode, path):
    """Which tui.json to mirror the plugin entry into, or None for "do not".

    "auto" writes ONLY when the config being written is the global takeover
    target itself. The ferry lane profiles (~/.config/ferry/opencode-*.json) and
    the --profiles-only / --no-opencode client scopes must never bring
    ~/.config/opencode into existence - that ABSENCE is what those scopes mean,
    and lib/ferry-clientbootstrap.test.py asserts it.
    """
    if mode == "off":
        return None
    if mode == "explicit":
        return os.path.expanduser(path) if path else None
    if os.path.abspath(os.path.dirname(os.path.expanduser(cfg_path))) != global_opencode_dir():
        return None
    return os.path.expanduser(path) if path else os.path.join(global_opencode_dir(), "tui.json")


# --- Cache hygiene: opencode NEVER invalidates ~/.cache/opencode/packages. ---
# The per-spec directory is path.join(global.cache, "packages", spec) with
# sanitize() a no-op off Windows, so the raw spec string acts as a PATH - the
# slashes in a tarball URL nest several levels deep, and node's path.join
# collapses the "//" exactly as normpath does (packages/core/src/npm.ts:43-47,79).
# With a name-prefixed spec the install SHORT-CIRCUITS on the mere existence of
# <dir>/node_modules/<name> (npm.ts:125-127) - no TTL, no version compare - so an
# interrupted install leaves an empty directory that is treated as installed
# forever, with no self-heal and no retry (loader.ts:178).
DEAD_CACHE_SPECS = (
    "github:sblattj/OpenCode-goal-plugin#v0.9.1",
    "github:sblattj/OpenCode-goal-plugin",
    "github:sblattj/opencode-goal-plugin",
    "opencode-goal-plugin@latest",
)


def cache_packages_root():
    base = os.environ.get("XDG_CACHE_HOME") or os.path.join(os.path.expanduser("~"), ".cache")
    return os.path.abspath(os.path.join(base, "opencode", "packages"))


def cache_dir_for(spec):
    """opencode's per-spec directory for `spec`.

    NORMPATH IS LOAD-BEARING. opencode builds this with node's path.join, which
    COLLAPSES the `//` after `https:`; python's os.path.join does not. So the
    canonical spec lands at
    `<packages>/opencode-goal-plugin@https:/github.com/sblattj/OpenCode-goal-plugin/archive/refs/tags/v0.11.0.tar.gz`
    with a SINGLE slash after `https:`, and a path built without normpath points
    at a directory that does not exist - which reads as "nothing to purge" and
    as "the plugin is not installed".
    """
    return os.path.normpath(os.path.join(cache_packages_root(), spec))


def purgeable(spec, path):
    """Refuse to delete anything that is not plainly one of OUR cache dirs.

    Three independent gates, all of which must hold:
      1. the path is strictly under <cache>/opencode/packages/ and walks no `..`;
      2. the SPEC it came from is one of ours (is_goal_spec / the package name);
      3. the package name appears as a PATH COMPONENT under packages/.

    (3) is deliberately not "the LEAF is the package name". A spec's slashes
    become directories, so the tarball form's leaf is a TAG
    (`v0.11.0.tar.gz`) and its first component is `opencode-goal-plugin@https:`,
    while the pre-v1.30.1 `github:` form's leaf is `OpenCode-goal-plugin#v0.9.1`
    and its first component is `github:sblattj`. A rule anchored to either end
    alone refuses half the directories this exists to clean. `packages/foo`
    matches nothing under any of them.
    """
    root = cache_packages_root()
    p = os.path.normpath(path)
    if not p.startswith(root + os.sep):
        return False
    rel = [s for s in os.path.relpath(p, root).split(os.sep) if s]
    if not rel or os.pardir in rel:
        return False
    marker = GOAL_PLUGIN_PKG.lower()
    if not (is_goal_spec(spec) or marker in str(spec).lower()):
        return False
    return any(marker in seg.lower() for seg in rel)


def purge_cache(specs):
    """Remove the per-spec directory of each spec — and ONLY that directory.

    NOT the whole `opencode-goal-plugin@https:` subtree: every tarball spec of
    this package shares that first component, so wiping it while migrating a
    stale `...v0.10.0.tar.gz` entry would also delete an already-good
    `...v0.11.0.tar.gz` install and leave an offline laptop with nothing. The
    exact per-spec directory is the minimal correct unit; empty scaffolding is
    pruned below.
    """
    removed = []
    root = cache_packages_root()
    for spec in specs:
        d = cache_dir_for(spec)
        if not purgeable(spec, d) or not os.path.isdir(d):
            continue
        shutil.rmtree(d, ignore_errors=True)
        if os.path.exists(d):
            continue
        removed.append(d)
        # Prune the now-empty scaffolding a nested spec left behind
        # (`packages/https:/github.com/...`). rmdir refuses a non-empty
        # directory, so a sibling install stops the walk on its own.
        parent = os.path.dirname(d)
        while parent.startswith(root + os.sep):
            try:
                os.rmdir(parent)
            except OSError:
                break
            parent = os.path.dirname(parent)
    return removed

# opencode 1.18.23 ships SEVEN built-in agents. Verified two ways so a future
# rename gets caught: the published schema's $defs.Config.properties.agent names
# exactly plan/build/general/explore/title/summary/compaction, and the installed
# binary contains each of those strings. `scout` is in NEITHER (0 occurrences in
# the 144MB binary) — ferry pinned it for months and the pin did nothing, because
# an unknown key just lands in `agent`'s additionalProperties and is never read.
DRIVER_AGENTS = ("build", "plan")
EXPLORE_AGENTS = ("explore",)
COMPACTION_AGENTS = ("compaction",)
HOUSE_AGENTS = ("title", "summary")

# --- The fan-out worker, split in two and banded by complexity. ---
# opencode's task tool takes NO model parameter: the driver dispatches by AGENT
# NAME, and each agent is pinned to exactly one lane in this config. The only
# thing the driver sees at dispatch time is the task tool's own description,
# which lists every non-primary agent as "- <name>: <description>" — so the
# complexity band has to live IN the description or the driver has nothing to
# route on. Hence one cheap worker (light) and one capable worker (standard),
# each carrying its band in prose.
#
# `general` is DISABLED rather than deleted. Dropping the key does not remove
# the agent: opencode ships `general` as a built-in, and an unpinned built-in
# reappears inheriting the primary model — i.e. every fan-out task would land
# on the expensive driver lane. `{"disable": true}` is the only way to take it
# off the task tool's menu.
LIGHT_AGENTS = ("light",)
STANDARD_AGENTS = ("standard",)
DISABLED_AGENTS = ("general",)
LIGHT_DESC = "Worker for tasks rated 0-50 of 100 complexity: exploration follow-ups, small fixes, easy implementation, mechanical edits with clear instructions. Full tool access. Default worker; use standard only when the task clearly needs deeper judgment."
STANDARD_DESC = "Worker for tasks rated 51-100 of 100 complexity: multi-file implementation, ambiguous debugging, design judgment. Full tool access. Use light for anything rated 50 or below."

# --- Role lanes plus the selectable medium lane. Never a real model id. ---
# The local lanes cap KV at 131072 (128k) tokens, so a 100k-token prompt plus
# opencode's 32k output reservation tips over into a clean 400 (max_tokens is
# reserved against the KV budget). 8k output keeps prompts up to ~123k
# admissible; a compaction summary never needs 32k anyway.
#
# `modalities` is NOT decoration. opencode gates attachments on it: a custom
# provider's model has capabilities.input.image == false unless its config
# entry says `modalities.input` includes "image" (models.dev knows nothing
# about a ferry lane, so there is no fallback). With the flag false, opencode
# still runs its Read tool on a pasted screenshot — and then REPLACES the image
# with the text `ERROR: Cannot read "x.png" (this model does not support image
# input). Inform the user.` before the request leaves the laptop. The lane
# behind `heavy` reads images fine; the model was told it could not. Measured
# 2026-09-05 (opencode 1.18.29) by capturing the bytes on the wire: without the
# declaration the request carried that ERROR line, with it the request carried
# an image_url part and GPT-6 Astra described the picture. Same for PDF, which
# the front passes as an input_file. The GPU pair stays text-only: the mlx
# servers behind local-orch/local-sub take no image input, and declaring one
# would send bytes they reject instead of the placeholder they now get.
# --- Query the host catalogue; never populate a config FROM it. ---
# The catalogue does NOT advertise the fallback deployments: they route by name
# but are not `public`, so they never appear in /v1/models. Those are reached by
# the ROUTER on overflow, not by a client picking one out of a menu, so they stay
# out of the config.
served = []
try:
    # An authed front door rejects a bare catalogue request, which would read
    # as "host down" and wire the lane pair unchecked — so carry the same
    # bearer the generated configs will use. No key => no header (unchanged).
    req = urllib.request.Request(f"{base}/models")
    if oc_key:
        req.add_header("Authorization", "Bearer %s" % oc_key)
    with urllib.request.urlopen(req, timeout=4) as r:
        served = [m.get("id") for m in json.load(r).get("data", []) if m.get("id")]
except Exception as e:
    print(f"    (Could not query {base}/models: {e}; wiring the lane pair unchecked)")

# A modern cloud host exposes `medium`, which carries the `standard` worker.
# When that capability is absent (or cannot be checked), standard joins light
# and explore on flash; compaction and housekeeping use super-flash. The GPU
# pair deliberately stays exactly as it was — it has only two lanes, so light,
# standard and explore all share local-sub.
if prefer_local:
    driver, light, standard, explore, compaction, house = (
        "local-orch", "local-sub", "local-sub", "local-sub", "local-sub",
        "local-sub")
    limits = {"limit": {"context": 131072, "output": 8192}}
else:
    driver, light, explore, house = "heavy", "flash", "flash", "super-flash"
    standard = "medium" if "medium" in served else "flash"
    compaction = house
    limits = {"modalities": {"input": ["text", "image", "pdf"],
                             "output": ["text"]}}
driver = force_model or driver
# --small-model (and therefore --super) moves the whole fan-out together: the
# band split is about which worker the driver PICKS, not about keeping two
# different lanes alive when the operator asked for one.
light = force_small or light
standard = force_small or standard
explore = force_small or explore
compaction = force_house or compaction
house = force_house or house

# --- Validate the selected lanes against the host catalogue. ---
if served:
    # Public selected lanes are checked. A selected lane that also backs
    # title/summary is exempt: some compatible hosts omit it from their public
    # catalogue, and the config must remain usable for their scheduled agents.
    # The same exemption covers local-sub and --super, where all non-driver
    # agents deliberately share that lane.
    missing = [l for l in dict.fromkeys((driver, light, standard, explore,
                                         compaction))
               if l not in served and l != house]
    if missing:
        print(f"    WARNING: host does not serve {', '.join(missing)}.")
        print(f"    Catalogue: {', '.join(served)}")

# --- Snapshot: the whole original, verbatim, before we touch anything. ---
# .jsonc because opencode's schema sets allowComments/allowTrailingCommas, so a
# hand-maintained config legitimately carries comments that json.dump cannot
# round-trip. The snapshot is where they survive.
SNAP_RE_TPL = r"^{stem}\.\d{{8}}T\d{{6}}Z(-\d+)?\.jsonc$"

def snapshot(path, keep):
    if not os.path.exists(path):
        return None
    d = os.path.dirname(path) or "."
    stem = os.path.splitext(os.path.basename(path))[0]
    ts = datetime.datetime.now(datetime.timezone.utc).strftime("%Y%m%dT%H%M%SZ")
    snap = os.path.join(d, f"{stem}.{ts}.jsonc")
    n = 1
    while os.path.exists(snap):     # two runs inside one second must not collide
        snap = os.path.join(d, f"{stem}.{ts}-{n}.jsonc")
        n += 1
    shutil.copy2(path, snap)
    if keep > 0:
        # Match only OUR snapshots: the timestamp shape, anchored to this stem.
        # A plain "{stem}.*.jsonc" glob would happily delete a user's own
        # opencode.notes.jsonc sitting in the same directory.
        pat = re.compile(SNAP_RE_TPL.format(stem=re.escape(stem)))
        olds = sorted(f for f in os.listdir(d) if pat.match(f))
        for old in olds[:-keep]:
            os.remove(os.path.join(d, old))
    return snap

# --- Load whatever is there (JSONC-tolerant). ---
cfg = None
if os.path.exists(cfg_path):
    raw = open(cfg_path).read()
    try:
        cfg = json.loads(raw)
    except json.JSONDecodeError:
        stripped = re.sub(r'("(?:\\.|[^"\\])*")|//[^\n]*|/\*.*?\*/',
                          lambda m: m.group(1) or '', raw, flags=re.S)
        stripped = re.sub(r',\s*([}\]])', r'\1', stripped)
        try:
            cfg = json.loads(stripped)
            print("    (config is JSONC; its comments survive in the snapshot, not in the rewrite)")
        except json.JSONDecodeError as e:
            print(f"    WARNING: existing config is unparseable ({e}); starting from a fresh one.")
            print("    The original is preserved verbatim in the snapshot below.")
            cfg = None

snap = snapshot(cfg_path, keep_snaps)
if cfg is None:
    cfg = {}
cfg.setdefault("$schema", SCHEMA)

# --- provider.ferry: the WIRING is ours; the picker's contents are not. ---
# npm/options/limits are rewritten every run — that is the drift this command
# exists to end. Two things here are NOT ours, and survive a takeover:
#
#   1. Extra lanes. A host config typically declares local-orch/local-sub too, so
#      the GPU pair is selectable from the picker without a hand edit. Rebuilding
#      `models` wholesale deleted them, and the deletion was invisible: the
#      command still reported success, and the lanes still resolved if you typed
#      one — they were just gone from the menu.
#   2. A hand-written `name` on any lane. It is the label a human reads in the
#      status bar, and /v1/models carries only the lane id, so ferry has no
#      better one to offer: never invented, never overwritten.
#
# A label naming the MODEL behind a lane goes stale by design — the whole point
# of the lane-name contract is that the model swaps host-side without touching a
# client config — so a label should name the lane's ROLE.
prov = cfg.setdefault("provider", {})
prev_ferry = prov.get("ferry") if isinstance(prov.get("ferry"), dict) else {}
prev_models = prev_ferry.get("models") if isinstance(prev_ferry.get("models"), dict) else {}
prev_options = prev_ferry.get("options") if isinstance(prev_ferry.get("options"), dict) else {}

# dict.fromkeys: multiple agents can share one lane, and the GPU pair has only
# two lanes, so declaring one model entry per agent would create duplicates.
models = {}
# `medium` is a cloud capability tier. Declare it when the host offers it so it
# appears in opencode's model picker and can serve the `standard` worker. Its
# resolved backend varies by fleet: domestic
# Terra accepts attachments, while international GLM-5.3 is text-only. A single
# opencode provider entry cannot vary modalities with X-Ferry-Fleet, so omit the
# declaration and preserve the safe text-only baseline across every fleet.
declared_lanes = (driver, light, standard, explore, compaction, house)
# Do not advertise a lane a pre-medium host does not serve. Explicit use still
# adds it through the selected agent lanes above, allowing a caller to request the
# new lane deliberately and receive the normal catalogue warning if absent.
if not prefer_local and "medium" in served:
    declared_lanes += ("medium",)
for lane in dict.fromkeys(declared_lanes):
    spec = {} if lane == "medium" and not prefer_local else dict(limits)
    prev_name = (prev_models.get(lane) or {}).get("name")
    if isinstance(prev_name, str) and prev_name:
        spec["name"] = prev_name
    models[lane] = spec
for lane, spec in prev_models.items():
    if lane not in models and isinstance(spec, dict):
        models[lane] = spec
extra_lanes = [l for l in models if l not in declared_lanes]

# Only baseURL/apiKey/headers are ours; every other options key a user
# hand-added (or a previous run wrote) survives untouched.
options = dict(prev_options)
# The bearer the front door expects: the master key when one is configured
# (client.json / --key), else the legacy 'local' placeholder.
options["baseURL"] = base
options["apiKey"] = oc_key or "local"
# Fleet identity, rewritten fresh every run - see front/ferry_front.py's
# resolver (docs/superpowers/specs/2026-09-04-fleets-design.md §4/§6).
# "{env:FERRY_FLEET}" is opencode's OWN env-substitution syntax; ferry must
# never resolve it, so a one-shot `FERRY_FLEET=international opencode-super`
# is read at opencode's load time, not at config-write time.
options["headers"] = {
    "X-Ferry-Client": client_name,
    "X-Ferry-Fleet": "{env:FERRY_FLEET}",
}

prov["ferry"] = {
    "npm": "@ai-sdk/openai-compatible",
    # Regenerated, not preserved: this one is DERIVED from --host, and a name
    # carried over from a previous host would label the picker with a box the
    # baseURL no longer points at.
    "name": f"Ferry ({host})",
    "options": options,
    "models": models,
}

# Raw spec strings this run rewrote, in the config AND in tui.json: those are
# the ~/.cache/opencode/packages directories nothing references any more.
migrated_from = []
goal_entry = None
tui_file = None
tui_snap = None

if set_default:
    # --- The takeover. Four keys replaced outright, one appended to. ---
    cfg["permission"] = "allow"          # schema: PermissionConfig accepts the
                                         # bare enum "ask" | "allow" | "deny"
    cfg["model"] = f"ferry/{driver}"
    # small_model follows the title/summary HOUSEKEEPER, not the fan-out workers. opencode's own schema
    # describes it as "small model to use for tasks like title generation", which
    # is the housekeeping role exactly; leaving it on light/standard/explore would send
    # every small task opencode has not got a named agent for to a fan-out lane.
    cfg["small_model"] = f"ferry/{house}"

    # Replaced WHOLESALE, not merged: a stale pin left behind here (a compaction
    # agent still naming a retired model id, say) is exactly the drift this
    # command exists to end. Anything custom is recoverable from the snapshot.
    agent = {a: {"model": f"ferry/{driver}"} for a in DRIVER_AGENTS}
    # Disabled, not deleted: a deleted key lets opencode's built-in `general`
    # come back on the primary (driver) model.
    agent.update({a: {"disable": True} for a in DISABLED_AGENTS})
    agent.update({a: {"description": LIGHT_DESC, "mode": "subagent",
                      "model": f"ferry/{light}"} for a in LIGHT_AGENTS})
    agent.update({a: {"description": STANDARD_DESC, "mode": "subagent",
                      "model": f"ferry/{standard}"} for a in STANDARD_AGENTS})
    agent.update({a: {"model": f"ferry/{explore}"} for a in EXPLORE_AGENTS})
    agent.update({a: {"model": f"ferry/{compaction}"} for a in COMPACTION_AGENTS})
    agent.update({a: {"model": f"ferry/{house}"} for a in HOUSE_AGENTS})
    cfg["agent"] = agent

    # Additive — a plugin list belongs to the user; we only ensure ours is in it.
    # Every earlier spelling of OUR entry is rewritten to the canonical spec (see
    # ensure_goal_plugin and the GOAL_PLUGIN comment block: the forms ferry wrote
    # before v1.30.1 install to disk and are then silently discarded).
    plugins, goal_entry, migrated = ensure_goal_plugin(cfg.get("plugin"))
    cfg["plugin"] = plugins
    migrated_from.extend(migrated)

    # MERGE, never take over: the plugin's /goal slash command needs a
    # top-level `command.goal` entry or it never appears in opencode. Only the
    # `goal` key is ours, and only when it is absent - a user who customised it
    # keeps their version verbatim, and no other command is touched.
    commands = cfg.get("command")
    if not isinstance(commands, dict):
        commands = {}
    if "goal" not in commands:
        commands["goal"] = dict(GOAL_COMMAND)
    cfg["command"] = commands

os.makedirs(os.path.dirname(cfg_path) or ".", exist_ok=True)
with open(cfg_path, "w") as f:
    json.dump(cfg, f, indent=2)
    f.write("\n")

# --- Mirror the plugin entry into tui.json (the sidebar half). ---
# Same migration, same dedupe, same snapshot policy as opencode.json; every
# other key in the file is left exactly as it was. The ENTRY differs, though:
# the TUI loader cannot load a module out of the colon-bearing package cache
# (see the GOAL_TUI_MARKER block), so tui.json names ferry's colon-free copy.
tui_state = tui_copy_state()
tui_copy_ready = bool(tui_state and tui_state["ok"]) and not tui_dir_unusable()
raw_goal = goal_entry[0] if isinstance(goal_entry, list) and goal_entry else goal_entry
if isinstance(raw_goal, str) and is_path_entry(raw_goal) and not is_managed_tui_entry(raw_goal):
    # A user's own fork: whatever satisfies the server half satisfies the TUI
    # half, and ferry has no copy of a fork to point at.
    tui_want = raw_goal
elif tui_copy_ready:
    tui_want = goal_tui_spec()
else:
    # BOOTSTRAP form: nothing is installed yet, so there is nothing to copy.
    # _ferry_sync_goal_tui_copy replaces this with the file:// URL later in
    # this same run, right after the pre-install populates the cache.
    tui_want = GOAL_PLUGIN
if set_default:
    tui_file = tui_target(cfg_path, tui_mode, tui_path)
if tui_file:
    tui_cfg = load_jsonc(tui_file, tui_file)
    if not isinstance(tui_cfg, dict):
        tui_cfg = {}
    tui_snap = snapshot(tui_file, keep_snaps)
    tui_plugins, _tui_entry, tui_migrated = ensure_goal_plugin(tui_cfg.get("plugin"), tui_want)
    migrated_from.extend(tui_migrated)
    tui_cfg.setdefault("$schema", TUI_SCHEMA)
    tui_cfg["plugin"] = tui_plugins
    os.makedirs(os.path.dirname(tui_file) or ".", exist_ok=True)
    with open(tui_file, "w") as f:
        json.dump(tui_cfg, f, indent=2)
        f.write("\n")

# --- Purge the cache directories this run just orphaned. ---
purged = []
if set_default and not keep_cache:
    specs = list(dict.fromkeys(list(migrated_from) + list(DEAD_CACHE_SPECS)))
    purged = purge_cache(specs)
    # The canonical directory is kept when it holds a real install and removed
    # only when it does NOT: an interrupted install leaves an empty
    # node_modules/<name> that opencode's existence-only short-circuit then
    # treats as installed forever.
    canon = cache_dir_for(GOAL_PLUGIN)
    if os.path.isdir(canon) and not os.path.exists(
            os.path.join(canon, "node_modules", GOAL_PLUGIN_PKG, "package.json")):
        purged += purge_cache([GOAL_PLUGIN])

# Hand the canonical spec to the pre-install pass, but only when ferry's own
# entry is the one in play: a local fork must not trigger an upstream install.
# Lines 4-7 are for _ferry_sync_goal_tui_copy, which runs after the install and
# needs to know which tui.json was written (empty when none was), where the
# managed copy lives, which cache directory to copy FROM, and whether that
# managed path is usable at all.
if set_default and goal_entry == GOAL_PLUGIN and spec_out:
    goal_cache_root = os.path.join(cache_dir_for(GOAL_PLUGIN), "node_modules", GOAL_PLUGIN_PKG)
    with open(spec_out, "w") as f:
        f.write(f"{GOAL_PLUGIN}\n{GOAL_PLUGIN_REF}\n{GOAL_PLUGIN_PKG}\n"
                f"{tui_file or ''}\n{goal_tui_dir()}\n{goal_cache_root}\n"
                f"{'1' if tui_dir_unusable() else '0'}\n")

print(f"    Wired opencode -> {base}")
print(f"    Provider: ferry   Lanes: {driver} (driver), {light} (light), {standard} (standard), {explore} (explore), {compaction} (compaction), {house} (title/summary)")
if extra_lanes:
    print(f"    Kept in picker: {', '.join(extra_lanes)} (declared in the config, not pinned by ferry)")
if set_default:
    print(f"    model={cfg['model']}  small_model={cfg['small_model']}  permission=allow")
    print(f"    Agents pinned:  {'/'.join(DRIVER_AGENTS)} -> ferry/{driver}")
    print(f"                    {'/'.join(LIGHT_AGENTS)} -> ferry/{light}; {'/'.join(STANDARD_AGENTS)} -> ferry/{standard}")
    print(f"                    {'/'.join(EXPLORE_AGENTS)} -> ferry/{explore}; {'/'.join(DISABLED_AGENTS)} -> disabled")
    print(f"                    {'/'.join(COMPACTION_AGENTS)} -> ferry/{compaction}; {'/'.join(HOUSE_AGENTS)} -> ferry/{house}")
    # Report the entry that actually SATISFIES the requirement, not the package
    # we would have added. Printing GOAL_PLUGIN unconditionally claimed an
    # install that never happened whenever a local fork was already present.
    label = goal_entry[0] if isinstance(goal_entry, list) and goal_entry else goal_entry
    # What tui.json now names, spelled out: the file:// copy is the only form
    # whose TUI half actually loads, so "written" is not enough to tell a good
    # run from one that only wired the server half.
    if not tui_file:
        tui_note = "skipped"
    elif tui_want == GOAL_PLUGIN:
        tui_note = "spec, TUI copy pending install"
    else:
        tui_note = tui_want
    print(f"    Plugin: {label}  (/goal command wired; tui.json: {tui_note})")
    if label != GOAL_PLUGIN:
        print(f"                    (counts as {GOAL_PLUGIN}; upstream not added)")
    if tui_file and tui_dir_unusable():
        print(f"    WARNING: {goal_tui_dir()} contains ':' or '#'. Bun splits a module")
        print("             path at the first colon, so opencode's TUI loader cannot load a")
        print("             plugin from there; tui.json keeps the spec and the sidebar half")
        print("             will not appear. Set XDG_DATA_HOME to a path without ':' or '#'.")
    elif tui_file and tui_state and (tui_state["spec"] != GOAL_PLUGIN
                                     or tui_state["version"] != GOAL_PLUGIN_REF.lstrip("v")):
        stale = f"    TUI plugin: copy holds {tui_state['version']}, expected {GOAL_PLUGIN_REF.lstrip('v')}"
        print(stale + ("; refreshed after the install below" if do_install
                       else "; rerun without --no-install"))
    for d in purged:
        print(f"    Cache purged:   {d}")
else:
    print("    --no-default: provider wired; permission/model/agent left alone.")
if snap:
    print(f"    Snapshot:       {snap}")
if tui_snap:
    print(f"    Snapshot:       {tui_snap}")
print(f"    Config written: {cfg_path}")
if tui_file:
    print(f"    TUI config:     {tui_file}")
PYEOF

  # --- Pre-install the plugin through the production loader. ---
  # The block above only WROTE a spec. Nothing installs it until opencode next
  # starts, and if the install fails there it fails silently forever (see
  # _ferry_preinstall_goal_plugin). Doing it here is what turns "the config
  # looks right" into "the plugin is on disk and loadable".
  local goal_spec="" goal_ref="" goal_pkg=""
  local goal_tui_file="" goal_tui_dir="" goal_cache_root="" goal_tui_bad="0"
  if [[ -s "$oc_specfile" ]]; then
    goal_spec="$(sed -n 1p "$oc_specfile")"
    goal_ref="$(sed -n 2p "$oc_specfile")"
    goal_pkg="$(sed -n 3p "$oc_specfile")"
    goal_tui_file="$(sed -n 4p "$oc_specfile")"
    goal_tui_dir="$(sed -n 5p "$oc_specfile")"
    goal_cache_root="$(sed -n 6p "$oc_specfile")"
    goal_tui_bad="$(sed -n 7p "$oc_specfile")"
  fi
  rm -f "$oc_specfile"

  # The doctrine for the plugin this run just wired. Gated on $goal_spec for the
  # same reason the pre-install below is: a non-empty spec means "this run wrote
  # a real config AND ferry's own goal entry is the one in play" — --no-default
  # and a pre-existing local fork both leave it empty, and neither should get a
  # skill describing wiring ferry did not do. Deliberately NOT gated on
  # (( do_install )): copying one file out of the checkout is local work, while
  # --no-install is about skipping the network fetch of the plugin package.
  [[ -n "$goal_spec" ]] && _ferry_install_goal_skill

  if (( do_install )) && [[ -n "$goal_spec" ]]; then
    if command -v opencode >/dev/null 2>&1; then
      _ferry_preinstall_goal_plugin "$goal_spec" "$goal_ref" "$goal_pkg"
      # Runs after the pre-install whether or not it warned: the sync re-checks
      # the cache for itself, and the TUI half is worthless until the copy
      # exists. It is what turns the bootstrap spec in tui.json into a file://
      # entry the TUI loader can actually load. Skipped only when the managed
      # path is itself unloadable (a ':' or '#' in $XDG_DATA_HOME) — the python
      # block above has already reported that, and the copy would be useless.
      if [[ "$goal_tui_bad" != "1" ]]; then
        _ferry_sync_goal_tui_copy "$goal_spec" "$goal_ref" "$goal_pkg" \
          "$goal_tui_file" "$goal_tui_dir" "$goal_cache_root"
      fi
    else
      echo "    Plugin:  opencode is not on PATH; skipping the pre-install."
      echo "             It will be fetched the first time opencode starts."
    fi
  fi
  return 0
}
# ferry claude — point Claude Code at the ferry endpoint by lane name.
#
# LiteLLM already serves Anthropic /v1/messages for every lane, so Claude Code
# needs only four env vars and two lane picks — no config file, no plugin. The
# cloud pair is heavy/flash (driver/worker), the GPU pair local-orch/local-sub;
# those are the same two roles `ferry opencode` wires, under Claude's own names:
# the main model, the haiku-slot (background/housekeeping calls), and the
# subagent model.
#
# Host and port are BAKED into the wrappers at install time on purpose: the
# function must work with ferry down, so there is no runtime lookup to fail.
# Re-running `ferry claude --host ...` rewrites them; that is the whole point of
# the marker strip below.

FERRY_CL_MARK_START="# >>> ferry claude profiles >>>"
FERRY_CL_MARK_END="# <<< ferry claude profiles <<<"

_ferry_install_claude_wrappers() {
  local cl_host="$1" cl_port="$2" cl_key="${3:-}"
  # Empty / absent key keeps the legacy 'local' bearer, so a front door without
  # a litellm master_key is unchanged. Anything else is baked in verbatim.
  [[ -z "$cl_key" ]] && cl_key="local"
  local rc="$HOME/.zshrc"
  touch "$rc"

  # Strip the canonical block, then re-add. Legacy `alias claude-ferry=` /
  # `alias claude-ferry-local=` / `alias claude-ferry-super=` lines go too: an
  # alias ABOVE a function of the same name makes zsh expand the alias inside
  # `name() {`, a parse error on every subsequent `source ~/.zshrc` (same
  # footgun client-bootstrap.sh strips).
  python3 - "$rc" "$FERRY_CL_MARK_START" "$FERRY_CL_MARK_END" <<'PYEOF'
import sys
rc, start, end = sys.argv[1], sys.argv[2], sys.argv[3]

with open(rc) as f:
    lines = f.readlines()

out, skip = [], False
for ln in lines:
    s = ln.rstrip("\n")
    if s == start:
        skip = True
        continue
    if s == end:
        skip = False
        continue
    if not skip:
        out.append(ln)

def is_legacy_alias(l):
    t = l.lstrip()
    return (t.startswith("alias claude-ferry=")
            or t.startswith("alias claude-ferry-local=")
            or t.startswith("alias claude-ferry-super="))

out = [l for l in out if not is_legacy_alias(l)]

while out and out[-1].strip() == "":
    out.pop()
with open(rc, "w") as f:
    f.writelines(out)
    if out:
        f.write("\n")
PYEOF
  if (( $? != 0 )); then
    echo "    WARNING: could not rewrite $rc; leaving the claude wrappers alone." >&2
    return 1
  fi

  # QUOTED heredoc, so the $HOME and $@ inside the functions survive verbatim.
  # Host/port/key ride as placeholder tokens and are baked in by the
  # substitution right after — a quoted heredoc cannot interpolate them itself.
  local block
  block=$(cat <<'EOF'
# >>> ferry claude profiles >>>
# Installed by `ferry claude` / host-reset.sh. ANTHROPIC_BASE_URL carries NO
# /v1 suffix: Claude Code appends /v1/messages itself, and LiteLLM already
# serves that path for every lane. AUTH_TOKEN is 'local' unless the front door
# runs litellm with a master_key (LITELLM_MASTER_KEY) — `ferry claude` bakes
# the real key from ~/.config/ferry/client.json's master_key (or --key) when
# one is configured, else the 'local' placeholder a keyless LAN setup expects.
#
# Host and port are baked in at install time: the wrapper must work with ferry
# down, so there is no runtime resolution to fail.
#
# There is deliberately no bare `claude` function — that is the user's personal
# tool, and wrapping it would hijack sessions they never asked to route
# anywhere. Env is scoped with `env`, so it reaches the claude child process
# only and never leaks into the interactive shell.
unalias claude-ferry claude-ferry-local claude-ferry-super 2>/dev/null

# claude-ferry: the CLOUD pair — heavy drives, flash runs subagents and the
# haiku-slot background calls.
claude-ferry() {
  local nl=$'\n'
  env ANTHROPIC_BASE_URL="http://__FERRY_CL_HOST__:__FERRY_CL_PORT__" \
      ANTHROPIC_AUTH_TOKEN=__FERRY_CL_KEY__ \
      ANTHROPIC_CUSTOM_HEADERS="X-Ferry-Client: __FERRY_CL_NAME__${FERRY_FLEET:+${nl}X-Ferry-Fleet: $FERRY_FLEET}" \
      ANTHROPIC_MODEL=heavy \
      ANTHROPIC_DEFAULT_HAIKU_MODEL=flash \
      CLAUDE_CODE_SUBAGENT_MODEL=flash \
      CLAUDE_CODE_MAX_OUTPUT_TOKENS=32000 \
      CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS=1 \
      CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC=1 \
      command claude "$@"
}

# claude-ferry-local: the GPU pair — local-orch drives, local-sub fans out.
# Nothing leaves this machine. DISABLE_THINKING=1 because reasoning tokens
# count against max_tokens AND the GPU lanes' KV budget, which is the same
# reason `ferry opencode` caps local-lane output at 8k.
claude-ferry-local() {
  local nl=$'\n'
  env ANTHROPIC_BASE_URL="http://__FERRY_CL_HOST__:__FERRY_CL_PORT__" \
      ANTHROPIC_AUTH_TOKEN=__FERRY_CL_KEY__ \
      ANTHROPIC_CUSTOM_HEADERS="X-Ferry-Client: __FERRY_CL_NAME__${FERRY_FLEET:+${nl}X-Ferry-Fleet: $FERRY_FLEET}" \
      ANTHROPIC_MODEL=local-orch \
      ANTHROPIC_DEFAULT_HAIKU_MODEL=local-sub \
      CLAUDE_CODE_SUBAGENT_MODEL=local-sub \
      CLAUDE_CODE_MAX_OUTPUT_TOKENS=32000 \
      CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS=1 \
      CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC=1 \
      CLAUDE_CODE_DISABLE_THINKING=1 \
      command claude "$@"
}

# claude-ferry-super: heavy drives; super-flash covers background tasks AND
# subagents. The cheapest cloud profile.
claude-ferry-super() {
  local nl=$'\n'
  env ANTHROPIC_BASE_URL="http://__FERRY_CL_HOST__:__FERRY_CL_PORT__" \
      ANTHROPIC_AUTH_TOKEN=__FERRY_CL_KEY__ \
      ANTHROPIC_CUSTOM_HEADERS="X-Ferry-Client: __FERRY_CL_NAME__${FERRY_FLEET:+${nl}X-Ferry-Fleet: $FERRY_FLEET}" \
      ANTHROPIC_MODEL=heavy \
      ANTHROPIC_DEFAULT_HAIKU_MODEL=super-flash \
      CLAUDE_CODE_SUBAGENT_MODEL=super-flash \
      CLAUDE_CODE_MAX_OUTPUT_TOKENS=32000 \
      CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS=1 \
      CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC=1 \
      command claude "$@"
}
# <<< ferry claude profiles <<<
EOF
)
  block="${block//__FERRY_CL_HOST__/$cl_host}"
  block="${block//__FERRY_CL_PORT__/$cl_port}"
  block="${block//__FERRY_CL_KEY__/$cl_key}"
  block="${block//__FERRY_CL_NAME__/$CLIENT_NAME}"
  print -r -- "$block" >> "$rc"

  echo ">>> claude shell wrappers installed in $rc:"
  echo "    claude-ferry        -> cloud pair: heavy drives, flash fans out"
  echo "    claude-ferry-local  -> GPU pair:   local-orch drives, local-sub fans out"
  echo "    claude-ferry-super  -> super pair: heavy drives, super-flash covers the rest"
  echo "    (bare 'claude' is untouched — run: source $rc)"
}

cmd_claude() {
  # Wire Claude Code to the ferry endpoint: install the claude-ferry[-local]
  # shell wrappers AND write ~/.config/ferry/claude.json recording which lane
  # plays which role. The JSON is the machine-readable twin of the wrappers —
  # other tooling (host-reset.sh, tests) reads the mapping instead of parsing
  # zshrc — so both are written in one pass and always agree.
  local cl_host="" cl_port="" cl_key="" _wrappers_only=0

  while [[ $# -gt 0 ]]; do
    case "$1" in
      --host)     cl_host="$2"; shift 2 ;;
      --port)     cl_port="$2"; shift 2 ;;
      --key)      cl_key="$2"; shift 2 ;;
      --wrappers) _wrappers_only=1; shift ;;
      --help|-h)
        cat <<'EOF'
ferry claude — point Claude Code at the ferry endpoint by lane name.

Usage:
  ferry claude [--host H] [--port P] [--key K] [--wrappers]

  (no flags)   Install the ~/.zshrc wrappers (claude-ferry / claude-ferry-local)
               and write ~/.config/ferry/claude.json recording the lane map.
               Host resolves from --host, else ~/.config/ferry/client.json;
               on a host machine (no client.json) it defaults to 127.0.0.1:8090.
  --wrappers   Install ONLY the zshrc wrappers (used by host-reset.sh).
  --host H     Endpoint host to bake into the wrappers.
  --port P     Endpoint port (default 8090).
  --key K      Bearer token baked into the wrappers and sent to the front door.
               Default: client.json's master_key when set, else 'local'.

Lane map:  cloud  main=heavy   background=flash
           local  main=local-orch  background=local-sub
EOF
        return 0 ;;
      *) echo "Unknown option for 'ferry claude': $1"; exit 1 ;;
    esac
  done

  # Resolve host: --host wins, else the client profile. Read client.json fresh
  # rather than trusting the boot-time CLIENT_HOST: `ferry claude --wrappers`
  # runs from host-reset.sh, which may run before/outside a normal CLI boot.
  if [[ -z "$cl_host" || -z "$cl_port" ]]; then
    local prof ph pp
    prof=$(python3 -c "import json, os, sys
try:
    d = json.load(open(os.path.expanduser(sys.argv[1])))
    print(d.get('host') or '', d.get('port') or '')
except Exception:
    pass" "$HOME/.config/ferry/client.json" 2>/dev/null)
    read -r ph pp <<< "$prof" 2>/dev/null
    if [[ -z "$cl_host" ]]; then
      cl_host="${ph:-}"
      if [[ -z "$cl_host" ]]; then
        if (( CLIENT_MODE )); then
          echo "Error: ~/.config/ferry/client.json has no 'host'. Re-run the client"
          echo "bootstrap, or pass --host <mdns-or-ip> explicitly."
          exit 1
        fi
        # HOST: the proxy is on this very machine, so loopback is the right
        # default — same reasoning as `ferry opencode`.
        cl_host="127.0.0.1"
        echo ">>> No --host and no client profile: this is the HOST, so wiring"
        echo "    Claude Code to its own proxy at http://127.0.0.1:8090."
      fi
    fi
    if [[ -z "$cl_port" ]]; then
      cl_port="$pp"
    fi
  fi
  cl_port="${cl_port:-8090}"

  # Key precedence: --key, else the client profile's optional master_key, else
  # empty (the installer then bakes the legacy 'local' token). Read fresh from
  # client.json — same reason host/port are — because `ferry claude --wrappers`
  # runs from host-reset.sh, which may run before/outside a normal CLI boot.
  # The value only lands in files, never on stdout.
  if [[ -z "$cl_key" ]]; then
    cl_key=$(python3 -c "import json, os, sys
try:
    print(json.load(open(os.path.expanduser(sys.argv[1]))).get('master_key') or '')
except Exception:
    pass" "$HOME/.config/ferry/client.json" 2>/dev/null)
  fi
  cl_key="${cl_key:-${CLIENT_MASTER_KEY:-}}"

  if (( _wrappers_only )); then
    _ferry_install_claude_wrappers "$cl_host" "$cl_port" "$cl_key"
    return $?
  fi

  _ferry_install_claude_wrappers "$cl_host" "$cl_port" "$cl_key"

  # Snapshot any existing claude.json before overwrite, so a lane re-mapping is
  # always reversible (same convention as `ferry opencode`'s config snapshots).
  python3 - "$cl_host" "$cl_port" "$cl_key" "$HOME/.config/ferry/claude.json" <<'PYEOF'
import datetime, json, os, shutil, sys

host, port, path = sys.argv[1], sys.argv[2], os.path.expanduser(sys.argv[4])
key = sys.argv[3]
if os.path.exists(path):
    ts = datetime.datetime.now(datetime.timezone.utc).strftime("%Y%m%dT%H%M%SZ")
    snap = f"{path}.{ts}.bak"
    n = 1
    while os.path.exists(snap):     # two runs inside one second must not collide
        snap = f"{path}.{ts}-{n}.bak"
        n += 1
    shutil.copy2(path, snap)
    print(f"    Snapshot:       {snap}")

cfg = {
    "host": host,
    "port": port,
    "lanes": {
        "cloud": {"main": "heavy", "background": "flash"},
        "local": {"main": "local-orch", "background": "local-sub"},
    },
}
# Mirrored only when it is a real key: 'local' (or empty) is the keyless LAN
# default, and recording it would make the mirror claim an auth setup that
# does not exist.
if key and key != "local":
    cfg["master_key"] = key
os.makedirs(os.path.dirname(path), exist_ok=True)
with open(path, "w") as f:
    json.dump(cfg, f, indent=2)
    f.write("\n")
print(f"    Wired claude    -> http://{host}:{port}")
print(f"    Lanes: cloud main=heavy background=flash | local main=local-orch background=local-sub")
print(f"    Config written: {path}")
PYEOF

  if ! command -v claude >/dev/null 2>&1; then
    # Informational, not fatal: the wrappers are installed so they are ready
    # the moment Claude Code is.
    echo "    NOTE: Claude Code isn't installed on this machine yet — wrappers are"
    echo "    in place regardless, so nothing more is needed once it is."
  fi
}
# ferry auth-claude — Claude Pro/Max subscription OAuth (browser PKCE).
#
# Wraps the Python OAuth engine (front/ferry_claude_oauth.py, Path A) so an
# operator can run `ferry auth-claude login|status|refresh|logout`. The token
# JSON lives in ferry's own config dir, NOT litellm's: the ChatGPT lane keeps
# ~/.config/litellm/chatgpt/auth.json, and this is its Claude twin at
# ~/.config/ferry/claude/auth.json — mode 0600, readable by this account only.
#
# The access/refresh tokens NEVER reach stdout: every line this module prints
# is derived (email, expiry, state, paths), and the only reader of the raw
# JSON is the summary helper, which emits exactly those safe fields.
#
# Cross-seat contract with front/ferry_claude_oauth.py (the engine seat):
#   login        argv `login --output <path>`; the FERRY_CLAUDE_AUTH_JSON env
#                var carries the same path. Runs the browser PKCE flow, writes
#                the token JSON (access_token / refresh_token / expires_at /
#                email or id_token), exits 0.
#   ensure-valid argv `ensure-valid --auth <path> --force`; same env var.
#                Forces a refresh, persisting the ROTATED refresh_token in
#                place at <path>; exits 0 on success, nonzero when the grant
#                is dead (re-run `ferry auth-claude login`).
# The zsh side never parses the engine's stdout for data — it re-reads the
# token file after every engine call, so the two sides evolve independently.

# Where the subscription tokens live. Overridable via the environment (tests,
# alternate profiles) — same override pattern as FERRY_SCHEMATRON_PORT.
FERRY_CLAUDE_AUTH_JSON="${FERRY_CLAUDE_AUTH_JSON:-$HOME/.config/ferry/claude/auth.json}"
# The OAuth engine script. Empty = resolve from $APP_DIR at call time, so
# loading this module never fails when front/ is not deployed yet.
FERRY_CLAUDE_OAUTH_PY="${FERRY_CLAUDE_OAUTH_PY:-}"

# _ferry_litellm_python — the venv interpreter that owns litellm. Reuses the
# serve module's resolver when it is loaded (the built monolith always loads
# it); otherwise the identical litellm-bin resolution inline, because this
# module must also stand alone with only ferry-core.zsh sourced (the test
# harness does exactly that). `litellm` is installed as a uv tool, so the
# python sitting next to the resolved binary is the only interpreter whose
# site-packages can import the engine's dependencies.
_ferry_litellm_python() {
  if (( $+functions[_ferry_front_python] )); then
    _ferry_front_python
    return $?
  fi
  local bin real py
  bin="$(command -v litellm 2>/dev/null)" || return 1
  [[ -n "$bin" ]] || return 1
  real="$(python3 -c 'import os,sys; print(os.path.realpath(sys.argv[1]))' "$bin" 2>/dev/null)" || return 1
  py="${real:h}/python"
  [[ -x "$py" ]] || return 1
  print -r -- "$py"
}

# _ferry_claude_oauth_script — the engine script path. Explicit
# FERRY_CLAUDE_OAUTH_PY wins (tests, side-by-side engines); otherwise resolve
# from $APP_DIR. Sourcing the modules from lib/ makes APP_DIR the lib dir
# (ferry-core computes it from $0), and the engine lives one level up in
# front/, so try the parent when the direct guess misses.
_ferry_claude_oauth_script() {
  if [[ -n "$FERRY_CLAUDE_OAUTH_PY" ]]; then
    print -r -- "$FERRY_CLAUDE_OAUTH_PY"
    return 0
  fi
  local d="$APP_DIR"
  [[ -f "$d/front/ferry_claude_oauth.py" ]] || d="${d:h}"
  print -r -- "$d/front/ferry_claude_oauth.py"
}

# _ferry_claude_engine — shared preflight for login/refresh: resolve the venv
# python and the engine script, verify both, print "python\nscript" on
# success. Fails with an operator-readable message on stderr otherwise.
_ferry_claude_engine() {
  local py script
  if ! py="$(_ferry_litellm_python)"; then
    echo "Error: no litellm venv on PATH — the OAuth engine needs its python." >&2
    echo "Is ferry installed here? Run: ferry install" >&2
    return 1
  fi
  script="$(_ferry_claude_oauth_script)"
  if [[ ! -f "$script" ]]; then
    echo "Error: OAuth engine missing: $script" >&2
    return 1
  fi
  print -r -- "$py"
  print -r -- "$script"
}

# _ferry_claude_auth_summary — the ONLY reader of the raw token JSON. Prints
# exactly three lines — email, expiry (UTC ISO), state (valid|expired|
# unknown) — and never the tokens themselves. A missing/unparsable file
# reports "unknown" fields rather than failing, so no caller ever sees a
# traceback or a secret on an error path.
_ferry_claude_auth_summary() {
  python3 - "$1" <<'PYEOF'
import base64, datetime, json, sys

try:
    with open(sys.argv[1]) as f:
        d = json.load(f)
except Exception:
    print("unknown")
    print("unknown")
    print("missing")
    raise SystemExit(0)

def jwt_email(tok):
    # Unverified payload peek ONLY to surface the account's email address;
    # the token itself is never emitted anywhere.
    try:
        payload = tok.split(".")[1]
        payload += "=" * (-len(payload) % 4)
        return json.loads(base64.urlsafe_b64decode(payload)).get("email") or ""
    except Exception:
        return ""

email = (d.get("email") or d.get("account_email")
         or jwt_email(d.get("id_token") or "") or d.get("account_id") or "unknown")
exp = d.get("expires_at")
if isinstance(exp, (int, float)) and exp > 0:
    iso = datetime.datetime.fromtimestamp(exp, datetime.timezone.utc).strftime("%Y-%m-%d %H:%M:%SZ")
    state = ("expired" if exp <= datetime.datetime.now(datetime.timezone.utc).timestamp()
             else "valid")
else:
    iso, state = "unknown", "unknown"
print(email)
print(iso)
print(state)
PYEOF
}

_ferry_auth_claude_usage() {
  cat <<'EOF'
ferry auth-claude — Claude Pro/Max subscription OAuth (browser PKCE).

Usage:
  ferry auth-claude login             Run the browser PKCE flow and store the
                                      tokens at ~/.config/ferry/claude/auth.json
                                      (mode 0600). Prints the account email.
  ferry auth-claude status            Show account, expiry, and VALID/EXPIRED.
                                      Exits 1 when logged out or expired.
                                      Never prints tokens.
  ferry auth-claude refresh           Force a token refresh and persist the
                                      rotated refresh token in place.
  ferry auth-claude logout [--force]  Delete the stored credentials; --force
                                      skips the 'yes' confirmation.

The token path defaults to ~/.config/ferry/claude/auth.json; redirect it with
the FERRY_CLAUDE_AUTH_JSON environment variable.
EOF
}

# _ferry_claude_print_summary — the shared status tail: account / expiry /
# state lines from the summary helper. Echoes the derived fields only.
_ferry_claude_print_summary() {
  local summary email="unknown" iso="unknown" state="unknown"
  summary="$(_ferry_claude_auth_summary "$FERRY_CLAUDE_AUTH_JSON" 2>/dev/null)"
  { read -r email && read -r iso && read -r state; } <<< "$summary" || true
  echo "    Account: $email"
  echo "    Expires: $iso"
  case "$state" in
    valid)   echo "    State:   VALID (token usable)"; return 0 ;;
    expired) echo "    State:   EXPIRED (run: ferry auth-claude refresh)"; return 1 ;;
    *)       echo "    State:   unknown (no parsable expires_at)"; return 0 ;;
  esac
}

cmd_auth_claude() {
  # `ferry auth-claude` — manage the Claude Pro/Max subscription credentials.
  # All token material stays inside $FERRY_CLAUDE_AUTH_JSON; stdout carries
  # only email, expiry, state, and paths.
  local sub="${1:-}"

  if [[ "$sub" == "--help" || "$sub" == "-h" ]]; then
    _ferry_auth_claude_usage
    return 0
  fi
  if [[ -z "$sub" ]]; then
    _ferry_auth_claude_usage
    return 1
  fi
  shift

  local force=0
  while [[ $# -gt 0 ]]; do
    case "$1" in
      --force)    force=1; shift ;;
      --help|-h)  _ferry_auth_claude_usage; return 0 ;;
      *) echo "Unknown option for 'ferry auth-claude $sub': $1" >&2; exit 1 ;;
    esac
  done

  case "$sub" in
    login)
      local py script
      { read -r py && read -r script; } <<< "$(_ferry_claude_engine)" || exit 1
      mkdir -p "${FERRY_CLAUDE_AUTH_JSON:h}"
      # The engine inherits the terminal so it can open the browser and print
      # its own progress. The token path rides BOTH as --output and as the
      # FERRY_CLAUDE_AUTH_JSON env var (the engine may read either).
      FERRY_CLAUDE_AUTH_JSON="$FERRY_CLAUDE_AUTH_JSON" \
        "$py" "$script" login --output "$FERRY_CLAUDE_AUTH_JSON" || exit 1
      if [[ ! -f "$FERRY_CLAUDE_AUTH_JSON" ]]; then
        echo "Error: the OAuth engine exited 0 but wrote no token file at" >&2
        echo "$FERRY_CLAUDE_AUTH_JSON" >&2
        exit 1
      fi
      # Enforce the mode even when the engine forgot: this file IS the
      # credential. Status lines come from the safe summary, never the engine
      # stdout (which is not parsed for data).
      chmod 600 "$FERRY_CLAUDE_AUTH_JSON"
      echo ">>> Claude subscription authorized."
      _ferry_claude_print_summary
      echo "    Tokens:  $FERRY_CLAUDE_AUTH_JSON (0600)"
      ;;

    status)
      if [[ ! -f "$FERRY_CLAUDE_AUTH_JSON" ]]; then
        echo "Not logged in (no $FERRY_CLAUDE_AUTH_JSON)."
        echo "Run: ferry auth-claude login"
        return 1
      fi
      echo ">>> Claude subscription auth: $FERRY_CLAUDE_AUTH_JSON"
      _ferry_claude_print_summary
      ;;

    refresh)
      if [[ ! -f "$FERRY_CLAUDE_AUTH_JSON" ]]; then
        echo "Not logged in (no $FERRY_CLAUDE_AUTH_JSON) — nothing to refresh."
        echo "Run: ferry auth-claude login"
        return 1
      fi
      local py script
      { read -r py && read -r script; } <<< "$(_ferry_claude_engine)" || exit 1
      # --force: the operator asked for a refresh, so refresh — not "maybe".
      # The engine must persist the ROTATED refresh_token in place or the
      # next refresh uses a dead token.
      FERRY_CLAUDE_AUTH_JSON="$FERRY_CLAUDE_AUTH_JSON" \
        "$py" "$script" ensure-valid --auth "$FERRY_CLAUDE_AUTH_JSON" --force || exit 1
      chmod 600 "$FERRY_CLAUDE_AUTH_JSON"
      echo ">>> Refreshed the Claude subscription token."
      _ferry_claude_print_summary
      ;;

    logout)
      if [[ ! -f "$FERRY_CLAUDE_AUTH_JSON" ]]; then
        echo "Not logged in (no $FERRY_CLAUDE_AUTH_JSON) — nothing to remove."
        return 0
      fi
      if (( ! force )); then
        local answer=""
        echo "Remove the Claude subscription credentials at"
        echo "  $FERRY_CLAUDE_AUTH_JSON"
        printf "Type 'yes' to confirm: "
        if ! read -r answer; then
          echo ""
          echo "No confirmation available (stdin closed) — pass --force." >&2
          return 1
        fi
        if [[ "$answer" != "yes" ]]; then
          echo "Aborted — credentials left in place."
          return 1
        fi
      fi
      rm -f "$FERRY_CLAUDE_AUTH_JSON"
      echo ">>> Removed $FERRY_CLAUDE_AUTH_JSON (logged out)."
      ;;

    *)
      echo "Unknown subcommand for 'ferry auth-claude': $sub" >&2
      echo "Run: ferry auth-claude --help" >&2
      exit 1
      ;;
  esac
}
# ferry fleet — read or switch which FLEET (routing set) a caller resolves
# bare lane names against. Talks to the front door's control plane at
# /v1/ferry/fleet (front/ferry_front.py): GET returns the fleet document,
# POST mutates the caller's own sticky selection, or (host only, --default)
# the host-wide default. Identity and auth follow the same rules as every
# other ferry client command: CLIENT_MODE=1 talks to
# $CLIENT_HOST:$CLIENT_PORT with $CLIENT_MASTER_KEY; CLIENT_MODE=0 (the host)
# talks to its own loopback front door with $LITELLM_MASTER_KEY.

cmd_fleet() {
  local verb="${1:-}"
  [[ $# -gt 0 ]] && shift
  local fleet="" flag=""

  case "$verb" in
    --help|-h|"")
      cat <<'EOF'
ferry fleet — read or switch which routing fleet bare lane names resolve to.

Usage:
  ferry fleet ls                    List every fleet with its primaries; '*'
                                     marks the default, 'you' marks your own
                                     resolved fleet.
  ferry fleet show                  Show who you are, your resolved fleet, the
                                     host-wide default, and every client's pick.
  ferry fleet use <fleet>           Select a fleet for yourself (sticky).
  ferry fleet use <fleet> --default [Host only] Set the host-wide default fleet.
  ferry fleet use --clear           Clear your own selection (follow the default).
  ferry fleet --help                This message.
EOF
      [[ "$verb" == "" ]] && exit 1
      return 0
      ;;
    ls|show) ;;
    use)
      fleet="${1:-}"
      if [[ "$fleet" == "--clear" ]]; then
        fleet=""
        flag="clear"
        shift
      else
        [[ $# -gt 0 ]] && shift
        if [[ "${1:-}" == "--default" ]]; then
          flag="default"
          shift
        fi
      fi
      if [[ "$flag" != "clear" && ( -z "$fleet" || "$fleet" == --* ) ]]; then
        echo "Usage: ferry fleet use <fleet> [--default] | ferry fleet use --clear" >&2
        exit 1
      fi
      if [[ $# -gt 0 ]]; then
        echo "Unknown option for 'ferry fleet use': $1" >&2
        exit 1
      fi
      ;;
    *)
      echo "Unknown 'ferry fleet' subcommand: $verb" >&2
      echo "Usage: ferry fleet ls | show | use <fleet> [--default] | use --clear" >&2
      exit 1
      ;;
  esac

  # --default on a client is refused before any HTTP call: a client has no
  # authority to move the host-wide default (the front door would 403 it
  # anyway, loopback-only), so this is a courtesy short-circuit, not the
  # security boundary.
  if [[ "$flag" == "default" && "$CLIENT_MODE" == "1" ]]; then
    echo "the default is the host's to set" >&2
    exit 1
  fi

  local base key route_config=""
  if [[ "$CLIENT_MODE" == "1" ]]; then
    base="http://$CLIENT_HOST:$CLIENT_PORT"
    key="$CLIENT_MASTER_KEY"
  else
    base="http://127.0.0.1:$PORT"
    key="${LITELLM_MASTER_KEY:-}"
    route_config="$DEFAULT_ROUTE_CONFIG"
  fi

  python3 - "$base" "$key" "$CLIENT_NAME" "$verb" "$fleet" "$flag" "$route_config" <<'PYEOF'
import json
import os
import re
import sys
import urllib.error
import urllib.request

base, key, name, verb, fleet, flag, route_config = sys.argv[1:8]

FLEET_PATH = "/v1/ferry/fleet"


def _headers():
    h = {"X-Ferry-Client": name}
    if key:
        h["Authorization"] = "Bearer " + key
    return h


def _fail(msg):
    print("ferry fleet: " + msg, file=sys.stderr)
    sys.exit(1)


def _request(method, payload=None):
    data = None
    headers = _headers()
    if payload is not None:
        data = json.dumps(payload).encode()
        headers["Content-Type"] = "application/json"
    req = urllib.request.Request(base + FLEET_PATH, data=data, headers=headers, method=method)
    try:
        with urllib.request.urlopen(req, timeout=10) as resp:
            return json.loads(resp.read())
    except urllib.error.HTTPError as e:
        body = e.read().decode()
        try:
            err = json.loads(body)
            # the fleet route replies {"error": {"message": ...}}; the front
            # door's body reader replies {"errors": [...]} -- accept both.
            if isinstance(err.get("error"), dict):
                msg = err["error"].get("message") or body
            elif isinstance(err.get("errors"), list):
                msg = "; ".join(str(x) for x in err["errors"]) or body
            else:
                msg = body
        except Exception:
            msg = body
        _fail(msg)
    except urllib.error.URLError as e:
        _fail("cannot reach the front door at " + base + ": " + str(e.reason))


def _get():
    return _request("GET")


def _keys_column(fleets):
    if not route_config:
        return None
    try:
        with open(route_config) as f:
            text = f.read()
    except OSError:
        # Unreadable route config: nothing was checked, so say unknown, not ok.
        return {fname: "?" for fname in fleets}
    out = {}
    for fname in fleets:
        pat = re.compile(
            r"model_name:\s*" + re.escape(fname) + r"\.[^\n]*\n(.*?)(?=\n\s*-\s*model_name:|\Z)",
            re.S,
        )
        missing = []
        for m in pat.finditer(text):
            for env_name in re.findall(r"os\.environ/([A-Za-z0-9_]+)", m.group(1)):
                if not os.environ.get(env_name) and env_name not in missing:
                    missing.append(env_name)
        out[fname] = "ok" if not missing else "missing: " + ", ".join(missing)
    return out


def _fmt(v):
    return v if v else "-"


def cmd_ls():
    doc = _get()
    fleets = doc["fleets"]
    names = list(fleets.keys())
    keys = _keys_column(fleets)
    header = ["FLEET", "HEAVY", "MEDIUM", "FLASH", "SUPER-FLASH"]
    if keys is not None:
        header.append("KEYS")
    rows = []
    for fname in names:
        lanes = fleets[fname]
        marks = []
        if fname == doc.get("default"):
            marks.append("*")
        if fname == doc.get("fleet"):
            marks.append("you")
        label = fname + ("  " + " ".join(marks) if marks else "")
        row = [label, _fmt(lanes.get("heavy")), _fmt(lanes.get("medium")), _fmt(lanes.get("flash")), _fmt(lanes.get("super-flash"))]
        if keys is not None:
            row.append(keys.get(fname, "ok"))
        rows.append(row)
    widths = [len(h) for h in header]
    for row in rows:
        for i, c in enumerate(row):
            widths[i] = max(widths[i], len(c))

    def line(cols):
        return "  ".join(c.ljust(widths[i]) for i, c in enumerate(cols))

    print(line(header))
    for row in rows:
        print(line(row))


def cmd_show():
    doc = _get()
    print("you: " + str(doc.get("you", "")))
    print("fleet: " + str(doc.get("fleet", "")))
    print("default: " + str(doc.get("default", "")))
    print("clients:")
    for identity, fname in doc.get("clients", {}).items():
        print("  " + identity + " -> " + str(fname))


def cmd_use():
    doc = _get()
    fleets = doc["fleets"]
    if flag != "clear" and fleet not in fleets:
        names = ", ".join(fleets.keys())
        _fail("unknown fleet '" + fleet + "'; fleets: " + names)
    if flag == "clear":
        payload = {"fleet": None}
    elif flag == "default":
        payload = {"fleet": fleet, "default": True}
    else:
        payload = {"fleet": fleet}
    resp = _request("POST", payload)
    print("fleet: " + str(resp.get("fleet")))


if verb == "ls":
    cmd_ls()
elif verb == "show":
    cmd_show()
elif verb == "use":
    cmd_use()
else:
    _fail("unknown verb '" + verb + "'")
PYEOF
}

cmd_dash() {
  # Live dashboard for ferry — two modes:
  #   ferry dash [--open] [--port P] ...   lightweight stdlib page (`ferry-dash`, localhost:8091),
  #                                        delegates to the sibling python script. Default mode,
  #                                        unchanged behavior.
  #   ferry dash --grafana [--open]        full Grafana+VictoriaMetrics observability stack
  #                                        (localhost:3001) — delegates to observ/bringup.sh.
  #   ferry dash --grafana --down|--stop   tear the Grafana stack down via observ/teardown.sh.
  # Scan args for --grafana / --down / --stop; everything else (--open, --port, --purge, ...)
  # is forwarded through unchanged to whichever script handles the request.
  local arg has_grafana=0 is_down=0
  local -a fwd_args
  for arg in "$@"; do
    case "$arg" in
      --grafana) has_grafana=1 ;;
      --down|--stop) is_down=1 ;;
      *) fwd_args+=("$arg") ;;
    esac
  done

  if (( has_grafana )); then
    local observ_dir="$APP_DIR/observ"
    if [[ ! -f "$observ_dir/bringup.sh" ]]; then
      echo "Error: the Grafana stack lives in observ/ — re-run 'ferry install' or pull latest."
      exit 1
    fi
    if (( is_down )); then
      exec bash "$observ_dir/teardown.sh" "${fwd_args[@]}"
    else
      exec bash "$observ_dir/bringup.sh" "${fwd_args[@]}"
    fi
  fi

  # ---- default: lightweight stdlib page ----
  local dash="$APP_DIR/ferry-dash"
  [[ -f "$dash" ]] || dash="$(command -v ferry-dash || true)"
  if [[ -z "$dash" || ! -f "$dash" ]]; then
    echo "Error: 'ferry-dash' not found next to 'ferry' or on PATH. Re-run: ferry install"
    exit 1
  fi
  if ! command -v python3 >/dev/null 2>&1; then
    echo "Error: 'ferry dash' needs python3 (any version — stdlib only)."
    exit 1
  fi
  exec python3 "$dash" "$@"
}
# ----------------- UPDATE -----------------
# `ferry update` — catch this machine up, whichever end of the wire it is on.
#
# It owns NO update logic. Both catch-up paths already existed and are the
# tested ones; what was missing was a single command that picks the right one,
# so nobody has to remember which half of the stack they are standing on:
#
#   host    ->  ./host-reset.sh      rebuild the CLI from lib/, re-link it,
#                                    validate the route config, bounce the proxy
#   client  ->  curl .../client-reset.sh | zsh
#                                    re-pull the CLI from the host, re-apply the
#                                    opencode takeover
#
# They are deliberately NOT mirrors of each other (see host-reset.sh's header): a
# client is stale because the HOST has a newer CLI to download, while a host is
# stale because its own `ferry` drifted from its own lib/. Same word, two
# different repairs — which is exactly why guessing wrong is easy and why this
# command exists.
#
# ROLE DETECTION IS NOT NEW HERE. ferry has always decided host-vs-client on the
# presence of ~/.config/ferry/client.json (ferry-core.zsh sets CLIENT_MODE from
# it, and ferry-hostreset.test.py pins that rule). This reuses that flag rather
# than inventing a second, divergent notion of role.
cmd_update() {
  local dry_run=0 full=0 forced=""

  while [[ $# -gt 0 ]]; do
    case "$1" in
      --dry-run) dry_run=1; shift ;;
      --full)    full=1; shift ;;
      --host)    forced="host"; shift ;;
      --client)  forced="client"; shift ;;
      -h|--help)
        echo "Usage: ferry update [--full] [--host|--client] [--dry-run]"
        echo
        echo "  Catches this machine up. Detects host vs client from"
        echo "  ~/.config/ferry/client.json; --host/--client override it."
        echo "  --full    [Host] also reload the GPU lanes (minutes, not seconds)"
        echo "  --dry-run print the command that would run, and stop"
        return 0
        ;;
      *)
        echo "Error: unknown option for 'ferry update': $1" >&2
        echo "       ferry update [--full] [--host|--client] [--dry-run]" >&2
        return 1
        ;;
    esac
  done

  local role
  if [[ -n "$forced" ]]; then
    role="$forced"
  elif (( CLIENT_MODE )); then
    role="client"
  else
    role="host"
  fi

  local cmd
  if [[ "$role" == "client" ]]; then
    # --full reloads the GPU lanes, which live on the host. Refusing beats
    # silently dropping the flag: a user who passed it believes something extra
    # happened, and on a client nothing extra can.
    if (( full )); then
      echo "Error: --full applies to the host (it reloads the GPU lanes)." >&2
      echo "       A client has none; drop the flag." >&2
      return 1
    fi
    # CLIENT_HOST is empty both when there is no profile at all (--client forced
    # on a host) and when the profile omits `host`. Either way there is nothing
    # to curl, and an unguarded template would dial the literal "http:///".
    if [[ -z "$CLIENT_HOST" ]]; then
      echo "Error: no host to update from — ~/.config/ferry/client.json is" >&2
      echo "       missing or has no 'host' key." >&2
      echo "       Bootstrap this client first:" >&2
      echo "         curl -fsSL http://<host>:8095/client-bootstrap.sh | zsh" >&2
      return 1
    fi
    cmd="curl -fsSL http://$CLIENT_HOST:$CLIENT_SHARE_PORT/client-reset.sh | zsh"
  else
    local reset="$APP_DIR/host-reset.sh"
    if [[ ! -f "$reset" ]]; then
      echo "Error: host-reset.sh not found next to the CLI ($reset)." >&2
      echo "       On a host, ~/.local/bin/ferry should symlink into the checkout." >&2
      return 1
    fi
    cmd="zsh $reset"
    (( full )) && cmd="$cmd --full"
  fi

  if (( dry_run )); then
    echo ">>> ferry update — detected role: $role"
    echo "    would run: $cmd"
    return 0
  fi

  echo ">>> ferry update — $role mode"
  eval "$cmd"
}
# ----------------- MIGRATE (client -> host) -----------------
# `ferry migrate` — turn THIS machine from a ferry CLIENT into a ferry HOST.
#
# The transition itself lives in client-to-host.sh, which must run from inside a
# checkout (it needs host-bootstrap.sh, host-reset.sh, the route template and
# lib/). But a machine bootstrapped as a client has only the single-file `ferry`
# CLI in ~/.local/bin and no checkout at all — so this command's ONE job is to
# make sure a checkout exists (use the one this CLI already lives in, otherwise
# clone one) and then hand off to the engine. Every flag is forwarded verbatim.
#
# It owns no migration logic of its own: keeping client-to-host.sh the single
# source of truth is deliberate, the same way `ferry update` delegates to
# host-reset.sh / client-reset.sh rather than re-implementing either.
cmd_migrate() {
  local dry_run=0 assume_yes=0 full=0 do_pull=0
  local dir="" repo="https://github.com/sblattj/llm-ferry.git"

  while [[ $# -gt 0 ]]; do
    case "$1" in
      --dry-run) dry_run=1; shift ;;
      -y|--yes)  assume_yes=1; shift ;;
      --full)    full=1; shift ;;
      --pull)    do_pull=1; shift ;;
      --no-pull) do_pull=0; shift ;;
      --dir)
        [[ $# -ge 2 && -n "$2" ]] || { echo "Error: --dir needs a path" >&2; return 1; }
        dir="$2"; shift 2 ;;
      --dir=*)   dir="${1#--dir=}"; shift ;;
      --repo)
        [[ $# -ge 2 && -n "$2" ]] || { echo "Error: --repo needs a URL" >&2; return 1; }
        repo="$2"; shift 2 ;;
      --repo=*)  repo="${1#--repo=}"; shift ;;
      -h|--help)
        echo "Usage: ferry migrate [--dir PATH] [--repo URL] [--full] [--pull] [--dry-run] [--yes]"
        echo
        echo "  Turn THIS machine from a ferry client into a host. Ensures a repo"
        echo "  checkout exists (this CLI's own, or one cloned to --dir), then runs"
        echo "  client-to-host.sh from it."
        echo "  --dir PATH   where to clone the repo if a checkout is needed"
        echo "                 (default: ~/gdev/llm-ferry if ~/gdev exists, else ~/llm-ferry)"
        echo "  --repo URL   clone source (default: $repo)"
        echo "  --full       also reload the GPU lanes at the end (minutes)"
        echo "  --pull       let host-reset git-pull the checkout first"
        echo "  --dry-run    print every step and change nothing"
        echo "  --yes, -y    skip the confirmation prompt"
        return 0
        ;;
      *)
        echo "Error: unknown option for 'ferry migrate': $1" >&2
        echo "       ferry migrate [--dir PATH] [--repo URL] [--full] [--pull] [--dry-run] [--yes]" >&2
        return 1
        ;;
    esac
  done

  # Default checkout location: honour a ~/gdev dev tree if it exists, else a
  # neutral ~/llm-ferry.
  if [[ -z "$dir" ]]; then
    if [[ -d "$HOME/gdev" ]]; then dir="$HOME/gdev/llm-ferry"; else dir="$HOME/llm-ferry"; fi
  fi

  # A checkout is anything carrying the engine plus the host scripts it needs.
  _is_checkout() { [[ -f "$1/client-to-host.sh" && -f "$1/host-bootstrap.sh" && -d "$1/lib" ]]; }

  local checkout=""
  if _is_checkout "$APP_DIR"; then
    # ferry is already running from a checkout (a host box, or a re-run).
    checkout="$APP_DIR"
    echo ">>> ferry migrate — using this checkout: $checkout"
  elif _is_checkout "$dir"; then
    checkout="$dir"
    echo ">>> ferry migrate — using existing checkout: $checkout"
  elif [[ -e "$dir" ]]; then
    echo "Error: $dir exists but is not an llm-ferry checkout." >&2
    echo "       Point --dir at a fresh path, or update that checkout (git pull)." >&2
    return 1
  else
    # No checkout anywhere — clone one. This is the normal client case: the CLI
    # is a lone file in ~/.local/bin with no repo behind it.
    if ! command -v git >/dev/null 2>&1; then
      echo "Error: git is needed to clone the repo, and it is not on PATH." >&2
      echo "       Install git, or clone $repo yourself and re-run:" >&2
      echo "         ferry migrate --dir <that checkout>" >&2
      return 1
    fi
    if (( dry_run )); then
      echo ">>> ferry migrate — would clone $repo -> $dir"
      echo "    [dry-run] git clone $repo $dir"
      echo "    [dry-run] then: zsh $dir/client-to-host.sh --dry-run"
      echo "    (the engine is not on disk yet, so its own dry-run cannot be previewed here)"
      return 0
    fi
    echo ">>> Cloning $repo -> $dir ..."
    git clone "$repo" "$dir" || { echo "Error: clone failed." >&2; return 1; }
    if ! _is_checkout "$dir"; then
      echo "Error: cloned $dir but it has no client-to-host.sh — is the repo up to date?" >&2
      return 1
    fi
    checkout="$dir"
  fi

  # An older checkout may predate this feature.
  if [[ ! -f "$checkout/client-to-host.sh" ]]; then
    echo "Error: $checkout has no client-to-host.sh — update it (git pull) or clone fresh with --dir." >&2
    return 1
  fi

  # Hand off. Flags forwarded verbatim; the engine does the confirming, backups,
  # and the actual work.
  local -a fwd
  (( dry_run ))    && fwd+=(--dry-run)
  (( assume_yes )) && fwd+=(--yes)
  (( full ))       && fwd+=(--full)
  (( do_pull ))    && fwd+=(--pull)

  echo ">>> Running the migration engine: $checkout/client-to-host.sh ${fwd[*]}"
  zsh "$checkout/client-to-host.sh" "${fwd[@]}"
}

# ----------------- PARSER -----------------

if [[ $# -lt 1 ]]; then
  usage
fi

COMMAND="$1"
shift

case "$COMMAND" in
  install)       cmd_install ;;
  up)            cmd_up "$@" ;;
  down)          cmd_down "$@" ;;
  reload)        cmd_reload ;;
  status)        cmd_status ;;
  share)         cmd_share ;;
  msg)           cmd_msg "$@" ;;
  log)           cmd_log ;;
  inbox)         cmd_inbox "$@" ;;
  relay)         cmd_relay "$@" ;;
  expose)        cmd_expose "$@" ;;
  expose-vnc)    cmd_expose_vnc "$@" ;;
  offer)         cmd_offer "$@" ;;
  pull)          cmd_pull "$@" ;;
  get)           cmd_get "$@" ;;
  receive)       cmd_receive "$@" ;;
  send)          cmd_send "$@" ;;
  drop)          cmd_drop "$@" ;;
  pickup)        cmd_pickup "$@" ;;
  serve-hf)      cmd_serve_hf "$@" ;;
  serve-proxy)   cmd_serve_proxy "$@" ;;
  serve-vnc)     cmd_serve_vnc "$@" ;;
  env)           cmd_env "$@" ;;
  opencode)      cmd_opencode "$@" ;;
  claude)        cmd_claude "$@" ;;
  auth-claude)   cmd_auth_claude "$@" ;;
  fleet)         cmd_fleet "$@" ;;
  update)        cmd_update "$@" ;;
  migrate)       cmd_migrate "$@" ;;
  dash)          cmd_dash "$@" ;;
  --help|-h)     usage ;;
  *)             echo "Unknown command: $COMMAND"; usage ;;
esac
