#!/bin/bash
# Driver for the 2026-09 engine comparison (loopback only). Run one engine at a time on an idle Mac.
# Usage: run_all.sh <engine> <outdir>
#   engine: rapid-mlx | rapid-mlx-pflash-off | omlx | mlx-lm | ollama | ollama-mlx | lmstudio
# Each engine: start server -> scenarios a b d e c(first) -> restart -> c(after_restart) -> stop.
# Only the PID this script started is ever signalled.
# v2 (run 2): every request asks for thinking explicitly (--thinking on), each engine gets a
# fresh HOME under its output dir (no cache or settings carried between runs), and the
# driver waits COOLDOWN seconds plus until the GPU reads idle before starting a server.
set -u
ENGINE=$1
OUT=$2
HERE=$(cd "$(dirname "$0")" && pwd)
W=${W:-/private/tmp/geo-compare-bench}          # scratch root (venvs, logs, oMLX model dir)
PY=${PY:-python3}
: "${HF_HOME:?set HF_HOME to your normal Hugging Face cache}"
SNAP=${SNAP:-$HF_HOME/hub/models--mlx-community--Qwen3.5-9B-4bit/snapshots/8b2b98c00a6b4d291155e4890773ca8f769aee53}
OL=${OLLAMA_BIN:-ollama}
mkdir -p "$OUT"
gpu_util() { ioreg -r -d 1 -w 0 -c IOAccelerator | grep -o '"Device Utilization %"=[0-9]*' | head -1; }
gpu_pct() { gpu_util | grep -o '[0-9]*$'; }
cooldown() {  # fixed rest, then up to 3 min for 10 consecutive idle (<10%) seconds
  sleep "${COOLDOWN:-60}"; local ok=0
  for _ in $(seq 1 180); do
    if [ "$(gpu_pct)" -lt 10 ] 2>/dev/null; then ok=$((ok+1)); else ok=0; fi
    [ "$ok" -ge 10 ] && break; sleep 1
  done
  echo "cooldown done $(date '+%H:%M:%S') idle_seconds=$ok $(gpu_util)" >>"$OUT/env.txt"
}
export HOME="$OUT/home"; mkdir -p "$HOME"   # fresh per engine run (HF_HOME stays the shared cache)
{ date; uname -a; sysctl -n machdep.cpu.brand_string hw.memsize; sw_vers -productVersion; echo "gpu before: $(gpu_util)"; } >>"$OUT/env.txt"
export RAPID_MLX_TELEMETRY=0 DO_NOT_TRACK=1

wait_ready() {  # $1 base url, $2 pid
  for _ in $(seq 1 600); do
    if curl -s -m 2 ${AUTH:+-H "Authorization: Bearer $AUTH"} "$1/models" >/dev/null 2>&1; then return 0; fi
    if [ -n "${2:-}" ] && ! kill -0 "$2" 2>/dev/null; then echo "server $2 died"; return 1; fi
    sleep 1
  done
  return 1
}

stop_pid() {  # graceful stop of a PID we started
  kill "$1" 2>/dev/null
  for _ in $(seq 1 60); do kill -0 "$1" 2>/dev/null || return 0; sleep 1; done
  kill -9 "$1" 2>/dev/null
}

case "$ENGINE" in
  rapid-mlx|rapid-mlx-pflash-off)
    # Defaults for this alias at 0.15.2 include MTP speculative decoding and PFlash
    # (lossy long-prompt compression). rapid-mlx-pflash-off keeps everything else
    # default but prefills the full prompt, like the other engines.
    BASE=http://127.0.0.1:18801/v1; MODEL=qwen3.5-9b-4bit; AUTH=
    EXTRA=; [ "$ENGINE" = rapid-mlx-pflash-off ] && EXTRA="--pflash off"
    start() { "$W/rmlx-venv/bin/rapid-mlx" serve qwen3.5-9b-4bit --port 18801 $EXTRA >>"$OUT/server.log" 2>&1 & PID=$!; echo "$PID rapid-mlx serve" >>"$W/PIDS.txt"; }
    stop() { stop_pid "$PID"; }
    ;;
  omlx)
    # --model-dir holds one symlink to the HF snapshot above; SSD cache in a fresh scratch dir.
    BASE=http://127.0.0.1:18802/v1; MODEL=Qwen3.5-9B-4bit; AUTH=localbench
    mkdir -p "$W/omlx-models" "$OUT/omlx-ssd-cache"
    ln -sfn "$SNAP" "$W/omlx-models/Qwen3.5-9B-4bit"
    start() { "$W/omlx-venv/bin/omlx" serve --model-dir "$W/omlx-models" --port 18802 --host 127.0.0.1 --api-key localbench --paged-ssd-cache-dir "$OUT/omlx-ssd-cache" >>"$OUT/server.log" 2>&1 & PID=$!; echo "$PID omlx serve" >>"$W/PIDS.txt"; }
    stop() { stop_pid "$PID"; }
    ;;
  mlx-lm)
    # Apple's reference server from the same venv as rapid-mlx (mlx-lm is its dependency).
    BASE=http://127.0.0.1:18804/v1; MODEL=$SNAP; AUTH=
    start() { "$W/rmlx-venv/bin/python" -m mlx_lm server --model "$SNAP" --host 127.0.0.1 --port 18804 >>"$OUT/server.log" 2>&1 & PID=$!; echo "$PID mlx_lm.server" >>"$W/PIDS.txt"; }
    stop() { stop_pid "$PID"; }
    ;;
  ollama|ollama-mlx)
    # ollama serve is started separately (see README: OLLAMA_NUM_PARALLEL, OLLAMA_CONTEXT_LENGTH); "restart" = unload the model.
    BASE=http://127.0.0.1:11434/v1; AUTH=
    if [ "$ENGINE" = ollama ]; then MODEL=qwen3.5:9b; else MODEL=qwen3.5:9b-mlx; fi
    start() { PID=; }
    stop() { OLLAMA_HOST=127.0.0.1:11434 "$OL" stop "$MODEL"; sleep 3; }
    ;;
  lmstudio)
    BASE=http://127.0.0.1:1234/v1; MODEL=qwen3.5-9b-4bit-bench; AUTH=
    LMS=${LMS_BIN:-lms}
    start() { "$LMS" load mlx-community/Qwen3.5-9B-4bit --identifier "$MODEL" --context-length 40960 --parallel 4 -y >>"$OUT/server.log" 2>&1; PID=; }
    stop() { "$LMS" unload "$MODEL" >>"$OUT/server.log" 2>&1; }
    ;;
esac

run() {  # scenario, phase
  "$PY" "$HERE/bench_compare.py" --engine "$ENGINE" --base-url "$BASE" --model "$MODEL" ${AUTH:+--api-key "$AUTH"} \
    --thinking on --timeout "${REQ_TIMEOUT:-300}" --scenario "$1" --phase "$2" --out "$OUT/$ENGINE-$1-$2.json" 2>>"$OUT/client.err" | tee -a "$OUT/summary.jsonl"
}

cooldown
start; wait_ready "$BASE" "${PID:-}" || exit 1
for s in ${SCENARIOS:-a b d e}; do run "$s" first; done
if [ "${SKIP_C:-0}" != 1 ]; then
  run c first
  stop; start; wait_ready "$BASE" "${PID:-}" || exit 1
  run c after_restart
fi
stop
echo "gpu after: $(gpu_util)" >>"$OUT/env.txt"
