{
 "run": "run3-m4pro-48gb",
 "date": "2026-09-26",
 "machine": "Mac mini, Apple M4 Pro, 48 GB unified memory, macOS 26.5.1, on AC power, idle before each engine (GPU 0% after a 60 s rest + 10 idle seconds)",
 "harness": "The committed run-2 bench_compare.py + run_all.sh, unchanged; run-2 protocol (every request sends chat_template_kwargs.enable_thinking=true, fresh HOME per engine, cooldown, order oMLX, Ollama, mlx-lm, Rapid-MLX PFlash off, Rapid-MLX defaults). Rapid-MLX defaults ran scenarios b and c (run 2 ran only b). Client: the rapid-mlx venv's Python 3.11.16 (uv-managed). 2026-09-26 22:14 to 22:57 PDT. Everything installed under ~/rmlx-compare-2026-09; RAPID_MLX_TELEMETRY=0, DO_NOT_TRACK=1.",
 "model": "Qwen3.5-9B at 4-bit (MLX engines: mlx-community/Qwen3.5-9B-4bit @ 8b2b98c, MTP sidecar mlx-community/Qwen3.5-9B-MTP-4bit @ 222dfd2; Ollama: qwen3.5:9b, GGUF Q4_K_M, id 6488c96fa5fa), same as run 2",
 "engines": {
  "rapid-mlx-pflash-off": {
   "label": "Rapid-MLX 0.15.2 (PFlash off)",
   "version": "rapid-mlx 0.15.2, mlx 0.32.2, mlx-lm 0.31.3",
   "command": "rapid-mlx serve qwen3.5-9b-4bit --port 18801 --pflash off",
   "notes": "Prefix-cache budget chosen at start: 7372.2 MB (first start), 7394.1 MB (after restart). No 'Cache entry too large' lines. Agent session turns 2-10 prefilled only 42-66 new tokens each (about 22.8k cached). Shutdown persistence skipped all 6 entries (predicted 6.45 s write vs the shutdown budget), so turn 1 after the restart was a full prefill."
  },
  "rapid-mlx": {
   "label": "Rapid-MLX 0.15.2 (defaults, PFlash on)",
   "version": "same",
   "command": "rapid-mlx serve qwen3.5-9b-4bit --port 18801",
   "notes": "Pure defaults (PFlash + MTP). Budgets 7406.2 / 7394.2 MB. Ran b and c only."
  },
  "omlx": {
   "label": "oMLX 0.7.0rc1 (SSD cache on)",
   "version": "omlx 0.7.0rc1, mlx 0.32.2, mlx-lm 0.32.0, mlx-vlm 0.7.1",
   "command": "omlx serve --model-dir <dir with one symlink to the snapshot> --port 18802 --host 127.0.0.1 --api-key localbench --paged-ssd-cache-dir <fresh dir>",
   "notes": "No prefill-memory-guard refusals on this machine."
  },
  "ollama": {
   "label": "Ollama 0.34.3 (qwen3.5:9b, Q4_K_M)",
   "version": "ollama 0.34.3 (standalone ollama-darwin.tgz binary run from the scratch dir, OLLAMA_MODELS in the scratch dir)",
   "command": "OLLAMA_NUM_PARALLEL=4 OLLAMA_CONTEXT_LENGTH=32768 OLLAMA_KEEP_ALIVE=60m ollama serve",
   "notes": "Again logs 'model architecture does not currently support parallel requests' for qwen35; the 4 streams ran serially. 'Restart' = ollama stop <model>."
  },
  "mlx-lm": {
   "label": "mlx-lm 0.31.3 (mlx_lm.server)",
   "version": "mlx-lm 0.31.3, mlx 0.32.2 (rapid-mlx venv)",
   "command": "python -m mlx_lm server --model <snapshot> --host 127.0.0.1 --port 18804",
   "notes": "No out-of-memory error at 48 GB; both agent sessions completed."
  }
 },
 "not_measured": {
  "lmstudio": "Not installed on this machine. Its headless installer (llmster/lms) installs into ~/.lmstudio with a background service rather than a self-contained scratch directory, which the run's rules exclude."
 },
 "machine_short": "an M4 Pro Mac mini with 48 GB",
 "model_short": "Qwen3.5-9B at 4-bit",
 "note": "Competitor numbers on the M4 Pro pages come from this run. Rapid-MLX was re-measured at main @1394f16 in run4-m4pro-48gb; this run's Rapid-MLX 0.15.2 rows are kept as the before-fix baseline."
}
