{
 "date": "2026-09-26",
 "machine": "MacBook Pro, Apple M3 Pro, 18 GB unified memory, macOS 15.6.1, on power",
 "model": "Qwen3.5-9B at 4-bit (MLX engines: mlx-community/Qwen3.5-9B-4bit @ 8b2b98c; Ollama: qwen3.5:9b, GGUF Q4_K_M, id 6488c96fa5fa)",
 "harness": "bench_compare.py + run_all.sh in the parent directory (run 2 versions); client Python 3.11.15; loopback only; one engine at a time. Run 2 (every number on the pages): 2026-09-26 19:16 to about 20:35 PDT, order oMLX, Ollama, mlx-lm, Rapid-MLX (PFlash off), Rapid-MLX (defaults); every request sends chat_template_kwargs.enable_thinking=true; each engine gets a fresh HOME under its output dir; the driver rests 60 s and waits for 10 idle GPU seconds before each engine. Run 1 (raw/run1, summary-run1.json, harness in raw/run1/harness): 18:03 to 19:01 PDT, fixed order with Rapid-MLX first, no cooldown, thinking left to each engine's default (Rapid-MLX turned it off), shared HOME; superseded because of those flaws.",
 "engines": {
  "rapid-mlx-pflash-off": {
   "label": "Rapid-MLX 0.15.2 (PFlash off)",
   "version": "rapid-mlx 0.15.2, mlx 0.32.2, mlx-lm 0.31.3 (pip, fresh venv)",
   "command": "rapid-mlx serve qwen3.5-9b-4bit --port 18801 --pflash off",
   "notes": "All other settings default, including MTP speculative decoding (on by default for this alias) and the prefix cache. Rapid-MLX turns thinking off by default for casual requests (its log: 'R12-T2F auto-disable'); run 2 asks for thinking explicitly on every request. Run 1 only: its first start loaded 7 prefix-cache entries left by an aborted earlier start (scenarios a/b prompts only)."
  },
  "rapid-mlx": {
   "label": "Rapid-MLX 0.15.2 (defaults, PFlash on)",
   "version": "rapid-mlx 0.15.2, mlx 0.32.2, mlx-lm 0.31.3 (pip, fresh venv)",
   "command": "rapid-mlx serve qwen3.5-9b-4bit --port 18801",
   "notes": "Pure defaults. For this alias the defaults include PFlash (long-prompt prefill compression: keeps 20% of a long prompt's tokens by default) and MTP speculative decoding. Run 2 ran only the cold-prefill test (scenario b) in this configuration; run 1 also ran the agent session, where PFlash did not engage (requests carry tools)."
  },
  "omlx": {
   "label": "oMLX 0.7.0rc1 (SSD cache on)",
   "version": "omlx 0.7.0rc1 (GitHub release wheel cp311, marked Latest on 2026-09-24), mlx 0.32.2, mlx-lm 0.32.0, mlx-vlm 0.7.1",
   "command": "omlx serve --model-dir <dir with one symlink to the snapshot> --port 18802 --host 127.0.0.1 --api-key localbench --paged-ssd-cache-dir <fresh empty dir>",
   "notes": "The SSD cache directory is set explicitly (its documented flag for the restart case); everything else default. The API key is only because oMLX refuses keyless inference by default. oMLX streams several tokens per SSE chunk."
  },
  "ollama": {
   "label": "Ollama 0.34.3 (qwen3.5:9b, Q4_K_M)",
   "version": "ollama 0.34.3 (Ollama.app)",
   "command": "OLLAMA_NUM_PARALLEL=4 OLLAMA_CONTEXT_LENGTH=32768 OLLAMA_KEEP_ALIVE=60m ollama serve",
   "notes": "Context raised from the default so the 16k and 23k-token prompts are not truncated. OLLAMA_NUM_PARALLEL=4 was set, but Ollama 0.34.3 logs 'model architecture does not currently support parallel requests' for qwen35 and serves one request at a time, so the 4-stream test ran serially. 'Restart' for the agent session = ollama stop <model> (unload, which drops its KV cache) and reload. Ollama's MLX engine (qwen3.5:9b-mlx) needs 32 GB+ and was not run on this 18 GB machine. Weights are GGUF Q4_K_M, not the MLX 4-bit file the other engines load."
  },
  "mlx-lm": {
   "label": "mlx-lm 0.31.3 (mlx_lm.server)",
   "version": "mlx-lm 0.31.3, mlx 0.32.2 (same venv as rapid-mlx)",
   "command": "python -m mlx_lm server --model <snapshot> --host 127.0.0.1 --port 18804",
   "notes": "Apple's reference server, defaults. Run 2 records its agent session up to the failed turn (per-request timeout 300 s).",
   "agent_session_note": "the server's generation thread died with a Metal out-of-memory error (see its server.log): on turn 4 of the first session and on turn 3 after the restart"
  }
 },
 "not_measured": {
  "lmstudio": "LM Studio is not installed on this laptop. It is installed on our Mac Studio M3 Ultra, but that machine's GPU was at 100% utilization from an unrelated fine-tuning job for the whole benchmark window (about nine hours of it remained), so we measured nothing there rather than publish contended numbers.",
  "m3_ultra": "Same reason: no Mac Studio numbers in this run.",
  "ollama-mlx": "Ollama's MLX engine requires 32 GB+ of memory."
 },
 "machine_short": "an M3 Pro MacBook Pro with 18 GB",
 "model_short": "Qwen3.5-9B at 4-bit"
}
