{
 "rapid_mlx_version": "Rapid-MLX 0.15.3, measured from a source checkout at commit 1394f16960d51e5d619c1aa45f1aa173454eb8c8, which 0.15.3 includes (the measured checkout reported package version string 0.15.2; mlx 0.32.2, mlx-lm 0.31.3). Includes #3794 (agent-session prefix cache: small-RAM budget floor + shutdown save). The benchmark was not re-run on the release tag.",
 "rapid_mlx_sha": "1394f16960d51e5d619c1aa45f1aa173454eb8c8",
 "competitors": "Not re-run: oMLX, Ollama and mlx-lm numbers on the pages come from the earlier run on the same machine with the same versions (see 'competitor_run').",
 "harness": "The committed run-2 bench_compare.py + run_all.sh, byte-identical; run-2 protocol (explicit thinking, fresh HOME per engine, 60 s rest + 10 idle GPU seconds). Order: Rapid-MLX PFlash off (scenarios a b d e c + restart), then Rapid-MLX defaults (b, c + restart). RAPID_MLX_TELEMETRY=0.",
 "run": "run4-m4pro-48gb",
 "date": "2026-09-27",
 "competitor_run": "run3-m4pro-48gb",
 "machine": "Mac mini, Apple M4 Pro, 48 GB unified memory, macOS 26.5.1, on AC power, idle before each engine",
 "machine_short": "an M4 Pro Mac mini with 48 GB",
 "model_short": "Qwen3.5-9B at 4-bit",
 "model": "Qwen3.5-9B at 4-bit (mlx-community/Qwen3.5-9B-4bit @ 8b2b98c, MTP sidecar @ 222dfd2)",
 "when": "2026-09-27 02:07 to about 02:25 PDT",
 "cache_findings": {
  "budget_mb": [
   6700.5,
   6725.3,
   6737.1,
   6737.3
  ],
  "agent_session_floor_mb": 4096.0,
  "cache_entry_too_large_lines": 0,
  "turns_2_10_new_prefill_tokens": "42-66 per turn (rest cached)",
  "shutdown_save": "SAVED 6/6 entries (5793 MB) at every shutdown, in 1.5-1.7 s; LOADED 6 entries at restart",
  "restart_turn1": "prefilled 3875 of 22819 tokens (18944 restored from the saved cache)"
 },
 "engines": {
  "rapid-mlx-pflash-off": {
   "label": "Rapid-MLX main @1394f16 (PFlash off)",
   "command": "rapid-mlx serve qwen3.5-9b-4bit --port 18801 --pflash off"
  },
  "rapid-mlx": {
   "label": "Rapid-MLX main @1394f16 (defaults, PFlash on)",
   "command": "rapid-mlx serve qwen3.5-9b-4bit --port 18801",
   "notes": "Ran scenarios b and c only."
  }
 }
}
