{
 "rapid_mlx_version": "Rapid-MLX 0.15.3, measured from a source checkout at commit 1394f16960d51e5d619c1aa45f1aa173454eb8c8, which 0.15.3 includes (the measured checkout reported package version string 0.15.2; mlx 0.32.2, mlx-lm 0.31.3). Includes #3794 (agent-session prefix cache: small-RAM budget floor + shutdown save). The benchmark was not re-run on the release tag.",
 "rapid_mlx_sha": "1394f16960d51e5d619c1aa45f1aa173454eb8c8",
 "competitors": "Not re-run: oMLX, Ollama and mlx-lm numbers on the pages come from the earlier run on the same machine with the same versions (see 'competitor_run').",
 "harness": "The committed run-2 bench_compare.py + run_all.sh, byte-identical; run-2 protocol (explicit thinking, fresh HOME per engine, 60 s rest + 10 idle GPU seconds). Order: Rapid-MLX PFlash off (scenarios a b d e c + restart), then Rapid-MLX defaults (b, c + restart). RAPID_MLX_TELEMETRY=0.",
 "run": "run4-m3pro-18gb",
 "date": "2026-09-27",
 "competitor_run": "run2",
 "machine": "MacBook Pro, Apple M3 Pro, 18 GB unified memory, macOS 15.6.1, on power",
 "machine_short": "an M3 Pro MacBook Pro with 18 GB",
 "model_short": "Qwen3.5-9B at 4-bit",
 "model": "Qwen3.5-9B at 4-bit (mlx-community/Qwen3.5-9B-4bit @ 8b2b98c, MTP sidecar @ 222dfd2)",
 "when": "2026-09-27 02:07 to about 02:30 PDT",
 "cache_findings": {
  "budget_mb": [
   2041.3,
   2041.3,
   2041.3,
   2041.3
  ],
  "agent_session_floor_mb": 2041.3,
  "cache_entry_too_large_lines": 0,
  "turns_2_10_new_prefill_tokens": "42-66 per turn (rest cached)",
  "shutdown_save": "PFlash-off config: first shutdown SAVED 1/1 entry (968 MB) in 1.2 s; second shutdown saved nothing ('shutdown budget would not fit entry 0/1', predicted 3.07 s). Defaults config: SAVED 1/1 at both shutdowns. LOADED 1 entry at restart in both configs.",
  "restart_turn1": "prefilled 3875 of 22819 tokens (18944 restored from the saved cache)"
 },
 "engines": {
  "rapid-mlx-pflash-off": {
   "label": "Rapid-MLX main @1394f16 (PFlash off)",
   "command": "rapid-mlx serve qwen3.5-9b-4bit --port 18801 --pflash off"
  },
  "rapid-mlx": {
   "label": "Rapid-MLX main @1394f16 (defaults, PFlash on)",
   "command": "rapid-mlx serve qwen3.5-9b-4bit --port 18801",
   "notes": "Ran scenarios b and c only."
  }
 }
}
