{
 "date": "2026-10-04",
 "measured": "Oct 4, 2026",
 "machine": "Mac mini, Apple M4 Pro, 48 GB unified memory, macOS 26.5.1, on AC power, idle before every server start (30 s rest, then 10 seconds of idle GPU; no other model server running)",
 "machine_short": "Mac mini M4 Pro, 48 GB",
 "machine_chip": "M4 Pro",
 "machine_inline": "Mac mini M4 Pro (48 GB)",
 "rapid": {
  "label": "Rapid-MLX 0.15.6",
  "version": "0.15.6",
  "install": "pip install rapid-mlx==0.15.6 (bundles mlx 0.32.3, mlx-lm 0.31.3)",
  "command": "rapid-mlx serve <alias> --host 127.0.0.1 --port <port>",
  "notes": "Defaults. For qwen3.5-9b-4bit and qwen3.6-35b-4bit the defaults turn on MTP speculative decoding with prompt lookup (copy-drafting from the prompt); for qwen3.5-4b-4bit MTP is off by default. Config rapid_mtp adds --speculative-config '{\"method\":\"mtp\"}' (4B only). Config rapid_nospec adds --no-spec-decode (attribution runs only, 1 rep)."
 },
 "baseline": {
  "label": "mlx-lm 0.32.0 (mlx_lm.server)",
  "version": "0.32.0",
  "install": "pip install mlx-lm==0.32.0 (with mlx 0.32.3)",
  "command": "python -m mlx_lm server --model <Hugging Face snapshot> --host 127.0.0.1 --port <port>",
  "notes": "Defaults. mlx_lm.generate (no server) was run on two tasks per model as a sanity check of the server baseline: raw/post/generate2.txt."
 },
 "request": "Streaming /v1/chat/completions, temperature 0, top_p 1, fixed max_tokens per task, chat_template_kwargs.enable_thinking=false, stream_options.include_usage",
 "protocol": "Each rep starts a fresh server per config with a fresh temporary HOME (no prefix, response or disk cache carries over), sends 2 discarded warm-up requests, then each task once. Config order rotates every rep. Decode tok/s = (completion_tokens - 1) / (last content chunk - first content chunk), completion_tokens from each engine's usage. Cells are medians over reps.",
 "models": {
  "9b": {"label": "Qwen3.5-9B 4-bit", "alias": "qwen3.5-9b-4bit", "weights": "mlx-community/Qwen3.5-9B-4bit @ 8b2b98c", "rapid_default": "MTP + prompt lookup (on by default)"},
  "35b": {"label": "Qwen3.6-35B-A3B 4-bit (MoE)", "alias": "qwen3.6-35b-4bit", "weights": "mlx-community/Qwen3.6-35B-A3B-4bit @ 38740b8", "rapid_default": "MTP + prompt lookup (on by default)"},
  "4b": {"label": "Qwen3.5-4B 4-bit", "alias": "qwen3.5-4b-4bit", "weights": "mlx-community/Qwen3.5-4B-MLX-4bit @ 32f3e8e", "rapid_default": "no speculative decoding (MTP off by default for this model)"}
 },
 "extras": {
  "harness": "The published /compare harness (blog/assets/compare-2026-09/bench_compare.py, unchanged), --thinking on, on qwen3.5-9b-4bit: scenario d (4 concurrent streams, 256 tokens each, 2 reps) and scenario c (22.7k-token agent session, 10 turns). Driver: extras.sh."
 },
 "sanitized": "Absolute paths under the benchmark host's home directory are replaced with ~ in the published files. Server logs are not published (they contain host paths); everything the pages quote is in raw/."
}
