{
 "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead",
 "kit_sha256_note": "one kit measured every row",
 "device": "MacStudio_M1_Ultra",
 "device_id": "m1-ultra",
 "board": "v2026.10.04_r5c_kit_pd1",
 "replaces_board": "v2026.10.03_r3_kit_pd1",
 "merge": false,
 "prompts": {
  "prompt_set": "pd1",
  "source": "Charles Darwin, On the Origin of Species (1859), Project Gutenberg eBook #1228, public domain",
  "sha256sums_sha256": "0a9649f5dcddf86caadb28c96d99cd0f47d1ffa3cd2b3629d657e50fcd9d779d",
  "kit_tgz_sha256": "a48fb30b33f285ceb51d27d7aa5d2586c7710a3306ae87d79fa98e1ebd0f7c16"
 },
 "runs_per_cell": 11,
 "complete": false,
 "build": "veizik-metal 2026.10.02-r5 candidate DMG b8fb122bc11b8c2ee4615737a38db29e083c9c45662e0f1b02b7f572bf688c86, CLI 1.2.0 (CLI binary sha256 a9eb4d9b8018432f27ce95abd0e0dc7bc63fa23a4695592d9a71858819f4713a)",
 "header": "# M1Ultra_r5 BENCH Apple_M1_Ultra mem=137438953472 macos=26.5.1 2026-10-04_13:00:21 load=2.51 veizik=veizik_1.2.0 mlx_lm/mlx=0.31.3/0.32.2 engs=vz,mq,mb runs=11",
 "method": "public repro kit bench.sh: customer CLI `veizik run <m> --tokens 128 --ignore-eos`, shipped defaults, no VZ_* env; MLX via mlx_lm EOS-masked (mlx_fixedlen.py); 1 discarded warm run per engine per cell; 11 interleaved rounds, order alternating; quiet gate before every run (GPU <= idle floor + 5, background-process CPU <= 10, load < 5, AC, no thermal warning); rounds with load>=5 before/after are discarded and re-run; median of 11.",
 "insufficient_run_rows": [],
 "discarded_load_rounds": 3,
 "discarded_rows_note": "discarded_load_rounds / discarded_foreign_rows count DISCARD rows (one per engine run) of the cells in this file only",
 "discarded_foreign_rows": 0,
 "excluded": [
  "qwen2.5-1.5b",
  "qwen3-1.7b",
  "qwen2.5-7b"
 ],
 "unmeasurable_decode_rows": [],
 "rows": [
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "prefill",
   "N": 512,
   "engine": "mlx_bf16",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.15,
    2.3,
    1.91,
    1.92,
    2.37,
    2.26,
    2.11,
    3.57,
    3.37,
    2.62,
    2.58
   ],
   "load_after": [
    2.3,
    2.2,
    1.92,
    1.93,
    2.26,
    2.16,
    2.02,
    3.37,
    3.42,
    2.81,
    2.58
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "as-shipped",
   "precision": "bf16 (HF base snapshot, unquantised)",
   "input_tokens": 512,
   "peak_memory": [
    "1.630GB",
    "1.630GB",
    "1.630GB",
    "1.630GB",
    "1.630GB",
    "1.630GB",
    "1.630GB",
    "1.630GB",
    "1.630GB",
    "1.630GB",
    "1.630GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "cc82bffad80e",
    "cc82bffad80e",
    "cc82bffad80e",
    "cc82bffad80e",
    "cc82bffad80e",
    "cc82bffad80e",
    "cc82bffad80e",
    "cc82bffad80e",
    "cc82bffad80e",
    "cc82bffad80e",
    "cc82bffad80e"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/prefill/512/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    4853.32,
    5232.066,
    5244.122,
    5144.975,
    4839.533,
    5073.122,
    5023.258,
    5032.924,
    5100.829,
    4886.94,
    5100.342
   ],
   "min": 4839.533,
   "max": 5244.122,
   "median": 5073.122,
   "spread_pct": 8.36,
   "status": "ok",
   "runs_as_reported": [
    4853.32,
    5232.066,
    5244.122,
    5144.975,
    4839.533,
    5073.122,
    5023.258,
    5032.924,
    5100.829,
    4886.94,
    5100.342
   ],
   "min_as_reported": 4839.533,
   "max_as_reported": 5244.122,
   "median_as_reported": 5073.122,
   "spread_pct_as_reported": 8.36,
   "rate_definition": "judged = mlx_lm prompt_tps = N / prompt_time (prompt pass through token 0), as reported",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "decode",
   "N": 512,
   "engine": "mlx_bf16",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.15,
    2.3,
    1.91,
    1.92,
    2.37,
    2.26,
    2.11,
    3.57,
    3.37,
    2.62,
    2.58
   ],
   "load_after": [
    2.3,
    2.2,
    1.92,
    1.93,
    2.26,
    2.16,
    2.02,
    3.37,
    3.42,
    2.81,
    2.58
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "as-shipped",
   "precision": "bf16 (HF base snapshot, unquantised)",
   "input_tokens": 512,
   "peak_memory": [
    "1.630GB",
    "1.630GB",
    "1.630GB",
    "1.630GB",
    "1.630GB",
    "1.630GB",
    "1.630GB",
    "1.630GB",
    "1.630GB",
    "1.630GB",
    "1.630GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "cc82bffad80e",
    "cc82bffad80e",
    "cc82bffad80e",
    "cc82bffad80e",
    "cc82bffad80e",
    "cc82bffad80e",
    "cc82bffad80e",
    "cc82bffad80e",
    "cc82bffad80e",
    "cc82bffad80e",
    "cc82bffad80e"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/decode/512/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    247.4545390625,
    248.66302343750002,
    248.5657890625,
    247.9119375,
    247.9069765625,
    248.22745312499998,
    249.6333828125,
    248.5380078125,
    248.1431171875,
    247.78890625,
    248.056796875
   ],
   "min": 247.4545390625,
   "max": 249.6333828125,
   "median": 248.1431171875,
   "spread_pct": 0.88,
   "status": "ok",
   "runs_as_reported": [
    249.403,
    250.621,
    250.523,
    249.864,
    249.859,
    250.182,
    251.599,
    250.495,
    250.097,
    249.74,
    250.01
   ],
   "min_as_reported": 249.403,
   "max_as_reported": 251.599,
   "median_as_reported": 250.097,
   "spread_pct_as_reported": 0.88,
   "rate_definition": "judged = generation_tps x (gen_tok-1)/gen_tok = 127 tokens / mlx_lm generation window; as_reported = mlx_lm generation_tps = 128/window",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "prefill",
   "N": 512,
   "engine": "mlx",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.25,
    1.91,
    1.99,
    1.93,
    2.23,
    2.16,
    2.2,
    3.53,
    3.42,
    2.76,
    2.62
   ],
   "load_after": [
    2.15,
    1.91,
    1.99,
    2.25,
    2.37,
    2.31,
    2.11,
    3.57,
    3.3,
    2.62,
    2.62
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "matched",
   "precision": "q4 g64 affine (mlx-community-style 4-bit g64, HF-converted)",
   "input_tokens": 512,
   "peak_memory": [
    "0.719GB",
    "0.719GB",
    "0.719GB",
    "0.719GB",
    "0.719GB",
    "0.719GB",
    "0.719GB",
    "0.719GB",
    "0.719GB",
    "0.719GB",
    "0.719GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "6c0feece1819",
    "6c0feece1819",
    "6c0feece1819",
    "6c0feece1819",
    "6c0feece1819",
    "6c0feece1819",
    "6c0feece1819",
    "6c0feece1819",
    "6c0feece1819",
    "6c0feece1819",
    "6c0feece1819"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/prefill/512/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    4662.422,
    4830.88,
    4854.352,
    4326.305,
    5155.632,
    4729.328,
    4680.92,
    5002.227,
    5107.253,
    4885.222,
    4788.982
   ],
   "min": 4326.305,
   "max": 5155.632,
   "median": 4830.88,
   "spread_pct": 19.17,
   "status": "ok",
   "runs_as_reported": [
    4662.422,
    4830.88,
    4854.352,
    4326.305,
    5155.632,
    4729.328,
    4680.92,
    5002.227,
    5107.253,
    4885.222,
    4788.982
   ],
   "min_as_reported": 4326.305,
   "max_as_reported": 5155.632,
   "median_as_reported": 4830.88,
   "spread_pct_as_reported": 19.17,
   "rate_definition": "judged = mlx_lm prompt_tps = N / prompt_time (prompt pass through token 0), as reported",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "variance_cause_scope_check": {
    "foreign_evidence": "none found (no in-run foreign watch in this harness; machine file activity checked for the cell window)",
    "rule": "device scope requires a clean quiet gate on every run of every engine in the cell and veizik and both MLX baselines all wide (spread > 10%)",
    "veizik_spread_pct": 26.33,
    "baseline_spread_pct": {
     "mlx": 19.17,
     "mlx_bf16": 8.36
    },
    "quiet_check_tally": {
     "veizik": {
      "runs": 11,
      "gpu_idle_pct_max": 0.0,
      "gpu_threshold_pct": 5.0,
      "runs_over_gpu_threshold": 0,
      "foreign_process_samples": 0,
      "foreign_watch": "recorded (fg)"
     },
     "mlx": {
      "runs": 11,
      "gpu_idle_pct_max": 0.0,
      "gpu_threshold_pct": 5.0,
      "runs_over_gpu_threshold": 0,
      "foreign_process_samples": 0,
      "foreign_watch": "recorded (fg)"
     },
     "mlx_bf16": {
      "runs": 11,
      "gpu_idle_pct_max": 0.0,
      "gpu_threshold_pct": 5.0,
      "runs_over_gpu_threshold": 0,
      "foreign_process_samples": 0,
      "foreign_watch": "recorded (fg)"
     }
    },
    "result": "device scope withheld: 7 of 11 kept runs per engine predate the 13:27 pause of mediaanalysisd/photoanalysisd and 4 follow it, so a device attribution cannot be shown"
   },
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "decode",
   "N": 512,
   "engine": "mlx",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.25,
    1.91,
    1.99,
    1.93,
    2.23,
    2.16,
    2.2,
    3.53,
    3.42,
    2.76,
    2.62
   ],
   "load_after": [
    2.15,
    1.91,
    1.99,
    2.25,
    2.37,
    2.31,
    2.11,
    3.57,
    3.3,
    2.62,
    2.62
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "matched",
   "precision": "q4 g64 affine (mlx-community-style 4-bit g64, HF-converted)",
   "input_tokens": 512,
   "peak_memory": [
    "0.719GB",
    "0.719GB",
    "0.719GB",
    "0.719GB",
    "0.719GB",
    "0.719GB",
    "0.719GB",
    "0.719GB",
    "0.719GB",
    "0.719GB",
    "0.719GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "6c0feece1819",
    "6c0feece1819",
    "6c0feece1819",
    "6c0feece1819",
    "6c0feece1819",
    "6c0feece1819",
    "6c0feece1819",
    "6c0feece1819",
    "6c0feece1819",
    "6c0feece1819",
    "6c0feece1819"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/decode/512/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    270.2728671875,
    270.6538671875,
    269.42256249999997,
    271.563703125,
    270.4465,
    271.4704375,
    270.89496875000003,
    271.4178515625,
    270.7253046875,
    271.2164375,
    271.2213984375
   ],
   "min": 269.42256249999997,
   "max": 271.563703125,
   "median": 270.89496875000003,
   "spread_pct": 0.79,
   "status": "ok",
   "runs_as_reported": [
    272.401,
    272.785,
    271.544,
    273.702,
    272.576,
    273.608,
    273.028,
    273.555,
    272.857,
    273.352,
    273.357
   ],
   "min_as_reported": 271.544,
   "max_as_reported": 273.702,
   "median_as_reported": 273.028,
   "spread_pct_as_reported": 0.79,
   "rate_definition": "judged = generation_tps x (gen_tok-1)/gen_tok = 127 tokens / mlx_lm generation window; as_reported = mlx_lm generation_tps = 128/window",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "prefill",
   "N": 512,
   "engine": "veizik",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.25,
    2.08,
    2.08,
    2.25,
    2.25,
    2.31,
    2.2,
    3.76,
    3.3,
    2.76,
    2.65
   ],
   "load_after": [
    2.25,
    2.08,
    2.08,
    2.25,
    2.23,
    2.31,
    2.2,
    3.53,
    3.3,
    2.76,
    2.65
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "row_id": "m1-ultra/qwen2.5-0.5b/veizik/prefill/512/kit_r5_pd1",
   "unit": "tok/s (core self-clock, CLI \"prefill : <ms> ms for <N> positions\" line)",
   "measured_binary_sha256": "8e9e22a82073be4814699b6271be95f1045e1eeb8a8ba59bd912692289d7d659",
   "measured_binary_digest_kind": "sha256 of __TEXT,__text section",
   "core_file_sha256": "c7678f0931a848f593dedd5b07c8d505f5d44f22b6fc1a7e713fa15232f23896",
   "measured_cli_sha256": "a9eb4d9b8018432f27ce95abd0e0dc7bc63fa23a4695592d9a71858819f4713a",
   "cli_digest_kind": "sha256 of the installed CLI binary (file)",
   "snapshot": "060db6499f32faf8b98477b0a26969ef7d8b9987",
   "precision_mode": "as-shipped",
   "precision": "veizik shipped low-bit core (installed release)",
   "command": "veizik run qwen2.5-0.5b --tokens 128 --ignore-eos \"<p512 prompt file>\"  (customer CLI, no VZ_* env)",
   "input_tokens": 512,
   "checkpoint": "Qwen base HF snapshot 060db649",
   "runs": [
    9744.358937201903,
    11009.744090838516,
    11195.655898182617,
    9497.922110436424,
    9108.85008612391,
    8862.5237530193,
    9955.980187256455,
    10686.488209018928,
    8961.870702301716,
    9081.559172313311,
    9714.490432965924
   ],
   "min": 8862.5237530193,
   "max": 11195.655898182617,
   "median": 9714.490432965924,
   "spread_pct": 26.33,
   "status": "ok",
   "runs_as_reported": [
    10264.64,
    11675.37,
    11886.24,
    9990.63,
    9563.48,
    9292.03,
    10480.21,
    11308.92,
    9392.94,
    9533.92,
    10232.43
   ],
   "min_as_reported": 9292.03,
   "max_as_reported": 11886.24,
   "median_as_reported": 10232.43,
   "spread_pct_as_reported": 27.92,
   "rate_definition": "judged = N / (prefill_ms + dec_ms/dec_tok) for qwen2.5 cores (their prefill clock stops before the last prompt step, which produces token 0, has completed; one decode step is added as an upper bound for it); as reported (N / prefill_ms) for qwen3-1.7b, whose clock stops at the completion of that step, the same span as mlx_lm prompt_time",
   "value_bound": "lower",
   "value_bound_note": "prefill: one whole decode step is added for the last prompt step, an upper bound on what the prefill clock leaves out; follow-up: have the qwen2.5 drivers end the prefill clock at the completion of the last prompt step so this becomes exact",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "variance_cause_scope_check": {
    "foreign_evidence": "none found (no in-run foreign watch in this harness; machine file activity checked for the cell window)",
    "rule": "device scope requires a clean quiet gate on every run of every engine in the cell and veizik and both MLX baselines all wide (spread > 10%)",
    "veizik_spread_pct": 26.33,
    "baseline_spread_pct": {
     "mlx": 19.17,
     "mlx_bf16": 8.36
    },
    "quiet_check_tally": {
     "veizik": {
      "runs": 11,
      "gpu_idle_pct_max": 0.0,
      "gpu_threshold_pct": 5.0,
      "runs_over_gpu_threshold": 0,
      "foreign_process_samples": 0,
      "foreign_watch": "recorded (fg)"
     },
     "mlx": {
      "runs": 11,
      "gpu_idle_pct_max": 0.0,
      "gpu_threshold_pct": 5.0,
      "runs_over_gpu_threshold": 0,
      "foreign_process_samples": 0,
      "foreign_watch": "recorded (fg)"
     },
     "mlx_bf16": {
      "runs": 11,
      "gpu_idle_pct_max": 0.0,
      "gpu_threshold_pct": 5.0,
      "runs_over_gpu_threshold": 0,
      "foreign_process_samples": 0,
      "foreign_watch": "recorded (fg)"
     }
    },
    "result": "device scope withheld: 7 of 11 kept runs per engine predate the 13:27 pause of mediaanalysisd/photoanalysisd and 4 follow it, so a device attribution cannot be shown"
   },
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "decode",
   "N": 512,
   "engine": "veizik",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.25,
    2.08,
    2.08,
    2.25,
    2.25,
    2.31,
    2.2,
    3.76,
    3.3,
    2.76,
    2.65
   ],
   "load_after": [
    2.25,
    2.08,
    2.08,
    2.25,
    2.23,
    2.31,
    2.2,
    3.53,
    3.3,
    2.76,
    2.65
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "row_id": "m1-ultra/qwen2.5-0.5b/veizik/decode/512/kit_r5_pd1",
   "unit": "tok/s (core self-clock, CLI \"generated\" line: 127 decoded tokens after the prompt-pass token, @ KV depth N; --ignore-eos)",
   "measured_binary_sha256": "8e9e22a82073be4814699b6271be95f1045e1eeb8a8ba59bd912692289d7d659",
   "measured_binary_digest_kind": "sha256 of __TEXT,__text section",
   "core_file_sha256": "c7678f0931a848f593dedd5b07c8d505f5d44f22b6fc1a7e713fa15232f23896",
   "measured_cli_sha256": "a9eb4d9b8018432f27ce95abd0e0dc7bc63fa23a4695592d9a71858819f4713a",
   "cli_digest_kind": "sha256 of the installed CLI binary (file)",
   "snapshot": "060db6499f32faf8b98477b0a26969ef7d8b9987",
   "precision_mode": "as-shipped",
   "precision": "veizik shipped low-bit core (installed release)",
   "command": "veizik run qwen2.5-0.5b --tokens 128 --ignore-eos \"<p512 prompt file>\"  (customer CLI, no VZ_* env)",
   "input_tokens": 512,
   "checkpoint": "Qwen base HF snapshot 060db649",
   "runs": [
    375.49,
    377.18,
    376.36,
    376.15,
    374.24,
    374.48,
    388.75,
    379.22,
    381.4,
    373.83,
    374.84
   ],
   "min": 373.83,
   "max": 388.75,
   "median": 376.15,
   "spread_pct": 3.99,
   "status": "ok",
   "runs_as_reported": [
    375.49,
    377.18,
    376.36,
    376.15,
    374.24,
    374.48,
    388.75,
    379.22,
    381.4,
    373.83,
    374.84
   ],
   "min_as_reported": 373.83,
   "max_as_reported": 388.75,
   "median_as_reported": 376.15,
   "spread_pct_as_reported": 3.99,
   "rate_definition": "judged = 127 decoded tokens / their time (CLI \"+ 127 decoded in X ms\"); as reported",
   "value_bound": "lower",
   "value_bound_note": "decode: the decode window also contains the completion of the last prompt step (token 0), so 127/window slightly understates the decode rate; follow-up: have the qwen2.5 drivers end the prefill clock at the completion of the last prompt step so this becomes exact",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "prefill",
   "N": 3968,
   "engine": "mlx_bf16",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.29,
    2.18,
    1.93,
    1.94,
    1.58,
    1.93,
    2.19,
    2.25,
    2.1,
    2.01,
    2.15
   ],
   "load_after": [
    2.18,
    2.01,
    2.02,
    1.94,
    1.93,
    1.86,
    2.25,
    2.31,
    2.01,
    2.09,
    2.05
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "as-shipped",
   "precision": "bf16 (HF base snapshot, unquantised)",
   "input_tokens": 3968,
   "peak_memory": [
    "1.780GB",
    "1.780GB",
    "1.780GB",
    "1.780GB",
    "1.780GB",
    "1.780GB",
    "1.783GB",
    "1.780GB",
    "1.780GB",
    "1.780GB",
    "1.780GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "92df8443248a",
    "92df8443248a",
    "92df8443248a",
    "92df8443248a",
    "92df8443248a",
    "92df8443248a",
    "92df8443248a",
    "92df8443248a",
    "92df8443248a",
    "92df8443248a",
    "92df8443248a"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/prefill/3968/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    10655.87,
    10566.792,
    10804.692,
    10903.031,
    10646.228,
    10867.005,
    10977.602,
    11091.685,
    10794.531,
    10886.414,
    10788.622
   ],
   "min": 10566.792,
   "max": 11091.685,
   "median": 10804.692,
   "spread_pct": 4.97,
   "status": "ok",
   "runs_as_reported": [
    10655.87,
    10566.792,
    10804.692,
    10903.031,
    10646.228,
    10867.005,
    10977.602,
    11091.685,
    10794.531,
    10886.414,
    10788.622
   ],
   "min_as_reported": 10566.792,
   "max_as_reported": 11091.685,
   "median_as_reported": 10804.692,
   "spread_pct_as_reported": 4.97,
   "rate_definition": "judged = mlx_lm prompt_tps = N / prompt_time (prompt pass through token 0), as reported",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "decode",
   "N": 3968,
   "engine": "mlx_bf16",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.29,
    2.18,
    1.93,
    1.94,
    1.58,
    1.93,
    2.19,
    2.25,
    2.1,
    2.01,
    2.15
   ],
   "load_after": [
    2.18,
    2.01,
    2.02,
    1.94,
    1.93,
    1.86,
    2.25,
    2.31,
    2.01,
    2.09,
    2.05
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "as-shipped",
   "precision": "bf16 (HF base snapshot, unquantised)",
   "input_tokens": 3968,
   "peak_memory": [
    "1.780GB",
    "1.780GB",
    "1.780GB",
    "1.780GB",
    "1.780GB",
    "1.780GB",
    "1.783GB",
    "1.780GB",
    "1.780GB",
    "1.780GB",
    "1.780GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "92df8443248a",
    "92df8443248a",
    "92df8443248a",
    "92df8443248a",
    "92df8443248a",
    "92df8443248a",
    "92df8443248a",
    "92df8443248a",
    "92df8443248a",
    "92df8443248a",
    "92df8443248a"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/decode/3968/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    221.6775078125,
    221.55546875000002,
    221.4959375,
    222.0188203125,
    221.69735156250002,
    221.922578125,
    223.0407734375,
    221.50982812499998,
    221.7806953125,
    222.13490625,
    221.96226562500001
   ],
   "min": 221.4959375,
   "max": 223.0407734375,
   "median": 221.7806953125,
   "spread_pct": 0.7,
   "status": "ok",
   "runs_as_reported": [
    223.423,
    223.3,
    223.24,
    223.767,
    223.443,
    223.67,
    224.797,
    223.254,
    223.527,
    223.884,
    223.71
   ],
   "min_as_reported": 223.24,
   "max_as_reported": 224.797,
   "median_as_reported": 223.527,
   "spread_pct_as_reported": 0.7,
   "rate_definition": "judged = generation_tps x (gen_tok-1)/gen_tok = 127 tokens / mlx_lm generation window; as_reported = mlx_lm generation_tps = 128/window",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "prefill",
   "N": 3968,
   "engine": "mlx",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.31,
    2.01,
    1.93,
    1.94,
    1.72,
    1.86,
    2.29,
    2.31,
    2.11,
    2.08,
    2.07
   ],
   "load_after": [
    2.31,
    2.01,
    1.93,
    1.94,
    1.58,
    2.35,
    2.29,
    2.21,
    2.1,
    2.08,
    2.15
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "matched",
   "precision": "q4 g64 affine (mlx-community-style 4-bit g64, HF-converted)",
   "input_tokens": 3968,
   "peak_memory": [
    "1.108GB",
    "1.108GB",
    "1.108GB",
    "1.108GB",
    "1.108GB",
    "1.108GB",
    "1.108GB",
    "1.108GB",
    "1.108GB",
    "1.108GB",
    "1.108GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "46c55acbdf1e",
    "46c55acbdf1e",
    "46c55acbdf1e",
    "46c55acbdf1e",
    "46c55acbdf1e",
    "46c55acbdf1e",
    "46c55acbdf1e",
    "46c55acbdf1e",
    "46c55acbdf1e",
    "46c55acbdf1e",
    "46c55acbdf1e"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/prefill/3968/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    9845.843,
    9926.939,
    9748.533,
    9819.953,
    9870.471,
    9288.603,
    9726.91,
    9836.594,
    9905.987,
    9600.464,
    9702.215
   ],
   "min": 9288.603,
   "max": 9926.939,
   "median": 9819.953,
   "spread_pct": 6.87,
   "status": "ok",
   "runs_as_reported": [
    9845.843,
    9926.939,
    9748.533,
    9819.953,
    9870.471,
    9288.603,
    9726.91,
    9836.594,
    9905.987,
    9600.464,
    9702.215
   ],
   "min_as_reported": 9288.603,
   "max_as_reported": 9926.939,
   "median_as_reported": 9819.953,
   "spread_pct_as_reported": 6.87,
   "rate_definition": "judged = mlx_lm prompt_tps = N / prompt_time (prompt pass through token 0), as reported",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "decode",
   "N": 3968,
   "engine": "mlx",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.31,
    2.01,
    1.93,
    1.94,
    1.72,
    1.86,
    2.29,
    2.31,
    2.11,
    2.08,
    2.07
   ],
   "load_after": [
    2.31,
    2.01,
    1.93,
    1.94,
    1.58,
    2.35,
    2.29,
    2.21,
    2.1,
    2.08,
    2.15
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "matched",
   "precision": "q4 g64 affine (mlx-community-style 4-bit g64, HF-converted)",
   "input_tokens": 3968,
   "peak_memory": [
    "1.108GB",
    "1.108GB",
    "1.108GB",
    "1.108GB",
    "1.108GB",
    "1.108GB",
    "1.108GB",
    "1.108GB",
    "1.108GB",
    "1.108GB",
    "1.108GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "46c55acbdf1e",
    "46c55acbdf1e",
    "46c55acbdf1e",
    "46c55acbdf1e",
    "46c55acbdf1e",
    "46c55acbdf1e",
    "46c55acbdf1e",
    "46c55acbdf1e",
    "46c55acbdf1e",
    "46c55acbdf1e",
    "46c55acbdf1e"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/decode/3968/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    238.7371796875,
    238.3045859375,
    241.55896093750002,
    238.71336718749998,
    239.3245546875,
    239.1856484375,
    238.6836015625,
    238.80167187499998,
    239.3364609375,
    239.23823437500002,
    239.1171875
   ],
   "min": 238.3045859375,
   "max": 241.55896093750002,
   "median": 239.1171875,
   "spread_pct": 1.37,
   "status": "ok",
   "runs_as_reported": [
    240.617,
    240.181,
    243.461,
    240.593,
    241.209,
    241.069,
    240.563,
    240.682,
    241.221,
    241.122,
    241.0
   ],
   "min_as_reported": 240.181,
   "max_as_reported": 243.461,
   "median_as_reported": 241.0,
   "spread_pct_as_reported": 1.37,
   "rate_definition": "judged = generation_tps x (gen_tok-1)/gen_tok = 127 tokens / mlx_lm generation window; as_reported = mlx_lm generation_tps = 128/window",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "prefill",
   "N": 3968,
   "engine": "veizik",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.08,
    2.01,
    1.93,
    1.86,
    1.86,
    2.35,
    2.32,
    2.21,
    2.11,
    2.08,
    2.08
   ],
   "load_after": [
    2.08,
    2.01,
    1.93,
    1.86,
    1.72,
    2.32,
    2.32,
    2.21,
    2.11,
    2.08,
    2.07
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "row_id": "m1-ultra/qwen2.5-0.5b/veizik/prefill/3968/kit_r5_pd1",
   "unit": "tok/s (core self-clock, CLI \"prefill : <ms> ms for <N> positions\" line)",
   "measured_binary_sha256": "8e9e22a82073be4814699b6271be95f1045e1eeb8a8ba59bd912692289d7d659",
   "measured_binary_digest_kind": "sha256 of __TEXT,__text section",
   "core_file_sha256": "c7678f0931a848f593dedd5b07c8d505f5d44f22b6fc1a7e713fa15232f23896",
   "measured_cli_sha256": "a9eb4d9b8018432f27ce95abd0e0dc7bc63fa23a4695592d9a71858819f4713a",
   "cli_digest_kind": "sha256 of the installed CLI binary (file)",
   "snapshot": "060db6499f32faf8b98477b0a26969ef7d8b9987",
   "precision_mode": "as-shipped",
   "precision": "veizik shipped low-bit core (installed release)",
   "command": "veizik run qwen2.5-0.5b --tokens 128 --ignore-eos \"<p3968 prompt file>\"  (customer CLI, no VZ_* env)",
   "input_tokens": 3968,
   "checkpoint": "Qwen base HF snapshot 060db649",
   "runs": [
    13082.19494047793,
    13034.575680119444,
    13152.369930030945,
    13033.192852672104,
    12915.215399844343,
    13077.751579449347,
    12984.268102503716,
    12846.787170048337,
    12881.80156673848,
    13017.63723798129,
    13102.342629785371
   ],
   "min": 12846.787170048337,
   "max": 13152.369930030945,
   "median": 13033.192852672104,
   "spread_pct": 2.38,
   "status": "ok",
   "runs_as_reported": [
    13225.61,
    13176.42,
    13297.32,
    13175.33,
    13054.56,
    13221.07,
    13125.95,
    12985.23,
    13019.61,
    13158.16,
    13245.96
   ],
   "min_as_reported": 12985.23,
   "max_as_reported": 13297.32,
   "median_as_reported": 13175.33,
   "spread_pct_as_reported": 2.4,
   "rate_definition": "judged = N / (prefill_ms + dec_ms/dec_tok) for qwen2.5 cores (their prefill clock stops before the last prompt step, which produces token 0, has completed; one decode step is added as an upper bound for it); as reported (N / prefill_ms) for qwen3-1.7b, whose clock stops at the completion of that step, the same span as mlx_lm prompt_time",
   "value_bound": "lower",
   "value_bound_note": "prefill: one whole decode step is added for the last prompt step, an upper bound on what the prefill clock leaves out; follow-up: have the qwen2.5 drivers end the prefill clock at the completion of the last prompt step so this becomes exact",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "decode",
   "N": 3968,
   "engine": "veizik",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.08,
    2.01,
    1.93,
    1.86,
    1.86,
    2.35,
    2.32,
    2.21,
    2.11,
    2.08,
    2.08
   ],
   "load_after": [
    2.08,
    2.01,
    1.93,
    1.86,
    1.72,
    2.32,
    2.32,
    2.21,
    2.11,
    2.08,
    2.07
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "row_id": "m1-ultra/qwen2.5-0.5b/veizik/decode/3968/kit_r5_pd1",
   "unit": "tok/s (core self-clock, CLI \"generated\" line: 127 decoded tokens after the prompt-pass token, @ KV depth N; --ignore-eos)",
   "measured_binary_sha256": "8e9e22a82073be4814699b6271be95f1045e1eeb8a8ba59bd912692289d7d659",
   "measured_binary_digest_kind": "sha256 of __TEXT,__text section",
   "core_file_sha256": "c7678f0931a848f593dedd5b07c8d505f5d44f22b6fc1a7e713fa15232f23896",
   "measured_cli_sha256": "a9eb4d9b8018432f27ce95abd0e0dc7bc63fa23a4695592d9a71858819f4713a",
   "cli_digest_kind": "sha256 of the installed CLI binary (file)",
   "snapshot": "060db6499f32faf8b98477b0a26969ef7d8b9987",
   "precision_mode": "as-shipped",
   "precision": "veizik shipped low-bit core (installed release)",
   "command": "veizik run qwen2.5-0.5b --tokens 128 --ignore-eos \"<p3968 prompt file>\"  (customer CLI, no VZ_* env)",
   "input_tokens": 3968,
   "checkpoint": "Qwen base HF snapshot 060db649",
   "runs": [
    304.04,
    305.15,
    304.07,
    304.47,
    304.92,
    304.04,
    303.16,
    303.67,
    306.71,
    307.2,
    304.54
   ],
   "min": 303.16,
   "max": 307.2,
   "median": 304.47,
   "spread_pct": 1.33,
   "status": "ok",
   "runs_as_reported": [
    304.04,
    305.15,
    304.07,
    304.47,
    304.92,
    304.04,
    303.16,
    303.67,
    306.71,
    307.2,
    304.54
   ],
   "min_as_reported": 303.16,
   "max_as_reported": 307.2,
   "median_as_reported": 304.47,
   "spread_pct_as_reported": 1.33,
   "rate_definition": "judged = 127 decoded tokens / their time (CLI \"+ 127 decoded in X ms\"); as reported",
   "value_bound": "lower",
   "value_bound_note": "decode: the decode window also contains the completion of the last prompt step (token 0), so 127/window slightly understates the decode rate; follow-up: have the qwen2.5 drivers end the prefill clock at the completion of the last prompt step so this becomes exact",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "prefill",
   "N": 8192,
   "engine": "mlx_bf16",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    3.52,
    3.24,
    2.12,
    2.02,
    2.16,
    2.15,
    1.96,
    1.88,
    1.56,
    1.52,
    1.59
   ],
   "load_after": [
    3.24,
    3.06,
    2.11,
    2.1,
    2.15,
    2.05,
    1.88,
    1.81,
    1.52,
    1.48,
    1.54
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "as-shipped",
   "precision": "bf16 (HF base snapshot, unquantised)",
   "input_tokens": 8192,
   "peak_memory": [
    "1.879GB",
    "1.879GB",
    "1.879GB",
    "1.879GB",
    "1.879GB",
    "1.879GB",
    "1.879GB",
    "1.879GB",
    "1.879GB",
    "1.879GB",
    "1.879GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "763ebf059043",
    "763ebf059043",
    "763ebf059043",
    "763ebf059043",
    "763ebf059043",
    "763ebf059043",
    "763ebf059043",
    "763ebf059043",
    "763ebf059043",
    "763ebf059043",
    "763ebf059043"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/prefill/8192/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    10194.073,
    10119.599,
    10233.746,
    10123.878,
    10081.332,
    10166.461,
    10100.941,
    10120.465,
    10182.677,
    9863.123,
    10113.423
   ],
   "min": 9863.123,
   "max": 10233.746,
   "median": 10120.465,
   "spread_pct": 3.76,
   "status": "ok",
   "runs_as_reported": [
    10194.073,
    10119.599,
    10233.746,
    10123.878,
    10081.332,
    10166.461,
    10100.941,
    10120.465,
    10182.677,
    9863.123,
    10113.423
   ],
   "min_as_reported": 9863.123,
   "max_as_reported": 10233.746,
   "median_as_reported": 10120.465,
   "spread_pct_as_reported": 3.76,
   "rate_definition": "judged = mlx_lm prompt_tps = N / prompt_time (prompt pass through token 0), as reported",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "decode",
   "N": 8192,
   "engine": "mlx_bf16",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    3.52,
    3.24,
    2.12,
    2.02,
    2.16,
    2.15,
    1.96,
    1.88,
    1.56,
    1.52,
    1.59
   ],
   "load_after": [
    3.24,
    3.06,
    2.11,
    2.1,
    2.15,
    2.05,
    1.88,
    1.81,
    1.52,
    1.48,
    1.54
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "as-shipped",
   "precision": "bf16 (HF base snapshot, unquantised)",
   "input_tokens": 8192,
   "peak_memory": [
    "1.879GB",
    "1.879GB",
    "1.879GB",
    "1.879GB",
    "1.879GB",
    "1.879GB",
    "1.879GB",
    "1.879GB",
    "1.879GB",
    "1.879GB",
    "1.879GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "763ebf059043",
    "763ebf059043",
    "763ebf059043",
    "763ebf059043",
    "763ebf059043",
    "763ebf059043",
    "763ebf059043",
    "763ebf059043",
    "763ebf059043",
    "763ebf059043",
    "763ebf059043"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/decode/8192/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    192.3345546875,
    192.7909609375,
    192.58359375,
    193.8605390625,
    192.7155546875,
    196.15546874999998,
    193.2017265625,
    194.0252421875,
    192.4605625,
    192.6421328125,
    192.4893359375
   ],
   "min": 192.3345546875,
   "max": 196.15546874999998,
   "median": 192.7155546875,
   "spread_pct": 1.99,
   "status": "ok",
   "runs_as_reported": [
    193.849,
    194.309,
    194.1,
    195.387,
    194.233,
    197.7,
    194.723,
    195.553,
    193.976,
    194.159,
    194.005
   ],
   "min_as_reported": 193.849,
   "max_as_reported": 197.7,
   "median_as_reported": 194.233,
   "spread_pct_as_reported": 1.99,
   "rate_definition": "judged = generation_tps x (gen_tok-1)/gen_tok = 127 tokens / mlx_lm generation window; as_reported = mlx_lm generation_tps = 128/window",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "prefill",
   "N": 8192,
   "engine": "mlx",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.09,
    3.06,
    2.14,
    2.1,
    2.09,
    2.05,
    2.04,
    1.76,
    1.57,
    1.48,
    1.64
   ],
   "load_after": [
    3.52,
    2.9,
    2.12,
    2.01,
    2.08,
    1.97,
    1.96,
    1.62,
    1.52,
    1.76,
    1.59
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "matched",
   "precision": "q4 g64 affine (mlx-community-style 4-bit g64, HF-converted)",
   "input_tokens": 8192,
   "peak_memory": [
    "1.232GB",
    "1.232GB",
    "1.232GB",
    "1.232GB",
    "1.232GB",
    "1.232GB",
    "1.232GB",
    "1.232GB",
    "1.232GB",
    "1.232GB",
    "1.232GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "f93e0c968dfc",
    "f93e0c968dfc",
    "f93e0c968dfc",
    "f93e0c968dfc",
    "f93e0c968dfc",
    "f93e0c968dfc",
    "f93e0c968dfc",
    "f93e0c968dfc",
    "f93e0c968dfc",
    "f93e0c968dfc",
    "f93e0c968dfc"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/prefill/8192/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    9204.475,
    9199.453,
    9205.027,
    9127.496,
    9230.929,
    9248.392,
    9248.42,
    9187.791,
    9150.468,
    9254.208,
    9197.904
   ],
   "min": 9127.496,
   "max": 9254.208,
   "median": 9204.475,
   "spread_pct": 1.39,
   "status": "ok",
   "runs_as_reported": [
    9204.475,
    9199.453,
    9205.027,
    9127.496,
    9230.929,
    9248.392,
    9248.42,
    9187.791,
    9150.468,
    9254.208,
    9197.904
   ],
   "min_as_reported": 9127.496,
   "max_as_reported": 9254.208,
   "median_as_reported": 9204.475,
   "spread_pct_as_reported": 1.39,
   "rate_definition": "judged = mlx_lm prompt_tps = N / prompt_time (prompt pass through token 0), as reported",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "decode",
   "N": 8192,
   "engine": "mlx",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.09,
    3.06,
    2.14,
    2.1,
    2.09,
    2.05,
    2.04,
    1.76,
    1.57,
    1.48,
    1.64
   ],
   "load_after": [
    3.52,
    2.9,
    2.12,
    2.01,
    2.08,
    1.97,
    1.96,
    1.62,
    1.52,
    1.76,
    1.59
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "matched",
   "precision": "q4 g64 affine (mlx-community-style 4-bit g64, HF-converted)",
   "input_tokens": 8192,
   "peak_memory": [
    "1.232GB",
    "1.232GB",
    "1.232GB",
    "1.232GB",
    "1.232GB",
    "1.232GB",
    "1.232GB",
    "1.232GB",
    "1.232GB",
    "1.232GB",
    "1.232GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "f93e0c968dfc",
    "f93e0c968dfc",
    "f93e0c968dfc",
    "f93e0c968dfc",
    "f93e0c968dfc",
    "f93e0c968dfc",
    "f93e0c968dfc",
    "f93e0c968dfc",
    "f93e0c968dfc",
    "f93e0c968dfc",
    "f93e0c968dfc"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/decode/8192/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    202.37053125,
    204.31125,
    203.1176484375,
    202.692,
    202.37846875,
    202.2346015625,
    201.7633125,
    204.464046875,
    202.56202343750002,
    202.4231171875,
    202.0063984375
   ],
   "min": 201.7633125,
   "max": 204.464046875,
   "median": 202.4231171875,
   "spread_pct": 1.34,
   "status": "ok",
   "runs_as_reported": [
    203.964,
    205.92,
    204.717,
    204.288,
    203.972,
    203.827,
    203.352,
    206.074,
    204.157,
    204.017,
    203.597
   ],
   "min_as_reported": 203.352,
   "max_as_reported": 206.074,
   "median_as_reported": 204.017,
   "spread_pct_as_reported": 1.34,
   "rate_definition": "judged = generation_tps x (gen_tok-1)/gen_tok = 127 tokens / mlx_lm generation window; as_reported = mlx_lm generation_tps = 128/window",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "prefill",
   "N": 8192,
   "engine": "veizik",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.1,
    2.23,
    2.14,
    2.01,
    2.09,
    1.97,
    2.13,
    1.62,
    1.57,
    1.76,
    1.7
   ],
   "load_after": [
    2.1,
    2.23,
    2.14,
    2.09,
    2.09,
    2.13,
    2.13,
    1.62,
    1.57,
    1.7,
    1.7
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "row_id": "m1-ultra/qwen2.5-0.5b/veizik/prefill/8192/kit_r5_pd1",
   "unit": "tok/s (core self-clock, CLI \"prefill : <ms> ms for <N> positions\" line)",
   "measured_binary_sha256": "8e9e22a82073be4814699b6271be95f1045e1eeb8a8ba59bd912692289d7d659",
   "measured_binary_digest_kind": "sha256 of __TEXT,__text section",
   "core_file_sha256": "c7678f0931a848f593dedd5b07c8d505f5d44f22b6fc1a7e713fa15232f23896",
   "measured_cli_sha256": "a9eb4d9b8018432f27ce95abd0e0dc7bc63fa23a4695592d9a71858819f4713a",
   "cli_digest_kind": "sha256 of the installed CLI binary (file)",
   "snapshot": "060db6499f32faf8b98477b0a26969ef7d8b9987",
   "precision_mode": "as-shipped",
   "precision": "veizik shipped low-bit core (installed release)",
   "command": "veizik run qwen2.5-0.5b --tokens 128 --ignore-eos \"<p8192 prompt file>\"  (customer CLI, no VZ_* env)",
   "input_tokens": 8192,
   "checkpoint": "Qwen base HF snapshot 060db649",
   "runs": [
    11317.469305760374,
    11271.423807914303,
    11399.90336450027,
    11300.980054658306,
    11267.974056964427,
    11291.628481042058,
    11276.65990428804,
    11282.716572033287,
    11283.683285999665,
    11301.352505055434,
    11354.481595170731
   ],
   "min": 11267.974056964427,
   "max": 11399.90336450027,
   "median": 11291.628481042058,
   "spread_pct": 1.17,
   "status": "ok",
   "runs_as_reported": [
    11374.0,
    11327.06,
    11456.94,
    11356.85,
    11323.6,
    11347.47,
    11332.57,
    11338.55,
    11338.94,
    11357.37,
    11411.28
   ],
   "min_as_reported": 11323.6,
   "max_as_reported": 11456.94,
   "median_as_reported": 11347.47,
   "spread_pct_as_reported": 1.18,
   "rate_definition": "judged = N / (prefill_ms + dec_ms/dec_tok) for qwen2.5 cores (their prefill clock stops before the last prompt step, which produces token 0, has completed; one decode step is added as an upper bound for it); as reported (N / prefill_ms) for qwen3-1.7b, whose clock stops at the completion of that step, the same span as mlx_lm prompt_time",
   "value_bound": "lower",
   "value_bound_note": "prefill: one whole decode step is added for the last prompt step, an upper bound on what the prefill clock leaves out; follow-up: have the qwen2.5 drivers end the prefill clock at the completion of the last prompt step so this becomes exact",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "decode",
   "N": 8192,
   "engine": "veizik",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.1,
    2.23,
    2.14,
    2.01,
    2.09,
    1.97,
    2.13,
    1.62,
    1.57,
    1.76,
    1.7
   ],
   "load_after": [
    2.1,
    2.23,
    2.14,
    2.09,
    2.09,
    2.13,
    2.13,
    1.62,
    1.57,
    1.7,
    1.7
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "row_id": "m1-ultra/qwen2.5-0.5b/veizik/decode/8192/kit_r5_pd1",
   "unit": "tok/s (core self-clock, CLI \"generated\" line: 127 decoded tokens after the prompt-pass token, @ KV depth N; --ignore-eos)",
   "measured_binary_sha256": "8e9e22a82073be4814699b6271be95f1045e1eeb8a8ba59bd912692289d7d659",
   "measured_binary_digest_kind": "sha256 of __TEXT,__text section",
   "core_file_sha256": "c7678f0931a848f593dedd5b07c8d505f5d44f22b6fc1a7e713fa15232f23896",
   "measured_cli_sha256": "a9eb4d9b8018432f27ce95abd0e0dc7bc63fa23a4695592d9a71858819f4713a",
   "cli_digest_kind": "sha256 of the installed CLI binary (file)",
   "snapshot": "060db6499f32faf8b98477b0a26969ef7d8b9987",
   "precision_mode": "as-shipped",
   "precision": "veizik shipped low-bit core (installed release)",
   "command": "veizik run qwen2.5-0.5b --tokens 128 --ignore-eos \"<p8192 prompt file>\"  (customer CLI, no VZ_* env)",
   "input_tokens": 8192,
   "checkpoint": "Qwen base HF snapshot 060db649",
   "runs": [
    277.95,
    280.13,
    279.52,
    280.44,
    280.02,
    280.09,
    279.0,
    279.7,
    282.64,
    279.72,
    278.46
   ],
   "min": 277.95,
   "max": 282.64,
   "median": 279.72,
   "spread_pct": 1.69,
   "status": "ok",
   "runs_as_reported": [
    277.95,
    280.13,
    279.52,
    280.44,
    280.02,
    280.09,
    279.0,
    279.7,
    282.64,
    279.72,
    278.46
   ],
   "min_as_reported": 277.95,
   "max_as_reported": 282.64,
   "median_as_reported": 279.72,
   "spread_pct_as_reported": 1.69,
   "rate_definition": "judged = 127 decoded tokens / their time (CLI \"+ 127 decoded in X ms\"); as reported",
   "value_bound": "lower",
   "value_bound_note": "decode: the decode window also contains the completion of the last prompt step (token 0), so 127/window slightly understates the decode rate; follow-up: have the qwen2.5 drivers end the prefill clock at the completion of the last prompt step so this becomes exact",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "prefill",
   "N": 16384,
   "engine": "mlx_bf16",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    1.08,
    1.22,
    1.68,
    1.57,
    1.64,
    1.54,
    1.39,
    2.01,
    2.2,
    2.02,
    1.91
   ],
   "load_after": [
    1.22,
    1.21,
    1.57,
    1.45,
    1.59,
    1.5,
    1.84,
    1.85,
    2.11,
    1.94,
    1.84
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "as-shipped",
   "precision": "bf16 (HF base snapshot, unquantised)",
   "input_tokens": 16384,
   "peak_memory": [
    "1.990GB",
    "1.990GB",
    "1.990GB",
    "1.990GB",
    "1.990GB",
    "1.990GB",
    "1.990GB",
    "1.990GB",
    "1.990GB",
    "1.990GB",
    "1.990GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "608d22bdecb6",
    "608d22bdecb6",
    "608d22bdecb6",
    "608d22bdecb6",
    "608d22bdecb6",
    "608d22bdecb6",
    "608d22bdecb6",
    "608d22bdecb6",
    "608d22bdecb6",
    "608d22bdecb6",
    "608d22bdecb6"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/prefill/16384/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    8076.11,
    8083.094,
    8056.168,
    8100.228,
    8084.812,
    7967.247,
    8106.921,
    8088.558,
    8073.437,
    8098.903,
    8076.747
   ],
   "min": 7967.247,
   "max": 8106.921,
   "median": 8083.094,
   "spread_pct": 1.75,
   "status": "ok",
   "runs_as_reported": [
    8076.11,
    8083.094,
    8056.168,
    8100.228,
    8084.812,
    7967.247,
    8106.921,
    8088.558,
    8073.437,
    8098.903,
    8076.747
   ],
   "min_as_reported": 7967.247,
   "max_as_reported": 8106.921,
   "median_as_reported": 8083.094,
   "spread_pct_as_reported": 1.75,
   "rate_definition": "judged = mlx_lm prompt_tps = N / prompt_time (prompt pass through token 0), as reported",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "decode",
   "N": 16384,
   "engine": "mlx_bf16",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    1.08,
    1.22,
    1.68,
    1.57,
    1.64,
    1.54,
    1.39,
    2.01,
    2.2,
    2.02,
    1.91
   ],
   "load_after": [
    1.22,
    1.21,
    1.57,
    1.45,
    1.59,
    1.5,
    1.84,
    1.85,
    2.11,
    1.94,
    1.84
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "as-shipped",
   "precision": "bf16 (HF base snapshot, unquantised)",
   "input_tokens": 16384,
   "peak_memory": [
    "1.990GB",
    "1.990GB",
    "1.990GB",
    "1.990GB",
    "1.990GB",
    "1.990GB",
    "1.990GB",
    "1.990GB",
    "1.990GB",
    "1.990GB",
    "1.990GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "608d22bdecb6",
    "608d22bdecb6",
    "608d22bdecb6",
    "608d22bdecb6",
    "608d22bdecb6",
    "608d22bdecb6",
    "608d22bdecb6",
    "608d22bdecb6",
    "608d22bdecb6",
    "608d22bdecb6",
    "608d22bdecb6"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/decode/16384/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    171.4609140625,
    171.73178125,
    172.0026484375,
    171.6573671875,
    171.987765625,
    171.8419140625,
    171.567078125,
    171.529375,
    171.77642968749998,
    171.48175,
    171.91732031249998
   ],
   "min": 171.4609140625,
   "max": 172.0026484375,
   "median": 171.73178125,
   "spread_pct": 0.32,
   "status": "ok",
   "runs_as_reported": [
    172.811,
    173.084,
    173.357,
    173.009,
    173.342,
    173.195,
    172.918,
    172.88,
    173.129,
    172.832,
    173.271
   ],
   "min_as_reported": 172.811,
   "max_as_reported": 173.357,
   "median_as_reported": 173.084,
   "spread_pct_as_reported": 0.32,
   "rate_definition": "judged = generation_tps x (gen_tok-1)/gen_tok = 127 tokens / mlx_lm generation window; as_reported = mlx_lm generation_tps = 128/window",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "prefill",
   "N": 16384,
   "engine": "mlx",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    1.46,
    1.21,
    1.74,
    1.45,
    1.44,
    1.5,
    1.42,
    1.85,
    2.04,
    1.94,
    1.73
   ],
   "load_after": [
    1.35,
    1.49,
    1.68,
    1.61,
    1.64,
    1.46,
    1.39,
    1.78,
    2.2,
    1.86,
    1.91
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "matched",
   "precision": "q4 g64 affine (mlx-community-style 4-bit g64, HF-converted)",
   "input_tokens": 16384,
   "peak_memory": [
    "1.362GB",
    "1.362GB",
    "1.362GB",
    "1.362GB",
    "1.362GB",
    "1.362GB",
    "1.362GB",
    "1.362GB",
    "1.362GB",
    "1.362GB",
    "1.362GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "addc190963a1",
    "addc190963a1",
    "addc190963a1",
    "addc190963a1",
    "addc190963a1",
    "addc190963a1",
    "addc190963a1",
    "addc190963a1",
    "addc190963a1",
    "addc190963a1",
    "addc190963a1"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/prefill/16384/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    7479.667,
    7474.017,
    7512.967,
    7490.749,
    7513.016,
    7493.721,
    7505.007,
    7493.284,
    7481.482,
    7487.95,
    7474.895
   ],
   "min": 7474.017,
   "max": 7513.016,
   "median": 7490.749,
   "spread_pct": 0.52,
   "status": "ok",
   "runs_as_reported": [
    7479.667,
    7474.017,
    7512.967,
    7490.749,
    7513.016,
    7493.721,
    7505.007,
    7493.284,
    7481.482,
    7487.95,
    7474.895
   ],
   "min_as_reported": 7474.017,
   "max_as_reported": 7513.016,
   "median_as_reported": 7490.749,
   "spread_pct_as_reported": 0.52,
   "rate_definition": "judged = mlx_lm prompt_tps = N / prompt_time (prompt pass through token 0), as reported",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "decode",
   "N": 16384,
   "engine": "mlx",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    1.46,
    1.21,
    1.74,
    1.45,
    1.44,
    1.5,
    1.42,
    1.85,
    2.04,
    1.94,
    1.73
   ],
   "load_after": [
    1.35,
    1.49,
    1.68,
    1.61,
    1.64,
    1.46,
    1.39,
    1.78,
    2.2,
    1.86,
    1.91
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "matched",
   "precision": "q4 g64 affine (mlx-community-style 4-bit g64, HF-converted)",
   "input_tokens": 16384,
   "peak_memory": [
    "1.362GB",
    "1.362GB",
    "1.362GB",
    "1.362GB",
    "1.362GB",
    "1.362GB",
    "1.362GB",
    "1.362GB",
    "1.362GB",
    "1.362GB",
    "1.362GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "addc190963a1",
    "addc190963a1",
    "addc190963a1",
    "addc190963a1",
    "addc190963a1",
    "addc190963a1",
    "addc190963a1",
    "addc190963a1",
    "addc190963a1",
    "addc190963a1",
    "addc190963a1"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/decode/16384/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    177.0052578125,
    176.9745,
    176.619296875,
    177.34359375,
    177.2523125,
    176.742328125,
    176.55282031250002,
    177.4656328125,
    177.9666875,
    177.095546875,
    176.7413359375
   ],
   "min": 176.55282031250002,
   "max": 177.9666875,
   "median": 177.0052578125,
   "spread_pct": 0.8,
   "status": "ok",
   "runs_as_reported": [
    178.399,
    178.368,
    178.01,
    178.74,
    178.648,
    178.134,
    177.943,
    178.863,
    179.368,
    178.49,
    178.133
   ],
   "min_as_reported": 177.943,
   "max_as_reported": 179.368,
   "median_as_reported": 178.399,
   "spread_pct_as_reported": 0.8,
   "rate_definition": "judged = generation_tps x (gen_tok-1)/gen_tok = 127 tokens / mlx_lm generation window; as_reported = mlx_lm generation_tps = 128/window",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "prefill",
   "N": 16384,
   "engine": "veizik",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    1.59,
    1.49,
    1.45,
    1.61,
    1.56,
    1.5,
    1.46,
    1.96,
    2.04,
    1.79,
    1.79
   ],
   "load_after": [
    1.59,
    1.45,
    1.45,
    1.61,
    1.56,
    1.5,
    1.46,
    1.96,
    2.04,
    1.79,
    1.73
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "row_id": "m1-ultra/qwen2.5-0.5b/veizik/prefill/16384/kit_r5_pd1",
   "unit": "tok/s (core self-clock, CLI \"prefill : <ms> ms for <N> positions\" line)",
   "measured_binary_sha256": "8e9e22a82073be4814699b6271be95f1045e1eeb8a8ba59bd912692289d7d659",
   "measured_binary_digest_kind": "sha256 of __TEXT,__text section",
   "core_file_sha256": "c7678f0931a848f593dedd5b07c8d505f5d44f22b6fc1a7e713fa15232f23896",
   "measured_cli_sha256": "a9eb4d9b8018432f27ce95abd0e0dc7bc63fa23a4695592d9a71858819f4713a",
   "cli_digest_kind": "sha256 of the installed CLI binary (file)",
   "snapshot": "060db6499f32faf8b98477b0a26969ef7d8b9987",
   "precision_mode": "as-shipped",
   "precision": "veizik shipped low-bit core (installed release)",
   "command": "veizik run qwen2.5-0.5b --tokens 128 --ignore-eos \"<p16384 prompt file>\"  (customer CLI, no VZ_* env)",
   "input_tokens": 16384,
   "checkpoint": "Qwen base HF snapshot 060db649",
   "runs": [
    8510.342077413483,
    8513.653957966042,
    8361.882137354216,
    8512.135866904868,
    8497.568619881584,
    8504.610184330859,
    8514.29819926248,
    8504.270762260243,
    8397.665438934497,
    8522.81158967831,
    8513.540364505014
   ],
   "min": 8361.882137354216,
   "max": 8522.81158967831,
   "median": 8510.342077413483,
   "spread_pct": 1.92,
   "status": "ok",
   "runs_as_reported": [
    8527.41,
    8530.96,
    8378.47,
    8529.4,
    8514.81,
    8521.85,
    8531.59,
    8521.57,
    8414.44,
    8540.03,
    8531.01
   ],
   "min_as_reported": 8378.47,
   "max_as_reported": 8540.03,
   "median_as_reported": 8527.41,
   "spread_pct_as_reported": 1.93,
   "rate_definition": "judged = N / (prefill_ms + dec_ms/dec_tok) for qwen2.5 cores (their prefill clock stops before the last prompt step, which produces token 0, has completed; one decode step is added as an upper bound for it); as reported (N / prefill_ms) for qwen3-1.7b, whose clock stops at the completion of that step, the same span as mlx_lm prompt_time",
   "value_bound": "lower",
   "value_bound_note": "prefill: one whole decode step is added for the last prompt step, an upper bound on what the prefill clock leaves out; follow-up: have the qwen2.5 drivers end the prefill clock at the completion of the last prompt step so this becomes exact",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "decode",
   "N": 16384,
   "engine": "veizik",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    1.59,
    1.49,
    1.45,
    1.61,
    1.56,
    1.5,
    1.46,
    1.96,
    2.04,
    1.79,
    1.79
   ],
   "load_after": [
    1.59,
    1.45,
    1.45,
    1.61,
    1.56,
    1.5,
    1.46,
    1.96,
    2.04,
    1.79,
    1.73
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "row_id": "m1-ultra/qwen2.5-0.5b/veizik/decode/16384/kit_r5_pd1",
   "unit": "tok/s (core self-clock, CLI \"generated\" line: 127 decoded tokens after the prompt-pass token, @ KV depth N; --ignore-eos)",
   "measured_binary_sha256": "8e9e22a82073be4814699b6271be95f1045e1eeb8a8ba59bd912692289d7d659",
   "measured_binary_digest_kind": "sha256 of __TEXT,__text section",
   "core_file_sha256": "c7678f0931a848f593dedd5b07c8d505f5d44f22b6fc1a7e713fa15232f23896",
   "measured_cli_sha256": "a9eb4d9b8018432f27ce95abd0e0dc7bc63fa23a4695592d9a71858819f4713a",
   "cli_digest_kind": "sha256 of the installed CLI binary (file)",
   "snapshot": "060db6499f32faf8b98477b0a26969ef7d8b9987",
   "precision_mode": "as-shipped",
   "precision": "veizik shipped low-bit core (installed release)",
   "command": "veizik run qwen2.5-0.5b --tokens 128 --ignore-eos \"<p16384 prompt file>\"  (customer CLI, no VZ_* env)",
   "input_tokens": 16384,
   "checkpoint": "Qwen base HF snapshot 060db649",
   "runs": [
    259.47,
    256.21,
    257.77,
    256.65,
    256.15,
    256.57,
    256.44,
    255.73,
    257.11,
    257.98,
    253.69
   ],
   "min": 253.69,
   "max": 259.47,
   "median": 256.57,
   "spread_pct": 2.28,
   "status": "ok",
   "runs_as_reported": [
    259.47,
    256.21,
    257.77,
    256.65,
    256.15,
    256.57,
    256.44,
    255.73,
    257.11,
    257.98,
    253.69
   ],
   "min_as_reported": 253.69,
   "max_as_reported": 259.47,
   "median_as_reported": 256.57,
   "spread_pct_as_reported": 2.28,
   "rate_definition": "judged = 127 decoded tokens / their time (CLI \"+ 127 decoded in X ms\"); as reported",
   "value_bound": "lower",
   "value_bound_note": "decode: the decode window also contains the completion of the last prompt step (token 0), so 127/window slightly understates the decode rate; follow-up: have the qwen2.5 drivers end the prefill clock at the completion of the last prompt step so this becomes exact",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "prefill",
   "N": 24576,
   "engine": "mlx_bf16",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.16,
    2.21,
    1.93,
    1.94,
    1.82,
    2.07,
    2.37,
    2.24,
    1.88,
    1.97,
    2.29
   ],
   "load_after": [
    2.21,
    2.02,
    1.94,
    1.8,
    1.99,
    2.15,
    2.26,
    2.06,
    1.97,
    2.05,
    2.4
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "as-shipped",
   "precision": "bf16 (HF base snapshot, unquantised)",
   "input_tokens": 24576,
   "peak_memory": [
    "1.994GB",
    "1.994GB",
    "1.994GB",
    "1.994GB",
    "1.994GB",
    "1.994GB",
    "1.994GB",
    "1.994GB",
    "1.994GB",
    "1.994GB",
    "1.994GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "ce58f1fc29df",
    "ce58f1fc29df",
    "ce58f1fc29df",
    "ce58f1fc29df",
    "ce58f1fc29df",
    "ce58f1fc29df",
    "ce58f1fc29df",
    "ce58f1fc29df",
    "ce58f1fc29df",
    "ce58f1fc29df",
    "ce58f1fc29df"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/prefill/24576/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    6608.601,
    6605.232,
    6617.866,
    6598.807,
    6597.231,
    6573.311,
    6610.911,
    6611.656,
    6618.337,
    6612.815,
    6624.013
   ],
   "min": 6573.311,
   "max": 6624.013,
   "median": 6610.911,
   "spread_pct": 0.77,
   "status": "ok",
   "runs_as_reported": [
    6608.601,
    6605.232,
    6617.866,
    6598.807,
    6597.231,
    6573.311,
    6610.911,
    6611.656,
    6618.337,
    6612.815,
    6624.013
   ],
   "min_as_reported": 6573.311,
   "max_as_reported": 6624.013,
   "median_as_reported": 6610.911,
   "spread_pct_as_reported": 0.77,
   "rate_definition": "judged = mlx_lm prompt_tps = N / prompt_time (prompt pass through token 0), as reported",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "decode",
   "N": 24576,
   "engine": "mlx_bf16",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.16,
    2.21,
    1.93,
    1.94,
    1.82,
    2.07,
    2.37,
    2.24,
    1.88,
    1.97,
    2.29
   ],
   "load_after": [
    2.21,
    2.02,
    1.94,
    1.8,
    1.99,
    2.15,
    2.26,
    2.06,
    1.97,
    2.05,
    2.4
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "as-shipped",
   "precision": "bf16 (HF base snapshot, unquantised)",
   "input_tokens": 24576,
   "peak_memory": [
    "1.994GB",
    "1.994GB",
    "1.994GB",
    "1.994GB",
    "1.994GB",
    "1.994GB",
    "1.994GB",
    "1.994GB",
    "1.994GB",
    "1.994GB",
    "1.994GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "ce58f1fc29df",
    "ce58f1fc29df",
    "ce58f1fc29df",
    "ce58f1fc29df",
    "ce58f1fc29df",
    "ce58f1fc29df",
    "ce58f1fc29df",
    "ce58f1fc29df",
    "ce58f1fc29df",
    "ce58f1fc29df",
    "ce58f1fc29df"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/decode/24576/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    155.8161015625,
    156.100859375,
    155.81213281249998,
    156.44415625,
    156.1385625,
    155.8855546875,
    155.882578125,
    156.4907890625,
    156.09490625,
    155.8538046875,
    158.2052890625
   ],
   "min": 155.81213281249998,
   "max": 158.2052890625,
   "median": 156.09490625,
   "spread_pct": 1.54,
   "status": "ok",
   "runs_as_reported": [
    157.043,
    157.33,
    157.039,
    157.676,
    157.368,
    157.113,
    157.11,
    157.723,
    157.324,
    157.081,
    159.451
   ],
   "min_as_reported": 157.039,
   "max_as_reported": 159.451,
   "median_as_reported": 157.324,
   "spread_pct_as_reported": 1.54,
   "rate_definition": "judged = generation_tps x (gen_tok-1)/gen_tok = 127 tokens / mlx_lm generation window; as_reported = mlx_lm generation_tps = 128/window",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "prefill",
   "N": 24576,
   "engine": "mlx",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.29,
    2.02,
    2.02,
    1.8,
    2.06,
    2.15,
    2.43,
    2.29,
    1.85,
    1.97,
    2.24
   ],
   "load_after": [
    2.16,
    2.11,
    1.93,
    2.06,
    1.89,
    2.49,
    2.4,
    2.11,
    1.88,
    1.81,
    2.29
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "matched",
   "precision": "q4 g64 affine (mlx-community-style 4-bit g64, HF-converted)",
   "input_tokens": 24576,
   "peak_memory": [
    "1.509GB",
    "1.509GB",
    "1.509GB",
    "1.509GB",
    "1.509GB",
    "1.509GB",
    "1.509GB",
    "1.509GB",
    "1.509GB",
    "1.509GB",
    "1.509GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "e08dc5c5ed45",
    "e08dc5c5ed45",
    "e08dc5c5ed45",
    "e08dc5c5ed45",
    "e08dc5c5ed45",
    "e08dc5c5ed45",
    "e08dc5c5ed45",
    "e08dc5c5ed45",
    "e08dc5c5ed45",
    "e08dc5c5ed45",
    "e08dc5c5ed45"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/prefill/24576/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    6211.984,
    6173.243,
    6215.448,
    6226.231,
    6235.94,
    6225.009,
    6236.258,
    6204.445,
    6227.205,
    6240.461,
    6230.742
   ],
   "min": 6173.243,
   "max": 6240.461,
   "median": 6226.231,
   "spread_pct": 1.09,
   "status": "ok",
   "runs_as_reported": [
    6211.984,
    6173.243,
    6215.448,
    6226.231,
    6235.94,
    6225.009,
    6236.258,
    6204.445,
    6227.205,
    6240.461,
    6230.742
   ],
   "min_as_reported": 6173.243,
   "max_as_reported": 6240.461,
   "median_as_reported": 6226.231,
   "spread_pct_as_reported": 1.09,
   "rate_definition": "judged = mlx_lm prompt_tps = N / prompt_time (prompt pass through token 0), as reported",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "decode",
   "N": 24576,
   "engine": "mlx",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.29,
    2.02,
    2.02,
    1.8,
    2.06,
    2.15,
    2.43,
    2.29,
    1.85,
    1.97,
    2.24
   ],
   "load_after": [
    2.16,
    2.11,
    1.93,
    2.06,
    1.89,
    2.49,
    2.4,
    2.11,
    1.88,
    1.81,
    2.29
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "matched",
   "precision": "q4 g64 affine (mlx-community-style 4-bit g64, HF-converted)",
   "input_tokens": 24576,
   "peak_memory": [
    "1.509GB",
    "1.509GB",
    "1.509GB",
    "1.509GB",
    "1.509GB",
    "1.509GB",
    "1.509GB",
    "1.509GB",
    "1.509GB",
    "1.509GB",
    "1.509GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "e08dc5c5ed45",
    "e08dc5c5ed45",
    "e08dc5c5ed45",
    "e08dc5c5ed45",
    "e08dc5c5ed45",
    "e08dc5c5ed45",
    "e08dc5c5ed45",
    "e08dc5c5ed45",
    "e08dc5c5ed45",
    "e08dc5c5ed45",
    "e08dc5c5ed45"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/decode/24576/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    155.2426171875,
    155.5124921875,
    155.3477890625,
    155.2287265625,
    155.311078125,
    155.838921875,
    155.24559374999998,
    155.0560859375,
    155.4291484375,
    155.1374453125,
    155.1384375
   ],
   "min": 155.0560859375,
   "max": 155.838921875,
   "median": 155.24559374999998,
   "spread_pct": 0.5,
   "status": "ok",
   "runs_as_reported": [
    156.465,
    156.737,
    156.571,
    156.451,
    156.534,
    157.066,
    156.468,
    156.277,
    156.653,
    156.359,
    156.36
   ],
   "min_as_reported": 156.277,
   "max_as_reported": 157.066,
   "median_as_reported": 156.468,
   "spread_pct_as_reported": 0.5,
   "rate_definition": "judged = generation_tps x (gen_tok-1)/gen_tok = 127 tokens / mlx_lm generation window; as_reported = mlx_lm generation_tps = 128/window",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "prefill",
   "N": 24576,
   "engine": "veizik",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    1.62,
    2.11,
    2.02,
    2.06,
    2.06,
    2.49,
    2.64,
    2.11,
    2.01,
    2.47,
    2.35
   ],
   "load_after": [
    2.29,
    2.02,
    2.02,
    1.9,
    2.06,
    2.61,
    2.43,
    2.1,
    1.85,
    2.47,
    2.24
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "row_id": "m1-ultra/qwen2.5-0.5b/veizik/prefill/24576/kit_r5_pd1",
   "unit": "tok/s (core self-clock, CLI \"prefill : <ms> ms for <N> positions\" line)",
   "measured_binary_sha256": "8e9e22a82073be4814699b6271be95f1045e1eeb8a8ba59bd912692289d7d659",
   "measured_binary_digest_kind": "sha256 of __TEXT,__text section",
   "core_file_sha256": "c7678f0931a848f593dedd5b07c8d505f5d44f22b6fc1a7e713fa15232f23896",
   "measured_cli_sha256": "a9eb4d9b8018432f27ce95abd0e0dc7bc63fa23a4695592d9a71858819f4713a",
   "cli_digest_kind": "sha256 of the installed CLI binary (file)",
   "snapshot": "060db6499f32faf8b98477b0a26969ef7d8b9987",
   "precision_mode": "as-shipped",
   "precision": "veizik shipped low-bit core (installed release)",
   "command": "veizik run qwen2.5-0.5b --tokens 128 --ignore-eos \"<p24576 prompt file>\"  (customer CLI, no VZ_* env)",
   "input_tokens": 24576,
   "checkpoint": "Qwen base HF snapshot 060db649",
   "runs": [
    6779.771026068699,
    6783.283842512512,
    6778.2681125723375,
    6731.698160040203,
    6776.601881954161,
    6782.871540360138,
    6764.67163226891,
    6771.627685523924,
    6778.719178712747,
    6625.590402954303,
    6769.812619411098
   ],
   "min": 6625.590402954303,
   "max": 6783.283842512512,
   "median": 6776.601881954161,
   "spread_pct": 2.38,
   "status": "ok",
   "runs_as_reported": [
    6788.95,
    6792.46,
    6787.41,
    6741.57,
    6785.74,
    6792.0,
    6773.8,
    6780.77,
    6787.94,
    6634.46,
    6778.99
   ],
   "min_as_reported": 6634.46,
   "max_as_reported": 6792.46,
   "median_as_reported": 6785.74,
   "spread_pct_as_reported": 2.38,
   "rate_definition": "judged = N / (prefill_ms + dec_ms/dec_tok) for qwen2.5 cores (their prefill clock stops before the last prompt step, which produces token 0, has completed; one decode step is added as an upper bound for it); as reported (N / prefill_ms) for qwen3-1.7b, whose clock stops at the completion of that step, the same span as mlx_lm prompt_time",
   "value_bound": "lower",
   "value_bound_note": "prefill: one whole decode step is added for the last prompt step, an upper bound on what the prefill clock leaves out; follow-up: have the qwen2.5 drivers end the prefill clock at the completion of the last prompt step so this becomes exact",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "decode",
   "N": 24576,
   "engine": "veizik",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    1.62,
    2.11,
    2.02,
    2.06,
    2.06,
    2.49,
    2.64,
    2.11,
    2.01,
    2.47,
    2.35
   ],
   "load_after": [
    2.29,
    2.02,
    2.02,
    1.9,
    2.06,
    2.61,
    2.43,
    2.1,
    1.85,
    2.47,
    2.24
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "row_id": "m1-ultra/qwen2.5-0.5b/veizik/decode/24576/kit_r5_pd1",
   "unit": "tok/s (core self-clock, CLI \"generated\" line: 127 decoded tokens after the prompt-pass token, @ KV depth N; --ignore-eos)",
   "measured_binary_sha256": "8e9e22a82073be4814699b6271be95f1045e1eeb8a8ba59bd912692289d7d659",
   "measured_binary_digest_kind": "sha256 of __TEXT,__text section",
   "core_file_sha256": "c7678f0931a848f593dedd5b07c8d505f5d44f22b6fc1a7e713fa15232f23896",
   "measured_cli_sha256": "a9eb4d9b8018432f27ce95abd0e0dc7bc63fa23a4695592d9a71858819f4713a",
   "cli_digest_kind": "sha256 of the installed CLI binary (file)",
   "snapshot": "060db6499f32faf8b98477b0a26969ef7d8b9987",
   "precision_mode": "as-shipped",
   "precision": "veizik shipped low-bit core (installed release)",
   "command": "veizik run qwen2.5-0.5b --tokens 128 --ignore-eos \"<p24576 prompt file>\"  (customer CLI, no VZ_* env)",
   "input_tokens": 24576,
   "checkpoint": "Qwen base HF snapshot 060db649",
   "runs": [
    204.03,
    204.33,
    204.71,
    187.07,
    204.73,
    205.25,
    204.28,
    204.44,
    203.1,
    201.56,
    203.4
   ],
   "min": 187.07,
   "max": 205.25,
   "median": 204.28,
   "spread_pct": 9.72,
   "status": "ok",
   "runs_as_reported": [
    204.03,
    204.33,
    204.71,
    187.07,
    204.73,
    205.25,
    204.28,
    204.44,
    203.1,
    201.56,
    203.4
   ],
   "min_as_reported": 187.07,
   "max_as_reported": 205.25,
   "median_as_reported": 204.28,
   "spread_pct_as_reported": 9.72,
   "rate_definition": "judged = 127 decoded tokens / their time (CLI \"+ 127 decoded in X ms\"); as reported",
   "value_bound": "lower",
   "value_bound_note": "decode: the decode window also contains the completion of the last prompt step (token 0), so 127/window slightly understates the decode rate; follow-up: have the qwen2.5 drivers end the prefill clock at the completion of the last prompt step so this becomes exact",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "prefill",
   "N": 30000,
   "engine": "mlx_bf16",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    1.91,
    1.77,
    2.21,
    2.02,
    2.21,
    2.57,
    2.26,
    2.14,
    1.93,
    2.65,
    2.23
   ],
   "load_after": [
    1.77,
    1.73,
    2.02,
    2.1,
    2.57,
    2.42,
    2.24,
    2.2,
    2.79,
    2.6,
    2.04
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "as-shipped",
   "precision": "bf16 (HF base snapshot, unquantised)",
   "input_tokens": 30000,
   "peak_memory": [
    "2.044GB",
    "2.044GB",
    "2.044GB",
    "2.044GB",
    "2.044GB",
    "2.044GB",
    "2.044GB",
    "2.044GB",
    "2.044GB",
    "2.044GB",
    "2.044GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "2a7c2c01ecf8",
    "2a7c2c01ecf8",
    "2a7c2c01ecf8",
    "2a7c2c01ecf8",
    "2a7c2c01ecf8",
    "2a7c2c01ecf8",
    "2a7c2c01ecf8",
    "2a7c2c01ecf8",
    "2a7c2c01ecf8",
    "2a7c2c01ecf8",
    "2a7c2c01ecf8"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/prefill/30000/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    5902.676,
    5892.602,
    5899.547,
    5905.164,
    5893.367,
    5896.535,
    5896.89,
    5898.124,
    5910.2,
    5893.797,
    5896.403
   ],
   "min": 5892.602,
   "max": 5910.2,
   "median": 5896.89,
   "spread_pct": 0.3,
   "status": "ok",
   "runs_as_reported": [
    5902.676,
    5892.602,
    5899.547,
    5905.164,
    5893.367,
    5896.535,
    5896.89,
    5898.124,
    5910.2,
    5893.797,
    5896.403
   ],
   "min_as_reported": 5892.602,
   "max_as_reported": 5910.2,
   "median_as_reported": 5896.89,
   "spread_pct_as_reported": 0.3,
   "rate_definition": "judged = mlx_lm prompt_tps = N / prompt_time (prompt pass through token 0), as reported",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "decode",
   "N": 30000,
   "engine": "mlx_bf16",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    1.91,
    1.77,
    2.21,
    2.02,
    2.21,
    2.57,
    2.26,
    2.14,
    1.93,
    2.65,
    2.23
   ],
   "load_after": [
    1.77,
    1.73,
    2.02,
    2.1,
    2.57,
    2.42,
    2.24,
    2.2,
    2.79,
    2.6,
    2.04
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "as-shipped",
   "precision": "bf16 (HF base snapshot, unquantised)",
   "input_tokens": 30000,
   "peak_memory": [
    "2.044GB",
    "2.044GB",
    "2.044GB",
    "2.044GB",
    "2.044GB",
    "2.044GB",
    "2.044GB",
    "2.044GB",
    "2.044GB",
    "2.044GB",
    "2.044GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "2a7c2c01ecf8",
    "2a7c2c01ecf8",
    "2a7c2c01ecf8",
    "2a7c2c01ecf8",
    "2a7c2c01ecf8",
    "2a7c2c01ecf8",
    "2a7c2c01ecf8",
    "2a7c2c01ecf8",
    "2a7c2c01ecf8",
    "2a7c2c01ecf8",
    "2a7c2c01ecf8"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/decode/30000/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    145.7037265625,
    145.35150000000002,
    145.4497265625,
    145.202671875,
    145.738453125,
    145.85553124999998,
    145.3187578125,
    145.6055,
    145.60649218749998,
    145.3842421875,
    145.73646875
   ],
   "min": 145.202671875,
   "max": 145.85553124999998,
   "median": 145.6055,
   "spread_pct": 0.45,
   "status": "ok",
   "runs_as_reported": [
    146.851,
    146.496,
    146.595,
    146.346,
    146.886,
    147.004,
    146.463,
    146.752,
    146.753,
    146.529,
    146.884
   ],
   "min_as_reported": 146.346,
   "max_as_reported": 147.004,
   "median_as_reported": 146.752,
   "spread_pct_as_reported": 0.45,
   "rate_definition": "judged = generation_tps x (gen_tok-1)/gen_tok = 127 tokens / mlx_lm generation window; as_reported = mlx_lm generation_tps = 128/window",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "prefill",
   "N": 30000,
   "engine": "mlx",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.16,
    1.73,
    1.97,
    2.1,
    2.43,
    2.42,
    2.33,
    2.2,
    2.02,
    2.47,
    2.46
   ],
   "load_after": [
    1.91,
    1.69,
    2.21,
    2.25,
    2.31,
    2.51,
    2.29,
    2.01,
    1.93,
    2.49,
    2.23
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "matched",
   "precision": "q4 g64 affine (mlx-community-style 4-bit g64, HF-converted)",
   "input_tokens": 30000,
   "peak_memory": [
    "1.563GB",
    "1.563GB",
    "1.563GB",
    "1.563GB",
    "1.563GB",
    "1.563GB",
    "1.563GB",
    "1.563GB",
    "1.563GB",
    "1.563GB",
    "1.563GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "ec5fe23db4c4",
    "ec5fe23db4c4",
    "ec5fe23db4c4",
    "ec5fe23db4c4",
    "ec5fe23db4c4",
    "ec5fe23db4c4",
    "ec5fe23db4c4",
    "ec5fe23db4c4",
    "ec5fe23db4c4",
    "ec5fe23db4c4",
    "ec5fe23db4c4"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/prefill/30000/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    5590.119,
    5593.764,
    5585.263,
    5593.773,
    5596.697,
    5596.881,
    5591.23,
    5587.725,
    5599.723,
    5585.297,
    5575.457
   ],
   "min": 5575.457,
   "max": 5599.723,
   "median": 5591.23,
   "spread_pct": 0.44,
   "status": "ok",
   "runs_as_reported": [
    5590.119,
    5593.764,
    5585.263,
    5593.773,
    5596.697,
    5596.881,
    5591.23,
    5587.725,
    5599.723,
    5585.297,
    5575.457
   ],
   "min_as_reported": 5575.457,
   "max_as_reported": 5599.723,
   "median_as_reported": 5591.23,
   "spread_pct_as_reported": 0.44,
   "rate_definition": "judged = mlx_lm prompt_tps = N / prompt_time (prompt pass through token 0), as reported",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "decode",
   "N": 30000,
   "engine": "mlx",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.16,
    1.73,
    1.97,
    2.1,
    2.43,
    2.42,
    2.33,
    2.2,
    2.02,
    2.47,
    2.46
   ],
   "load_after": [
    1.91,
    1.69,
    2.21,
    2.25,
    2.31,
    2.51,
    2.29,
    2.01,
    1.93,
    2.49,
    2.23
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "unit": "tok/s (mlx_lm generate self-clock, EOS masked, 128 tokens)",
   "precision_mode": "matched",
   "precision": "q4 g64 affine (mlx-community-style 4-bit g64, HF-converted)",
   "input_tokens": 30000,
   "peak_memory": [
    "1.563GB",
    "1.563GB",
    "1.563GB",
    "1.563GB",
    "1.563GB",
    "1.563GB",
    "1.563GB",
    "1.563GB",
    "1.563GB",
    "1.563GB",
    "1.563GB"
   ],
   "mlx_dtype": null,
   "out_sha": [
    "ec5fe23db4c4",
    "ec5fe23db4c4",
    "ec5fe23db4c4",
    "ec5fe23db4c4",
    "ec5fe23db4c4",
    "ec5fe23db4c4",
    "ec5fe23db4c4",
    "ec5fe23db4c4",
    "ec5fe23db4c4",
    "ec5fe23db4c4",
    "ec5fe23db4c4"
   ],
   "engine_version": "mlx-lm/mlx 0.31.3 0.32.2",
   "paired_veizik_rows": "m1-ultra/qwen2.5-0.5b/veizik/decode/30000/kit_r5_pd1",
   "paired_note": "same interleaved round (n=11, order alternating), same prompt file",
   "runs": [
    143.7372109375,
    144.079515625,
    144.1380546875,
    143.8513125,
    143.6776796875,
    144.5805703125,
    143.64890625,
    143.716375,
    143.821546875,
    143.9931953125,
    143.66478125
   ],
   "min": 143.64890625,
   "max": 144.5805703125,
   "median": 143.821546875,
   "spread_pct": 0.65,
   "status": "ok",
   "runs_as_reported": [
    144.869,
    145.214,
    145.273,
    144.984,
    144.809,
    145.719,
    144.78,
    144.848,
    144.954,
    145.127,
    144.796
   ],
   "min_as_reported": 144.78,
   "max_as_reported": 145.719,
   "median_as_reported": 144.954,
   "spread_pct_as_reported": 0.65,
   "rate_definition": "judged = generation_tps x (gen_tok-1)/gen_tok = 127 tokens / mlx_lm generation window; as_reported = mlx_lm generation_tps = 128/window",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "prefill",
   "N": 30000,
   "engine": "veizik",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.27,
    1.95,
    1.88,
    2.25,
    2.42,
    2.51,
    2.36,
    2.01,
    1.93,
    2.49,
    2.41
   ],
   "load_after": [
    2.16,
    1.88,
    1.97,
    2.42,
    2.55,
    2.36,
    2.33,
    1.93,
    2.02,
    2.53,
    2.46
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "row_id": "m1-ultra/qwen2.5-0.5b/veizik/prefill/30000/kit_r5_pd1",
   "unit": "tok/s (core self-clock, CLI \"prefill : <ms> ms for <N> positions\" line)",
   "measured_binary_sha256": "8e9e22a82073be4814699b6271be95f1045e1eeb8a8ba59bd912692289d7d659",
   "measured_binary_digest_kind": "sha256 of __TEXT,__text section",
   "core_file_sha256": "c7678f0931a848f593dedd5b07c8d505f5d44f22b6fc1a7e713fa15232f23896",
   "measured_cli_sha256": "a9eb4d9b8018432f27ce95abd0e0dc7bc63fa23a4695592d9a71858819f4713a",
   "cli_digest_kind": "sha256 of the installed CLI binary (file)",
   "snapshot": "060db6499f32faf8b98477b0a26969ef7d8b9987",
   "precision_mode": "as-shipped",
   "precision": "veizik shipped low-bit core (installed release)",
   "command": "veizik run qwen2.5-0.5b --tokens 128 --ignore-eos \"<p30000 prompt file>\"  (customer CLI, no VZ_* env)",
   "input_tokens": 30000,
   "checkpoint": "Qwen base HF snapshot 060db649",
   "runs": [
    5886.24715215329,
    5892.516262614121,
    5878.41031307071,
    5892.6858025776355,
    5876.624914938167,
    5892.591402892881,
    5876.1374991498105,
    5887.368175336543,
    5877.893755169702,
    5891.063447398098,
    5871.029703642672
   ],
   "min": 5871.029703642672,
   "max": 5892.6858025776355,
   "median": 5886.24715215329,
   "spread_pct": 0.37,
   "status": "ok",
   "runs_as_reported": [
    5892.06,
    5898.32,
    5884.17,
    5898.46,
    5882.41,
    5898.4,
    5881.91,
    5893.18,
    5883.66,
    5896.88,
    5876.8
   ],
   "min_as_reported": 5876.8,
   "max_as_reported": 5898.46,
   "median_as_reported": 5892.06,
   "spread_pct_as_reported": 0.37,
   "rate_definition": "judged = N / (prefill_ms + dec_ms/dec_tok) for qwen2.5 cores (their prefill clock stops before the last prompt step, which produces token 0, has completed; one decode step is added as an upper bound for it); as reported (N / prefill_ms) for qwen3-1.7b, whose clock stops at the completion of that step, the same span as mlx_lm prompt_time",
   "value_bound": "lower",
   "value_bound_note": "prefill: one whole decode step is added for the last prompt step, an upper bound on what the prefill clock leaves out; follow-up: have the qwen2.5 drivers end the prefill clock at the completion of the last prompt step so this becomes exact",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  },
  {
   "device_id": "m1-ultra",
   "host": "MacStudio_M1_Ultra",
   "model": "qwen2.5-0.5b",
   "model_id": "Qwen/Qwen2.5-0.5B",
   "axis": "decode",
   "N": 30000,
   "engine": "veizik",
   "session": "kit_r5_pd1",
   "measured_from": "v2026.10.04_r5c_kit_pd1/M1Ultra_r5_bench.txt",
   "load_before": [
    2.27,
    1.95,
    1.88,
    2.25,
    2.42,
    2.51,
    2.36,
    2.01,
    1.93,
    2.49,
    2.41
   ],
   "load_after": [
    2.16,
    1.88,
    1.97,
    2.42,
    2.55,
    2.36,
    2.33,
    1.93,
    2.02,
    2.53,
    2.46
   ],
   "power": [
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power",
    "AC_Power"
   ],
   "therm": [
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0",
    "0"
   ],
   "measurement_runs": 11,
   "row_id": "m1-ultra/qwen2.5-0.5b/veizik/decode/30000/kit_r5_pd1",
   "unit": "tok/s (core self-clock, CLI \"generated\" line: 127 decoded tokens after the prompt-pass token, @ KV depth N; --ignore-eos)",
   "measured_binary_sha256": "8e9e22a82073be4814699b6271be95f1045e1eeb8a8ba59bd912692289d7d659",
   "measured_binary_digest_kind": "sha256 of __TEXT,__text section",
   "core_file_sha256": "c7678f0931a848f593dedd5b07c8d505f5d44f22b6fc1a7e713fa15232f23896",
   "measured_cli_sha256": "a9eb4d9b8018432f27ce95abd0e0dc7bc63fa23a4695592d9a71858819f4713a",
   "cli_digest_kind": "sha256 of the installed CLI binary (file)",
   "snapshot": "060db6499f32faf8b98477b0a26969ef7d8b9987",
   "precision_mode": "as-shipped",
   "precision": "veizik shipped low-bit core (installed release)",
   "command": "veizik run qwen2.5-0.5b --tokens 128 --ignore-eos \"<p30000 prompt file>\"  (customer CLI, no VZ_* env)",
   "input_tokens": 30000,
   "checkpoint": "Qwen base HF snapshot 060db649",
   "runs": [
    198.93,
    199.66,
    200.22,
    200.68,
    199.24,
    199.57,
    199.74,
    198.95,
    200.08,
    199.04,
    199.35
   ],
   "min": 198.93,
   "max": 200.68,
   "median": 199.57,
   "spread_pct": 0.88,
   "status": "ok",
   "runs_as_reported": [
    198.93,
    199.66,
    200.22,
    200.68,
    199.24,
    199.57,
    199.74,
    198.95,
    200.08,
    199.04,
    199.35
   ],
   "min_as_reported": 198.93,
   "max_as_reported": 200.68,
   "median_as_reported": 199.57,
   "spread_pct_as_reported": 0.88,
   "rate_definition": "judged = 127 decoded tokens / their time (CLI \"+ 127 decoded in X ms\"); as reported",
   "value_bound": "lower",
   "value_bound_note": "decode: the decode window also contains the completion of the last prompt step (token 0), so 127/window slightly understates the decode rate; follow-up: have the qwen2.5 drivers end the prefill clock at the completion of the last prompt step so this becomes exact",
   "quiet_gate": {
    "gpu_idle_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "background_proc_cpu_pct": [
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0,
     0.0
    ],
    "threshold": {
     "gpu_pct": [
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0,
      5.0
     ],
     "background_proc_cpu_pct": [
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0,
      10.0
     ],
     "idle_floor_pct": 0.0,
     "margin_pct": 5.0,
     "idle_floor_samples": "50,0,0,0,0,0,0,0,0,0",
     "rule": "gpu threshold = this device idle floor (median of 10 ioreg samples at start) + margin",
     "source": "qb_gpu/qb_dcpu columns in the raw rows; floor from the # idle_floor header"
    },
    "basis": "kit bench.sh quiet gate: median of 5 ioreg Device Utilization % samples before the run and summed CPU% of the background media-analysis and screen-sharing processes; the run waits until both are at or under threshold; the gate waits, never stops a process"
   },
   "protocol_id": "apple-compare-v2",
   "n_runs": 11,
   "kit_sha256": "1eba32f9dd33d45a95260e15471fb525690516ac386046a05a6cefc155ae0ead"
  }
 ]
}