diff --git a/benchmarks/cuda_gb10_2026-07-12_postreboot_single_falcon-h1-tiny-90m-instruct-4bit.csv b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_falcon-h1-tiny-90m-instruct-4bit.csv new file mode 100644 index 000000000..acd79c492 --- /dev/null +++ b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_falcon-h1-tiny-90m-instruct-4bit.csv @@ -0,0 +1,2 @@ +model,model_path,prompt_tokens,generated_tokens,prefill_ms,prefill_tok_s,decode_ms,decode_tok_s,date,hardware,mlx_version,build_type,max_tokens,prompt,prompt_target_len +falcon-h1-tiny-90m-instruct-4bit,models/falcon-h1-tiny-90m-instruct-4bit,17,30,14.24,1193.95,84.64,354.43,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", diff --git a/benchmarks/cuda_gb10_2026-07-12_postreboot_single_gemma-4-26b-a4b-it-4bit.csv b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_gemma-4-26b-a4b-it-4bit.csv new file mode 100644 index 000000000..11644ad8e --- /dev/null +++ b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_gemma-4-26b-a4b-it-4bit.csv @@ -0,0 +1,4 @@ +model,model_path,prompt_tokens,generated_tokens,prefill_ms,prefill_tok_s,decode_ms,decode_tok_s,date,hardware,mlx_version,build_type,max_tokens,prompt,prompt_target_len +gemma-4-26b-a4b-it-4bit,models/gemma-4-26b-a4b-it-4bit,20,26,148.21,134.94,432.81,60.07,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", +gemma-4-26b-a4b-it-4bit,models/gemma-4-26b-a4b-it-4bit,20,26,148.87,134.35,436.46,59.57,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", +gemma-4-26b-a4b-it-4bit,models/gemma-4-26b-a4b-it-4bit,20,26,151.83,131.73,434.17,59.88,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", diff --git a/benchmarks/cuda_gb10_2026-07-12_postreboot_single_gemma-4-26b-a4b-it-qat-4bit.csv b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_gemma-4-26b-a4b-it-qat-4bit.csv new file mode 100644 index 000000000..1eec02f91 --- /dev/null +++ b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_gemma-4-26b-a4b-it-qat-4bit.csv @@ -0,0 +1,4 @@ +model,model_path,prompt_tokens,generated_tokens,prefill_ms,prefill_tok_s,decode_ms,decode_tok_s,date,hardware,mlx_version,build_type,max_tokens,prompt,prompt_target_len +gemma-4-26b-a4b-it-qat-4bit,models/gemma-4-26b-a4b-it-qat-4bit,20,26,150.25,133.11,484.35,53.68,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", +gemma-4-26b-a4b-it-qat-4bit,models/gemma-4-26b-a4b-it-qat-4bit,20,26,149.90,133.43,486.17,53.48,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", +gemma-4-26b-a4b-it-qat-4bit,models/gemma-4-26b-a4b-it-qat-4bit,20,26,149.73,133.57,484.07,53.71,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", diff --git a/benchmarks/cuda_gb10_2026-07-12_postreboot_single_glm4-flash-4bit.csv b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_glm4-flash-4bit.csv new file mode 100644 index 000000000..5d63dc806 --- /dev/null +++ b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_glm4-flash-4bit.csv @@ -0,0 +1,4 @@ +model,model_path,prompt_tokens,generated_tokens,prefill_ms,prefill_tok_s,decode_ms,decode_tok_s,date,hardware,mlx_version,build_type,max_tokens,prompt,prompt_target_len +glm4-flash-4bit,models/glm4-flash-4bit,12,18,120.71,99.41,379.61,47.42,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", +glm4-flash-4bit,models/glm4-flash-4bit,12,18,113.79,105.46,359.52,50.07,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", +glm4-flash-4bit,models/glm4-flash-4bit,12,18,113.37,105.85,387.74,46.42,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", diff --git a/benchmarks/cuda_gb10_2026-07-12_postreboot_single_granite-4.0-h-350m-4bit.csv b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_granite-4.0-h-350m-4bit.csv new file mode 100644 index 000000000..ec9e81d8e --- /dev/null +++ b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_granite-4.0-h-350m-4bit.csv @@ -0,0 +1,2 @@ +model,model_path,prompt_tokens,generated_tokens,prefill_ms,prefill_tok_s,decode_ms,decode_tok_s,date,hardware,mlx_version,build_type,max_tokens,prompt,prompt_target_len +granite-4.0-h-350m-4bit,models/granite-4.0-h-350m-4bit,37,19,19.92,1857.41,110.97,171.22,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", diff --git a/benchmarks/cuda_gb10_2026-07-12_postreboot_single_granite-4.0-h-tiny-4bit.csv b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_granite-4.0-h-tiny-4bit.csv new file mode 100644 index 000000000..d0ea8a542 --- /dev/null +++ b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_granite-4.0-h-tiny-4bit.csv @@ -0,0 +1,2 @@ +model,model_path,prompt_tokens,generated_tokens,prefill_ms,prefill_tok_s,decode_ms,decode_tok_s,date,hardware,mlx_version,build_type,max_tokens,prompt,prompt_target_len +granite-4.0-h-tiny-4bit,models/granite-4.0-h-tiny-4bit,15,42,62.89,238.51,414.33,101.37,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", diff --git a/benchmarks/cuda_gb10_2026-07-12_postreboot_single_lfm2-8b-a1b-4bit.csv b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_lfm2-8b-a1b-4bit.csv new file mode 100644 index 000000000..551d8ee2a --- /dev/null +++ b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_lfm2-8b-a1b-4bit.csv @@ -0,0 +1,2 @@ +model,model_path,prompt_tokens,generated_tokens,prefill_ms,prefill_tok_s,decode_ms,decode_tok_s,date,hardware,mlx_version,build_type,max_tokens,prompt,prompt_target_len +lfm2-8b-a1b-4bit,models/lfm2-8b-a1b-4bit,17,37,114.22,148.84,223.53,165.53,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", diff --git a/benchmarks/cuda_gb10_2026-07-12_postreboot_single_mamba2-130m.csv b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_mamba2-130m.csv new file mode 100644 index 000000000..441599d55 --- /dev/null +++ b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_mamba2-130m.csv @@ -0,0 +1,4 @@ +model,model_path,prompt_tokens,generated_tokens,prefill_ms,prefill_tok_s,decode_ms,decode_tok_s,date,hardware,mlx_version,build_type,max_tokens,prompt,prompt_target_len +mamba2-130m,models/mamba2-130m,7,100,7.84,892.53,675.77,147.98,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", +mamba2-130m,models/mamba2-130m,7,100,7.84,892.83,492.66,202.98,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", +mamba2-130m,models/mamba2-130m,7,100,6.99,1001.24,488.35,204.77,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", diff --git a/benchmarks/cuda_gb10_2026-07-12_postreboot_single_nemotron-3-nano-omni-30b-a3b-reasoning-4bit.csv b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_nemotron-3-nano-omni-30b-a3b-reasoning-4bit.csv new file mode 100644 index 000000000..79d7dbff6 --- /dev/null +++ b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_nemotron-3-nano-omni-30b-a3b-reasoning-4bit.csv @@ -0,0 +1,2 @@ +model,model_path,prompt_tokens,generated_tokens,prefill_ms,prefill_tok_s,decode_ms,decode_tok_s,date,hardware,mlx_version,build_type,max_tokens,prompt,prompt_target_len +nemotron-3-nano-omni-30b-a3b-reasoning-4bit,models/nemotron-3-nano-omni-30b-a3b-reasoning-4bit,23,20,190.12,120.97,241.32,82.88,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", diff --git a/benchmarks/cuda_gb10_2026-07-12_postreboot_single_nemotron-h-30b-4bit.csv b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_nemotron-h-30b-4bit.csv new file mode 100644 index 000000000..38675f7ab --- /dev/null +++ b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_nemotron-h-30b-4bit.csv @@ -0,0 +1,2 @@ +model,model_path,prompt_tokens,generated_tokens,prefill_ms,prefill_tok_s,decode_ms,decode_tok_s,date,hardware,mlx_version,build_type,max_tokens,prompt,prompt_target_len +nemotron-h-30b-4bit,models/nemotron-h-30b-4bit,23,46,203.10,113.24,526.27,87.41,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", diff --git a/benchmarks/cuda_gb10_2026-07-12_postreboot_single_nemotron-nas-30b-4bit.csv b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_nemotron-nas-30b-4bit.csv new file mode 100644 index 000000000..0f7d03a39 --- /dev/null +++ b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_nemotron-nas-30b-4bit.csv @@ -0,0 +1,2 @@ +model,model_path,prompt_tokens,generated_tokens,prefill_ms,prefill_tok_s,decode_ms,decode_tok_s,date,hardware,mlx_version,build_type,max_tokens,prompt,prompt_target_len +nemotron-nas-30b-4bit,models/nemotron-nas-30b-4bit,23,46,194.94,117.99,535.47,85.91,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", diff --git a/benchmarks/cuda_gb10_2026-07-12_postreboot_single_plamo-2-1b.csv b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_plamo-2-1b.csv new file mode 100644 index 000000000..a1cd35ef1 --- /dev/null +++ b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_plamo-2-1b.csv @@ -0,0 +1,2 @@ +model,model_path,prompt_tokens,generated_tokens,prefill_ms,prefill_tok_s,decode_ms,decode_tok_s,date,hardware,mlx_version,build_type,max_tokens,prompt,prompt_target_len +plamo-2-1b,models/plamo-2-1b,7,100,35.28,198.43,2134.72,46.84,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", diff --git a/benchmarks/cuda_gb10_2026-07-12_postreboot_single_qwen3-30b-a3b-4bit.csv b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_qwen3-30b-a3b-4bit.csv new file mode 100644 index 000000000..d8b411dc9 --- /dev/null +++ b/benchmarks/cuda_gb10_2026-07-12_postreboot_single_qwen3-30b-a3b-4bit.csv @@ -0,0 +1,2 @@ +model,model_path,prompt_tokens,generated_tokens,prefill_ms,prefill_tok_s,decode_ms,decode_tok_s,date,hardware,mlx_version,build_type,max_tokens,prompt,prompt_target_len +qwen3-30b-a3b-4bit,models/qwen3-30b-a3b-4bit,19,34,141.41,134.37,360.23,94.39,2026-07-12,NVIDIA_GB10_122GB,0.4.0-rc.1,release,100,"Hello, how are you today?", diff --git a/docs/benchmark_results/model_tests.md b/docs/benchmark_results/model_tests.md index c917407ae..58ea022df 100644 --- a/docs/benchmark_results/model_tests.md +++ b/docs/benchmark_results/model_tests.md @@ -71,7 +71,7 @@ The table below summarizes the current cross-hardware decode readings for select | Gemma-3-1B | 1B | 227.38 | 395.02 | 278.52 | | EXAONE-3.5-2.4B | 2.4B | 199.11 | 287.66 | 141.83 | | SmolLM3-3B | 3B | 131.45 | 234.21 | 104.24 | -| Nemotron-H-30B | 30B | 91.75 | 176.95 | 79.94¶ | +| Nemotron-H-30B | 30B | 91.75 | 176.95 | 87.41¶ | | Qwen3-MoE-30B | 30B | 83.42 | 175.48 | 89.06† | | Llama-3.1-8B | 8B | 106.63 | 117.46 | 50.53 | | Qwen2.5-7B | 7B | 108.47 | 126.20 | 54.56 | @@ -82,11 +82,11 @@ The table below summarizes the current cross-hardware decode readings for select *Qwen3-0.6B on GB10 again stopped at 9 tokens before EOS (2026-07-12); the 283.90 tok/s figure is from that short window and is not directly comparable to full-length runs. †Qwen3-MoE-30B (`qwen3-moe-4bit`) **failed** on GB10 at 0.3.0 (Metal-only fused-MoE kernel aborted on CUDA); the CUDA fused decode-MoE kernel (#319) restored it at 0.3.1, and at 89.06 tok/s it stays ahead of M1 Ultra (83.75). §GPT-OSS-120B and Solar-Open-100B were excluded from the 2026-07-12 GB10 sweep by the memory gate (weights > ~51 GiB, `SKIP:oom_estimate`); their figures are carried from the 2026-06-17 / 0.3.1 sweep. -¶Nemotron-H-30B doubled vs both earlier GB10 records (40.32 on 2026-06-17) with no SSM-related code change in between; the whole SSM/hybrid cluster reads 2-3x higher on 2026-07-12 and should be re-verified after a fresh boot (see the GB10 file's notable-changes list). +¶Nemotron-H-30B doubled vs the 2026-06-17 record (40.32) because the fused single-token SSM decode kernel was ported to CUDA on 2026-07-10 (#727); the post-reboot re-verification (#755) confirmed the gain on a fresh host (87.41, post-reboot single). The whole SSM/hybrid cluster carries the same attribution (see the GB10 file's notable-changes list). M1 Ultra column is from 2026-07-12 with mlxcel 0.4.0-rc.1 / MLX pin `57c66cac` / `--cooldown 30 --big-cooldown 30`, using the `mlxcel-bench-decode` same-process harness. M5 Max column is from the 2026-07-11/12 full re-sweep with mlxcel 0.4.0-rc.1 / MLX pin `57c66cac` / `--cooldown 30 --big-cooldown 30`, same-process `mlxcel-bench-decode` harness. -GB10 column is from 2026-07-12 with mlxcel 0.4.0-rc.1 / MLX pin `57c66cac` (0.32.1) / CUDA 13.0 (SM 12.1) / `--cooldown 15 --big-cooldown 45`, using the `mlxcel-bench-decode` same-process warm harness, except the two `§`-marked memory-gated rows carried from 2026-06-17 / 0.3.1. +GB10 column is from 2026-07-12 with mlxcel 0.4.0-rc.1 / MLX pin `57c66cac` (0.32.1) / CUDA 13.0 (SM 12.1) / `--cooldown 15 --big-cooldown 45`, using the `mlxcel-bench-decode` same-process warm harness, except the two `§`-marked memory-gated rows carried from 2026-06-17 / 0.3.1 and the `¶`-marked Nemotron-H row, which is the post-reboot single from the same day (#755, `--cooldown 30`). All three columns now share mlxcel 0.4.0-rc.1 and the MLX pin `57c66cac`, so the Apple Silicon gap reflects hardware delta. M5 Max stays roughly 1.73x faster than M1 Ultra on the selected 16 rows (avg ~1.73x, median ~1.77x). The largest MoE rows show the M5 Max advantage: qwen3-moe-30b runs at 175.48 vs 83.42 tok/s (2.10x), gpt-oss-120b at 113.90 vs 59.29 (1.92x), and solar-open-100b at 65.40 vs 35.02 (1.87x). On GB10 the CUDA fused decode-MoE kernel (#319) keeps qwen3-moe-30b (89.06) just ahead of M1 Ultra (83.42). For Qwen2.5-0.5B the 4-bit row is the directly comparable cross-hardware figure; the bf16 variant runs at 295.65 tok/s on M1 Ultra and 401.49 tok/s on M5 Max. diff --git a/docs/benchmark_results/model_tests_gb10.md b/docs/benchmark_results/model_tests_gb10.md index ea560dcd7..10ce81ca5 100644 --- a/docs/benchmark_results/model_tests_gb10.md +++ b/docs/benchmark_results/model_tests_gb10.md @@ -93,8 +93,8 @@ Prefill/Decode are the measured-pass figures from `mlxcel-bench-decode`. Notes r | gemma-4-12b-it-4bit-gs32 | ✅ | 215.87 | 21.61 | 40 tok; local uniform-requant variant (#685 tooling) | | gemma-4-12b-it-4bit-uniform | ✅ | 192.16 | 21.15 | 27 tok; local uniform-requant variant (#685 tooling) | | gemma-4-12b-it-assistant-4bit | ⚪ | - | - | MTP/DFlash drafter (needs a target; not standalone) | -| gemma-4-26b-a4b-it-4bit | ✅ | 114.81 | 50.19 | 26 tok | -| gemma-4-26b-a4b-it-qat-4bit | ✅ | 114.43 | 45.29 | 26 tok | +| gemma-4-26b-a4b-it-4bit | ✅ | 134.35 | 59.88 | 26 tok; median of 3 post-reboot runs (#755): 59.57-60.07, back above the 0.3.1 record (58.59) | +| gemma-4-26b-a4b-it-qat-4bit | ✅ | 133.43 | 53.68 | 26 tok; median of 3 post-reboot runs (#755): 53.48-53.71, above the 0.3.1 record (50.33) | | gemma-4-31b-4bit | ✅ | 23.08 | 8.84 | | | gemma-4-31b-it-4bit | ✅ | 52.37 | 8.33 | 26 tok | | gemma-4-31b-it-assistant-bf16 | ⚪ | - | - | MTP/DFlash drafter (needs a target; not standalone) | @@ -126,8 +126,8 @@ Prefill/Decode are the measured-pass figures from `mlxcel-bench-decode`. Notes r | Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | |-------|--------|-----------------|----------------|-------| | granite-3.3-2b-instruct-4bit | ✅ | 1779.32 | 123.98 | 31 tok | -| granite-4.0-h-350m-4bit | ✅ | 2207.98 | 259.69 | 18 tok | -| granite-4.0-h-tiny-4bit | ✅ | 256.74 | 100.28 | 44 tok | +| granite-4.0-h-350m-4bit | ✅ | 1857.41 | 171.22 | 19 tok; post-reboot single (#755); the overnight-sweep run read 259.69 (tiny launch-bound models swing widely run-to-run on this host) | +| granite-4.0-h-tiny-4bit | ✅ | 238.51 | 101.37 | 42 tok; post-reboot single (#755) | | granite-4.1-3b-4bit | ✅ | 291.06 | 78.85 | 7 tok | | granite-4.1-8b-4bit | ✅ | 177.90 | 18.43 | 1 tok | | granite-speech-4.1-2b-nar-mlx | ⚪ | - | - | not a standalone text-gen model | @@ -140,13 +140,13 @@ Prefill/Decode are the measured-pass figures from `mlxcel-bench-decode`. Notes r | dots.llm1.inst-mixed-4-6bit | ✅ | 25.42 | 22.04 | memory-gate skip on 2026-07-12 (see version note); figures from 2026-06-17 / 0.3.1 | | gpt-oss-120b-4bit | ✅ | 57.75 | 50.48 | memory-gate skip on 2026-07-12 (see version note); figures from 2026-06-17 / 0.3.1 | | gpt-oss-20b-mxfp4 | ✅ | 104.58 | 79.36 | | -| lfm2-8b-a1b-4bit | ✅ | 150.06 | 161.87 | 37 tok | +| lfm2-8b-a1b-4bit | ✅ | 148.84 | 165.53 | 37 tok; post-reboot single (#755 control) | | llama-4-scout-17b-4bit | ✅ | 27.72 | 21.46 | memory-gate skip on 2026-07-12 (see version note); figures from 2026-07-09 / 0.4.0-rc.1 | | minimax-m2-3bit | ✅ | 26.95 | 22.03 | memory-gate skip on 2026-07-12 (see version note); figures from 2026-06-17 / 0.3.1 | | mixtral-8x7b-4bit | ✅ | 12.63 | 28.42 | 73 tok | | phi-3.5-moe-4bit | ✅ | 30.18 | 53.32 | | | qwen1.5-moe-a2.7b-4bit | ✅ | 260.27 | 122.13 | | -| qwen3-30b-a3b-4bit | ✅ | 127.33 | 92.41 | 34 tok | +| qwen3-30b-a3b-4bit | ✅ | 134.37 | 94.39 | 34 tok; post-reboot single (#755 control) | | qwen3-moe-4bit | ✅ | 129.28 | 89.06 | 34 tok | ## MLA (Multi-head Latent Attention) @@ -166,21 +166,21 @@ Prefill/Decode are the measured-pass figures from `mlxcel-bench-decode`. Notes r | Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | |-------|--------|-----------------|----------------|-------| -| nemotron-3-nano-omni-30b-a3b-reasoning-4bit | ✅ | 123.57 | 80.86 | 20 tok | -| nemotron-h-30b-4bit | ✅ | 117.23 | 79.94 | 46 tok | -| nemotron-nas-30b-4bit | ✅ | 114.13 | 82.72 | 46 tok | +| nemotron-3-nano-omni-30b-a3b-reasoning-4bit | ✅ | 120.97 | 82.88 | 20 tok; post-reboot single (#755) | +| nemotron-h-30b-4bit | ✅ | 113.24 | 87.41 | 46 tok; post-reboot single (#755) | +| nemotron-nas-30b-4bit | ✅ | 117.99 | 85.91 | 46 tok; post-reboot single (#755) | ## SSM / Mamba / Hybrid | Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | |-------|--------|-----------------|----------------|-------| -| falcon-h1-tiny-90m-instruct-4bit | ✅ | 1419.44 | 413.00 | 30 tok | +| falcon-h1-tiny-90m-instruct-4bit | ✅ | 1193.95 | 354.43 | 30 tok; post-reboot single (#755); the overnight-sweep run read 413.00 | | falcon-mamba-7b-4bit | ✅ | 92.54 | 22.06 | 2 tok | | jamba-v0.1-4bit | ✅ | 523.67 | 89.63 | | | lfm2-350m-8bit | ✅ | 3270.66 | 393.84 | 13 tok; decode regression fixed (#748): was ~40 tok/s, restored to the 0.3.1 envelope by computing the single-step depthwise short conv as a broadcast multiply-sum instead of a tiny bf16 `conv1d` (MLX 0.32.1 routed that to cuDNN's per-channel grouped-conv engine on CUDA) | -| mamba2-130m | ✅ | 890.29 | 162.23 | decode conv audited (#752): runs on the fast `conv1d_c1_k1_nhwc` engine (576 launches = 24 layers x 24 tokens, one per op), not the `convolve_common_engine` per-channel path; the -10.4% sweep delta is environmental, not conv dispatch | +| mamba2-130m | ✅ | 892.83 | 202.98 | median of 3 post-reboot runs (#755): decode read 147.98 / 202.98 / 204.77 across identical same-boot runs, so the earlier -10.4% sweep delta was run-to-run variance, not a regression; decode conv audited (#752): runs on the fast `conv1d_c1_k1_nhwc` engine (576 launches = 24 layers x 24 tokens, one per op), not the `convolve_common_engine` per-channel path | | mamba2-1.3b-4bit | ✅ | 329.13 | 83.40 | | -| plamo-2-1b | ✅ | 199.12 | 44.54 | | +| plamo-2-1b | ✅ | 198.43 | 46.84 | post-reboot single (#755) | ## Chinese / Asian Language Models @@ -188,7 +188,7 @@ Prefill/Decode are the measured-pass figures from `mlxcel-bench-decode`. Notes r |-------|--------|-----------------|----------------|-------| | baichuan-m1-14b-4bit | ✅ | 73.62 | 23.18 | 7 tok | | ernie-4.5-0.3b-4bit | ✅ | 5857.00 | 625.30 | | -| glm4-flash-4bit | ✅ | 99.04 | 45.72 | 16 tok | +| glm4-flash-4bit | ✅ | 105.46 | 47.42 | 18 tok; median of 3 post-reboot runs (#755); identical 100-token runs span 39.2-54.8 tok/s on this host, and the 0.3.1 record (53.33) sits inside that envelope | | glm-5.1-4bit | ⚪ | - | - | not tested (weights not downloaded; empty directory) | | glm-5-4bit | ⚪ | - | - | incomplete checkpoint: weights present but no tokenizer | | hunyuan-13b | ✅ | 19.05 | 14.79 | | @@ -372,9 +372,9 @@ Models that accept image input and generated tokens under the `"What is in this ### Notable changes vs the 2026-06-17 (0.3.1) and 2026-07-09 (rc.1 subset) records -- **SSM / hybrid / NAS decode reads 2-3x higher than every earlier record**: granite-4.0-h-350m 86.60 → 259.69, granite-4.0-h-tiny 33.84 → 100.28, falcon-h1-tiny 110.42 → 413.00, nemotron-h-30b 40.32 → 79.94, nemotron-nas-30b 37.33 → 82.72, nemotron-omni-30b 38.45 → 80.86, plamo-2-1b 35.14 → 44.54. Two of these (granite-350m, falcon-h1-tiny) were re-measured as recently as 2026-07-09 at the low values, and no SSM-related code has landed since, so the delta is environmental rather than a code change. The affected cluster is exactly the launch-latency-sensitive family. Re-verify after the planned pre-release reboot before treating these as the release numbers; the post-reboot session is shared with the #755 re-measurement plan. +- **SSM / hybrid / NAS decode reads 1.4-3.4x higher than the 0.3.1 record, attributed to #727 and confirmed post-reboot (#755)**: granite-4.0-h-350m 86.60 → 171.22, granite-4.0-h-tiny 33.84 → 101.37, falcon-h1-tiny 110.42 → 354.43, nemotron-h-30b 40.32 → 87.41, nemotron-nas-30b 37.33 → 85.91, nemotron-omni-30b 38.45 → 82.88, plamo-2-1b 35.14 → 46.84 (post-reboot singles; the table rows above carry these values). An earlier revision of this note called the delta environmental because "no SSM-related code has landed since" the low 2026-07-09 readings. That was wrong: the fused single-token SSM decode kernel was ported to CUDA on 2026-07-10 (#727, one launch replacing the ~55-op SSD scan graph per SSM layer for granite-4.0-h, falcon-h1, plamo-2, nemotron-h), the day after those singles ran, and its own PR measured granite-4.0-h-350m at 4.5x. The post-reboot re-measurement (#755) confirms the gains persist on a fresh host, so these are real release numbers, not an artifact. The tiny launch-bound checkpoints still swing widely between individual runs (granite-350m read 259.69 in the overnight sweep vs 171.22 post-reboot; falcon-h1-tiny 413.00 vs 354.43); see the run-to-run variance note below. - **lfm2-350m-8bit decode regressed ~10x, now fixed (#748)**: 409.01 (0.3.1) → 39.84 at rc.1, restored to 393.84 by the fix. Root cause: MLX 0.32.1 dispatches the single-step (L=1) bf16 depthwise short conv on CUDA to cuDNN's generic `convolve_common_engine`, which launches one kernel per channel (~1024 for a 350M LFM2) and consumed 88.6% of decode GPU time; computing that decode step as a broadcast multiply-and-sum over the `L_cache` taps removes the grouped-conv dispatch. The sibling `lfm2-8b-a1b-4bit` was never affected (its conv runs on the fast `conv1d_c1_k1_nhwc` kernel; 157.73 → 161.87 → 161.39). Prefill still uses `conv1d` and was always healthy. -- **Moderate decode drops worth watching**: gemma-4-26b-a4b-it-4bit 58.59 → 50.19 (-14%), gemma-4-26b-a4b-it-qat-4bit 50.33 → 45.29 (-10%), glm4-flash-4bit 53.33 → 45.72 (-14%). These are NOT the #748 conv-dispatch pattern (correcting the PR #751 hypothesis on the record, per #752): neither model is a conv architecture (`gemma4` MoE and `glm4_moe_lite` respectively; neither file calls `conv1d`), so their decode conv cannot fall into cuDNN's grouped-conv engine. The drops need separate attribution and are out of scope for the conv audit; tracked in #755, with a post-reboot re-measurement as the gating first step alongside the SSM-cluster re-verification above. +- **Moderate decode drops, resolved as no code regression (#755)**: the overnight sweep read gemma-4-26b-a4b-it-4bit 58.59 → 50.19 (-14%), gemma-4-26b-a4b-it-qat-4bit 50.33 → 45.29 (-10%), glm4-flash-4bit 53.33 → 45.72 (-14%), and these were never the #748 conv-dispatch pattern (neither `gemma4` MoE nor `glm4_moe_lite` calls `conv1d`, per #752). The post-reboot re-measurement settles them in two different ways. The gemma pair recovered decisively and repeatably: 59.57-60.07 (n=3) and 53.48-53.71 (n=3), at or above the 0.3.1 records with ±0.5% spread. The 26B MoE is bandwidth-bound and does not flap between runs, so its sweep-day depression was the stale ~5.5-day host/driver state and a fresh boot removes it. glm4-flash did not "recover" because there was nothing to recover from: twelve identical 100-token greedy runs on the freshly booted host span 39.2-54.8 tok/s in a bimodal pattern (a ~41 tok/s mode and a ~52.5 tok/s mode; the generated tokens are identical, so the work is constant), and the 0.3.1 record (53.33) sits inside the fast mode, meaning a single sweep run simply draws from this distribution. mamba2-130m behaves the same (147.98 then 202.98-208.25 across identical same-boot runs, vs the 181.05 record). Kill-switch A/B rules out the code suspects: `MLXCEL_QMV_MULTIROW=0` reads 49.55 (inside the envelope; the #740 multirow path keeps `M*B == 1` classic decode on the stock kernel by design), `MLXCEL_SSM_CUDA_KERNEL=0` is unchanged on mamba2 (208.25; pure mamba2 never uses the fused kernel), and #732 shipped only trace tooling plus a default-off normalization. The slow mode is not SM clock capping either (slow runs were sampled at a pinned 2411-2424 MHz SM clock), so its host-level mechanism remains unidentified, but the fast mode reproducing the 0.3.1 number rules out a code-level regression. Practical consequence: single-run decode deltas within roughly ±25% on launch-bound models (small dense/SSM checkpoints and small MoEs like glm4-flash) are below this host's run-to-run noise floor and should not be read as regressions without repeats. Separately, glm4-flash's templated greedy output shortened from 100 tokens (0.3.1) to 18 (rc.1), a different greedy continuation rather than a failure, so its sweep-to-sweep decode averages additionally stopped being length-comparable. - **SSM / hybrid L=1 conv dispatch audited across the family (#752), no further regression found**: after #748/#751 fixed LFM2, every other decode-path depthwise-conv family was measured under `nsys` on a warm GB10 CUDA decode (`MLX_USE_CUDA_GRAPHS=0` for a complete kernel histogram). All run on the fast `conv1d_c1_k1_nhwc` cuDNN engine, not the per-channel `convolve_common_engine` that regressed LFM2: mamba2-130m (bf16), mamba2-1.3b-4bit, falcon-mamba-7b-4bit, falcon-h1-tiny-90m-4bit, granite-4.0-h-350m-4bit, jamba-v0.1-4bit, plamo-2-1b (f32 activations), qwen3.5-0.8b-4bit (covers the gated-delta conv), and nemotron-h-30b-4bit. The proof is the launch count: each shows conv instances equal to (conv layers x decode tokens), one launch per op, whereas the slow engine launches one kernel per channel (thousands per op). LFM2's slow-engine dispatch is specific to its `conv_L_cache = 3` / hidden-1024 shape and does not reproduce at the SSM `conv_kernel = 4` widths, so no family is adapted. All of this was measured on MLX pin `57c66cac` (0.32.1), the same pin whose CUDA conv dispatch heuristic produced the lfm2 slow path, so these REFUTED verdicts are point-in-time and should be re-checked against newer MLX pins. The #751 short-conv decode helper was still lifted into a shared `models::conv_decode` module so any future family that does regress can adopt it in one line. - **Text-only prefill for several VLM-capable models is an order of magnitude higher than 0.3.1** (aya-vision-8b 124.91 → 1354.53, pixtral-12b 35.97 → 120.39, youtu-vl 134.27 → 451.42, mistral-small-3.1-24b 63.68 → 891.89), consistent with the text-prompt path no longer paying vision-tower costs rather than with a kernel speedup. - **Decode improvements >10% on dense/MoE models**: apertus-8b +21%, gemma3-4b / gemma-3-4b-it +19%, gemma3-1b +15%, gemma3n-e2b +10%, qwen3-0.6b-4bit +13%, aya-expanse-8b +12%, plus the VLM-side gains below.