diff --git a/benchmarks/cuda_gb10_2026-06-17.csv b/benchmarks/cuda_gb10_2026-06-17.csv new file mode 100644 index 000000000..ca3cb08d3 --- /dev/null +++ b/benchmarks/cuda_gb10_2026-06-17.csv @@ -0,0 +1,149 @@ +model,model_path,prompt_tokens,generated_tokens,prefill_ms,prefill_tok_s,decode_ms,decode_tok_s,date,hardware,mlx_version,build_type,max_tokens,prompt +apertus-8b-instruct-2509-4bit,./models/apertus-8b-instruct-2509-4bit/,66,31,133.36,494.90,735.47,42.15,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +aya-expanse-8b-4bit,./models/aya-expanse-8b-4bit/,8,100,44.74,178.81,2088.40,47.88,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +aya-vision-8b,./models/aya-vision-8b/,8,87,64.05,124.91,1864.06,46.67,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +baichuan-m1-14b-4bit,./models/baichuan-m1-14b-4bit/,9,7,121.10,74.32,313.06,22.36,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +bitnet-b1.58-2b-4t,./models/bitnet-b1.58-2b-4t/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?",FAIL:bench +bitnet-b1.58-2b-4t-4bit,./models/bitnet-b1.58-2b-4t-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?",FAIL:bench +bunny-llama3-8b-4bit,./models/bunny-llama3-8b-4bit/,18,40,41.48,433.97,819.52,48.81,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +command-r7b-4bit,./models/command-r7b-4bit/,8,100,62.86,127.27,2092.77,47.78,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +deepseek-coder-1.3b-4bit,./models/deepseek-coder-1.3b-4bit/,76,100,16.74,4540.35,1151.31,86.86,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +deepseek-r1-distill-7b-4bit,./models/deepseek-r1-distill-7b-4bit/,13,100,61.94,209.88,1832.89,54.56,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +deepseek-v2-lite-4bit,./models/deepseek-v2-lite-4bit/,15,100,93.71,160.07,1032.96,96.81,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +deepseek-v3-4bit,./models/deepseek-v3-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?",FAIL:bench +diffusiongemma-26b-a4b-it-4bit,./models/diffusiongemma-26b-a4b-it-4bit/,20,27,170.07,117.60,716.53,37.68,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +docling-layout-heron-mlx-bf16,./models/docling-layout-heron-mlx-bf16/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?",FAIL:bench +dots.llm1.inst-mixed-4-6bit,./models/dots.llm1.inst-mixed-4-6bit/,18,39,708.04,25.42,1769.22,22.04,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +ernie-4.5-0.3b-4bit,./models/ernie-4.5-0.3b-4bit/,36,100,5.64,6384.30,152.68,654.97,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +exaone-3.5-2.4b-4bit,./models/exaone-3.5-2.4b-4bit/,41,100,29.49,1390.18,730.30,136.93,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +exaone4-1.2b-4bit,./models/exaone4-1.2b-4bit/,19,53,13.05,1456.06,246.83,214.72,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +falcon-h1-tiny-90m-instruct-4bit,./models/falcon-h1-tiny-90m-instruct-4bit/,17,100,13.13,1294.33,970.97,102.99,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +falcon-mamba-7b-4bit,./models/falcon-mamba-7b-4bit/,16,2,159.56,100.28,89.84,22.26,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma2-2b-4bit,./models/gemma2-2b-4bit/,16,27,25.01,639.86,246.47,109.55,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma-2b-4bit,./models/gemma-2b-4bit/,16,49,27.42,583.56,521.23,94.01,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma3-1b-4bit,./models/gemma3-1b-4bit/,16,34,12.77,1252.95,140.58,241.85,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma3-4b-4bit,./models/gemma3-4b-4bit/,16,84,38.97,410.60,1100.86,76.30,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma-3-4b-it-4bit,./models/gemma-3-4b-it-4bit/,16,84,38.91,411.18,1103.63,76.11,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma3n-e2b-4bit,./models/gemma3n-e2b-4bit/,16,72,37.36,428.25,900.42,79.96,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma3n-e4b-4bit,./models/gemma3n-e4b-4bit/,16,74,56.86,281.38,1381.38,53.57,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma-4-12b-it-4bit,./models/gemma-4-12b-it-4bit/,20,100,98.69,202.66,6802.77,14.70,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma-4-12b-it-assistant-4bit,./models/gemma-4-12b-it-assistant-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?",FAIL:bench +gemma-4-26b-a4b-it-4bit,./models/gemma-4-26b-a4b-it-4bit/,20,26,152.90,130.81,443.75,58.59,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma-4-26b-a4b-it-qat-4bit,./models/gemma-4-26b-a4b-it-qat-4bit/,20,26,147.00,136.06,516.55,50.33,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma-4-31b-4bit,./models/gemma-4-31b-4bit/,8,100,261.60,30.58,11197.89,8.93,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma-4-31b-it-4bit,./models/gemma-4-31b-it-4bit/,20,26,284.68,70.25,3048.11,8.53,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma-4-31b-it-assistant-bf16,./models/gemma-4-31b-it-assistant-bf16/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?",FAIL:bench +gemma-4-31b-it-nvfp4,./models/gemma-4-31b-it-nvfp4/,20,26,1230.39,16.26,29291.84,0.89,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma-4-31b-it-qat-4bit,./models/gemma-4-31b-it-qat-4bit/,20,26,248.41,80.51,4647.92,5.59,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma-4-e2b-it-4bit,./models/gemma-4-e2b-it-4bit/,20,100,26.26,761.69,959.22,104.25,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma-4-e2b-it-8bit,./models/gemma-4-e2b-it-8bit/,20,100,32.88,608.29,1711.56,58.43,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma-4-e2b-it-qat-4bit,./models/gemma-4-e2b-it-qat-4bit/,16,100,31.79,503.36,1506.22,66.39,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma-4-e4b-it-4bit,./models/gemma-4-e4b-it-4bit/,20,100,56.30,355.22,2024.65,49.39,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma-4-e4b-it-8bit,./models/gemma-4-e4b-it-8bit/,20,45,68.41,292.35,1644.91,27.36,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma-4-e4b-it-qat-4bit,./models/gemma-4-e4b-it-qat-4bit/,16,33,58.46,273.71,980.24,33.67,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +glm4-flash-4bit,./models/glm4-flash-4bit/,12,100,112.51,106.65,1875.20,53.33,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +glm-5.1-4bit,./models/glm-5.1-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?",FAIL:bench +glm-5-4bit,./models/glm-5-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?",FAIL:bench +gpt-oss-120b-4bit,./models/gpt-oss-120b-4bit/,73,82,1264.09,57.75,1624.32,50.48,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gpt-oss-20b-mxfp4,./models/gpt-oss-20b-mxfp4/,73,100,578.61,126.16,1294.55,77.25,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +granite-3.3-2b-instruct-4bit,./models/granite-3.3-2b-instruct-4bit/,64,28,35.52,1801.88,246.16,113.75,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +granite-4.0-h-350m-4bit,./models/granite-4.0-h-350m-4bit/,37,19,21.58,1714.62,296.88,64.00,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +granite-4.0-h-tiny-4bit,./models/granite-4.0-h-tiny-4bit/,15,53,58.41,256.82,1566.33,33.84,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +granite-4.1-3b-4bit,./models/granite-4.1-3b-4bit/,15,7,45.60,328.96,94.93,73.74,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +granite-4.1-8b-4bit,./models/granite-4.1-8b-4bit/,15,1,90.26,166.20,54.19,18.45,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +granite-speech-4.1-2b-nar-mlx,./models/granite-speech-4.1-2b-nar-mlx/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?",FAIL:bench +hunyuan-13b,./models/hunyuan-13b/,15,100,798.64,18.78,6758.48,14.80,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +hunyuan-1.8b-4bit,./models/hunyuan-1.8b-4bit/,14,41,20.41,685.98,265.00,154.72,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +hunyuan-a13b-instruct-4bit,./models/hunyuan-a13b-instruct-4bit/,27,100,1345.49,20.07,6638.69,15.06,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +internlm2-7b-4bit,./models/internlm2-7b-4bit/,17,100,43.75,388.59,1996.42,50.09,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +internlm3-8b-4bit,./models/internlm3-8b-4bit/,27,100,51.65,522.79,2278.62,43.89,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +internvl3-1b,./models/internvl3-1b/,36,37,7.90,4559.06,78.15,473.46,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +jamba-v0.1-4bit,./models/jamba-v0.1-4bit/,44,100,81.31,541.13,1169.90,85.48,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +lfm2-350m-8bit,./models/lfm2-350m-8bit/,17,13,6.14,2768.23,31.78,409.01,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +lfm2-8b-a1b-4bit,./models/lfm2-8b-a1b-4bit/,17,37,116.87,145.46,234.58,157.73,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +llama-3.1-8b-4bit,./models/llama-3.1-8b-4bit/,98,100,75.69,1294.78,2036.61,49.10,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +llama-3.1-8b-bf16,./models/llama-3.1-8b-bf16/,99,87,81.83,1209.77,5864.14,14.84,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +llama-3.2-1b-4bit,./models/llama-3.2-1b-4bit/,99,100,12.70,7793.45,384.15,260.32,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +llama-4-scout-17b-4bit,./models/llama-4-scout-17b-4bit/,69,100,2446.74,28.20,4774.95,20.94,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +llava-1.5-7b-4bit,./models/llava-1.5-7b-4bit/,8,100,39.01,205.07,1816.08,55.06,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +llava-interleave-qwen-0.5b-bf16,./models/llava-interleave-qwen-0.5b-bf16/,26,49,10.27,2531.16,237.20,206.58,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +llava-next-mistral-7b-4bit,./models/llava-next-mistral-7b-4bit/,15,100,40.07,374.38,1937.76,51.61,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +mamba2-130m,./models/mamba2-130m/,7,100,7.11,984.73,552.33,181.05,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +mamba2-1.3b-4bit,./models/mamba2-1.3b-4bit/,7,100,24.67,283.78,1228.92,81.37,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +meta-llama-3.1-8b-instruct-4bit,./models/meta-llama-3.1-8b-instruct-4bit/,98,100,74.23,1320.16,2043.57,48.93,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +mimo-7b-4bit,./models/mimo-7b-4bit/,26,100,73.16,355.38,1909.37,52.37,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +minicpm-2b-4bit,./models/minicpm-2b-4bit/,14,100,29.36,476.85,827.24,120.88,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +minicpm3-4b-4bit,./models/minicpm3-4b-4bit/,17,100,61.71,275.48,1780.65,56.16,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +minicpm-v-4.6-bf16,./models/minicpm-v-4.6-bf16/,7,100,22.04,317.54,906.23,110.35,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +minicpm-v-4.6-mxfp4,./models/minicpm-v-4.6-mxfp4/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?",FAIL:bench +minimax-m2-3bit,./models/minimax-m2-3bit/,29,100,1075.96,26.95,4539.49,22.03,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +ministral-3b-4bit,./models/ministral-3b-4bit/,542,34,86.30,6280.07,338.17,100.54,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +mistral-small-3.1-24b-4bit,./models/mistral-small-3.1-24b-4bit/,8,100,125.63,63.68,6218.68,16.08,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +mixtral-8x7b-4bit,./models/mixtral-8x7b-4bit/,15,73,1204.06,12.46,2614.94,27.92,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +molmo2-4b,./models/molmo2-4b/,15,33,76.14,197.01,1242.49,26.56,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +molmo-7b,./models/molmo-7b/,11,24,54.01,203.66,713.44,33.64,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +nemotron-3-nano-omni-30b-a3b-reasoning-4bit,./models/nemotron-3-nano-omni-30b-a3b-reasoning-4bit/,23,20,195.31,117.76,520.11,38.45,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +nemotron-h-30b-4bit,./models/nemotron-h-30b-4bit/,23,46,197.92,116.21,1140.78,40.32,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +nemotron-nas-30b-4bit,./models/nemotron-nas-30b-4bit/,23,46,203.48,113.03,1232.29,37.33,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +olmo-1b-4bit,./models/olmo-1b-4bit/,7,100,26.80,261.22,1011.07,98.90,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +olmo2-7b-4bit,./models/olmo2-7b-4bit/,21,27,67.42,311.46,507.83,53.17,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +olmo3-32b-4bit,./models/olmo3-32b-4bit/,63,100,205.65,306.34,8581.39,11.65,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +paligemma2-3b-6bit,./models/paligemma2-3b-6bit/,7,0,43.59,160.60,10.17,0.00,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +phi-2-4bit,./models/phi-2-4bit/,7,1,52.65,132.96,28.32,35.31,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +phi-3.5-mini-4bit,./models/phi-3.5-mini-4bit/,10,40,34.42,290.51,437.87,91.35,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +phi-3.5-moe-4bit,./models/phi-3.5-moe-4bit/,10,100,346.38,28.87,1994.92,50.13,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +phi-3.5-vision-4bit,./models/phi-3.5-vision-4bit/,15,43,34.44,435.53,472.27,91.05,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +phi-3-mini-4bit,./models/phi-3-mini-4bit/,10,22,35.17,284.33,238.51,92.24,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +phi-4-4bit,./models/phi-4-4bit/,14,100,84.59,165.50,3677.77,27.19,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +pixtral-12b,./models/pixtral-12b/,7,100,194.61,35.97,3048.58,32.80,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +pixtral-12b-4bit,./models/pixtral-12b-4bit/,7,100,194.33,36.02,3044.53,32.85,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +plamo-2-1b,./models/plamo-2-1b/,7,100,36.86,189.89,2910.24,34.36,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen1.5-moe-a2.7b-4bit,./models/qwen1.5-moe-a2.7b-4bit/,25,100,95.70,261.24,796.67,125.52,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen2-0.5b,./models/qwen2-0.5b/,36,100,8.83,4075.86,208.50,479.62,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen2.5-0.5b-4bit,./models/qwen2.5-0.5b-4bit/,36,100,9.51,3784.53,205.90,485.68,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen2.5-0.5b-bf16,./models/qwen2.5-0.5b-bf16/,36,100,10.54,3416.94,500.26,199.89,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen2.5-1.5b-4bit,./models/qwen2.5-1.5b-4bit/,26,100,23.60,1101.93,497.77,200.90,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen2.5-1.5b-instruct-4bit,./models/qwen2.5-1.5b-instruct-4bit/,36,100,23.40,1538.21,496.05,201.59,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen2.5-7b,./models/qwen2.5-7b/,36,100,60.53,594.78,1865.59,53.60,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen2.5-7b-4bit,./models/qwen2.5-7b-4bit/,36,100,59.14,608.73,1881.23,53.16,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen2.5-7b-8bit,./models/qwen2.5-7b-8bit/,36,100,54.25,663.57,3313.80,30.18,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen2.5-7b-instruct-4bit,./models/qwen2.5-7b-instruct-4bit/,36,100,58.27,617.78,1842.23,54.28,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen2.5-vl-3b,./models/qwen2.5-vl-3b/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?",FAIL:bench +qwen2.5-vl-3b-4bit,./models/qwen2.5-vl-3b-4bit/,26,39,70.04,371.22,650.73,59.93,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen2-vl-2b,./models/qwen2-vl-2b/,26,35,44.59,583.13,345.35,101.35,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen2-vl-2b-4bit,./models/qwen2-vl-2b-4bit/,26,35,42.83,607.09,347.07,100.84,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3-0.6b,./models/qwen3-0.6b/,19,9,10.77,1764.03,30.58,294.34,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3-0.6b-4bit,./models/qwen3-0.6b-4bit/,19,9,11.93,1592.92,30.98,290.52,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3-1.7b-4bit,./models/qwen3-1.7b-4bit/,19,14,16.85,1127.29,82.33,170.05,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3-30b-a3b-4bit,./models/qwen3-30b-a3b-4bit/,19,34,142.42,133.40,374.87,90.70,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3-4b-4bit,./models/qwen3-4b-4bit/,19,36,35.68,532.50,446.91,80.55,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3.5-0.8b-4bit,./models/qwen3.5-0.8b-4bit/,19,18,29.65,640.75,102.89,174.95,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3.5-0.8b-optiq-4bit,./models/qwen3.5-0.8b-optiq-4bit/,19,19,29.49,644.30,121.08,156.91,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3.5-27b-4bit,./models/qwen3.5-27b-4bit/,19,30,319.95,59.38,2457.81,12.21,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3.5-27b-dflash,./models/qwen3.5-27b-dflash/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?",FAIL:bench +qwen3.5-2b-4bit,./models/qwen3.5-2b-4bit/,19,32,36.91,514.72,256.93,124.55,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3.5-35b-a3b-4bit,./models/qwen3.5-35b-a3b-4bit/,19,31,131.69,144.28,482.38,64.26,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3.5-4b-4bit,./models/qwen3.5-4b-4bit/,19,31,76.63,247.95,499.57,62.05,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3.5-4b-dflash,./models/qwen3.5-4b-dflash/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?",FAIL:bench +qwen3.5-9b-4bit,./models/qwen3.5-9b-4bit/,19,31,109.93,172.83,796.56,38.92,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3.5-9b-bf16,./models/qwen3.5-9b-bf16/,19,31,112.25,169.27,2364.43,13.11,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3.6-35b-a3b-4bit,./models/qwen3.6-35b-a3b-4bit/,19,28,128.23,148.18,444.51,62.99,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3-8b-4bit,./models/qwen3-8b-4bit/,19,33,80.36,236.45,693.97,47.55,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3-moe-4bit,./models/qwen3-moe-4bit/,19,34,140.42,135.30,378.45,89.84,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3-next-480b-4bit,./models/qwen3-next-480b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?",SKIP:oom_estimate +qwen3-vl-2b,./models/qwen3-vl-2b/,15,59,18.09,828.99,359.87,163.95,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3-vl-2b-4bit,./models/qwen3-vl-2b-4bit/,15,59,17.85,840.44,357.26,165.15,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3-vl-30b-a3b-4bit,./models/qwen3-vl-30b-a3b-4bit/,15,35,118.38,126.71,420.40,83.25,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3-vl-32b-4bit,./models/qwen3-vl-32b-4bit/,15,37,178.42,84.07,3377.61,10.95,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3-vl-4b-4bit,./models/qwen3-vl-4b-4bit/,15,49,41.71,359.60,647.65,75.66,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3-vl-4b-instruct-4bit,./models/qwen3-vl-4b-instruct-4bit/,15,49,42.81,350.36,646.61,75.78,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3-vl-8b-4bit,./models/qwen3-vl-8b-4bit/,15,57,70.41,213.04,1198.29,47.57,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +qwen3-vl-8b-instruct-4bit,./models/qwen3-vl-8b-instruct-4bit/,15,57,65.92,227.56,1196.98,47.62,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +seed-oss-36b-instruct-4bit,./models/seed-oss-36b-instruct-4bit/,15,100,190.26,78.84,9779.79,10.23,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +smollm-135m-4bit,./models/smollm-135m-4bit/,16,100,6.59,2429.28,153.36,652.08,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +smollm3-3b-4bit,./models/smollm3-3b-4bit/,75,51,45.57,1645.91,489.86,104.11,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +solar-open-100b-4bit,./models/solar-open-100b-4bit/,73,100,1561.26,46.76,5443.79,18.37,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +stablelm-1.6b-4bit,./models/stablelm-1.6b-4bit/,26,100,15.49,1678.51,504.35,198.28,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +starcoder2-3b-4bit,./models/starcoder2-3b-4bit/,7,100,32.53,215.17,982.11,101.82,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +youtu-vl-4b-instruct,./models/youtu-vl-4b-instruct/,7,100,52.13,134.27,4558.89,21.94,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" +gemma3n-e4b-bf16,./models/gemma3n-e4b-bf16/,16,69,58.14,275.17,3232.57,21.35,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?" diff --git a/benchmarks/cuda_gb10_vlm_2026-06-17.csv b/benchmarks/cuda_gb10_vlm_2026-06-17.csv new file mode 100644 index 000000000..173574b56 --- /dev/null +++ b/benchmarks/cuda_gb10_vlm_2026-06-17.csv @@ -0,0 +1,149 @@ +model,model_path,prompt_tokens,generated_tokens,prefill_ms,prefill_tok_s,decode_ms,decode_tok_s,date,hardware,mlx_version,build_type,max_tokens,prompt +apertus-8b-instruct-2509-4bit,./models/apertus-8b-instruct-2509-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +aya-expanse-8b-4bit,./models/aya-expanse-8b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +aya-vision-8b,./models/aya-vision-8b/,176,100,278.26,632.49,2205.52,45.34,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +baichuan-m1-14b-4bit,./models/baichuan-m1-14b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +bitnet-b1.58-2b-4t,./models/bitnet-b1.58-2b-4t/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +bitnet-b1.58-2b-4t-4bit,./models/bitnet-b1.58-2b-4t-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +bunny-llama3-8b-4bit,./models/bunny-llama3-8b-4bit/,746,37,443.61,1681.67,820.51,45.09,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +command-r7b-4bit,./models/command-r7b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +deepseek-coder-1.3b-4bit,./models/deepseek-coder-1.3b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +deepseek-r1-distill-7b-4bit,./models/deepseek-r1-distill-7b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +deepseek-v2-lite-4bit,./models/deepseek-v2-lite-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +deepseek-v3-4bit,./models/deepseek-v3-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +diffusiongemma-26b-a4b-it-4bit,./models/diffusiongemma-26b-a4b-it-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +docling-layout-heron-mlx-bf16,./models/docling-layout-heron-mlx-bf16/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +dots.llm1.inst-mixed-4-6bit,./models/dots.llm1.inst-mixed-4-6bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +ernie-4.5-0.3b-4bit,./models/ernie-4.5-0.3b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +exaone-3.5-2.4b-4bit,./models/exaone-3.5-2.4b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +exaone4-1.2b-4bit,./models/exaone4-1.2b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +falcon-h1-tiny-90m-instruct-4bit,./models/falcon-h1-tiny-90m-instruct-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +falcon-mamba-7b-4bit,./models/falcon-mamba-7b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +gemma2-2b-4bit,./models/gemma2-2b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +gemma-2b-4bit,./models/gemma-2b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +gemma3-1b-4bit,./models/gemma3-1b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +gemma3-4b-4bit,./models/gemma3-4b-4bit/,275,23,270.12,1018.08,314.37,73.16,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +gemma-3-4b-it-4bit,./models/gemma-3-4b-it-4bit/,275,16,276.63,994.12,227.66,70.28,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +gemma3n-e2b-4bit,./models/gemma3n-e2b-4bit/,273,25,124.45,2193.58,341.97,73.10,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +gemma3n-e4b-4bit,./models/gemma3n-e4b-4bit/,273,33,186.07,1467.15,629.42,52.43,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +gemma-4-12b-it-4bit,./models/gemma-4-12b-it-4bit/,277,100,300.10,923.02,6805.58,14.69,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +gemma-4-12b-it-assistant-4bit,./models/gemma-4-12b-it-assistant-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +gemma-4-26b-a4b-it-4bit,./models/gemma-4-26b-a4b-it-4bit/,277,28,1806.83,153.31,512.41,54.64,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +gemma-4-26b-a4b-it-qat-4bit,./models/gemma-4-26b-a4b-it-qat-4bit/,277,30,1848.60,149.84,613.12,48.93,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +gemma-4-31b-4bit,./models/gemma-4-31b-4bit/,265,8,834.15,317.69,1022.84,7.82,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +gemma-4-31b-it-4bit,./models/gemma-4-31b-it-4bit/,277,25,999.02,277.27,3028.28,8.26,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +gemma-4-31b-it-assistant-bf16,./models/gemma-4-31b-it-assistant-bf16/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +gemma-4-31b-it-nvfp4,./models/gemma-4-31b-it-nvfp4/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +gemma-4-31b-it-qat-4bit,./models/gemma-4-31b-it-qat-4bit/,277,29,914.83,302.79,5195.92,5.58,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +gemma-4-e2b-it-4bit,./models/gemma-4-e2b-it-4bit/,277,100,109.53,2528.92,960.01,104.17,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +gemma-4-e2b-it-8bit,./models/gemma-4-e2b-it-8bit/,277,100,129.52,2138.74,1711.41,58.43,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +gemma-4-e2b-it-qat-4bit,./models/gemma-4-e2b-it-qat-4bit/,273,100,122.12,2235.54,1505.32,66.43,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +gemma-4-e4b-it-4bit,./models/gemma-4-e4b-it-4bit/,277,100,196.79,1407.62,2020.00,49.50,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +gemma-4-e4b-it-8bit,./models/gemma-4-e4b-it-8bit/,277,51,243.23,1138.82,1890.96,26.97,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +gemma-4-e4b-it-qat-4bit,./models/gemma-4-e4b-it-qat-4bit/,273,32,222.25,1228.36,960.50,33.32,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +glm4-flash-4bit,./models/glm4-flash-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +glm-5.1-4bit,./models/glm-5.1-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +glm-5-4bit,./models/glm-5-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +gpt-oss-120b-4bit,./models/gpt-oss-120b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +gpt-oss-20b-mxfp4,./models/gpt-oss-20b-mxfp4/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +granite-3.3-2b-instruct-4bit,./models/granite-3.3-2b-instruct-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +granite-4.0-h-350m-4bit,./models/granite-4.0-h-350m-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +granite-4.0-h-tiny-4bit,./models/granite-4.0-h-tiny-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +granite-4.1-3b-4bit,./models/granite-4.1-3b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +granite-4.1-8b-4bit,./models/granite-4.1-8b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +granite-speech-4.1-2b-nar-mlx,./models/granite-speech-4.1-2b-nar-mlx/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +hunyuan-13b,./models/hunyuan-13b/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +hunyuan-1.8b-4bit,./models/hunyuan-1.8b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +hunyuan-a13b-instruct-4bit,./models/hunyuan-a13b-instruct-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +internlm2-7b-4bit,./models/internlm2-7b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +internlm3-8b-4bit,./models/internlm3-8b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +internvl3-1b,./models/internvl3-1b/,293,8,149.62,1958.25,20.14,397.14,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +jamba-v0.1-4bit,./models/jamba-v0.1-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +lfm2-350m-8bit,./models/lfm2-350m-8bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +lfm2-8b-a1b-4bit,./models/lfm2-8b-a1b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +llama-3.1-8b-4bit,./models/llama-3.1-8b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +llama-3.1-8b-bf16,./models/llama-3.1-8b-bf16/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +llama-3.2-1b-4bit,./models/llama-3.2-1b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +llama-4-scout-17b-4bit,./models/llama-4-scout-17b-4bit/,230,100,7891.24,29.15,4867.51,20.54,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +llava-1.5-7b-4bit,./models/llava-1.5-7b-4bit/,583,100,338.69,1721.32,1920.25,52.08,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +llava-interleave-qwen-0.5b-bf16,./models/llava-interleave-qwen-0.5b-bf16/,754,32,36.54,20633.47,170.10,188.13,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +llava-next-mistral-7b-4bit,./models/llava-next-mistral-7b-4bit/,590,100,273.18,2159.71,1958.20,51.07,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +mamba2-130m,./models/mamba2-130m/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +mamba2-1.3b-4bit,./models/mamba2-1.3b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +meta-llama-3.1-8b-instruct-4bit,./models/meta-llama-3.1-8b-instruct-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +mimo-7b-4bit,./models/mimo-7b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +minicpm-2b-4bit,./models/minicpm-2b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +minicpm3-4b-4bit,./models/minicpm3-4b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +minicpm-v-4.6-bf16,./models/minicpm-v-4.6-bf16/,32,23,54.82,583.76,221.99,103.61,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +minicpm-v-4.6-mxfp4,./models/minicpm-v-4.6-mxfp4/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +minimax-m2-3bit,./models/minimax-m2-3bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +ministral-3b-4bit,./models/ministral-3b-4bit/,3566,100,1412.52,2524.57,1119.15,89.35,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +mistral-small-3.1-24b-4bit,./models/mistral-small-3.1-24b-4bit/,3032,100,6006.22,504.81,6522.41,15.33,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +mixtral-8x7b-4bit,./models/mixtral-8x7b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +molmo2-4b,./models/molmo2-4b/,438,46,566.31,773.43,1754.82,26.21,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +molmo-7b,./models/molmo-7b/,327,2,298.39,1095.89,83.99,23.81,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +nemotron-3-nano-omni-30b-a3b-reasoning-4bit,./models/nemotron-3-nano-omni-30b-a3b-reasoning-4bit/,279,6,1897.55,147.03,182.60,32.86,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +nemotron-h-30b-4bit,./models/nemotron-h-30b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +nemotron-nas-30b-4bit,./models/nemotron-nas-30b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +olmo-1b-4bit,./models/olmo-1b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +olmo2-7b-4bit,./models/olmo2-7b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +olmo3-32b-4bit,./models/olmo3-32b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +paligemma2-3b-6bit,./models/paligemma2-3b-6bit/,1032,2,588.10,1754.82,45.10,44.34,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +phi-2-4bit,./models/phi-2-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +phi-3.5-mini-4bit,./models/phi-3.5-mini-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +phi-3.5-moe-4bit,./models/phi-3.5-moe-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +phi-3.5-vision-4bit,./models/phi-3.5-vision-4bit/,773,19,546.95,1413.29,236.04,80.50,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +phi-3-mini-4bit,./models/phi-3-mini-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +phi-4-4bit,./models/phi-4-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +pixtral-12b,./models/pixtral-12b/,4102,100,5244.25,782.19,3301.86,30.29,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +pixtral-12b-4bit,./models/pixtral-12b-4bit/,4102,100,5307.30,772.90,3312.04,30.19,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +plamo-2-1b,./models/plamo-2-1b/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen1.5-moe-a2.7b-4bit,./models/qwen1.5-moe-a2.7b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen2-0.5b,./models/qwen2-0.5b/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen2.5-0.5b-4bit,./models/qwen2.5-0.5b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen2.5-0.5b-bf16,./models/qwen2.5-0.5b-bf16/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen2.5-1.5b-4bit,./models/qwen2.5-1.5b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen2.5-1.5b-instruct-4bit,./models/qwen2.5-1.5b-instruct-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen2.5-7b,./models/qwen2.5-7b/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen2.5-7b-4bit,./models/qwen2.5-7b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen2.5-7b-8bit,./models/qwen2.5-7b-8bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen2.5-7b-instruct-4bit,./models/qwen2.5-7b-instruct-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen2.5-vl-3b,./models/qwen2.5-vl-3b/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen2.5-vl-3b-4bit,./models/qwen2.5-vl-3b-4bit/,91,64,152.09,598.32,1072.12,59.70,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +qwen2-vl-2b,./models/qwen2-vl-2b/,91,12,137.80,660.36,131.01,91.60,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +qwen2-vl-2b-4bit,./models/qwen2-vl-2b-4bit/,91,12,138.70,656.09,136.15,88.14,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +qwen3-0.6b,./models/qwen3-0.6b/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen3-0.6b-4bit,./models/qwen3-0.6b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen3-1.7b-4bit,./models/qwen3-1.7b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen3-30b-a3b-4bit,./models/qwen3-30b-a3b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen3-4b-4bit,./models/qwen3-4b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen3.5-0.8b-4bit,./models/qwen3.5-0.8b-4bit/,69,100,71.60,963.74,494.97,202.03,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +qwen3.5-0.8b-optiq-4bit,./models/qwen3.5-0.8b-optiq-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen3.5-27b-4bit,./models/qwen3.5-27b-4bit/,69,100,664.77,103.80,7874.08,12.70,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +qwen3.5-27b-dflash,./models/qwen3.5-27b-dflash/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen3.5-2b-4bit,./models/qwen3.5-2b-4bit/,69,47,93.59,737.26,368.24,127.63,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +qwen3.5-35b-a3b-4bit,./models/qwen3.5-35b-a3b-4bit/,69,100,379.97,181.59,1667.23,59.98,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +qwen3.5-4b-4bit,./models/qwen3.5-4b-4bit/,69,49,162.14,425.56,788.57,62.14,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +qwen3.5-4b-dflash,./models/qwen3.5-4b-dflash/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen3.5-9b-4bit,./models/qwen3.5-9b-4bit/,69,100,221.62,311.34,2512.10,39.81,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +qwen3.5-9b-bf16,./models/qwen3.5-9b-bf16/,69,100,204.09,338.09,7383.55,13.54,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +qwen3.6-35b-a3b-4bit,./models/qwen3.6-35b-a3b-4bit/,69,100,394.21,175.03,1603.89,62.35,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +qwen3-8b-4bit,./models/qwen3-8b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen3-moe-4bit,./models/qwen3-moe-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +qwen3-next-480b-4bit,./models/qwen3-next-480b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"Hello, how are you today?",SKIP:oom_estimate +qwen3-vl-2b,./models/qwen3-vl-2b/,65,52,30.51,2130.17,393.29,132.22,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +qwen3-vl-2b-4bit,./models/qwen3-vl-2b-4bit/,65,84,30.45,2134.98,635.28,132.22,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +qwen3-vl-30b-a3b-4bit,./models/qwen3-vl-30b-a3b-4bit/,65,73,324.60,200.25,1148.44,63.56,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +qwen3-vl-32b-4bit,./models/qwen3-vl-32b-4bit/,65,51,430.50,150.99,4919.63,10.37,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +qwen3-vl-4b-4bit,./models/qwen3-vl-4b-4bit/,65,37,64.90,1001.53,559.33,66.15,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +qwen3-vl-4b-instruct-4bit,./models/qwen3-vl-4b-instruct-4bit/,65,37,64.82,1002.80,569.02,65.02,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +qwen3-vl-8b-4bit,./models/qwen3-vl-8b-4bit/,65,30,109.03,596.15,746.03,40.21,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +qwen3-vl-8b-instruct-4bit,./models/qwen3-vl-8b-instruct-4bit/,65,30,112.57,577.42,743.75,40.34,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +seed-oss-36b-instruct-4bit,./models/seed-oss-36b-instruct-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +smollm-135m-4bit,./models/smollm-135m-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +smollm3-3b-4bit,./models/smollm3-3b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +solar-open-100b-4bit,./models/solar-open-100b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +stablelm-1.6b-4bit,./models/stablelm-1.6b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +starcoder2-3b-4bit,./models/starcoder2-3b-4bit/,,,,,,,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?",FAIL:bench +youtu-vl-4b-instruct,./models/youtu-vl-4b-instruct/,57,0,68.90,827.31,40.53,0.00,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" +gemma3n-e4b-bf16,./models/gemma3n-e4b-bf16/,273,24,145.37,1878.01,1162.51,20.65,2026-06-17,NVIDIA_GB10_122GB,0.3.1,release,100,"What is in this image?" diff --git a/docs/benchmark_results/model_tests.md b/docs/benchmark_results/model_tests.md index 24994a523..efee59975 100644 --- a/docs/benchmark_results/model_tests.md +++ b/docs/benchmark_results/model_tests.md @@ -12,7 +12,7 @@ M5 Max, and mlx-lm / mlx-vlm baselines, see |----------|------|--------|-------------| | Mac Studio M1 Ultra 128GB | [model_tests_m1ultra.md](model_tests_m1ultra.md) | Active | 2026-06-15 | | MacBook Pro M5 Max 128GB | [model_tests_m5max.md](model_tests_m5max.md) | Active | 2026-06-15 | -| NVIDIA GB10 (DGX Spark) | [model_tests_gb10.md](model_tests_gb10.md) | Active | 2026-05-28 | +| NVIDIA GB10 (DGX Spark) | [model_tests_gb10.md](model_tests_gb10.md) | Active | 2026-06-17 | ## Benchmark CSVs @@ -36,6 +36,8 @@ Current source-of-truth data lives in `benchmarks/`: | `metal_m1ultra_vlm_2026-05-19.csv` | M1 Ultra | 2026-05-19 (mlxcel 0.0.28, MLX commit 84961223; >65GB skipped) | VLM | | `pylm_m1ultra_2026-05-19.csv` | M1 Ultra | 2026-05-19 (mlx-lm 0.31.3 baseline, https://github.com/ml-explore/mlx-lm @ `df1d3f3`; >65GB skipped) | Text | | `pylm_m1ultra_vlm_2026-05-19.csv` | M1 Ultra | 2026-05-19 (mlx-vlm baseline, https://github.com/Blaizzy/mlx-vlm @ `d85ca4d`; >65GB skipped) | VLM | +| `cuda_gb10_2026-06-17.csv` | GB10 | 2026-06-17 (mlxcel 0.3.1 [CSV relabeled; Cargo.toml 0.3.0 until release], MLX pin a6ec7123, CUDA 13.0 / SM 12.1, post-#319 CUDA fused decode-MoE; full text re-benchmark, 148 models, 133 pass / 14 fail / 1 OOM-skip) | Text | +| `cuda_gb10_vlm_2026-06-17.csv` | GB10 | 2026-06-17 (mlxcel 0.3.1; full VLM re-benchmark, 53 measured image rows) | VLM | | `cuda_gb10_2026-05-28.csv` | GB10 | 2026-05-28 (full text re-benchmark, mlxcel 0.1.0, MLX commit 84961223, warm same-process harness `c9a77f2`, `--cooldown 0`; 109 models, 8 fail/skip) | Text | | `cuda_gb10_vlm_2026-05-28.csv` | GB10 | 2026-05-28 (full VLM re-benchmark, mlxcel 0.1.0; 38 measured VLM rows, 0 image-path failures) | VLM | | `cuda_gb10_2026-05-19.csv` | GB10 | 2026-05-19 (mlxcel 0.0.27, MLX 0.31.2) | Text | @@ -49,40 +51,41 @@ The table below summarizes the current cross-hardware decode readings for select | Model | Params | M1 Ultra | M5 Max | GB10 | |-------|--------|----------|--------|------| -| SmolLM-135M | 135M | 374.92 | 916.80 | 643.04 | -| ERNIE-4.5-0.3B | 300M | 495.71 | 1072.92 | 682.24 | -| Qwen2.5-0.5B (4bit) | 500M | 343.91 | 678.95 | 502.51 | -| Llama-3.2-1B | 1B | 364.36 | 552.96 | 253.63 | -| Qwen3-0.6B | 600M | 275.55 | 565.88 | 317.75* | -| StableLM-1.6B | 1.6B | 270.88 | 424.38 | 197.05 | -| Gemma-3-1B | 1B | 229.70 | 396.72 | 256.48 | -| EXAONE-3.5-2.4B | 2.4B | 197.73 | 287.70 | 146.48 | -| SmolLM3-3B | 3B | 126.29 | 232.99 | 100.66 | -| Nemotron-H-30B | 30B | 91.54 | 176.10 | 32.92 | -| Qwen3-MoE-30B | 30B | 83.75 | 175.63 | 57.49 | -| Llama-3.1-8B | 8B | 107.89 | 116.61 | 49.15 | -| Qwen2.5-7B | 7B | 111.50 | 126.26 | 53.73 | -| Mixtral-8x7B | 47B | 54.25 | 65.37 | 28.00 | -| GPT-OSS-120B | 120B (MoE) | 58.41 | 113.91 | 50.63 | -| Solar-Open-100B | 100B (MoE) | 32.96 | 65.39 | 18.52 | - -*Qwen3-0.6B on GB10 produced only 9 tokens before EOS at 2026-05-28; the 317.75 tok/s decode rate is from that short window and is not directly comparable to full-length runs. +| SmolLM-135M | 135M | 374.92 | 916.80 | 652.08 | +| ERNIE-4.5-0.3B | 300M | 495.71 | 1072.92 | 654.97 | +| Qwen2.5-0.5B (4bit) | 500M | 343.91 | 678.95 | 485.68 | +| Llama-3.2-1B | 1B | 364.36 | 552.96 | 260.32 | +| Qwen3-0.6B | 600M | 275.55 | 565.88 | 294.34* | +| StableLM-1.6B | 1.6B | 270.88 | 424.38 | 198.28 | +| Gemma-3-1B | 1B | 229.70 | 396.72 | 241.85 | +| EXAONE-3.5-2.4B | 2.4B | 197.73 | 287.70 | 136.93 | +| SmolLM3-3B | 3B | 126.29 | 232.99 | 104.11 | +| Nemotron-H-30B | 30B | 91.54 | 176.10 | 40.32 | +| Qwen3-MoE-30B | 30B | 83.75 | 175.63 | 89.84† | +| Llama-3.1-8B | 8B | 107.89 | 116.61 | 49.10 | +| Qwen2.5-7B | 7B | 111.50 | 126.26 | 53.16 | +| Mixtral-8x7B | 47B | 54.25 | 65.37 | 27.92 | +| GPT-OSS-120B | 120B (MoE) | 58.41 | 113.91 | 50.48 | +| Solar-Open-100B | 100B (MoE) | 32.96 | 65.39 | 18.37 | + +*Qwen3-0.6B on GB10 again stopped at 9 tokens before EOS (2026-06-17); the 294.34 tok/s figure is from that short window and is not directly comparable to full-length runs. +†Qwen3-MoE-30B (`qwen3-moe-4bit`) **failed** on GB10 at 0.3.0 (Metal-only fused-MoE kernel aborted on CUDA); the CUDA fused decode-MoE kernel (#319) restores it at 0.3.1, and at 89.84 tok/s it now edges past M1 Ultra (83.75). M1 Ultra column is from 2026-06-15 with mlxcel 0.2.1 / MLX pin commit `a6ec712` (0.32.0-dev) / no cooldown, using the `mlxcel-bench-decode` same-process harness (post #289 bf16-scale fix and #291 quantized-embedding fix). M5 Max column is from the 2026-06-15 full re-sweep with mlxcel 0.2.1 / MLX pin `a6ec7123` / same-process `mlxcel-bench-decode` harness (bare run). -GB10 column is from 2026-05-28 with mlxcel 0.1.0 / MLX pin `84961223` / `--cooldown 0`, using the `mlxcel-bench-decode` same-process warm harness (PR `c9a77f2`). -Both Apple Silicon columns now share mlxcel 0.2.1 and the same MLX pin `a6ec712`, so the gap reflects hardware delta. M5 Max stays roughly 1.76x faster than M1 Ultra on the selected 16 rows (avg ~1.76x, median ~1.88x). The largest MoE rows show the M5 Max advantage: qwen3-moe-30b runs at 175.63 vs 83.75 tok/s (2.10x), gpt-oss-120b at 113.91 vs 58.41 (1.95x), and solar-open-100b at 65.39 vs 32.96 (1.98x). The GB10 column still predates the MLX bump (0.1.0) and is pending a 0.2.1 refresh. +GB10 column is from 2026-06-17 with mlxcel 0.3.1 / MLX pin `a6ec7123` / CUDA 13.0 (SM 12.1) / `--cooldown 0`, using the `mlxcel-bench-decode` same-process warm harness. 0.3.1 adds the CUDA fused decode-MoE kernel (#319): the Qwen MoE rows that aborted at 0.3.0 now run, and `qwen3-moe-30b` (89.84) edges past M1 Ultra (83.75). (CSV `mlx_version` relabeled to 0.3.1; the Cargo.toml bump lands at release.) +Both Apple Silicon columns now share mlxcel 0.2.1 and the same MLX pin `a6ec712`, so the gap reflects hardware delta. M5 Max stays roughly 1.76x faster than M1 Ultra on the selected 16 rows (avg ~1.76x, median ~1.88x). The largest MoE rows show the M5 Max advantage: qwen3-moe-30b runs at 175.63 vs 83.75 tok/s (2.10x), gpt-oss-120b at 113.91 vs 58.41 (1.95x), and solar-open-100b at 65.39 vs 32.96 (1.98x). The GB10 column is now on mlxcel 0.3.1 with the CUDA fused decode-MoE kernel (#319); the Qwen MoE rows that failed at 0.3.0 run on CUDA at 0.3.1 and can exceed M1 Ultra (qwen3-moe-30b 89.84 vs 83.75). For Qwen2.5-0.5B the 4-bit row is the directly comparable cross-hardware figure; on M1 Ultra `qwen2.5-0.5b-bf16` now runs after the #289 fix (298.92 tok/s), and the bf16 variant runs on M5 Max at 404.68 tok/s. -## Overall Status (mlxcel 0.2.1 on M1 Ultra and M5 Max, 0.1.0 on GB10) +## Overall Status (mlxcel 0.2.1 on M1 Ultra and M5 Max, 0.3.1 on GB10) | Metric | Count | |--------|-------| | Supported model architectures | 89+ ModelType variants | | Text models tested (M1 Ultra, 2026-06-15) | 136 pass, 2 partial, 4 fail, 9 skip/non-standalone (151 dirs; adds apertus, seed-oss, dots.llm1, granite family, lfm2, plamo-2, falcon-h1, BitNet; diffusiongemma loads via #291) | | Text models tested (M5 Max, 2026-06-15) | 131 pass, 5 partial, 14 fail/skip (0.2.1 full sweep; post-sweep: qwen2.5-vl-3b-4bit fixed by re-download, oversized bf16 hunyuan dropped; neither a code regression) | -| Text models tested (GB10, 2026-05-28) | 101 pass, 8 fail/skip (109 total) | -| VLM models tested (GB10, 2026-05-28) | 38 pass, 0 image-path fail (38 measured) | +| Text models tested (GB10, 2026-06-17) | 133 pass, 14 fail, 1 OOM-skip (148 total; 0.3.1 with the CUDA fused decode-MoE kernel #319 — 9 MoE models flipped FAIL→pass vs 0.3.0: 5 Qwen MoE + gemma-4-26b-a4b x2 + dots.llm1 + diffusiongemma) | +| VLM models tested (GB10, 2026-06-17) | 53 measured image rows (0.3.1) | | VLM models tested (M5 Max, 2026-06-15) | 54 valid VLM rows (0.2.1 full VLM re-sweep; adds qwen3-vl-4b/8b, minicpm-v-4.6-bf16, nemotron-omni, youtu-vl; qwen2.5-vl-3b-4bit restored after re-download) | | VLM models tested (M1 Ultra, 2026-06-15) | 55 measured VLM rows (53 pass + 2 partial) | | Beating mlx-lm on M1 Ultra (text, >=100%) | 24/74 (32%, 6-15 vs pinned 5-19 baseline) | diff --git a/docs/benchmark_results/model_tests_gb10.md b/docs/benchmark_results/model_tests_gb10.md index 00209cee5..e4be32142 100644 --- a/docs/benchmark_results/model_tests_gb10.md +++ b/docs/benchmark_results/model_tests_gb10.md @@ -1,268 +1,369 @@ # Model Compatibility & Performance Tests (NVIDIA GB10) -Compatibility and performance testing for mlxcel models on **NVIDIA GB10 (DGX Spark)**, running the CUDA backend. +Compatibility and decode/prefill performance for mlxcel models on **NVIDIA GB10 (DGX Spark)** running the CUDA backend. This is the **0.3.1** benchmark, run on the merged CUDA fused decode-MoE kernel (#319). ## Test Environment | Item | Value | |------|-------| -| **Hardware** | NVIDIA GB10 (DGX Spark), 122 GB unified memory, ~273 GB/s LPDDR5x | +| **Hardware** | NVIDIA GB10 (DGX Spark), 122 GB unified LPDDR5x | | **OS** | Linux (aarch64), kernel 6.17 | -| **Backend** | CUDA 13.0 | -| **mlxcel version** | 0.1.0 | -| **MLX version** | pinned commit `84961223` (via mlxcel-core; CSV `mlx_version` field records 0.31.2) | -| **Harness** | same-process `mlxcel-bench-decode`, warm prefill (PR `c9a77f2`), `--cooldown 0` | +| **Backend** | CUDA 13.0 (driver via `libcuda.so.1`, cuDNN 9) | +| **mlxcel version** | 0.3.1 | +| **MLX pin** | mlxcel-core commit `a6ec7123` | +| **CUDA build** | `MLX_CUDA_ARCHITECTURES=121 cargo build --release --features cuda` (GB10 = SM 12.1) | +| **Harness** | same-process `mlxcel-bench-decode`, warm prefill, pre-warm on, `--cooldown 0` | | **Test Prompt** | "Hello, how are you today?" (text) / "What is in this image?" (VLM) | | **Max Tokens** | 100 | -| **Test Date** | 2026-05-28 | -| **CSV** | `benchmarks/cuda_gb10_2026-05-28.csv`, `benchmarks/cuda_gb10_vlm_2026-05-28.csv` | +| **Test Date** | 2026-06-17 | +| **CSVs** | `benchmarks/cuda_gb10_2026-06-17.csv`, `benchmarks/cuda_gb10_vlm_2026-06-17.csv` | + +> Version note: this is the 0.3.1 benchmark, run on the post-#319 code. The `Cargo.toml` bump to 0.3.1 happens at release, so the CSVs were stamped 0.3.0 by the harness and relabeled to 0.3.1 to match the line they measure. +> +> Build note: a default `cargo build --release` on Linux produces a CPU-only binary (default features are `surgery` only), which silently runs MLX on the Grace CPU at ~0.4 tok/s. The GB10 benchmark must be built with `--features cuda`. +> +> The CUDA fused decode-MoE kernel (#319) landed for 0.3.1: nine MoE models that aborted at 0.3.0 on the Metal-only kernel (`[metal_kernel] No Metal back-end`) now run on the CUDA fused path, faster than `gather_qmm`. See the Summary and `fused-moe-decode-kernel-design.md`. ## Legend -- ✅ Pass: model loads and produces output within the configured token budget +- ✅ Pass: model loads and produces tokens within the configured budget - ❌ Fail: warmup/bench failure, OOM skip, or 0 tokens generated +Prefill/Decode are the measured-pass figures from `mlxcel-bench-decode`. Notes record an early-EOS token count (when a model stopped before 100) or the failure cause. Models are grouped by architecture family; VLM-capable models appear once under text (text-prompt pass) and again in the image-input table at the end. + ## Basic Transformers -| Model | Test Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | -|-------|------------|--------|-----------------|----------------|-------| -| llama-3.2-1b-4bit | Llama-3.2-1B-4bit | ✅ | 6858.95 | 253.63 | 31 tok | -| llama-3.1-8b-4bit | Llama-3.1-8B-Instruct-4bit | ✅ | 1361.89 | 49.15 | | -| llama-3.1-8b-bf16 | Llama-3.1-8B-Instruct (bf16) | ✅ | 1208.56 | 14.81 | 87 tok | -| phi-2-4bit | phi-2-hf-4bit | ✅ | 135.15 | 36.49 | 1 tok | -| phi-3-mini-4bit | Phi-3-mini-4k-instruct-4bit | ✅ | 280.73 | 93.19 | 25 tok | -| phi-3.5-mini-4bit | Phi-3.5-mini-instruct-4bit | ✅ | 290.58 | 92.50 | 40 tok | -| phi-4-4bit | Phi-4-4bit | ✅ | 161.06 | 27.52 | | -| qwen2-0.5b | Qwen2.5-0.5B (bf16) | ✅ | 3870.91 | 496.44 | | -| qwen2.5-0.5b-4bit | Qwen2.5-0.5B-Instruct-4bit | ✅ | 3705.50 | 502.51 | | -| qwen2.5-0.5b-bf16 | Qwen2.5-0.5B (bf16) | ✅ | 3136.60 | 202.87 | | -| qwen2.5-7b | Qwen2.5-7B (bf16) | ✅ | 634.86 | 54.07 | | -| qwen2.5-7b-4bit | Qwen2.5-7B-Instruct-4bit | ✅ | 617.25 | 53.73 | | -| qwen2.5-7b-8bit | Qwen2.5-7B-8bit | ✅ | 684.10 | 29.98 | | -| qwen3-0.6b | Qwen3-0.6B (bf16) | ✅ | 2021.77 | 317.75 | 9 tok (EOS) | -| qwen3-0.6b-4bit | Qwen3-0.6B-4bit | ✅ | 1956.00 | 314.62 | 9 tok (EOS) | -| qwen3-1.7b-4bit | Qwen3-1.7B-4bit | ✅ | 1134.38 | 167.90 | 14 tok | -| qwen3-4b-4bit | Qwen3-4B-4bit | ✅ | 488.30 | 81.37 | 36 tok | -| qwen3-8b-4bit | Qwen3-8B-4bit | ✅ | 252.73 | 48.71 | 33 tok | -| smollm-135m-4bit | SmolLM-135M-Instruct-4bit | ✅ | 3001.35 | 643.04 | | -| smollm3-3b-4bit | SmolLM3-3B-4bit | ✅ | 1628.18 | 100.66 | 18 tok | -| stablelm-1.6b-4bit | stablelm-2-1_6b-chat-4bit | ✅ | 1546.33 | 197.05 | | -| starcoder2-3b-4bit | starcoder2-3b-4bit | ✅ | 220.65 | 102.42 | | -| olmo-1b-4bit | OLMo-1B-hf-4bit | ✅ | 262.62 | 98.26 | | -| olmo2-7b-4bit | OLMo2-7B-4bit | ✅ | 316.99 | 53.17 | 27 tok | -| olmo3-32b-4bit | OLMo3-32B-4bit | ✅ | 309.94 | 11.70 | | -| minicpm-2b-4bit | MiniCPM-2B-sft-bf16-4bit | ✅ | 434.56 | 122.27 | | -| mimo-7b-4bit | MiMo-7B-RL-4bit | ✅ | 358.62 | 53.33 | | +| Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | +|-------|--------|-----------------|----------------|-------| +| apertus-8b-instruct-2509-4bit | ✅ | 494.90 | 42.15 | 31 tok | +| llama-3.1-8b-4bit | ✅ | 1294.78 | 49.10 | | +| llama-3.1-8b-bf16 | ✅ | 1209.77 | 14.84 | 87 tok | +| llama-3.2-1b-4bit | ✅ | 7793.45 | 260.32 | | +| meta-llama-3.1-8b-instruct-4bit | ✅ | 1320.16 | 48.93 | | +| mimo-7b-4bit | ✅ | 355.38 | 52.37 | | +| minicpm-2b-4bit | ✅ | 476.85 | 120.88 | | +| olmo-1b-4bit | ✅ | 261.22 | 98.90 | | +| olmo2-7b-4bit | ✅ | 311.46 | 53.17 | 27 tok | +| olmo3-32b-4bit | ✅ | 306.34 | 11.65 | | +| phi-2-4bit | ✅ | 132.96 | 35.31 | 1 tok | +| phi-3.5-mini-4bit | ✅ | 290.51 | 91.35 | 40 tok | +| phi-3-mini-4bit | ✅ | 284.33 | 92.24 | 22 tok | +| phi-4-4bit | ✅ | 165.50 | 27.19 | | +| qwen2-0.5b | ✅ | 4075.86 | 479.62 | | +| qwen2.5-0.5b-4bit | ✅ | 3784.53 | 485.68 | | +| qwen2.5-0.5b-bf16 | ✅ | 3416.94 | 199.89 | | +| qwen2.5-1.5b-4bit | ✅ | 1101.93 | 200.90 | | +| qwen2.5-1.5b-instruct-4bit | ✅ | 1538.21 | 201.59 | | +| qwen2.5-7b | ✅ | 594.78 | 53.60 | | +| qwen2.5-7b-4bit | ✅ | 608.73 | 53.16 | | +| qwen2.5-7b-8bit | ✅ | 663.57 | 30.18 | | +| qwen2.5-7b-instruct-4bit | ✅ | 617.78 | 54.28 | | +| qwen3-0.6b | ✅ | 1764.03 | 294.34 | 9 tok | +| qwen3-0.6b-4bit | ✅ | 1592.92 | 290.52 | 9 tok | +| qwen3-1.7b-4bit | ✅ | 1127.29 | 170.05 | 14 tok | +| qwen3-4b-4bit | ✅ | 532.50 | 80.55 | 36 tok | +| qwen3-8b-4bit | ✅ | 236.45 | 47.55 | 33 tok | +| smollm-135m-4bit | ✅ | 2429.28 | 652.08 | | +| smollm3-3b-4bit | ✅ | 1645.91 | 104.11 | 51 tok | +| stablelm-1.6b-4bit | ✅ | 1678.51 | 198.28 | | +| starcoder2-3b-4bit | ✅ | 215.17 | 101.82 | | ## Gemma Family -| Model | Test Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | -|-------|------------|--------|-----------------|----------------|-------| -| gemma-2b-4bit | gemma-2b-it-4bit | ✅ | 601.35 | 100.06 | 41 tok | -| gemma2-2b-4bit | gemma-2-2b-it-4bit | ✅ | 665.31 | 117.38 | 27 tok | -| gemma3-1b-4bit | gemma-3-1b-it-4bit | ✅ | 977.29 | 256.48 | 34 tok | -| gemma3-4b-4bit | gemma-3-4b-it-4bit | ✅ | 392.79 | 80.17 | 72 tok | -| gemma3n-e2b-4bit | gemma-3n-E2B-it-4bit | ✅ | 415.19 | 81.83 | 68 tok | -| gemma3n-e4b-4bit | gemma-3n-E4B-it-4bit | ✅ | 260.59 | 53.53 | 74 tok | -| gemma3n-e4b-bf16 | gemma-3n-E4B-it (bf16) | ✅ | 273.15 | 21.61 | 69 tok | - -### Gemma 4 - -| Model | Test Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | -|-------|------------|--------|-----------------|----------------|-------| -| gemma-4-e2b-it-4bit | Gemma-4-E2B-it-4bit | ✅ | 707.60 | 98.70 | 28 tok | -| gemma-4-e2b-it-8bit | Gemma-4-E2B-it-8bit | ✅ | 522.63 | 58.24 | | -| gemma-4-e4b-it-4bit | Gemma-4-E4B-it-4bit | ✅ | 325.35 | 47.58 | 33 tok | -| gemma-4-e4b-it-8bit | Gemma-4-E4B-it-8bit | ✅ | 272.71 | 27.22 | 33 tok | -| gemma-4-26b-a4b-it-4bit | Gemma-4-26B-A4B-it-4bit | ❌ | - | FAIL | warmup failure | -| gemma-4-31b-4bit | Gemma-4-31B-4bit | ✅ | 23.32 | 8.79 | | -| gemma-4-31b-it-4bit | Gemma-4-31B-it-4bit | ✅ | 48.97 | 8.06 | 26 tok | -| Gemma-4-31b-it-nvfp4 | Gemma-4-31B-it (NVFP4) | ✅ | 16.52 | 0.90 | 26 tok | +| Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | +|-------|--------|-----------------|----------------|-------| +| gemma2-2b-4bit | ✅ | 639.86 | 109.55 | 27 tok | +| gemma-2b-4bit | ✅ | 583.56 | 94.01 | 49 tok | +| gemma3-1b-4bit | ✅ | 1252.95 | 241.85 | 34 tok | +| gemma3-4b-4bit | ✅ | 410.60 | 76.30 | 84 tok | +| gemma-3-4b-it-4bit | ✅ | 411.18 | 76.11 | 84 tok | +| gemma3n-e2b-4bit | ✅ | 428.25 | 79.96 | 72 tok | +| gemma3n-e4b-4bit | ✅ | 281.38 | 53.57 | 74 tok | +| gemma3n-e4b-bf16 | ✅ | 275.17 | 21.35 | 69 tok | + +## Gemma 4 + +| Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | +|-------|--------|-----------------|----------------|-------| +| diffusiongemma-26b-a4b-it-4bit | ✅ | 117.60 | 37.68 | 27 tok | +| gemma-4-12b-it-4bit | ✅ | 202.66 | 14.70 | | +| gemma-4-12b-it-assistant-4bit | ❌ | - | - | MTP/DFlash drafter (needs a target; not standalone) | +| gemma-4-26b-a4b-it-4bit | ✅ | 130.81 | 58.59 | 26 tok | +| gemma-4-26b-a4b-it-qat-4bit | ✅ | 136.06 | 50.33 | 26 tok | +| gemma-4-31b-4bit | ✅ | 30.58 | 8.93 | | +| gemma-4-31b-it-4bit | ✅ | 70.25 | 8.53 | 26 tok | +| gemma-4-31b-it-assistant-bf16 | ❌ | - | - | MTP/DFlash drafter (needs a target; not standalone) | +| gemma-4-31b-it-nvfp4 | ✅ | 16.26 | 0.89 | 26 tok | +| gemma-4-31b-it-qat-4bit | ✅ | 80.51 | 5.59 | 26 tok | +| gemma-4-e2b-it-4bit | ✅ | 761.69 | 104.25 | | +| gemma-4-e2b-it-8bit | ✅ | 608.29 | 58.43 | | +| gemma-4-e2b-it-qat-4bit | ✅ | 503.36 | 66.39 | | +| gemma-4-e4b-it-4bit | ✅ | 355.22 | 49.39 | | +| gemma-4-e4b-it-8bit | ✅ | 292.35 | 27.36 | 45 tok | +| gemma-4-e4b-it-qat-4bit | ✅ | 273.71 | 33.67 | 33 tok | ## EXAONE -| Model | Test Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | -|-------|------------|--------|-----------------|----------------|-------| -| exaone-3.5-2.4b-4bit | EXAONE-3.5-2.4B-Instruct-4bit | ✅ | 1391.24 | 146.48 | | -| exaone4-1.2b-4bit | exaone-4.0-1.2b-4bit | ✅ | 1136.29 | 225.62 | 53 tok | +| Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | +|-------|--------|-----------------|----------------|-------| +| exaone-3.5-2.4b-4bit | ✅ | 1390.18 | 136.93 | | +| exaone4-1.2b-4bit | ✅ | 1456.06 | 214.72 | 53 tok | ## Cohere / Command R -| Model | Test Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | -|-------|------------|--------|-----------------|----------------|-------| -| command-r7b-4bit | c4ai-command-r7b-4bit | ✅ | 124.23 | 52.12 | | -| aya-expanse-8b-4bit | aya-expanse-8b-4bit | ✅ | 159.55 | 52.89 | | +| Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | +|-------|--------|-----------------|----------------|-------| +| aya-expanse-8b-4bit | ✅ | 178.81 | 47.88 | | +| command-r7b-4bit | ✅ | 127.27 | 47.78 | | + +## Granite (IBM) + +| Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | +|-------|--------|-----------------|----------------|-------| +| granite-3.3-2b-instruct-4bit | ✅ | 1801.88 | 113.75 | 28 tok | +| granite-4.0-h-350m-4bit | ✅ | 1714.62 | 64.00 | 19 tok | +| granite-4.0-h-tiny-4bit | ✅ | 256.82 | 33.84 | 53 tok | +| granite-4.1-3b-4bit | ✅ | 328.96 | 73.74 | 7 tok | +| granite-4.1-8b-4bit | ✅ | 166.20 | 18.45 | 1 tok | +| granite-speech-4.1-2b-nar-mlx | ❌ | - | - | not a standalone text-gen model | ## MoE (Mixture of Experts) -| Model | Test Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | -|-------|------------|--------|-----------------|----------------|-------| -| mixtral-8x7b-4bit | Mixtral-8x7B-Instruct-v0.1-4bit | ✅ | 12.60 | 28.00 | 73 tok | -| qwen1.5-moe-a2.7b-4bit | Qwen1.5-MoE-A2.7B-Chat-4bit | ✅ | 248.96 | 112.09 | | -| qwen3-moe-4bit | Qwen3-30B-A3B-4bit | ✅ | 133.23 | 57.49 | 33 tok | -| qwen3-30b-a3b-4bit | Qwen3-30B-A3B-4bit | ✅ | 134.11 | 53.55 | 34 tok | -| phi-3.5-moe-4bit | Phi-3.5-MoE-instruct-4bit | ✅ | 28.99 | 51.35 | | -| minimax-m2-3bit | MiniMax-M2-3bit | ✅ | 26.72 | 21.85 | | -| gpt-oss-20b-mxfp4 | gpt-oss-20b-MXFP4 | ✅ | 126.41 | 77.94 | | -| gpt-oss-120b-4bit | gpt-oss-120b-4bit | ✅ | 54.57 | 50.63 | 73 tok | -| deepseek-v2-lite-4bit | DeepSeek-V2-Lite-Chat-4bit | ✅ | 156.31 | 99.07 | | -| deepseek-v3-4bit | DeepSeek-V3-0324-4bit | ❌ | - | FAIL | warmup failure | -| llama-4-scout-17b-4bit | Llama-4-Scout-17B-16E-4bit | ✅ | 27.67 | 20.88 | | +| Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | +|-------|--------|-----------------|----------------|-------| +| deepseek-v2-lite-4bit | ✅ | 160.07 | 96.81 | | +| deepseek-v3-4bit | ❌ | - | - | warmup failure (also failed at 0.1.0) | +| dots.llm1.inst-mixed-4-6bit | ✅ | 25.42 | 22.04 | 39 tok | +| gpt-oss-120b-4bit | ✅ | 57.75 | 50.48 | 82 tok | +| gpt-oss-20b-mxfp4 | ✅ | 126.16 | 77.25 | | +| lfm2-8b-a1b-4bit | ✅ | 145.46 | 157.73 | 37 tok | +| llama-4-scout-17b-4bit | ✅ | 28.20 | 20.94 | | +| minimax-m2-3bit | ✅ | 26.95 | 22.03 | | +| mixtral-8x7b-4bit | ✅ | 12.46 | 27.92 | 73 tok | +| phi-3.5-moe-4bit | ✅ | 28.87 | 50.13 | | +| qwen1.5-moe-a2.7b-4bit | ✅ | 261.24 | 125.52 | | +| qwen3-30b-a3b-4bit | ✅ | 133.40 | 90.70 | 34 tok | +| qwen3-moe-4bit | ✅ | 135.30 | 89.84 | 34 tok | ## MLA (Multi-head Latent Attention) -| Model | Test Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | -|-------|------------|--------|-----------------|----------------|-------| -| minicpm3-4b-4bit | MiniCPM3-4B-4bit | ✅ | 282.58 | 57.55 | | +| Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | +|-------|--------|-----------------|----------------|-------| +| minicpm3-4b-4bit | ✅ | 275.48 | 56.16 | | ## DeepSeek Family -| Model | Test Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | -|-------|------------|--------|-----------------|----------------|-------| -| deepseek-coder-1.3b-4bit | deepseek-coder-1.3b-4bit | ✅ | 4655.97 | 92.61 | | -| deepseek-r1-distill-7b-4bit | DeepSeek-R1-Distill-Qwen-7B-4bit | ✅ | 210.07 | 58.55 | | +| Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | +|-------|--------|-----------------|----------------|-------| +| deepseek-coder-1.3b-4bit | ✅ | 4540.35 | 86.86 | | +| deepseek-r1-distill-7b-4bit | ✅ | 209.88 | 54.56 | | ## Nemotron Family -| Model | Test Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | -|-------|------------|--------|-----------------|----------------|-------| -| nemotron-h-30b-4bit | Nemotron-H-30B-4bit | ✅ | 108.15 | 32.92 | 46 tok | -| nemotron-nas-30b-4bit | Nemotron-NAS-30B-A3B-4bit | ✅ | 105.00 | 32.98 | 46 tok | +| Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | +|-------|--------|-----------------|----------------|-------| +| nemotron-3-nano-omni-30b-a3b-reasoning-4bit | ✅ | 117.76 | 38.45 | 20 tok | +| nemotron-h-30b-4bit | ✅ | 116.21 | 40.32 | 46 tok | +| nemotron-nas-30b-4bit | ✅ | 113.03 | 37.33 | 46 tok | -## SSM / Mamba / Hybrid Models +## SSM / Mamba / Hybrid -| Model | Test Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | -|-------|------------|--------|-----------------|----------------|-------| -| falcon-mamba-7b-4bit | Falcon-Mamba-7B-4bit | ✅ | 83.89 | 22.09 | 2 tok | -| mamba2-1.3b-4bit | mamba2-1.3b-4bit | ✅ | 277.75 | 80.50 | | -| jamba-v0.1-4bit | Jamba-v0.1-4bit | ✅ | 529.88 | 85.42 | | +| Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | +|-------|--------|-----------------|----------------|-------| +| falcon-h1-tiny-90m-instruct-4bit | ✅ | 1294.33 | 102.99 | | +| falcon-mamba-7b-4bit | ✅ | 100.28 | 22.26 | 2 tok | +| jamba-v0.1-4bit | ✅ | 541.13 | 85.48 | | +| lfm2-350m-8bit | ✅ | 2768.23 | 409.01 | 13 tok | +| mamba2-130m | ✅ | 984.73 | 181.05 | | +| mamba2-1.3b-4bit | ✅ | 283.78 | 81.37 | | +| plamo-2-1b | ✅ | 189.89 | 34.36 | | ## Chinese / Asian Language Models -| Model | Test Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | -|-------|------------|--------|-----------------|----------------|-------| -| baichuan-m1-14b-4bit | Baichuan-M1-14B-Instruct-4bit | ✅ | 75.66 | 24.06 | 7 tok | -| glm4-flash-4bit | GLM-4-Flash-4bit | ✅ | 102.70 | 55.04 | | -| GLM-5.1-4bit | GLM-5.1-4bit | ❌ | - | FAIL | warmup failure | -| internlm2-7b-4bit | InternLM2-7B-4bit | ✅ | 387.86 | 50.26 | | -| internlm3-8b-4bit | internlm3-8b-instruct-4bit | ✅ | 530.75 | 43.89 | | -| ernie-4.5-0.3b-4bit | ERNIE-4.5-0.3B-Instruct-4bit | ✅ | 5403.22 | 682.24 | | -| hunyuan-13b | Hunyuan-Large (bf16, 13B) | ✅ | 19.31 | 15.15 | | -| hunyuan-4bit | Hunyuan-Large-Instruct-4bit | ✅ | 19.87 | 14.86 | | -| hunyuan-dense-4bit | Hunyuan-1.8B-Instruct-4bit | ✅ | 668.84 | 158.90 | 41 tok | -| hunyuan-1.8b-4bit | Hunyuan-1.8B-Instruct-4bit | ✅ | 732.00 | 157.78 | 41 tok | -| hunyuan-large-4bit | Hunyuan-Large-Instruct-4bit | ✅ | 20.04 | 15.10 | | -| hunyuan-moe-a13b-bf16 | Hunyuan-MoE-A13B (bf16) | ✅ | 19.00 | 14.79 | | +| Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | +|-------|--------|-----------------|----------------|-------| +| baichuan-m1-14b-4bit | ✅ | 74.32 | 22.36 | 7 tok | +| ernie-4.5-0.3b-4bit | ✅ | 6384.30 | 654.97 | | +| glm4-flash-4bit | ✅ | 106.65 | 53.33 | | +| glm-5.1-4bit | ❌ | - | - | warmup failure | +| glm-5-4bit | ❌ | - | - | warmup failure | +| hunyuan-13b | ✅ | 18.78 | 14.80 | | +| hunyuan-1.8b-4bit | ✅ | 685.98 | 154.72 | 41 tok | +| hunyuan-a13b-instruct-4bit | ✅ | 20.07 | 15.06 | | +| internlm2-7b-4bit | ✅ | 388.59 | 50.09 | | +| internlm3-8b-4bit | ✅ | 522.79 | 43.89 | | +| seed-oss-36b-instruct-4bit | ✅ | 78.84 | 10.23 | | ## Mistral Family -| Model | Test Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | -|-------|------------|--------|-----------------|----------------|-------| -| ministral-3b-4bit | Ministral-3B-Instruct-4bit | ✅ | 6316.20 | 101.17 | 34 tok | -| mistral-small-3.1-24b-4bit | mistral-small-3.1-24b-4bit | ✅ | 65.18 | 16.08 | | -| pixtral-12b | pixtral-12b (bf16) | ✅ | 36.69 | 33.09 | | -| pixtral-12b-4bit | pixtral-12b-4bit | ✅ | 36.00 | 33.06 | | +| Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | +|-------|--------|-----------------|----------------|-------| +| ministral-3b-4bit | ✅ | 6280.07 | 100.54 | 34 tok | +| mistral-small-3.1-24b-4bit | ✅ | 63.68 | 16.08 | | +| pixtral-12b | ✅ | 35.97 | 32.80 | | +| pixtral-12b-4bit | ✅ | 36.02 | 32.85 | | + +## BitNet + +| Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | +|-------|--------|-----------------|----------------|-------| +| bitnet-b1.58-2b-4t | ❌ | - | - | ternary BitNet; fails warmup on CUDA | +| bitnet-b1.58-2b-4t-4bit | ❌ | - | - | ternary BitNet; fails warmup on CUDA | ## VLM-capable Models (text-only pass) -| Model | Test Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | -|-------|------------|--------|-----------------|----------------|-------| -| aya-vision-8b | aya-vision-8b | ✅ | 117.37 | 52.16 | | -| bunny-llama3-8b-4bit | Bunny-Llama-3-8B-V-4bit | ✅ | 380.57 | 52.89 | 40 tok | -| llava-1.5-7b-4bit | llava-1.5-7b-4bit | ✅ | 207.24 | 56.02 | | -| llava-next-mistral-7b-4bit | llava-v1.6-mistral-7b-4bit | ✅ | 371.54 | 52.64 | | -| llava-interleave-qwen-0.5b-bf16 | llava-interleave-qwen-0.5b-bf16 | ✅ | 2448.92 | 212.61 | 49 tok | -| molmo2-4b | molmo2-4b | ✅ | 188.38 | 26.76 | 33 tok | -| molmo-7b | Molmo-7B | ✅ | 212.01 | 33.62 | 24 tok | -| paligemma2-3b-6bit | paligemma2-3b | ❌ | 164.80 | 0.00 | 0 tokens generated | -| phi-3.5-vision-4bit | Phi-3.5-vision-instruct-4bit | ✅ | 426.67 | 91.41 | 43 tok | -| internvl3-1b | InternVL3-1B | ✅ | 4041.24 | 479.58 | 37 tok | -| qwen2-vl-2b | Qwen2-VL-2B (bf16) | ✅ | 670.87 | 101.92 | 35 tok | -| qwen2-vl-2b-4bit | Qwen2-VL-2B-Instruct-4bit | ✅ | 683.57 | 101.28 | 35 tok | -| qwen2.5-vl-3b | Qwen2.5-VL-3B (bf16) | ❌ | - | FAIL | warmup failure | -| qwen2.5-vl-3b-4bit | Qwen2.5-VL-3B-Instruct-4bit | ❌ | - | FAIL | warmup failure | -| qwen3-vl-2b | Qwen3-VL-2B (bf16) | ✅ | 848.87 | 166.97 | 61 tok | -| qwen3-vl-2b-4bit | Qwen3-VL-2B-Instruct-4bit | ✅ | 754.03 | 165.01 | 33 tok | -| qwen3-vl-30b-a3b-4bit | Qwen3-VL-30B-A3B-4bit | ✅ | 126.26 | 56.10 | 34 tok | -| qwen3-vl-32b-4bit | Qwen3-VL-32B-4bit | ✅ | 86.60 | 10.92 | 37 tok | - -## Qwen3.5 / Qwen3-next (new architectures) - -| Model | Test Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | -|-------|------------|--------|-----------------|----------------|-------| -| qwen3.5-0.8b-4bit | Qwen3.5-0.8B-4bit | ✅ | 632.45 | 172.27 | 18 tok | -| qwen3.5-2b-4bit | Qwen3.5-2B-4bit | ✅ | 512.33 | 127.49 | 31 tok | -| qwen3.5-4b-4bit | Qwen3.5-4B-4bit | ✅ | 250.83 | 63.05 | 31 tok | -| qwen3.5-9b-4bit | Qwen3.5-9B-4bit | ✅ | 164.08 | 39.19 | 31 tok | -| qwen3.5-9b-bf16 | Qwen3.5-9B (bf16) | ✅ | 163.30 | 13.03 | 31 tok | -| qwen3.5-27b-4bit | Qwen3.5-27B-4bit | ✅ | 59.42 | 12.36 | 30 tok | -| qwen3.5-35b-a3b-4bit | Qwen3.5-35B-A3B-4bit | ✅ | 144.11 | 48.71 | 31 tok | -| Qwen3.5-397B-A17B-4bit | Qwen3.5-397B-A17B-4bit | ❌ | - | SKIP | OOM skip (capacity) | -| qwen3-next-480b-4bit | Qwen3-next-480B-4bit | ❌ | - | SKIP | OOM skip (capacity) | -| solar-open-100b-4bit | Solar-Open-100B-4bit | ✅ | 49.53 | 18.52 | | +| Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | +|-------|--------|-----------------|----------------|-------| +| aya-vision-8b | ✅ | 124.91 | 46.67 | 87 tok | +| bunny-llama3-8b-4bit | ✅ | 433.97 | 48.81 | 40 tok | +| docling-layout-heron-mlx-bf16 | ❌ | - | - | not a standalone text-gen model | +| internvl3-1b | ✅ | 4559.06 | 473.46 | 37 tok | +| llava-1.5-7b-4bit | ✅ | 205.07 | 55.06 | | +| llava-interleave-qwen-0.5b-bf16 | ✅ | 2531.16 | 206.58 | 49 tok | +| llava-next-mistral-7b-4bit | ✅ | 374.38 | 51.61 | | +| minicpm-v-4.6-bf16 | ✅ | 317.54 | 110.35 | | +| minicpm-v-4.6-mxfp4 | ❌ | - | - | mxfp4 warmup failure (bf16 variant works) | +| molmo2-4b | ✅ | 197.01 | 26.56 | 33 tok | +| molmo-7b | ✅ | 203.66 | 33.64 | 24 tok | +| paligemma2-3b-6bit | ❌ | 160.60 | 0.00 | 0 tokens generated | +| phi-3.5-vision-4bit | ✅ | 435.53 | 91.05 | 43 tok | +| qwen2.5-vl-3b | ❌ | - | - | warmup failure (also failed at 0.1.0) | +| qwen2.5-vl-3b-4bit | ✅ | 371.22 | 59.93 | 39 tok | +| qwen2-vl-2b | ✅ | 583.13 | 101.35 | 35 tok | +| qwen2-vl-2b-4bit | ✅ | 607.09 | 100.84 | 35 tok | +| qwen3-vl-2b | ✅ | 828.99 | 163.95 | 59 tok | +| qwen3-vl-2b-4bit | ✅ | 840.44 | 165.15 | 59 tok | +| qwen3-vl-30b-a3b-4bit | ✅ | 126.71 | 83.25 | 35 tok | +| qwen3-vl-32b-4bit | ✅ | 84.07 | 10.95 | 37 tok | +| qwen3-vl-4b-4bit | ✅ | 359.60 | 75.66 | 49 tok | +| qwen3-vl-4b-instruct-4bit | ✅ | 350.36 | 75.78 | 49 tok | +| qwen3-vl-8b-4bit | ✅ | 213.04 | 47.57 | 57 tok | +| qwen3-vl-8b-instruct-4bit | ✅ | 227.56 | 47.62 | 57 tok | +| youtu-vl-4b-instruct | ✅ | 134.27 | 21.94 | | + +## Qwen3.5 / Qwen3.6 / Qwen3-next + +| Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | +|-------|--------|-----------------|----------------|-------| +| qwen3.5-0.8b-4bit | ✅ | 640.75 | 174.95 | 18 tok | +| qwen3.5-0.8b-optiq-4bit | ✅ | 644.30 | 156.91 | 19 tok | +| qwen3.5-27b-4bit | ✅ | 59.38 | 12.21 | 30 tok | +| qwen3.5-27b-dflash | ❌ | - | - | MTP/DFlash drafter (needs a target; not standalone) | +| qwen3.5-2b-4bit | ✅ | 514.72 | 124.55 | 32 tok | +| qwen3.5-35b-a3b-4bit | ✅ | 144.28 | 64.26 | 31 tok | +| qwen3.5-4b-4bit | ✅ | 247.95 | 62.05 | 31 tok | +| qwen3.5-4b-dflash | ❌ | - | - | MTP/DFlash drafter (needs a target; not standalone) | +| qwen3.5-9b-4bit | ✅ | 172.83 | 38.92 | 31 tok | +| qwen3.5-9b-bf16 | ✅ | 169.27 | 13.11 | 31 tok | +| qwen3.6-35b-a3b-4bit | ✅ | 148.18 | 62.99 | 28 tok | +| qwen3-next-480b-4bit | ❌ | - | - | OOM skip (capacity, weights > mem budget) | + +## Solar + +| Model | Status | Prefill (tok/s) | Decode (tok/s) | Notes | +|-------|--------|-----------------|----------------|-------| +| solar-open-100b-4bit | ✅ | 46.76 | 18.37 | | ## VLM Benchmark (image input) -| Model | Test Model | Status | Generated Tokens | Prefill (tok/s) | Decode (tok/s) | -|-------|------------|--------|------------------|-----------------|----------------| -| aya-vision-8b | aya-vision-8b | ✅ | 100 | 673.71 | 38.94 | -| bunny-llama3-8b-4bit | Bunny-Llama-3-8B-V-4bit | ✅ | 37 | 1680.92 | 38.30 | -| gemma3-4b-4bit | gemma-3-4b-it-4bit | ✅ | 14 | 1012.14 | 69.83 | -| gemma3n-e2b-4bit | gemma-3n-E2B-it-4bit | ✅ | 29 | 2223.72 | 74.87 | -| gemma3n-e4b-4bit | gemma-3n-E4B-it-4bit | ✅ | 33 | 1422.82 | 51.44 | -| gemma3n-e4b-bf16 | gemma-3n-E4B-it (bf16) | ✅ | 24 | 2014.06 | 20.96 | -| gemma-4-31b-4bit | Gemma-4-31B-4bit | ✅ | 8 | 322.22 | 7.82 | -| gemma-4-31b-it-4bit | Gemma-4-31B-it-4bit | ✅ | 27 | 331.15 | 8.46 | -| gemma-4-e2b-it-4bit | Gemma-4-E2B-it-4bit | ✅ | 20 | 2510.79 | 95.14 | -| gemma-4-e2b-it-8bit | Gemma-4-E2B-it-8bit | ✅ | 7 | 2135.14 | 48.47 | -| gemma-4-e4b-it-4bit | Gemma-4-E4B-it-4bit | ✅ | 47 | 1321.27 | 47.15 | -| gemma-4-e4b-it-8bit | Gemma-4-E4B-it-8bit | ✅ | 11 | 1117.64 | 25.43 | -| llama-4-scout-17b-4bit | Llama-4-Scout-17B-16E-4bit | ✅ | 100 | 28.39 | 20.65 | -| llava-1.5-7b-4bit | llava-1.5-7b-4bit | ✅ | 100 | 1704.28 | 53.27 | -| llava-interleave-qwen-0.5b-bf16 | llava-interleave-qwen-0.5b-bf16 | ✅ | 32 | 19737.74 | 191.14 | -| llava-next-mistral-7b-4bit | llava-v1.6-mistral-7b-4bit | ✅ | 100 | 2170.28 | 52.26 | -| ministral-3b-4bit | Ministral-3B-Instruct-4bit | ✅ | 100 | 2529.43 | 90.70 | -| mistral-small-3.1-24b-4bit | mistral-small-3.1-24b-4bit | ✅ | 100 | 473.60 | 15.59 | -| molmo2-4b | molmo2-4b | ✅ | 46 | 762.92 | 26.60 | -| paligemma2-3b-6bit | paligemma2-3b | ✅ | 2 | 1846.20 | 42.61 | -| phi-3.5-vision-4bit | Phi-3.5-vision-instruct-4bit | ✅ | 19 | 1383.21 | 80.90 | -| pixtral-12b | pixtral-12b (bf16) | ✅ | 100 | 562.66 | 29.86 | -| pixtral-12b-4bit | pixtral-12b-4bit | ✅ | 100 | 749.72 | 30.63 | -| qwen2-vl-2b | Qwen2-VL-2B (bf16) | ✅ | 12 | 677.83 | 90.87 | -| qwen2-vl-2b-4bit | Qwen2-VL-2B-Instruct-4bit | ✅ | 12 | 696.50 | 92.67 | -| qwen3-vl-2b | Qwen3-VL-2B (bf16) | ✅ | 84 | 2192.64 | 131.85 | -| qwen3-vl-2b-4bit | Qwen3-VL-2B-Instruct-4bit | ✅ | 80 | 2260.71 | 132.73 | -| qwen3-vl-30b-a3b-4bit | Qwen3-VL-30B-A3B-4bit | ✅ | 72 | 184.85 | 45.15 | -| qwen3-vl-32b-4bit | Qwen3-VL-32B-4bit | ✅ | 55 | 158.46 | 10.40 | -| qwen3.5-27b-4bit | Qwen3.5-27B-4bit | ✅ | 100 | 103.03 | 12.75 | -| qwen3.5-35b-a3b-4bit | Qwen3.5-35B-A3B-4bit | ✅ | 100 | 179.45 | 45.23 | -| qwen3.5-9b-bf16 | Qwen3.5-9B (bf16) | ✅ | 100 | 347.94 | 13.58 | -| qwen3.5-2b-4bit | Qwen3.5-2B-4bit | ✅ | 47 | 719.30 | 123.77 | -| qwen3.5-4b-4bit | Qwen3.5-4B-4bit | ✅ | 49 | 427.82 | 59.77 | -| qwen3.5-9b-4bit | Qwen3.5-9B-4bit | ✅ | 100 | 306.12 | 39.09 | -| qwen3.5-0.8b-4bit | Qwen3.5-0.8B-4bit | ✅ | 100 | 1013.86 | 206.67 | -| internvl3-1b | InternVL3-1B | ✅ | 8 | 1898.25 | 406.33 | -| molmo-7b | Molmo-7B | ✅ | 2 | 1121.48 | 23.66 | +Models that accept image input and generated tokens under the `"What is in this image?"` prompt with `tests/fixtures/test_image.png`. + +| Model | Status | Generated Tokens | Prefill (tok/s) | Decode (tok/s) | +|-------|--------|------------------|-----------------|----------------| +| aya-vision-8b | ✅ | 100 | 632.49 | 45.34 | +| bunny-llama3-8b-4bit | ✅ | 37 | 1681.67 | 45.09 | +| gemma3-4b-4bit | ✅ | 23 | 1018.08 | 73.16 | +| gemma-3-4b-it-4bit | ✅ | 16 | 994.12 | 70.28 | +| gemma3n-e2b-4bit | ✅ | 25 | 2193.58 | 73.10 | +| gemma3n-e4b-4bit | ✅ | 33 | 1467.15 | 52.43 | +| gemma-4-12b-it-4bit | ✅ | 100 | 923.02 | 14.69 | +| gemma-4-26b-a4b-it-4bit | ✅ | 28 | 153.31 | 54.64 | +| gemma-4-26b-a4b-it-qat-4bit | ✅ | 30 | 149.84 | 48.93 | +| gemma-4-31b-4bit | ✅ | 8 | 317.69 | 7.82 | +| gemma-4-31b-it-4bit | ✅ | 25 | 277.27 | 8.26 | +| gemma-4-31b-it-qat-4bit | ✅ | 29 | 302.79 | 5.58 | +| gemma-4-e2b-it-4bit | ✅ | 100 | 2528.92 | 104.17 | +| gemma-4-e2b-it-8bit | ✅ | 100 | 2138.74 | 58.43 | +| gemma-4-e2b-it-qat-4bit | ✅ | 100 | 2235.54 | 66.43 | +| gemma-4-e4b-it-4bit | ✅ | 100 | 1407.62 | 49.50 | +| gemma-4-e4b-it-8bit | ✅ | 51 | 1138.82 | 26.97 | +| gemma-4-e4b-it-qat-4bit | ✅ | 32 | 1228.36 | 33.32 | +| internvl3-1b | ✅ | 8 | 1958.25 | 397.14 | +| llama-4-scout-17b-4bit | ✅ | 100 | 29.15 | 20.54 | +| llava-1.5-7b-4bit | ✅ | 100 | 1721.32 | 52.08 | +| llava-interleave-qwen-0.5b-bf16 | ✅ | 32 | 20633.47 | 188.13 | +| llava-next-mistral-7b-4bit | ✅ | 100 | 2159.71 | 51.07 | +| minicpm-v-4.6-bf16 | ✅ | 23 | 583.76 | 103.61 | +| ministral-3b-4bit | ✅ | 100 | 2524.57 | 89.35 | +| mistral-small-3.1-24b-4bit | ✅ | 100 | 504.81 | 15.33 | +| molmo2-4b | ✅ | 46 | 773.43 | 26.21 | +| molmo-7b | ✅ | 2 | 1095.89 | 23.81 | +| nemotron-3-nano-omni-30b-a3b-reasoning-4bit | ✅ | 6 | 147.03 | 32.86 | +| paligemma2-3b-6bit | ✅ | 2 | 1754.82 | 44.34 | +| phi-3.5-vision-4bit | ✅ | 19 | 1413.29 | 80.50 | +| pixtral-12b | ✅ | 100 | 782.19 | 30.29 | +| pixtral-12b-4bit | ✅ | 100 | 772.90 | 30.19 | +| qwen2.5-vl-3b-4bit | ✅ | 64 | 598.32 | 59.70 | +| qwen2-vl-2b | ✅ | 12 | 660.36 | 91.60 | +| qwen2-vl-2b-4bit | ✅ | 12 | 656.09 | 88.14 | +| qwen3.5-0.8b-4bit | ✅ | 100 | 963.74 | 202.03 | +| qwen3.5-27b-4bit | ✅ | 100 | 103.80 | 12.70 | +| qwen3.5-2b-4bit | ✅ | 47 | 737.26 | 127.63 | +| qwen3.5-35b-a3b-4bit | ✅ | 100 | 181.59 | 59.98 | +| qwen3.5-4b-4bit | ✅ | 49 | 425.56 | 62.14 | +| qwen3.5-9b-4bit | ✅ | 100 | 311.34 | 39.81 | +| qwen3.5-9b-bf16 | ✅ | 100 | 338.09 | 13.54 | +| qwen3.6-35b-a3b-4bit | ✅ | 100 | 175.03 | 62.35 | +| qwen3-vl-2b | ✅ | 52 | 2130.17 | 132.22 | +| qwen3-vl-2b-4bit | ✅ | 84 | 2134.98 | 132.22 | +| qwen3-vl-30b-a3b-4bit | ✅ | 73 | 200.25 | 63.56 | +| qwen3-vl-32b-4bit | ✅ | 51 | 150.99 | 10.37 | +| qwen3-vl-4b-4bit | ✅ | 37 | 1001.53 | 66.15 | +| qwen3-vl-4b-instruct-4bit | ✅ | 37 | 1002.80 | 65.02 | +| qwen3-vl-8b-4bit | ✅ | 30 | 596.15 | 40.21 | +| qwen3-vl-8b-instruct-4bit | ✅ | 30 | 577.42 | 40.34 | +| gemma3n-e4b-bf16 | ✅ | 24 | 1878.01 | 20.65 | --- ## Summary -**Test date**: 2026-05-28 | **Hardware**: NVIDIA GB10 (DGX Spark) | **mlxcel**: 0.1.0 | **MLX**: pin `84961223` +**Test date**: 2026-06-17 | **Hardware**: NVIDIA GB10 (DGX Spark) | **mlxcel**: 0.3.1 (CUDA 13.0, SM 12.1) | **MLX pin**: `a6ec7123` | Metric | Count | |--------|-------| -| **Total text models attempted** | 109 | -| **Pass (✅)** | 101 | -| **Fail / skip / 0-token (❌)** | 8 (5 fail, 2 OOM skip, 1 zero-token) | -| **VLM models measured (image)** | 38 | -| **VLM Pass (✅)** | 38 | -| **VLM image-path failures (❌, 0 tokens)** | 0 | - -The remaining VLM-CSV rows are text-only models that fail warmup under an image prompt; their text-suite results are in the per-family tables above. - -### Failing / skipped models - -- **Warmup/bench failures:** `deepseek-v3-4bit`, `gemma-4-26b-a4b-it-4bit`, `GLM-5.1-4bit`, `qwen2.5-vl-3b`, `qwen2.5-vl-3b-4bit` +| **Total text models attempted** | 148 | +| **Pass (✅)** | 133 | +| **Fail / 0-token (❌)** | 14 | +| **OOM-skipped (capacity)** | 1 | +| **VLM models measured (image input)** | 53 | + +### CUDA fused decode-MoE kernel (#319) + +The 0.3.1 line ported the fused decode-MoE kernel to CUDA (#319). Nine MoE models that aborted at 0.3.0 on the Metal-only kernel (`[metal_kernel] No Metal back-end`) now run on the CUDA fused path, faster than the `gather_qmm` fallback: + +| Model | 0.3.0 | 0.3.1 | +|-------|-------|-------| +| qwen3-moe-4bit (Qwen3-30B-A3B) | ❌ FAIL | 89.84 | +| qwen3-30b-a3b-4bit | ❌ FAIL | 90.70 | +| qwen3.5-35b-a3b-4bit | ❌ FAIL | 64.26 | +| qwen3.6-35b-a3b-4bit | ❌ FAIL | 62.99 | +| qwen1.5-moe-a2.7b-4bit | ❌ FAIL | 125.52 | +| gemma-4-26b-a4b-it-4bit | ❌ FAIL | 58.59 | +| gemma-4-26b-a4b-it-qat-4bit | ❌ FAIL | 50.33 | +| dots.llm1.inst-mixed-4-6bit | ❌ FAIL | 22.04 | +| diffusiongemma-26b-a4b-it-4bit | ❌ FAIL | 37.68 | + +`lfm2-8b-a1b-4bit` (141.8 → 157.7) and `qwen3-vl-30b-a3b-4bit` (57.1 → 83.3) were already passing on the legacy path and got faster on the fused path. Greedy output is byte-identical to `gather_qmm`. The CUDA fused crossover is much higher than Metal's (~13-14k vs ~4096), but the cap stays 4096 (see `fused-moe-decode-kernel-design.md`). + +### Failing / skipped models (by cause) + +- **BitNet (ternary; fails CUDA warmup):** `bitnet-b1.58-2b-4t`, `bitnet-b1.58-2b-4t-4bit` +- **Other model-specific failures:** `deepseek-v3-4bit`, `glm-5-4bit`, `glm-5.1-4bit` +- **Not standalone text-gen models:** `docling-layout-heron-mlx-bf16` (document layout), `granite-speech-4.1-2b-nar-mlx` (speech) +- **MTP/DFlash drafter checkpoints (need a target; not standalone):** `gemma-4-12b-it-assistant-4bit`, `gemma-4-31b-it-assistant-bf16`, `qwen3.5-27b-dflash`, `qwen3.5-4b-dflash` +- **VLM warmup failure under image setup:** `qwen2.5-vl-3b` (bf16), `minicpm-v-4.6-mxfp4` (mxfp4; the bf16 variant passes) - **Zero tokens generated:** `paligemma2-3b-6bit` -- **OOM-skipped (capacity, not a real failure):** `Qwen3.5-397B-A17B-4bit`, `qwen3-next-480b-4bit` +- **OOM-skipped (capacity, weights exceed the memory budget):** `qwen3-next-480b-4bit` + +These remaining failures are tracked in #315.