[
  {
    "model": "Qwen3.5-397B-A17B",
    "family": "Qwen3.5-397B-A17B",
    "model_band": "200B+",
    "rows": 2,
    "chip_count": 2,
    "chips": [
      "M3 Ultra (512 GB)",
      "M5 Max (128 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "flash-moe",
      "MLX"
    ],
    "quantizations": [
      "4bit",
      "q4.1bit"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 0,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 40.2,
    "fastest_chip": "M3 Ultra (512 GB)",
    "fastest_quantization": "q4.1bit",
    "max_context_tokens": null
  },
  {
    "model": "Qwen 3 235B-A22B",
    "family": "Qwen 3",
    "model_band": "200B+",
    "rows": 8,
    "chip_count": 6,
    "chips": [
      "M2 Ultra (192 GB)",
      "M3 Ultra (256 GB)",
      "M3 Ultra (512 GB)",
      "M4 Max (128 GB)",
      "M4 Ultra (192 GB)",
      "M5 Max (128 GB)"
    ],
    "runtime_count": 3,
    "runtimes": [
      "LM Studio",
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "3bit",
      "4bit",
      "Q4",
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 6,
    "smallest_published_fit_gb": 100,
    "smallest_published_fit_chip": "M4 Max (128 GB)",
    "fastest_avg_tok_s": 30,
    "fastest_chip": "M4 Max (128 GB)",
    "fastest_quantization": "3bit",
    "max_context_tokens": 10000
  },
  {
    "model": "Qwen3.5-122B-A10B",
    "family": "Qwen3.5-122B-A10B",
    "model_band": "200B+",
    "rows": 6,
    "chip_count": 3,
    "chips": [
      "M3 Ultra (256 GB)",
      "M5 Max (128 GB)",
      "M5 Pro (64 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "llama.cpp",
      "MLX"
    ],
    "quantizations": [
      "4bit",
      "8bit",
      "IQ1_M",
      "MXFP4"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 1,
    "smallest_published_fit_gb": 40.8,
    "smallest_published_fit_chip": "M5 Pro (64 GB)",
    "fastest_avg_tok_s": 65.85,
    "fastest_chip": "M5 Max (128 GB)",
    "fastest_quantization": "4bit",
    "max_context_tokens": 32778
  },
  {
    "model": "gpt-oss 120B",
    "family": "gpt-oss",
    "model_band": "200B+",
    "rows": 2,
    "chip_count": 2,
    "chips": [
      "M4 Ultra (192 GB)",
      "M5 Max (128 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 2,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 10,
    "fastest_chip": "M4 Ultra (192 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Mistral Small 4 119B",
    "family": "Mistral Small 4",
    "model_band": "200B+",
    "rows": 3,
    "chip_count": 2,
    "chips": [
      "M4 Ultra (192 GB)",
      "M5 Max (128 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 3,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 45,
    "fastest_chip": "M4 Ultra (192 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Qwen 2.5 72B",
    "family": "Qwen 2.5",
    "model_band": "70B",
    "rows": 2,
    "chip_count": 2,
    "chips": [
      "M4 Ultra (192 GB)",
      "M5 Max (128 GB)"
    ],
    "runtime_count": 1,
    "runtimes": [
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 2,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 15,
    "fastest_chip": "M4 Ultra (192 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "DeepSeek R1 Distill Llama 70B",
    "family": "DeepSeek R1 Distill Llama",
    "model_band": "70B",
    "rows": 2,
    "chip_count": 2,
    "chips": [
      "M4 Ultra (192 GB)",
      "M5 Max (128 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 2,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 16,
    "fastest_chip": "M4 Ultra (192 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Llama 3.3 70B",
    "family": "Llama 3.3",
    "model_band": "70B",
    "rows": 17,
    "chip_count": 8,
    "chips": [
      "M1 Ultra (64 GB)",
      "M2 Max (38-core GPU, 64 GB)",
      "M3 Max (GPU count not published, 64 GB)",
      "M3 Ultra (512 GB)",
      "M4 Max (128 GB)",
      "M4 Pro (64 GB)",
      "M4 Ultra (192 GB)",
      "M5 Max (128 GB)"
    ],
    "runtime_count": 4,
    "runtimes": [
      "llama.cpp",
      "LM Studio",
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "4bit",
      "8bit",
      "Q4_K - Medium",
      "Q8_0"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 9,
    "time_to_first_token_rows": 6,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 18,
    "fastest_chip": "M4 Ultra (192 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": 39686
  },
  {
    "model": "Qwen3.5-35B-A3B",
    "family": "Qwen3.5-35B-A3B",
    "model_band": "27B-32B",
    "rows": 13,
    "chip_count": 9,
    "chips": [
      "M1 Max (64 GB)",
      "M3 Max (96 GB)",
      "M3 Ultra (256 GB)",
      "M4 (16 GB)",
      "M4 Max (48 GB)",
      "M5 Max (48 GB)",
      "M5 Max (64 GB)",
      "M5 Max (128 GB)",
      "M5 Pro (64 GB)"
    ],
    "runtime_count": 5,
    "runtimes": [
      "llama.cpp",
      "LM Studio (llama.cpp)",
      "LM Studio (MLX)",
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "4bit",
      "8bit",
      "Q4_K - Medium",
      "Q4_K_L"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 6,
    "smallest_published_fit_gb": 19.6,
    "smallest_published_fit_chip": "M3 Ultra (256 GB)",
    "fastest_avg_tok_s": 128,
    "fastest_chip": "M5 Max (48 GB)",
    "fastest_quantization": "4bit",
    "max_context_tokens": 8000
  },
  {
    "model": "Qwen3.6-35B-A3B",
    "family": "Qwen3.6-35B-A3B",
    "model_band": "27B-32B",
    "rows": 4,
    "chip_count": 4,
    "chips": [
      "M4 Max (48 GB)",
      "M4 Pro (24 GB)",
      "M5 Max (64 GB)",
      "M5 Max (128 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 4,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 55,
    "fastest_chip": "M5 Max (128 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "DeepSeek R1 Distill Qwen 32B",
    "family": "DeepSeek R1 Distill Qwen",
    "model_band": "27B-32B",
    "rows": 3,
    "chip_count": 3,
    "chips": [
      "M3 Max (36 GB)",
      "M4 Max (48 GB)",
      "M5 Max (64 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "LM Studio",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 3,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 27,
    "fastest_chip": "M5 Max (64 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Qwen 3 32B",
    "family": "Qwen 3",
    "model_band": "27B-32B",
    "rows": 26,
    "chip_count": 7,
    "chips": [
      "M3 Max (GPU count not published, 64 GB)",
      "M4 Max (40-core GPU, 64 GB)",
      "M4 Max (64 GB)",
      "M4 Max (128 GB)",
      "M4 Pro (32 GB)",
      "M4 Ultra (192 GB)",
      "M5 Max (128 GB)"
    ],
    "runtime_count": 4,
    "runtimes": [
      "llama.cpp",
      "LM Studio",
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium",
      "Q8_0"
    ],
    "verified_rows": 1,
    "prompt_eval_rows": 20,
    "time_to_first_token_rows": 25,
    "smallest_published_fit_gb": 20,
    "smallest_published_fit_chip": "M4 Max (40-core GPU, 64 GB)",
    "fastest_avg_tok_s": 32,
    "fastest_chip": "M4 Ultra (192 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": 128000
  },
  {
    "model": "Gemma 4 31B",
    "family": "Gemma 4",
    "model_band": "27B-32B",
    "rows": 4,
    "chip_count": 4,
    "chips": [
      "M4 Max (48 GB)",
      "M4 Pro (24 GB)",
      "M5 Max (64 GB)",
      "M5 Max (128 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 4,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 26,
    "fastest_chip": "M5 Max (128 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Nemotron Cascade 2 30B-A3B",
    "family": "Nemotron Cascade 2",
    "model_band": "27B-32B",
    "rows": 3,
    "chip_count": 3,
    "chips": [
      "M4 Max (48 GB)",
      "M4 Pro (24 GB)",
      "M5 Max (64 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 3,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 35,
    "fastest_chip": "M5 Max (64 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Nemotron-3-Nano-30B-A3B",
    "family": "Nemotron-3-Nano-30B-A3B",
    "model_band": "27B-32B",
    "rows": 1,
    "chip_count": 1,
    "chips": [
      "M1 Max (64 GB)"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llama.cpp"
    ],
    "quantizations": [
      "Q4_K_XL"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 0,
    "smallest_published_fit_gb": 22,
    "smallest_published_fit_chip": "M1 Max (64 GB)",
    "fastest_avg_tok_s": 43.7,
    "fastest_chip": "M1 Max (64 GB)",
    "fastest_quantization": "Q4_K_XL",
    "max_context_tokens": 4096
  },
  {
    "model": "Qwen 3 30B-A3B",
    "family": "Qwen 3",
    "model_band": "27B-32B",
    "rows": 10,
    "chip_count": 7,
    "chips": [
      "M3 Max (36 GB)",
      "M4 Max (40-core GPU, 64 GB)",
      "M4 Max (48 GB)",
      "M4 Max (128 GB)",
      "M4 Pro (24 GB)",
      "M4 Pro (48 GB)",
      "M5 Max (64 GB)"
    ],
    "runtime_count": 3,
    "runtimes": [
      "LM Studio",
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "8bit",
      "Q4",
      "Q4_K - Medium",
      "Q5",
      "Q6",
      "Q8"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 4,
    "time_to_first_token_rows": 5,
    "smallest_published_fit_gb": 16.12,
    "smallest_published_fit_chip": "M4 Max (40-core GPU, 64 GB)",
    "fastest_avg_tok_s": 92.09,
    "fastest_chip": "M4 Max (40-core GPU, 64 GB)",
    "fastest_quantization": "Q4",
    "max_context_tokens": 10000
  },
  {
    "model": "Qwen3-Coder-30B-A3B",
    "family": "Qwen3-Coder-30B-A3B",
    "model_band": "27B-32B",
    "rows": 1,
    "chip_count": 1,
    "chips": [
      "M1 Max (64 GB)"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llama.cpp"
    ],
    "quantizations": [
      "IQ4_XS"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 0,
    "smallest_published_fit_gb": 16.1,
    "smallest_published_fit_chip": "M1 Max (64 GB)",
    "fastest_avg_tok_s": 58.5,
    "fastest_chip": "M1 Max (64 GB)",
    "fastest_quantization": "IQ4_XS",
    "max_context_tokens": 4096
  },
  {
    "model": "Gemma 3 27B",
    "family": "Gemma 3",
    "model_band": "27B-32B",
    "rows": 9,
    "chip_count": 8,
    "chips": [
      "M3 Max (96 GB)",
      "M3 Ultra (512 GB)",
      "M4 (32 GB)",
      "M4 Max (48 GB)",
      "M4 Max (128 GB)",
      "M4 Pro (24 GB)",
      "M5 Max (64 GB)",
      "M5 Max (128 GB)"
    ],
    "runtime_count": 4,
    "runtimes": [
      "llama.cpp",
      "LM Studio",
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "bf16",
      "Q4_0",
      "Q4_K - Medium",
      "Q6_K",
      "Q8_0"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 2,
    "time_to_first_token_rows": 7,
    "smallest_published_fit_gb": 52.57,
    "smallest_published_fit_chip": "M3 Ultra (512 GB)",
    "fastest_avg_tok_s": 42,
    "fastest_chip": "M5 Max (64 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": 131072
  },
  {
    "model": "Qwen3.5-27B",
    "family": "Qwen3.5-27B",
    "model_band": "27B-32B",
    "rows": 18,
    "chip_count": 10,
    "chips": [
      "M1 Max (64 GB)",
      "M2 Max (38-core GPU, 96 GB)",
      "M2 Ultra (GPU count not published, 128 GB)",
      "M3 Ultra (256 GB)",
      "M4 (16 GB)",
      "M4 Max",
      "M4 Pro (48 GB)",
      "M5 Max (48 GB)",
      "M5 Max (128 GB)",
      "M5 Pro (64 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "llama.cpp",
      "MLX"
    ],
    "quantizations": [
      "4bit",
      "8bit",
      "Q4_K",
      "Q4_K - Medium",
      "Q6_K",
      "Q8_0"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 6,
    "time_to_first_token_rows": 2,
    "smallest_published_fit_gb": 15.3,
    "smallest_published_fit_chip": "M3 Ultra (256 GB)",
    "fastest_avg_tok_s": 38,
    "fastest_chip": "M3 Ultra (256 GB)",
    "fastest_quantization": "4bit",
    "max_context_tokens": 16384
  },
  {
    "model": "Qwen3.6-27B",
    "family": "Qwen3.6-27B",
    "model_band": "27B-32B",
    "rows": 1,
    "chip_count": 1,
    "chips": [
      "M4 Max (128 GB)"
    ],
    "runtime_count": 1,
    "runtimes": [
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 0,
    "smallest_published_fit_gb": 29,
    "smallest_published_fit_chip": "M4 Max (128 GB)",
    "fastest_avg_tok_s": 16.56,
    "fastest_chip": "M4 Max (128 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Gemma 4 26B-A4B",
    "family": "Gemma 4",
    "model_band": "27B-32B",
    "rows": 4,
    "chip_count": 4,
    "chips": [
      "M4 Max (48 GB)",
      "M4 Pro (24 GB)",
      "M5 Max (128 GB)",
      "M5 Pro (24 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 4,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 50,
    "fastest_chip": "M5 Max (128 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Devstral Small 2 24B",
    "family": "Devstral Small 2",
    "model_band": "27B-32B",
    "rows": 8,
    "chip_count": 3,
    "chips": [
      "M1 Ultra (64 GB)",
      "M3 Ultra (256 GB)",
      "M4 (16 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "llama.cpp",
      "MLX"
    ],
    "quantizations": [
      "4bit",
      "8bit",
      "Q4_0",
      "Q4_1",
      "Q4_K - Medium",
      "Q8"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 7,
    "smallest_published_fit_gb": 13.4,
    "smallest_published_fit_chip": "M3 Ultra (256 GB)",
    "fastest_avg_tok_s": 47,
    "fastest_chip": "M3 Ultra (256 GB)",
    "fastest_quantization": "4bit",
    "max_context_tokens": null
  },
  {
    "model": "Llama 4 Scout 17B-16E",
    "family": "Llama 4 Scout",
    "model_band": "27B-32B",
    "rows": 3,
    "chip_count": 2,
    "chips": [
      "M4 Ultra (192 GB)",
      "M5 Max (128 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 3,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 30,
    "fastest_chip": "M4 Ultra (192 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Ministral 3 14B",
    "family": "Ministral 3",
    "model_band": "14B",
    "rows": 3,
    "chip_count": 3,
    "chips": [
      "M3 (16 GB)",
      "M4 Pro (24 GB)",
      "M5 Max (64 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "LM Studio",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 3,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 58,
    "fastest_chip": "M5 Max (64 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Phi-4 14B",
    "family": "Phi-4",
    "model_band": "14B",
    "rows": 3,
    "chip_count": 3,
    "chips": [
      "M2 (16 GB)",
      "M4 (16 GB)",
      "M5 Max (64 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 3,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 62,
    "fastest_chip": "M5 Max (64 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Qwen 3 14B",
    "family": "Qwen 3",
    "model_band": "14B",
    "rows": 4,
    "chip_count": 4,
    "chips": [
      "M3 (16 GB)",
      "M4 Pro (24 GB)",
      "M5 Max (64 GB)",
      "M5 Max (128 GB)"
    ],
    "runtime_count": 3,
    "runtimes": [
      "LM Studio",
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium",
      "Q8_0"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 4,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 58,
    "fastest_chip": "M5 Max (64 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "qwen-2-5-14b-instruct",
    "family": "qwen-2-5-14b-instruct",
    "model_band": "14B",
    "rows": 52,
    "chip_count": 52,
    "chips": [
      "M1 (7-core GPU, 16 GB)",
      "M1 (8-core GPU, 16 GB)",
      "M1 Max (24-core GPU, 32 GB)",
      "M1 Max (24-core GPU, 64 GB)",
      "M1 Max (32-core GPU, 32 GB)",
      "M1 Max (32-core GPU, 64 GB)",
      "M1 Pro (14-core GPU, 16 GB)",
      "M1 Pro (14-core GPU, 32 GB)",
      "M1 Pro (16-core GPU, 16 GB)",
      "M1 Pro (16-core GPU, 32 GB)",
      "M1 Ultra (48-core GPU, 128 GB)",
      "M1 Ultra (64-core GPU, 128 GB)",
      "M1 Ultra (GPU count not published, 128 GB)",
      "M2 (8-core GPU, 16 GB)",
      "M2 (10-core GPU, 16 GB)",
      "M2 (10-core GPU, 24 GB)",
      "M2 Max (30-core GPU, 64 GB)",
      "M2 Max (38-core GPU, 32 GB)",
      "M2 Max (38-core GPU, 64 GB)",
      "M2 Max (38-core GPU, 96 GB)",
      "M2 Pro (16-core GPU, 16 GB)",
      "M2 Pro (19-core GPU, 32 GB)",
      "M2 Ultra (60-core GPU, 64 GB)",
      "M2 Ultra (76-core GPU, 128 GB)",
      "M3 (10-core GPU, 24 GB)",
      "M3 Max (30-core GPU, 36 GB)",
      "M3 Max (30-core GPU, 96 GB)",
      "M3 Max (40-core GPU, 64 GB)",
      "M3 Max (40-core GPU, 128 GB)",
      "M3 Pro (14-core GPU, 18 GB)",
      "M3 Pro (14-core GPU, 36 GB)",
      "M3 Pro (18-core GPU, 18 GB)",
      "M3 Pro (18-core GPU, 36 GB)",
      "M3 Ultra (60-core GPU, 96 GB)",
      "M3 Ultra (80-core GPU, 256 GB)",
      "M3 Ultra (80-core GPU, 512 GB)",
      "M4 (8-core GPU, 16 GB)",
      "M4 (10-core GPU, 16 GB)",
      "M4 (10-core GPU, 24 GB)",
      "M4 (10-core GPU, 32 GB)",
      "M4 Max (32-core GPU, 36 GB)",
      "M4 Max (40-core GPU, 48 GB)",
      "M4 Max (40-core GPU, 64 GB)",
      "M4 Max (40-core GPU, 128 GB)",
      "M4 Pro (16-core GPU, 24 GB)",
      "M4 Pro (16-core GPU, 48 GB)",
      "M4 Pro (16-core GPU, 64 GB)",
      "M4 Pro (20-core GPU, 24 GB)",
      "M4 Pro (20-core GPU, 48 GB)",
      "M4 Pro (20-core GPU, 64 GB)",
      "M5 (10-core GPU, 32 GB)",
      "M5 Max (32-core GPU, 36 GB)"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 52,
    "time_to_first_token_rows": 52,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 36.71,
    "fastest_chip": "M3 Ultra (80-core GPU, 256 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Gemma 3 12B",
    "family": "Gemma 3",
    "model_band": "14B",
    "rows": 4,
    "chip_count": 4,
    "chips": [
      "M2 (16 GB)",
      "M3 (16 GB)",
      "M4 Pro (24 GB)",
      "M5 Max (64 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 4,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 68,
    "fastest_chip": "M5 Max (64 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Qwen3.5-9B",
    "family": "Qwen3.5-9B",
    "model_band": "14B",
    "rows": 13,
    "chip_count": 9,
    "chips": [
      "M1 (16 GB)",
      "M1 Pro (16 GB)",
      "M3 (16 GB)",
      "M3 Ultra (256 GB)",
      "M4 (16 GB)",
      "M4 Pro (24 GB)",
      "M5 Max (64 GB)",
      "M5 Max (128 GB)",
      "M5 Pro (64 GB)"
    ],
    "runtime_count": 4,
    "runtimes": [
      "llama.cpp",
      "LM Studio",
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "4bit",
      "Q4_0",
      "Q4_K - Medium",
      "Q4_K - Small",
      "Q6_K",
      "Q8_0"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 10,
    "smallest_published_fit_gb": 5.1,
    "smallest_published_fit_chip": "M3 Ultra (256 GB)",
    "fastest_avg_tok_s": 106,
    "fastest_chip": "M3 Ultra (256 GB)",
    "fastest_quantization": "4bit",
    "max_context_tokens": null
  },
  {
    "model": "DeepSeek R1 Distill Llama 8B",
    "family": "DeepSeek R1 Distill Llama",
    "model_band": "7B-8B",
    "rows": 5,
    "chip_count": 4,
    "chips": [
      "M1 (16 GB)",
      "M2 (16 GB)",
      "M4 (16 GB)",
      "M5 Max (64 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium",
      "Q8_0"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 5,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 97,
    "fastest_chip": "M5 Max (64 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Llama 3.1 8B",
    "family": "Llama 3.1",
    "model_band": "7B-8B",
    "rows": 6,
    "chip_count": 6,
    "chips": [
      "M1 (16 GB)",
      "M2 (8 GB)",
      "M3 Pro (18 GB)",
      "M4 (16 GB)",
      "M5 Max (64 GB)",
      "M5 Max (128 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium",
      "Q8_0"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 6,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 138,
    "fastest_chip": "M5 Max (128 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "llama-3-1-8b-instruct",
    "family": "llama-3-1-8b-instruct",
    "model_band": "7B-8B",
    "rows": 52,
    "chip_count": 52,
    "chips": [
      "M1 (7-core GPU, 8 GB)",
      "M1 (7-core GPU, 16 GB)",
      "M1 (8-core GPU, 8 GB)",
      "M1 Max (24-core GPU, 64 GB)",
      "M1 Max (32-core GPU, 32 GB)",
      "M1 Max (32-core GPU, 64 GB)",
      "M1 Pro (14-core GPU, 16 GB)",
      "M1 Pro (14-core GPU, 32 GB)",
      "M1 Pro (16-core GPU, 16 GB)",
      "M1 Pro (16-core GPU, 32 GB)",
      "M1 Ultra (48-core GPU, 128 GB)",
      "M1 Ultra (64-core GPU, 128 GB)",
      "M1 Ultra (GPU count not published, 128 GB)",
      "M2 (8-core GPU, 8 GB)",
      "M2 (8-core GPU, 16 GB)",
      "M2 (10-core GPU, 16 GB)",
      "M2 (10-core GPU, 24 GB)",
      "M2 Max (30-core GPU, 32 GB)",
      "M2 Max (38-core GPU, 32 GB)",
      "M2 Max (38-core GPU, 96 GB)",
      "M2 Pro (16-core GPU, 16 GB)",
      "M2 Pro (16-core GPU, 32 GB)",
      "M2 Pro (19-core GPU, 32 GB)",
      "M2 Ultra (60-core GPU, 64 GB)",
      "M3 (10-core GPU, 16 GB)",
      "M3 (10-core GPU, 24 GB)",
      "M3 (GPU count not published, 16 GB)",
      "M3 Max (30-core GPU, 36 GB)",
      "M3 Max (30-core GPU, 96 GB)",
      "M3 Max (40-core GPU, 64 GB)",
      "M3 Max (40-core GPU, 128 GB)",
      "M3 Pro (14-core GPU, 18 GB)",
      "M3 Pro (14-core GPU, 36 GB)",
      "M3 Pro (18-core GPU, 18 GB)",
      "M3 Pro (18-core GPU, 36 GB)",
      "M3 Ultra (80-core GPU, 256 GB)",
      "M3 Ultra (80-core GPU, 512 GB)",
      "M4 (8-core GPU, 16 GB)",
      "M4 (10-core GPU, 16 GB)",
      "M4 (10-core GPU, 24 GB)",
      "M4 (10-core GPU, 32 GB)",
      "M4 Max (32-core GPU, 36 GB)",
      "M4 Max (40-core GPU, 48 GB)",
      "M4 Max (40-core GPU, 64 GB)",
      "M4 Max (40-core GPU, 128 GB)",
      "M4 Pro (16-core GPU, 24 GB)",
      "M4 Pro (16-core GPU, 48 GB)",
      "M4 Pro (20-core GPU, 24 GB)",
      "M4 Pro (20-core GPU, 48 GB)",
      "M4 Pro (20-core GPU, 64 GB)",
      "M5 (10-core GPU, 32 GB)",
      "M5 Max (32-core GPU, 36 GB)"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 52,
    "time_to_first_token_rows": 52,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 63.3,
    "fastest_chip": "M3 Ultra (80-core GPU, 256 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Ministral 3 8B",
    "family": "Ministral 3",
    "model_band": "7B-8B",
    "rows": 3,
    "chip_count": 3,
    "chips": [
      "M3 (16 GB)",
      "M4 (16 GB)",
      "M5 Max (64 GB)"
    ],
    "runtime_count": 3,
    "runtimes": [
      "LM Studio",
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 3,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 98,
    "fastest_chip": "M5 Max (64 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Qwen 3 8B",
    "family": "Qwen 3",
    "model_band": "7B-8B",
    "rows": 6,
    "chip_count": 5,
    "chips": [
      "M1 (16 GB)",
      "M2 (16 GB)",
      "M4 Max (128 GB)",
      "M4 Pro (24 GB)",
      "M5 Max (128 GB)"
    ],
    "runtime_count": 3,
    "runtimes": [
      "LM Studio",
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium",
      "Q8_0"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 6,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 98,
    "fastest_chip": "M5 Max (128 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": 10000
  },
  {
    "model": "Llama 2 7B",
    "family": "Llama 2",
    "model_band": "7B-8B",
    "rows": 5,
    "chip_count": 5,
    "chips": [
      "M1 Pro (16-core GPU)",
      "M2 Ultra (76-core GPU, 192 GB)",
      "M3 Max (40-core GPU, 48 GB)",
      "M3 Pro (18-core GPU)",
      "M4 (10-core GPU, 16 GB)"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llama.cpp"
    ],
    "quantizations": [
      "Q4_0"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 5,
    "time_to_first_token_rows": 0,
    "smallest_published_fit_gb": 3.56,
    "smallest_published_fit_chip": "M1 Pro (16-core GPU)",
    "fastest_avg_tok_s": 94.27,
    "fastest_chip": "M2 Ultra (76-core GPU, 192 GB)",
    "fastest_quantization": "Q4_0",
    "max_context_tokens": 512
  },
  {
    "model": "Mistral 7B v0.3",
    "family": "Mistral",
    "model_band": "7B-8B",
    "rows": 6,
    "chip_count": 4,
    "chips": [
      "M1 (16 GB)",
      "M3 (16 GB)",
      "M4 Pro (24 GB)",
      "M5 Max (64 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium",
      "Q8_0"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 6,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 122,
    "fastest_chip": "M5 Max (64 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "qwen-2-5-7b-instruct",
    "family": "qwen-2-5-7b-instruct",
    "model_band": "7B-8B",
    "rows": 1,
    "chip_count": 1,
    "chips": [
      "M4 Max (128 GB)"
    ],
    "runtime_count": 1,
    "runtimes": [
      "LM Studio"
    ],
    "quantizations": [
      "Q8_0"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 1,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 49.67,
    "fastest_chip": "M4 Max (128 GB)",
    "fastest_quantization": "Q8_0",
    "max_context_tokens": 10000
  },
  {
    "model": "qwen-3-0-6b",
    "family": "qwen-3-0-6b",
    "model_band": "7B-8B",
    "rows": 1,
    "chip_count": 1,
    "chips": [
      "M4 Max (128 GB)"
    ],
    "runtime_count": 1,
    "runtimes": [
      "LM Studio"
    ],
    "quantizations": [
      "Q8_0"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 1,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 184.45,
    "fastest_chip": "M4 Max (128 GB)",
    "fastest_quantization": "Q8_0",
    "max_context_tokens": 10000
  },
  {
    "model": "Gemma 3 4B",
    "family": "Gemma 3",
    "model_band": "0.5B-4B",
    "rows": 5,
    "chip_count": 5,
    "chips": [
      "M1 (8 GB)",
      "M3 Pro (18 GB)",
      "M4 Max (128 GB)",
      "M4 Pro (24 GB)",
      "M5 Max (64 GB)"
    ],
    "runtime_count": 3,
    "runtimes": [
      "LM Studio",
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_0",
      "Q4_K - Medium",
      "Q8_0"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 5,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 132,
    "fastest_chip": "M5 Max (64 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": 4096
  },
  {
    "model": "Gemma 4 E4B",
    "family": "Gemma 4 E4B",
    "model_band": "0.5B-4B",
    "rows": 5,
    "chip_count": 5,
    "chips": [
      "M1 (8 GB)",
      "M3 (16 GB)",
      "M4 Pro (24 GB)",
      "M5 Max (128 GB)",
      "M5 Pro (24 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 5,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 128,
    "fastest_chip": "M5 Max (128 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Qwen 3 4B",
    "family": "Qwen 3",
    "model_band": "0.5B-4B",
    "rows": 9,
    "chip_count": 4,
    "chips": [
      "M2 (8 GB)",
      "M4 Max (40-core GPU, 64 GB)",
      "M4 Pro (24 GB)",
      "M5 Max (64 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4",
      "Q4_G32",
      "Q4_K - Medium",
      "Q5",
      "Q5_G32",
      "Q6",
      "Q8"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 6,
    "time_to_first_token_rows": 3,
    "smallest_published_fit_gb": 2.54,
    "smallest_published_fit_chip": "M4 Max (40-core GPU, 64 GB)",
    "fastest_avg_tok_s": 149.07,
    "fastest_chip": "M4 Max (40-core GPU, 64 GB)",
    "fastest_quantization": "Q4_G32",
    "max_context_tokens": 2048
  },
  {
    "model": "Qwen3.5-4B",
    "family": "Qwen3.5-4B",
    "model_band": "0.5B-4B",
    "rows": 2,
    "chip_count": 2,
    "chips": [
      "M4 (16 GB)",
      "M5 Max (64 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 2,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 148,
    "fastest_chip": "M5 Max (64 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Phi-4 Mini Instruct 3.8B",
    "family": "Phi-4 Mini Instruct",
    "model_band": "0.5B-4B",
    "rows": 7,
    "chip_count": 6,
    "chips": [
      "M1 (16 GB)",
      "M2 (8 GB)",
      "M3 (16 GB)",
      "M4 Max (48 GB)",
      "M4 Pro (24 GB)",
      "M5 Max (64 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium",
      "Q8_0"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 7,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 142,
    "fastest_chip": "M5 Max (64 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Gemma 4 E2B",
    "family": "Gemma 4 E2B",
    "model_band": "0.5B-4B",
    "rows": 4,
    "chip_count": 4,
    "chips": [
      "M1 (8 GB)",
      "M3 (16 GB)",
      "M4 Pro (24 GB)",
      "M5 Max (128 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 4,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 158,
    "fastest_chip": "M5 Max (128 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "llama-3-2-1b-instruct",
    "family": "llama-3-2-1b-instruct",
    "model_band": "0.5B-4B",
    "rows": 63,
    "chip_count": 63,
    "chips": [
      "M1 (7-core GPU, 8 GB)",
      "M1 (7-core GPU, 16 GB)",
      "M1 (8-core GPU, 8 GB)",
      "M1 (8-core GPU, 16 GB)",
      "M1 Max (24-core GPU, 32 GB)",
      "M1 Max (24-core GPU, 64 GB)",
      "M1 Max (32-core GPU, 32 GB)",
      "M1 Max (32-core GPU, 64 GB)",
      "M1 Pro (14-core GPU, 16 GB)",
      "M1 Pro (14-core GPU, 32 GB)",
      "M1 Pro (16-core GPU, 16 GB)",
      "M1 Pro (16-core GPU, 32 GB)",
      "M1 Ultra (48-core GPU, 128 GB)",
      "M1 Ultra (64-core GPU, 128 GB)",
      "M1 Ultra (GPU count not published, 128 GB)",
      "M2 (8-core GPU, 8 GB)",
      "M2 (8-core GPU, 16 GB)",
      "M2 (10-core GPU, 8 GB)",
      "M2 (10-core GPU, 16 GB)",
      "M2 (10-core GPU, 24 GB)",
      "M2 Max (30-core GPU, 32 GB)",
      "M2 Max (38-core GPU, 32 GB)",
      "M2 Pro (16-core GPU, 16 GB)",
      "M2 Pro (16-core GPU, 32 GB)",
      "M2 Pro (19-core GPU, 16 GB)",
      "M2 Pro (19-core GPU, 32 GB)",
      "M2 Ultra (60-core GPU, 64 GB)",
      "M2 Ultra (60-core GPU, 128 GB)",
      "M2 Ultra (60-core GPU, 192 GB)",
      "M2 Ultra (GPU count not published, 128 GB)",
      "M3 (10-core GPU, 16 GB)",
      "M3 (10-core GPU, 24 GB)",
      "M3 (GPU count not published, 16 GB)",
      "M3 Max (30-core GPU, 36 GB)",
      "M3 Max (30-core GPU, 96 GB)",
      "M3 Max (40-core GPU, 48 GB)",
      "M3 Max (40-core GPU, 64 GB)",
      "M3 Max (40-core GPU, 128 GB)",
      "M3 Pro (14-core GPU, 18 GB)",
      "M3 Pro (14-core GPU, 36 GB)",
      "M3 Pro (18-core GPU, 18 GB)",
      "M3 Pro (18-core GPU, 36 GB)",
      "M3 Ultra (80-core GPU, 256 GB)",
      "M3 Ultra (80-core GPU, 512 GB)",
      "M4 (8-core GPU, 16 GB)",
      "M4 (10-core GPU, 16 GB)",
      "M4 (10-core GPU, 24 GB)",
      "M4 (10-core GPU, 32 GB)",
      "M4 (GPU count not published, 16 GB)",
      "M4 Max (32-core GPU, 36 GB)",
      "M4 Max (40-core GPU, 48 GB)",
      "M4 Max (40-core GPU, 64 GB)",
      "M4 Max (40-core GPU, 128 GB)",
      "M4 Max (GPU count not published, 128 GB)",
      "M4 Pro (16-core GPU, 24 GB)",
      "M4 Pro (16-core GPU, 48 GB)",
      "M4 Pro (16-core GPU, 64 GB)",
      "M4 Pro (20-core GPU, 24 GB)",
      "M4 Pro (20-core GPU, 48 GB)",
      "M4 Pro (20-core GPU, 64 GB)",
      "M5 (10-core GPU, 16 GB)",
      "M5 (10-core GPU, 32 GB)",
      "M5 Max (32-core GPU, 36 GB)"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "quantizations": [
      "Q4_K - Medium"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 63,
    "time_to_first_token_rows": 63,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 228.99,
    "fastest_chip": "M5 Max (32-core GPU, 36 GB)",
    "fastest_quantization": "Q4_K - Medium",
    "max_context_tokens": null
  },
  {
    "model": "Qwen 3 0.6B",
    "family": "Qwen 3",
    "model_band": "0.5B-4B",
    "rows": 1,
    "chip_count": 1,
    "chips": [
      "M3 Ultra (256 GB)"
    ],
    "runtime_count": 1,
    "runtimes": [
      "MLX"
    ],
    "quantizations": [
      "4bit"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 0,
    "smallest_published_fit_gb": null,
    "smallest_published_fit_chip": null,
    "fastest_avg_tok_s": 370,
    "fastest_chip": "M3 Ultra (256 GB)",
    "fastest_quantization": "4bit",
    "max_context_tokens": null
  },
  {
    "model": "Devstral Small 1.1",
    "family": "Devstral Small 1.1",
    "model_band": "unknown",
    "rows": 4,
    "chip_count": 4,
    "chips": [
      "M2 (24 GB)",
      "M3 Ultra (512 GB)",
      "M4 Max (128 GB)",
      "M4 Pro (48 GB)"
    ],
    "runtime_count": 1,
    "runtimes": [
      "LM Studio"
    ],
    "quantizations": [
      "4bit",
      "6bit"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 1,
    "smallest_published_fit_gb": 18.51,
    "smallest_published_fit_chip": "M4 Pro (48 GB)",
    "fastest_avg_tok_s": 43,
    "fastest_chip": "M3 Ultra (512 GB)",
    "fastest_quantization": "4bit",
    "max_context_tokens": 131072
  },
  {
    "model": "GLM-4.5-Air",
    "family": "GLM-4.5-Air",
    "model_band": "unknown",
    "rows": 5,
    "chip_count": 2,
    "chips": [
      "M3 Ultra (256 GB)",
      "M4 Max (128 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "LM Studio",
      "MLX"
    ],
    "quantizations": [
      "4bit"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 4,
    "time_to_first_token_rows": 0,
    "smallest_published_fit_gb": 60.3,
    "smallest_published_fit_chip": "M3 Ultra (256 GB)",
    "fastest_avg_tok_s": 54,
    "fastest_chip": "M3 Ultra (256 GB)",
    "fastest_quantization": "4bit",
    "max_context_tokens": 62037
  },
  {
    "model": "GLM-4.7-Flash",
    "family": "GLM-4.7-Flash",
    "model_band": "unknown",
    "rows": 2,
    "chip_count": 2,
    "chips": [
      "M1 Max (64 GB)",
      "M3 Ultra (256 GB)"
    ],
    "runtime_count": 2,
    "runtimes": [
      "llama.cpp",
      "MLX"
    ],
    "quantizations": [
      "8bit",
      "Q4_K_XL"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 0,
    "smallest_published_fit_gb": 17,
    "smallest_published_fit_chip": "M1 Max (64 GB)",
    "fastest_avg_tok_s": 58,
    "fastest_chip": "M3 Ultra (256 GB)",
    "fastest_quantization": "8bit",
    "max_context_tokens": 4096
  },
  {
    "model": "GLM-5",
    "family": "GLM-5",
    "model_band": "unknown",
    "rows": 5,
    "chip_count": 1,
    "chips": [
      "M3 Ultra (512 GB)"
    ],
    "runtime_count": 1,
    "runtimes": [
      "MLX"
    ],
    "quantizations": [
      "4bit"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 5,
    "time_to_first_token_rows": 5,
    "smallest_published_fit_gb": 391.82,
    "smallest_published_fit_chip": "M3 Ultra (512 GB)",
    "fastest_avg_tok_s": 16.7,
    "fastest_chip": "M3 Ultra (512 GB)",
    "fastest_quantization": "4bit",
    "max_context_tokens": 32768
  },
  {
    "model": "Qwen3-Coder-Next",
    "family": "Qwen3-Coder-Next",
    "model_band": "unknown",
    "rows": 6,
    "chip_count": 2,
    "chips": [
      "M3 Ultra (256 GB)",
      "M5 Max (128 GB)"
    ],
    "runtime_count": 1,
    "runtimes": [
      "MLX"
    ],
    "quantizations": [
      "4bit",
      "6bit",
      "8bit"
    ],
    "verified_rows": 0,
    "prompt_eval_rows": 4,
    "time_to_first_token_rows": 0,
    "smallest_published_fit_gb": 44.9,
    "smallest_published_fit_chip": "M3 Ultra (256 GB)",
    "fastest_avg_tok_s": 79.3,
    "fastest_chip": "M5 Max (128 GB)",
    "fastest_quantization": "8bit",
    "max_context_tokens": 65545
  }
]
