[
  {
    "chip": "M1 (7-core GPU, 8 GB)",
    "rows": 2,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 2,
    "time_to_first_token_rows": 2,
    "fastest_avg_tok_s": 38.52,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M1 (7-core GPU, 16 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 37.87,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M1 (8 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "Gemma 3 4B",
      "Gemma 4 E2B",
      "Gemma 4 E4B"
    ],
    "runtime_count": 1,
    "runtimes": [
      "Ollama"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 58,
    "fastest_model": "Gemma 4 E2B",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M1 (8-core GPU, 8 GB)",
    "rows": 2,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 2,
    "time_to_first_token_rows": 2,
    "fastest_avg_tok_s": 40.41,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M1 (8-core GPU, 16 GB)",
    "rows": 2,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 2,
    "time_to_first_token_rows": 2,
    "fastest_avg_tok_s": 40.18,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M1 (16 GB)",
    "rows": 6,
    "verified_rows": 0,
    "model_count": 6,
    "models": [
      "DeepSeek R1 Distill Llama 8B",
      "Llama 3.1 8B",
      "Mistral 7B v0.3",
      "Phi-4 Mini Instruct 3.8B",
      "Qwen 3 8B",
      "Qwen3.5-9B"
    ],
    "runtime_count": 1,
    "runtimes": [
      "Ollama"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 6,
    "fastest_avg_tok_s": 58,
    "fastest_model": "Phi-4 Mini Instruct 3.8B",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M1 Max (24-core GPU, 32 GB)",
    "rows": 2,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 2,
    "time_to_first_token_rows": 2,
    "fastest_avg_tok_s": 93.86,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M1 Max (24-core GPU, 64 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 105.66,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M1 Max (32-core GPU, 32 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 125.77,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M1 Max (32-core GPU, 64 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 120.67,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M1 Max (64 GB)",
    "rows": 10,
    "verified_rows": 0,
    "model_count": 5,
    "models": [
      "GLM-4.7-Flash",
      "Nemotron-3-Nano-30B-A3B",
      "Qwen3-Coder-30B-A3B",
      "Qwen3.5-27B",
      "Qwen3.5-35B-A3B"
    ],
    "runtime_count": 4,
    "runtimes": [
      "llama.cpp",
      "LM Studio (llama.cpp)",
      "LM Studio (MLX)",
      "MLX"
    ],
    "max_context_tokens": 8000,
    "prompt_eval_rows": 5,
    "time_to_first_token_rows": 0,
    "fastest_avg_tok_s": 58.5,
    "fastest_model": "Qwen3-Coder-30B-A3B",
    "fastest_quantization": "IQ4_XS",
    "largest_published_fit_gb": 22,
    "largest_published_fit_model": "Nemotron-3-Nano-30B-A3B"
  },
  {
    "chip": "M1 Pro (14-core GPU, 16 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 71.81,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M1 Pro (14-core GPU, 32 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 71.02,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M1 Pro (16 GB)",
    "rows": 2,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "Qwen3.5-9B"
    ],
    "runtime_count": 2,
    "runtimes": [
      "llama.cpp",
      "MLX"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 0,
    "fastest_avg_tok_s": 30,
    "fastest_model": "Qwen3.5-9B",
    "fastest_quantization": "4bit",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M1 Pro (16-core GPU, 16 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 78.16,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M1 Pro (16-core GPU, 32 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 77.24,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M1 Pro (16-core GPU)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "Llama 2 7B"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llama.cpp"
    ],
    "max_context_tokens": 512,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 0,
    "fastest_avg_tok_s": 36.41,
    "fastest_model": "Llama 2 7B",
    "fastest_quantization": "Q4_0",
    "largest_published_fit_gb": 3.56,
    "largest_published_fit_model": "Llama 2 7B"
  },
  {
    "chip": "M1 Ultra (48-core GPU, 128 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 137.97,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M1 Ultra (64 GB)",
    "rows": 5,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "Devstral Small 2 24B",
      "Llama 3.3 70B"
    ],
    "runtime_count": 3,
    "runtimes": [
      "llama.cpp",
      "LM Studio",
      "MLX"
    ],
    "max_context_tokens": 3991,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 5,
    "fastest_avg_tok_s": 29.68,
    "fastest_model": "Devstral Small 2 24B",
    "fastest_quantization": "4bit",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M1 Ultra (64-core GPU, 128 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 151.07,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M1 Ultra (GPU count not published, 128 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 57.07,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 (8 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "Llama 3.1 8B",
      "Phi-4 Mini Instruct 3.8B",
      "Qwen 3 4B"
    ],
    "runtime_count": 1,
    "runtimes": [
      "Ollama"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 72,
    "fastest_model": "Phi-4 Mini Instruct 3.8B",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 (8-core GPU, 8 GB)",
    "rows": 2,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 2,
    "time_to_first_token_rows": 2,
    "fastest_avg_tok_s": 34.48,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 (8-core GPU, 16 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 35.32,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 (10-core GPU, 8 GB)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "llama-3-2-1b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 1,
    "fastest_avg_tok_s": 56.46,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 (10-core GPU, 16 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 55.57,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 (10-core GPU, 24 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 54.76,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 (16 GB)",
    "rows": 4,
    "verified_rows": 0,
    "model_count": 4,
    "models": [
      "DeepSeek R1 Distill Llama 8B",
      "Gemma 3 12B",
      "Phi-4 14B",
      "Qwen 3 8B"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 4,
    "fastest_avg_tok_s": 58,
    "fastest_model": "DeepSeek R1 Distill Llama 8B",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 (24 GB)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "Devstral Small 1.1"
    ],
    "runtime_count": 1,
    "runtimes": [
      "LM Studio"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 0,
    "fastest_avg_tok_s": 6,
    "fastest_model": "Devstral Small 1.1",
    "fastest_quantization": "4bit",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 Max (30-core GPU, 32 GB)",
    "rows": 2,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 2,
    "time_to_first_token_rows": 2,
    "fastest_avg_tok_s": 127.59,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 Max (30-core GPU, 64 GB)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 1,
    "fastest_avg_tok_s": 14.49,
    "fastest_model": "qwen-2-5-14b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 Max (38-core GPU, 32 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 152.98,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 Max (38-core GPU, 64 GB)",
    "rows": 2,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "Llama 3.3 70B",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 2,
    "runtimes": [
      "llamafile",
      "LM Studio"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 1,
    "fastest_avg_tok_s": 21.96,
    "fastest_model": "qwen-2-5-14b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 Max (38-core GPU, 96 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "qwen-2-5-14b-instruct",
      "Qwen3.5-27B"
    ],
    "runtime_count": 2,
    "runtimes": [
      "llamafile",
      "MLX"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 2,
    "time_to_first_token_rows": 2,
    "fastest_avg_tok_s": 46.41,
    "fastest_model": "llama-3-1-8b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": 30,
    "largest_published_fit_model": "Qwen3.5-27B"
  },
  {
    "chip": "M2 Pro (16-core GPU, 16 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 91.15,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 Pro (16-core GPU, 32 GB)",
    "rows": 2,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 2,
    "time_to_first_token_rows": 2,
    "fastest_avg_tok_s": 91.45,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 Pro (19-core GPU, 16 GB)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "llama-3-2-1b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 1,
    "fastest_avg_tok_s": 99.52,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 Pro (19-core GPU, 32 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 100.26,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 Ultra (60-core GPU, 64 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 174.09,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 Ultra (60-core GPU, 128 GB)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "llama-3-2-1b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 1,
    "fastest_avg_tok_s": 176.39,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 Ultra (60-core GPU, 192 GB)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "llama-3-2-1b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 1,
    "fastest_avg_tok_s": 169.76,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 Ultra (76-core GPU, 128 GB)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 1,
    "fastest_avg_tok_s": 36.63,
    "fastest_model": "qwen-2-5-14b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 Ultra (76-core GPU, 192 GB)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "Llama 2 7B"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llama.cpp"
    ],
    "max_context_tokens": 512,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 0,
    "fastest_avg_tok_s": 94.27,
    "fastest_model": "Llama 2 7B",
    "fastest_quantization": "Q4_0",
    "largest_published_fit_gb": 3.56,
    "largest_published_fit_model": "Llama 2 7B"
  },
  {
    "chip": "M2 Ultra (192 GB)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "Qwen 3 235B-A22B"
    ],
    "runtime_count": 1,
    "runtimes": [
      "MLX"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 0,
    "fastest_avg_tok_s": 26.4,
    "fastest_model": "Qwen 3 235B-A22B",
    "fastest_quantization": "Q4",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M2 Ultra (GPU count not published, 128 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "llama-3-2-1b-instruct",
      "Qwen3.5-27B"
    ],
    "runtime_count": 2,
    "runtimes": [
      "llamafile",
      "MLX"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 1,
    "fastest_avg_tok_s": 120.41,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 (10-core GPU, 16 GB)",
    "rows": 2,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 2,
    "time_to_first_token_rows": 2,
    "fastest_avg_tok_s": 67.19,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 (10-core GPU, 24 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 64.71,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 (16 GB)",
    "rows": 9,
    "verified_rows": 0,
    "model_count": 9,
    "models": [
      "Gemma 3 12B",
      "Gemma 4 E2B",
      "Gemma 4 E4B",
      "Ministral 3 8B",
      "Ministral 3 14B",
      "Mistral 7B v0.3",
      "Phi-4 Mini Instruct 3.8B",
      "Qwen 3 14B",
      "Qwen3.5-9B"
    ],
    "runtime_count": 3,
    "runtimes": [
      "LM Studio",
      "MLX",
      "Ollama"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 9,
    "fastest_avg_tok_s": 95,
    "fastest_model": "Phi-4 Mini Instruct 3.8B",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 (GPU count not published, 16 GB)",
    "rows": 2,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 2,
    "time_to_first_token_rows": 2,
    "fastest_avg_tok_s": 61.58,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 Max (30-core GPU, 36 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 132.99,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 Max (30-core GPU, 96 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 132.87,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 Max (36 GB)",
    "rows": 2,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "DeepSeek R1 Distill Qwen 32B",
      "Qwen 3 30B-A3B"
    ],
    "runtime_count": 1,
    "runtimes": [
      "Ollama"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 2,
    "fastest_avg_tok_s": 28,
    "fastest_model": "Qwen 3 30B-A3B",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 Max (40-core GPU, 48 GB)",
    "rows": 2,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "Llama 2 7B",
      "llama-3-2-1b-instruct"
    ],
    "runtime_count": 2,
    "runtimes": [
      "llama.cpp",
      "llamafile"
    ],
    "max_context_tokens": 512,
    "prompt_eval_rows": 2,
    "time_to_first_token_rows": 1,
    "fastest_avg_tok_s": 148.96,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": 3.56,
    "largest_published_fit_model": "Llama 2 7B"
  },
  {
    "chip": "M3 Max (40-core GPU, 64 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 106.97,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 Max (40-core GPU, 128 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 146.34,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 Max (96 GB)",
    "rows": 2,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "Gemma 3 27B",
      "Qwen3.5-35B-A3B"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 2,
    "fastest_avg_tok_s": 28,
    "fastest_model": "Gemma 3 27B",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "rows": 24,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "Llama 3.3 70B",
      "Qwen 3 32B"
    ],
    "runtime_count": 2,
    "runtimes": [
      "llama.cpp",
      "Ollama"
    ],
    "max_context_tokens": 32170,
    "prompt_eval_rows": 24,
    "time_to_first_token_rows": 20,
    "fastest_avg_tok_s": 10.41,
    "fastest_model": "Qwen 3 32B",
    "fastest_quantization": "Q8_0",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 Pro (14-core GPU, 18 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 88.11,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 Pro (14-core GPU, 36 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 88.24,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 Pro (18 GB)",
    "rows": 2,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "Gemma 3 4B",
      "Llama 3.1 8B"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 2,
    "fastest_avg_tok_s": 88,
    "fastest_model": "Gemma 3 4B",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 Pro (18-core GPU, 18 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 85.63,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 Pro (18-core GPU, 36 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 89.76,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 Pro (18-core GPU)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "Llama 2 7B"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llama.cpp"
    ],
    "max_context_tokens": 512,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 0,
    "fastest_avg_tok_s": 30.74,
    "fastest_model": "Llama 2 7B",
    "fastest_quantization": "Q4_0",
    "largest_published_fit_gb": 3.56,
    "largest_published_fit_model": "Llama 2 7B"
  },
  {
    "chip": "M3 Ultra (60-core GPU, 96 GB)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 1,
    "fastest_avg_tok_s": 34.44,
    "fastest_model": "qwen-2-5-14b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 Ultra (80-core GPU, 256 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 177.9,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 Ultra (80-core GPU, 512 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 178.84,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M3 Ultra (256 GB)",
    "rows": 13,
    "verified_rows": 0,
    "model_count": 10,
    "models": [
      "Devstral Small 2 24B",
      "GLM-4.5-Air",
      "GLM-4.7-Flash",
      "Qwen 3 0.6B",
      "Qwen 3 235B-A22B",
      "Qwen3-Coder-Next",
      "Qwen3.5-9B",
      "Qwen3.5-27B",
      "Qwen3.5-35B-A3B",
      "Qwen3.5-122B-A10B"
    ],
    "runtime_count": 1,
    "runtimes": [
      "MLX"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 0,
    "fastest_avg_tok_s": 370,
    "fastest_model": "Qwen 3 0.6B",
    "fastest_quantization": "4bit",
    "largest_published_fit_gb": 129.8,
    "largest_published_fit_model": "Qwen3.5-122B-A10B"
  },
  {
    "chip": "M3 Ultra (512 GB)",
    "rows": 13,
    "verified_rows": 0,
    "model_count": 6,
    "models": [
      "Devstral Small 1.1",
      "Gemma 3 27B",
      "GLM-5",
      "Llama 3.3 70B",
      "Qwen 3 235B-A22B",
      "Qwen3.5-397B-A17B"
    ],
    "runtime_count": 2,
    "runtimes": [
      "LM Studio",
      "MLX"
    ],
    "max_context_tokens": 128000,
    "prompt_eval_rows": 9,
    "time_to_first_token_rows": 7,
    "fastest_avg_tok_s": 43,
    "fastest_model": "Devstral Small 1.1",
    "fastest_quantization": "4bit",
    "largest_published_fit_gb": 415.41,
    "largest_published_fit_model": "GLM-5"
  },
  {
    "chip": "M4 (8-core GPU, 16 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 65.86,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 (10-core GPU, 16 GB)",
    "rows": 4,
    "verified_rows": 0,
    "model_count": 4,
    "models": [
      "Llama 2 7B",
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 2,
    "runtimes": [
      "llama.cpp",
      "llamafile"
    ],
    "max_context_tokens": 512,
    "prompt_eval_rows": 4,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 76.16,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": 3.56,
    "largest_published_fit_model": "Llama 2 7B"
  },
  {
    "chip": "M4 (10-core GPU, 24 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 75.43,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 (10-core GPU, 32 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 75.57,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 (16 GB)",
    "rows": 14,
    "verified_rows": 0,
    "model_count": 9,
    "models": [
      "DeepSeek R1 Distill Llama 8B",
      "Devstral Small 2 24B",
      "Llama 3.1 8B",
      "Ministral 3 8B",
      "Phi-4 14B",
      "Qwen3.5-4B",
      "Qwen3.5-9B",
      "Qwen3.5-27B",
      "Qwen3.5-35B-A3B"
    ],
    "runtime_count": 4,
    "runtimes": [
      "llama.cpp",
      "LM Studio",
      "MLX",
      "Ollama"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 14,
    "fastest_avg_tok_s": 92,
    "fastest_model": "Qwen3.5-4B",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 (32 GB)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "Gemma 3 27B"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llama.cpp"
    ],
    "max_context_tokens": 512,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 0,
    "fastest_avg_tok_s": 5.72,
    "fastest_model": "Gemma 3 27B",
    "fastest_quantization": "Q4_0",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 (GPU count not published, 16 GB)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "llama-3-2-1b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 1,
    "fastest_avg_tok_s": 67.95,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 Max",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "Qwen3.5-27B"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llama.cpp"
    ],
    "max_context_tokens": 16384,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 0,
    "fastest_avg_tok_s": 16.69,
    "fastest_model": "Qwen3.5-27B",
    "fastest_quantization": "Q4_K",
    "largest_published_fit_gb": 16.4,
    "largest_published_fit_model": "Qwen3.5-27B"
  },
  {
    "chip": "M4 Max (32-core GPU, 36 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 166.45,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 Max (40-core GPU, 48 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 178.99,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 Max (40-core GPU, 64 GB)",
    "rows": 14,
    "verified_rows": 1,
    "model_count": 6,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "Qwen 3 4B",
      "Qwen 3 30B-A3B",
      "Qwen 3 32B",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 3,
    "runtimes": [
      "llamafile",
      "MLX",
      "Ollama"
    ],
    "max_context_tokens": 128000,
    "prompt_eval_rows": 13,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 180.24,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": 29.78,
    "largest_published_fit_model": "Qwen 3 30B-A3B"
  },
  {
    "chip": "M4 Max (40-core GPU, 128 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 182.56,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 Max (48 GB)",
    "rows": 9,
    "verified_rows": 0,
    "model_count": 9,
    "models": [
      "DeepSeek R1 Distill Qwen 32B",
      "Gemma 3 27B",
      "Gemma 4 26B-A4B",
      "Gemma 4 31B",
      "Nemotron Cascade 2 30B-A3B",
      "Phi-4 Mini Instruct 3.8B",
      "Qwen 3 30B-A3B",
      "Qwen3.5-35B-A3B",
      "Qwen3.6-35B-A3B"
    ],
    "runtime_count": 3,
    "runtimes": [
      "LM Studio",
      "MLX",
      "Ollama"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 9,
    "fastest_avg_tok_s": 125,
    "fastest_model": "Phi-4 Mini Instruct 3.8B",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 Max (64 GB)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "Qwen 3 32B"
    ],
    "runtime_count": 1,
    "runtimes": [
      "MLX"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 1,
    "fastest_avg_tok_s": 22,
    "fastest_model": "Qwen 3 32B",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 Max (128 GB)",
    "rows": 19,
    "verified_rows": 0,
    "model_count": 12,
    "models": [
      "Devstral Small 1.1",
      "Gemma 3 4B",
      "Gemma 3 27B",
      "GLM-4.5-Air",
      "Llama 3.3 70B",
      "Qwen 3 8B",
      "Qwen 3 30B-A3B",
      "Qwen 3 32B",
      "Qwen 3 235B-A22B",
      "qwen-2-5-7b-instruct",
      "qwen-3-0-6b",
      "Qwen3.6-27B"
    ],
    "runtime_count": 4,
    "runtimes": [
      "llama.cpp",
      "LM Studio",
      "MLX",
      "Ollama"
    ],
    "max_context_tokens": 131072,
    "prompt_eval_rows": 6,
    "time_to_first_token_rows": 11,
    "fastest_avg_tok_s": 184.45,
    "fastest_model": "qwen-3-0-6b",
    "fastest_quantization": "Q8_0",
    "largest_published_fit_gb": 100,
    "largest_published_fit_model": "Qwen 3 235B-A22B"
  },
  {
    "chip": "M4 Max (GPU count not published, 128 GB)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "llama-3-2-1b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 1,
    "fastest_avg_tok_s": 156.3,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 Pro (16-core GPU, 24 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 110.95,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 Pro (16-core GPU, 48 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 111,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 Pro (16-core GPU, 64 GB)",
    "rows": 2,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 2,
    "time_to_first_token_rows": 2,
    "fastest_avg_tok_s": 111.94,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 Pro (20-core GPU, 24 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 119.25,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 Pro (20-core GPU, 48 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 118.9,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 Pro (20-core GPU, 64 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 118.56,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 Pro (24 GB)",
    "rows": 19,
    "verified_rows": 0,
    "model_count": 17,
    "models": [
      "Gemma 3 4B",
      "Gemma 3 12B",
      "Gemma 3 27B",
      "Gemma 4 26B-A4B",
      "Gemma 4 31B",
      "Gemma 4 E2B",
      "Gemma 4 E4B",
      "Ministral 3 14B",
      "Mistral 7B v0.3",
      "Nemotron Cascade 2 30B-A3B",
      "Phi-4 Mini Instruct 3.8B",
      "Qwen 3 4B",
      "Qwen 3 8B",
      "Qwen 3 14B",
      "Qwen 3 30B-A3B",
      "Qwen3.5-9B",
      "Qwen3.6-35B-A3B"
    ],
    "runtime_count": 3,
    "runtimes": [
      "LM Studio",
      "MLX",
      "Ollama"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 19,
    "fastest_avg_tok_s": 118,
    "fastest_model": "Qwen 3 4B",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 Pro (32 GB)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "Qwen 3 32B"
    ],
    "runtime_count": 1,
    "runtimes": [
      "Ollama"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 1,
    "fastest_avg_tok_s": 15,
    "fastest_model": "Qwen 3 32B",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 Pro (48 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "Devstral Small 1.1",
      "Qwen 3 30B-A3B",
      "Qwen3.5-27B"
    ],
    "runtime_count": 2,
    "runtimes": [
      "LM Studio",
      "MLX"
    ],
    "max_context_tokens": 131072,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 1,
    "fastest_avg_tok_s": 55,
    "fastest_model": "Qwen 3 30B-A3B",
    "fastest_quantization": "8bit",
    "largest_published_fit_gb": 18.51,
    "largest_published_fit_model": "Devstral Small 1.1"
  },
  {
    "chip": "M4 Pro (64 GB)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "Llama 3.3 70B"
    ],
    "runtime_count": 1,
    "runtimes": [
      "Ollama"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 1,
    "fastest_avg_tok_s": 5,
    "fastest_model": "Llama 3.3 70B",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M4 Ultra (192 GB)",
    "rows": 8,
    "verified_rows": 0,
    "model_count": 8,
    "models": [
      "DeepSeek R1 Distill Llama 70B",
      "gpt-oss 120B",
      "Llama 3.3 70B",
      "Llama 4 Scout 17B-16E",
      "Mistral Small 4 119B",
      "Qwen 2.5 72B",
      "Qwen 3 32B",
      "Qwen 3 235B-A22B"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 8,
    "fastest_avg_tok_s": 45,
    "fastest_model": "Mistral Small 4 119B",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M5 (10-core GPU, 16 GB)",
    "rows": 1,
    "verified_rows": 0,
    "model_count": 1,
    "models": [
      "llama-3-2-1b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 1,
    "time_to_first_token_rows": 1,
    "fastest_avg_tok_s": 98.08,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M5 (10-core GPU, 32 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 98.37,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M5 Max (32-core GPU, 36 GB)",
    "rows": 3,
    "verified_rows": 0,
    "model_count": 3,
    "models": [
      "llama-3-1-8b-instruct",
      "llama-3-2-1b-instruct",
      "qwen-2-5-14b-instruct"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llamafile"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 3,
    "time_to_first_token_rows": 3,
    "fastest_avg_tok_s": 228.99,
    "fastest_model": "llama-3-2-1b-instruct",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M5 Max (48 GB)",
    "rows": 4,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "Qwen3.5-27B",
      "Qwen3.5-35B-A3B"
    ],
    "runtime_count": 2,
    "runtimes": [
      "llama.cpp",
      "MLX"
    ],
    "max_context_tokens": 8000,
    "prompt_eval_rows": 4,
    "time_to_first_token_rows": 0,
    "fastest_avg_tok_s": 128,
    "fastest_model": "Qwen3.5-35B-A3B",
    "fastest_quantization": "4bit",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M5 Max (64 GB)",
    "rows": 23,
    "verified_rows": 0,
    "model_count": 20,
    "models": [
      "DeepSeek R1 Distill Llama 8B",
      "DeepSeek R1 Distill Qwen 32B",
      "Gemma 3 4B",
      "Gemma 3 12B",
      "Gemma 3 27B",
      "Gemma 4 31B",
      "Llama 3.1 8B",
      "Ministral 3 8B",
      "Ministral 3 14B",
      "Mistral 7B v0.3",
      "Nemotron Cascade 2 30B-A3B",
      "Phi-4 14B",
      "Phi-4 Mini Instruct 3.8B",
      "Qwen 3 4B",
      "Qwen 3 14B",
      "Qwen 3 30B-A3B",
      "Qwen3.5-4B",
      "Qwen3.5-9B",
      "Qwen3.5-35B-A3B",
      "Qwen3.6-35B-A3B"
    ],
    "runtime_count": 2,
    "runtimes": [
      "MLX",
      "Ollama"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 23,
    "fastest_avg_tok_s": 148,
    "fastest_model": "Qwen3.5-4B",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M5 Max (128 GB)",
    "rows": 33,
    "verified_rows": 0,
    "model_count": 23,
    "models": [
      "DeepSeek R1 Distill Llama 70B",
      "Gemma 3 27B",
      "Gemma 4 26B-A4B",
      "Gemma 4 31B",
      "Gemma 4 E2B",
      "Gemma 4 E4B",
      "gpt-oss 120B",
      "Llama 3.1 8B",
      "Llama 3.3 70B",
      "Llama 4 Scout 17B-16E",
      "Mistral Small 4 119B",
      "Qwen 2.5 72B",
      "Qwen 3 8B",
      "Qwen 3 14B",
      "Qwen 3 32B",
      "Qwen 3 235B-A22B",
      "Qwen3-Coder-Next",
      "Qwen3.5-9B",
      "Qwen3.5-27B",
      "Qwen3.5-35B-A3B",
      "Qwen3.5-122B-A10B",
      "Qwen3.5-397B-A17B",
      "Qwen3.6-35B-A3B"
    ],
    "runtime_count": 4,
    "runtimes": [
      "flash-moe",
      "llama.cpp",
      "MLX",
      "Ollama"
    ],
    "max_context_tokens": 65545,
    "prompt_eval_rows": 8,
    "time_to_first_token_rows": 22,
    "fastest_avg_tok_s": 158,
    "fastest_model": "Gemma 4 E2B",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": 92.61,
    "largest_published_fit_model": "Qwen3-Coder-Next"
  },
  {
    "chip": "M5 Pro (24 GB)",
    "rows": 2,
    "verified_rows": 0,
    "model_count": 2,
    "models": [
      "Gemma 4 26B-A4B",
      "Gemma 4 E4B"
    ],
    "runtime_count": 1,
    "runtimes": [
      "Ollama"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 2,
    "fastest_avg_tok_s": 92,
    "fastest_model": "Gemma 4 E4B",
    "fastest_quantization": "Q4_K - Medium",
    "largest_published_fit_gb": null,
    "largest_published_fit_model": null
  },
  {
    "chip": "M5 Pro (64 GB)",
    "rows": 4,
    "verified_rows": 0,
    "model_count": 4,
    "models": [
      "Qwen3.5-9B",
      "Qwen3.5-27B",
      "Qwen3.5-35B-A3B",
      "Qwen3.5-122B-A10B"
    ],
    "runtime_count": 1,
    "runtimes": [
      "llama.cpp"
    ],
    "max_context_tokens": null,
    "prompt_eval_rows": 0,
    "time_to_first_token_rows": 4,
    "fastest_avg_tok_s": 41.9,
    "fastest_model": "Qwen3.5-35B-A3B",
    "fastest_quantization": "Q4_K_L",
    "largest_published_fit_gb": 40.8,
    "largest_published_fit_model": "Qwen3.5-122B-A10B"
  }
]
