[
  {
    "chip": "M1 Max (64 GB)",
    "model": "GLM-4.7-Flash",
    "quantization": "Q4_K_XL",
    "runtime": "llama.cpp",
    "ram_required_gb": 17,
    "ram_required_note": "~17 GB",
    "context_tokens": 4096,
    "prompt_eval_tok_s": 99.4,
    "avg_tok_s": 36.8,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rdx2c7/ran_3_popular_30b_moe_models_on_my_apple_silicon/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M1 Max (64 GB)",
    "model": "Nemotron-3-Nano-30B-A3B",
    "quantization": "Q4_K_XL",
    "runtime": "llama.cpp",
    "ram_required_gb": 22,
    "ram_required_note": "~22 GB",
    "context_tokens": 4096,
    "prompt_eval_tok_s": 136.9,
    "avg_tok_s": 43.7,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rdx2c7/ran_3_popular_30b_moe_models_on_my_apple_silicon/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M1 Max (64 GB)",
    "model": "Qwen3-Coder-30B-A3B",
    "quantization": "IQ4_XS",
    "runtime": "llama.cpp",
    "ram_required_gb": 16.1,
    "ram_required_note": "~16.1 GB",
    "context_tokens": 4096,
    "prompt_eval_tok_s": 132.1,
    "avg_tok_s": 58.5,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rdx2c7/ran_3_popular_30b_moe_models_on_my_apple_silicon/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (40-core GPU, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": 20,
    "ram_required_note": "~20 GB",
    "context_tokens": 128000,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 22,
    "source": "Silicon Score Lab",
    "verified": true,
    "source_url": null,
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (40-core GPU, 64 GB)",
    "model": "Qwen 3 4B",
    "quantization": "Q4",
    "runtime": "MLX",
    "ram_required_gb": 2.54,
    "ram_required_note": "~2.54 GB",
    "context_tokens": 2048,
    "prompt_eval_tok_s": 2976.68,
    "avg_tok_s": 148.08,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/awni/c1790e4c3a39be6e8f1c4afd42423d2d",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (40-core GPU, 64 GB)",
    "model": "Qwen 3 4B",
    "quantization": "Q4_G32",
    "runtime": "MLX",
    "ram_required_gb": 2.78,
    "ram_required_note": "~2.78 GB",
    "context_tokens": 2048,
    "prompt_eval_tok_s": 2838.42,
    "avg_tok_s": 149.07,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/awni/c1790e4c3a39be6e8f1c4afd42423d2d",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (40-core GPU, 64 GB)",
    "model": "Qwen 3 4B",
    "quantization": "Q5",
    "runtime": "MLX",
    "ram_required_gb": 3.26,
    "ram_required_note": "~3.26 GB",
    "context_tokens": 2048,
    "prompt_eval_tok_s": 2736.15,
    "avg_tok_s": 143.2,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/awni/c1790e4c3a39be6e8f1c4afd42423d2d",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (40-core GPU, 64 GB)",
    "model": "Qwen 3 4B",
    "quantization": "Q5_G32",
    "runtime": "MLX",
    "ram_required_gb": 3.5,
    "ram_required_note": "~3.5 GB",
    "context_tokens": 2048,
    "prompt_eval_tok_s": 2754.53,
    "avg_tok_s": 142.98,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/awni/c1790e4c3a39be6e8f1c4afd42423d2d",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (40-core GPU, 64 GB)",
    "model": "Qwen 3 4B",
    "quantization": "Q6",
    "runtime": "MLX",
    "ram_required_gb": 3.98,
    "ram_required_note": "~3.98 GB",
    "context_tokens": 2048,
    "prompt_eval_tok_s": 2735.69,
    "avg_tok_s": 136.58,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/awni/c1790e4c3a39be6e8f1c4afd42423d2d",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (40-core GPU, 64 GB)",
    "model": "Qwen 3 4B",
    "quantization": "Q8",
    "runtime": "MLX",
    "ram_required_gb": 5.06,
    "ram_required_note": "~5.06 GB",
    "context_tokens": 2048,
    "prompt_eval_tok_s": 1780.63,
    "avg_tok_s": 111.55,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/awni/c1790e4c3a39be6e8f1c4afd42423d2d",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (40-core GPU, 64 GB)",
    "model": "Qwen 3 30B-A3B",
    "quantization": "Q4",
    "runtime": "MLX",
    "ram_required_gb": 16.12,
    "ram_required_note": "~16.12 GB",
    "context_tokens": 2048,
    "prompt_eval_tok_s": 822.59,
    "avg_tok_s": 92.09,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/awni/c1790e4c3a39be6e8f1c4afd42423d2d",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (40-core GPU, 64 GB)",
    "model": "Qwen 3 30B-A3B",
    "quantization": "Q5",
    "runtime": "MLX",
    "ram_required_gb": 18.09,
    "ram_required_note": "~18.09 GB",
    "context_tokens": 2048,
    "prompt_eval_tok_s": 819.83,
    "avg_tok_s": 84.89,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/awni/c1790e4c3a39be6e8f1c4afd42423d2d",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (40-core GPU, 64 GB)",
    "model": "Qwen 3 30B-A3B",
    "quantization": "Q6",
    "runtime": "MLX",
    "ram_required_gb": 21.87,
    "ram_required_note": "~21.87 GB",
    "context_tokens": 2048,
    "prompt_eval_tok_s": 817.63,
    "avg_tok_s": 76.74,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/awni/c1790e4c3a39be6e8f1c4afd42423d2d",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (40-core GPU, 64 GB)",
    "model": "Qwen 3 30B-A3B",
    "quantization": "Q8",
    "runtime": "MLX",
    "ram_required_gb": 29.78,
    "ram_required_note": "~29.78 GB",
    "context_tokens": 2048,
    "prompt_eval_tok_s": 772.59,
    "avg_tok_s": 52.58,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/awni/c1790e4c3a39be6e8f1c4afd42423d2d",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "qwen-3-0-6b",
    "quantization": "Q8_0",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 10000,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 184.45,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/estsauver/a70c929398479f3166f3d69bcededac3",
    "time_to_first_token_s": 0.04
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "Qwen 3 8B",
    "quantization": "Q4_K - Medium",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 10000,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 63.15,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/estsauver/a70c929398479f3166f3d69bcededac3",
    "time_to_first_token_s": 0.14
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "Qwen 3 30B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 10000,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 70.15,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/estsauver/a70c929398479f3166f3d69bcededac3",
    "time_to_first_token_s": 0.62
  },
  {
    "chip": "M4 Pro (48 GB)",
    "model": "Qwen 3 30B-A3B",
    "quantization": "8bit",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 55,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1m6tf9v/m4_pro_owners_i_want_your_biased_hottakes/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "Gemma 3 27B",
    "quantization": "Q8_0",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 131072,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 14.49,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/estsauver/a70c929398479f3166f3d69bcededac3",
    "time_to_first_token_s": 1.71
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "Gemma 3 27B",
    "quantization": "Q8_0",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 10000,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 12.99,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/estsauver/a70c929398479f3166f3d69bcededac3",
    "time_to_first_token_s": 0.79
  },
  {
    "chip": "M4 (32 GB)",
    "model": "Gemma 3 27B",
    "quantization": "Q4_0",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 512,
    "prompt_eval_tok_s": 47.51,
    "avg_tok_s": 5.72,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/zachrattner/52a6b56d70ed024b18c992ef14b89656",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "Gemma 3 4B",
    "quantization": "Q4_0",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 4096,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 100.54,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/estsauver/a70c929398479f3166f3d69bcededac3",
    "time_to_first_token_s": 0.14
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "qwen-2-5-7b-instruct",
    "quantization": "Q8_0",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 10000,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 49.67,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/estsauver/a70c929398479f3166f3d69bcededac3",
    "time_to_first_token_s": 0.26
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "Qwen 3 235B-A22B",
    "quantization": "Q4_K - Medium",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 10000,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 8.08,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/estsauver/a70c929398479f3166f3d69bcededac3",
    "time_to_first_token_s": 1.87
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "Qwen 3 235B-A22B",
    "quantization": "3bit",
    "runtime": "MLX",
    "ram_required_gb": 100,
    "ram_required_note": "~100 GB",
    "context_tokens": 1400,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 30,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kfmoyx/qwen3_235b_pairs_extremely_well_with_a_macbook/",
    "time_to_first_token_s": 14
  },
  {
    "chip": "M2 Ultra (192 GB)",
    "model": "Qwen 3 235B-A22B",
    "quantization": "Q4",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 26.4,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kr9d9b/how_are_you_running_qwen3235b_locally/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (256 GB)",
    "model": "Qwen 3 235B-A22B",
    "quantization": "Q4",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 27,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kr9d9b/how_are_you_running_qwen3235b_locally/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (512 GB)",
    "model": "Qwen 3 235B-A22B",
    "quantization": "4bit",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 27.36,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1lumsd2/mac_studio_512gb_online/",
    "time_to_first_token_s": 1.73
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "Llama 3.3 70B",
    "quantization": "4bit",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 11.8,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1i7b3r1/i_did_a_quick_test_of_macbook_m4_max_128_gb/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "Llama 3.3 70B",
    "quantization": "8bit",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 6.5,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1i7b3r1/i_did_a_quick_test_of_macbook_m4_max_128_gb/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "Llama 3.3 70B",
    "quantization": "Q8_0",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 35550,
    "prompt_eval_tok_s": 58,
    "avg_tok_s": 4.5,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1hax4ue/llama_32_3b_and_llama_33_70b_models_on_a_mac_mini/",
    "time_to_first_token_s": 600
  },
  {
    "chip": "M3 Ultra (512 GB)",
    "model": "Llama 3.3 70B",
    "quantization": "4bit",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 7800,
    "prompt_eval_tok_s": 150,
    "avg_tok_s": 15.5,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1jdvk7c/any_m3_ultra_test_requests_for_mlx_models_in_lm/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (512 GB)",
    "model": "Llama 3.3 70B",
    "quantization": "8bit",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 7800,
    "prompt_eval_tok_s": 150,
    "avg_tok_s": 8.5,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1jdvk7c/any_m3_ultra_test_requests_for_mlx_models_in_lm/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (512 GB)",
    "model": "Llama 3.3 70B",
    "quantization": "4bit",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 39686,
    "prompt_eval_tok_s": 103,
    "avg_tok_s": 9.57,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1jdvk7c/any_m3_ultra_test_requests_for_mlx_models_in_lm/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (512 GB)",
    "model": "Llama 3.3 70B",
    "quantization": "8bit",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 39686,
    "prompt_eval_tok_s": 101,
    "avg_tok_s": 6.53,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1jdvk7c/any_m3_ultra_test_requests_for_mlx_models_in_lm/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M1 Ultra (64 GB)",
    "model": "Llama 3.3 70B",
    "quantization": "4bit",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 3991,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 12.55,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1he2v2n/speed_test_llama3370b_on_2xrtx3090_vs_m3max_64gb/",
    "time_to_first_token_s": 49.8
  },
  {
    "chip": "M2 Max (38-core GPU, 64 GB)",
    "model": "Llama 3.3 70B",
    "quantization": "4bit",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 8.8,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1hvwwsq/exolab_nvidias_digits_outperforms_apples_m4_chips/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Llama 3.3 70B",
    "quantization": "Q4_K - Medium",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 258,
    "prompt_eval_tok_s": 67.86,
    "avg_tok_s": 8.15,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1he2v2n/speed_test_llama3370b_on_2xrtx3090_vs_m3max_64gb/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Llama 3.3 70B",
    "quantization": "Q4_K - Medium",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 8013,
    "prompt_eval_tok_s": 65.17,
    "avg_tok_s": 7.48,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1he2v2n/speed_test_llama3370b_on_2xrtx3090_vs_m3max_64gb/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Llama 3.3 70B",
    "quantization": "Q4_K - Medium",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 16001,
    "prompt_eval_tok_s": 59.5,
    "avg_tok_s": 7,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1he2v2n/speed_test_llama3370b_on_2xrtx3090_vs_m3max_64gb/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Llama 3.3 70B",
    "quantization": "Q4_K - Medium",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 32170,
    "prompt_eval_tok_s": 50.32,
    "avg_tok_s": 6.13,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1he2v2n/speed_test_llama3370b_on_2xrtx3090_vs_m3max_64gb/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 264,
    "prompt_eval_tok_s": 153.63,
    "avg_tok_s": 10.41,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 1.72
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 264,
    "prompt_eval_tok_s": 152.12,
    "avg_tok_s": 10.35,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 1.74
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 450,
    "prompt_eval_tok_s": 171.37,
    "avg_tok_s": 10.28,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 2.63
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 450,
    "prompt_eval_tok_s": 169.53,
    "avg_tok_s": 10.33,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 2.65
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 723,
    "prompt_eval_tok_s": 164.83,
    "avg_tok_s": 10.29,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 4.39
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 723,
    "prompt_eval_tok_s": 163.79,
    "avg_tok_s": 10.27,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 4.41
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 1219,
    "prompt_eval_tok_s": 169.15,
    "avg_tok_s": 10.19,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 7.21
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 1219,
    "prompt_eval_tok_s": 168.32,
    "avg_tok_s": 10.11,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 7.24
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 1858,
    "prompt_eval_tok_s": 166.81,
    "avg_tok_s": 10.09,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 11.14
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 1858,
    "prompt_eval_tok_s": 166.96,
    "avg_tok_s": 10.1,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 11.13
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 2979,
    "prompt_eval_tok_s": 162.22,
    "avg_tok_s": 9.89,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 18.36
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 2979,
    "prompt_eval_tok_s": 161.46,
    "avg_tok_s": 9.88,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 18.45
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 4669,
    "prompt_eval_tok_s": 154.16,
    "avg_tok_s": 9.67,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 30.29
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 4669,
    "prompt_eval_tok_s": 153.03,
    "avg_tok_s": 9.66,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 30.51
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 7948,
    "prompt_eval_tok_s": 140.11,
    "avg_tok_s": 9.2,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 56.73
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 7948,
    "prompt_eval_tok_s": 138.99,
    "avg_tok_s": 9.18,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 57.18
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 12416,
    "prompt_eval_tok_s": 127.96,
    "avg_tok_s": 8.6,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 97.03
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 12416,
    "prompt_eval_tok_s": 127.08,
    "avg_tok_s": 8.57,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 97.7
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 20172,
    "prompt_eval_tok_s": 111.18,
    "avg_tok_s": 7.58,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 181.44
  },
  {
    "chip": "M3 Max (GPU count not published, 64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q8_0",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 20172,
    "prompt_eval_tok_s": 111.8,
    "avg_tok_s": 7.53,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kk0ghi/speed_comparison_with_qwen332bq8_0_ollama/",
    "time_to_first_token_s": 180.43
  },
  {
    "chip": "M4 Pro (64 GB)",
    "model": "Llama 3.3 70B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 5,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1hax4ue/llama_32_3b_and_llama_33_70b_models_on_a_mac_mini/",
    "time_to_first_token_s": 3.75
  },
  {
    "chip": "M1 Pro (16-core GPU)",
    "model": "Llama 2 7B",
    "quantization": "Q4_0",
    "runtime": "llama.cpp",
    "ram_required_gb": 3.56,
    "ram_required_note": "~3.56 GB",
    "context_tokens": 512,
    "prompt_eval_tok_s": 266.25,
    "avg_tok_s": 36.41,
    "source": "reference run",
    "verified": false,
    "source_url": "https://github.com/ggml-org/llama.cpp/discussions/4167",
    "time_to_first_token_s": null
  },
  {
    "chip": "M2 Ultra (76-core GPU, 192 GB)",
    "model": "Llama 2 7B",
    "quantization": "Q4_0",
    "runtime": "llama.cpp",
    "ram_required_gb": 3.56,
    "ram_required_note": "~3.56 GB",
    "context_tokens": 512,
    "prompt_eval_tok_s": 1238.48,
    "avg_tok_s": 94.27,
    "source": "reference run",
    "verified": false,
    "source_url": "https://github.com/ggml-org/llama.cpp/discussions/4167",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Pro (18-core GPU)",
    "model": "Llama 2 7B",
    "quantization": "Q4_0",
    "runtime": "llama.cpp",
    "ram_required_gb": 3.56,
    "ram_required_note": "~3.56 GB",
    "context_tokens": 512,
    "prompt_eval_tok_s": 341.67,
    "avg_tok_s": 30.74,
    "source": "reference run",
    "verified": false,
    "source_url": "https://github.com/ggml-org/llama.cpp/discussions/4167",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Max (40-core GPU, 48 GB)",
    "model": "Llama 2 7B",
    "quantization": "Q4_0",
    "runtime": "llama.cpp",
    "ram_required_gb": 3.56,
    "ram_required_note": "~3.56 GB",
    "context_tokens": 512,
    "prompt_eval_tok_s": 690.99,
    "avg_tok_s": 65.85,
    "source": "reference run",
    "verified": false,
    "source_url": "https://github.com/ggml-org/llama.cpp/discussions/4167",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 (10-core GPU, 16 GB)",
    "model": "Llama 2 7B",
    "quantization": "Q4_0",
    "runtime": "llama.cpp",
    "ram_required_gb": 3.56,
    "ram_required_note": "~3.56 GB",
    "context_tokens": 512,
    "prompt_eval_tok_s": 221.29,
    "avg_tok_s": 24.11,
    "source": "reference run",
    "verified": false,
    "source_url": "https://github.com/ggml-org/llama.cpp/discussions/4167",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q4_K - Medium",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 10000,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 11.72,
    "source": "reference run",
    "verified": false,
    "source_url": "https://gist.github.com/estsauver/a70c929398479f3166f3d69bcededac3",
    "time_to_first_token_s": 0.81
  },
  {
    "chip": "M1 (7-core GPU, 8 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 108.71905633333331,
    "avg_tok_s": 13.385236666666666,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1281",
    "time_to_first_token_s": 11.513906754444445
  },
  {
    "chip": "M1 (7-core GPU, 8 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 532.9676827037036,
    "avg_tok_s": 38.51773918518518,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1281",
    "time_to_first_token_s": 2.4294918625925925
  },
  {
    "chip": "M1 (7-core GPU, 16 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 83.93792277777777,
    "avg_tok_s": 9.387483333333332,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/223",
    "time_to_first_token_s": 14.82420303238889
  },
  {
    "chip": "M1 (7-core GPU, 16 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 530.8859519722223,
    "avg_tok_s": 37.872518722222225,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/223",
    "time_to_first_token_s": 2.4528570579444446
  },
  {
    "chip": "M1 (7-core GPU, 16 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 40.612479388888886,
    "avg_tok_s": 4.806673055555556,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/223",
    "time_to_first_token_s": 30.980614497666668
  },
  {
    "chip": "M1 (8-core GPU, 8 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 133.8709728888889,
    "avg_tok_s": 14.552260666666667,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/17",
    "time_to_first_token_s": 9.648145671333332
  },
  {
    "chip": "M1 (8-core GPU, 8 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 607.4231519333333,
    "avg_tok_s": 40.41441422222222,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/17",
    "time_to_first_token_s": 2.1677008980666663
  },
  {
    "chip": "M1 (8-core GPU, 16 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 585.7259774074074,
    "avg_tok_s": 40.18323162962963,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/432",
    "time_to_first_token_s": 2.24642097837037
  },
  {
    "chip": "M1 (8-core GPU, 16 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 53.26808444444444,
    "avg_tok_s": 5.429473666666667,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/432",
    "time_to_first_token_s": 24.55238074088889
  },
  {
    "chip": "M1 Max (24-core GPU, 32 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1582.032366,
    "avg_tok_s": 93.859194,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1943",
    "time_to_first_token_s": 0.7243009050000001
  },
  {
    "chip": "M1 Max (24-core GPU, 32 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 155.93443555555555,
    "avg_tok_s": 17.355148333333332,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1943",
    "time_to_first_token_s": 7.862980407555557
  },
  {
    "chip": "M1 Max (24-core GPU, 64 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 306.18835666666666,
    "avg_tok_s": 32.11692033333334,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/688",
    "time_to_first_token_s": 3.9939120186666677
  },
  {
    "chip": "M1 Max (24-core GPU, 64 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1669.4850333333334,
    "avg_tok_s": 105.66463296296295,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/688",
    "time_to_first_token_s": 0.6893040093333332
  },
  {
    "chip": "M1 Max (24-core GPU, 64 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 140.19277200000002,
    "avg_tok_s": 15.135508527777777,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/688",
    "time_to_first_token_s": 8.830140840277778
  },
  {
    "chip": "M1 Max (32-core GPU, 32 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 359.8733261944444,
    "avg_tok_s": 35.40364952777778,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/597",
    "time_to_first_token_s": 3.4528548829722223
  },
  {
    "chip": "M1 Max (32-core GPU, 32 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 2025.1921614074072,
    "avg_tok_s": 125.77495338888887,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/597",
    "time_to_first_token_s": 0.5596487252407406
  },
  {
    "chip": "M1 Max (32-core GPU, 32 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 195.8191797407407,
    "avg_tok_s": 20.07296611111111,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/597",
    "time_to_first_token_s": 6.301562639796296
  },
  {
    "chip": "M1 Max (32-core GPU, 64 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 380.7280225277778,
    "avg_tok_s": 37.75919747222222,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/364",
    "time_to_first_token_s": 3.12193795138889
  },
  {
    "chip": "M1 Max (32-core GPU, 64 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1972.074889361111,
    "avg_tok_s": 120.67175636111112,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/364",
    "time_to_first_token_s": 0.5782512951666667
  },
  {
    "chip": "M1 Max (32-core GPU, 64 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 185.93018677777775,
    "avg_tok_s": 19.01644427777778,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/364",
    "time_to_first_token_s": 6.8818776490777775
  },
  {
    "chip": "M1 Pro (14-core GPU, 16 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 176.96636692592594,
    "avg_tok_s": 20.14017855555555,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/412",
    "time_to_first_token_s": 6.991519057185186
  },
  {
    "chip": "M1 Pro (14-core GPU, 16 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1040.067598027778,
    "avg_tok_s": 71.81051661111111,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/412",
    "time_to_first_token_s": 1.1483909050555556
  },
  {
    "chip": "M1 Pro (14-core GPU, 16 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 92.4920569777778,
    "avg_tok_s": 10.8301478,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/412",
    "time_to_first_token_s": 13.4879233686
  },
  {
    "chip": "M1 Pro (14-core GPU, 32 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 173.08191844444443,
    "avg_tok_s": 19.971878777777775,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/418",
    "time_to_first_token_s": 7.230578513999999
  },
  {
    "chip": "M1 Pro (14-core GPU, 32 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1063.8479185555557,
    "avg_tok_s": 71.01946466666666,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/418",
    "time_to_first_token_s": 1.1336526296666667
  },
  {
    "chip": "M1 Pro (14-core GPU, 32 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 88.86977233333333,
    "avg_tok_s": 10.399665111111112,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1602",
    "time_to_first_token_s": 14.166348648222224
  },
  {
    "chip": "M1 Pro (16-core GPU, 16 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 204.5403022222222,
    "avg_tok_s": 21.88205522222222,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/929",
    "time_to_first_token_s": 6.089909109518518
  },
  {
    "chip": "M1 Pro (16-core GPU, 16 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1158.147661425926,
    "avg_tok_s": 78.15808337037036,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/929",
    "time_to_first_token_s": 1.0274150409074074
  },
  {
    "chip": "M1 Pro (16-core GPU, 16 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 106.6591231111111,
    "avg_tok_s": 11.90798088888889,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/929",
    "time_to_first_token_s": 11.83519856488889
  },
  {
    "chip": "M1 Pro (16-core GPU, 32 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 200.28393007407408,
    "avg_tok_s": 21.698664740740742,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/4",
    "time_to_first_token_s": 6.2272433256296305
  },
  {
    "chip": "M1 Pro (16-core GPU, 32 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1165.9555193555557,
    "avg_tok_s": 77.23935995555556,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/4",
    "time_to_first_token_s": 1.0365038463555556
  },
  {
    "chip": "M1 Pro (16-core GPU, 32 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 104.4587133888889,
    "avg_tok_s": 11.619242305555556,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/4",
    "time_to_first_token_s": 12.083205306861112
  },
  {
    "chip": "M1 Ultra (48-core GPU, 128 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 534.3503048148149,
    "avg_tok_s": 48.89361162962964,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/306",
    "time_to_first_token_s": 2.1556520740370373
  },
  {
    "chip": "M1 Ultra (48-core GPU, 128 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 2582.382230222222,
    "avg_tok_s": 137.97023270370372,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/306",
    "time_to_first_token_s": 0.4251138565185186
  },
  {
    "chip": "M1 Ultra (48-core GPU, 128 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 289.8969081111111,
    "avg_tok_s": 27.773420277777777,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/306",
    "time_to_first_token_s": 4.044066250166667
  },
  {
    "chip": "M1 Ultra (64-core GPU, 128 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 667.5689454444446,
    "avg_tok_s": 54.30790400000001,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1087",
    "time_to_first_token_s": 1.6929770138888889
  },
  {
    "chip": "M1 Ultra (64-core GPU, 128 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 2977.086063277778,
    "avg_tok_s": 151.0678697777778,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1087",
    "time_to_first_token_s": 0.3675812592777778
  },
  {
    "chip": "M1 Ultra (64-core GPU, 128 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 371.7215576111111,
    "avg_tok_s": 32.417089222222224,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1087",
    "time_to_first_token_s": 3.126813851888889
  },
  {
    "chip": "M1 Ultra (GPU count not published, 128 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 71.55644355555556,
    "avg_tok_s": 15.24847544444444,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/307",
    "time_to_first_token_s": 17.47675613911111
  },
  {
    "chip": "M1 Ultra (GPU count not published, 128 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 283.7757922222222,
    "avg_tok_s": 57.07094177777779,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/307",
    "time_to_first_token_s": 4.787078634222223
  },
  {
    "chip": "M1 Ultra (GPU count not published, 128 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 38.83732133333333,
    "avg_tok_s": 7.9883163333333345,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/307",
    "time_to_first_token_s": 32.935429014111115
  },
  {
    "chip": "M2 (8-core GPU, 8 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 148.82072661111113,
    "avg_tok_s": 18.342969166666663,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/245",
    "time_to_first_token_s": 8.390452365666668
  },
  {
    "chip": "M2 (8-core GPU, 8 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 523.2109411111111,
    "avg_tok_s": 34.48367238888889,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/245",
    "time_to_first_token_s": 2.7225200694166665
  },
  {
    "chip": "M2 (8-core GPU, 16 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 113.95904577777776,
    "avg_tok_s": 12.906848888888891,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1939",
    "time_to_first_token_s": 10.946365963333333
  },
  {
    "chip": "M2 (8-core GPU, 16 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 523.1656983888889,
    "avg_tok_s": 35.320437777777784,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1939",
    "time_to_first_token_s": 3.046989798555555
  },
  {
    "chip": "M2 (8-core GPU, 16 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 59.9820798888889,
    "avg_tok_s": 6.960648666666668,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1939",
    "time_to_first_token_s": 21.432082513888886
  },
  {
    "chip": "M2 (10-core GPU, 8 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 804.8249559999999,
    "avg_tok_s": 56.46111377777777,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/847",
    "time_to_first_token_s": 1.6694184073333336
  },
  {
    "chip": "M2 (10-core GPU, 16 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 140.73035655555557,
    "avg_tok_s": 14.68716522222222,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/499",
    "time_to_first_token_s": 9.125759541666666
  },
  {
    "chip": "M2 (10-core GPU, 16 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 801.8366849333333,
    "avg_tok_s": 55.565495822222225,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/499",
    "time_to_first_token_s": 1.589305650066667
  },
  {
    "chip": "M2 (10-core GPU, 16 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 74.75765844444444,
    "avg_tok_s": 8.076025111111111,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/499",
    "time_to_first_token_s": 17.296274291666677
  },
  {
    "chip": "M2 (10-core GPU, 24 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 141.73761555555558,
    "avg_tok_s": 14.668569888888888,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/202",
    "time_to_first_token_s": 9.023242666666668
  },
  {
    "chip": "M2 (10-core GPU, 24 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 813.7094306666667,
    "avg_tok_s": 54.7565285,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/202",
    "time_to_first_token_s": 1.5797905915
  },
  {
    "chip": "M2 (10-core GPU, 24 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 70.25649544444445,
    "avg_tok_s": 7.345236666666668,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/202",
    "time_to_first_token_s": 18.027677185222224
  },
  {
    "chip": "M2 Max (30-core GPU, 32 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 325.001513,
    "avg_tok_s": 31.185154777777775,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1079",
    "time_to_first_token_s": 3.7657567592222216
  },
  {
    "chip": "M2 Max (30-core GPU, 32 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 2031.665092888889,
    "avg_tok_s": 127.58507908333334,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1079",
    "time_to_first_token_s": 0.5575502119166668
  },
  {
    "chip": "M2 Max (30-core GPU, 64 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 149.0023927777778,
    "avg_tok_s": 14.487775777777776,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1473",
    "time_to_first_token_s": 7.678328407666665
  },
  {
    "chip": "M2 Max (38-core GPU, 32 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 473.660562,
    "avg_tok_s": 44.66334788888888,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1291",
    "time_to_first_token_s": 2.4878861063333333
  },
  {
    "chip": "M2 Max (38-core GPU, 32 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 2551.365566666667,
    "avg_tok_s": 152.97702255555558,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1291",
    "time_to_first_token_s": 0.44196865744444436
  },
  {
    "chip": "M2 Max (38-core GPU, 32 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 225.5740881111111,
    "avg_tok_s": 20.586609111111116,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1291",
    "time_to_first_token_s": 4.983570564777778
  },
  {
    "chip": "M2 Max (38-core GPU, 64 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 222.7697787777778,
    "avg_tok_s": 21.958976814814815,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1577",
    "time_to_first_token_s": 5.511166165148148
  },
  {
    "chip": "M2 Max (38-core GPU, 96 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 484.057812,
    "avg_tok_s": 46.41150744444444,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/515",
    "time_to_first_token_s": 2.4422629954444446
  },
  {
    "chip": "M2 Max (38-core GPU, 96 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 252.470057,
    "avg_tok_s": 25.230792444444447,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/515",
    "time_to_first_token_s": 4.798504768333334
  },
  {
    "chip": "M2 Pro (16-core GPU, 16 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 224.56031674074077,
    "avg_tok_s": 24.331960740740744,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1220",
    "time_to_first_token_s": 5.574148605037037
  },
  {
    "chip": "M2 Pro (16-core GPU, 16 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1328.2018266666669,
    "avg_tok_s": 91.14859188888887,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1220",
    "time_to_first_token_s": 0.9017461759444444
  },
  {
    "chip": "M2 Pro (16-core GPU, 16 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 118.980526,
    "avg_tok_s": 13.350706388888891,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1220",
    "time_to_first_token_s": 10.621650037111111
  },
  {
    "chip": "M2 Pro (16-core GPU, 32 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 208.68323722222223,
    "avg_tok_s": 23.809723111111115,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1298",
    "time_to_first_token_s": 6.06785997211111
  },
  {
    "chip": "M2 Pro (16-core GPU, 32 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1281.403522111111,
    "avg_tok_s": 91.45281944444444,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1298",
    "time_to_first_token_s": 0.9141989862222222
  },
  {
    "chip": "M2 Pro (19-core GPU, 16 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1457.8697797777777,
    "avg_tok_s": 99.52299688888888,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1862",
    "time_to_first_token_s": 0.8091630742222221
  },
  {
    "chip": "M2 Pro (19-core GPU, 32 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 261.761128,
    "avg_tok_s": 26.311742333333328,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/470",
    "time_to_first_token_s": 4.701914784666666
  },
  {
    "chip": "M2 Pro (19-core GPU, 32 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1487.5642871333334,
    "avg_tok_s": 100.2557703111111,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/470",
    "time_to_first_token_s": 0.8063303454888889
  },
  {
    "chip": "M2 Pro (19-core GPU, 32 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 137.32948566666667,
    "avg_tok_s": 14.116209888888887,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/470",
    "time_to_first_token_s": 9.101328361111113
  },
  {
    "chip": "M2 Ultra (60-core GPU, 64 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 703.2592348888888,
    "avg_tok_s": 59.480982,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/2309",
    "time_to_first_token_s": 1.6398754306666665
  },
  {
    "chip": "M2 Ultra (60-core GPU, 64 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 3290.1508422222223,
    "avg_tok_s": 174.08661300000003,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/2309",
    "time_to_first_token_s": 0.3351846761111111
  },
  {
    "chip": "M2 Ultra (60-core GPU, 64 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 381.12729266666673,
    "avg_tok_s": 34.21236922222223,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/2309",
    "time_to_first_token_s": 3.0647734165555556
  },
  {
    "chip": "M2 Ultra (60-core GPU, 128 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 3295.522693222222,
    "avg_tok_s": 176.3882528888889,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/617",
    "time_to_first_token_s": 0.33357317588888885
  },
  {
    "chip": "M2 Ultra (60-core GPU, 192 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 3272.209784333334,
    "avg_tok_s": 169.76082655555555,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/273",
    "time_to_first_token_s": 0.3388111018888889
  },
  {
    "chip": "M2 Ultra (76-core GPU, 128 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 470.5342577777778,
    "avg_tok_s": 36.626677,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1222",
    "time_to_first_token_s": 2.484028763888889
  },
  {
    "chip": "M2 Ultra (GPU count not published, 128 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 659.8306766666667,
    "avg_tok_s": 120.4092842222222,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/621",
    "time_to_first_token_s": 1.9676880275555553
  },
  {
    "chip": "M3 (10-core GPU, 16 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 138.71456447222224,
    "avg_tok_s": 13.499802861111112,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/414",
    "time_to_first_token_s": 9.024639849583332
  },
  {
    "chip": "M3 (10-core GPU, 16 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 931.5106880888889,
    "avg_tok_s": 67.18826562222223,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/414",
    "time_to_first_token_s": 1.4093825610888888
  },
  {
    "chip": "M3 (10-core GPU, 24 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 109.8980422777778,
    "avg_tok_s": 10.22738411111111,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/562",
    "time_to_first_token_s": 11.670486025444445
  },
  {
    "chip": "M3 (10-core GPU, 24 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 916.5074562888888,
    "avg_tok_s": 64.71307648888889,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/562",
    "time_to_first_token_s": 1.4258484739777777
  },
  {
    "chip": "M3 (10-core GPU, 24 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 65.80348074074074,
    "avg_tok_s": 6.133619851851852,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/562",
    "time_to_first_token_s": 19.91686526244444
  },
  {
    "chip": "M3 (GPU count not published, 16 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 33.58097844444444,
    "avg_tok_s": 11.83115477777778,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1139",
    "time_to_first_token_s": 40.49901329166667
  },
  {
    "chip": "M3 (GPU count not published, 16 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 229.7350115555556,
    "avg_tok_s": 61.584239,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1139",
    "time_to_first_token_s": 6.018140875
  },
  {
    "chip": "M3 Max (30-core GPU, 36 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 443.4985221666666,
    "avg_tok_s": 37.47979072222222,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/805",
    "time_to_first_token_s": 2.725718886444444
  },
  {
    "chip": "M3 Max (30-core GPU, 36 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 2553.2959846666668,
    "avg_tok_s": 132.9925948888889,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/805",
    "time_to_first_token_s": 0.4669405787777778
  },
  {
    "chip": "M3 Max (30-core GPU, 36 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 226.180183,
    "avg_tok_s": 19.768525888888888,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/805",
    "time_to_first_token_s": 5.407046111222224
  },
  {
    "chip": "M3 Max (30-core GPU, 96 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 456.8050157777777,
    "avg_tok_s": 37.74030144444444,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/789",
    "time_to_first_token_s": 2.6877091111111118
  },
  {
    "chip": "M3 Max (30-core GPU, 96 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 2519.9553065555556,
    "avg_tok_s": 132.87379466666664,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/789",
    "time_to_first_token_s": 0.4758228193333332
  },
  {
    "chip": "M3 Max (30-core GPU, 96 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 238.76133155555556,
    "avg_tok_s": 20.766045,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/789",
    "time_to_first_token_s": 5.162620243055555
  },
  {
    "chip": "M3 Max (40-core GPU, 48 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 3399.1241412222225,
    "avg_tok_s": 148.95889516666665,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1114",
    "time_to_first_token_s": 0.34880203711111113
  },
  {
    "chip": "M3 Max (40-core GPU, 64 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 377.04858177777777,
    "avg_tok_s": 25.448832000000003,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1469",
    "time_to_first_token_s": 3.3699854838888887
  },
  {
    "chip": "M3 Max (40-core GPU, 64 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 2521.617585666667,
    "avg_tok_s": 106.97443855555557,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1469",
    "time_to_first_token_s": 0.520643525388889
  },
  {
    "chip": "M3 Max (40-core GPU, 64 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 200.82679227777783,
    "avg_tok_s": 13.801572,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1469",
    "time_to_first_token_s": 6.657162884277778
  },
  {
    "chip": "M3 Max (40-core GPU, 128 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 587.0791163333333,
    "avg_tok_s": 45.792675277777775,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/750",
    "time_to_first_token_s": 2.0508701621111114
  },
  {
    "chip": "M3 Max (40-core GPU, 128 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 3291.9091724444447,
    "avg_tok_s": 146.33974311111112,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/750",
    "time_to_first_token_s": 0.3584107959999999
  },
  {
    "chip": "M3 Max (40-core GPU, 128 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 302.37069566666673,
    "avg_tok_s": 25.48905755555555,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/750",
    "time_to_first_token_s": 4.070913819333334
  },
  {
    "chip": "M3 Pro (14-core GPU, 18 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 199.51528166666668,
    "avg_tok_s": 19.10761944444444,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/263",
    "time_to_first_token_s": 6.427559875000002
  },
  {
    "chip": "M3 Pro (14-core GPU, 18 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1343.9570948888888,
    "avg_tok_s": 88.10604988888889,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/263",
    "time_to_first_token_s": 0.9594814676666664
  },
  {
    "chip": "M3 Pro (14-core GPU, 18 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 117.03684152777775,
    "avg_tok_s": 11.886003166666669,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/263",
    "time_to_first_token_s": 10.918680665472221
  },
  {
    "chip": "M3 Pro (14-core GPU, 36 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 223.43918544444443,
    "avg_tok_s": 21.455100444444447,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/862",
    "time_to_first_token_s": 5.651565199111111
  },
  {
    "chip": "M3 Pro (14-core GPU, 36 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1327.8959733333334,
    "avg_tok_s": 88.23620972222221,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/862",
    "time_to_first_token_s": 0.9533793772777779
  },
  {
    "chip": "M3 Pro (14-core GPU, 36 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 119.77962366666664,
    "avg_tok_s": 12.10437772222222,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/862",
    "time_to_first_token_s": 10.648140152777778
  },
  {
    "chip": "M3 Pro (18-core GPU, 18 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 279.21399711111115,
    "avg_tok_s": 20.80849266666667,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/401",
    "time_to_first_token_s": 4.529956342555556
  },
  {
    "chip": "M3 Pro (18-core GPU, 18 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1573.6998116666664,
    "avg_tok_s": 85.63227766666667,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/401",
    "time_to_first_token_s": 0.8258373006666666
  },
  {
    "chip": "M3 Pro (18-core GPU, 18 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 144.77590885185188,
    "avg_tok_s": 11.63579448148148,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/401",
    "time_to_first_token_s": 8.883480046185184
  },
  {
    "chip": "M3 Pro (18-core GPU, 36 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 283.8059724444445,
    "avg_tok_s": 22.06217488888889,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/428",
    "time_to_first_token_s": 4.460495907444445
  },
  {
    "chip": "M3 Pro (18-core GPU, 36 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1586.168225333333,
    "avg_tok_s": 89.75982788888888,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/428",
    "time_to_first_token_s": 0.8094999791666667
  },
  {
    "chip": "M3 Pro (18-core GPU, 36 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 147.43616580555556,
    "avg_tok_s": 11.988618277777778,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/428",
    "time_to_first_token_s": 8.729495730361112
  },
  {
    "chip": "M3 Ultra (60-core GPU, 96 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 444.62243322222224,
    "avg_tok_s": 34.43578599999999,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/479",
    "time_to_first_token_s": 2.672898578888889
  },
  {
    "chip": "M3 Ultra (80-core GPU, 256 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1062.184084222222,
    "avg_tok_s": 63.304543444444434,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1359",
    "time_to_first_token_s": 1.1019065137777775
  },
  {
    "chip": "M3 Ultra (80-core GPU, 256 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 4999.496534111112,
    "avg_tok_s": 177.900228,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1359",
    "time_to_first_token_s": 0.22653121744444446
  },
  {
    "chip": "M3 Ultra (80-core GPU, 256 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 568.2534414444444,
    "avg_tok_s": 36.711573666666666,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1359",
    "time_to_first_token_s": 2.077086245444445
  },
  {
    "chip": "M3 Ultra (80-core GPU, 512 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1109.4627288333336,
    "avg_tok_s": 62.67566477777777,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/404",
    "time_to_first_token_s": 1.060675240722222
  },
  {
    "chip": "M3 Ultra (80-core GPU, 512 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 5600.99501711111,
    "avg_tok_s": 178.83792144444442,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/404",
    "time_to_first_token_s": 0.20634095138888886
  },
  {
    "chip": "M3 Ultra (80-core GPU, 512 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 577.4241827222222,
    "avg_tok_s": 35.80064133333334,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/404",
    "time_to_first_token_s": 2.0507156840555556
  },
  {
    "chip": "M4 (8-core GPU, 16 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 134.02995044444447,
    "avg_tok_s": 15.26954622222222,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1394",
    "time_to_first_token_s": 9.18359174066667
  },
  {
    "chip": "M4 (8-core GPU, 16 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 897.0363834222222,
    "avg_tok_s": 65.86037953333333,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1394",
    "time_to_first_token_s": 1.4411192250444445
  },
  {
    "chip": "M4 (8-core GPU, 16 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 61.06916633333333,
    "avg_tok_s": 7.234274277777778,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1394",
    "time_to_first_token_s": 20.47596530788889
  },
  {
    "chip": "M4 (10-core GPU, 16 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 166.82581797222224,
    "avg_tok_s": 15.983290222222225,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/32",
    "time_to_first_token_s": 7.980498432833332
  },
  {
    "chip": "M4 (10-core GPU, 16 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1091.1117940111112,
    "avg_tok_s": 76.16243562222222,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/32",
    "time_to_first_token_s": 1.1983527194666668
  },
  {
    "chip": "M4 (10-core GPU, 16 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 83.05415,
    "avg_tok_s": 8.726884422222223,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/32",
    "time_to_first_token_s": 15.662719702777778
  },
  {
    "chip": "M4 (10-core GPU, 24 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 148.96748985185187,
    "avg_tok_s": 15.935378370370373,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1085",
    "time_to_first_token_s": 8.236942598851853
  },
  {
    "chip": "M4 (10-core GPU, 24 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1035.9960293703705,
    "avg_tok_s": 75.4330094074074,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1085",
    "time_to_first_token_s": 1.2527001605185184
  },
  {
    "chip": "M4 (10-core GPU, 24 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 93.30326355555556,
    "avg_tok_s": 9.238281555555554,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1085",
    "time_to_first_token_s": 13.714502268555552
  },
  {
    "chip": "M4 (10-core GPU, 32 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 166.14551366666666,
    "avg_tok_s": 16.809439277777777,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/546",
    "time_to_first_token_s": 7.554811377277778
  },
  {
    "chip": "M4 (10-core GPU, 32 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1069.8806100277775,
    "avg_tok_s": 75.56749613888888,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/546",
    "time_to_first_token_s": 1.2130988726666667
  },
  {
    "chip": "M4 (10-core GPU, 32 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 79.33710781481481,
    "avg_tok_s": 8.58771625925926,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/546",
    "time_to_first_token_s": 15.81842497840741
  },
  {
    "chip": "M4 (GPU count not published, 16 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 238.95714649999996,
    "avg_tok_s": 67.95033205555555,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/35",
    "time_to_first_token_s": 5.743805437444445
  },
  {
    "chip": "M4 Max (32-core GPU, 36 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 570.2711772,
    "avg_tok_s": 48.071329866666666,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/162",
    "time_to_first_token_s": 2.1545828000444445
  },
  {
    "chip": "M4 Max (32-core GPU, 36 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 3268.880837361111,
    "avg_tok_s": 166.45011455555556,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/162",
    "time_to_first_token_s": 0.366560225611111
  },
  {
    "chip": "M4 Max (32-core GPU, 36 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 273.40494424444444,
    "avg_tok_s": 24.60766388888889,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/162",
    "time_to_first_token_s": 4.441582821466667
  },
  {
    "chip": "M4 Max (40-core GPU, 48 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 663.3646092592592,
    "avg_tok_s": 55.0887822962963,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1188",
    "time_to_first_token_s": 1.7937055694814816
  },
  {
    "chip": "M4 Max (40-core GPU, 48 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 3750.568106722222,
    "avg_tok_s": 178.9852262222222,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1188",
    "time_to_first_token_s": 0.30928531144444443
  },
  {
    "chip": "M4 Max (40-core GPU, 48 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 347.03543255555553,
    "avg_tok_s": 30.13459711111111,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1188",
    "time_to_first_token_s": 3.4913595508333337
  },
  {
    "chip": "M4 Max (40-core GPU, 64 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 557.1200590555555,
    "avg_tok_s": 47.095953944444446,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/776",
    "time_to_first_token_s": 2.1445155647222225
  },
  {
    "chip": "M4 Max (40-core GPU, 64 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 3857.499544666666,
    "avg_tok_s": 180.24040433333337,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/776",
    "time_to_first_token_s": 0.30436414811111107
  },
  {
    "chip": "M4 Max (40-core GPU, 64 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 286.74860906666663,
    "avg_tok_s": 25.874467866666663,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/776",
    "time_to_first_token_s": 4.498645011088889
  },
  {
    "chip": "M4 Max (40-core GPU, 128 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 617.642594037037,
    "avg_tok_s": 51.63987292592592,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/6",
    "time_to_first_token_s": 1.9420189397222225
  },
  {
    "chip": "M4 Max (40-core GPU, 128 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 3833.843851847222,
    "avg_tok_s": 182.5577164513889,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/6",
    "time_to_first_token_s": 0.30469161659027777
  },
  {
    "chip": "M4 Max (40-core GPU, 128 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 326.6194374666667,
    "avg_tok_s": 28.74764654814815,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/6",
    "time_to_first_token_s": 3.7867302675777776
  },
  {
    "chip": "M4 Max (GPU count not published, 128 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 679.7101832222222,
    "avg_tok_s": 156.3000006666667,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/8",
    "time_to_first_token_s": 1.9661574258888888
  },
  {
    "chip": "M4 Pro (16-core GPU, 24 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 298.0409531777778,
    "avg_tok_s": 30.508745533333336,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/380",
    "time_to_first_token_s": 4.201011161111111
  },
  {
    "chip": "M4 Pro (16-core GPU, 24 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1823.817922962963,
    "avg_tok_s": 110.94556159259258,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/380",
    "time_to_first_token_s": 0.6715746882962962
  },
  {
    "chip": "M4 Pro (16-core GPU, 24 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 144.2954926666667,
    "avg_tok_s": 15.160405666666666,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/380",
    "time_to_first_token_s": 8.903966308944444
  },
  {
    "chip": "M4 Pro (16-core GPU, 48 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 302.390073,
    "avg_tok_s": 30.176443111111105,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/550",
    "time_to_first_token_s": 4.143549694333333
  },
  {
    "chip": "M4 Pro (16-core GPU, 48 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1754.577648111111,
    "avg_tok_s": 111.00253472222222,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/550",
    "time_to_first_token_s": 0.6816085969999999
  },
  {
    "chip": "M4 Pro (16-core GPU, 48 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 161.09253933333332,
    "avg_tok_s": 16.762545222222222,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/550",
    "time_to_first_token_s": 7.843727675888888
  },
  {
    "chip": "M4 Pro (16-core GPU, 64 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1858.9418227777776,
    "avg_tok_s": 111.93509922222222,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/2277",
    "time_to_first_token_s": 0.6703511943333332
  },
  {
    "chip": "M4 Pro (16-core GPU, 64 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 151.01082800000003,
    "avg_tok_s": 16.132661555555554,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/2277",
    "time_to_first_token_s": 8.460523194444445
  },
  {
    "chip": "M4 Pro (20-core GPU, 24 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 359.5577164,
    "avg_tok_s": 32.533236711111115,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/561",
    "time_to_first_token_s": 3.419087840755556
  },
  {
    "chip": "M4 Pro (20-core GPU, 24 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 2128.7541195555555,
    "avg_tok_s": 119.24874207407407,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/561",
    "time_to_first_token_s": 0.5725979937407406
  },
  {
    "chip": "M4 Pro (20-core GPU, 24 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 190.44284311111113,
    "avg_tok_s": 17.982970555555557,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/561",
    "time_to_first_token_s": 6.6294910853968245
  },
  {
    "chip": "M4 Pro (20-core GPU, 48 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 361.8889191604938,
    "avg_tok_s": 32.67357938271605,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/777",
    "time_to_first_token_s": 3.435018682604938
  },
  {
    "chip": "M4 Pro (20-core GPU, 48 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 2134.135712188034,
    "avg_tok_s": 118.8963297948718,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/777",
    "time_to_first_token_s": 0.5720378254615385
  },
  {
    "chip": "M4 Pro (20-core GPU, 48 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 189.7956117692308,
    "avg_tok_s": 17.981375188034185,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/777",
    "time_to_first_token_s": 6.62966483617094
  },
  {
    "chip": "M4 Pro (20-core GPU, 64 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 349.4138904444444,
    "avg_tok_s": 32.90144462962963,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/369",
    "time_to_first_token_s": 3.563006918222223
  },
  {
    "chip": "M4 Pro (20-core GPU, 64 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 2145.125925888889,
    "avg_tok_s": 118.5574338888889,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/369",
    "time_to_first_token_s": 0.5718198889722222
  },
  {
    "chip": "M4 Pro (20-core GPU, 64 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 182.97843991666667,
    "avg_tok_s": 18.008562833333333,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/369",
    "time_to_first_token_s": 6.9151872395
  },
  {
    "chip": "M5 (10-core GPU, 16 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1244.9073107777776,
    "avg_tok_s": 98.08477433333331,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/1716",
    "time_to_first_token_s": 0.995430389
  },
  {
    "chip": "M5 (10-core GPU, 32 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 210.70696922222223,
    "avg_tok_s": 22.29346603703704,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/2912",
    "time_to_first_token_s": 5.997633993703704
  },
  {
    "chip": "M5 (10-core GPU, 32 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 1271.5481898333337,
    "avg_tok_s": 98.37004172222224,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/2912",
    "time_to_first_token_s": 0.989687108888889
  },
  {
    "chip": "M5 (10-core GPU, 32 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 110.4192515,
    "avg_tok_s": 11.536774722222225,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/2912",
    "time_to_first_token_s": 11.637239780333331
  },
  {
    "chip": "M5 Max (32-core GPU, 36 GB)",
    "model": "llama-3-1-8b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 630.2577913333333,
    "avg_tok_s": 61.634733222222216,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/3105",
    "time_to_first_token_s": 1.914077976777778
  },
  {
    "chip": "M5 Max (32-core GPU, 36 GB)",
    "model": "llama-3-2-1b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 3620.6608126666665,
    "avg_tok_s": 228.99372777777776,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/3105",
    "time_to_first_token_s": 0.32193552311111107
  },
  {
    "chip": "M5 Max (32-core GPU, 36 GB)",
    "model": "qwen-2-5-14b-instruct",
    "quantization": "Q4_K - Medium",
    "runtime": "llamafile",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": 343.244466,
    "avg_tok_s": 34.28566722222222,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.localscore.ai/accelerator/3105",
    "time_to_first_token_s": 3.5801585140000003
  },
  {
    "chip": "M5 Max (48 GB)",
    "model": "Qwen3.5-35B-A3B",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 8000,
    "prompt_eval_tok_s": 3235,
    "avg_tok_s": 128,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1s4bggo/benchmarked_qwen35_35b_moe_27b_dense_122b_moe/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M1 Max (64 GB)",
    "model": "Qwen3.5-35B-A3B",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 8000,
    "prompt_eval_tok_s": 431,
    "avg_tok_s": 57.6,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1s4bggo/benchmarked_qwen35_35b_moe_27b_dense_122b_moe/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M1 Max (64 GB)",
    "model": "Qwen3.5-35B-A3B",
    "quantization": "4bit",
    "runtime": "LM Studio (MLX)",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 57,
    "source": "reference run",
    "verified": false,
    "source_url": "https://famstack.dev/guides/mlx-vs-gguf-apple-silicon/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M1 Max (64 GB)",
    "model": "Qwen3.5-35B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "LM Studio (llama.cpp)",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 29,
    "source": "reference run",
    "verified": false,
    "source_url": "https://famstack.dev/guides/mlx-vs-gguf-apple-silicon/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M5 Max (48 GB)",
    "model": "Qwen3.5-27B",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 8000,
    "prompt_eval_tok_s": 779,
    "avg_tok_s": 31.3,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1s4bggo/benchmarked_qwen35_35b_moe_27b_dense_122b_moe/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M1 Max (64 GB)",
    "model": "Qwen3.5-27B",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 8000,
    "prompt_eval_tok_s": 67,
    "avg_tok_s": 15,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1s4bggo/benchmarked_qwen35_35b_moe_27b_dense_122b_moe/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M1 Max (64 GB)",
    "model": "Qwen3.5-27B",
    "quantization": "Q8_0",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 10.5,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1s0xmq0/question_llamacpp_performance_on_m1_max_qwen_27b/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M1 Max (64 GB)",
    "model": "Qwen3.5-27B",
    "quantization": "Q6_K",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 12,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1s0xmq0/question_llamacpp_performance_on_m1_max_qwen_27b/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M1 Max (64 GB)",
    "model": "Qwen3.5-27B",
    "quantization": "Q4_K - Medium",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 11.5,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1s0xmq0/question_llamacpp_performance_on_m1_max_qwen_27b/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M5 Max (48 GB)",
    "model": "Qwen3.5-35B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 8000,
    "prompt_eval_tok_s": 783,
    "avg_tok_s": 89.4,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1s4bggo/benchmarked_qwen35_35b_moe_27b_dense_122b_moe/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M5 Max (48 GB)",
    "model": "Qwen3.5-27B",
    "quantization": "Q4_K - Medium",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 8000,
    "prompt_eval_tok_s": 171,
    "avg_tok_s": 23.7,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1s4bggo/benchmarked_qwen35_35b_moe_27b_dense_122b_moe/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (256 GB)",
    "model": "Qwen3.5-122B-A10B",
    "quantization": "8bit",
    "runtime": "MLX",
    "ram_required_gb": 129.8,
    "ram_required_note": "~129.8 GB",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 43,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rkcvqa/benchmarked_11_mlx_models_on_m3_ultra_heres_which/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (256 GB)",
    "model": "Qwen3.5-122B-A10B",
    "quantization": "MXFP4",
    "runtime": "MLX",
    "ram_required_gb": 65,
    "ram_required_note": "~65 GB",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 57,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rkcvqa/benchmarked_11_mlx_models_on_m3_ultra_heres_which/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (256 GB)",
    "model": "Qwen3-Coder-Next",
    "quantization": "6bit",
    "runtime": "MLX",
    "ram_required_gb": 64.8,
    "ram_required_note": "~64.8 GB",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 66,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rkcvqa/benchmarked_11_mlx_models_on_m3_ultra_heres_which/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (256 GB)",
    "model": "Qwen3-Coder-Next",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": 44.9,
    "ram_required_note": "~44.9 GB",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 74,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rkcvqa/benchmarked_11_mlx_models_on_m3_ultra_heres_which/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (256 GB)",
    "model": "GLM-4.5-Air",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": 60.3,
    "ram_required_note": "~60.3 GB",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 54,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rkcvqa/benchmarked_11_mlx_models_on_m3_ultra_heres_which/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "GLM-4.5-Air",
    "quantization": "4bit",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 10758,
    "prompt_eval_tok_s": 183.25,
    "avg_tok_s": 27.75,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1mt3epi/m4_max_generation_speed_vs_context_size/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "GLM-4.5-Air",
    "quantization": "4bit",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 21002,
    "prompt_eval_tok_s": 120.82,
    "avg_tok_s": 17.99,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1mt3epi/m4_max_generation_speed_vs_context_size/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "GLM-4.5-Air",
    "quantization": "4bit",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 31182,
    "prompt_eval_tok_s": 75.5,
    "avg_tok_s": 15.07,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1mt3epi/m4_max_generation_speed_vs_context_size/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "GLM-4.5-Air",
    "quantization": "4bit",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 62037,
    "prompt_eval_tok_s": 41.46,
    "avg_tok_s": 9.58,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1mt3epi/m4_max_generation_speed_vs_context_size/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (256 GB)",
    "model": "Devstral Small 2 24B",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": 13.4,
    "ram_required_note": "~13.4 GB",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 47,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rkcvqa/benchmarked_11_mlx_models_on_m3_ultra_heres_which/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (256 GB)",
    "model": "GLM-4.7-Flash",
    "quantization": "8bit",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 58,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rkcvqa/benchmarked_11_mlx_models_on_m3_ultra_heres_which/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (256 GB)",
    "model": "Qwen 3 0.6B",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 370,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rkcvqa/benchmarked_11_mlx_models_on_m3_ultra_heres_which/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (256 GB)",
    "model": "Qwen3.5-9B",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": 5.1,
    "ram_required_note": "~5.1 GB",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 106,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rkcvqa/benchmarked_11_mlx_models_on_m3_ultra_heres_which/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (256 GB)",
    "model": "Qwen3.5-27B",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": 15.3,
    "ram_required_note": "~15.3 GB",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 38,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rkcvqa/benchmarked_11_mlx_models_on_m3_ultra_heres_which/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (256 GB)",
    "model": "Qwen3.5-35B-A3B",
    "quantization": "8bit",
    "runtime": "MLX",
    "ram_required_gb": 36.9,
    "ram_required_note": "~36.9 GB",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 80,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rkcvqa/benchmarked_11_mlx_models_on_m3_ultra_heres_which/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (256 GB)",
    "model": "Qwen3.5-35B-A3B",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": 19.6,
    "ram_required_note": "~19.6 GB",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 95,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rkcvqa/benchmarked_11_mlx_models_on_m3_ultra_heres_which/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 (16 GB)",
    "model": "Qwen3.5-9B",
    "quantization": "Q4_0",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 4.0648,
    "source": "reference run",
    "verified": false,
    "source_url": "https://huggingface.co/Manojb/macmini-16gb-bench-gguf-mlx/raw/main/SUMMARY.md",
    "time_to_first_token_s": 0.5057
  },
  {
    "chip": "M4 (16 GB)",
    "model": "Qwen3.5-9B",
    "quantization": "Q4_K - Small",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 3.142,
    "source": "reference run",
    "verified": false,
    "source_url": "https://huggingface.co/Manojb/macmini-16gb-bench-gguf-mlx/raw/main/SUMMARY.md",
    "time_to_first_token_s": 0.8445
  },
  {
    "chip": "M4 (16 GB)",
    "model": "Qwen3.5-9B",
    "quantization": "Q6_K",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 2.1849,
    "source": "reference run",
    "verified": false,
    "source_url": "https://huggingface.co/Manojb/macmini-16gb-bench-gguf-mlx/raw/main/SUMMARY.md",
    "time_to_first_token_s": 1.5031
  },
  {
    "chip": "M4 (16 GB)",
    "model": "Devstral Small 2 24B",
    "quantization": "Q4_0",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 3.3641,
    "source": "reference run",
    "verified": false,
    "source_url": "https://huggingface.co/Manojb/macmini-16gb-bench-gguf-mlx/raw/main/SUMMARY.md",
    "time_to_first_token_s": 2.6874
  },
  {
    "chip": "M4 (16 GB)",
    "model": "Devstral Small 2 24B",
    "quantization": "Q4_1",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 0.0542,
    "source": "reference run",
    "verified": false,
    "source_url": "https://huggingface.co/Manojb/macmini-16gb-bench-gguf-mlx/raw/main/SUMMARY.md",
    "time_to_first_token_s": 17.48
  },
  {
    "chip": "M4 (16 GB)",
    "model": "Devstral Small 2 24B",
    "quantization": "Q4_K - Medium",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 0.0198,
    "source": "reference run",
    "verified": false,
    "source_url": "https://huggingface.co/Manojb/macmini-16gb-bench-gguf-mlx/raw/main/SUMMARY.md",
    "time_to_first_token_s": 65.32
  },
  {
    "chip": "M4 (16 GB)",
    "model": "Qwen3.5-35B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 1.2956,
    "source": "reference run",
    "verified": false,
    "source_url": "https://huggingface.co/Manojb/macmini-16gb-bench-gguf-mlx/raw/main/SUMMARY.md",
    "time_to_first_token_s": 13.0595
  },
  {
    "chip": "M4 (16 GB)",
    "model": "Qwen3.5-27B",
    "quantization": "Q4_K - Medium",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 0.0084,
    "source": "reference run",
    "verified": false,
    "source_url": "https://huggingface.co/Manojb/macmini-16gb-bench-gguf-mlx/raw/main/SUMMARY.md",
    "time_to_first_token_s": 66.9331
  },
  {
    "chip": "M1 Ultra (64 GB)",
    "model": "Devstral Small 2 24B",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 29.68,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1plbjqg/devstral_small_2_on_macos/",
    "time_to_first_token_s": 6.63
  },
  {
    "chip": "M1 Ultra (64 GB)",
    "model": "Devstral Small 2 24B",
    "quantization": "8bit",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 22.32,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1plbjqg/devstral_small_2_on_macos/",
    "time_to_first_token_s": 7.57
  },
  {
    "chip": "M1 Ultra (64 GB)",
    "model": "Devstral Small 2 24B",
    "quantization": "Q4_K - Medium",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 25.3,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1plbjqg/devstral_small_2_on_macos/",
    "time_to_first_token_s": 5.89
  },
  {
    "chip": "M1 Ultra (64 GB)",
    "model": "Devstral Small 2 24B",
    "quantization": "Q8",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 23.37,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1plbjqg/devstral_small_2_on_macos/",
    "time_to_first_token_s": 5.66
  },
  {
    "chip": "M1 Pro (16 GB)",
    "model": "Qwen3.5-9B",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 30,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rj3ay3/qwen359b_4bit_quant_acting_weird/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M1 Pro (16 GB)",
    "model": "Qwen3.5-9B",
    "quantization": "4bit",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 15,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rj3ay3/qwen359b_4bit_quant_acting_weird/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Pro (48 GB)",
    "model": "Qwen3.5-27B",
    "quantization": "8bit",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 8.5,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1roetd9/qwen_35_27b_macbook_m4_pro_48gb/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M2 Max (38-core GPU, 96 GB)",
    "model": "Qwen3.5-27B",
    "quantization": "8bit",
    "runtime": "MLX",
    "ram_required_gb": 30,
    "ram_required_note": "~30 GB",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 12,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rkkeqh/qwen35_models_ultra_slow_for_anyone_else_compared/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M2 Ultra (GPU count not published, 128 GB)",
    "model": "Qwen3.5-27B",
    "quantization": "8bit",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 20.6,
    "source": "reference run",
    "verified": false,
    "source_url": "https://github.com/ml-explore/mlx-lm/pull/990",
    "time_to_first_token_s": null
  },
  {
    "chip": "M2 Ultra (GPU count not published, 128 GB)",
    "model": "Qwen3.5-27B",
    "quantization": "8bit",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 27.1,
    "source": "reference run",
    "verified": false,
    "source_url": "https://github.com/ml-explore/mlx-lm/pull/990",
    "time_to_first_token_s": null
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen3.5-27B",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 31.6,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rzkw4x/m5_max_128g_performance_tests_i_just_got_my_new/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen3.5-27B",
    "quantization": "Q6_K",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 8192,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 16.5,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rzkw4x/m5_max_128g_performance_tests_i_just_got_my_new/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen3.5-122B-A10B",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": 71.91,
    "ram_required_note": "~71.91 GB",
    "context_tokens": 4106,
    "prompt_eval_tok_s": 881.466,
    "avg_tok_s": 65.853,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rqnpvj/m5_max_just_arrived_benchmarks_incoming/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen3.5-122B-A10B",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": 73.803,
    "ram_required_note": "~73.803 GB",
    "context_tokens": 16394,
    "prompt_eval_tok_s": 1239.734,
    "avg_tok_s": 60.639,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rqnpvj/m5_max_just_arrived_benchmarks_incoming/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen3.5-122B-A10B",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": 76.397,
    "ram_required_note": "~76.397 GB",
    "context_tokens": 32778,
    "prompt_eval_tok_s": 1067.824,
    "avg_tok_s": 54.923,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rqnpvj/m5_max_just_arrived_benchmarks_incoming/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen3-Coder-Next",
    "quantization": "8bit",
    "runtime": "MLX",
    "ram_required_gb": 87.068,
    "ram_required_note": "~87.068 GB",
    "context_tokens": 4105,
    "prompt_eval_tok_s": 754.927,
    "avg_tok_s": 79.296,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rqnpvj/m5_max_just_arrived_benchmarks_incoming/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen3-Coder-Next",
    "quantization": "8bit",
    "runtime": "MLX",
    "ram_required_gb": 88.176,
    "ram_required_note": "~88.176 GB",
    "context_tokens": 16393,
    "prompt_eval_tok_s": 1802.144,
    "avg_tok_s": 74.293,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rqnpvj/m5_max_just_arrived_benchmarks_incoming/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen3-Coder-Next",
    "quantization": "8bit",
    "runtime": "MLX",
    "ram_required_gb": 89.652,
    "ram_required_note": "~89.652 GB",
    "context_tokens": 32777,
    "prompt_eval_tok_s": 1887.158,
    "avg_tok_s": 68.624,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rqnpvj/m5_max_just_arrived_benchmarks_incoming/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen3-Coder-Next",
    "quantization": "8bit",
    "runtime": "MLX",
    "ram_required_gb": 92.605,
    "ram_required_note": "~92.605 GB",
    "context_tokens": 65545,
    "prompt_eval_tok_s": 1432.73,
    "avg_tok_s": 48.212,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rqnpvj/m5_max_just_arrived_benchmarks_incoming/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M5 Pro (64 GB)",
    "model": "Qwen3.5-9B",
    "quantization": "Q4_K - Medium",
    "runtime": "llama.cpp",
    "ram_required_gb": 13.8,
    "ram_required_note": "~13.8 GB",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 25,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.sharpai.org/benchmark/",
    "time_to_first_token_s": 0.765
  },
  {
    "chip": "M5 Pro (64 GB)",
    "model": "Qwen3.5-27B",
    "quantization": "Q4_K - Medium",
    "runtime": "llama.cpp",
    "ram_required_gb": 24.9,
    "ram_required_note": "~24.9 GB",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 10,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.sharpai.org/benchmark/",
    "time_to_first_token_s": 2.156
  },
  {
    "chip": "M5 Pro (64 GB)",
    "model": "Qwen3.5-35B-A3B",
    "quantization": "Q4_K_L",
    "runtime": "llama.cpp",
    "ram_required_gb": 27.2,
    "ram_required_note": "~27.2 GB",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 41.9,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.sharpai.org/benchmark/",
    "time_to_first_token_s": 0.435
  },
  {
    "chip": "M5 Pro (64 GB)",
    "model": "Qwen3.5-122B-A10B",
    "quantization": "IQ1_M",
    "runtime": "llama.cpp",
    "ram_required_gb": 40.8,
    "ram_required_note": "~40.8 GB",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 18,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.sharpai.org/benchmark/",
    "time_to_first_token_s": 1.627
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen3.5-397B-A17B",
    "quantization": "4bit",
    "runtime": "flash-moe",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 12.99,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1s30vs2/took_the_48gb_flashmoe_benchmark_and_ran_it_on/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (512 GB)",
    "model": "Qwen3.5-397B-A17B",
    "quantization": "q4.1bit",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 40.2,
    "source": "reference run",
    "verified": false,
    "source_url": "https://huggingface.co/inferencerlabs/Qwen3.5-397B-A17B-MLX-4.1bit",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (512 GB)",
    "model": "GLM-5",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": 391.82,
    "ram_required_note": "~391.82 GB",
    "context_tokens": 1024,
    "prompt_eval_tok_s": 187,
    "avg_tok_s": 16.7,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rdkze3/m3_ultra_512gb_realworld_performance_of/",
    "time_to_first_token_s": 5.477
  },
  {
    "chip": "M3 Ultra (512 GB)",
    "model": "GLM-5",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": 394.07,
    "ram_required_note": "~394.07 GB",
    "context_tokens": 4096,
    "prompt_eval_tok_s": 180.1,
    "avg_tok_s": 13.7,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rdkze3/m3_ultra_512gb_realworld_performance_of/",
    "time_to_first_token_s": 22.745
  },
  {
    "chip": "M3 Ultra (512 GB)",
    "model": "GLM-5",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": 396.69,
    "ram_required_note": "~396.69 GB",
    "context_tokens": 8192,
    "prompt_eval_tok_s": 154.1,
    "avg_tok_s": 13.2,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rdkze3/m3_ultra_512gb_realworld_performance_of/",
    "time_to_first_token_s": 53.169
  },
  {
    "chip": "M3 Ultra (512 GB)",
    "model": "GLM-5",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": 402.72,
    "ram_required_note": "~402.72 GB",
    "context_tokens": 16384,
    "prompt_eval_tok_s": 117.4,
    "avg_tok_s": 12,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rdkze3/m3_ultra_512gb_realworld_performance_of/",
    "time_to_first_token_s": 139.545
  },
  {
    "chip": "M3 Ultra (512 GB)",
    "model": "GLM-5",
    "quantization": "4bit",
    "runtime": "MLX",
    "ram_required_gb": 415.41,
    "ram_required_note": "~415.41 GB",
    "context_tokens": 32768,
    "prompt_eval_tok_s": 77.7,
    "avg_tok_s": 10.7,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rdkze3/m3_ultra_512gb_realworld_performance_of/",
    "time_to_first_token_s": 421.955
  },
  {
    "chip": "M4 Pro (48 GB)",
    "model": "Devstral Small 1.1",
    "quantization": "6bit",
    "runtime": "LM Studio",
    "ram_required_gb": 18.51,
    "ram_required_note": "~18.51 GB",
    "context_tokens": 131072,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 12.88,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1nkfvrl/local_llm_coding_stack_24gb_minimum_ideal_36gb/",
    "time_to_first_token_s": 5.91
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "Devstral Small 1.1",
    "quantization": "4bit",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 33,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1m1t19r/any_experiences_running_llms_on_a_macbook/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M2 (24 GB)",
    "model": "Devstral Small 1.1",
    "quantization": "4bit",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 6,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1m1t19r/any_experiences_running_llms_on_a_macbook/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (512 GB)",
    "model": "Devstral Small 1.1",
    "quantization": "4bit",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 43,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1pry2v7/people_using_devstral_2_123b_how_has_it_been/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Gemma 3 27B",
    "quantization": "Q6_K",
    "runtime": "llama.cpp",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": 8192,
    "prompt_eval_tok_s": 391,
    "avg_tok_s": 20,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1s0czc4/round_2_followup_m5_max_128g_performance_tests_i/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M3 Ultra (512 GB)",
    "model": "Gemma 3 27B",
    "quantization": "bf16",
    "runtime": "LM Studio",
    "ram_required_gb": 52.57,
    "ram_required_note": "~52.57 GB",
    "context_tokens": 128000,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 11.19,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1kfi8xh/benchmark_quickanddirty_test_of_5_models_on_a_mac/",
    "time_to_first_token_s": 1.72
  },
  {
    "chip": "M4 Max",
    "model": "Qwen3.5-27B",
    "quantization": "Q4_K",
    "runtime": "llama.cpp",
    "ram_required_gb": 16.4,
    "ram_required_note": "~16.4 GB",
    "context_tokens": 2048,
    "prompt_eval_tok_s": 222.23,
    "avg_tok_s": 16.69,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rlyukb/m4_max_llamacpp_benchmarks_of_qwen35_35b_and_27b/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max",
    "model": "Qwen3.5-27B",
    "quantization": "Q4_K",
    "runtime": "llama.cpp",
    "ram_required_gb": 16.4,
    "ram_required_note": "~16.4 GB",
    "context_tokens": 8192,
    "prompt_eval_tok_s": 209.3,
    "avg_tok_s": 16.14,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rlyukb/m4_max_llamacpp_benchmarks_of_qwen35_35b_and_27b/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M4 Max",
    "model": "Qwen3.5-27B",
    "quantization": "Q4_K",
    "runtime": "llama.cpp",
    "ram_required_gb": 16.4,
    "ram_required_note": "~16.4 GB",
    "context_tokens": 16384,
    "prompt_eval_tok_s": 195.44,
    "avg_tok_s": 15.75,
    "source": "reference run",
    "verified": false,
    "source_url": "https://www.reddit.com/r/LocalLLaMA/comments/1rlyukb/m4_max_llamacpp_benchmarks_of_qwen35_35b_and_27b/",
    "time_to_first_token_s": null
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Gemma 4 26B-A4B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 50,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.csv",
    "time_to_first_token_s": 0.5
  },
  {
    "chip": "M4 Max (48 GB)",
    "model": "Gemma 4 26B-A4B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 40,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.csv",
    "time_to_first_token_s": 0.7
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Gemma 4 26B-A4B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 28,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.csv",
    "time_to_first_token_s": 1
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Gemma 4 31B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 26,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.csv",
    "time_to_first_token_s": 0.7
  },
  {
    "chip": "M4 Max (48 GB)",
    "model": "Gemma 4 31B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 18,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.csv",
    "time_to_first_token_s": 1.1
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Gemma 4 31B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 14,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.csv",
    "time_to_first_token_s": 1.4
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen3.6-35B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 55,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.csv",
    "time_to_first_token_s": 0.6
  },
  {
    "chip": "M4 Max (48 GB)",
    "model": "Qwen3.6-35B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 42,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.csv",
    "time_to_first_token_s": 0.9
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Qwen3.6-35B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 32,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.csv",
    "time_to_first_token_s": 1.2
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Gemma 3 27B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 42,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.9
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Gemma 3 27B",
    "quantization": "Q4_K - Medium",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 25,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.5
  },
  {
    "chip": "M4 Max (48 GB)",
    "model": "Gemma 3 27B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 35,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.1
  },
  {
    "chip": "M3 Max (96 GB)",
    "model": "Gemma 3 27B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 28,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.3
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Gemma 4 E2B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 158,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.1
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Gemma 4 E2B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 95,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.2
  },
  {
    "chip": "M3 (16 GB)",
    "model": "Gemma 4 E2B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 82,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.3
  },
  {
    "chip": "M1 (8 GB)",
    "model": "Gemma 4 E2B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 58,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.5
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Gemma 4 E4B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 128,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.2
  },
  {
    "chip": "M5 Pro (24 GB)",
    "model": "Gemma 4 E4B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 92,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.3
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Gemma 4 E4B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 78,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.3
  },
  {
    "chip": "M3 (16 GB)",
    "model": "Gemma 4 E4B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 62,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.4
  },
  {
    "chip": "M1 (8 GB)",
    "model": "Gemma 4 E4B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 42,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.6
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Mistral Small 4 119B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 38,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.9
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Mistral Small 4 119B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 42,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.8
  },
  {
    "chip": "M4 Ultra (192 GB)",
    "model": "Mistral Small 4 119B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 45,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.7
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Llama 3.3 70B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 12,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 2.8
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Llama 3.3 70B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 15,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 2.4
  },
  {
    "chip": "M4 Ultra (192 GB)",
    "model": "Llama 3.3 70B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 18,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 2
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "DeepSeek R1 Distill Llama 70B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 11,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 3
  },
  {
    "chip": "M4 Ultra (192 GB)",
    "model": "DeepSeek R1 Distill Llama 70B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 16,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 2.2
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen 2.5 72B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 10,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 3.2
  },
  {
    "chip": "M4 Ultra (192 GB)",
    "model": "Qwen 2.5 72B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 15,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 2.5
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen 3 235B-A22B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 15,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 2.2
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen 3 235B-A22B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 18,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.8
  },
  {
    "chip": "M4 Ultra (192 GB)",
    "model": "Qwen 3 235B-A22B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 22,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.5
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Nemotron Cascade 2 30B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 35,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.9
  },
  {
    "chip": "M4 Max (48 GB)",
    "model": "Nemotron Cascade 2 30B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 28,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.2
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Nemotron Cascade 2 30B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 22,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.5
  },
  {
    "chip": "M4 Max (128 GB)",
    "model": "Qwen3.6-27B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": 29,
    "ram_required_note": "~29 GB",
    "context_tokens": null,
    "prompt_eval_tok_s": 114.49,
    "avg_tok_s": 16.56,
    "source": "reference run",
    "verified": false,
    "source_url": "https://huggingface.co/batiai/Qwen3.6-27B-GGUF",
    "time_to_first_token_s": null
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Llama 4 Scout 17B-16E",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 22,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.8
  },
  {
    "chip": "M4 Ultra (192 GB)",
    "model": "Llama 4 Scout 17B-16E",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 30,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.3
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Llama 4 Scout 17B-16E",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 26,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.5
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "gpt-oss 120B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 7,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 5.5
  },
  {
    "chip": "M4 Ultra (192 GB)",
    "model": "gpt-oss 120B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 10,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 4.2
  },
  {
    "chip": "M5 Pro (24 GB)",
    "model": "Gemma 4 26B-A4B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 35,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.8
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Gemma 4 31B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 22,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.9
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Qwen3.6-35B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 48,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.8
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Gemma 3 4B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 132,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.3
  },
  {
    "chip": "M1 (8 GB)",
    "model": "Gemma 3 4B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 48,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.8
  },
  {
    "chip": "M3 Pro (18 GB)",
    "model": "Gemma 3 4B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 88,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.4
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Qwen3.5-9B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 105,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.5
  },
  {
    "chip": "M4 (16 GB)",
    "model": "Qwen3.5-9B",
    "quantization": "Q4_K - Medium",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 72,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.6
  },
  {
    "chip": "M3 (16 GB)",
    "model": "Qwen3.5-9B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 58,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.7
  },
  {
    "chip": "M1 (16 GB)",
    "model": "Qwen3.5-9B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 35,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.1
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen 3 8B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 98,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.4
  },
  {
    "chip": "M2 (16 GB)",
    "model": "Qwen 3 8B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 55,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.7
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Phi-4 14B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 62,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.6
  },
  {
    "chip": "M4 (16 GB)",
    "model": "Phi-4 14B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 38,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1
  },
  {
    "chip": "M2 (16 GB)",
    "model": "Phi-4 14B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 28,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.3
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Qwen 3 30B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 62,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.7
  },
  {
    "chip": "M4 Max (48 GB)",
    "model": "Qwen 3 30B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 42,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.9
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Qwen 3 30B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 35,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1
  },
  {
    "chip": "M3 Max (36 GB)",
    "model": "Qwen 3 30B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 28,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.3
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Qwen3.5-35B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 52,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.9
  },
  {
    "chip": "M4 Max (48 GB)",
    "model": "Qwen3.5-35B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 34,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.2
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen3.5-35B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 48,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.9
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 28,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.1
  },
  {
    "chip": "M4 Max (64 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 22,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.4
  },
  {
    "chip": "M4 Pro (32 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 15,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.8
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "DeepSeek R1 Distill Qwen 32B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 27,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.2
  },
  {
    "chip": "M4 Max (48 GB)",
    "model": "DeepSeek R1 Distill Qwen 32B",
    "quantization": "Q4_K - Medium",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 18,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.8
  },
  {
    "chip": "M3 Max (36 GB)",
    "model": "DeepSeek R1 Distill Qwen 32B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 14,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 2
  },
  {
    "chip": "M3 Max (96 GB)",
    "model": "Qwen3.5-35B-A3B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 22,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.5
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Qwen 3 8B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 82,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.5
  },
  {
    "chip": "M1 (16 GB)",
    "model": "Qwen 3 8B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 38,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1
  },
  {
    "chip": "M4 Ultra (192 GB)",
    "model": "Qwen 3 32B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 32,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Qwen3.5-9B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 92,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.4
  },
  {
    "chip": "M2 (8 GB)",
    "model": "Qwen 3 4B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 52,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.7
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Qwen 3 4B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 118,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.3
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Mistral 7B v0.3",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 122,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.3
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Mistral 7B v0.3",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 98,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.4
  },
  {
    "chip": "M3 (16 GB)",
    "model": "Mistral 7B v0.3",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 62,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.6
  },
  {
    "chip": "M1 (16 GB)",
    "model": "Mistral 7B v0.3",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 42,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.2
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Llama 3.1 8B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 138,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.3
  },
  {
    "chip": "M4 (16 GB)",
    "model": "Llama 3.1 8B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 75,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.6
  },
  {
    "chip": "M3 Pro (18 GB)",
    "model": "Llama 3.1 8B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 62,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.7
  },
  {
    "chip": "M1 (16 GB)",
    "model": "Llama 3.1 8B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 40,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.1
  },
  {
    "chip": "M2 (8 GB)",
    "model": "Llama 3.1 8B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 48,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.8
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Qwen 3 4B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 135,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.2
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Qwen3.5-4B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 148,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.2
  },
  {
    "chip": "M4 (16 GB)",
    "model": "Qwen3.5-4B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 92,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.4
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Qwen 3 8B",
    "quantization": "Q8_0",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 68,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.5
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Mistral 7B v0.3",
    "quantization": "Q8_0",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 88,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.4
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Mistral 7B v0.3",
    "quantization": "Q8_0",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 65,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.5
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen3.5-9B",
    "quantization": "Q8_0",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 78,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.5
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Llama 3.1 8B",
    "quantization": "Q8_0",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 82,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.4
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Gemma 3 4B",
    "quantization": "Q8_0",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 95,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.3
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Qwen 3 14B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 58,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.8
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Qwen 3 14B",
    "quantization": "Q4_K - Medium",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 38,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1
  },
  {
    "chip": "M3 (16 GB)",
    "model": "Qwen 3 14B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 30,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.2
  },
  {
    "chip": "M5 Max (128 GB)",
    "model": "Qwen 3 14B",
    "quantization": "Q8_0",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 42,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.9
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Phi-4 Mini Instruct 3.8B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 142,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.3
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Phi-4 Mini Instruct 3.8B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 108,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.4
  },
  {
    "chip": "M3 (16 GB)",
    "model": "Phi-4 Mini Instruct 3.8B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 95,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.3
  },
  {
    "chip": "M2 (8 GB)",
    "model": "Phi-4 Mini Instruct 3.8B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 72,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.5
  },
  {
    "chip": "M1 (16 GB)",
    "model": "Phi-4 Mini Instruct 3.8B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 58,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.6
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Phi-4 Mini Instruct 3.8B",
    "quantization": "Q8_0",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 112,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.3
  },
  {
    "chip": "M4 Max (48 GB)",
    "model": "Phi-4 Mini Instruct 3.8B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 125,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.3
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Gemma 3 12B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 68,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.6
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Gemma 3 12B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 52,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.7
  },
  {
    "chip": "M3 (16 GB)",
    "model": "Gemma 3 12B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 38,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.1
  },
  {
    "chip": "M2 (16 GB)",
    "model": "Gemma 3 12B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 32,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.3
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "DeepSeek R1 Distill Llama 8B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 97,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.5
  },
  {
    "chip": "M4 (16 GB)",
    "model": "DeepSeek R1 Distill Llama 8B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 78,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.5
  },
  {
    "chip": "M2 (16 GB)",
    "model": "DeepSeek R1 Distill Llama 8B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 58,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.8
  },
  {
    "chip": "M1 (16 GB)",
    "model": "DeepSeek R1 Distill Llama 8B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 38,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.2
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "DeepSeek R1 Distill Llama 8B",
    "quantization": "Q8_0",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 75,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.5
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Ministral 3 8B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 98,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.4
  },
  {
    "chip": "M3 (16 GB)",
    "model": "Ministral 3 8B",
    "quantization": "Q4_K - Medium",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 55,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.7
  },
  {
    "chip": "M5 Max (64 GB)",
    "model": "Ministral 3 14B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 58,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.7
  },
  {
    "chip": "M4 Pro (24 GB)",
    "model": "Ministral 3 14B",
    "quantization": "Q4_K - Medium",
    "runtime": "Ollama",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 40,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.9
  },
  {
    "chip": "M4 (16 GB)",
    "model": "Ministral 3 8B",
    "quantization": "Q4_K - Medium",
    "runtime": "MLX",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 72,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 0.5
  },
  {
    "chip": "M3 (16 GB)",
    "model": "Ministral 3 14B",
    "quantization": "Q4_K - Medium",
    "runtime": "LM Studio",
    "ram_required_gb": null,
    "ram_required_note": "not published",
    "context_tokens": null,
    "prompt_eval_tok_s": null,
    "avg_tok_s": 30,
    "source": "reference run",
    "verified": false,
    "source_url": "https://llmcheck.net/data/benchmarks.json",
    "time_to_first_token_s": 1.2
  }
]
