{
 "generated": "2026-10-09T04:33+08:00",
 "engines": [
  {
   "engine": "llama.cpp",
   "repo": "ggml-org/llama.cpp",
   "role": "GGUF on anything: CUDA, ROCm, Vulkan, Metal, SYCL, CPU. Default for single-user boxes.",
   "version": "b11514",
   "date": "2026-10-08",
   "url": "https://github.com/ggml-org/llama.cpp/releases/tag/b11514"
  },
  {
   "engine": "ik_llama.cpp",
   "repo": "ikawrakow/ik_llama.cpp",
   "role": "llama.cpp fork, faster CPU / hybrid MoE offload and IQ quants.",
   "version": "main@04b4ebc",
   "date": "2026-10-08",
   "url": "https://github.com/ikawrakow/ik_llama.cpp"
  },
  {
   "engine": "vLLM",
   "repo": "vllm-project/vllm",
   "role": "Multi-user serving (PagedAttention, FP8 KV, MTP). NVIDIA, AMD Instinct, Intel XPU.",
   "version": "v0.31.0",
   "date": "2026-10-05",
   "url": "https://github.com/vllm-project/vllm/releases/tag/v0.31.0"
  },
  {
   "engine": "SGLang",
   "repo": "sgl-project/sglang",
   "role": "Agentic / multi-turn serving (RadixAttention). NVIDIA, AMD Instinct.",
   "version": "v0.5.21",
   "date": "2026-10-02",
   "url": "https://github.com/sgl-project/sglang/releases/tag/v0.5.21"
  },
  {
   "engine": "TensorRT-LLM",
   "repo": "NVIDIA/TensorRT-LLM",
   "role": "Max throughput on Hopper/Blackwell, native NVFP4.",
   "version": "v1.3.0rc29",
   "date": "2026-09-29",
   "url": "https://github.com/NVIDIA/TensorRT-LLM/releases/tag/v1.3.0rc29"
  },
  {
   "engine": "Ollama",
   "repo": "ollama/ollama",
   "role": "Zero-config pull-and-run; easiest path for non-specialists.",
   "version": "v0.40.1",
   "date": "2026-10-07",
   "url": "https://github.com/ollama/ollama/releases/tag/v0.40.1"
  },
  {
   "engine": "llama-swap",
   "repo": "mostlygeek/llama-swap",
   "role": "Hot-swap several models behind one OpenAI port.",
   "version": "v262",
   "date": "2026-10-03",
   "url": "https://github.com/mostlygeek/llama-swap/releases/tag/v262"
  },
  {
   "engine": "MLX",
   "repo": "ml-explore/mlx",
   "role": "Apple Silicon array framework.",
   "version": "v0.32.3",
   "date": "2026-09-29",
   "url": "https://github.com/ml-explore/mlx/releases/tag/v0.32.3"
  },
  {
   "engine": "mlx-lm",
   "repo": "ml-explore/mlx-lm",
   "role": "Apple Silicon LLM serving (mlx_lm.server is OpenAI-compatible).",
   "version": "v0.31.3",
   "date": "2026-04-22",
   "url": "https://github.com/ml-explore/mlx-lm/releases/tag/v0.31.3"
  },
  {
   "engine": "mlx-vlm",
   "repo": "Blaizzy/mlx-vlm",
   "role": "Apple Silicon vision-language models.",
   "version": "v0.7.6",
   "date": "2026-10-05",
   "url": "https://github.com/Blaizzy/mlx-vlm/releases/tag/v0.7.6",
   "stale": true
  },
  {
   "engine": "ExLlamaV3",
   "repo": "turboderp-org/exllamav3",
   "role": "Fastest single-stream on one NVIDIA GPU, EXL3 quants.",
   "version": "v1.6.0",
   "date": "2026-10-07",
   "url": "https://github.com/turboderp-org/exllamav3/releases/tag/v1.6.0"
  },
  {
   "engine": "TabbyAPI",
   "repo": "theroyallab/tabbyAPI",
   "role": "OpenAI server for ExLlamaV3.",
   "version": "main@2fd6cc7",
   "date": "2026-10-06",
   "url": "https://github.com/theroyallab/tabbyAPI"
  },
  {
   "engine": "KTransformers",
   "repo": "kvcache-ai/ktransformers",
   "role": "Huge MoE (DeepSeek-class) with CPU expert offload.",
   "version": "v0.7.1",
   "date": "2026-09-15",
   "url": "https://github.com/kvcache-ai/ktransformers/releases/tag/v0.7.1"
  },
  {
   "engine": "Lemonade",
   "repo": "lemonade-sdk/lemonade",
   "role": "AMD Ryzen AI: NPU + iGPU, auto backend pick.",
   "version": "v2026.41.1",
   "date": "2026-10-07",
   "url": "https://github.com/lemonade-sdk/lemonade/releases/tag/v2026.41.1"
  },
  {
   "engine": "OpenVINO",
   "repo": "openvinotoolkit/openvino",
   "role": "Intel CPU / Arc / NPU.",
   "version": "2026.4.1",
   "date": "2026-10-01",
   "url": "https://github.com/openvinotoolkit/openvino/releases/tag/2026.4.1"
  },
  {
   "engine": "LocalAI",
   "repo": "mudler/LocalAI",
   "role": "All-in-one OpenAI-compatible server (LLM + image + audio).",
   "version": "v4.11.0",
   "date": "2026-10-02",
   "url": "https://github.com/mudler/LocalAI/releases/tag/v4.11.0"
  },
  {
   "engine": "mistral.rs",
   "repo": "EricLBuehler/mistral.rs",
   "role": "Rust engine with in-situ quantization.",
   "version": "v0.9.4",
   "date": "2026-09-24",
   "url": "https://github.com/EricLBuehler/mistral.rs/releases/tag/v0.9.4"
  }
 ],
 "tokenmark": {
  "generated": "2026-10-07T21:14:44.466Z",
  "counts": {
   "configs": 258,
   "models": 118,
   "platforms": 4,
   "releases": 127
  },
  "platforms": [
   "strix-halo",
   "dgx-spark",
   "mac-ultra",
   "mac-max"
  ],
  "configs": [
   {
    "model": "Qwen3.6-35B-A3B-MTP",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "vulkan",
    "ctx": 262144,
    "mode": "mtp-3",
    "decode_tps": 66,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 41,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 39,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 131072,
    "mode": "mtp",
    "decode_tps": 24.6,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731",
    "quant": "UD-IQ2_XXS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "vulkan",
    "ctx": 131072,
    "mode": null,
    "decode_tps": 12,
    "date": ""
   },
   {
    "model": "Qwen3 0.6B",
    "quant": "Q8_0",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 266,
    "date": ""
   },
   {
    "model": "Qwen3 0.6B",
    "quant": "Q8_0",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 208.7,
    "date": ""
   },
   {
    "model": "LFM2.5 8B-A1B",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 176.5,
    "date": ""
   },
   {
    "model": "Qwen3-30B-A3B-Instruct-2507",
    "quant": "IQ4_XS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 103.2,
    "date": ""
   },
   {
    "model": "Qwen3-Coder 30B-A3B",
    "quant": "Q4_K_S",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 98,
    "date": ""
   },
   {
    "model": "Qwen3-Coder 30B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 97.1,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-heretic",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 96,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash 0731 1M [DUAL 2xSpark]",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 95.5,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B-NVFP4-Fast (Unsloth)",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 92.6,
    "date": ""
   },
   {
    "model": "GPT-OSS-20B",
    "quant": "Q4_K_XL",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 90.5,
    "date": ""
   },
   {
    "model": "Qwen3-Coder 30B-A3B",
    "quant": "IQ4_XS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 90.4,
    "date": ""
   },
   {
    "model": "Qwen3 30B-A3B NEO-MAX",
    "quant": "IQ4_XS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 87.4,
    "date": ""
   },
   {
    "model": "Ornith-1.0-35B-AEON",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 84.8,
    "date": ""
   },
   {
    "model": "Qwen3.6 35B-A3B",
    "quant": "Q4_0",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 81.3,
    "date": ""
   },
   {
    "model": "Qwen3-30B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 79.1,
    "date": ""
   },
   {
    "model": "Nemotron Cascade 2 30B-A3B",
    "quant": "IQ4_XS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 79,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 77.4,
    "date": ""
   },
   {
    "model": "Nemotron 3 Nano 30B-A3B",
    "quant": "IQ4_XS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 76,
    "date": ""
   },
   {
    "model": "Qwen3.5 35B-A3B",
    "quant": "IQ4_XS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 75.2,
    "date": ""
   },
   {
    "model": "Gemma 4 26B-A4B IT QAT",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 74.8,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "mtp-3",
    "decode_tps": 74.6,
    "date": ""
   },
   {
    "model": "Qwen3-Coder 30B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 73.7,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "mtp-2",
    "decode_tps": 72.8,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "mtp-3",
    "decode_tps": 68.3,
    "date": ""
   },
   {
    "model": "Qwen AgentWorld 35B-A3B",
    "quant": "IQ4_XS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 65.7,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "mtp-3",
    "decode_tps": 65.6,
    "date": ""
   },
   {
    "model": "Qwen3.5 35B-A3B",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 64.9,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "mtp-2",
    "decode_tps": 64.5,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "mtp-2",
    "decode_tps": 64.4,
    "date": ""
   },
   {
    "model": "Nemotron 3 Nano Omni 30B-A3B Reasoning",
    "quant": "MXFP4",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 64.3,
    "date": ""
   },
   {
    "model": "Qwen3.6 35B-A3B",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 63.8,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 62.8,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "Q4_K_XL",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 62.7,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 62.2,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "mtp-3",
    "decode_tps": 62.1,
    "date": ""
   },
   {
    "model": "Qwen3-Coder-Next 80B-A3B",
    "quant": "IQ4_XS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 61.9,
    "date": ""
   },
   {
    "model": "Nemotron Labs Audex 30B-A3B text-only",
    "quant": "MXFP4",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 60.7,
    "date": ""
   },
   {
    "model": "Qwen3.6 35B-A3B",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "ollama",
    "ctx": 2048,
    "mode": null,
    "decode_tps": 60.6,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "mtp-2",
    "decode_tps": 58.9,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "baseline",
    "decode_tps": 58.7,
    "date": ""
   },
   {
    "model": "NVIDIA Nemotron 3 Nano Omni 30B-A3B Reasoning",
    "quant": "MXFP4",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 56.6,
    "date": ""
   },
   {
    "model": "Ornith-1.0-35B",
    "quant": "Q8_0",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 56.1,
    "date": ""
   },
   {
    "model": "Qwen3-Next 80B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 54.9,
    "date": ""
   },
   {
    "model": "Qwen3.5 35B-A3B",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 54.7,
    "date": ""
   },
   {
    "model": "Gemma 4 26B-A4B IT",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 54.2,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "baseline",
    "decode_tps": 54.1,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 53.5,
    "date": ""
   },
   {
    "model": "Nemotron 3 Nano Omni 30B-A3B",
    "quant": "NVFP4",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 53.2,
    "date": ""
   },
   {
    "model": "llama-2-7b",
    "quant": "Q4_0",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 53,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 53,
    "date": ""
   },
   {
    "model": "Qwen3.6 35B-A3B",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 52.7,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash DSpark 1M [DUAL 2xSpark]",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 52.6,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 51.7,
    "date": ""
   },
   {
    "model": "gpt-oss-120b",
    "quant": "MXFP4",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 50.6,
    "date": ""
   },
   {
    "model": "llama-2-7b",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 50.4,
    "date": ""
   },
   {
    "model": "Qwen3-Coder-Next-80B-A3B",
    "quant": "Q4_K_XL",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 50.1,
    "date": ""
   },
   {
    "model": "Qwen3-Next 80B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 49.6,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 49.2,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 48.9,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "baseline",
    "decode_tps": 48.7,
    "date": ""
   },
   {
    "model": "Gemma 4 26B-A4B",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 48.5,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": null,
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 48.2,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 48.1,
    "date": ""
   },
   {
    "model": "Nemotron-3-Nano-30B-A3B",
    "quant": null,
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 47.9,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 47.9,
    "date": ""
   },
   {
    "model": "gpt-oss-20b-F16",
    "quant": null,
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 47.8,
    "date": ""
   },
   {
    "model": "Kimi-Linear-48B-A3B",
    "quant": "Q8_0",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 47.2,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 46.5,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "baseline",
    "decode_tps": 46,
    "date": ""
   },
   {
    "model": "shisa-v2-llama3.1-8b.i1",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 43.5,
    "date": ""
   },
   {
    "model": "gemma-4-26B-A4B-it",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 41.8,
    "date": ""
   },
   {
    "model": "Nemotron-3-Nano-Omni-30B-A3B",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 41.7,
    "date": ""
   },
   {
    "model": "Nemotron-3-Nano-30B-A3B",
    "quant": "FP8",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 40.7,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": "mtp-3",
    "decode_tps": 37.3,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731",
    "quant": "UD-IQ3_XXS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 524288,
    "mode": "dspark",
    "decode_tps": 35.7,
    "date": ""
   },
   {
    "model": "Qwen3.8-Flash-Next",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 35.1,
    "date": ""
   },
   {
    "model": "gpt-oss-120b-F16",
    "quant": null,
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 33.9,
    "date": ""
   },
   {
    "model": "Nemotron-Labs-3-Puzzle-75B-A9B",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 32.8,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash (0731)",
    "quant": null,
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 32.8,
    "date": ""
   },
   {
    "model": "Kolibri-1",
    "quant": "FP8",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 32,
    "date": ""
   },
   {
    "model": "MiniMax-M3 428B [DUAL 2xSpark]",
    "quant": "GPTQ",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 30.2,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731",
    "quant": null,
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "rocm",
    "ctx": 131072,
    "mode": "dspark",
    "decode_tps": 29.6,
    "date": ""
   },
   {
    "model": "Gemma 4 12B IT QAT",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 29.3,
    "date": ""
   },
   {
    "model": "Gemma-4-12B-IT",
    "quant": "Q4_K_M",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 28.4,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B-AEON",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": "dflash-10",
    "decode_tps": 27.9,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B",
    "quant": null,
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 65536,
    "mode": "main-low",
    "decode_tps": 27.8,
    "date": ""
   },
   {
    "model": "Laguna-S-2.1",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 27.3,
    "date": ""
   },
   {
    "model": "Qwen3.8-Flash-Next",
    "quant": "IQ4_XS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 27.2,
    "date": ""
   },
   {
    "model": "Qwen-AgentWorld-35B-A3B",
    "quant": "BF16",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 26.9,
    "date": ""
   },
   {
    "model": "Laguna-S-2.1 [nspec=15]",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 25.9,
    "date": ""
   },
   {
    "model": "GLM-5.3-Flash",
    "quant": null,
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 25.1,
    "date": ""
   },
   {
    "model": "Nemotron-3-Super-120B-A12B",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 24.5,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B",
    "quant": null,
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 65536,
    "mode": "main-off",
    "decode_tps": 23.9,
    "date": ""
   },
   {
    "model": "GLM-4.5-Air",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 23.4,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B-NVFP4 (Unsloth)",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": "mtp-2",
    "decode_tps": 23.3,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "sglang",
    "ctx": null,
    "mode": "dspark-8",
    "decode_tps": 22.8,
    "date": ""
   },
   {
    "model": "dots.llm1.inst",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 22.7,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B",
    "quant": null,
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 65536,
    "mode": "main-medium",
    "decode_tps": 22.6,
    "date": ""
   },
   {
    "model": "Gemma-4-26B-A4B-IT",
    "quant": "BF16",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 22.3,
    "date": ""
   },
   {
    "model": "Qwen3.5-122B-A10B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 21.6,
    "date": ""
   },
   {
    "model": "MiMo-V2.5 Omni [DUAL 2xSpark]",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 21.5,
    "date": ""
   },
   {
    "model": "Qwen3.8 27B",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "ollama",
    "ctx": 4096,
    "mode": null,
    "decode_tps": 20.4,
    "date": ""
   },
   {
    "model": "Llama-4-Scout-17B-16E-Instruct",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 20.2,
    "date": ""
   },
   {
    "model": "Step-3.7-Flash",
    "quant": "IQ4_XS",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 19.9,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731",
    "quant": "UD-IQ3_XXS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 19.8,
    "date": ""
   },
   {
    "model": "Nex-N2-Pro-397B-A17B",
    "quant": null,
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 18.9,
    "date": ""
   },
   {
    "model": "Nemotron 3 Super 120B-A12B",
    "quant": "IQ4_XS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 18.9,
    "date": ""
   },
   {
    "model": "Hunyuan-A13B-Instruct",
    "quant": "UD-Q6_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 18.4,
    "date": ""
   },
   {
    "model": "Llama 4 Scout 109B",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 18.3,
    "date": ""
   },
   {
    "model": "SuperQwen3.8-27B-abliterated",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 17.3,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731",
    "quant": "UD-IQ2_XXS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 16.2,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731",
    "quant": "UD-IQ2_XXS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 16.1,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731",
    "quant": "UD-IQ3_XXS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 16,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B",
    "quant": null,
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 131072,
    "mode": "main-low",
    "decode_tps": 16,
    "date": ""
   },
   {
    "model": "Qwen3-235B-A22B-Instruct-2507",
    "quant": "UD-Q3_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 15.9,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731",
    "quant": "UD-IQ3_XXS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 15.8,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B",
    "quant": null,
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 131072,
    "mode": "main-medium",
    "decode_tps": 15.6,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B",
    "quant": null,
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 131072,
    "mode": "main-xhigh",
    "decode_tps": 15,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731",
    "quant": "UD-IQ2_XXS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 14.8,
    "date": ""
   },
   {
    "model": "Mistral-Small-3.1-24B-Instruct-2503",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 14.7,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B",
    "quant": null,
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 131072,
    "mode": "main-off",
    "decode_tps": 14.7,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731",
    "quant": "UD-IQ3_XXS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 14.5,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B",
    "quant": null,
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 65536,
    "mode": "variance-low",
    "decode_tps": 14,
    "date": ""
   },
   {
    "model": "Qwen3.5-397B-A17B [DUAL 2xSpark]",
    "quant": "IQ4_NL",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 13.6,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "mtp-3",
    "decode_tps": 13.5,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B",
    "quant": null,
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 65536,
    "mode": "variance-medium",
    "decode_tps": 13.5,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "mtp-3",
    "decode_tps": 13.3,
    "date": ""
   },
   {
    "model": "DeepSeek V4 Flash 284B",
    "quant": "UD-IQ2_XXS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 13.3,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B",
    "quant": null,
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 65536,
    "mode": "main-xhigh",
    "decode_tps": 13.3,
    "date": ""
   },
   {
    "model": "Qwen3.6 27B MTP NVFP4 v3",
    "quant": "NVFP4",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 13.2,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731",
    "quant": "UD-IQ2_XXS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 13.1,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B",
    "quant": null,
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 65536,
    "mode": "budget-medium",
    "decode_tps": 12.9,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "mtp-3",
    "decode_tps": 12.8,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "mtp-2",
    "decode_tps": 12.4,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "mtp-3",
    "decode_tps": 12.2,
    "date": ""
   },
   {
    "model": "gemma-3-27b-it",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 12.1,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "mtp-2",
    "decode_tps": 11.7,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "mtp-2",
    "decode_tps": 11.7,
    "date": ""
   },
   {
    "model": "Gemma 4 31B IT QAT",
    "quant": "Q4_0",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 11.4,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "mtp-2",
    "decode_tps": 10.8,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731",
    "quant": "UD-IQ2_XXS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 9.1,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731",
    "quant": "UD-IQ3_XXS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 9,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "Q8_0",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 7.8,
    "date": ""
   },
   {
    "model": "Qwen3.6 27B MTP",
    "quant": "Q8_0",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 7.7,
    "date": ""
   },
   {
    "model": "Qwopus3.6-27B-Coder",
    "quant": "Q8_0",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 7.6,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 6.6,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 6.6,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 6.6,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 6.6,
    "date": ""
   },
   {
    "model": "Gemma-4-31B-IT",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 6.6,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "baseline",
    "decode_tps": 6.5,
    "date": ""
   },
   {
    "model": "Qwen3-32B",
    "quant": "Q8_0",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 6.4,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "baseline",
    "decode_tps": 6.3,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "baseline",
    "decode_tps": 6.3,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "UD-Q8_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "baseline",
    "decode_tps": 6.1,
    "date": ""
   },
   {
    "model": "shisa-v2-llama3.3-70b.i1",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 5.1,
    "date": ""
   },
   {
    "model": "Llama 3.1 70B",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "ollama",
    "ctx": null,
    "mode": null,
    "decode_tps": 4.7,
    "date": ""
   },
   {
    "model": "Ornith-1.5-35B-A3B",
    "quant": "FP4",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": null,
    "ctx": null,
    "mode": "mtp",
    "decode_tps": 105.6,
    "date": ""
   },
   {
    "model": "NVIDIA Nemotron 3.5 Lightning 30B",
    "quant": "FP4",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": null,
    "ctx": null,
    "mode": "mtp",
    "decode_tps": 95.2,
    "date": ""
   },
   {
    "model": "Gemma-4-26B-A4B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 78,
    "date": ""
   },
   {
    "model": "Nex N2.5 Mini 35B-A3B",
    "quant": "FP4",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": null,
    "ctx": null,
    "mode": null,
    "decode_tps": 76.9,
    "date": ""
   },
   {
    "model": "Ornith 1.5 35B-A3B",
    "quant": "FP4",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": null,
    "ctx": null,
    "mode": null,
    "decode_tps": 76.9,
    "date": ""
   },
   {
    "model": "Nex N2.5 Mini 35B-A3B",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": null,
    "ctx": null,
    "mode": null,
    "decode_tps": 72.8,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 262144,
    "mode": null,
    "decode_tps": 72,
    "date": ""
   },
   {
    "model": "Ornith 1.5 35B-A3B",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": null,
    "ctx": null,
    "mode": null,
    "decode_tps": 71.6,
    "date": ""
   },
   {
    "model": "GLM-5.3-Flash",
    "quant": null,
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": 262144,
    "mode": null,
    "decode_tps": 38,
    "date": ""
   },
   {
    "model": "Qwen 3.8 27B",
    "quant": "FP4",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": null,
    "ctx": null,
    "mode": "mtp-4",
    "decode_tps": 33.8,
    "date": ""
   },
   {
    "model": "DeepSeek V4 Flash 284B",
    "quant": "IQ2_XXS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": null,
    "ctx": null,
    "mode": "mtp",
    "decode_tps": 32,
    "date": ""
   },
   {
    "model": "Qwen 3.8 27B",
    "quant": "FP4",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": null,
    "ctx": 32768,
    "mode": "mtp",
    "decode_tps": 26.9,
    "date": ""
   },
   {
    "model": "Muse-Glimmer-30B",
    "quant": "FP4",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "rocm",
    "ctx": null,
    "mode": "dflash",
    "decode_tps": 26.5,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "vulkan",
    "ctx": 262144,
    "mode": "mtp",
    "decode_tps": 23.5,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B VL",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "mtp",
    "decode_tps": 22,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B-DFlash2",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "vulkan",
    "ctx": 32768,
    "mode": "dflash2",
    "decode_tps": 21.2,
    "date": ""
   },
   {
    "model": "Qwen3.6-27B",
    "quant": "UD-Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 262144,
    "mode": "mtp-5",
    "decode_tps": 21,
    "date": ""
   },
   {
    "model": "Qwen 3.8 27B",
    "quant": "FP8",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": "mtp",
    "decode_tps": 19,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B",
    "quant": "Q8_0",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "vulkan",
    "ctx": 131072,
    "mode": "mtp",
    "decode_tps": 17,
    "date": ""
   },
   {
    "model": "Qwen 3.8 27B",
    "quant": "FP4",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "vulkan",
    "ctx": 16384,
    "mode": null,
    "decode_tps": 13.9,
    "date": ""
   },
   {
    "model": "Qwen 3.8 27B",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": null,
    "ctx": null,
    "mode": null,
    "decode_tps": 12.4,
    "date": ""
   },
   {
    "model": "Qwen 3.8 27B",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 32768,
    "mode": null,
    "decode_tps": 12.3,
    "date": ""
   },
   {
    "model": "Qwen 3.8 27B",
    "quant": "FP16",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 5,
    "date": ""
   },
   {
    "model": "llama-3.2-1b",
    "quant": "MLX-4BIT",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "mlx",
    "ctx": 512,
    "mode": null,
    "decode_tps": 527.7,
    "date": ""
   },
   {
    "model": "llama-3.2-1b",
    "quant": "Q4_K_M",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 439.6,
    "date": ""
   },
   {
    "model": "Qwen3.8-Flash-Next",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "sglang",
    "ctx": 262144,
    "mode": "c6",
    "decode_tps": 306,
    "date": ""
   },
   {
    "model": "Qwen3.8-Flash-Next",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "sglang",
    "ctx": 262144,
    "mode": null,
    "decode_tps": 275.4,
    "date": ""
   },
   {
    "model": "Qwen3.8-Flash-Next",
    "quant": "FP8",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": null,
    "ctx": null,
    "mode": null,
    "decode_tps": 266.8,
    "date": ""
   },
   {
    "model": "qwen3-4b",
    "quant": "MLX-4BIT",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "mlx",
    "ctx": 512,
    "mode": null,
    "decode_tps": 186,
    "date": ""
   },
   {
    "model": "qwen3-4b",
    "quant": "Q4_K_M",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 159.7,
    "date": ""
   },
   {
    "model": "qwen3.6-35b-a3b",
    "quant": "MLX-4BIT",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "mlx",
    "ctx": 512,
    "mode": null,
    "decode_tps": 151.1,
    "date": ""
   },
   {
    "model": "qwen3-30b-a3b",
    "quant": "Q4_K_M",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 148.5,
    "date": ""
   },
   {
    "model": "qwen3-30b-a3b",
    "quant": "MLX-4BIT",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "mlx",
    "ctx": 512,
    "mode": null,
    "decode_tps": 147.2,
    "date": ""
   },
   {
    "model": "Qwen3 1.7B",
    "quant": "Q4_K_M",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": 2048,
    "mode": null,
    "decode_tps": 146.1,
    "date": ""
   },
   {
    "model": "qwen3.6-35b-a3b",
    "quant": "Q4_K_M",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 124.3,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731",
    "quant": "4-bit",
    "hardware": "Apple Mac Studio (M-series Ultra)",
    "hw": "mac-ultra",
    "backend": "mlx",
    "ctx": 13900,
    "mode": "batch8",
    "decode_tps": 112,
    "date": ""
   },
   {
    "model": "qwen3-8b",
    "quant": "MLX-4BIT",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "mlx",
    "ctx": 512,
    "mode": null,
    "decode_tps": 110.5,
    "date": ""
   },
   {
    "model": "Qwen3.6 35B-A3B",
    "quant": "MLX-4BIT",
    "hardware": "Apple Mac Studio (M-series Ultra)",
    "hw": "mac-ultra",
    "backend": "mlx",
    "ctx": 32000,
    "mode": null,
    "decode_tps": 109.1,
    "date": ""
   },
   {
    "model": "qwen3-8b",
    "quant": "Q4_K_M",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 97.7,
    "date": ""
   },
   {
    "model": "Ministral 3B",
    "quant": "Q4_K_M",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": 2048,
    "mode": null,
    "decode_tps": 86.6,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "FP8",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "metrale",
    "ctx": 2048,
    "mode": "mtp-1",
    "decode_tps": 85,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-DSpark",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": 1000000,
    "mode": "peak",
    "decode_tps": 84.3,
    "date": ""
   },
   {
    "model": "Qwen3 30B (MoE)",
    "quant": "Q4_K_M",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": 2048,
    "mode": null,
    "decode_tps": 83.8,
    "date": ""
   },
   {
    "model": "GLM-5.3-Flash",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "sglang",
    "ctx": 131072,
    "mode": "dflash2-c12",
    "decode_tps": 83.2,
    "date": ""
   },
   {
    "model": "Qwen3.8-Flash-Next",
    "quant": "MLX-4BIT",
    "hardware": "Apple Mac Studio (M-series Ultra)",
    "hw": "mac-ultra",
    "backend": "mlx",
    "ctx": null,
    "mode": "mtp",
    "decode_tps": 76.4,
    "date": ""
   },
   {
    "model": "gemma-3-12b",
    "quant": "MLX-4BIT",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "mlx",
    "ctx": 512,
    "mode": null,
    "decode_tps": 66.9,
    "date": ""
   },
   {
    "model": "Qwen3.8-Flash-Next",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "sglang",
    "ctx": 262144,
    "mode": "c1",
    "decode_tps": 63,
    "date": ""
   },
   {
    "model": "qwen3-14b",
    "quant": "MLX-4BIT",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "mlx",
    "ctx": 512,
    "mode": null,
    "decode_tps": 62.5,
    "date": ""
   },
   {
    "model": "gemma-3-12b",
    "quant": "Q4_K_M",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 61.5,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731",
    "quant": null,
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "cuda",
    "ctx": null,
    "mode": null,
    "decode_tps": 59,
    "date": ""
   },
   {
    "model": "qwen3-14b",
    "quant": "Q4_K_M",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 57.4,
    "date": ""
   },
   {
    "model": "Qwen3 14B",
    "quant": "MLX-4BIT",
    "hardware": "Apple Mac Studio (M-series Ultra)",
    "hw": "mac-ultra",
    "backend": "mlx",
    "ctx": 32000,
    "mode": null,
    "decode_tps": 55.4,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731-JA-REAP-K216",
    "quant": "EXL3-3BPW",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": null,
    "ctx": 256000,
    "mode": null,
    "decode_tps": 55,
    "date": ""
   },
   {
    "model": "gpt-oss-120b",
    "quant": "MXFP4",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 128,
    "mode": null,
    "decode_tps": 52.9,
    "date": ""
   },
   {
    "model": "gpt-oss-120b",
    "quant": "MXFP4",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 128,
    "mode": null,
    "decode_tps": 50.2,
    "date": ""
   },
   {
    "model": "Qwen3-Coder-Next",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": null,
    "ctx": 64000,
    "mode": null,
    "decode_tps": 50,
    "date": ""
   },
   {
    "model": "Qwen3.8-Flash-Next",
    "quant": "W4B",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "halogen",
    "ctx": 32768,
    "mode": "mtp",
    "decode_tps": 46,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731",
    "quant": "4-bit",
    "hardware": "Apple Mac Studio (M-series Ultra)",
    "hw": "mac-ultra",
    "backend": "mlx",
    "ctx": 13900,
    "mode": "single",
    "decode_tps": 45,
    "date": ""
   },
   {
    "model": "Qwen3.8-Flash-Next",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": 262144,
    "mode": "mtp-2",
    "decode_tps": 44.2,
    "date": ""
   },
   {
    "model": "Qwen3.8 27B",
    "quant": "MLX-4BIT",
    "hardware": "Apple Mac Studio (M-series Ultra)",
    "hw": "mac-ultra",
    "backend": "mlx",
    "ctx": 32000,
    "mode": null,
    "decode_tps": 42.6,
    "date": ""
   },
   {
    "model": "Qwen3 8B",
    "quant": "Q4_K_M",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": 2048,
    "mode": null,
    "decode_tps": 42,
    "date": ""
   },
   {
    "model": "mistral-small-3.1-24b",
    "quant": "MLX-4BIT",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "mlx",
    "ctx": 512,
    "mode": null,
    "decode_tps": 41.1,
    "date": ""
   },
   {
    "model": "Ministral 8B",
    "quant": "Q4_K_M",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": 2048,
    "mode": null,
    "decode_tps": 40.1,
    "date": ""
   },
   {
    "model": "Qwen3.8-Flash-Next",
    "quant": "W4B",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "halogen",
    "ctx": 1500,
    "mode": "baseline",
    "decode_tps": 37.6,
    "date": ""
   },
   {
    "model": "mistral-small-3.1-24b",
    "quant": "Q4_K_M",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 37.2,
    "date": ""
   },
   {
    "model": "Qwen3.8-Flash-Next-REAP-288",
    "quant": "MLX-4BIT",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "mlx",
    "ctx": null,
    "mode": null,
    "decode_tps": 37,
    "date": ""
   },
   {
    "model": "GLM-5.3-Flash",
    "quant": "FP8",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": null,
    "decode_tps": 36,
    "date": ""
   },
   {
    "model": "Qwen3.8-Flash-Next",
    "quant": "IQ3_XXS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 156000,
    "mode": "mtp",
    "decode_tps": 35.7,
    "date": ""
   },
   {
    "model": "GLM-5.3-Flash",
    "quant": "MLX-4BIT",
    "hardware": "Apple Mac Studio (M-series Ultra)",
    "hw": "mac-ultra",
    "backend": "mlx",
    "ctx": null,
    "mode": null,
    "decode_tps": 34.2,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731",
    "quant": "2.58bpw-mix",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": null,
    "ctx": null,
    "mode": "dspark",
    "decode_tps": 34.2,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": 1000000,
    "mode": "prose",
    "decode_tps": 33.2,
    "date": ""
   },
   {
    "model": "gemma-3-27b",
    "quant": "MLX-4BIT",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "mlx",
    "ctx": 512,
    "mode": null,
    "decode_tps": 31.8,
    "date": ""
   },
   {
    "model": "Qwen3.6-35B-A3B",
    "quant": "BF16",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "mlx",
    "ctx": null,
    "mode": null,
    "decode_tps": 30.8,
    "date": ""
   },
   {
    "model": "gemma-3-27b",
    "quant": "Q4_K_M",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 29.9,
    "date": ""
   },
   {
    "model": "DeepSeek-V4-Flash-0731-MXFP4-MLX-Abliterated",
    "quant": "MXFP4",
    "hardware": "Apple Mac Studio (M-series Ultra)",
    "hw": "mac-ultra",
    "backend": "mlx",
    "ctx": 1048576,
    "mode": null,
    "decode_tps": 29.6,
    "date": ""
   },
   {
    "model": "GLM-5.3-Flash",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "sglang",
    "ctx": 131072,
    "mode": "dflash2-c1",
    "decode_tps": 28.6,
    "date": ""
   },
   {
    "model": "qwen3-32b",
    "quant": "MLX-4BIT",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "mlx",
    "ctx": 512,
    "mode": null,
    "decode_tps": 28.4,
    "date": ""
   },
   {
    "model": "Qwen3.8-Flash-Next-REAP-288",
    "quant": "MLX-4BIT",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "mlx",
    "ctx": null,
    "mode": null,
    "decode_tps": 28,
    "date": ""
   },
   {
    "model": "Qwen3.8-Flash-Next",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": null,
    "mode": "mtp-1",
    "decode_tps": 27.6,
    "date": ""
   },
   {
    "model": "Qwen3.8-Flash-Next",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": 262144,
    "mode": "baseline",
    "decode_tps": 27.3,
    "date": ""
   },
   {
    "model": "Qwen3.8-Flash-Next",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": 1000000,
    "mode": null,
    "decode_tps": 26.7,
    "date": ""
   },
   {
    "model": "qwen3-32b",
    "quant": "Q4_K_M",
    "hardware": "Apple M-series Max (MacBook Pro / Mac Studio Max)",
    "hw": "mac-max",
    "backend": "llama.cpp",
    "ctx": 512,
    "mode": null,
    "decode_tps": 26.4,
    "date": ""
   },
   {
    "model": "SuperQwen3.8-27b-abliterated",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "vllm",
    "ctx": 262043,
    "mode": "spec-k5",
    "decode_tps": 25.8,
    "date": ""
   },
   {
    "model": "Ministral 14B",
    "quant": "Q4_K_M",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": 2048,
    "mode": null,
    "decode_tps": 25.7,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B",
    "quant": "NVFP4",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "metrale",
    "ctx": 2048,
    "mode": "mtp",
    "decode_tps": 25.1,
    "date": ""
   },
   {
    "model": "Qwen3.8-Flash-Next",
    "quant": "Q4_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 24,
    "date": ""
   },
   {
    "model": "GLM-5.3-Flash Abliterated",
    "quant": "MLX-4BIT",
    "hardware": "Apple Mac Studio (M-series Ultra)",
    "hw": "mac-ultra",
    "backend": "mlx",
    "ctx": 16384,
    "mode": "mtp",
    "decode_tps": 24,
    "date": ""
   },
   {
    "model": "Qwen3.8-Flash-Next",
    "quant": "IQ3_XXS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 156000,
    "mode": "baseline",
    "decode_tps": 23.5,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B",
    "quant": "UD-Q6_K_XL",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 131072,
    "mode": "dflash2",
    "decode_tps": 17,
    "date": ""
   },
   {
    "model": "Motif-3 315B",
    "quant": "UD-IQ2_XXS",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": 32768,
    "mode": null,
    "decode_tps": 16.5,
    "date": ""
   },
   {
    "model": "SuperQwen3.8-Flash-Next-abliterated",
    "quant": "FP8",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "sglang",
    "ctx": 8192,
    "mode": null,
    "decode_tps": 16.2,
    "date": ""
   },
   {
    "model": "Qwen3.8-27B",
    "quant": "IQ4_XS",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": 65536,
    "mode": null,
    "decode_tps": 14.1,
    "date": ""
   },
   {
    "model": "Qwen 3.8 27B",
    "quant": "FP4",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 14,
    "date": ""
   },
   {
    "model": "Qwen 3.8 27B",
    "quant": "Q4_K_M",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 12.3,
    "date": ""
   },
   {
    "model": "Qwen3 32B (Dense)",
    "quant": "Q4_K_M",
    "hardware": "NVIDIA DGX Spark (GB10)",
    "hw": "dgx-spark",
    "backend": "llama.cpp",
    "ctx": 2048,
    "mode": null,
    "decode_tps": 10.5,
    "date": ""
   },
   {
    "model": "Qwen 3.8 27B",
    "quant": "FP8",
    "hardware": "AMD Strix Halo (Ryzen AI Max+ 395)",
    "hw": "strix-halo",
    "backend": "llama.cpp",
    "ctx": null,
    "mode": null,
    "decode_tps": 7.7,
    "date": ""
   }
  ]
 },
 "hf_trending": [
  {
   "id": "ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-GGUF",
   "format": "GGUF",
   "created": "2026-09-07",
   "downloads": 3405442,
   "url": "https://huggingface.co/ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-GGUF"
  },
  {
   "id": "unsloth/embeddinggemma-2-GGUF",
   "format": "GGUF",
   "created": "2026-10-06",
   "downloads": 29692,
   "url": "https://huggingface.co/unsloth/embeddinggemma-2-GGUF"
  },
  {
   "id": "ISTA-DASLab/Qwen3.8-27B-GSQ-RCO-GGUF",
   "format": "GGUF",
   "created": "2026-08-28",
   "downloads": 1517150,
   "url": "https://huggingface.co/ISTA-DASLab/Qwen3.8-27B-GSQ-RCO-GGUF"
  },
  {
   "id": "unsloth/Qwen3.8-27B-GGUF",
   "format": "GGUF",
   "created": "2026-08-13",
   "downloads": 6518857,
   "url": "https://huggingface.co/unsloth/Qwen3.8-27B-GGUF"
  },
  {
   "id": "ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-Coder-GGUF",
   "format": "GGUF",
   "created": "2026-09-26",
   "downloads": 599869,
   "url": "https://huggingface.co/ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-Coder-GGUF"
  },
  {
   "id": "nvidia/Nemotron-3-Diarization",
   "format": "GGUF",
   "created": "2026-09-01",
   "downloads": 66040,
   "url": "https://huggingface.co/nvidia/Nemotron-3-Diarization"
  },
  {
   "id": "unsloth/Qwen3.8-Flash-Next-GGUF",
   "format": "GGUF",
   "created": "2026-08-26",
   "downloads": 1315187,
   "url": "https://huggingface.co/unsloth/Qwen3.8-Flash-Next-GGUF"
  },
  {
   "id": "unsloth/Qwen-Image-2.1-GGUF",
   "format": "GGUF",
   "created": "2026-09-21",
   "downloads": 805462,
   "url": "https://huggingface.co/unsloth/Qwen-Image-2.1-GGUF"
  },
  {
   "id": "ggml-org/embeddinggemma-2-GGUF",
   "format": "GGUF",
   "created": "2026-10-06",
   "downloads": 9518,
   "url": "https://huggingface.co/ggml-org/embeddinggemma-2-GGUF"
  },
  {
   "id": "LiquidAI/d1-3B-GGUF",
   "format": "GGUF",
   "created": "2026-10-06",
   "downloads": 2353,
   "url": "https://huggingface.co/LiquidAI/d1-3B-GGUF"
  },
  {
   "id": "ggml-org/Laya-GGUF",
   "format": "GGUF",
   "created": "2026-10-01",
   "downloads": 10112,
   "url": "https://huggingface.co/ggml-org/Laya-GGUF"
  },
  {
   "id": "nvidia/nemotron-3.5-asr-streaming-0.6b",
   "format": "GGUF",
   "created": "2026-05-15",
   "downloads": 1289845,
   "url": "https://huggingface.co/nvidia/nemotron-3.5-asr-streaming-0.6b"
  },
  {
   "id": "LiquidAI/d1-omni-600M-GGUF",
   "format": "GGUF",
   "created": "2026-10-06",
   "downloads": 826,
   "url": "https://huggingface.co/LiquidAI/d1-omni-600M-GGUF"
  },
  {
   "id": "bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF",
   "format": "GGUF",
   "created": "2026-09-21",
   "downloads": 496776,
   "url": "https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF"
  },
  {
   "id": "ggml-org/Clef-Flash-GGUF",
   "format": "GGUF",
   "created": "2026-10-02",
   "downloads": 26306,
   "url": "https://huggingface.co/ggml-org/Clef-Flash-GGUF"
  },
  {
   "id": "unsloth/gemma-4-26B-A4B-it-GGUF",
   "format": "GGUF",
   "created": "2026-04-01",
   "downloads": 565548,
   "url": "https://huggingface.co/unsloth/gemma-4-26B-A4B-it-GGUF"
  },
  {
   "id": "mlx-community/clef-flash-4bit",
   "format": "MLX",
   "created": "2026-10-01",
   "downloads": 3804,
   "url": "https://huggingface.co/mlx-community/clef-flash-4bit"
  },
  {
   "id": "mlx-community/Qwen3.8-27B-4bit",
   "format": "MLX",
   "created": "2026-08-14",
   "downloads": 429036,
   "url": "https://huggingface.co/mlx-community/Qwen3.8-27B-4bit"
  },
  {
   "id": "mlx-community/clef-4bit",
   "format": "MLX",
   "created": "2026-10-01",
   "downloads": 1355,
   "url": "https://huggingface.co/mlx-community/clef-4bit"
  },
  {
   "id": "mlx-community/clef-8bit",
   "format": "MLX",
   "created": "2026-10-02",
   "downloads": 1085,
   "url": "https://huggingface.co/mlx-community/clef-8bit"
  },
  {
   "id": "mlx-community/embeddinggemma-2-bf16",
   "format": "MLX",
   "created": "2026-10-06",
   "downloads": 1001,
   "url": "https://huggingface.co/mlx-community/embeddinggemma-2-bf16"
  },
  {
   "id": "mlx-community/Llama-3-Groq-8B-Tool-Use-4bit",
   "format": "MLX",
   "created": "2024-07-17",
   "downloads": 603,
   "url": "https://huggingface.co/mlx-community/Llama-3-Groq-8B-Tool-Use-4bit"
  },
  {
   "id": "mlx-community/Qwen3.5-9B-MLX-4bit",
   "format": "MLX",
   "created": "2026-03-02",
   "downloads": 31373,
   "url": "https://huggingface.co/mlx-community/Qwen3.5-9B-MLX-4bit"
  },
  {
   "id": "mlx-community/Qwen3.8-27B-8bit",
   "format": "MLX",
   "created": "2026-08-14",
   "downloads": 78188,
   "url": "https://huggingface.co/mlx-community/Qwen3.8-27B-8bit"
  }
 ],
 "models": [
  {
   "name": "Qwen3.8-27B",
   "params_b": 27.3,
   "architecture": "dense",
   "active_params_b": null,
   "capabilities": [
    "coding",
    "vision",
    "chat",
    "general"
   ],
   "gguf_repo": "unsloth/Qwen3.8-27B-GGUF",
   "mmproj": true,
   "quants": [
    {
     "name": "UD-Q2_K_XL",
     "size_mb": 9829,
     "bpw": 2.88
    },
    {
     "name": "UD-Q3_K_XL",
     "size_mb": 13146,
     "bpw": 3.85
    },
    {
     "name": "UD-IQ4_XS",
     "size_mb": 14253,
     "bpw": 4.17
    },
    {
     "name": "UD-Q4_K_XL",
     "size_mb": 17559,
     "bpw": 5.14
    },
    {
     "name": "UD-Q5_K_XL",
     "size_mb": 20877,
     "bpw": 6.11
    },
    {
     "name": "UD-Q6_K_XL",
     "size_mb": 25299,
     "bpw": 7.41
    },
    {
     "name": "Q8_0",
     "size_mb": 29047,
     "bpw": 8.51
    },
    {
     "name": "UD-Q8_K_XL",
     "size_mb": 31458,
     "bpw": 9.21
    }
   ],
   "notes": "Dense 27B with native vision (needs the mmproj file). Strong all-rounder that fits one 24-32 GB GPU at 4-bit.",
   "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
   "fetched": "2026-10-09T04:33+08:00"
  },
  {
   "name": "Qwen3.8-Flash-Next",
   "params_b": 176.9,
   "architecture": "moe",
   "active_params_b": 6,
   "capabilities": [
    "coding",
    "vision",
    "chat",
    "general"
   ],
   "gguf_repo": "unsloth/Qwen3.8-Flash-Next-GGUF",
   "mmproj": true,
   "quants": [
    {
     "name": "UD-Q2_K_XL",
     "size_mb": 78869,
     "bpw": 3.57
    },
    {
     "name": "UD-Q3_K_XL",
     "size_mb": 89986,
     "bpw": 4.07
    },
    {
     "name": "UD-IQ4_XS",
     "size_mb": 93683,
     "bpw": 4.24
    },
    {
     "name": "UD-Q4_K_XL",
     "size_mb": 111335,
     "bpw": 5.03
    },
    {
     "name": "UD-Q5_K_XL",
     "size_mb": 158286,
     "bpw": 7.16
    },
    {
     "name": "UD-Q6_K_XL",
     "size_mb": 169165,
     "bpw": 7.65
    },
    {
     "name": "Q8_0",
     "size_mb": 188225,
     "bpw": 8.51
    }
   ],
   "notes": "MoE, ~6B active of 125B plus 51B n-gram embedding. Native vision (mmproj). MTP draft files in the repo's MTP/ folder speed up decode.",
   "source": "https://huggingface.co/Qwen/Qwen3.8-Flash-Next",
   "fetched": "2026-10-09T04:33+08:00"
  },
  {
   "name": "GLM-5.3-Flash",
   "params_b": 320.8,
   "architecture": "moe",
   "active_params_b": 18,
   "capabilities": [
    "coding",
    "chat",
    "general",
    "vision"
   ],
   "gguf_repo": "unsloth/GLM-5.3-Flash-GGUF",
   "mmproj": true,
   "quants": [
    {
     "name": "UD-Q2_K_XL",
     "size_mb": 108720,
     "bpw": 2.71
    },
    {
     "name": "UD-Q3_K_XL",
     "size_mb": 147536,
     "bpw": 3.68
    },
    {
     "name": "UD-IQ4_XS",
     "size_mb": 156822,
     "bpw": 3.91
    },
    {
     "name": "UD-Q4_K_XL",
     "size_mb": 199707,
     "bpw": 4.98
    },
    {
     "name": "UD-Q5_K_XL",
     "size_mb": 240306,
     "bpw": 5.99
    },
    {
     "name": "UD-Q6_K_XL",
     "size_mb": 291833,
     "bpw": 7.28
    },
    {
     "name": "Q8_0",
     "size_mb": 340982,
     "bpw": 8.5
    }
   ],
   "notes": "MoE, 320B total, 18B active. Needs 128 GB+ even at 2-bit.",
   "source": "https://huggingface.co/zai-org/GLM-5.3-Flash",
   "fetched": "2026-10-09T04:33+08:00"
  },
  {
   "name": "gpt-oss-120b",
   "params_b": 116.8,
   "architecture": "moe",
   "active_params_b": 5.1,
   "capabilities": [
    "coding",
    "chat",
    "general"
   ],
   "gguf_repo": "unsloth/gpt-oss-120b-GGUF",
   "mmproj": false,
   "quants": [
    {
     "name": "Q4_K_M",
     "size_mb": 62769,
     "bpw": 4.3
    },
    {
     "name": "Q5_K_M",
     "size_mb": 62890,
     "bpw": 4.31
    },
    {
     "name": "UD-Q4_K_XL",
     "size_mb": 63016,
     "bpw": 4.32
    },
    {
     "name": "Q6_K",
     "size_mb": 63284,
     "bpw": 4.33
    },
    {
     "name": "UD-Q6_K_XL",
     "size_mb": 63284,
     "bpw": 4.33
    },
    {
     "name": "Q8_0",
     "size_mb": 63387,
     "bpw": 4.34
    },
    {
     "name": "UD-Q8_K_XL",
     "size_mb": 64473,
     "bpw": 4.41
    }
   ],
   "notes": "MoE, 117B total, 5.1B active. Ships natively in MXFP4, so every GGUF quant is about the same size: take the native one.",
   "source": "https://huggingface.co/openai/gpt-oss-120b",
   "fetched": "2026-10-09T04:33+08:00"
  },
  {
   "name": "gpt-oss-20b",
   "params_b": 20.9,
   "architecture": "moe",
   "active_params_b": 3.6,
   "capabilities": [
    "coding",
    "chat",
    "general"
   ],
   "gguf_repo": "unsloth/gpt-oss-20b-GGUF",
   "mmproj": false,
   "quants": [
    {
     "name": "Q4_K_M",
     "size_mb": 11625,
     "bpw": 4.45
    },
    {
     "name": "Q5_K_M",
     "size_mb": 11717,
     "bpw": 4.48
    },
    {
     "name": "UD-Q4_K_XL",
     "size_mb": 11872,
     "bpw": 4.54
    },
    {
     "name": "Q6_K",
     "size_mb": 12041,
     "bpw": 4.61
    },
    {
     "name": "UD-Q6_K_XL",
     "size_mb": 12041,
     "bpw": 4.61
    },
    {
     "name": "Q8_0",
     "size_mb": 12110,
     "bpw": 4.63
    },
    {
     "name": "UD-Q8_K_XL",
     "size_mb": 13195,
     "bpw": 5.05
    }
   ],
   "notes": "MoE, 21B total, 3.6B active. Native MXFP4, fits a 16 GB GPU.",
   "source": "https://huggingface.co/openai/gpt-oss-20b",
   "fetched": "2026-10-09T04:33+08:00"
  },
  {
   "name": "Qwen3-Coder-Next",
   "params_b": 79.7,
   "architecture": "moe",
   "active_params_b": 3,
   "capabilities": [
    "coding",
    "general"
   ],
   "gguf_repo": "unsloth/Qwen3-Coder-Next-GGUF",
   "mmproj": false,
   "quants": [
    {
     "name": "UD-Q2_K_XL",
     "size_mb": 26761,
     "bpw": 2.69
    },
    {
     "name": "UD-Q3_K_XL",
     "size_mb": 36283,
     "bpw": 3.64
    },
    {
     "name": "UD-IQ4_XS",
     "size_mb": 38429,
     "bpw": 3.86
    },
    {
     "name": "IQ4_XS",
     "size_mb": 42676,
     "bpw": 4.29
    },
    {
     "name": "MXFP4",
     "size_mb": 48034,
     "bpw": 4.82
    },
    {
     "name": "Q4_K_M",
     "size_mb": 48528,
     "bpw": 4.87
    },
    {
     "name": "UD-Q4_K_XL",
     "size_mb": 49608,
     "bpw": 4.98
    },
    {
     "name": "Q5_K_M",
     "size_mb": 56848,
     "bpw": 5.71
    },
    {
     "name": "UD-Q5_K_XL",
     "size_mb": 59541,
     "bpw": 5.98
    },
    {
     "name": "Q6_K",
     "size_mb": 65614,
     "bpw": 6.59
    },
    {
     "name": "UD-Q6_K_XL",
     "size_mb": 73119,
     "bpw": 7.34
    },
    {
     "name": "Q8_0",
     "size_mb": 84812,
     "bpw": 8.52
    },
    {
     "name": "UD-Q8_K_XL",
     "size_mb": 86349,
     "bpw": 8.67
    }
   ],
   "notes": "MoE, 80B total, 3B active. Coding agent model.",
   "source": "https://huggingface.co/Qwen/Qwen3-Coder-Next",
   "fetched": "2026-10-09T04:33+08:00"
  },
  {
   "name": "Qwen3-Coder-30B-A3B-Instruct",
   "params_b": 30.5,
   "architecture": "moe",
   "active_params_b": 3.3,
   "capabilities": [
    "coding",
    "general"
   ],
   "gguf_repo": "unsloth/Qwen3-Coder-30B-A3B-Instruct-GGUF",
   "mmproj": false,
   "quants": [
    {
     "name": "UD-Q2_K_XL",
     "size_mb": 11789,
     "bpw": 3.09
    },
    {
     "name": "UD-Q3_K_XL",
     "size_mb": 13806,
     "bpw": 3.62
    },
    {
     "name": "IQ4_XS",
     "size_mb": 16378,
     "bpw": 4.29
    },
    {
     "name": "UD-Q4_K_XL",
     "size_mb": 17665,
     "bpw": 4.63
    },
    {
     "name": "Q4_K_M",
     "size_mb": 18557,
     "bpw": 4.86
    },
    {
     "name": "Q5_K_M",
     "size_mb": 21726,
     "bpw": 5.69
    },
    {
     "name": "UD-Q5_K_XL",
     "size_mb": 21740,
     "bpw": 5.7
    },
    {
     "name": "Q6_K",
     "size_mb": 25093,
     "bpw": 6.57
    },
    {
     "name": "UD-Q6_K_XL",
     "size_mb": 26340,
     "bpw": 6.9
    },
    {
     "name": "Q8_0",
     "size_mb": 32484,
     "bpw": 8.51
    },
    {
     "name": "UD-Q8_K_XL",
     "size_mb": 35990,
     "bpw": 9.43
    }
   ],
   "notes": "MoE, 30.5B total, 3.3B active. Very fast coder for 24 GB GPUs and unified-memory boxes.",
   "source": "https://huggingface.co/Qwen/Qwen3-Coder-30B-A3B-Instruct",
   "fetched": "2026-10-09T04:33+08:00"
  },
  {
   "name": "Qwen3.6-35B-A3B",
   "params_b": 34.7,
   "architecture": "moe",
   "active_params_b": 3,
   "capabilities": [
    "coding",
    "vision",
    "chat",
    "general"
   ],
   "gguf_repo": "unsloth/Qwen3.6-35B-A3B-GGUF",
   "mmproj": true,
   "quants": [
    {
     "name": "UD-Q2_K_XL",
     "size_mb": 12291,
     "bpw": 2.84
    },
    {
     "name": "UD-Q3_K_XL",
     "size_mb": 16846,
     "bpw": 3.89
    },
    {
     "name": "UD-IQ4_XS",
     "size_mb": 17731,
     "bpw": 4.09
    },
    {
     "name": "MXFP4",
     "size_mb": 21706,
     "bpw": 5.01
    },
    {
     "name": "UD-Q4_K_XL",
     "size_mb": 22360,
     "bpw": 5.16
    },
    {
     "name": "UD-Q5_K_XL",
     "size_mb": 26593,
     "bpw": 6.14
    },
    {
     "name": "UD-Q6_K_XL",
     "size_mb": 31844,
     "bpw": 7.35
    },
    {
     "name": "Q8_0",
     "size_mb": 36903,
     "bpw": 8.52
    },
    {
     "name": "UD-Q8_K_XL",
     "size_mb": 38451,
     "bpw": 8.87
    }
   ],
   "notes": "MoE, 35B total, ~3B active. MTP variant unsloth/Qwen3.6-35B-A3B-MTP-GGUF decodes faster.",
   "source": "https://huggingface.co/unsloth/Qwen3.6-35B-A3B-GGUF",
   "fetched": "2026-10-09T04:33+08:00"
  },
  {
   "name": "Gemma-4-26B-A4B-it",
   "params_b": 25.2,
   "architecture": "moe",
   "active_params_b": 3.8,
   "capabilities": [
    "vision",
    "chat",
    "general"
   ],
   "gguf_repo": "unsloth/gemma-4-26B-A4B-it-GGUF",
   "mmproj": true,
   "quants": [
    {
     "name": "UD-Q2_K_XL",
     "size_mb": 10547,
     "bpw": 3.34
    },
    {
     "name": "UD-Q3_K_XL",
     "size_mb": 12907,
     "bpw": 4.09
    },
    {
     "name": "UD-IQ4_XS",
     "size_mb": 13597,
     "bpw": 4.31
    },
    {
     "name": "MXFP4",
     "size_mb": 16551,
     "bpw": 5.25
    },
    {
     "name": "UD-Q4_K_XL",
     "size_mb": 17011,
     "bpw": 5.39
    },
    {
     "name": "UD-Q5_K_XL",
     "size_mb": 21218,
     "bpw": 6.73
    },
    {
     "name": "UD-Q6_K_XL",
     "size_mb": 23295,
     "bpw": 7.39
    },
    {
     "name": "Q8_0",
     "size_mb": 26860,
     "bpw": 8.52
    },
    {
     "name": "UD-Q8_K_XL",
     "size_mb": 27636,
     "bpw": 8.76
    }
   ],
   "notes": "MoE, 25.2B total, 3.8B active. Multimodal (mmproj). QAT build: unsloth/gemma-4-26B-A4B-it-qat-GGUF.",
   "source": "https://huggingface.co/google/gemma-4-26B-A4B-it",
   "fetched": "2026-10-09T04:33+08:00"
  },
  {
   "name": "Nemotron-3-Nano-30B-A3B",
   "params_b": 31.6,
   "architecture": "moe",
   "active_params_b": 3.5,
   "capabilities": [
    "chat",
    "general"
   ],
   "gguf_repo": "unsloth/Nemotron-3-Nano-30B-A3B-GGUF",
   "mmproj": false,
   "quants": [
    {
     "name": "IQ4_XS",
     "size_mb": 18169,
     "bpw": 4.6
    },
    {
     "name": "UD-Q2_K_XL",
     "size_mb": 19920,
     "bpw": 5.05
    },
    {
     "name": "UD-Q3_K_XL",
     "size_mb": 19939,
     "bpw": 5.05
    },
    {
     "name": "UD-Q4_K_XL",
     "size_mb": 22834,
     "bpw": 5.78
    },
    {
     "name": "Q4_K_M",
     "size_mb": 24574,
     "bpw": 6.23
    },
    {
     "name": "Q5_K_M",
     "size_mb": 26149,
     "bpw": 6.62
    },
    {
     "name": "UD-Q5_K_XL",
     "size_mb": 27506,
     "bpw": 6.97
    },
    {
     "name": "Q6_K",
     "size_mb": 33508,
     "bpw": 8.49
    },
    {
     "name": "UD-Q6_K_XL",
     "size_mb": 33508,
     "bpw": 8.49
    },
    {
     "name": "Q8_0",
     "size_mb": 33585,
     "bpw": 8.51
    },
    {
     "name": "UD-Q8_K_XL",
     "size_mb": 40448,
     "bpw": 10.25
    }
   ],
   "notes": "Hybrid Mamba MoE, 30B total, 3.5B active, 1M context.",
   "source": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16",
   "fetched": "2026-10-09T04:33+08:00"
  },
  {
   "name": "Gemma-4-E4B-it",
   "params_b": 7.5,
   "architecture": "dense",
   "active_params_b": null,
   "capabilities": [
    "vision",
    "chat",
    "general"
   ],
   "gguf_repo": "unsloth/gemma-4-E4B-it-GGUF",
   "mmproj": true,
   "quants": [
    {
     "name": "UD-Q2_K_XL",
     "size_mb": 3757,
     "bpw": 4.0
    },
    {
     "name": "UD-Q3_K_XL",
     "size_mb": 4588,
     "bpw": 4.88
    },
    {
     "name": "IQ4_XS",
     "size_mb": 4715,
     "bpw": 5.02
    },
    {
     "name": "Q4_K_M",
     "size_mb": 4977,
     "bpw": 5.3
    },
    {
     "name": "UD-Q4_K_XL",
     "size_mb": 5126,
     "bpw": 5.45
    },
    {
     "name": "Q5_K_M",
     "size_mb": 5482,
     "bpw": 5.83
    },
    {
     "name": "UD-Q5_K_XL",
     "size_mb": 6656,
     "bpw": 7.08
    },
    {
     "name": "Q6_K",
     "size_mb": 7075,
     "bpw": 7.53
    },
    {
     "name": "UD-Q6_K_XL",
     "size_mb": 7458,
     "bpw": 7.94
    },
    {
     "name": "Q8_0",
     "size_mb": 8193,
     "bpw": 8.72
    },
    {
     "name": "UD-Q8_K_XL",
     "size_mb": 8712,
     "bpw": 9.27
    }
   ],
   "notes": "Small multimodal model for laptops and 8 GB GPUs (mmproj for vision).",
   "source": "https://huggingface.co/google/gemma-4-E4B-it",
   "fetched": "2026-10-09T04:33+08:00"
  },
  {
   "name": "Qwen3.5-9B",
   "params_b": 9.0,
   "architecture": "dense",
   "active_params_b": null,
   "capabilities": [
    "chat",
    "general",
    "vision"
   ],
   "gguf_repo": "unsloth/Qwen3.5-9B-GGUF",
   "mmproj": true,
   "quants": [
    {
     "name": "UD-Q2_K_XL",
     "size_mb": 4122,
     "bpw": 3.68
    },
    {
     "name": "UD-Q3_K_XL",
     "size_mb": 5054,
     "bpw": 4.52
    },
    {
     "name": "IQ4_XS",
     "size_mb": 5169,
     "bpw": 4.62
    },
    {
     "name": "Q4_K_M",
     "size_mb": 5681,
     "bpw": 5.08
    },
    {
     "name": "UD-Q4_K_XL",
     "size_mb": 5966,
     "bpw": 5.33
    },
    {
     "name": "Q5_K_M",
     "size_mb": 6578,
     "bpw": 5.88
    },
    {
     "name": "UD-Q5_K_XL",
     "size_mb": 6744,
     "bpw": 6.03
    },
    {
     "name": "Q6_K",
     "size_mb": 7458,
     "bpw": 6.66
    },
    {
     "name": "UD-Q6_K_XL",
     "size_mb": 8757,
     "bpw": 7.82
    },
    {
     "name": "Q8_0",
     "size_mb": 9528,
     "bpw": 8.51
    },
    {
     "name": "UD-Q8_K_XL",
     "size_mb": 12974,
     "bpw": 11.59
    }
   ],
   "notes": "Small dense model for 8-12 GB GPUs.",
   "source": "https://huggingface.co/unsloth/Qwen3.5-9B-GGUF",
   "fetched": "2026-10-09T04:33+08:00"
  },
  {
   "name": "Qwen3.5-4B",
   "params_b": 4.2,
   "architecture": "dense",
   "active_params_b": null,
   "capabilities": [
    "chat",
    "general",
    "vision"
   ],
   "gguf_repo": "unsloth/Qwen3.5-4B-GGUF",
   "mmproj": true,
   "quants": [
    {
     "name": "UD-Q2_K_XL",
     "size_mb": 1941,
     "bpw": 3.69
    },
    {
     "name": "UD-Q3_K_XL",
     "size_mb": 2436,
     "bpw": 4.63
    },
    {
     "name": "IQ4_XS",
     "size_mb": 2477,
     "bpw": 4.71
    },
    {
     "name": "Q4_K_M",
     "size_mb": 2741,
     "bpw": 5.21
    },
    {
     "name": "UD-Q4_K_XL",
     "size_mb": 2912,
     "bpw": 5.54
    },
    {
     "name": "Q5_K_M",
     "size_mb": 3144,
     "bpw": 5.98
    },
    {
     "name": "UD-Q5_K_XL",
     "size_mb": 3251,
     "bpw": 6.18
    },
    {
     "name": "Q6_K",
     "size_mb": 3526,
     "bpw": 6.71
    },
    {
     "name": "UD-Q6_K_XL",
     "size_mb": 4146,
     "bpw": 7.89
    },
    {
     "name": "Q8_0",
     "size_mb": 4482,
     "bpw": 8.53
    },
    {
     "name": "UD-Q8_K_XL",
     "size_mb": 5952,
     "bpw": 11.32
    }
   ],
   "notes": "Tiny dense model for CPU-only boxes and 6 GB GPUs.",
   "source": "https://huggingface.co/unsloth/Qwen3.5-4B-GGUF",
   "fetched": "2026-10-09T04:33+08:00"
  }
 ],
 "hardware": [
  {
   "id": "rtx-3090",
   "name": "NVIDIA GeForce RTX 3090",
   "vendor": "nvidia",
   "class": "consumer-gpu",
   "arch": "Ampere sm_86",
   "mem_gb": 24,
   "mem_type": "GDDR6X",
   "bw_gbps": 936,
   "fp4": false,
   "fp8": false,
   "engines": [
    "llama.cpp",
    "exllamav3",
    "vllm",
    "ollama"
   ],
   "source": "https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3090-3090ti/",
   "confidence": "high",
   "match": [
    "rtx 3090"
   ],
   "exclude": [
    "3090 ti",
    "laptop",
    "mobile"
   ]
  },
  {
   "id": "rtx-4090",
   "name": "NVIDIA GeForce RTX 4090",
   "vendor": "nvidia",
   "class": "consumer-gpu",
   "arch": "Ada sm_89",
   "mem_gb": 24,
   "mem_type": "GDDR6X",
   "bw_gbps": 1008,
   "fp4": false,
   "fp8": true,
   "engines": [
    "llama.cpp",
    "exllamav3",
    "vllm",
    "sglang",
    "ollama"
   ],
   "source": "https://www.nvidia.com/en-us/geforce/graphics-cards/40-series/rtx-4090/",
   "confidence": "high",
   "match": [
    "rtx 4090"
   ],
   "exclude": [
    "laptop",
    "mobile",
    "4090 d"
   ]
  },
  {
   "id": "rtx-5090",
   "name": "NVIDIA GeForce RTX 5090",
   "vendor": "nvidia",
   "class": "consumer-gpu",
   "arch": "Blackwell sm_120",
   "mem_gb": 32,
   "mem_type": "GDDR7",
   "bw_gbps": 1792,
   "fp4": true,
   "fp8": true,
   "engines": [
    "llama.cpp",
    "exllamav3",
    "vllm",
    "sglang",
    "tensorrt-llm",
    "ollama"
   ],
   "source": "https://www.nvidia.com/en-us/geforce/graphics-cards/50-series/rtx-5090/",
   "confidence": "high",
   "match": [
    "rtx 5090"
   ],
   "exclude": [
    "laptop",
    "mobile"
   ]
  },
  {
   "id": "rtx-pro-6000-bw",
   "name": "NVIDIA RTX PRO 6000 Blackwell",
   "vendor": "nvidia",
   "class": "workstation-gpu",
   "arch": "Blackwell sm_120",
   "mem_gb": 96,
   "mem_type": "GDDR7 ECC",
   "bw_gbps": 1792,
   "fp4": true,
   "fp8": true,
   "engines": [
    "vllm",
    "sglang",
    "tensorrt-llm",
    "llama.cpp",
    "exllamav3"
   ],
   "source": "https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-6000/",
   "confidence": "high",
   "match": [
    "rtx pro 6000"
   ],
   "exclude": [
    "max-q"
   ]
  },
  {
   "id": "dgx-spark",
   "name": "NVIDIA DGX Spark (GB10)",
   "vendor": "nvidia",
   "class": "unified-memory-box",
   "arch": "Blackwell GB10 sm_121, ARM64",
   "mem_gb": 128,
   "mem_type": "LPDDR5X unified",
   "bw_gbps": 273,
   "fp4": true,
   "fp8": true,
   "engines": [
    "llama.cpp",
    "vllm",
    "sglang",
    "tensorrt-llm",
    "ollama"
   ],
   "source": "https://www.nvidia.com/en-us/products/workstations/dgx-spark/",
   "confidence": "high",
   "match": [
    "gb10",
    "dgx spark"
   ],
   "exclude": []
  },
  {
   "id": "jetson-agx-thor",
   "name": "NVIDIA Jetson AGX Thor",
   "vendor": "nvidia",
   "class": "edge",
   "arch": "Blackwell, ARM64",
   "mem_gb": 128,
   "mem_type": "LPDDR5X unified",
   "bw_gbps": 273,
   "fp4": true,
   "fp8": true,
   "engines": [
    "llama.cpp",
    "vllm",
    "tensorrt-llm"
   ],
   "source": "https://www.nvidia.com/en-us/autonomous-machines/embedded-systems/jetson-thor/",
   "confidence": "high",
   "match": [
    "thor"
   ],
   "exclude": []
  },
  {
   "id": "h100-sxm",
   "name": "NVIDIA H100 SXM",
   "vendor": "nvidia",
   "class": "datacenter-gpu",
   "arch": "Hopper sm_90",
   "mem_gb": 80,
   "mem_type": "HBM3",
   "bw_gbps": 3350,
   "fp4": false,
   "fp8": true,
   "engines": [
    "vllm",
    "sglang",
    "tensorrt-llm"
   ],
   "source": "https://www.nvidia.com/en-us/data-center/h100/",
   "confidence": "high",
   "match": [
    "h100 80gb hbm3",
    "h100 sxm"
   ],
   "exclude": [
    "pcie",
    "nvl"
   ]
  },
  {
   "id": "h200",
   "name": "NVIDIA H200",
   "vendor": "nvidia",
   "class": "datacenter-gpu",
   "arch": "Hopper sm_90",
   "mem_gb": 141,
   "mem_type": "HBM3e",
   "bw_gbps": 4800,
   "fp4": false,
   "fp8": true,
   "engines": [
    "vllm",
    "sglang",
    "tensorrt-llm"
   ],
   "source": "https://www.nvidia.com/en-us/data-center/h200/",
   "confidence": "high",
   "match": [
    "h200"
   ],
   "exclude": [
    "nvl"
   ]
  },
  {
   "id": "b200",
   "name": "NVIDIA B200",
   "vendor": "nvidia",
   "class": "datacenter-gpu",
   "arch": "Blackwell sm_100",
   "mem_gb": 180,
   "mem_type": "HBM3e",
   "bw_gbps": 8000,
   "fp4": true,
   "fp8": true,
   "engines": [
    "vllm",
    "sglang",
    "tensorrt-llm"
   ],
   "source": "https://www.nvidia.com/en-us/data-center/dgx-b200/",
   "confidence": "medium",
   "notes": "180 GB usable of 192 GB physical; vendors quote 7.7 to 8 TB/s.",
   "match": [
    "b200"
   ],
   "exclude": []
  },
  {
   "id": "strix-halo-395",
   "name": "AMD Ryzen AI Max+ 395 (Strix Halo)",
   "vendor": "amd",
   "class": "unified-memory-box",
   "arch": "RDNA 3.5 gfx1151 iGPU + XDNA2 NPU",
   "mem_gb": 128,
   "mem_type": "LPDDR5X-8000 unified",
   "bw_gbps": 256,
   "fp4": false,
   "fp8": false,
   "engines": [
    "llama.cpp",
    "lemonade",
    "ollama",
    "vllm"
   ],
   "notes": "llama.cpp Vulkan usually wins decode, ROCm wins prompt processing. vLLM only via community gfx1151 builds.",
   "source": "https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-395.html",
   "confidence": "high",
   "match": [
    "gfx1151",
    "8060s",
    "ai max+ 395",
    "ai max 395",
    "strix halo"
   ],
   "exclude": []
  },
  {
   "id": "rx-7900-xtx",
   "name": "AMD Radeon RX 7900 XTX",
   "vendor": "amd",
   "class": "consumer-gpu",
   "arch": "RDNA 3 gfx1100",
   "mem_gb": 24,
   "mem_type": "GDDR6",
   "bw_gbps": 960,
   "fp4": false,
   "fp8": false,
   "engines": [
    "llama.cpp",
    "ollama",
    "vllm"
   ],
   "source": "https://www.amd.com/en/products/graphics/desktops/radeon/7000-series/amd-radeon-rx-7900xtx.html",
   "confidence": "high",
   "match": [
    "7900 xtx"
   ],
   "exclude": []
  },
  {
   "id": "rx-9070-xt",
   "name": "AMD Radeon RX 9070 XT",
   "vendor": "amd",
   "class": "consumer-gpu",
   "arch": "RDNA 4 gfx1201",
   "mem_gb": 16,
   "mem_type": "GDDR6",
   "bw_gbps": 640,
   "fp4": false,
   "fp8": true,
   "engines": [
    "llama.cpp",
    "ollama"
   ],
   "source": "https://www.amd.com/en/products/graphics/desktops/radeon/9000-series/amd-radeon-rx-9070xt.html",
   "confidence": "high",
   "match": [
    "9070 xt"
   ],
   "exclude": []
  },
  {
   "id": "mi300x",
   "name": "AMD Instinct MI300X",
   "vendor": "amd",
   "class": "datacenter-gpu",
   "arch": "CDNA 3 gfx942",
   "mem_gb": 192,
   "mem_type": "HBM3",
   "bw_gbps": 5300,
   "fp4": false,
   "fp8": true,
   "engines": [
    "vllm",
    "sglang"
   ],
   "source": "https://www.amd.com/en/products/accelerators/instinct/mi300/mi300x.html",
   "confidence": "high",
   "match": [
    "mi300x"
   ],
   "exclude": []
  },
  {
   "id": "m4-max",
   "name": "Apple M4 Max (40-core GPU)",
   "vendor": "apple",
   "class": "unified-memory-box",
   "arch": "Apple Silicon, Metal",
   "mem_gb": 128,
   "mem_type": "LPDDR5X unified",
   "bw_gbps": 546,
   "fp4": false,
   "fp8": false,
   "engines": [
    "mlx-lm",
    "llama.cpp",
    "ollama",
    "lm-studio"
   ],
   "notes": "Max memory depends on configuration (36 to 128 GB). mlx_lm.server is OpenAI-compatible, no wrapper needed.",
   "source": "https://www.apple.com/macbook-pro/specs/",
   "confidence": "high",
   "match": [
    "m4 max"
   ],
   "exclude": []
  },
  {
   "id": "m3-ultra",
   "name": "Apple M3 Ultra",
   "vendor": "apple",
   "class": "unified-memory-box",
   "arch": "Apple Silicon, Metal",
   "mem_gb": 512,
   "mem_type": "LPDDR5 unified",
   "bw_gbps": 819,
   "fp4": false,
   "fp8": false,
   "engines": [
    "mlx-lm",
    "llama.cpp",
    "ollama",
    "lm-studio"
   ],
   "notes": "Max memory depends on configuration (96 to 512 GB).",
   "source": "https://www.apple.com/mac-studio/specs/",
   "confidence": "high",
   "match": [
    "m3 ultra"
   ],
   "exclude": []
  },
  {
   "id": "arc-b580",
   "name": "Intel Arc B580",
   "vendor": "intel",
   "class": "consumer-gpu",
   "arch": "Xe2 Battlemage",
   "mem_gb": 12,
   "mem_type": "GDDR6",
   "bw_gbps": 456,
   "fp4": false,
   "fp8": false,
   "engines": [
    "llama.cpp",
    "openvino",
    "ollama"
   ],
   "notes": "llama.cpp via SYCL or Vulkan backend.",
   "source": "https://www.intel.com/content/www/us/en/products/sku/241598/intel-arc-b580-graphics/specifications.html",
   "confidence": "high",
   "match": [
    "arc b580"
   ],
   "exclude": []
  },
  {
   "id": "arc-pro-b60",
   "name": "Intel Arc Pro B60",
   "vendor": "intel",
   "class": "workstation-gpu",
   "arch": "Xe2 Battlemage",
   "mem_gb": 24,
   "mem_type": "GDDR6",
   "bw_gbps": 456,
   "fp4": false,
   "fp8": false,
   "engines": [
    "llama.cpp",
    "openvino",
    "vllm"
   ],
   "source": "https://download.intel.com/newsroom/2025/client-computing/Intel-Arc-Pro-B60-Data-Sheet.pdf",
   "confidence": "high",
   "match": [
    "arc pro b60"
   ],
   "exclude": []
  },
  {
   "id": "rtx-3060-12gb",
   "name": "NVIDIA GeForce RTX 3060 12GB",
   "vendor": "nvidia",
   "class": "consumer-gpu",
   "arch": "Ampere sm_86",
   "mem_gb": 12,
   "mem_type": "GDDR6",
   "bw_gbps": 360,
   "fp4": false,
   "fp8": false,
   "engines": [
    "llama.cpp",
    "exllamav3",
    "ollama",
    "vllm"
   ],
   "source": "https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3060-3060ti/",
   "confidence": "medium",
   "match": [
    "rtx 3060"
   ],
   "exclude": [
    "3060 ti",
    "laptop",
    "mobile"
   ],
   "notes": "The 8 GB RTX 3060 variant has 240 GB/s."
  },
  {
   "id": "rtx-4060-ti",
   "name": "NVIDIA GeForce RTX 4060 Ti (16GB)",
   "vendor": "nvidia",
   "class": "consumer-gpu",
   "arch": "Ada sm_89",
   "mem_gb": 16,
   "mem_type": "GDDR6",
   "bw_gbps": 288,
   "fp4": false,
   "fp8": true,
   "engines": [
    "llama.cpp",
    "exllamav3",
    "ollama",
    "vllm"
   ],
   "source": "https://www.nvidia.com/en-us/geforce/graphics-cards/40-series/rtx-4060-4060ti/",
   "confidence": "high",
   "match": [
    "rtx 4060 ti"
   ],
   "exclude": [
    "laptop",
    "mobile"
   ],
   "notes": "8 GB and 16 GB variants share the 288 GB/s bus."
  },
  {
   "id": "rtx-4070-ti-super",
   "name": "NVIDIA GeForce RTX 4070 Ti SUPER",
   "vendor": "nvidia",
   "class": "consumer-gpu",
   "arch": "Ada sm_89",
   "mem_gb": 16,
   "mem_type": "GDDR6X",
   "bw_gbps": 672,
   "fp4": false,
   "fp8": true,
   "engines": [
    "llama.cpp",
    "exllamav3",
    "ollama",
    "vllm"
   ],
   "source": "https://www.nvidia.com/en-us/geforce/graphics-cards/40-series/rtx-4070-family/",
   "confidence": "high",
   "match": [
    "rtx 4070 ti super"
   ],
   "exclude": [
    "laptop",
    "mobile"
   ]
  },
  {
   "id": "rtx-4080-super",
   "name": "NVIDIA GeForce RTX 4080 SUPER",
   "vendor": "nvidia",
   "class": "consumer-gpu",
   "arch": "Ada sm_89",
   "mem_gb": 16,
   "mem_type": "GDDR6X",
   "bw_gbps": 736,
   "fp4": false,
   "fp8": true,
   "engines": [
    "llama.cpp",
    "exllamav3",
    "ollama",
    "vllm"
   ],
   "source": "https://www.nvidia.com/en-us/geforce/graphics-cards/40-series/rtx-4080-family/",
   "confidence": "high",
   "match": [
    "rtx 4080 super"
   ],
   "exclude": [
    "laptop",
    "mobile"
   ]
  },
  {
   "id": "rtx-4080",
   "name": "NVIDIA GeForce RTX 4080",
   "vendor": "nvidia",
   "class": "consumer-gpu",
   "arch": "Ada sm_89",
   "mem_gb": 16,
   "mem_type": "GDDR6X",
   "bw_gbps": 717,
   "fp4": false,
   "fp8": true,
   "engines": [
    "llama.cpp",
    "exllamav3",
    "ollama",
    "vllm"
   ],
   "source": "https://www.nvidia.com/en-us/geforce/graphics-cards/40-series/rtx-4080-family/",
   "confidence": "high",
   "match": [
    "rtx 4080"
   ],
   "exclude": [
    "super",
    "laptop",
    "mobile"
   ]
  },
  {
   "id": "rtx-5060-ti",
   "name": "NVIDIA GeForce RTX 5060 Ti (16GB)",
   "vendor": "nvidia",
   "class": "consumer-gpu",
   "arch": "Blackwell sm_120",
   "mem_gb": 16,
   "mem_type": "GDDR7",
   "bw_gbps": 448,
   "fp4": true,
   "fp8": true,
   "engines": [
    "llama.cpp",
    "exllamav3",
    "ollama",
    "vllm"
   ],
   "source": "https://www.nvidia.com/en-us/geforce/graphics-cards/50-series/rtx-5060-family/",
   "confidence": "high",
   "match": [
    "rtx 5060 ti"
   ],
   "exclude": [
    "laptop",
    "mobile"
   ]
  },
  {
   "id": "rtx-5070-ti",
   "name": "NVIDIA GeForce RTX 5070 Ti",
   "vendor": "nvidia",
   "class": "consumer-gpu",
   "arch": "Blackwell sm_120",
   "mem_gb": 16,
   "mem_type": "GDDR7",
   "bw_gbps": 896,
   "fp4": true,
   "fp8": true,
   "engines": [
    "llama.cpp",
    "exllamav3",
    "ollama",
    "vllm"
   ],
   "source": "https://www.nvidia.com/en-us/geforce/graphics-cards/50-series/rtx-5070-family/",
   "confidence": "high",
   "match": [
    "rtx 5070 ti"
   ],
   "exclude": [
    "laptop",
    "mobile"
   ]
  },
  {
   "id": "rtx-5080",
   "name": "NVIDIA GeForce RTX 5080",
   "vendor": "nvidia",
   "class": "consumer-gpu",
   "arch": "Blackwell sm_120",
   "mem_gb": 16,
   "mem_type": "GDDR7",
   "bw_gbps": 960,
   "fp4": true,
   "fp8": true,
   "engines": [
    "llama.cpp",
    "exllamav3",
    "ollama",
    "vllm"
   ],
   "source": "https://www.nvidia.com/en-us/geforce/graphics-cards/50-series/rtx-5080/",
   "confidence": "high",
   "match": [
    "rtx 5080"
   ],
   "exclude": [
    "laptop",
    "mobile"
   ]
  },
  {
   "id": "rtx-6000-ada",
   "name": "NVIDIA RTX 6000 Ada Generation",
   "vendor": "nvidia",
   "class": "workstation-gpu",
   "arch": "Ada sm_89",
   "mem_gb": 48,
   "mem_type": "GDDR6 ECC",
   "bw_gbps": 960,
   "fp4": false,
   "fp8": true,
   "engines": [
    "vllm",
    "sglang",
    "llama.cpp",
    "exllamav3"
   ],
   "source": "https://www.nvidia.com/en-us/products/workstations/rtx-6000/",
   "confidence": "high",
   "match": [
    "rtx 6000 ada"
   ],
   "exclude": []
  },
  {
   "id": "rtx-a6000",
   "name": "NVIDIA RTX A6000",
   "vendor": "nvidia",
   "class": "workstation-gpu",
   "arch": "Ampere sm_86",
   "mem_gb": 48,
   "mem_type": "GDDR6 ECC",
   "bw_gbps": 768,
   "fp4": false,
   "fp8": false,
   "engines": [
    "vllm",
    "llama.cpp",
    "exllamav3",
    "sglang"
   ],
   "source": "https://www.nvidia.com/en-us/products/workstations/rtx-a6000/",
   "confidence": "high",
   "match": [
    "rtx a6000"
   ],
   "exclude": [
    "ada"
   ]
  },
  {
   "id": "l40s",
   "name": "NVIDIA L40S",
   "vendor": "nvidia",
   "class": "datacenter-gpu",
   "arch": "Ada sm_89",
   "mem_gb": 48,
   "mem_type": "GDDR6 ECC",
   "bw_gbps": 864,
   "fp4": false,
   "fp8": true,
   "engines": [
    "vllm",
    "sglang",
    "tensorrt-llm",
    "llama.cpp"
   ],
   "source": "https://www.nvidia.com/en-us/data-center/l40s/",
   "confidence": "high",
   "match": [
    "l40s"
   ],
   "exclude": []
  },
  {
   "id": "a100-80gb",
   "name": "NVIDIA A100 80GB",
   "vendor": "nvidia",
   "class": "datacenter-gpu",
   "arch": "Ampere sm_80",
   "mem_gb": 80,
   "mem_type": "HBM2e",
   "bw_gbps": 1935,
   "fp4": false,
   "fp8": false,
   "engines": [
    "vllm",
    "sglang",
    "tensorrt-llm"
   ],
   "source": "https://www.nvidia.com/en-us/data-center/a100/",
   "confidence": "high",
   "match": [
    "a100 80gb",
    "a100 sxm4 80gb"
   ],
   "exclude": [],
   "notes": "PCIe 1935 GB/s; the SXM part reaches 2039 GB/s."
  },
  {
   "id": "m4-pro",
   "name": "Apple M4 Pro",
   "vendor": "apple",
   "class": "unified-memory-box",
   "arch": "Apple Silicon, Metal",
   "mem_gb": 64,
   "mem_type": "LPDDR5X unified",
   "bw_gbps": 273,
   "fp4": false,
   "fp8": false,
   "engines": [
    "mlx-lm",
    "llama.cpp",
    "ollama",
    "lm-studio"
   ],
   "source": "https://www.apple.com/mac-mini/specs/",
   "confidence": "high",
   "match": [
    "m4 pro"
   ],
   "exclude": []
  },
  {
   "id": "m3-max",
   "name": "Apple M3 Max",
   "vendor": "apple",
   "class": "unified-memory-box",
   "arch": "Apple Silicon, Metal",
   "mem_gb": 128,
   "mem_type": "LPDDR5 unified",
   "bw_gbps": 300,
   "fp4": false,
   "fp8": false,
   "engines": [
    "mlx-lm",
    "llama.cpp",
    "ollama",
    "lm-studio"
   ],
   "source": "https://support.apple.com/specs",
   "confidence": "medium",
   "match": [
    "m3 max"
   ],
   "exclude": [],
   "notes": "300 GB/s on the 30-core GPU part, 400 GB/s on the 40-core part; we take the lower."
  },
  {
   "id": "m2-max",
   "name": "Apple M2 Max",
   "vendor": "apple",
   "class": "unified-memory-box",
   "arch": "Apple Silicon, Metal",
   "mem_gb": 96,
   "mem_type": "LPDDR5 unified",
   "bw_gbps": 400,
   "fp4": false,
   "fp8": false,
   "engines": [
    "mlx-lm",
    "llama.cpp",
    "ollama",
    "lm-studio"
   ],
   "source": "https://support.apple.com/specs",
   "confidence": "high",
   "match": [
    "m2 max"
   ],
   "exclude": []
  },
  {
   "id": "m1-max",
   "name": "Apple M1 Max",
   "vendor": "apple",
   "class": "unified-memory-box",
   "arch": "Apple Silicon, Metal",
   "mem_gb": 64,
   "mem_type": "LPDDR5 unified",
   "bw_gbps": 400,
   "fp4": false,
   "fp8": false,
   "engines": [
    "mlx-lm",
    "llama.cpp",
    "ollama",
    "lm-studio"
   ],
   "source": "https://support.apple.com/specs",
   "confidence": "high",
   "match": [
    "m1 max"
   ],
   "exclude": []
  },
  {
   "id": "m2-ultra",
   "name": "Apple M2 Ultra",
   "vendor": "apple",
   "class": "unified-memory-box",
   "arch": "Apple Silicon, Metal",
   "mem_gb": 192,
   "mem_type": "LPDDR5 unified",
   "bw_gbps": 800,
   "fp4": false,
   "fp8": false,
   "engines": [
    "mlx-lm",
    "llama.cpp",
    "ollama",
    "lm-studio"
   ],
   "source": "https://support.apple.com/specs",
   "confidence": "high",
   "match": [
    "m2 ultra"
   ],
   "exclude": []
  },
  {
   "id": "m1-ultra",
   "name": "Apple M1 Ultra",
   "vendor": "apple",
   "class": "unified-memory-box",
   "arch": "Apple Silicon, Metal",
   "mem_gb": 128,
   "mem_type": "LPDDR5 unified",
   "bw_gbps": 800,
   "fp4": false,
   "fp8": false,
   "engines": [
    "mlx-lm",
    "llama.cpp",
    "ollama",
    "lm-studio"
   ],
   "source": "https://support.apple.com/specs",
   "confidence": "high",
   "match": [
    "m1 ultra"
   ],
   "exclude": []
  }
 ],
 "matrix": {
  "rtx-3090": {
   "general": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "MXFP4",
     "size_gb": 21.7,
     "est_tps": [
      117,
      205
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q5_K_XL",
     "size_gb": 21.7,
     "est_tps": [
      102,
      179
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q5_K_XL",
     "size_gb": 20.9,
     "est_tps": [
      24,
      33
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "MXFP4",
     "size_gb": 21.7,
     "est_tps": [
      117,
      205
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q5_K_XL",
     "size_gb": 21.7,
     "est_tps": [
      102,
      179
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q5_K_XL",
     "size_gb": 20.9,
     "est_tps": [
      24,
      33
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "MXFP4",
     "size_gb": 21.7,
     "est_tps": [
      117,
      205
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q5_K_XL",
     "size_gb": 20.9,
     "est_tps": [
      24,
      33
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q5_K_XL",
     "size_gb": 21.2,
     "est_tps": [
      83,
      145
     ]
    }
   ]
  },
  "rtx-4090": {
   "general": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "MXFP4",
     "size_gb": 21.7,
     "est_tps": [
      123,
      214
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q5_K_XL",
     "size_gb": 21.7,
     "est_tps": [
      107,
      188
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q5_K_XL",
     "size_gb": 20.9,
     "est_tps": [
      26,
      36
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "MXFP4",
     "size_gb": 21.7,
     "est_tps": [
      123,
      214
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q5_K_XL",
     "size_gb": 21.7,
     "est_tps": [
      107,
      188
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q5_K_XL",
     "size_gb": 20.9,
     "est_tps": [
      26,
      36
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "MXFP4",
     "size_gb": 21.7,
     "est_tps": [
      123,
      214
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q5_K_XL",
     "size_gb": 20.9,
     "est_tps": [
      26,
      36
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q5_K_XL",
     "size_gb": 21.2,
     "est_tps": [
      88,
      153
     ]
    }
   ]
  },
  "rtx-5090": {
   "general": [
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q2_K_XL",
     "size_gb": 26.8,
     "est_tps": [
      204,
      357
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q5_K_XL",
     "size_gb": 26.6,
     "est_tps": [
      149,
      261
     ]
    },
    {
     "model": "Nemotron-3-Nano-30B-A3B",
     "quant": "UD-Q5_K_XL",
     "size_gb": 27.5,
     "est_tps": [
      129,
      226
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q2_K_XL",
     "size_gb": 26.8,
     "est_tps": [
      204,
      357
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q5_K_XL",
     "size_gb": 26.6,
     "est_tps": [
      149,
      261
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q6_K_XL",
     "size_gb": 26.3,
     "est_tps": [
      134,
      234
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q5_K_XL",
     "size_gb": 26.6,
     "est_tps": [
      149,
      261
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "Q8_0",
     "size_gb": 29.0,
     "est_tps": [
      33,
      45
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q8_K_XL",
     "size_gb": 27.6,
     "est_tps": [
      107,
      188
     ]
    }
   ]
  },
  "rtx-pro-6000-bw": {
   "general": [
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      135,
      236
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-Q2_K_XL",
     "size_gb": 78.9,
     "est_tps": [
      138,
      242
     ]
    },
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q8_K_XL",
     "size_gb": 86.3,
     "est_tps": [
      124,
      218
     ]
    }
   ],
   "coding": [
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      135,
      236
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-Q2_K_XL",
     "size_gb": 78.9,
     "est_tps": [
      138,
      242
     ]
    },
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q8_K_XL",
     "size_gb": 86.3,
     "est_tps": [
      124,
      218
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-Q2_K_XL",
     "size_gb": 78.9,
     "est_tps": [
      138,
      242
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      123,
      215
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 31.5,
     "est_tps": [
      30,
      42
     ]
    }
   ]
  },
  "dgx-spark": {
   "general": [
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-IQ4_XS",
     "size_gb": 93.7,
     "est_tps": [
      31,
      54
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      34,
      60
     ]
    },
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q8_K_XL",
     "size_gb": 86.3,
     "est_tps": [
      30,
      53
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-IQ4_XS",
     "size_gb": 93.7,
     "est_tps": [
      31,
      54
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      34,
      60
     ]
    },
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q8_K_XL",
     "size_gb": 86.3,
     "est_tps": [
      30,
      53
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-IQ4_XS",
     "size_gb": 93.7,
     "est_tps": [
      31,
      54
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      29,
      52
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 31.5,
     "est_tps": [
      4.8,
      6.5
     ]
    }
   ]
  },
  "jetson-agx-thor": {
   "general": [
    {
     "model": "GLM-5.3-Flash",
     "quant": "UD-Q2_K_XL",
     "size_gb": 108.7,
     "est_tps": [
      17,
      29
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-Q4_K_XL",
     "size_gb": 111.3,
     "est_tps": [
      26,
      46
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      34,
      60
     ]
    }
   ],
   "coding": [
    {
     "model": "GLM-5.3-Flash",
     "quant": "UD-Q2_K_XL",
     "size_gb": 108.7,
     "est_tps": [
      17,
      29
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-Q4_K_XL",
     "size_gb": 111.3,
     "est_tps": [
      26,
      46
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      34,
      60
     ]
    }
   ],
   "vision": [
    {
     "model": "GLM-5.3-Flash",
     "quant": "UD-Q2_K_XL",
     "size_gb": 108.7,
     "est_tps": [
      17,
      29
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-Q4_K_XL",
     "size_gb": 111.3,
     "est_tps": [
      26,
      46
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      29,
      52
     ]
    }
   ]
  },
  "h100-sxm": {
   "general": [
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      179,
      313
     ]
    },
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q6_K_XL",
     "size_gb": 73.1,
     "est_tps": [
      180,
      315
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      167,
      293
     ]
    }
   ],
   "coding": [
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      179,
      313
     ]
    },
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q6_K_XL",
     "size_gb": 73.1,
     "est_tps": [
      180,
      315
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      167,
      293
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      167,
      293
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 31.5,
     "est_tps": [
      55,
      77
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q8_K_XL",
     "size_gb": 27.6,
     "est_tps": [
      151,
      265
     ]
    }
   ]
  },
  "h200": {
   "general": [
    {
     "model": "GLM-5.3-Flash",
     "quant": "UD-Q2_K_XL",
     "size_gb": 108.7,
     "est_tps": [
      150,
      262
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-Q4_K_XL",
     "size_gb": 111.3,
     "est_tps": [
      183,
      320
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      201,
      353
     ]
    }
   ],
   "coding": [
    {
     "model": "GLM-5.3-Flash",
     "quant": "UD-Q2_K_XL",
     "size_gb": 108.7,
     "est_tps": [
      150,
      262
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-Q4_K_XL",
     "size_gb": 111.3,
     "est_tps": [
      183,
      320
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      201,
      353
     ]
    }
   ],
   "vision": [
    {
     "model": "GLM-5.3-Flash",
     "quant": "UD-Q2_K_XL",
     "size_gb": 108.7,
     "est_tps": [
      150,
      262
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-Q4_K_XL",
     "size_gb": 111.3,
     "est_tps": [
      183,
      320
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      191,
      334
     ]
    }
   ]
  },
  "b200": {
   "general": [
    {
     "model": "GLM-5.3-Flash",
     "quant": "UD-IQ4_XS",
     "size_gb": 156.8,
     "est_tps": [
      160,
      280
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-Q5_K_XL",
     "size_gb": 158.3,
     "est_tps": [
      193,
      338
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      228,
      400
     ]
    }
   ],
   "coding": [
    {
     "model": "GLM-5.3-Flash",
     "quant": "UD-IQ4_XS",
     "size_gb": 156.8,
     "est_tps": [
      160,
      280
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-Q5_K_XL",
     "size_gb": 158.3,
     "est_tps": [
      193,
      338
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      228,
      400
     ]
    }
   ],
   "vision": [
    {
     "model": "GLM-5.3-Flash",
     "quant": "UD-IQ4_XS",
     "size_gb": 156.8,
     "est_tps": [
      160,
      280
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-Q5_K_XL",
     "size_gb": 158.3,
     "est_tps": [
      193,
      338
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      220,
      386
     ]
    }
   ]
  },
  "strix-halo-395": {
   "general": [
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-IQ4_XS",
     "size_gb": 93.7,
     "est_tps": [
      29,
      51
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      32,
      57
     ]
    },
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q8_K_XL",
     "size_gb": 86.3,
     "est_tps": [
      28,
      50
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-IQ4_XS",
     "size_gb": 93.7,
     "est_tps": [
      29,
      51
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      32,
      57
     ]
    },
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q8_K_XL",
     "size_gb": 86.3,
     "est_tps": [
      28,
      50
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-IQ4_XS",
     "size_gb": 93.7,
     "est_tps": [
      29,
      51
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      28,
      49
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 31.5,
     "est_tps": [
      4.5,
      6.1
     ]
    }
   ]
  },
  "rx-7900-xtx": {
   "general": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "MXFP4",
     "size_gb": 21.7,
     "est_tps": [
      119,
      209
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q5_K_XL",
     "size_gb": 21.7,
     "est_tps": [
      104,
      182
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q5_K_XL",
     "size_gb": 20.9,
     "est_tps": [
      25,
      34
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "MXFP4",
     "size_gb": 21.7,
     "est_tps": [
      119,
      209
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q5_K_XL",
     "size_gb": 21.7,
     "est_tps": [
      104,
      182
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q5_K_XL",
     "size_gb": 20.9,
     "est_tps": [
      25,
      34
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "MXFP4",
     "size_gb": 21.7,
     "est_tps": [
      119,
      209
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q5_K_XL",
     "size_gb": 20.9,
     "est_tps": [
      25,
      34
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q5_K_XL",
     "size_gb": 21.2,
     "est_tps": [
      85,
      148
     ]
    }
   ]
  },
  "rx-9070-xt": {
   "general": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      24,
      33
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q3_K_XL",
     "size_gb": 13.8,
     "est_tps": [
      107,
      188
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-IQ4_XS",
     "size_gb": 13.6,
     "est_tps": [
      87,
      152
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      24,
      33
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q3_K_XL",
     "size_gb": 13.8,
     "est_tps": [
      107,
      188
     ]
    },
    {
     "model": "gpt-oss-20b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 13.2,
     "est_tps": [
      81,
      141
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      24,
      33
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-IQ4_XS",
     "size_gb": 13.6,
     "est_tps": [
      87,
      152
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q2_K_XL",
     "size_gb": 12.3,
     "est_tps": [
      131,
      228
     ]
    }
   ]
  },
  "mi300x": {
   "general": [
    {
     "model": "GLM-5.3-Flash",
     "quant": "UD-IQ4_XS",
     "size_gb": 156.8,
     "est_tps": [
      131,
      229
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-Q6_K_XL",
     "size_gb": 169.2,
     "est_tps": [
      161,
      282
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      207,
      363
     ]
    }
   ],
   "coding": [
    {
     "model": "GLM-5.3-Flash",
     "quant": "UD-IQ4_XS",
     "size_gb": 156.8,
     "est_tps": [
      131,
      229
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-Q6_K_XL",
     "size_gb": 169.2,
     "est_tps": [
      161,
      282
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      207,
      363
     ]
    }
   ],
   "vision": [
    {
     "model": "GLM-5.3-Flash",
     "quant": "UD-IQ4_XS",
     "size_gb": 156.8,
     "est_tps": [
      131,
      229
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-Q6_K_XL",
     "size_gb": 169.2,
     "est_tps": [
      161,
      282
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      197,
      345
     ]
    }
   ]
  },
  "m4-max": {
   "general": [
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-IQ4_XS",
     "size_gb": 93.7,
     "est_tps": [
      55,
      97
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      61,
      107
     ]
    },
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q8_K_XL",
     "size_gb": 86.3,
     "est_tps": [
      54,
      95
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-IQ4_XS",
     "size_gb": 93.7,
     "est_tps": [
      55,
      97
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      61,
      107
     ]
    },
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q8_K_XL",
     "size_gb": 86.3,
     "est_tps": [
      54,
      95
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-IQ4_XS",
     "size_gb": 93.7,
     "est_tps": [
      55,
      97
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      53,
      93
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 31.5,
     "est_tps": [
      9.5,
      13
     ]
    }
   ]
  },
  "m3-ultra": {
   "general": [
    {
     "model": "GLM-5.3-Flash",
     "quant": "Q8_0",
     "size_gb": 341.0,
     "est_tps": [
      16,
      28
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "Q8_0",
     "size_gb": 188.2,
     "est_tps": [
      44,
      76
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      83,
      145
     ]
    }
   ],
   "coding": [
    {
     "model": "GLM-5.3-Flash",
     "quant": "Q8_0",
     "size_gb": 341.0,
     "est_tps": [
      16,
      28
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "Q8_0",
     "size_gb": 188.2,
     "est_tps": [
      44,
      76
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      83,
      145
     ]
    }
   ],
   "vision": [
    {
     "model": "GLM-5.3-Flash",
     "quant": "Q8_0",
     "size_gb": 341.0,
     "est_tps": [
      16,
      28
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "Q8_0",
     "size_gb": 188.2,
     "est_tps": [
      44,
      76
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      73,
      128
     ]
    }
   ]
  },
  "arc-b580": {
   "general": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q2_K_XL",
     "size_gb": 9.8,
     "est_tps": [
      25,
      34
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q2_K_XL",
     "size_gb": 10.5,
     "est_tps": [
      82,
      143
     ]
    },
    {
     "model": "Qwen3.5-9B",
     "quant": "Q8_0",
     "size_gb": 9.5,
     "est_tps": [
      26,
      35
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q2_K_XL",
     "size_gb": 9.8,
     "est_tps": [
      25,
      34
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q2_K_XL",
     "size_gb": 10.5,
     "est_tps": [
      82,
      143
     ]
    },
    {
     "model": "Qwen3.5-9B",
     "quant": "Q8_0",
     "size_gb": 9.5,
     "est_tps": [
      26,
      35
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q2_K_XL",
     "size_gb": 9.8,
     "est_tps": [
      25,
      34
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q2_K_XL",
     "size_gb": 10.5,
     "est_tps": [
      82,
      143
     ]
    },
    {
     "model": "Qwen3.5-9B",
     "quant": "Q8_0",
     "size_gb": 9.5,
     "est_tps": [
      26,
      35
     ]
    }
   ]
  },
  "arc-pro-b60": {
   "general": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "MXFP4",
     "size_gb": 21.7,
     "est_tps": [
      72,
      127
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q5_K_XL",
     "size_gb": 21.7,
     "est_tps": [
      61,
      107
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q5_K_XL",
     "size_gb": 20.9,
     "est_tps": [
      12,
      16
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "MXFP4",
     "size_gb": 21.7,
     "est_tps": [
      72,
      127
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q5_K_XL",
     "size_gb": 21.7,
     "est_tps": [
      61,
      107
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q5_K_XL",
     "size_gb": 20.9,
     "est_tps": [
      12,
      16
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "MXFP4",
     "size_gb": 21.7,
     "est_tps": [
      72,
      127
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q5_K_XL",
     "size_gb": 20.9,
     "est_tps": [
      12,
      16
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q5_K_XL",
     "size_gb": 21.2,
     "est_tps": [
      48,
      83
     ]
    }
   ]
  },
  "rtx-3060-12gb": {
   "general": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q2_K_XL",
     "size_gb": 9.8,
     "est_tps": [
      20,
      27
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q2_K_XL",
     "size_gb": 10.5,
     "est_tps": [
      69,
      121
     ]
    },
    {
     "model": "Qwen3.5-9B",
     "quant": "Q8_0",
     "size_gb": 9.5,
     "est_tps": [
      20,
      28
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q2_K_XL",
     "size_gb": 9.8,
     "est_tps": [
      20,
      27
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q2_K_XL",
     "size_gb": 10.5,
     "est_tps": [
      69,
      121
     ]
    },
    {
     "model": "Qwen3.5-9B",
     "quant": "Q8_0",
     "size_gb": 9.5,
     "est_tps": [
      20,
      28
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q2_K_XL",
     "size_gb": 9.8,
     "est_tps": [
      20,
      27
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q2_K_XL",
     "size_gb": 10.5,
     "est_tps": [
      69,
      121
     ]
    },
    {
     "model": "Qwen3.5-9B",
     "quant": "Q8_0",
     "size_gb": 9.5,
     "est_tps": [
      20,
      28
     ]
    }
   ]
  },
  "rtx-4060-ti": {
   "general": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      11,
      15
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q3_K_XL",
     "size_gb": 13.8,
     "est_tps": [
      61,
      106
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-IQ4_XS",
     "size_gb": 13.6,
     "est_tps": [
      47,
      82
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      11,
      15
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q3_K_XL",
     "size_gb": 13.8,
     "est_tps": [
      61,
      106
     ]
    },
    {
     "model": "gpt-oss-20b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 13.2,
     "est_tps": [
      43,
      75
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      11,
      15
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-IQ4_XS",
     "size_gb": 13.6,
     "est_tps": [
      47,
      82
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q2_K_XL",
     "size_gb": 12.3,
     "est_tps": [
      78,
      137
     ]
    }
   ]
  },
  "rtx-4070-ti-super": {
   "general": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      25,
      35
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q3_K_XL",
     "size_gb": 13.8,
     "est_tps": [
      110,
      193
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-IQ4_XS",
     "size_gb": 13.6,
     "est_tps": [
      90,
      157
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      25,
      35
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q3_K_XL",
     "size_gb": 13.8,
     "est_tps": [
      110,
      193
     ]
    },
    {
     "model": "gpt-oss-20b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 13.2,
     "est_tps": [
      84,
      146
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      25,
      35
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-IQ4_XS",
     "size_gb": 13.6,
     "est_tps": [
      90,
      157
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q2_K_XL",
     "size_gb": 12.3,
     "est_tps": [
      134,
      235
     ]
    }
   ]
  },
  "rtx-4080-super": {
   "general": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      28,
      38
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q3_K_XL",
     "size_gb": 13.8,
     "est_tps": [
      117,
      204
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-IQ4_XS",
     "size_gb": 13.6,
     "est_tps": [
      96,
      167
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      28,
      38
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q3_K_XL",
     "size_gb": 13.8,
     "est_tps": [
      117,
      204
     ]
    },
    {
     "model": "gpt-oss-20b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 13.2,
     "est_tps": [
      89,
      156
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      28,
      38
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-IQ4_XS",
     "size_gb": 13.6,
     "est_tps": [
      96,
      167
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q2_K_XL",
     "size_gb": 12.3,
     "est_tps": [
      140,
      246
     ]
    }
   ]
  },
  "rtx-4080": {
   "general": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      27,
      37
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q3_K_XL",
     "size_gb": 13.8,
     "est_tps": [
      115,
      201
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-IQ4_XS",
     "size_gb": 13.6,
     "est_tps": [
      94,
      165
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      27,
      37
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q3_K_XL",
     "size_gb": 13.8,
     "est_tps": [
      115,
      201
     ]
    },
    {
     "model": "gpt-oss-20b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 13.2,
     "est_tps": [
      88,
      153
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      27,
      37
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-IQ4_XS",
     "size_gb": 13.6,
     "est_tps": [
      94,
      165
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q2_K_XL",
     "size_gb": 12.3,
     "est_tps": [
      139,
      243
     ]
    }
   ]
  },
  "rtx-5060-ti": {
   "general": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      17,
      23
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q3_K_XL",
     "size_gb": 13.8,
     "est_tps": [
      85,
      148
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-IQ4_XS",
     "size_gb": 13.6,
     "est_tps": [
      67,
      117
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      17,
      23
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q3_K_XL",
     "size_gb": 13.8,
     "est_tps": [
      85,
      148
     ]
    },
    {
     "model": "gpt-oss-20b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 13.2,
     "est_tps": [
      62,
      108
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      17,
      23
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-IQ4_XS",
     "size_gb": 13.6,
     "est_tps": [
      67,
      117
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q2_K_XL",
     "size_gb": 12.3,
     "est_tps": [
      106,
      185
     ]
    }
   ]
  },
  "rtx-5070-ti": {
   "general": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      33,
      46
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q3_K_XL",
     "size_gb": 13.8,
     "est_tps": [
      130,
      228
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-IQ4_XS",
     "size_gb": 13.6,
     "est_tps": [
      109,
      190
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      33,
      46
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q3_K_XL",
     "size_gb": 13.8,
     "est_tps": [
      130,
      228
     ]
    },
    {
     "model": "gpt-oss-20b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 13.2,
     "est_tps": [
      102,
      178
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      33,
      46
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-IQ4_XS",
     "size_gb": 13.6,
     "est_tps": [
      109,
      190
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q2_K_XL",
     "size_gb": 12.3,
     "est_tps": [
      155,
      270
     ]
    }
   ]
  },
  "rtx-5080": {
   "general": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      36,
      49
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q3_K_XL",
     "size_gb": 13.8,
     "est_tps": [
      135,
      237
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-IQ4_XS",
     "size_gb": 13.6,
     "est_tps": [
      113,
      198
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      36,
      49
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q3_K_XL",
     "size_gb": 13.8,
     "est_tps": [
      135,
      237
     ]
    },
    {
     "model": "gpt-oss-20b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 13.2,
     "est_tps": [
      106,
      186
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-IQ4_XS",
     "size_gb": 14.3,
     "est_tps": [
      36,
      49
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-IQ4_XS",
     "size_gb": 13.6,
     "est_tps": [
      113,
      198
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q2_K_XL",
     "size_gb": 12.3,
     "est_tps": [
      159,
      279
     ]
    }
   ]
  },
  "rtx-6000-ada": {
   "general": [
    {
     "model": "Qwen3-Coder-Next",
     "quant": "IQ4_XS",
     "size_gb": 42.7,
     "est_tps": [
      130,
      228
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      82,
      144
     ]
    },
    {
     "model": "Nemotron-3-Nano-30B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 40.4,
     "est_tps": [
      66,
      115
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3-Coder-Next",
     "quant": "IQ4_XS",
     "size_gb": 42.7,
     "est_tps": [
      130,
      228
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      82,
      144
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q8_K_XL",
     "size_gb": 36.0,
     "est_tps": [
      73,
      128
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      82,
      144
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 31.5,
     "est_tps": [
      17,
      23
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q8_K_XL",
     "size_gb": 27.6,
     "est_tps": [
      70,
      122
     ]
    }
   ]
  },
  "rtx-a6000": {
   "general": [
    {
     "model": "Qwen3-Coder-Next",
     "quant": "IQ4_XS",
     "size_gb": 42.7,
     "est_tps": [
      114,
      200
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      70,
      122
     ]
    },
    {
     "model": "Nemotron-3-Nano-30B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 40.4,
     "est_tps": [
      55,
      97
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3-Coder-Next",
     "quant": "IQ4_XS",
     "size_gb": 42.7,
     "est_tps": [
      114,
      200
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      70,
      122
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q8_K_XL",
     "size_gb": 36.0,
     "est_tps": [
      62,
      108
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      70,
      122
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 31.5,
     "est_tps": [
      13,
      18
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q8_K_XL",
     "size_gb": 27.6,
     "est_tps": [
      59,
      103
     ]
    }
   ]
  },
  "l40s": {
   "general": [
    {
     "model": "Qwen3-Coder-Next",
     "quant": "IQ4_XS",
     "size_gb": 42.7,
     "est_tps": [
      123,
      215
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      76,
      133
     ]
    },
    {
     "model": "Nemotron-3-Nano-30B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 40.4,
     "est_tps": [
      61,
      106
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3-Coder-Next",
     "quant": "IQ4_XS",
     "size_gb": 42.7,
     "est_tps": [
      123,
      215
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      76,
      133
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q8_K_XL",
     "size_gb": 36.0,
     "est_tps": [
      68,
      119
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      76,
      133
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 31.5,
     "est_tps": [
      15,
      20
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q8_K_XL",
     "size_gb": 27.6,
     "est_tps": [
      64,
      113
     ]
    }
   ]
  },
  "a100-80gb": {
   "general": [
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      140,
      245
     ]
    },
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q6_K_XL",
     "size_gb": 73.1,
     "est_tps": [
      142,
      248
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      128,
      224
     ]
    }
   ],
   "coding": [
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      140,
      245
     ]
    },
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q6_K_XL",
     "size_gb": 73.1,
     "est_tps": [
      142,
      248
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      128,
      224
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      128,
      224
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 31.5,
     "est_tps": [
      33,
      45
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q8_K_XL",
     "size_gb": 27.6,
     "est_tps": [
      113,
      197
     ]
    }
   ]
  },
  "m4-pro": {
   "general": [
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q4_K_XL",
     "size_gb": 49.6,
     "est_tps": [
      49,
      85
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      29,
      52
     ]
    },
    {
     "model": "Nemotron-3-Nano-30B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 40.4,
     "est_tps": [
      22,
      39
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q4_K_XL",
     "size_gb": 49.6,
     "est_tps": [
      49,
      85
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      29,
      52
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q8_K_XL",
     "size_gb": 36.0,
     "est_tps": [
      26,
      45
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      29,
      52
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 31.5,
     "est_tps": [
      4.8,
      6.5
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q8_K_XL",
     "size_gb": 27.6,
     "est_tps": [
      24,
      42
     ]
    }
   ]
  },
  "m3-max": {
   "general": [
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-IQ4_XS",
     "size_gb": 93.7,
     "est_tps": [
      33,
      58
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      37,
      65
     ]
    },
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q8_K_XL",
     "size_gb": 86.3,
     "est_tps": [
      33,
      57
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-IQ4_XS",
     "size_gb": 93.7,
     "est_tps": [
      33,
      58
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      37,
      65
     ]
    },
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q8_K_XL",
     "size_gb": 86.3,
     "est_tps": [
      33,
      57
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-IQ4_XS",
     "size_gb": 93.7,
     "est_tps": [
      33,
      58
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      32,
      56
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 31.5,
     "est_tps": [
      5.2,
      7.1
     ]
    }
   ]
  },
  "m2-max": {
   "general": [
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      47,
      83
     ]
    },
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q6_K_XL",
     "size_gb": 73.1,
     "est_tps": [
      48,
      85
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      41,
      72
     ]
    }
   ],
   "coding": [
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      47,
      83
     ]
    },
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q6_K_XL",
     "size_gb": 73.1,
     "est_tps": [
      48,
      85
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      41,
      72
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      41,
      72
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 31.5,
     "est_tps": [
      7.0,
      9.5
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q8_K_XL",
     "size_gb": 27.6,
     "est_tps": [
      34,
      59
     ]
    }
   ]
  },
  "m1-max": {
   "general": [
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q4_K_XL",
     "size_gb": 49.6,
     "est_tps": [
      66,
      115
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      41,
      72
     ]
    },
    {
     "model": "Nemotron-3-Nano-30B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 40.4,
     "est_tps": [
      32,
      56
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q4_K_XL",
     "size_gb": 49.6,
     "est_tps": [
      66,
      115
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      41,
      72
     ]
    },
    {
     "model": "Qwen3-Coder-30B-A3B-Instruct",
     "quant": "UD-Q8_K_XL",
     "size_gb": 36.0,
     "est_tps": [
      36,
      63
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      41,
      72
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 31.5,
     "est_tps": [
      7.0,
      9.5
     ]
    },
    {
     "model": "Gemma-4-26B-A4B-it",
     "quant": "UD-Q8_K_XL",
     "size_gb": 27.6,
     "est_tps": [
      34,
      59
     ]
    }
   ]
  },
  "m2-ultra": {
   "general": [
    {
     "model": "GLM-5.3-Flash",
     "quant": "UD-Q3_K_XL",
     "size_gb": 147.5,
     "est_tps": [
      34,
      60
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-Q4_K_XL",
     "size_gb": 111.3,
     "est_tps": [
      65,
      114
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      81,
      142
     ]
    }
   ],
   "coding": [
    {
     "model": "GLM-5.3-Flash",
     "quant": "UD-Q3_K_XL",
     "size_gb": 147.5,
     "est_tps": [
      34,
      60
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-Q4_K_XL",
     "size_gb": 111.3,
     "est_tps": [
      65,
      114
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      81,
      142
     ]
    }
   ],
   "vision": [
    {
     "model": "GLM-5.3-Flash",
     "quant": "UD-Q3_K_XL",
     "size_gb": 147.5,
     "est_tps": [
      34,
      60
     ]
    },
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-Q4_K_XL",
     "size_gb": 111.3,
     "est_tps": [
      65,
      114
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      72,
      126
     ]
    }
   ]
  },
  "m1-ultra": {
   "general": [
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-IQ4_XS",
     "size_gb": 93.7,
     "est_tps": [
      74,
      130
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      81,
      142
     ]
    },
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q8_K_XL",
     "size_gb": 86.3,
     "est_tps": [
      73,
      128
     ]
    }
   ],
   "coding": [
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-IQ4_XS",
     "size_gb": 93.7,
     "est_tps": [
      74,
      130
     ]
    },
    {
     "model": "gpt-oss-120b",
     "quant": "UD-Q8_K_XL",
     "size_gb": 64.5,
     "est_tps": [
      81,
      142
     ]
    },
    {
     "model": "Qwen3-Coder-Next",
     "quant": "UD-Q8_K_XL",
     "size_gb": 86.3,
     "est_tps": [
      73,
      128
     ]
    }
   ],
   "vision": [
    {
     "model": "Qwen3.8-Flash-Next",
     "quant": "UD-IQ4_XS",
     "size_gb": 93.7,
     "est_tps": [
      74,
      130
     ]
    },
    {
     "model": "Qwen3.6-35B-A3B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 38.5,
     "est_tps": [
      72,
      126
     ]
    },
    {
     "model": "Qwen3.8-27B",
     "quant": "UD-Q8_K_XL",
     "size_gb": 31.5,
     "est_tps": [
      14,
      19
     ]
    }
   ]
  }
 },
 "unknown_hardware": [
  "Apple M-series Max (MacBook Pro / Mac Studio Max)",
  "Apple Mac Studio (M-series Ultra)"
 ],
 "prices": {
  "generated": "2026-10-09T04:33+08:00",
  "currency": "SGD",
  "gst_rate": 0.09,
  "fx": {
   "usd_sgd": 1.282,
   "date": "2026-10-08",
   "source": "https://api.frankfurter.dev/v1/latest?base=USD&symbols=SGD"
  },
  "items": [
   {
    "id": "strix-halo-128",
    "device": "strix-halo-395",
    "label": "AMD Ryzen AI Max+ 395 mini PC, 128 GB",
    "class": "small",
    "from_sgd": 4324.74,
    "to_sgd": 5093.66,
    "sources_ok": 0,
    "sources": [
     {
      "seller": "GMKtec (EVO-X2)",
      "trust": "Official store",
      "url": "https://www.gmktec.com/products/amd-ryzen%e2%84%a2-ai-max-395-evo-x2-ai-mini-pc",
      "ships_from": "overseas",
      "checked": "2026-10-09T04:33+08:00",
      "ok": false,
      "error": "HTTPError: HTTP Error 429: Too Many Requests"
     },
     {
      "seller": "Bosgame (M5)",
      "trust": "Official store",
      "url": "https://www.bosgame.com/products/bosgame-m5-ai-mini-desktop-ryzen-ai-max-395-96gb-128gb-2tb",
      "ships_from": "overseas",
      "checked": "2026-10-09T04:33+08:00",
      "ok": false,
      "error": "HTTPError: HTTP Error 429: Too Many Requests"
     },
     {
      "seller": "ALLSTARS (Singapore)",
      "trust": "Local distributor, 3yr warranty",
      "url": "https://allstars.com.sg/shop/pc/pc-complete/gmktec-evo-x2-ai-mini-pc-amd-ryzen-ai-max-395-128gb-lpddr5-2tb-gen4-ssd-2-5g-lan-wifi7-licensed-windows-11-pro/",
      "ships_from": "sg",
      "checked": "2026-10-09T04:33+08:00",
      "ok": false,
      "out_of_stock": true,
      "listed": 5599.0,
      "currency": "SGD",
      "error": "ValueError: out of stock (page still lists SGD 5,599)"
     },
     {
      "seller": "Framework (Desktop)",
      "trust": "Official store, configure to order",
      "url": "https://frame.work/sg/en/desktop",
      "ships_from": "overseas",
      "checked": "2026-10-09T04:33+08:00",
      "ok": false,
      "link_only": true
     }
    ],
    "stale": true,
    "last_read": "2026-10-06T04:34+08:00"
   },
   {
    "id": "gb10-128",
    "device": "dgx-spark",
    "label": "NVIDIA GB10 box, 128 GB (DGX Spark or ASUS Ascent GX10)",
    "class": "small",
    "from_sgd": 6624.0,
    "to_sgd": 8096.0,
    "sources_ok": 0,
    "sources": [
     {
      "seller": "SourceIT (Singapore), ASUS Ascent GX10 1 TB",
      "trust": "Authorised reseller, local warranty, GeBIZ",
      "url": "https://sourceit.com.sg/products/asus-ascent-gx10-compact-desktop-ai-supercomputer-1tb-gx10-gg0007bn",
      "ships_from": "sg",
      "checked": "2026-10-09T04:33+08:00",
      "ok": false,
      "error": "HTTPError: HTTP Error 429: Too Many Requests"
     },
     {
      "seller": "SourceIT (Singapore), DGX Spark 4 TB",
      "trust": "Authorised reseller, local warranty, GeBIZ",
      "url": "https://sourceit.com.sg/products/nvidia-dgx-spark-ai-supercomputer-4tb-940-54242-0007-000",
      "ships_from": "sg",
      "checked": "2026-10-09T04:33+08:00",
      "ok": false,
      "error": "HTTPError: HTTP Error 429: Too Many Requests"
     },
     {
      "seller": "Vii PC Trade (Singapore), DGX Spark 4 TB",
      "trust": "Local reseller, 1yr warranty",
      "url": "https://www.viipc.sg/products/nvidia-dgx-spark-ai-server-enterprise-gpu-computing-platform-4tb-128gb-ai-desktop",
      "ships_from": "sg",
      "checked": "2026-10-09T04:33+08:00",
      "ok": false,
      "error": "HTTPError: HTTP Error 429: Too Many Requests"
     }
    ],
    "stale": true,
    "last_read": "2026-10-07T04:36+08:00"
   },
   {
    "id": "mac-studio",
    "device": "m5-ultra",
    "label": "Apple Mac Studio, M5 Ultra (96 GB base)",
    "class": "small",
    "from_sgd": 7999.0,
    "to_sgd": 9949.0,
    "sources_ok": 1,
    "sources": [
     {
      "seller": "Apple (SG)",
      "trust": "Official store, configure to order",
      "url": "https://www.apple.com/sg/shop/buy-mac/mac-studio",
      "ships_from": "sg",
      "checked": "2026-10-09T04:33+08:00",
      "ok": true,
      "currency": "SGD",
      "prices": [
       7999.0,
       9949.0
      ],
      "prices_sgd": [
       7999.0,
       9949.0
      ]
     }
    ],
    "last_read": "2026-10-09T04:33+08:00"
   },
   {
    "id": "mac-studio-max",
    "device": "m5-max",
    "label": "Apple Mac Studio, M5 Max (base configs)",
    "class": "small",
    "from_sgd": 3499.0,
    "to_sgd": 4399.0,
    "sources_ok": 1,
    "sources": [
     {
      "seller": "Apple (SG)",
      "trust": "Official store, configure to order",
      "url": "https://www.apple.com/sg/shop/buy-mac/mac-studio",
      "ships_from": "sg",
      "checked": "2026-10-09T04:33+08:00",
      "ok": true,
      "currency": "SGD",
      "prices": [
       3499.0,
       4399.0
      ],
      "prices_sgd": [
       3499.0,
       4399.0
      ]
     }
    ],
    "last_read": "2026-10-09T04:33+08:00"
   },
   {
    "id": "macbook-pro-max",
    "device": "m5-max",
    "label": "Apple MacBook Pro 14 or 16 inch, M5 Max (base configs)",
    "class": "small",
    "from_sgd": 5799.0,
    "to_sgd": 6999.0,
    "sources_ok": 1,
    "sources": [
     {
      "seller": "Apple (SG)",
      "trust": "Official store, configure to order",
      "url": "https://www.apple.com/sg/shop/buy-mac/macbook-pro",
      "ships_from": "sg",
      "checked": "2026-10-09T04:33+08:00",
      "ok": true,
      "currency": "SGD",
      "prices": [
       5799.0,
       5999.0,
       6699.0,
       6999.0
      ],
      "prices_sgd": [
       5799.0,
       5999.0,
       6699.0,
       6999.0
      ]
     }
    ],
    "last_read": "2026-10-09T04:33+08:00"
   },
   {
    "id": "rtx-pro-6000",
    "device": "rtx-pro-6000",
    "label": "NVIDIA RTX PRO 6000 Blackwell 96 GB (card only)",
    "class": "ws",
    "from_sgd": 28611.08,
    "to_sgd": 28611.08,
    "sources_ok": 0,
    "sources": [
     {
      "seller": "SourceIT (Singapore)",
      "trust": "Authorised reseller, local warranty, GeBIZ",
      "url": "https://sourceit.com.sg/products/nvidia-rtx-pro\u2122-6000-blackwell-workstation-edition-900-5g144-2500-000",
      "ships_from": "sg",
      "checked": "2026-10-09T04:33+08:00",
      "ok": false,
      "error": "HTTPError: HTTP Error 429: Too Many Requests"
     }
    ],
    "stale": true,
    "last_read": "2026-10-06T04:34+08:00"
   }
  ],
  "unavailable": [
   "strix-halo-128 / ALLSTARS (Singapore): ValueError: out of stock (page still lists SGD 5,599)"
  ],
  "errors": [
   "strix-halo-128 / GMKtec (EVO-X2): HTTPError: HTTP Error 429: Too Many Requests",
   "strix-halo-128 / Bosgame (M5): HTTPError: HTTP Error 429: Too Many Requests",
   "gb10-128 / SourceIT (Singapore), ASUS Ascent GX10 1 TB: HTTPError: HTTP Error 429: Too Many Requests",
   "gb10-128 / SourceIT (Singapore), DGX Spark 4 TB: HTTPError: HTTP Error 429: Too Many Requests",
   "gb10-128 / Vii PC Trade (Singapore), DGX Spark 4 TB: HTTPError: HTTP Error 429: Too Many Requests",
   "rtx-pro-6000 / SourceIT (Singapore): HTTPError: HTTP Error 429: Too Many Requests"
  ]
 },
 "errors": [
  "github Blaizzy/mlx-vlm: Remote end closed connection without response",
  "price strix-halo-128 / GMKtec (EVO-X2): HTTPError: HTTP Error 429: Too Many Requests",
  "price strix-halo-128 / Bosgame (M5): HTTPError: HTTP Error 429: Too Many Requests",
  "price gb10-128 / SourceIT (Singapore), ASUS Ascent GX10 1 TB: HTTPError: HTTP Error 429: Too Many Requests",
  "price gb10-128 / SourceIT (Singapore), DGX Spark 4 TB: HTTPError: HTTP Error 429: Too Many Requests",
  "price gb10-128 / Vii PC Trade (Singapore), DGX Spark 4 TB: HTTPError: HTTP Error 429: Too Many Requests",
  "price rtx-pro-6000 / SourceIT (Singapore): HTTPError: HTTP Error 429: Too Many Requests"
 ]
}