{"generated":"2026-09-10T06:44:15.420Z","community":{"owners_est":150000,"owners_low":93373,"owners_high":233433,"per_platform":{"strix-halo":{"anchor_stars":1915,"low":38300,"high":95750},"dgx-spark":{"anchor_stars":452,"low":9040,"high":22600},"mac":{"anchor_stars":6905,"low":46033,"high":115083}},"signals":{"sources":26,"stars_total":3425,"downloads_total":26789,"publishers":23},"assumptions":{"star_rate":{"low":0.02,"high":0.05},"big_memory_mac_share":0.3333333333333333},"method":"Owners per platform = stars on the anchor setup repo / assumed star rate (2-5% of users star an essential niche tool); Mac scaled by the assumed share of MLX users on a 96 GB+ machine. Signals are measured from tracked sources; ratios are assumptions and are published as such.","updated":"2026-09-10T06:44:15.421Z"},"counts":{"configs":148,"models":69,"platforms":4,"releases":101},"platforms":["strix-halo","dgx-spark","mac-ultra","mac-max"],"featured":[{"model":"GLM-5.3-Flash","vendor":"Zhipu AI","quant":"MLX-4BIT","format":null,"hardware":"mac-ultra","hardwareLabel":"Apple Mac Studio (M-series Ultra)","oem":null,"oemLabel":null,"backend":"mlx","variant":null,"ctx":null,"decode_tps":34.2,"prefill_tps":null,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.09,"credibilityLabel":"low","author":"drowzeys","url":"https://github.com/drowzeys/keys-Mac-oMLX-0.6.3.2RC-Dual-ANE-GLM-5.3-Flash-Abliterated-oQ4","doc_url":"https://github.com/drowzeys/keys-Mac-oMLX-0.6.3.2RC-Dual-ANE-GLM-5.3-Flash-Abliterated-oQ4/blob/HEAD/README.md","signals":{"stars":1,"forks":0,"watchers":1,"open_issues":0}},{"model":"DeepSeek V4 Flash 284B","vendor":"DeepSeek","quant":"UD-IQ2_XXS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","backend":"llama.cpp","variant":"vulkan","ctx":512,"decode_tps":13.3,"prefill_tps":155.6,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","signals":{"stars":323,"forks":22}},{"model":"Llama 4 Scout 109B","vendor":"Meta","quant":"Q4_K_M","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","backend":"llama.cpp","variant":"vulkan","ctx":512,"decode_tps":18.3,"prefill_tps":331,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","signals":{"stars":323,"forks":22}},{"model":"Gemma 4 31B IT QAT","vendor":"Google","quant":"Q4_0","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","backend":"llama.cpp","variant":"vulkan","ctx":512,"decode_tps":11.4,"prefill_tps":308.3,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","signals":{"stars":323,"forks":22}},{"model":"Qwen3.8-27B","vendor":"Alibaba","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"bosgame-m5","oemLabel":"Bosgame M5","backend":"llama.cpp","variant":"vulkan","ctx":131072,"decode_tps":24.6,"prefill_tps":null,"throughput_kind":"single-stream","trust":"zach-verified","trustLabel":"Verified (ran it)","credibility":0.58,"credibilityLabel":"medium","author":"sypherin","url":"https://github.com/sypherin/strix-halo-setup","doc_url":"https://github.com/sypherin/strix-halo-setup/blob/HEAD/README.md","signals":{"stars":72,"forks":4}},{"model":"Step-3.7-Flash","vendor":"StepFun","quant":"IQ4_XS","format":"GGUF","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"backend":"llama.cpp","variant":null,"ctx":null,"decode_tps":19.9,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/step-3.7-flash-llamacpp.json","signals":{"stars":4,"forks":0}},{"model":"Qwopus3.6-27B-Coder","vendor":"Other","quant":"Q8_0","format":"GGUF","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"backend":"llama.cpp","variant":null,"ctx":null,"decode_tps":7.6,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/qwopus3.6-27b-coder-q8-llamacpp.json","signals":{"stars":4,"forks":0}},{"model":"Mistral-Small-3.1-24B-Instruct-2503","vendor":"Mistral AI","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"backend":"llama.cpp","variant":"rocm","ctx":null,"decode_tps":14.7,"prefill_tps":368.5,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.75,"credibilityLabel":"high","author":"lhl","url":"https://github.com/lhl/strix-halo-testing","doc_url":"https://github.com/lhl/strix-halo-testing/blob/HEAD/llm-bench/Mistral-Small-3.1-24B-Instruct-2503-UD-Q4_K_XL/results.jsonl","signals":{"stars":252,"forks":18}}],"configs":[{"model":"Qwen3.6-35B-A3B-MTP","quant":"UD-Q4_K_XL","format":null,"hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":null,"variant":"vulkan","ctx":262144,"mode":"mtp-3","concurrency":1,"num_gpus":1,"decode_tps":66,"prefill_tps":null,"throughput_kind":"single-stream","trust":"zach-verified","trustLabel":"Verified (ran it)","credibility":0.58,"credibilityLabel":"medium","author":"sypherin","url":"https://github.com/sypherin/strix-halo-setup","doc_url":"https://github.com/sypherin/strix-halo-setup/blob/HEAD/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"stars":73,"forks":4,"watchers":1,"open_issues":0}},{"model":"Qwen3.6-35B-A3B","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"bosgame-m5","oemLabel":"Bosgame M5","vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"vulkan","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":41,"prefill_tps":null,"throughput_kind":"single-stream","trust":"zach-verified","trustLabel":"Verified (ran it)","credibility":0.58,"credibilityLabel":"medium","author":"sypherin","url":"https://github.com/sypherin/strix-halo-setup","doc_url":"https://github.com/sypherin/strix-halo-setup/blob/HEAD/README.md","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":72,"forks":4}},{"model":"Qwen3.6-35B-A3B","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"bosgame-m5","oemLabel":"Bosgame M5","vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":39,"prefill_tps":null,"throughput_kind":"single-stream","trust":"zach-verified","trustLabel":"Verified (ran it)","credibility":0.58,"credibilityLabel":"medium","author":"sypherin","url":"https://github.com/sypherin/strix-halo-setup","doc_url":"https://github.com/sypherin/strix-halo-setup/blob/HEAD/README.md","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":72,"forks":4}},{"model":"Qwen3.8-27B","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"bosgame-m5","oemLabel":"Bosgame M5","vendor":"Alibaba","version":3.8,"backend":"llama.cpp","variant":"vulkan","ctx":131072,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":24.6,"prefill_tps":null,"throughput_kind":"single-stream","trust":"zach-verified","trustLabel":"Verified (ran it)","credibility":0.58,"credibilityLabel":"medium","author":"sypherin","url":"https://github.com/sypherin/strix-halo-setup","doc_url":"https://github.com/sypherin/strix-halo-setup/blob/HEAD/README.md","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":72,"forks":4}},{"model":"Qwen3 0.6B","quant":"Q8_0","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":266,"prefill_tps":13112,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3 0.6B","quant":"Q8_0","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3,"backend":"llama.cpp","variant":"rocm","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":208.7,"prefill_tps":4666.1,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"LFM2.5 8B-A1B","quant":"Q4_K_M","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Other","version":2.5,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":176.5,"prefill_tps":3398.4,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3-30B-A3B-Instruct-2507","quant":"IQ4_XS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":103.2,"prefill_tps":1438.1,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3-Coder 30B-A3B","quant":"Q4_K_S","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":98,"prefill_tps":1406.5,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3-Coder 30B-A3B","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":97.1,"prefill_tps":1400,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3.6-35B-A3B-NVFP4-Fast (Unsloth)","quant":"NVFP4","format":"NVFP4","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"vllm","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":92.6,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/qwen3.6-35b-nvfp4-unsloth-fast.json","last_updated":"2026-09-01","first_seen":"2026-08-28","signals":{"stars":4,"forks":0}},{"model":"Qwen3.6-35B-A3B-NVFP4-Fast (Unsloth)","quant":"NVFP4","format":"NVFP4","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"vllm","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":92.6,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/qwen3.6-35b-nvfp4-unsloth-fast.json","last_updated":"2026-09-06","first_seen":"2026-09-06","signals":{"stars":4,"forks":0}},{"model":"GPT-OSS-20B","quant":"Q4_K_XL","format":"GGUF","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"OpenAI","version":0,"backend":"llama.cpp","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":90.5,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/gpt-oss-20b-llamacpp.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":4,"forks":0}},{"model":"Qwen3-Coder 30B-A3B","quant":"IQ4_XS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":90.4,"prefill_tps":1372.3,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3 30B-A3B NEO-MAX","quant":"IQ4_XS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":87.4,"prefill_tps":1396.1,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3.6 35B-A3B","quant":"Q4_0","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":81.3,"prefill_tps":1243.5,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3-30B-A3B","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":79.1,"prefill_tps":687.9,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.75,"credibilityLabel":"high","author":"lhl","url":"https://github.com/lhl/strix-halo-testing","doc_url":"https://github.com/lhl/strix-halo-testing/blob/HEAD/llm-bench/Qwen3-30B-A3B-UD-Q4_K_XL/results.jsonl","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":252,"forks":18}},{"model":"Nemotron Cascade 2 30B-A3B","quant":"IQ4_XS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"NVIDIA","version":2,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":79,"prefill_tps":1325.3,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Nemotron 3 Nano 30B-A3B","quant":"IQ4_XS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"NVIDIA","version":3,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":76,"prefill_tps":1312.5,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3.5 35B-A3B","quant":"IQ4_XS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3.5,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":75.2,"prefill_tps":1170.3,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Gemma 4 26B-A4B IT QAT","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Google","version":4,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":74.8,"prefill_tps":1432,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3.6-35B-A3B","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"vulkan","ctx":null,"mode":"mtp-3","concurrency":null,"num_gpus":null,"decode_tps":74.6,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/benchmark/results-mtp/summary.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":1915,"forks":192}},{"model":"Qwen3-Coder 30B-A3B","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3,"backend":"llama.cpp","variant":"rocm","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":73.7,"prefill_tps":1285.3,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3.6-35B-A3B","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"vulkan","ctx":null,"mode":"mtp-2","concurrency":null,"num_gpus":null,"decode_tps":72.8,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/benchmark/results-mtp/summary.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":1915,"forks":192}},{"model":"Qwen3.6-35B-A3B","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":"mtp-3","concurrency":null,"num_gpus":null,"decode_tps":68.3,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/benchmark/results-mtp/summary.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":1915,"forks":192}},{"model":"Qwen AgentWorld 35B-A3B","quant":"IQ4_XS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":0,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":65.7,"prefill_tps":1182.8,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3.5 35B-A3B","quant":"Q4_K_M","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3.5,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":64.9,"prefill_tps":1080,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3.6-35B-A3B","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":"mtp-2","concurrency":null,"num_gpus":null,"decode_tps":64.5,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/benchmark/results-mtp/summary.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":1915,"forks":192}},{"model":"Qwen3.6 35B-A3B","quant":"Q4_K_M","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":63.8,"prefill_tps":1064,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3.6-35B-A3B","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"vulkan-radv","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":62.8,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"Qwen3.6-35B-A3B","quant":"Q4_K_XL","format":"GGUF","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":62.7,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/qwen3.6-35b-q4-llamacpp.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":4,"forks":0}},{"model":"Qwen3.6-35B-A3B","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"vulkan-radv-performance","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":62.2,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"Qwen3-Coder-Next 80B-A3B","quant":"IQ4_XS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":61.9,"prefill_tps":739,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3.6 35B-A3B","quant":"Q4_K_M","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3.6,"backend":"ollama","variant":"vulkan","ctx":2048,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":60.6,"prefill_tps":793.1,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3.6-35B-A3B","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"vulkan","ctx":null,"mode":"baseline","concurrency":null,"num_gpus":null,"decode_tps":58.7,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/benchmark/results-mtp/summary.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":1915,"forks":192}},{"model":"Ornith-1.0-35B","quant":"Q8_0","format":"GGUF","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Other","version":1,"backend":"llama.cpp","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":56.1,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/ornith-35b-q8-llamacpp.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":4,"forks":0}},{"model":"Qwen3-Next 80B-A3B","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":54.9,"prefill_tps":657,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3.5 35B-A3B","quant":"Q4_K_M","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3.5,"backend":"llama.cpp","variant":"rocm","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":54.7,"prefill_tps":1047,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Gemma 4 26B-A4B IT","quant":"Q4_K_M","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Google","version":4,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":54.2,"prefill_tps":1323.4,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3.6-35B-A3B","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"rocm-7.14","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":53.5,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"llama-2-7b","quant":"Q4_0","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Meta","version":2,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":53,"prefill_tps":1327,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.75,"credibilityLabel":"high","author":"lhl","url":"https://github.com/lhl/strix-halo-testing","doc_url":"https://github.com/lhl/strix-halo-testing/blob/HEAD/llm-bench/llama-2-7b.Q4_0/results.jsonl","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":252,"forks":18}},{"model":"Qwen3.6-35B-A3B","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"rocm-7.14-pr26592","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":53,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-01","first_seen":"2026-08-29","signals":{"stars":1897,"forks":192}},{"model":"Qwen3.6 35B-A3B","quant":"Q4_K_M","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"rocm","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":52.7,"prefill_tps":1186.2,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3.6-35B-A3B","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"rocm-7.2.4","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":51.7,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/ryzen-ai-halo-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"llama-2-7b","quant":"Q4_K_M","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Meta","version":2,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":50.4,"prefill_tps":1083,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.75,"credibilityLabel":"high","author":"lhl","url":"https://github.com/lhl/strix-halo-testing","doc_url":"https://github.com/lhl/strix-halo-testing/blob/HEAD/llm-bench/llama-2-7b.Q4_K_M/results.jsonl","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":252,"forks":18}},{"model":"Qwen3-Coder-Next-80B-A3B","quant":"Q4_K_XL","format":"GGUF","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3,"backend":"llama.cpp","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":50.1,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/qwen3-coder-next-q4-llamacpp.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":4,"forks":0}},{"model":"Qwen3-Next 80B-A3B","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3,"backend":"llama.cpp","variant":"rocm","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":49.6,"prefill_tps":800.4,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3.6-35B-A3B","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"vulkan-radv","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":49.2,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"Qwen3.6-35B-A3B","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"vulkan-radv-performance","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":48.9,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"Qwen3.6-35B-A3B","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":"baseline","concurrency":null,"num_gpus":null,"decode_tps":48.7,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/benchmark/results-mtp/summary.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":1915,"forks":192}},{"model":"Gemma 4 26B-A4B","quant":"Q4_K_M","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Google","version":4,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":48.5,"prefill_tps":1142,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3.6-35B-A3B","quant":null,"format":"GGUF","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":48.2,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/qwen3.6-35b-q8-llamacpp.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":4,"forks":0}},{"model":"Qwen3.6-35B-A3B","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"rocm-7.14","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":48.1,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"Nemotron-3-Nano-30B-A3B","quant":null,"format":"GGUF","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"NVIDIA","version":3,"backend":"llama.cpp","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":47.9,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/nemotron3-nano-30b-q8-llamacpp.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":4,"forks":0}},{"model":"Qwen3.6-35B-A3B","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"rocm-7.14-pr26592","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":47.9,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-01","first_seen":"2026-08-29","signals":{"stars":1897,"forks":192}},{"model":"gpt-oss-20b-F16","quant":null,"format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"OpenAI","version":0,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":47.8,"prefill_tps":1213.3,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.75,"credibilityLabel":"high","author":"lhl","url":"https://github.com/lhl/strix-halo-testing","doc_url":"https://github.com/lhl/strix-halo-testing/blob/HEAD/llm-bench/gpt-oss-20b-F16/results.jsonl","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":252,"forks":18}},{"model":"Kimi-Linear-48B-A3B","quant":"Q8_0","format":"GGUF","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Moonshot","version":0,"backend":"llama.cpp","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":47.2,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/kimi-linear-48b-q8-llamacpp.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":4,"forks":0}},{"model":"Qwen3.6-35B-A3B","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"rocm-7.2.4","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":46.5,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/ryzen-ai-halo-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"shisa-v2-llama3.1-8b.i1","quant":"Q4_K_M","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Meta","version":2,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":43.5,"prefill_tps":878.2,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.75,"credibilityLabel":"high","author":"lhl","url":"https://github.com/lhl/strix-halo-testing","doc_url":"https://github.com/lhl/strix-halo-testing/blob/HEAD/llm-bench/shisa-v2-llama3.1-8b.i1-Q4_K_M/results.jsonl","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":252,"forks":18}},{"model":"gemma-4-26B-A4B-it","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Google","version":4,"backend":"llama.cpp","variant":"rocm-7.2.4","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":41.8,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/ryzen-ai-halo-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"Nemotron-3-Nano-30B-A3B","quant":"FP8","format":"FP8","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"NVIDIA","version":3,"backend":"vllm","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":40.7,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/nemotron3-nano-30b-fp8.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":4,"forks":0}},{"model":"gpt-oss-120b-F16","quant":null,"format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"OpenAI","version":0,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":33.9,"prefill_tps":534.1,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.75,"credibilityLabel":"high","author":"lhl","url":"https://github.com/lhl/strix-halo-testing","doc_url":"https://github.com/lhl/strix-halo-testing/blob/HEAD/llm-bench/gpt-oss-120b-F16/results.jsonl","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":252,"forks":18}},{"model":"MiniMax-M3 428B [DUAL 2xSpark]","quant":"GPTQ","format":"NVFP4","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"MiniMax","version":3,"backend":"vllm","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":30.2,"prefill_tps":null,"throughput_kind":"multi-gpu","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/minimax-m3-dual.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":4,"forks":0}},{"model":"Gemma 4 12B IT QAT","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Google","version":4,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":29.3,"prefill_tps":816.3,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Gemma-4-12B-IT","quant":"Q4_K_M","format":"GGUF","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Google","version":4,"backend":"llama.cpp","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":28.4,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/gemma-4-12b-it-llamacpp.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":4,"forks":0}},{"model":"Qwen3.8-Flash-Next","quant":"IQ4_XS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3.8,"backend":"llama.cpp","variant":"vulkan","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":27.2,"prefill_tps":394.7,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-09-06","signals":{"stars":323,"forks":22}},{"model":"Qwen-AgentWorld-35B-A3B","quant":"BF16","format":"BF16","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":0,"backend":"vllm","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":26.9,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/qwen-agentworld-35b-bf16.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":4,"forks":0}},{"model":"GLM-4.5-Air","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Zhipu AI","version":4.5,"backend":"llama.cpp","variant":"vulkan","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":23.4,"prefill_tps":179.9,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.75,"credibilityLabel":"high","author":"lhl","url":"https://github.com/lhl/strix-halo-testing","doc_url":"https://github.com/lhl/strix-halo-testing/blob/HEAD/llm-bench/GLM-4.5-Air-UD-Q4_K_XL/results.jsonl","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":252,"forks":18}},{"model":"Qwen3.6-27B-NVFP4 (Unsloth)","quant":"NVFP4","format":"NVFP4","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"vllm","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":23.3,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/qwen3.6-27b-nvfp4-unsloth.json","last_updated":"2026-09-01","first_seen":"2026-08-28","signals":{"stars":4,"forks":0}},{"model":"Qwen3.6-27B-NVFP4 (Unsloth)","quant":"NVFP4","format":"NVFP4","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"vllm","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":23.3,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/qwen3.6-27b-nvfp4-unsloth.json","last_updated":"2026-09-06","first_seen":"2026-09-06","signals":{"stars":4,"forks":0}},{"model":"dots.llm1.inst","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"RedNote","version":1,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":22.7,"prefill_tps":182,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.75,"credibilityLabel":"high","author":"lhl","url":"https://github.com/lhl/strix-halo-testing","doc_url":"https://github.com/lhl/strix-halo-testing/blob/HEAD/llm-bench/dots.llm1.inst-UD-Q4_K_XL/results.jsonl","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":252,"forks":18}},{"model":"Gemma-4-26B-A4B-IT","quant":"BF16","format":"BF16","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Google","version":4,"backend":"vllm","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":22.3,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/gemma-4-26b-a4b-bf16.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":4,"forks":0}},{"model":"Qwen3.5-122B-A10B","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.5,"backend":"llama.cpp","variant":"rocm-7.2.4","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":21.6,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/ryzen-ai-halo-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"Qwen3.8 27B","quant":"Q4_K_M","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3.8,"backend":"ollama","variant":"vulkan","ctx":4096,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":20.4,"prefill_tps":292.5,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Llama-4-Scout-17B-16E-Instruct","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Meta","version":4,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":20.2,"prefill_tps":306.4,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.75,"credibilityLabel":"high","author":"lhl","url":"https://github.com/lhl/strix-halo-testing","doc_url":"https://github.com/lhl/strix-halo-testing/blob/HEAD/llm-bench/Llama-4-Scout-17B-16E-Instruct-UD-Q4_K_XL/results.jsonl","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":252,"forks":18}},{"model":"Step-3.7-Flash","quant":"IQ4_XS","format":"GGUF","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"StepFun","version":3.7,"backend":"llama.cpp","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":19.9,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/step-3.7-flash-llamacpp.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":4,"forks":0}},{"model":"DeepSeek-V4-Flash-0731","quant":"UD-IQ3_XXS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":"llama.cpp","variant":"vulkan-radv-performance","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":19.8,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"Nex-N2-Pro-397B-A17B","quant":null,"format":"GGUF","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Other","version":2,"backend":"llama.cpp","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":18.9,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/nex-n2-pro-iq1m-llamacpp.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":4,"forks":0}},{"model":"Nemotron 3 Super 120B-A12B","quant":"IQ4_XS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"NVIDIA","version":3,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":18.9,"prefill_tps":297.1,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Hunyuan-A13B-Instruct","quant":"UD-Q6_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Tencent","version":0,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":18.4,"prefill_tps":297.5,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.75,"credibilityLabel":"high","author":"lhl","url":"https://github.com/lhl/strix-halo-testing","doc_url":"https://github.com/lhl/strix-halo-testing/blob/HEAD/llm-bench/Hunyuan-A13B-Instruct-UD-Q6_K_XL/results.jsonl","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":252,"forks":18}},{"model":"Llama 4 Scout 109B","quant":"Q4_K_M","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Meta","version":4,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":18.3,"prefill_tps":331,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"DeepSeek-V4-Flash-0731","quant":"UD-IQ2_XXS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":"llama.cpp","variant":"rocm-7.14","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":16.2,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"DeepSeek-V4-Flash-0731","quant":"UD-IQ2_XXS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":"llama.cpp","variant":"rocm-7.2.4","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":16.1,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"DeepSeek-V4-Flash-0731","quant":"UD-IQ3_XXS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":"llama.cpp","variant":"rocm-7.14","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":16,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"Qwen3-235B-A22B-Instruct-2507","quant":"UD-Q3_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3,"backend":"llama.cpp","variant":"vulkan","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":15.9,"prefill_tps":117.1,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.75,"credibilityLabel":"high","author":"lhl","url":"https://github.com/lhl/strix-halo-testing","doc_url":"https://github.com/lhl/strix-halo-testing/blob/HEAD/llm-bench/Qwen3-235B-A22B-Instruct-2507-UD-Q3_K_XL/results.jsonl","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":252,"forks":18}},{"model":"DeepSeek-V4-Flash-0731","quant":"UD-IQ3_XXS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":"llama.cpp","variant":"rocm-7.2.4","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":15.8,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"DeepSeek-V4-Flash-0731","quant":"UD-IQ2_XXS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":"llama.cpp","variant":"rocm-7.14-pr26592","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":14.8,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-01","first_seen":"2026-08-29","signals":{"stars":1897,"forks":192}},{"model":"Mistral-Small-3.1-24B-Instruct-2503","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Mistral AI","version":3.1,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":14.7,"prefill_tps":368.5,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.75,"credibilityLabel":"high","author":"lhl","url":"https://github.com/lhl/strix-halo-testing","doc_url":"https://github.com/lhl/strix-halo-testing/blob/HEAD/llm-bench/Mistral-Small-3.1-24B-Instruct-2503-UD-Q4_K_XL/results.jsonl","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":252,"forks":18}},{"model":"DeepSeek-V4-Flash-0731","quant":"UD-IQ3_XXS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":"llama.cpp","variant":"rocm-7.14-pr26592","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":14.5,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-01","first_seen":"2026-08-29","signals":{"stars":1897,"forks":192}},{"model":"Qwen3.5-397B-A17B [DUAL 2xSpark]","quant":"IQ4_NL","format":"GGUF","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.5,"backend":"llama.cpp","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":13.6,"prefill_tps":null,"throughput_kind":"multi-gpu","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/qwen3.5-397b-gguf-dual.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":4,"forks":0}},{"model":"Qwen3.6-27B","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":"mtp-3","concurrency":null,"num_gpus":null,"decode_tps":13.5,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/benchmark/results-mtp/summary.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":1915,"forks":192}},{"model":"Qwen3.6-27B","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"vulkan","ctx":null,"mode":"mtp-3","concurrency":null,"num_gpus":null,"decode_tps":13.3,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/benchmark/results-mtp/summary.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":1915,"forks":192}},{"model":"DeepSeek V4 Flash 284B","quant":"UD-IQ2_XXS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"DeepSeek","version":4,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":13.3,"prefill_tps":155.6,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3.6 27B MTP NVFP4 v3","quant":"NVFP4","format":"NVFP4","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":13.2,"prefill_tps":374,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.77,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-01","first_seen":"2026-08-30","signals":{"stars":314,"forks":22}},{"model":"Qwen3.6 27B MTP NVFP4 v3","quant":"NVFP4","format":"NVFP4","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":13.2,"prefill_tps":374,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-09-06","signals":{"stars":323,"forks":22}},{"model":"DeepSeek-V4-Flash-0731","quant":"UD-IQ2_XXS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":"llama.cpp","variant":"vulkan-radv-performance","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":13.1,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"Qwen3.6-27B","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":"mtp-2","concurrency":null,"num_gpus":null,"decode_tps":12.4,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/benchmark/results-mtp/summary.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":1915,"forks":192}},{"model":"gemma-3-27b-it","quant":"UD-Q4_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Google","version":3,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":12.1,"prefill_tps":302.2,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.75,"credibilityLabel":"high","author":"lhl","url":"https://github.com/lhl/strix-halo-testing","doc_url":"https://github.com/lhl/strix-halo-testing/blob/HEAD/llm-bench/gemma-3-27b-it-UD-Q4_K_XL/results.jsonl","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":252,"forks":18}},{"model":"Qwen3.6-27B","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"vulkan","ctx":null,"mode":"mtp-2","concurrency":null,"num_gpus":null,"decode_tps":11.7,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/benchmark/results-mtp/summary.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":1915,"forks":192}},{"model":"Gemma 4 31B IT QAT","quant":"Q4_0","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Google","version":4,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":11.4,"prefill_tps":308.3,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"DeepSeek-V4-Flash-0731","quant":"UD-IQ2_XXS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":"llama.cpp","variant":"vulkan-radv","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":9.1,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"DeepSeek-V4-Flash-0731","quant":"UD-IQ3_XXS","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":"llama.cpp","variant":"vulkan-radv","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":9,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"Qwen3.6-27B","quant":"Q8_0","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"rocm-7.2.4","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":7.8,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/ryzen-ai-halo-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"Qwen3.6 27B MTP","quant":"Q8_0","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"vulkan","ctx":512,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":7.7,"prefill_tps":341.9,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwopus3.6-27B-Coder","quant":"Q8_0","format":"GGUF","hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Other","version":3.6,"backend":"llama.cpp","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":null,"decode_tps":7.6,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.21,"credibilityLabel":"low","author":"jvr0x","url":"https://github.com/jvr0x/dgx-spark-bench","doc_url":"https://github.com/jvr0x/dgx-spark-bench/blob/HEAD/results/qwopus3.6-27b-coder-q8-llamacpp.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":4,"forks":0}},{"model":"Qwen3.6-27B","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"rocm-7.14","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":6.6,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"Qwen3.6-27B","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"rocm-7.14-pr26592","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":6.6,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-01","first_seen":"2026-08-29","signals":{"stars":1897,"forks":192}},{"model":"Qwen3.6-27B","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"rocm-7.2.4","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":6.6,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"Qwen3.6-27B","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"vulkan-radv","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":6.6,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"Qwen3.6-27B","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"vulkan-radv-performance","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":6.6,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/docs/toolbox-performance-results.json","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"stars":1915,"forks":192}},{"model":"Qwen3.6-27B","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":"baseline","concurrency":null,"num_gpus":null,"decode_tps":6.5,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/benchmark/results-mtp/summary.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":1915,"forks":192}},{"model":"Qwen3-32B","quant":"Q8_0","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":6.4,"prefill_tps":226.1,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.75,"credibilityLabel":"high","author":"lhl","url":"https://github.com/lhl/strix-halo-testing","doc_url":"https://github.com/lhl/strix-halo-testing/blob/HEAD/llm-bench/Qwen3-32B-Q8_0/results.jsonl","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":252,"forks":18}},{"model":"Qwen3.6-27B","quant":"UD-Q8_K_XL","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.6,"backend":"llama.cpp","variant":"vulkan","ctx":null,"mode":"baseline","concurrency":null,"num_gpus":null,"decode_tps":6.3,"prefill_tps":null,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":1,"credibilityLabel":"high","author":"kyuz0","url":"https://github.com/kyuz0/amd-strix-halo-toolboxes","doc_url":"https://github.com/kyuz0/amd-strix-halo-toolboxes/blob/HEAD/benchmark/results-mtp/summary.json","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":1915,"forks":192}},{"model":"shisa-v2-llama3.3-70b.i1","quant":"Q4_K_M","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Meta","version":2,"backend":"llama.cpp","variant":"rocm","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":5.1,"prefill_tps":94.7,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.75,"credibilityLabel":"high","author":"lhl","url":"https://github.com/lhl/strix-halo-testing","doc_url":"https://github.com/lhl/strix-halo-testing/blob/HEAD/llm-bench/shisa-v2-llama3.3-70b.i1-Q4_K_M/results.jsonl","last_updated":"2026-09-06","first_seen":"2026-08-28","signals":{"stars":252,"forks":18}},{"model":"Llama 3.1 70B","quant":"Q4_K_M","format":"GGUF","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":"beelink-gtr9","oemLabel":"Beelink GTR9","vendor":"Meta","version":3.1,"backend":"ollama","variant":"vulkan","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":4.7,"prefill_tps":79.6,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","credibility":0.78,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv","last_updated":"2026-09-06","first_seen":"2026-08-30","signals":{"stars":323,"forks":22}},{"model":"Qwen3.8-Flash-Next","quant":"NVFP4","format":null,"hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.8,"backend":"sglang","variant":null,"ctx":262144,"mode":"c6","concurrency":6,"num_gpus":2,"decode_tps":306,"prefill_tps":null,"throughput_kind":"aggregate","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0,"credibilityLabel":"none","author":"randomllama","url":"https://huggingface.co/randomllama/Qwen3.8-Flash-Next-DGX-Spark-field-notes","doc_url":"https://huggingface.co/randomllama/Qwen3.8-Flash-Next-DGX-Spark-field-notes/blob/main/README.md","last_updated":"2026-09-02","first_seen":"2026-09-02","signals":{"downloads":0,"likes":0}},{"model":"Qwen3.8-Flash-Next","quant":"NVFP4","format":null,"hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.8,"backend":"sglang","variant":null,"ctx":262144,"mode":null,"concurrency":16,"num_gpus":2,"decode_tps":275.4,"prefill_tps":null,"throughput_kind":"aggregate","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.13,"credibilityLabel":"low","author":"PixelML","url":"https://huggingface.co/PixelML/Qwen3.8-Flash-Next-NVFP4-Dual-DGX-Spark","doc_url":"https://huggingface.co/PixelML/Qwen3.8-Flash-Next-NVFP4-Dual-DGX-Spark/blob/main/README.md","last_updated":"2026-09-06","first_seen":"2026-08-29","signals":{"downloads":167,"likes":0}},{"model":"Qwen3.8-Flash-Next","quant":"FP8","format":null,"hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.8,"backend":null,"variant":null,"ctx":null,"mode":null,"concurrency":32,"num_gpus":1,"decode_tps":266.8,"prefill_tps":null,"throughput_kind":"aggregate","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.26,"credibilityLabel":"low","author":"jschmied","url":"https://github.com/jschmied/qwen38-flash-next-gb10","doc_url":"https://github.com/jschmied/qwen38-flash-next-gb10/blob/HEAD/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"stars":6,"forks":0,"watchers":6,"open_issues":0}},{"model":"DeepSeek-V4-Flash-0731","quant":"4-bit","format":null,"hardware":"mac-ultra","hardwareLabel":"Apple Mac Studio (M-series Ultra)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":"mlx","variant":null,"ctx":13900,"mode":"batch8","concurrency":8,"num_gpus":2,"decode_tps":112,"prefill_tps":1012,"throughput_kind":"aggregate","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.18,"credibilityLabel":"low","author":"avlp12","url":"https://github.com/avlp12/two-macstudio-m3u","doc_url":"https://github.com/avlp12/two-macstudio-m3u/blob/HEAD/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"stars":1,"forks":1,"watchers":1,"open_issues":0}},{"model":"DeepSeek-V4-Flash-DSpark","quant":"NVFP4","format":null,"hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":"vllm","variant":null,"ctx":1000000,"mode":"peak","concurrency":null,"num_gpus":2,"decode_tps":84.3,"prefill_tps":null,"throughput_kind":"multi-gpu","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.83,"credibilityLabel":"high","author":"tonyd2wild","url":"https://github.com/tonyd2wild/DeepSeek-v4-Flash-0731-DSpark-1M-NVFP4-KV-2x-DGX-Spark","doc_url":"https://github.com/tonyd2wild/DeepSeek-v4-Flash-0731-DSpark-1M-NVFP4-KV-2x-DGX-Spark/blob/HEAD/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"stars":452,"forks":54,"watchers":452,"open_issues":15}},{"model":"GLM-5.3-Flash","quant":"NVFP4","format":null,"hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Zhipu AI","version":5.3,"backend":"sglang","variant":null,"ctx":131072,"mode":"dflash2-c12","concurrency":12,"num_gpus":2,"decode_tps":83.2,"prefill_tps":null,"throughput_kind":"aggregate","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.26,"credibilityLabel":"low","author":"randomllama","url":"https://huggingface.co/randomllama/GLM-5.3-Flash-DFlash2-SGLang-2x-DGX-Spark","doc_url":"https://huggingface.co/randomllama/GLM-5.3-Flash-DFlash2-SGLang-2x-DGX-Spark/blob/main/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"downloads":0,"likes":2}},{"model":"Qwen3.8-Flash-Next","quant":"NVFP4","format":null,"hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.8,"backend":"sglang","variant":null,"ctx":262144,"mode":"c1","concurrency":1,"num_gpus":2,"decode_tps":63,"prefill_tps":null,"throughput_kind":"multi-gpu","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0,"credibilityLabel":"none","author":"randomllama","url":"https://huggingface.co/randomllama/Qwen3.8-Flash-Next-DGX-Spark-field-notes","doc_url":"https://huggingface.co/randomllama/Qwen3.8-Flash-Next-DGX-Spark-field-notes/blob/main/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"downloads":0,"likes":0}},{"model":"DeepSeek-V4-Flash-0731","quant":null,"format":null,"hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":"nvidia-dgx-spark","oemLabel":"NVIDIA DGX Spark","vendor":"DeepSeek","version":4,"backend":null,"variant":"cuda","ctx":null,"mode":null,"concurrency":12,"num_gpus":1,"decode_tps":59,"prefill_tps":null,"throughput_kind":"aggregate","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.8,"credibilityLabel":"high","author":"Entrpi","url":"https://github.com/Entrpi/ds4-on-spark","doc_url":"https://github.com/Entrpi/ds4-on-spark/blob/HEAD/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"stars":384,"forks":26,"watchers":384,"open_issues":4}},{"model":"DeepSeek-V4-Flash-0731-JA-REAP-K216","quant":"EXL3-3BPW","format":null,"hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":null,"variant":null,"ctx":256000,"mode":null,"concurrency":1,"num_gpus":1,"decode_tps":55,"prefill_tps":null,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.1,"credibilityLabel":"low","author":"Laplace1313","url":"https://huggingface.co/Laplace1313/DeepSeek-V4-Flash-0731-JA-REAP-K216-EXL3-3bpw-DGX-Spark","doc_url":"https://huggingface.co/Laplace1313/DeepSeek-V4-Flash-0731-JA-REAP-K216-EXL3-3bpw-DGX-Spark/blob/main/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"downloads":118,"likes":0}},{"model":"DeepSeek-V4-Flash-0731","quant":"4-bit","format":null,"hardware":"mac-ultra","hardwareLabel":"Apple Mac Studio (M-series Ultra)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":"mlx","variant":null,"ctx":13900,"mode":"single","concurrency":1,"num_gpus":2,"decode_tps":45,"prefill_tps":1012,"throughput_kind":"multi-gpu","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.18,"credibilityLabel":"low","author":"avlp12","url":"https://github.com/avlp12/two-macstudio-m3u","doc_url":"https://github.com/avlp12/two-macstudio-m3u/blob/HEAD/README.md","last_updated":"2026-09-02","first_seen":"2026-09-02","signals":{"stars":1,"forks":1,"watchers":1,"open_issues":0}},{"model":"Qwen3.8-Flash-Next","quant":"NVFP4","format":null,"hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.8,"backend":"vllm","variant":null,"ctx":262144,"mode":"mtp-2","concurrency":1,"num_gpus":1,"decode_tps":44.2,"prefill_tps":null,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0,"credibilityLabel":"none","author":"YSLAB-ai","url":"https://huggingface.co/YSLAB-ai/Qwen3.8-Flash-Next-NVFP4-BF16PLE-DGX-Spark","doc_url":"https://huggingface.co/YSLAB-ai/Qwen3.8-Flash-Next-NVFP4-BF16PLE-DGX-Spark/blob/main/README.md","last_updated":"2026-09-01","first_seen":"2026-09-01","signals":{"downloads":0,"likes":0}},{"model":"Qwen3.8-Flash-Next-REAP-288","quant":"MLX-4BIT","format":null,"hardware":"mac-max","hardwareLabel":"Apple M-series Max (MacBook Pro / Mac Studio Max)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.8,"backend":"mlx","variant":"pmlx","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":37,"prefill_tps":null,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.56,"credibilityLabel":"medium","author":"sh0wie","url":"https://huggingface.co/sh0wie/Qwen3.8-Flash-Next-REAP-288-MLX-4bit","doc_url":"https://huggingface.co/sh0wie/Qwen3.8-Flash-Next-REAP-288-MLX-4bit/blob/main/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"downloads":1055,"likes":19}},{"model":"GLM-5.3-Flash","quant":"FP8","format":null,"hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":"nvidia-dgx-spark","oemLabel":"NVIDIA DGX Spark","vendor":"Zhipu AI","version":5.3,"backend":"vllm","variant":null,"ctx":null,"mode":null,"concurrency":null,"num_gpus":4,"decode_tps":36,"prefill_tps":null,"throughput_kind":"multi-gpu","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0,"credibilityLabel":"none","author":"mpfaffenberger","url":"https://github.com/mpfaffenberger/GLM-5.3-Flash-TP4-4x-DGX-Spark","doc_url":"https://github.com/mpfaffenberger/GLM-5.3-Flash-TP4-4x-DGX-Spark/blob/HEAD/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"stars":0,"forks":0,"watchers":0,"open_issues":0}},{"model":"DeepSeek-V4-Flash-0731","quant":"UD-IQ3_XXS","format":null,"hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":"llama.cpp","variant":"vulkan","ctx":524288,"mode":null,"concurrency":null,"num_gpus":1,"decode_tps":35.7,"prefill_tps":226.8,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.38,"credibilityLabel":"low","author":"pepuscz","url":"https://github.com/pepuscz/strix-halo-deepseek-v4-flash","doc_url":"https://github.com/pepuscz/strix-halo-deepseek-v4-flash/blob/HEAD/README.md","last_updated":"2026-08-31","first_seen":"2026-08-29","signals":{"stars":13,"forks":2}},{"model":"Qwen3.8-Flash-Next","quant":"IQ3_XXS","format":null,"hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.8,"backend":"llama.cpp","variant":null,"ctx":156000,"mode":"mtp","concurrency":null,"num_gpus":null,"decode_tps":35.7,"prefill_tps":null,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.42,"credibilityLabel":"medium","author":"EasiiX","url":"https://huggingface.co/EasiiX/Qwen3.8-Flash-Next-MTP-Strix-Halo-GGUF","doc_url":"https://huggingface.co/EasiiX/Qwen3.8-Flash-Next-MTP-Strix-Halo-GGUF/blob/main/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"downloads":1411,"likes":3}},{"model":"GLM-5.3-Flash","quant":"MLX-4BIT","format":null,"hardware":"mac-ultra","hardwareLabel":"Apple Mac Studio (M-series Ultra)","oem":null,"oemLabel":null,"vendor":"Zhipu AI","version":5.3,"backend":"mlx","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":1,"decode_tps":34.2,"prefill_tps":null,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.09,"credibilityLabel":"low","author":"drowzeys","url":"https://github.com/drowzeys/keys-Mac-oMLX-0.6.3.2RC-Dual-ANE-GLM-5.3-Flash-Abliterated-oQ4","doc_url":"https://github.com/drowzeys/keys-Mac-oMLX-0.6.3.2RC-Dual-ANE-GLM-5.3-Flash-Abliterated-oQ4/blob/HEAD/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"stars":1,"forks":0,"watchers":1,"open_issues":0}},{"model":"DeepSeek-V4-Flash-0731","quant":"2.58bpw-mix","format":null,"hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":null,"variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":1,"decode_tps":34.2,"prefill_tps":null,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.75,"credibilityLabel":"high","author":"otheru","url":"https://huggingface.co/otheru/DeepSeek-V4-Flash-Strix-Halo-GGUF","doc_url":"https://huggingface.co/otheru/DeepSeek-V4-Flash-Strix-Halo-GGUF/blob/main/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"downloads":22777,"likes":20}},{"model":"DeepSeek-V4-Flash-0731","quant":"NVFP4","format":null,"hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":"vllm","variant":null,"ctx":1000000,"mode":"prose","concurrency":null,"num_gpus":2,"decode_tps":33.2,"prefill_tps":null,"throughput_kind":"multi-gpu","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.83,"credibilityLabel":"high","author":"tonyd2wild","url":"https://github.com/tonyd2wild/DeepSeek-v4-Flash-0731-DSpark-1M-NVFP4-KV-2x-DGX-Spark","doc_url":"https://github.com/tonyd2wild/DeepSeek-v4-Flash-0731-DSpark-1M-NVFP4-KV-2x-DGX-Spark/blob/HEAD/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"stars":452,"forks":54,"watchers":452,"open_issues":15}},{"model":"DeepSeek-V4-Flash-0731-MXFP4-MLX-Abliterated","quant":"MXFP4","format":null,"hardware":"mac-ultra","hardwareLabel":"Apple Mac Studio (M-series Ultra)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":"mlx","variant":null,"ctx":1048576,"mode":null,"concurrency":1,"num_gpus":1,"decode_tps":29.6,"prefill_tps":454,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.26,"credibilityLabel":"low","author":"drowzeys","url":"https://github.com/drowzeys/keys-Mac-oMLX-0.6.3.2RC-DeepSeekV4F-0731-MXFP4-Abliterated-Dual-ANE-CPU","doc_url":"https://github.com/drowzeys/keys-Mac-oMLX-0.6.3.2RC-DeepSeekV4F-0731-MXFP4-Abliterated-Dual-ANE-CPU/blob/HEAD/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"stars":6,"forks":0,"watchers":6,"open_issues":0}},{"model":"DeepSeek-V4-Flash-0731","quant":"UD-IQ3_XXS","format":null,"hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"DeepSeek","version":4,"backend":"llama.cpp","variant":"rocm","ctx":131072,"mode":null,"concurrency":null,"num_gpus":1,"decode_tps":29.6,"prefill_tps":142.8,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.38,"credibilityLabel":"low","author":"pepuscz","url":"https://github.com/pepuscz/strix-halo-deepseek-v4-flash","doc_url":"https://github.com/pepuscz/strix-halo-deepseek-v4-flash/blob/HEAD/README.md","last_updated":"2026-08-31","first_seen":"2026-08-29","signals":{"stars":13,"forks":2}},{"model":"GLM-5.3-Flash","quant":"NVFP4","format":null,"hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Zhipu AI","version":5.3,"backend":"sglang","variant":null,"ctx":131072,"mode":"dflash2-c1","concurrency":1,"num_gpus":2,"decode_tps":28.6,"prefill_tps":null,"throughput_kind":"multi-gpu","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.26,"credibilityLabel":"low","author":"randomllama","url":"https://huggingface.co/randomllama/GLM-5.3-Flash-DFlash2-SGLang-2x-DGX-Spark","doc_url":"https://huggingface.co/randomllama/GLM-5.3-Flash-DFlash2-SGLang-2x-DGX-Spark/blob/main/README.md","last_updated":"2026-09-02","first_seen":"2026-09-02","signals":{"downloads":0,"likes":2}},{"model":"Qwen3.8-Flash-Next-REAP-288","quant":"MLX-4BIT","format":null,"hardware":"mac-max","hardwareLabel":"Apple M-series Max (MacBook Pro / Mac Studio Max)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.8,"backend":"mlx","variant":null,"ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":28,"prefill_tps":null,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.56,"credibilityLabel":"medium","author":"sh0wie","url":"https://huggingface.co/sh0wie/Qwen3.8-Flash-Next-REAP-288-MLX-4bit","doc_url":"https://huggingface.co/sh0wie/Qwen3.8-Flash-Next-REAP-288-MLX-4bit/blob/main/README.md","last_updated":"2026-09-01","first_seen":"2026-09-01","signals":{"downloads":1055,"likes":19}},{"model":"Qwen3.8-Flash-Next","quant":"NVFP4","format":null,"hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.8,"backend":"vllm","variant":null,"ctx":null,"mode":"mtp-1","concurrency":null,"num_gpus":null,"decode_tps":27.6,"prefill_tps":null,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.3,"credibilityLabel":"low","author":"sayyidfareed","url":"https://huggingface.co/sayyidfareed/Qwen3.8-Flash-Next-DGX-Spark-1M-Recipe","doc_url":"https://huggingface.co/sayyidfareed/Qwen3.8-Flash-Next-DGX-Spark-1M-Recipe/blob/main/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"downloads":0,"likes":3}},{"model":"Qwen3.8-Flash-Next","quant":"NVFP4","format":null,"hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.8,"backend":"vllm","variant":null,"ctx":262144,"mode":"baseline","concurrency":1,"num_gpus":1,"decode_tps":27.3,"prefill_tps":null,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0,"credibilityLabel":"none","author":"YSLAB-ai","url":"https://huggingface.co/YSLAB-ai/Qwen3.8-Flash-Next-NVFP4-BF16PLE-DGX-Spark","doc_url":"https://huggingface.co/YSLAB-ai/Qwen3.8-Flash-Next-NVFP4-BF16PLE-DGX-Spark/blob/main/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"downloads":0,"likes":0}},{"model":"Qwen3.8-Flash-Next","quant":"NVFP4","format":null,"hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.8,"backend":"vllm","variant":null,"ctx":1000000,"mode":null,"concurrency":1,"num_gpus":1,"decode_tps":26.7,"prefill_tps":null,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.3,"credibilityLabel":"low","author":"sayyidfareed","url":"https://huggingface.co/sayyidfareed/Qwen3.8-Flash-Next-DGX-Spark-1M-Recipe","doc_url":"https://huggingface.co/sayyidfareed/Qwen3.8-Flash-Next-DGX-Spark-1M-Recipe/blob/main/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"downloads":0,"likes":3}},{"model":"SuperQwen3.8-27b-abliterated","quant":"NVFP4","format":null,"hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.8,"backend":"vllm","variant":null,"ctx":262043,"mode":null,"concurrency":1,"num_gpus":1,"decode_tps":25.8,"prefill_tps":null,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.56,"credibilityLabel":"medium","author":"Jiunsong","url":"https://huggingface.co/Jiunsong/SuperQwen3.8-27b-abliterated-NVFP4-DGX-Spark","doc_url":"https://huggingface.co/Jiunsong/SuperQwen3.8-27b-abliterated-NVFP4-DGX-Spark/blob/main/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"downloads":1207,"likes":19}},{"model":"Qwen3.8-Flash-Next","quant":"Q4_K_XL","format":null,"hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.8,"backend":"llama.cpp","variant":"vulkan","ctx":null,"mode":null,"concurrency":null,"num_gpus":null,"decode_tps":24,"prefill_tps":null,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0,"credibilityLabel":"none","author":"routhjim","url":"https://github.com/routhjim/fn-expert-swap","doc_url":"https://github.com/routhjim/fn-expert-swap/blob/HEAD/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"stars":0,"forks":0,"watchers":0,"open_issues":0}},{"model":"GLM-5.3-Flash Abliterated","quant":"MLX-4BIT","format":null,"hardware":"mac-ultra","hardwareLabel":"Apple Mac Studio (M-series Ultra)","oem":null,"oemLabel":null,"vendor":"Zhipu AI","version":5.3,"backend":"mlx","variant":null,"ctx":16384,"mode":"mtp","concurrency":1,"num_gpus":1,"decode_tps":24,"prefill_tps":365,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.09,"credibilityLabel":"low","author":"drowzeys","url":"https://github.com/drowzeys/keys-Mac-oMLX-0.6.3.2RC-Dual-ANE-GLM-5.3-Flash-Abliterated-oQ4","doc_url":"https://github.com/drowzeys/keys-Mac-oMLX-0.6.3.2RC-Dual-ANE-GLM-5.3-Flash-Abliterated-oQ4/blob/HEAD/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"stars":1,"forks":0,"watchers":1,"open_issues":0}},{"model":"Qwen3.8-Flash-Next","quant":"IQ3_XXS","format":null,"hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.8,"backend":"llama.cpp","variant":null,"ctx":156000,"mode":"baseline","concurrency":null,"num_gpus":null,"decode_tps":23.5,"prefill_tps":null,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.42,"credibilityLabel":"medium","author":"EasiiX","url":"https://huggingface.co/EasiiX/Qwen3.8-Flash-Next-MTP-Strix-Halo-GGUF","doc_url":"https://huggingface.co/EasiiX/Qwen3.8-Flash-Next-MTP-Strix-Halo-GGUF/blob/main/README.md","last_updated":"2026-09-01","first_seen":"2026-09-01","signals":{"downloads":1411,"likes":3}},{"model":"Qwen3.8-27B","quant":"UD-Q5_K_XL","format":null,"hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.8,"backend":"llama.cpp","variant":null,"ctx":null,"mode":"mtp","concurrency":1,"num_gpus":1,"decode_tps":23,"prefill_tps":null,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.32,"credibilityLabel":"low","author":"PieBru","url":"https://github.com/PieBru/Qwen-3.8-27B_Strix-Halo_gfx1151","doc_url":"https://github.com/PieBru/Qwen-3.8-27B_Strix-Halo_gfx1151/blob/HEAD/README.md","last_updated":"2026-08-31","first_seen":"2026-08-29","signals":{"stars":10,"forks":0}},{"model":"Qwen3.8 27B","quant":"Q4_K_M","format":null,"hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.8,"backend":"ollama","variant":null,"ctx":null,"mode":null,"concurrency":1,"num_gpus":1,"decode_tps":20.4,"prefill_tps":292.5,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.77,"credibilityLabel":"high","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide","doc_url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"stars":309,"forks":22}},{"model":"Qwen3.8-27B","quant":"UD-Q6_K_XL","format":null,"hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.8,"backend":"llama.cpp","variant":null,"ctx":131072,"mode":null,"concurrency":1,"num_gpus":1,"decode_tps":17,"prefill_tps":346,"throughput_kind":"single-stream","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.32,"credibilityLabel":"low","author":"PieBru","url":"https://github.com/PieBru/Qwen-3.8-27B_Strix-Halo_gfx1151","doc_url":"https://github.com/PieBru/Qwen-3.8-27B_Strix-Halo_gfx1151/blob/HEAD/README.md","last_updated":"2026-08-31","first_seen":"2026-08-29","signals":{"stars":10,"forks":0}},{"model":"SuperQwen3.8-Flash-Next-abliterated","quant":"FP8","format":null,"hardware":"dgx-spark","hardwareLabel":"NVIDIA DGX Spark (GB10)","oem":null,"oemLabel":null,"vendor":"Alibaba","version":3.8,"backend":"sglang","variant":null,"ctx":8192,"mode":null,"concurrency":1,"num_gpus":2,"decode_tps":16.2,"prefill_tps":null,"throughput_kind":"multi-gpu","trust":"linked-extract","trustLabel":"Extracted, source-linked","credibility":0.27,"credibilityLabel":"low","author":"Jiunsong","url":"https://huggingface.co/Jiunsong/SuperQwen3.8-Flash-Next-abliterated-FP8-DGX-Spark","doc_url":"https://huggingface.co/Jiunsong/SuperQwen3.8-Flash-Next-abliterated-FP8-DGX-Spark/blob/main/README.md","last_updated":"2026-08-29","first_seen":"2026-08-29","signals":{"downloads":54,"likes":2}}],"intel":[{"id":"https://huggingface.co/nerkyor/Qwen3.8-27B-EfficientThink-Uncensored-K3-Opus5-Grok4.6-GPT5.6Sol-SFT-SimPO-DFlash2-GGUF","headline":"nerkyor/Qwen3.8-27B-EfficientThink-Uncensored-K3-Opus5-Grok4.6-GPT5.6Sol-SFT-SimPO-DFlash2-GGUF","url":"https://huggingface.co/nerkyor/Qwen3.8-27B-EfficientThink-Uncensored-K3-Opus5-Grok4.6-GPT5.6Sol-SFT-SimPO-DFlash2-GGUF/blob/main/README.md","source_name":"hf-model","one_liner":"This 27B model fits comfortably in 128GB, offering a notable new SFT/SimPO recipe for efficient, uncensored local inference.","category":"model","published_at":"2026-09-06T11:06:57.000Z","image_url":null},{"id":"https://github.com/Gilamonster-Foundation/newt-agent","headline":"Gilamonster-Foundation/newt-agent","url":"https://github.com/Gilamonster-Foundation/newt-agent/blob/HEAD/README.md","source_name":"github-repo","one_liner":"Newt-agent brings experimental agentic coding to your DGX Spark, letting you run local inference you own on your 128GB box.","category":"dgx-spark","published_at":"2026-09-06T10:59:30Z","image_url":null},{"id":"https://huggingface.co/ngocle2303/e2_qlora_qwen3_8b_gguf","headline":"ngocle2303/e2_qlora_qwen3_8b_gguf","url":"https://huggingface.co/ngocle2303/e2_qlora_qwen3_8b_gguf/blob/main/README.md","source_name":"hf-model","one_liner":"Qwen3-8B fits easily in 128GB, but this niche E2-QLoRA recipe offers no clear speed or capability advantage over standard GGUFs.","category":"model","published_at":"2026-09-06T10:59:00.000Z","image_url":null},{"id":"https://github.com/Avarok-Cybersecurity/atlas","headline":"Avarok-Cybersecurity/atlas","url":"https://github.com/Avarok-Cybersecurity/atlas/blob/HEAD/README.md","source_name":"github-repo","one_liner":"Atlas is a pure Rust inference engine for DGX Spark, but it is a framework, not a model, so it does not directly fit in 128GB.","category":"dgx-spark","published_at":"2026-09-06T10:54:54Z","image_url":null},{"id":"https://github.com/maikzz32/strix-halo-vllm-cluster","headline":"maikzz32/strix-halo-vllm-cluster","url":"https://github.com/maikzz32/strix-halo-vllm-cluster/blob/HEAD/README.md","source_name":"github-repo","one_liner":"It enables vLLM cluster orchestration on Strix Halo, letting you run larger models across multiple 128GB boxes that won't fit on a single one.","category":"strix-halo","published_at":"2026-09-06T10:54:47Z","image_url":null},{"id":"https://github.com/NNNtrance/GLM-5.3-Flash-EXL3-DGX-Spark","headline":"NNNtrance/GLM-5.3-Flash-EXL3-DGX-Spark","url":"https://github.com/NNNtrance/GLM-5.3-Flash-EXL3-DGX-Spark/blob/HEAD/README.md","source_name":"github-repo","one_liner":"This release enables running GLM-5.3-Flash on a single DGX Spark via EXL3 4bpw, proving the model fits comfortably within the 128GB memory limit.","category":"dgx-spark","published_at":"2026-09-06T10:54:34Z","image_url":null},{"id":"https://github.com/madeye/qwen38-flash-next-on-dgx-spark","headline":"madeye/qwen38-flash-next-on-dgx-spark","url":"https://github.com/madeye/qwen38-flash-next-on-dgx-spark/blob/HEAD/README.md","source_name":"github-repo","one_liner":"This release matters because it demonstrates a viable, high-performance Qwen3.8-Flash recipe specifically optimized for the DGX Spark's 128GB unified memory.","category":"dgx-spark","published_at":"2026-09-06T10:54:15Z","image_url":null},{"id":"https://github.com/hgnnmnn/ai-stack","headline":"hgnnmnn/ai-stack","url":"https://github.com/hgnnmnn/ai-stack/blob/HEAD/README.md","source_name":"github-repo","one_liner":"This release matters because it provides a ready-made, self-hosted inference stack specifically optimized for AMD Strix Halo's Vulkan backend, simplifying local LLM deployment.","category":"strix-halo","published_at":"2026-09-06T10:54:07Z","image_url":null},{"id":"https://huggingface.co/LuffyTheFox/Qwen3.8-27B-Uncensored-Genesis-NVFP4-GGUF","headline":"LuffyTheFox/Qwen3.8-27B-Uncensored-Genesis-NVFP4-GGUF","url":"https://huggingface.co/LuffyTheFox/Qwen3.8-27B-Uncensored-Genesis-NVFP4-GGUF/blob/main/README.md","source_name":"hf-model","one_liner":"This NVFP4 GGUF quant lets you run the new 27B Qwen3.8 uncensored model comfortably on a single 128GB box.","category":"model","published_at":"2026-09-06T10:45:43.000Z","image_url":null},{"id":"https://huggingface.co/mradermacher/loes-qwen3.8-27b-i1-GGUF","headline":"mradermacher/loes-qwen3.8-27b-i1-GGUF","url":"https://huggingface.co/mradermacher/loes-qwen3.8-27b-i1-GGUF/blob/main/README.md","source_name":"hf-model","one_liner":"This Qwen3 27B release fits comfortably in 128GB, offering a high-quality local option for your single-box setup.","category":"model","published_at":"2026-09-06T10:44:20.000Z","image_url":null},{"id":"stanford-intelligence-per-watt-2511.07885","headline":"Intelligence per Watt (Stanford): local models answer 88.7% of 1M real queries, IPW up 5.3x since 2023","url":"https://arxiv.org/abs/2511.07885","source_name":"pinned","one_liner":"Stanford measured task accuracy per watt across 20+ local models and 8 accelerators on 1 million real queries: the best local model of 20B active parameters or fewer answers 88.7%, intelligence per watt rose 5.3x from 2023 to 2025, and a local-first router that is right 80% of the time cuts energy 64% and cost 59% against cloud only. The only local box they profiled is an Apple M4 Max; the Apache-2.0 harness reads AMD power, so Strix Halo numbers are ours to add.","category":"research","published_at":"2026-09-06","image_url":null,"pinned":true},{"id":"nvidia-ifa-2026-local-ai-optimizations","headline":"NVIDIA IFA 2026: llama.cpp up to 1.9x, vLLM 1.4x on two DGX Sparks, one-click local setup","url":"https://blogs.nvidia.com/blog/local-ai-ifa-next-gen-agents-nv-pair-rtx-spark/","source_name":"pinned","one_liner":"llama.cpp hits up to 1.9x on an RTX 5090 (kernel opts, better speculative decoding, faster prefill) and vLLM 1.4x on two DGX Sparks; one-click Windows setup ships via Ollama and LM Studio, and NVIDIA PAIR spreads inference across your local-network PCs. Same box, bigger and faster local models, so a lot of current bench numbers are now stale.","category":"optimization","published_at":"2026-09-03","image_url":null,"pinned":true}],"articles":[{"id":"https://vllm.ai/blog/2026-08-23-speculative-decoding-amd-gpus","headline":"Speculative Decoding in vLLM on AMD GPUs","url":"https://vllm.ai/blog/2026-08-23-speculative-decoding-amd-gpus","source_name":"article","one_liner":"Optimizes inference speed on AMD hardware, directly benefiting users running LLMs on Strix Halo or other AMD-based single-box setups.","category":"news","published_at":"2026-09-07T09:26:41Z","image_url":null,"via":"Hacker News"},{"id":"https://hugovergnes.github.io/little-lm-3-8b","headline":"Training a 3.8B LLM to 0.384 CORE for $998 – Hugo Vergnes","url":"https://hugovergnes.github.io/little-lm-3-8b/","source_name":"article","one_liner":"Demonstrates cost-effective training of small models, directly relevant to users running LLMs on single-box hardware.","category":"news","published_at":"2026-09-10T02:04:11Z","image_url":null,"via":"Hacker News"},{"id":"https://www.phoronix.com/news/NVIDIA-615.71.09-Linux-Driver","headline":"NVIDIA 615.71.09 Linux Driver Released With Vulkan Improvements","url":"https://www.phoronix.com/news/NVIDIA-615.71.09-Linux-Driver","source_name":"article","one_liner":"Vulkan improvements directly enhance GPU inference performance on NVIDIA hardware like the DGX Spark.","category":"news","published_at":"2026-09-09T14:02:37.000Z","image_url":null,"via":"Phoronix"},{"id":"https://www.phoronix.com/news/Experiment-Nouveau-NVIDIA-GB10","headline":"Experimental Patches Get Nouveau+NVK Working On NVIDIA DGX Spark GB10","url":"https://www.phoronix.com/news/Experiment-Nouveau-NVIDIA-GB10","source_name":"article","one_liner":"Enables open-source Vulkan rendering on DGX Spark, removing proprietary driver dependencies for local LLM inference and visualization.","category":"news","published_at":"2026-09-09T10:13:00.000Z","image_url":null,"via":"Phoronix"}],"latestReleases":[{"title":"nerkyor/Qwen3.8-27B-EfficientThink-Uncensored-K3-Opus5-Grok4.6-GPT5.6Sol-SFT-SimPO-DFlash2-GGUF","url":"https://huggingface.co/nerkyor/Qwen3.8-27B-EfficientThink-Uncensored-K3-Opus5-Grok4.6-GPT5.6Sol-SFT-SimPO-DFlash2-GGUF","doc_url":"https://huggingface.co/nerkyor/Qwen3.8-27B-EfficientThink-Uncensored-K3-Opus5-Grok4.6-GPT5.6Sol-SFT-SimPO-DFlash2-GGUF/blob/main/README.md","kind":"hf-model","platform":null,"desc":"","stars":null,"forks":null,"downloads":6874,"likes":4,"format":"GGUF","date":"2026-09-06","seed":false},{"title":"Gilamonster-Foundation/newt-agent","url":"https://github.com/Gilamonster-Foundation/newt-agent","doc_url":"https://github.com/Gilamonster-Foundation/newt-agent/blob/HEAD/README.md","kind":"github-repo","platform":"dgx-spark","desc":"experimental agentic coder for ollama, llama.cpp, and vLLM ... inference you OWN","stars":6,"forks":1,"downloads":null,"likes":null,"format":null,"date":"2026-09-06","seed":false},{"title":"ngocle2303/e2_qlora_qwen3_8b_gguf","url":"https://huggingface.co/ngocle2303/e2_qlora_qwen3_8b_gguf","doc_url":"https://huggingface.co/ngocle2303/e2_qlora_qwen3_8b_gguf/blob/main/README.md","kind":"hf-model","platform":null,"desc":"","stars":null,"forks":null,"downloads":0,"likes":0,"format":"GGUF","date":"2026-09-06","seed":false},{"title":"Avarok-Cybersecurity/atlas","url":"https://github.com/Avarok-Cybersecurity/atlas","doc_url":"https://github.com/Avarok-Cybersecurity/atlas/blob/HEAD/README.md","kind":"github-repo","platform":"dgx-spark","desc":"Pure Rust Inference Engine","stars":677,"forks":102,"downloads":null,"likes":null,"format":null,"date":"2026-09-06","seed":false},{"title":"maikzz32/strix-halo-vllm-cluster","url":"https://github.com/maikzz32/strix-halo-vllm-cluster","doc_url":"https://github.com/maikzz32/strix-halo-vllm-cluster/blob/HEAD/README.md","kind":"github-repo","platform":"strix-halo","desc":"","stars":0,"forks":0,"downloads":null,"likes":null,"format":null,"date":"2026-09-06","seed":false},{"title":"NNNtrance/GLM-5.3-Flash-EXL3-DGX-Spark","url":"https://github.com/NNNtrance/GLM-5.3-Flash-EXL3-DGX-Spark","doc_url":"https://github.com/NNNtrance/GLM-5.3-Flash-EXL3-DGX-Spark/blob/HEAD/README.md","kind":"github-repo","platform":"dgx-spark","desc":"Run zai-org/GLM-5.3-Flash as an EXL3 4bpw checkpoint on NVIDIA DGX Spark (GB10) with vLLM and cuda-exl3 — measured two-n","stars":2,"forks":0,"downloads":null,"likes":null,"format":null,"date":"2026-09-06","seed":false},{"title":"madeye/qwen38-flash-next-on-dgx-spark","url":"https://github.com/madeye/qwen38-flash-next-on-dgx-spark","doc_url":"https://github.com/madeye/qwen38-flash-next-on-dgx-spark/blob/HEAD/README.md","kind":"github-repo","platform":"dgx-spark","desc":"","stars":2,"forks":0,"downloads":null,"likes":null,"format":null,"date":"2026-09-06","seed":false},{"title":"hgnnmnn/ai-stack","url":"https://github.com/hgnnmnn/ai-stack","doc_url":"https://github.com/hgnnmnn/ai-stack/blob/HEAD/README.md","kind":"github-repo","platform":"strix-halo","desc":"Self-hosted LLM inference stack: llama.cpp Backends + LiteLLM Gateway on AMD Strix Halo (Vulkan/RADV)","stars":0,"forks":0,"downloads":null,"likes":null,"format":null,"date":"2026-09-06","seed":false},{"title":"LuffyTheFox/Qwen3.8-27B-Uncensored-Genesis-NVFP4-GGUF","url":"https://huggingface.co/LuffyTheFox/Qwen3.8-27B-Uncensored-Genesis-NVFP4-GGUF","doc_url":"https://huggingface.co/LuffyTheFox/Qwen3.8-27B-Uncensored-Genesis-NVFP4-GGUF/blob/main/README.md","kind":"hf-model","platform":null,"desc":"","stars":null,"forks":null,"downloads":0,"likes":0,"format":"GGUF","date":"2026-09-06","seed":false},{"title":"mradermacher/loes-qwen3.8-27b-i1-GGUF","url":"https://huggingface.co/mradermacher/loes-qwen3.8-27b-i1-GGUF","doc_url":"https://huggingface.co/mradermacher/loes-qwen3.8-27b-i1-GGUF/blob/main/README.md","kind":"hf-model","platform":null,"desc":"","stars":null,"forks":null,"downloads":2,"likes":0,"format":"GGUF","date":"2026-09-06","seed":false},{"title":"mradermacher/Qwen2.5-3B-Instruct-Palimpzest-Map-GGUF","url":"https://huggingface.co/mradermacher/Qwen2.5-3B-Instruct-Palimpzest-Map-GGUF","doc_url":"https://huggingface.co/mradermacher/Qwen2.5-3B-Instruct-Palimpzest-Map-GGUF/blob/main/README.md","kind":"hf-model","platform":null,"desc":"","stars":null,"forks":null,"downloads":0,"likes":0,"format":"GGUF","date":"2026-09-06","seed":false},{"title":"HarithSami/qwen2.5-14b-instruct-arabic-yt-merged-Q4_K_M-GGUF","url":"https://huggingface.co/HarithSami/qwen2.5-14b-instruct-arabic-yt-merged-Q4_K_M-GGUF","doc_url":"https://huggingface.co/HarithSami/qwen2.5-14b-instruct-arabic-yt-merged-Q4_K_M-GGUF/blob/main/README.md","kind":"hf-model","platform":null,"desc":"","stars":null,"forks":null,"downloads":0,"likes":0,"format":"GGUF","date":"2026-09-06","seed":false},{"title":"0rvar/xwen","url":"https://github.com/0rvar/xwen","doc_url":"https://github.com/0rvar/xwen/blob/HEAD/README.md","kind":"github-repo","platform":"mac-max","desc":"Qwen 3.6, tuned to the wall for 128GB M5 Max","stars":0,"forks":0,"downloads":null,"likes":null,"format":null,"date":"2026-09-06","seed":false},{"title":"HarithSami/qwen2.5-14b-instruct-arabic-yt-gguf","url":"https://huggingface.co/HarithSami/qwen2.5-14b-instruct-arabic-yt-gguf","doc_url":"https://huggingface.co/HarithSami/qwen2.5-14b-instruct-arabic-yt-gguf/blob/main/README.md","kind":"hf-model","platform":null,"desc":"","stars":null,"forks":null,"downloads":0,"likes":0,"format":"safetensors","date":"2026-09-06","seed":false},{"title":"mudler/Qwen3.8-Flash-Next-APEX-GGUF","url":"https://huggingface.co/mudler/Qwen3.8-Flash-Next-APEX-GGUF","doc_url":"https://huggingface.co/mudler/Qwen3.8-Flash-Next-APEX-GGUF/blob/main/README.md","kind":"hf-model","platform":null,"desc":"","stars":null,"forks":null,"downloads":0,"likes":10,"format":"GGUF","date":"2026-09-06","seed":false},{"title":"ggml-org/llama.cpp","url":"https://github.com/ggml-org/llama.cpp","doc_url":"https://github.com/ggml-org/llama.cpp/blob/HEAD/README.md","kind":"github-repo","platform":"dgx-spark","desc":"LLM inference in C/C++","stars":127218,"forks":22826,"downloads":null,"likes":null,"format":null,"date":"2026-09-06","seed":true},{"title":"1bit-MONSTER/1bit-MONSTER","url":"https://github.com/1bit-MONSTER/1bit-MONSTER","doc_url":"https://github.com/1bit-MONSTER/1bit-MONSTER/blob/HEAD/README.md","kind":"github-repo","platform":"strix-halo","desc":"One engine, any model. 100% Hugging Face model coverage &  Zero python during runtime. ","stars":17,"forks":6,"downloads":null,"likes":null,"format":null,"date":"2026-09-06","seed":false},{"title":"kksoftwareag/strix-halo-qwen-3.8-flash-next","url":"https://github.com/kksoftwareag/strix-halo-qwen-3.8-flash-next","doc_url":"https://github.com/kksoftwareag/strix-halo-qwen-3.8-flash-next/blob/HEAD/README.md","kind":"github-repo","platform":"strix-halo","desc":"llama.cpp TUI für Qwen3.8-Flash-Next auf AMD Strix Halo","stars":0,"forks":0,"downloads":null,"likes":null,"format":null,"date":"2026-09-06","seed":false},{"title":"Osmantic/ODS","url":"https://github.com/Osmantic/ODS","doc_url":"https://github.com/Osmantic/ODS/blob/HEAD/README.md","kind":"github-repo","platform":"strix-halo","desc":"Turn your PC, Mac, or Linux box into an AI server.  LLM inference, chat UI, voice, agents, workflows, RAG, and image gen","stars":6181,"forks":888,"downloads":null,"likes":null,"format":null,"date":"2026-09-06","seed":false},{"title":"pugant/strix-nebulosa","url":"https://github.com/pugant/strix-nebulosa","doc_url":"https://github.com/pugant/strix-nebulosa/blob/HEAD/README.md","kind":"github-repo","platform":"strix-halo","desc":"llama.cpp lab for AMD Strix Halo — the owl that looks 180B and runs in 64 GB. Quants, dual-drafter routing, and NO-GOs p","stars":5,"forks":0,"downloads":null,"likes":null,"format":null,"date":"2026-09-06","seed":false},{"title":"genishs/strix-halo-monitor","url":"https://github.com/genishs/strix-halo-monitor","doc_url":"https://github.com/genishs/strix-halo-monitor/blob/HEAD/README.md","kind":"github-repo","platform":"strix-halo","desc":"Real-time training/scoring monitor for AMD Strix Halo (gfx1151) unified-memory APUs — GTT, quant/step/scoring progress, ","stars":0,"forks":0,"downloads":null,"likes":null,"format":null,"date":"2026-09-06","seed":false},{"title":"wireframeslayout/strix-halo-onexplayer-super-x","url":"https://github.com/wireframeslayout/strix-halo-onexplayer-super-x","doc_url":"https://github.com/wireframeslayout/strix-halo-onexplayer-super-x/blob/HEAD/README.md","kind":"github-repo","platform":"strix-halo","desc":"Local LLM benchmarks on an ONEXPLAYER SUPER X handheld (Ryzen AI MAX+ 395 / Strix Halo): power-delivery effects, MTP spe","stars":0,"forks":0,"downloads":null,"likes":null,"format":null,"date":"2026-09-06","seed":false},{"title":"nightmedia/Qwen3.6-35B-A3B-Brainwaves-qx64-hi-mlx","url":"https://huggingface.co/nightmedia/Qwen3.6-35B-A3B-Brainwaves-qx64-hi-mlx","doc_url":"https://huggingface.co/nightmedia/Qwen3.6-35B-A3B-Brainwaves-qx64-hi-mlx/blob/main/README.md","kind":"hf-model","platform":null,"desc":"","stars":null,"forks":null,"downloads":0,"likes":0,"format":"MLX","date":"2026-09-06","seed":false},{"title":"ml-explore/mlx-lm","url":"https://github.com/ml-explore/mlx-lm","doc_url":"https://github.com/ml-explore/mlx-lm/blob/HEAD/README.md","kind":"github-repo","platform":"mac-ultra","desc":"Run LLMs with MLX","stars":6905,"forks":1026,"downloads":null,"likes":null,"format":null,"date":"2026-09-06","seed":true},{"title":"MiaAI-Lab/DeepSeek-v4-Flash-DSpark-2x-DGX-Spark","url":"https://github.com/MiaAI-Lab/DeepSeek-v4-Flash-DSpark-2x-DGX-Spark","doc_url":"https://github.com/MiaAI-Lab/DeepSeek-v4-Flash-DSpark-2x-DGX-Spark/blob/HEAD/README.md","kind":"github-repo","platform":"dgx-spark","desc":"DeepSeek-v4-Flash 0731 recipe for 2x DGX Sparks","stars":1247,"forks":173,"downloads":null,"likes":null,"format":null,"date":"2026-09-06","seed":false},{"title":"neopolita/Qwen3.8-Flash-Next-113B-A5B-Niwaki-3bit-mlx","url":"https://huggingface.co/neopolita/Qwen3.8-Flash-Next-113B-A5B-Niwaki-3bit-mlx","doc_url":"https://huggingface.co/neopolita/Qwen3.8-Flash-Next-113B-A5B-Niwaki-3bit-mlx/blob/main/README.md","kind":"hf-model","platform":null,"desc":"","stars":null,"forks":null,"downloads":0,"likes":0,"format":"MLX","date":"2026-09-06","seed":false},{"title":"neopolita/Qwen3.6-35B-A3B-Saikei-8bit-mlx","url":"https://huggingface.co/neopolita/Qwen3.6-35B-A3B-Saikei-8bit-mlx","doc_url":"https://huggingface.co/neopolita/Qwen3.6-35B-A3B-Saikei-8bit-mlx/blob/main/README.md","kind":"hf-model","platform":null,"desc":"","stars":null,"forks":null,"downloads":46,"likes":0,"format":"MLX","date":"2026-09-05","seed":false},{"title":"MiaAI-Lab/Qwen3.8-27B-SGLang-DGX-Spark","url":"https://github.com/MiaAI-Lab/Qwen3.8-27B-SGLang-DGX-Spark","doc_url":"https://github.com/MiaAI-Lab/Qwen3.8-27B-SGLang-DGX-Spark/blob/HEAD/README.md","kind":"github-repo","platform":"dgx-spark","desc":"Qwen3.8 27B on SGLang for DGX Spark","stars":395,"forks":41,"downloads":null,"likes":null,"format":null,"date":"2026-09-05","seed":false},{"title":"srv-sngh/Qwen3.8-27B-mlx-4bit","url":"https://huggingface.co/srv-sngh/Qwen3.8-27B-mlx-4bit","doc_url":"https://huggingface.co/srv-sngh/Qwen3.8-27B-mlx-4bit/blob/main/README.md","kind":"hf-model","platform":null,"desc":"","stars":null,"forks":null,"downloads":225,"likes":0,"format":"MLX","date":"2026-09-05","seed":false},{"title":"spark-arena/sparkrun","url":"https://github.com/spark-arena/sparkrun","doc_url":"https://github.com/spark-arena/sparkrun/blob/HEAD/README.md","kind":"github-repo","platform":"dgx-spark","desc":"sparkrun - launch, manage, and stop LLM inference workloads on NVIDIA DGX Spark systems.  Live support: https://discord.","stars":493,"forks":53,"downloads":null,"likes":null,"format":null,"date":"2026-09-05","seed":false}],"published_at":"2026-09-10T06:44:40.353Z"}