{"query":{"hardware":"Strix Halo","tasks":["coding"],"prefer":"balanced","limit":5},"count":5,"generated":"2026-09-18T21:15:29.462Z","recommendations":[{"model":"DeepSeek V4 Flash 284B","vendor":"DeepSeek","config":{"quant":"UD-IQ2_XXS","backend":"llama.cpp","variant":"vulkan","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","ctx":512,"decode_tps":13.3,"prefill_tps":155.6,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv"},"why":["13.3 tok/s decode on AMD Strix Halo (Ryzen AI Max+ 395)","284B-class model","from a structured benchmark table"]},{"model":"Qwen3-Coder-Next 80B-A3B","vendor":"Alibaba","config":{"quant":"IQ4_XS","backend":"llama.cpp","variant":"vulkan","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","ctx":512,"decode_tps":61.9,"prefill_tps":739,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv"},"why":["61.9 tok/s decode on AMD Strix Halo (Ryzen AI Max+ 395)","80B-class model","coding-focused","from a structured benchmark table"]},{"model":"Qwen3-Coder 30B-A3B","vendor":"Alibaba","config":{"quant":"Q4_K_S","backend":"llama.cpp","variant":"vulkan","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","ctx":512,"decode_tps":98,"prefill_tps":1406.5,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv"},"why":["98 tok/s decode on AMD Strix Halo (Ryzen AI Max+ 395)","30B-class model","coding-focused","from a structured benchmark table"]},{"model":"Qwen3-235B-A22B-Instruct-2507","vendor":"Alibaba","config":{"quant":"UD-Q3_K_XL","backend":"llama.cpp","variant":"vulkan","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","ctx":null,"decode_tps":15.9,"prefill_tps":117.1,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","author":"lhl","url":"https://github.com/lhl/strix-halo-testing/blob/HEAD/llm-bench/Qwen3-235B-A22B-Instruct-2507-UD-Q3_K_XL/results.jsonl"},"why":["15.9 tok/s decode on AMD Strix Halo (Ryzen AI Max+ 395)","235B-class model","from a structured benchmark table"]},{"model":"Qwen3.8-Flash-Next","vendor":"Alibaba","config":{"quant":"IQ4_XS","backend":"llama.cpp","variant":"vulkan","hardware":"strix-halo","hardwareLabel":"AMD Strix Halo (Ryzen AI Max+ 395)","ctx":null,"decode_tps":27.2,"prefill_tps":394.7,"throughput_kind":"single-stream","trust":"structured-table","trustLabel":"Structured table","author":"hogeheer","url":"https://github.com/hogeheer499-commits/strix-halo-guide/blob/HEAD/data/benchmarks.csv"},"why":["27.2 tok/s decode on AMD Strix Halo (Ryzen AI Max+ 395)","177B-class model","from a structured benchmark table"]}]}