From 34a932b6df8fdbeb3d8ef722fc422d1669024e78 Mon Sep 17 00:00:00 2001 From: Cary Palmer <24235924+professorpalmer@users.noreply.github.com> Date: Tue, 6 Oct 2026 20:52:17 -0500 Subject: [PATCH] cuda: take the planar PTQ1_0 layout at one column on Turing as well as Ampere One-column PTQ1_0 decode kept the Ada SOA_ISUM layout on sm_75. The planar PT kernel from #218 is faster there, as it is on Ampere: RTX 2060 SUPER, Bonsai 2 27B, q8_0 K/V, 4k context, greedy, 2 prompts x 400 tokens x 2 reps, stock clocks: SOA_ISUM 38.07 tok/s, PT 43.34 / 43.22 (+13.7%). GGML_CUDA_BATCH_INVARIANT=1 output is byte-identical before and after the change. --- ggml/src/ggml-cuda/common.cuh | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/ggml/src/ggml-cuda/common.cuh b/ggml/src/ggml-cuda/common.cuh index 7e133e3d6398..aba0c5914a2b 100644 --- a/ggml/src/ggml-cuda/common.cuh +++ b/ggml/src/ggml-cuda/common.cuh @@ -1283,7 +1283,8 @@ int ggml_cuda_get_device(); // Under GGML_CUDA_BATCH_INVARIANT the one-column case must run the same arithmetic as 2-8 // columns, so it takes the planar layout (the SoA vec-dot sums in a different order). // Ampere (sm_80/86, including the 3060/3090/170HX): the #218 PT kernel wins at one column -// too (+5.9% tg128 vs SoA on a 3060). Ada and newer keep SOA_ISUM at one column (4070 win). +// too (+5.9% tg128 vs SoA on a 3060). Turing (sm_75) as well (+13.7% decode on a 2060 SUPER). +// Ada and newer keep SOA_ISUM at one column (4070 win). // Lives here, after ggml_cuda_info(), because the body reads the current device's cc. static inline ggml_cuda_q8_1_layout ggml_cuda_q8_1_layout_host(ggml_type type_src0, int ncols_dst, bool has_ids) { const ggml_cuda_q8_1_layout l = ggml_cuda_q8_1_layout_for(type_src0, ncols_dst, has_ids); @@ -1292,7 +1293,7 @@ static inline ggml_cuda_q8_1_layout ggml_cuda_q8_1_layout_host(ggml_type type_sr } if (l == GGML_CUDA_Q8_1_SOA_ISUM) { const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc; - if (GGML_CUDA_CC_IS_NVIDIA(cc) && cc >= GGML_CUDA_CC_AMPERE && cc < GGML_CUDA_CC_ADA_LOVELACE) { + if (GGML_CUDA_CC_IS_NVIDIA(cc) && cc >= GGML_CUDA_CC_TURING && cc < GGML_CUDA_CC_ADA_LOVELACE) { return GGML_CUDA_Q8_1_PT; } }