From f9a3042128db085921e939ab07e7dca26cb5a3df Mon Sep 17 00:00:00 2001 From: bri-prism <288398250+bri-prism@users.noreply.github.com> Date: Sun, 27 Sep 2026 17:15:12 -0700 Subject: [PATCH] sycl: pick the src1 fp16 conversion from the runtime oneDNN switch ggml_sycl_mul_mat_batched_sycl chose how to convert src1 to fp16 at compile time (#if GGML_SYCL_DNNL), but chose the GEMM at runtime (g_ggml_sycl_enable_dnn). In a oneDNN-enabled build run with GGML_SYCL_ENABLE_DNN=0, src1 was converted into the strided oneDNN layout, and s11/s12/s13 were then reset to contiguous strides for the MKL gemm_batch path. The fallback read a strided buffer as contiguous and gave wrong results whenever src1 was not contiguous. Gate the oneDNN conversion on the runtime switch and use the contiguous _nc conversion otherwise. test-backend-ops -o MUL_MAT -b SYCL0 on Arc B390 with GGML_SYCL_ENABLE_DNN=0: 1297/1329 -> 1329/1329. The 32 cases that failed were f16 x f32 with non-contiguous views (k_v != 0). The default oneDNN path is unchanged at 1329/1329. --- ggml/src/ggml-sycl/ggml-sycl.cpp | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/ggml/src/ggml-sycl/ggml-sycl.cpp b/ggml/src/ggml-sycl/ggml-sycl.cpp index 761b7f9c664d..b51214360feb 100644 --- a/ggml/src/ggml-sycl/ggml-sycl.cpp +++ b/ggml/src/ggml-sycl/ggml-sycl.cpp @@ -3540,6 +3540,7 @@ static void ggml_sycl_mul_mat_batched_sycl(ggml_backend_sycl_context & ctx, cons " : converting src1 to fp16"); #if GGML_SYCL_DNNL + if (g_ggml_sycl_enable_dnn) { // iterate tensor dims and find the slowest moving dim and stride int last_dim=0; int last_str=0; @@ -3565,13 +3566,15 @@ static void ggml_sycl_mul_mat_batched_sycl(ggml_backend_sycl_context & ctx, cons const to_fp16_sycl_t to_fp16_sycl = ggml_get_to_fp16_sycl(src1->type, dst); GGML_ASSERT(to_fp16_sycl != nullptr); to_fp16_sycl(src1_f16, src1_f16_alloc.get(), ne_src1, queue); -# else + } else +#endif + { const int64_t ne_src1 = ggml_nelements(src1); src1_f16_alloc.alloc(ne_src1); const to_fp16_nc_sycl_t to_fp16_nc_sycl = ggml_get_to_fp16_nc_sycl(src1->type); GGML_ASSERT(to_fp16_nc_sycl != nullptr); to_fp16_nc_sycl(src1_f16, src1_f16_alloc.get(), ne10, ne11, ne12, ne13, s11, s12, s13, queue); -#endif + } src1_f16 = src1_f16_alloc.get(); s11 = ne10;