From 0e640a180155fe91a15099fd872f886b6e346953 Mon Sep 17 00:00:00 2001 From: lingyezhixing <144504450+lingyezhixing@users.noreply.github.com> Date: Tue, 22 Sep 2026 00:11:29 +0800 Subject: [PATCH] cuda: fix sm_70 tile compilation error (llama/29224) The 5-argument load_ldmatrix added in 1884824fd only defines tile<16,8>, so the Volta tile<8,4> does not match. See https://github.com/ggml-org/llama.cpp/issues/29222 for details. Building on 1884824fd, generalize the tile shape of the 5-argument load_ldmatrix from <16,8> to , so the non-swizzle branch forwards to the 3-argument loader for any shape. Local compilation and testing passed. Assisted-by: DeepSeek V4.1 Flash (OpenCode) --- ggml/src/ggml-cuda/mma.cuh | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/ggml/src/ggml-cuda/mma.cuh b/ggml/src/ggml-cuda/mma.cuh index 6af2b6a14..3583ba5e1 100644 --- a/ggml/src/ggml-cuda/mma.cuh +++ b/ggml/src/ggml-cuda/mma.cuh @@ -873,14 +873,16 @@ namespace ggml_cuda_mma { } // Load from tile element (i0, j0), swz tells if the tile is stored swizzled. - template + template static __device__ __forceinline__ void load_ldmatrix( - tile<16, 8, T, dl> & t, const T * __restrict__ tile_base, const int i0, const int j0, const int stride) { + tile & t, const T * __restrict__ tile_base, const int i0, const int j0, const int stride) { if constexpr (!swz) { load_ldmatrix(t, tile_base + i0*stride + j0, stride); return; } #if defined(TURING_MMA_AVAILABLE) + static_assert(I == 16, "bad tile width"); + static_assert(J == 8, "bad tile height"); const int i = i0 + threadIdx.x % t.I; const int j = j0 + (threadIdx.x / t.I) * (t.J / 2); int * xi = (int *) t.x;