From 73f09bd7beb9907af4d1edddca4e55c137f6a9f6 Mon Sep 17 00:00:00 2001 From: sb32445 Date: Fri, 2 Oct 2026 09:56:43 +0200 Subject: [PATCH 1/2] cuda: use the transposing concat kernel on all GPUs, not only on GB10 build_conv_state() concatenates the conv state [3, C] with a transposed [C, n_tokens] view for every Gated-Delta-Net layer. The shared-memory transpose kernel concat_dim0_transpose_u32 already exists but was enabled only for cc 12.1 (DGX Spark); other GPUs used the generic non-contiguous kernel with one block per output row and about 6 useful elements per block. RTX 4070 (cc 8.9), C=10240, f32: 2 tokens 8.8 -> 2.5 us, 3 tokens 8.8 -> 2.5 us per call (test-backend-ops, local test case). Bonsai 2 27B with MTP (n-max 2), 48 GDN layers: 104.4 -> 105.6 tok/s (+1.2 %, 3 alternating runs, outputs identical). No change for 1 token (contiguous case, other kernel). GGML_CUDA_CONCAT_TRANSPOSE=0 restores the old choice. test-backend-ops CONCAT passed on CUDA0 (195 cases incl. odd sizes C=7 and C=100, up to 33 tokens). Co-Authored-By: Claude Sonnet 5.5 --- ggml/src/ggml-cuda/concat.cu | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/ggml/src/ggml-cuda/concat.cu b/ggml/src/ggml-cuda/concat.cu index f59708892021..2b097f2c0277 100644 --- a/ggml/src/ggml-cuda/concat.cu +++ b/ggml/src/ggml-cuda/concat.cu @@ -1,4 +1,5 @@ #include "concat.cuh" +#include #include @@ -200,7 +201,8 @@ static void concat_cuda(const ggml_tensor * src0, const ggml_tensor * src1, ggml dim3 grid_dim(dst->ne[1], dst->ne[2], dst->ne[3]); if constexpr (sizeof(T) == sizeof(uint32_t)) { - const bool transpose_dim0 = ggml_cuda_info().devices[ggml_cuda_get_device()].cc == GGML_CUDA_CC_DGX_SPARK && + static const bool transpose_any_cc = [] { const char * e = getenv("GGML_CUDA_CONCAT_TRANSPOSE"); return e ? atoi(e) != 0 : true; }(); + const bool transpose_dim0 = (transpose_any_cc || ggml_cuda_info().devices[ggml_cuda_get_device()].cc == GGML_CUDA_CC_DGX_SPARK) && dim == 0 && src0->ne[2] == 1 && src0->ne[3] == 1 && src1->ne[2] == 1 && src1->ne[3] == 1 && dst->ne[2] == 1 && dst->ne[3] == 1 && src0->ne[0] <= 8 && src0->nb[0] == sizeof(uint32_t) && src0->nb[1] == (uint64_t) src0->ne[0]*sizeof(uint32_t) && From 97aec1b10907a5b83201d40c8d42850bb3ee9736 Mon Sep 17 00:00:00 2001 From: sb32445 Date: Sat, 3 Oct 2026 14:57:41 +0200 Subject: [PATCH 2/2] cuda: remove the GGML_CUDA_CONCAT_TRANSPOSE switch The environment switch of the previous commit was only there to measure the change; the transposing kernel is now used on all GPUs unconditionally. Co-Authored-By: Claude Sonnet 5.5 --- ggml/src/ggml-cuda/concat.cu | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/ggml/src/ggml-cuda/concat.cu b/ggml/src/ggml-cuda/concat.cu index 2b097f2c0277..49dc9b697fe3 100644 --- a/ggml/src/ggml-cuda/concat.cu +++ b/ggml/src/ggml-cuda/concat.cu @@ -1,5 +1,4 @@ #include "concat.cuh" -#include #include @@ -201,9 +200,7 @@ static void concat_cuda(const ggml_tensor * src0, const ggml_tensor * src1, ggml dim3 grid_dim(dst->ne[1], dst->ne[2], dst->ne[3]); if constexpr (sizeof(T) == sizeof(uint32_t)) { - static const bool transpose_any_cc = [] { const char * e = getenv("GGML_CUDA_CONCAT_TRANSPOSE"); return e ? atoi(e) != 0 : true; }(); - const bool transpose_dim0 = (transpose_any_cc || ggml_cuda_info().devices[ggml_cuda_get_device()].cc == GGML_CUDA_CC_DGX_SPARK) && - dim == 0 && src0->ne[2] == 1 && src0->ne[3] == 1 && src1->ne[2] == 1 && src1->ne[3] == 1 && + const bool transpose_dim0 = dim == 0 && src0->ne[2] == 1 && src0->ne[3] == 1 && src1->ne[2] == 1 && src1->ne[3] == 1 && dst->ne[2] == 1 && dst->ne[3] == 1 && src0->ne[0] <= 8 && src0->nb[0] == sizeof(uint32_t) && src0->nb[1] == (uint64_t) src0->ne[0]*sizeof(uint32_t) && src1->nb[1] == sizeof(uint32_t) && dst->nb[0] == sizeof(uint32_t) &&