From a55badae9f0e5a0d609127c0406e9a8d20ef9b83 Mon Sep 17 00:00:00 2001 From: sudoingX <200180104+sudoingX@users.noreply.github.com> Date: Sat, 19 Sep 2026 08:22:44 +0000 Subject: [PATCH 4/5] Add: GGML_CUDA_BATCH_INVARIANT for batch-invariant small-batch kernels Off by default, nothing changes. When set, kernels are picked so that the per-column arithmetic of a token does not depend on how many columns (1 to 8) share the launch: F16 and BF16 matrices with at most 512 rows take the mul_mat_vec_f kernel for 1 to 8 columns on Ampere and newer (its K partition depends on K only), flash attention with up to 8 queries takes the vector kernel on Turing and newer instead of switching to the tensor-core kernel at 2 queries, and the KV split (parallel_blocks) is sized as for one query tile so the partial-softmax combine order does not depend on the batch. Together with the batch-invariant PTQ1_0 mat-vec this makes the logits of a token bit-identical whether it is decoded alone or verified inside a speculative batch of 2, 3 or 4 tokens (measured on Ternary Bonsai 2 27B: greedy output with draft-mtp equals greedy output without it on three prompts at n-max 1, 2 and 3). Batches of 5 to 8 agree with each other but not with 1 to 4. Costs throughput at depth (the vector attention kernel at 3 queries over 42K keys). --- ggml/src/ggml-cuda/common.cuh | 8 ++++++++ ggml/src/ggml-cuda/fattn-common.cuh | 5 ++++- ggml/src/ggml-cuda/fattn.cu | 5 +++++ ggml/src/ggml-cuda/mmvf.cu | 6 ++++++ 4 files changed, 23 insertions(+), 1 deletion(-) diff --git a/ggml/src/ggml-cuda/common.cuh b/ggml/src/ggml-cuda/common.cuh index ef929d3d7..2600e6d86 100644 --- a/ggml/src/ggml-cuda/common.cuh +++ b/ggml/src/ggml-cuda/common.cuh @@ -173,6 +173,14 @@ static int ggml_cuda_highest_compiled_arch(const int arch) { // --------------------------------------------------------------------------------------------------------- +// GGML_CUDA_BATCH_INVARIANT=1: prefer kernels whose per-column arithmetic does not depend on the +// number of columns in the batch (1 to 8), so that a token decoded alone and a token verified +// inside a speculative batch see the same logits bit for bit. Costs some throughput at 2 to 8 columns. +static inline bool ggml_cuda_batch_invariant() { + static const bool enabled = getenv("GGML_CUDA_BATCH_INVARIANT") != nullptr; + return enabled; +} + #define MATRIX_ROW_PADDING 512 // last row of quant. matrices is a multiple of this to avoid out-of-bounds memory accesses #define GGML_CUDA_MAX_STREAMS 8 diff --git a/ggml/src/ggml-cuda/fattn-common.cuh b/ggml/src/ggml-cuda/fattn-common.cuh index e67cc7fdf..ae4318cbc 100644 --- a/ggml/src/ggml-cuda/fattn-common.cuh +++ b/ggml/src/ggml-cuda/fattn-common.cuh @@ -1154,11 +1154,14 @@ void launch_fattn( // If ntiles_total % blocks_per_wave != 0 then some efficiency is lost due to tail effects. // Test whether parallel_blocks can be set to a higher value for better efficiency. + // Batch-invariant mode: size the KV split as for a single query tile, so the order in which the + // partial softmax results are combined does not depend on how many queries are in the batch. + const int ntiles_dst_eff = ggml_cuda_batch_invariant() ? ntiles_dst / ntiles_x : ntiles_dst; const int blocks_per_wave = nsm * max_blocks_per_sm; int nwaves_best = 0; int efficiency_percent_best = 0; for (int parallel_blocks_test = parallel_blocks; parallel_blocks_test <= ntiles_KV; ++parallel_blocks_test) { - const int nblocks_total = ntiles_dst * parallel_blocks_test; + const int nblocks_total = ntiles_dst_eff * parallel_blocks_test; const int nwaves = (nblocks_total + blocks_per_wave - 1) / blocks_per_wave; const int efficiency_percent = 100 * nblocks_total / (nwaves*blocks_per_wave); diff --git a/ggml/src/ggml-cuda/fattn.cu b/ggml/src/ggml-cuda/fattn.cu index ab7a3b297..dacbe4884 100644 --- a/ggml/src/ggml-cuda/fattn.cu +++ b/ggml/src/ggml-cuda/fattn.cu @@ -460,6 +460,11 @@ static best_fattn_kernel ggml_cuda_get_best_fattn_kernel(const int device, const // If Turing tensor cores are available, use them: if (turing_mma_available(cc) && Q->ne[0] != 40 && Q->ne[0] != 72) { if (can_use_vector_kernel) { + // batch-invariant mode: the same (vector) kernel for 1 to 8 queries, so a token verified in a + // speculative batch attends with the same arithmetic as a token decoded alone + if (ggml_cuda_batch_invariant() && Q->ne[1] <= 8 && Q->ne[3] == 1) { + return BEST_FATTN_KERNEL_VEC; + } if (!ggml_is_quantized(K->type) && !ggml_is_quantized(V->type)) { if (cc >= GGML_CUDA_CC_ADA_LOVELACE && Q->ne[1] == 1 && Q->ne[3] == 1 && !(gqa_ratio > 4 && K->ne[1] >= 8192)) { return BEST_FATTN_KERNEL_VEC; diff --git a/ggml/src/ggml-cuda/mmvf.cu b/ggml/src/ggml-cuda/mmvf.cu index d7dbc8b99..9810f3edc 100644 --- a/ggml/src/ggml-cuda/mmvf.cu +++ b/ggml/src/ggml-cuda/mmvf.cu @@ -821,6 +821,9 @@ bool ggml_cuda_should_use_mmvf(enum ggml_type type, int cc, const int64_t * src0 if (GGML_CUDA_CC_IS_NVIDIA(cc)) { const bool src0_small = (src0_ne[1] <= 512 || src0_ne[2]*src0_ne[3] == 1); if (ampere_mma_available(cc)) { + if (ggml_cuda_batch_invariant()) { + return src0_small && ne11 <= MMVF_MAX_BATCH_SIZE; + } return src0_small && ne11 == 1; } if (cc >= GGML_CUDA_CC_ADA_LOVELACE) { @@ -847,6 +850,9 @@ bool ggml_cuda_should_use_mmvf(enum ggml_type type, int cc, const int64_t * src0 if (GGML_CUDA_CC_IS_NVIDIA(cc)) { const bool src0_small = (src0_ne[1] <= 512 || src0_ne[2]*src0_ne[3] == 1); if (ampere_mma_available(cc)) { + if (ggml_cuda_batch_invariant()) { + return src0_small && ne11 <= MMVF_MAX_BATCH_SIZE; + } return src0_small && ne11 == 1; } if (cc >= GGML_CUDA_CC_ADA_LOVELACE) { -- 2.34.1