Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 13 additions & 0 deletions ggml/src/ggml-cuda/common.cuh
Original file line number Diff line number Diff line change
Expand Up @@ -173,6 +173,19 @@ static int ggml_cuda_highest_compiled_arch(const int arch) {

// ---------------------------------------------------------------------------------------------------------

// GGML_CUDA_BATCH_INVARIANT=1: for batches of 1 to 4 columns, pick kernels whose per-column
// arithmetic does not depend on the column count on the paths this flag covers: the F16 and BF16
// mat-vec paths it selects, the PTQ1_0 mat-vec (1 to 4 columns; 5 and above take the MMQ tile
// path, which this flag does not touch), and flash attention up to 8 queries. On those paths a
// token decoded alone and a token verified inside a speculative batch see the same logits bit
// for bit. Other weight types and attention shapes outside those kernels can still pick
// batch-dependent kernels, so this is not a whole-model guarantee. Costs some throughput at
// 2 to 4 columns.
static inline bool ggml_cuda_batch_invariant() {
static const bool enabled = getenv("GGML_CUDA_BATCH_INVARIANT") != nullptr;
return enabled;
}

#define MATRIX_ROW_PADDING 512 // last row of quant. matrices is a multiple of this to avoid out-of-bounds memory accesses

#define GGML_CUDA_MAX_STREAMS 8
Expand Down
5 changes: 4 additions & 1 deletion ggml/src/ggml-cuda/fattn-common.cuh
Original file line number Diff line number Diff line change
Expand Up @@ -1154,11 +1154,14 @@ void launch_fattn(

// If ntiles_total % blocks_per_wave != 0 then some efficiency is lost due to tail effects.
// Test whether parallel_blocks can be set to a higher value for better efficiency.
// Batch-invariant mode: size the KV split as for a single query tile, so the order in which the
// partial softmax results are combined does not depend on how many queries are in the batch.
const int ntiles_dst_eff = ggml_cuda_batch_invariant() ? ntiles_dst / ntiles_x : ntiles_dst;
const int blocks_per_wave = nsm * max_blocks_per_sm;
int nwaves_best = 0;
int efficiency_percent_best = 0;
for (int parallel_blocks_test = parallel_blocks; parallel_blocks_test <= ntiles_KV; ++parallel_blocks_test) {
const int nblocks_total = ntiles_dst * parallel_blocks_test;
const int nblocks_total = ntiles_dst_eff * parallel_blocks_test;
const int nwaves = (nblocks_total + blocks_per_wave - 1) / blocks_per_wave;
const int efficiency_percent = 100 * nblocks_total / (nwaves*blocks_per_wave);

Expand Down
6 changes: 6 additions & 0 deletions ggml/src/ggml-cuda/fattn.cu
Original file line number Diff line number Diff line change
Expand Up @@ -460,6 +460,12 @@ static best_fattn_kernel ggml_cuda_get_best_fattn_kernel(const int device, const
// If Turing tensor cores are available, use them:
if (turing_mma_available(cc) && Q->ne[0] != 40 && Q->ne[0] != 72) {
if (can_use_vector_kernel) {
// batch-invariant mode: the same (vector) kernel for up to 8 queries, so a token verified in a
// speculative batch attends with the same arithmetic as a token decoded alone (the whole-model
// guarantee is 1 to 4 queries, see ggml_cuda_batch_invariant)
if (ggml_cuda_batch_invariant() && Q->ne[1] <= 8 && Q->ne[3] == 1) {
return BEST_FATTN_KERNEL_VEC;
}
if (!ggml_is_quantized(K->type) && !ggml_is_quantized(V->type)) {
if (cc >= GGML_CUDA_CC_ADA_LOVELACE && Q->ne[1] == 1 && Q->ne[3] == 1 && !(gqa_ratio > 4 && K->ne[1] >= 8192)) {
return BEST_FATTN_KERNEL_VEC;
Expand Down
8 changes: 8 additions & 0 deletions ggml/src/ggml-cuda/mmvf.cu
Original file line number Diff line number Diff line change
Expand Up @@ -821,6 +821,9 @@ bool ggml_cuda_should_use_mmvf(enum ggml_type type, int cc, const int64_t * src0
if (GGML_CUDA_CC_IS_NVIDIA(cc)) {
const bool src0_small = (src0_ne[1] <= 512 || src0_ne[2]*src0_ne[3] == 1);
if (ampere_mma_available(cc)) {
if (ggml_cuda_batch_invariant()) {
return src0_small && ne11 <= MMVF_MAX_BATCH_SIZE;
}
return src0_small && ne11 == 1;
}
if (cc >= GGML_CUDA_CC_ADA_LOVELACE) {
Expand All @@ -847,6 +850,11 @@ bool ggml_cuda_should_use_mmvf(enum ggml_type type, int cc, const int64_t * src0
if (GGML_CUDA_CC_IS_NVIDIA(cc)) {
const bool src0_small = (src0_ne[1] <= 512 || src0_ne[2]*src0_ne[3] == 1);
if (ampere_mma_available(cc)) {
// a few dozen rows (the qwen35 gated-delta-net gate projections) run faster as a
// mat-vec than through the tensor-core path at 2 to 8 columns: 3.4 vs 10.5 us on an RTX 3060
if (ggml_cuda_batch_invariant() || src0_ne[1] <= 64) {
return src0_small && ne11 <= MMVF_MAX_BATCH_SIZE;
}
return src0_small && ne11 == 1;
}
if (cc >= GGML_CUDA_CC_ADA_LOVELACE) {
Expand Down
Loading