CUDA: fix FA kernel selection logic (#21271)
This commit is contained in:
@@ -340,7 +340,14 @@ static best_fattn_kernel ggml_cuda_get_best_fattn_kernel(const int device, const
|
|||||||
case 128:
|
case 128:
|
||||||
case 112:
|
case 112:
|
||||||
case 256:
|
case 256:
|
||||||
|
if (V->ne[0] != K->ne[0]) {
|
||||||
|
return BEST_FATTN_KERNEL_NONE;
|
||||||
|
}
|
||||||
|
break;
|
||||||
case 512:
|
case 512:
|
||||||
|
if (V->ne[0] != K->ne[0]) {
|
||||||
|
return BEST_FATTN_KERNEL_NONE;
|
||||||
|
}
|
||||||
if (!gqa_opt_applies) {
|
if (!gqa_opt_applies) {
|
||||||
return BEST_FATTN_KERNEL_NONE;
|
return BEST_FATTN_KERNEL_NONE;
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user