CUDA: size routed MoE MMQ N-tiles from typical expert width on RDNA3 (#24546)
* adjust ncols_picker for routed MoE in mul_mat_q_case function * Adding CDNA, RDNA2 and RDNA4 * fix: update mmq_use_routed_moe_ncols_picker to include NVIDIA + Volta support * feat: enhance mmq configuration for various architectures with moe_ncols_min_cc support * refactor: replace moe_ncols_min_cc with use_typical_moe_ncols in mmq configuration files * HIP: mmq: enable typical moe ncols on RDNA4 --------- Co-authored-by: Carl Philipp Klemm <carl@uvos.xyz>
This commit is contained in:
co-authored by
Carl Philipp Klemm
parent
4735997382
commit
0c963452ea
@@ -1,4 +1,5 @@
|
||||
static constexpr __host__ __device__ ggml_cuda_mmq_config ggml_cuda_mmq_get_config_blackwell(ggml_type type, int J, bool fallback) {
|
||||
constexpr bool use_typical_moe_ncols = false;
|
||||
CASE(GGML_TYPE_MXFP4, 256, 1, 128, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_FP4, MMQ_ITER_K_FP4, true, true);
|
||||
CASE(GGML_TYPE_MXFP4, 256, 1, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_FP4, MMQ_ITER_K_FP4, true, true);
|
||||
CASE(GGML_TYPE_MXFP4, 256, 1, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_FP4, MMQ_ITER_K_FP4, true, true);
|
||||
|
||||
Reference in New Issue
Block a user