vulkan: Switch MUL_MAT_VEC to 4 K per iteration for F16/32 (#22887)
* vulkan: Switch MUL_MAT_VEC to 4 K per iteration for F16/32 Against mesa git, this shows a 4.8% performance improvement for tg128 on Qwen3.5-9B:BF16 on Intel BMG. Note that this breaks some tests until the last commit which fixes OOB A reads. * vulkan: Use aligned loads in mul_mat_vec when available Against mesa git, this shows a 3.3% performance improvement for tg128 on Qwen3.5-9B:BF16 on Intel BMG. * Make explicit that `num_rows` is <= `NUM_ROWS` in mul_mat_vec Mesa's UUB logic can't see through conditionals, limiting its ability to understand the bounds on the `num_rows` field in the cleanup run. Making it explicit that `num_rows` is, indeed, always <= `NUM_ROWS` helps mesa make slightly better codegen. Against mesa git, this currently shows a 1% performance improvement in tg128 on Qwen3.5-9B:BF16 on Intel BMG. * vulkan: Fix OOB A reads in MUL_MAT_VEC for odd sizes There was a TODO to fix the OOB reads from the A matrix which we do here. It is within performance noise (+<0.1%) in tg128 for Qwen3.5-9B:BF16 on Intel BMG.
This commit is contained in:
@@ -5,21 +5,60 @@
|
||||
#include "types.glsl"
|
||||
|
||||
#if defined(DATA_A_F32)
|
||||
FLOAT_TYPE dequantize1(uint ib, uint iqs, uint a_offset) {
|
||||
return data_a[a_offset + ib];
|
||||
}
|
||||
vec2 dequantize(uint ib, uint iqs, uint a_offset) {
|
||||
return vec2(data_a[a_offset + ib], data_a[a_offset + ib + 1]);
|
||||
}
|
||||
vec4 dequantize4(uint ib, uint iqs, uint a_offset) {
|
||||
return vec4(data_a[a_offset + ib ], data_a[a_offset + ib + 1],
|
||||
data_a[a_offset + ib + 2], data_a[a_offset + ib + 3]);
|
||||
}
|
||||
vec4 dequantize4_2aligned(uint ib, uint iqs, uint a_offset) {
|
||||
return vec4(data_a[a_offset + ib ], data_a[a_offset + ib + 1],
|
||||
data_a[a_offset + ib + 2], data_a[a_offset + ib + 3]);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
#if defined(DATA_A_F16)
|
||||
FLOAT_TYPE dequantize1(uint ib, uint iqs, uint a_offset) {
|
||||
return data_a[a_offset + ib];
|
||||
}
|
||||
vec2 dequantize(uint ib, uint iqs, uint a_offset) {
|
||||
return vec2(data_a[a_offset + ib], data_a[a_offset + ib + 1]);
|
||||
}
|
||||
vec4 dequantize4(uint ib, uint iqs, uint a_offset) {
|
||||
return vec4(data_a[a_offset + ib ], data_a[a_offset + ib + 1],
|
||||
data_a[a_offset + ib + 2], data_a[a_offset + ib + 3]);
|
||||
}
|
||||
vec4 dequantize4_2aligned(uint ib, uint iqs, uint a_offset) {
|
||||
const vec2 a = data_a_packed32[(a_offset + ib)/2];
|
||||
const vec2 b = data_a_packed32[(a_offset + ib)/2 + 1];
|
||||
return vec4(a, b);
|
||||
}
|
||||
#endif
|
||||
|
||||
#if defined(DATA_A_BF16)
|
||||
FLOAT_TYPE dequantize1(uint ib, uint iqs, uint a_offset) {
|
||||
return bf16_to_fp32(data_a[a_offset + ib]);
|
||||
}
|
||||
vec2 dequantize(uint ib, uint iqs, uint a_offset) {
|
||||
return vec2(bf16_to_fp32(data_a[a_offset + ib]), bf16_to_fp32(data_a[a_offset + ib + 1]));
|
||||
}
|
||||
vec4 dequantize4(uint ib, uint iqs, uint a_offset) {
|
||||
return vec4(bf16_to_fp32(data_a[a_offset + ib ]), bf16_to_fp32(data_a[a_offset + ib + 1]),
|
||||
bf16_to_fp32(data_a[a_offset + ib + 2]), bf16_to_fp32(data_a[a_offset + ib + 3]));
|
||||
}
|
||||
vec4 dequantize4_2aligned(uint ib, uint iqs, uint a_offset) {
|
||||
const uint a = data_a_packed32[(a_offset + ib)/2];
|
||||
const uint b = data_a_packed32[(a_offset + ib)/2 + 1];
|
||||
return vec4(uintBitsToFloat((a & 0x0000ffff) << 16),
|
||||
uintBitsToFloat( a & 0xffff0000),
|
||||
uintBitsToFloat((b & 0x0000ffff) << 16),
|
||||
uintBitsToFloat( b & 0xffff0000));
|
||||
}
|
||||
#endif
|
||||
|
||||
#if defined(DATA_A_Q4_0)
|
||||
|
||||
Reference in New Issue
Block a user