sycl: fuse mul_mat(gate) + mul_mat(up) + GLU for q4_K dense FFN (#26779)
Measured on Arc Pro B70 (Battlemage, Level Zero), llama-bench -r 20, two
interleaved rounds, tg128:
qwen2.5-3B-Instruct Q4_K_M 154.18 -> 158.53 t/s +2.8%
gemma-2-2b-it Q4_K_M 162.45 -> 165.62 t/s +2.0%
llama-batched-bench on qwen2.5-3B, S_TG by batch size:
B=1 142.72 -> 147.57 t/s +3.4%
B=2 243.72 -> 268.26 t/s +10.1%
B=4 359.58 -> 398.02 t/s +10.7%
B=8 449.75 -> 505.63 t/s +12.4%
This commit is contained in:
@@ -57,4 +57,20 @@ bool ggml_sycl_mul_mat_vec_q_id_reorder(
|
||||
size_t src1_row_stride,
|
||||
dpct::queue_ptr stream);
|
||||
|
||||
// Fused dense-FFN GEMV: writes glu(gate . y, up . y) instead of the two mat-vec results.
|
||||
// vx / vgate must share shape, stride and reorder layout. Returns false if unhandled.
|
||||
bool ggml_sycl_mul_mat_vec_q_glu_reorder(
|
||||
enum ggml_type src0_type,
|
||||
enum ggml_glu_op glu_op,
|
||||
const void * vx,
|
||||
const void * vgate,
|
||||
const void * vy,
|
||||
float * dst,
|
||||
int ncols, // K, shared by both weights
|
||||
int nrows, // output rows, i.e. weight ne[1]
|
||||
int ncols_dst, // activation columns, 1..MMVQ_MAX_BATCH_SIZE
|
||||
int stride_col_y_bytes, // bytes between activation columns in vy
|
||||
int stride_col_dst, // floats between output columns in dst
|
||||
dpct::queue_ptr stream);
|
||||
|
||||
#endif // GGML_SYCL_MMVQ_HPP
|
||||
|
||||
Reference in New Issue
Block a user