Measured on Arc Pro B70 (Battlemage, Level Zero), llama-bench -r 20, two
interleaved rounds, tg128:
qwen2.5-3B-Instruct Q4_K_M 154.18 -> 158.53 t/s +2.8%
gemma-2-2b-it Q4_K_M 162.45 -> 165.62 t/s +2.0%
llama-batched-bench on qwen2.5-3B, S_TG by batch size:
B=1 142.72 -> 147.57 t/s +3.4%
B=2 243.72 -> 268.26 t/s +10.1%
B=4 359.58 -> 398.02 t/s +10.7%
B=8 449.75 -> 505.63 t/s +12.4%
77 lines
3.2 KiB
C++
77 lines
3.2 KiB
C++
//
|
|
// MIT license
|
|
// Copyright (C) 2024 Intel Corporation
|
|
// SPDX-License-Identifier: MIT
|
|
//
|
|
|
|
//
|
|
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
|
// See https://llvm.org/LICENSE.txt for license information.
|
|
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
|
//
|
|
|
|
#ifndef GGML_SYCL_MMVQ_HPP
|
|
#define GGML_SYCL_MMVQ_HPP
|
|
|
|
#include "common.hpp"
|
|
|
|
|
|
void ggml_sycl_op_mul_mat_vec_q(
|
|
ggml_backend_sycl_context & ctx,
|
|
const ggml_tensor *src0, const ggml_tensor *src1, ggml_tensor *dst,
|
|
const char *src0_dd_i, const float *src1_ddf_i, const char *src1_ddq_i,
|
|
float *dst_dd_i, const int64_t row_low, const int64_t row_high,
|
|
const int64_t src1_ncols, const int64_t src1_padded_row_size,
|
|
const dpct::queue_ptr &stream);
|
|
|
|
// Requires standard (non-reorder) block layout for src0.
|
|
// Returns false if src0_type isn't handled; caller should fall back.
|
|
bool ggml_sycl_mul_mat_vec_q_id(
|
|
enum ggml_type src0_type,
|
|
const void * vx_base, // start of stacked expert weights
|
|
const void * vy, // pre-quantized src1 (Q8_1)
|
|
const int32_t * ids_dev, // device-side int32, length n_experts_used
|
|
float * dst_base,
|
|
int ncols,
|
|
int nrows,
|
|
int n_experts_used,
|
|
size_t expert_weight_stride, // bytes between experts in vx_base
|
|
size_t dst_row_stride, // bytes between dst rows
|
|
size_t src1_row_stride, // 0 = shared src1, else per-expert stride in bytes
|
|
dpct::queue_ptr stream);
|
|
|
|
// Reorder (SoA) variant of the fused MoE expert GEMV.
|
|
// vx_base: each expert slice (stride expert_weight_stride == src0->nb[2]) is a self-contained reorder/SoA layout.
|
|
// vy: src1 quantized with quantize_and_reorder_q8_1_soa (per-row SoA). Returns false if src0_type isn't handled.
|
|
bool ggml_sycl_mul_mat_vec_q_id_reorder(
|
|
enum ggml_type src0_type,
|
|
const void * vx_base,
|
|
const void * vy,
|
|
const int32_t * ids_dev,
|
|
float * dst_base,
|
|
int ncols,
|
|
int nrows,
|
|
int n_experts_used,
|
|
size_t expert_weight_stride,
|
|
size_t dst_row_stride,
|
|
size_t src1_row_stride,
|
|
dpct::queue_ptr stream);
|
|
|
|
// Fused dense-FFN GEMV: writes glu(gate . y, up . y) instead of the two mat-vec results.
|
|
// vx / vgate must share shape, stride and reorder layout. Returns false if unhandled.
|
|
bool ggml_sycl_mul_mat_vec_q_glu_reorder(
|
|
enum ggml_type src0_type,
|
|
enum ggml_glu_op glu_op,
|
|
const void * vx,
|
|
const void * vgate,
|
|
const void * vy,
|
|
float * dst,
|
|
int ncols, // K, shared by both weights
|
|
int nrows, // output rows, i.e. weight ne[1]
|
|
int ncols_dst, // activation columns, 1..MMVQ_MAX_BATCH_SIZE
|
|
int stride_col_y_bytes, // bytes between activation columns in vy
|
|
int stride_col_dst, // floats between output columns in dst
|
|
dpct::queue_ptr stream);
|
|
|
|
#endif // GGML_SYCL_MMVQ_HPP
|