* metal: dequantize q8_0 KV to f16 before flash attention Add a preprocessing pass for GGML_OP_FLASH_ATTN_EXT on the Metal backend: when the KV cache is quantized (Q8_0 for now), dequantize K and V into a contiguous F16 scratch buffer and run the existing F16 flash attention kernels on it, instead of the in-kernel dequantization path. - new kernel kernel_flash_attn_ext_dequant_to_f16<block_t, QK, deq_t4x4>: one thread per quant block (K then V), stride-aware so permuted KV is supported; instantiated for Q8_0 (extending to Q4_0/Q4_1/Q5_0/Q5_1 is one instantiation + one gate case) - the gate is type-only: dequantize whenever the KV is quantized, regardless of head sizes, GQA ratio or n_kv; the attention kernels themselves are untouched - the F16 copies live in the op's own scratch allocation (ggml_metal_op_flash_attn_ext_extra_dequant_f16); the KV pad kernel reads the dequantized buffers when the path is active - the FA pipeline getters gain a use_f16_kv flag selecting the existing f16 kernels and contiguous strides - ref: https://github.com/ggml-org/llama.cpp/pull/25556 Verification (M2 Ultra): - test-backend-ops test -o FLASH_ATTN_EXT: 4798/4798 pass, including the new q8_0 eval cases (decode/prompt, permuted, sinks+ALiBi+softcap, kv=113 pad path, kv=16384) - llama-perplexity on Qwen2.5-0.5B with -ctk q8_0 -ctv q8_0 matches the f16 KV reference (PPL 1.0008 vs 1.0008) Assisted-by: pi:llama.cpp/Qwen3.8-27B * metal : launch the FA KV dequant kernel separately for K and V Simplify kernel_flash_attn_ext_dequant_to_f16: it now dequantizes a single tensor (its own ne/nb and dst) with no is_v branching, and the op dispatches it twice with the same pipeline - once for K and once for V. The kargs struct shrinks to a single ne/nb set plus nblocks. Assisted-by: pi:llama.cpp/Qwen3.8-27B * metal : dequantize q4_0, q4_1, q5_0 and q5_1 KV to f16 before flash attention The dequant pass now covers all quantized KV types supported by the Metal flash attention kernels. The dequant kernel, kargs, scratch allocation and dispatch are type-generic, so each type is one kernel instantiation plus one gate case. Assisted-by: pi:llama.cpp/Qwen3.8-27B * metal : skip the redundant V dequant when V is a view of K In MLA-based models, the V of the FA op is a view of K (the first ne20 elements of each K row); the dequantized V is then a view of the dequantized K, so skip the second dequant dispatch, do not reserve the V scratch region, and let the pad and attention kernels read V from the K F16 buffer with K's strides. The detection follows the CUDA backend: V->view_src && (V->view_src == K || (V->view_src == K->view_src && V->view_offs == K->view_offs)) Also fix the FA pipeline getters: ns10/ns20 are function constants baked into the kernels and must be the actual K/V row widths as seen by the kernel. The dispatch now passes them explicitly (nb11_attn/nb10_attn, nb21_attn/nb20_attn) instead of the getters assuming contiguous F16 KV (ns20 = dv), which was wrong when V is read from K with K's row pitch (e.g. 576 vs 512). New test cases: 576/512 q8_0 (MLA shape, V is a view of K) at kv=113 (KV pad), nb=1 (vec) and nb=64 (non-vec). Assisted-by: pi:llama.cpp/Qwen3.8-27B * test : remove backend-specific wording from test-backend-ops comments Assisted-by: pi:llama.cpp/Qwen3.8-27B * pi : avoid backend mentions in test-backend-ops comments Assisted-by: pi:llama.cpp/Qwen3.8-27B * metal : rename the FA dequant_f16 identifiers to kv_f16 Assisted-by: pi:llama.cpp/Qwen3.8-27B * cont : clean-up * cont : remove TODO
106 lines
4.9 KiB
C
106 lines
4.9 KiB
C
#pragma once
|
|
|
|
#include "ggml-metal-device.h"
|
|
|
|
#ifdef __cplusplus
|
|
extern "C" {
|
|
#endif
|
|
|
|
typedef struct ggml_metal_op * ggml_metal_op_t;
|
|
|
|
ggml_metal_op_t ggml_metal_op_init(
|
|
ggml_metal_device_t dev,
|
|
ggml_metal_cmd_buf_t cmd_buf,
|
|
struct ggml_cgraph * gf,
|
|
int idx_start,
|
|
int idx_end,
|
|
bool use_fusion,
|
|
bool use_concurrency,
|
|
bool use_capture,
|
|
int debug_graph,
|
|
int debug_fusion);
|
|
|
|
void ggml_metal_op_free(ggml_metal_op_t ctx);
|
|
|
|
int ggml_metal_op_n_nodes(ggml_metal_op_t ctx);
|
|
|
|
int ggml_metal_op_encode(ggml_metal_op_t ctx, int idx);
|
|
|
|
//
|
|
// available ops:
|
|
//
|
|
|
|
// tokens per expert
|
|
size_t ggml_metal_op_mul_mat_id_extra_tpe(const struct ggml_tensor * op);
|
|
|
|
// id map [n_tokens, n_expert]
|
|
size_t ggml_metal_op_mul_mat_id_extra_ids(const struct ggml_tensor * op);
|
|
|
|
// return true if we should use the FA vector kernel for this op
|
|
bool ggml_metal_op_flash_attn_ext_use_vec(const struct ggml_tensor * op);
|
|
|
|
size_t ggml_metal_op_flash_attn_ext_extra_pad(const struct ggml_tensor * op);
|
|
size_t ggml_metal_op_flash_attn_ext_extra_blk(const struct ggml_tensor * op);
|
|
size_t ggml_metal_op_flash_attn_ext_extra_tmp(const struct ggml_tensor * op);
|
|
size_t ggml_metal_op_flash_attn_ext_extra_kv_f16(const struct ggml_tensor * op);
|
|
|
|
int ggml_metal_op_concat (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_repeat (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_acc (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_unary (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_glu (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_sum (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_sum_rows (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_cumsum (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_get_rows (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_set_rows (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_diag (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_lightning_indexer (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_dsv4_hc (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_soft_max (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_ssm_conv (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_ssm_scan (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_rwkv (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_gated_delta_net (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_solve_tri (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_set (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_cpy (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_pool_1d (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_pool_2d (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_fwht (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_mul_mat (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_mul_mat_id (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_add_id (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_flash_attn_ext (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_bin (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_silu_back (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_l2_norm (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_group_norm (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_norm (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_rope (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_im2col (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_conv_2d (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_conv_2d_dw (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_conv_3d (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_conv_transpose_1d (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_conv_transpose_2d (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_col2im_1d (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_snake_fused (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_upscale (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_pad (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_pad_reflect_1d (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_roll (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_arange (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_timestep_embedding(ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_argmax (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_argsort (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_top_k (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_tri (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_opt_step_adamw (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_opt_step_sgd (ggml_metal_op_t ctx, int idx);
|
|
int ggml_metal_op_count_equal (ggml_metal_op_t ctx, int idx);
|
|
|
|
#ifdef __cplusplus
|
|
}
|
|
#endif
|