* spec : add DFlash2 support (local convolution + candidate selector) (#27342) * support DFlash2 * Add p_min in DFlash2 Assisted-by: Claude Opus 5 * Revert unnecessary changes Assisted-by: Claude Opus 5 * Revert draft sampling in rejection sampling Assisted-by: Claude Opus 5 * Refactor code structure Assisted-by: Claude Opus 5 * Delete embedding scaling Assisted-by: Claude Opus 5 * Gate output transforms on DFlash2 Assisted-by: Claude Opus 5 * Optimize Dflash 2 cost Assisted-by: Claude Opus 5 * Avoid using atoi Assisted-by: Claude Opus 5 * Modify comments Assisted-by: Claude Opus 5 * Move llama_model_dflash_selector_top_k to llama-ext.h Assisted-by: Claude Opus 5 * Formatting Assisted-by: Claude Opus 5 * Apply patch to fix the mrope bug Assisted-by: Claude Opus 5 * fix ci Assisted-by: Claude Opus 5 * Fix graph number calculation Assisted-by: Claude Opus 5 * rename hid and unary Assisted-by: Claude Opus 5 --------- Co-authored-by: Jian Chen <jianchen0311@gmail.com> Co-authored-by: Xuan-Son Nguyen <son@huggingface.co> * revert top-k.cu changes --------- Co-authored-by: Zihan Zhang <tiancaizhangdaxian@sjtu.edu.cn> Co-authored-by: Jian Chen <jianchen0311@gmail.com>
This commit is contained in:
co-authored by
Jian Chen
Zihan Zhang
parent
58546250cf
commit
b10f9ca58c
@@ -344,6 +344,12 @@ static const std::map<llm_kv, const char *> LLM_KV_NAMES = {
|
||||
{ LLM_KV_NORM_BEFORE_RESIDUAL, "%s.norm_before_residual" },
|
||||
{ LLM_KV_NORM_BEFORE_FC, "%s.norm_before_fc" },
|
||||
|
||||
{ LLM_KV_DFLASH_BLOCK_SIZE, "%s.block_size" },
|
||||
{ LLM_KV_DFLASH_CONV_KERNEL_SIZE, "%s.conv_kernel_size" },
|
||||
{ LLM_KV_DFLASH_CONV_GROUP_SIZE, "%s.conv_group_size" },
|
||||
{ LLM_KV_DFLASH_SELECTOR_RANK, "%s.selector_rank" },
|
||||
{ LLM_KV_DFLASH_SELECTOR_TOP_K, "%s.selector_top_k" },
|
||||
|
||||
{ LLM_KV_SHORTCONV_L_CACHE, "%s.shortconv.l_cache" },
|
||||
// sentence-transformers dense modules feature dims
|
||||
{ LLM_KV_DENSE_2_FEAT_IN, "%s.dense_2_feat_in" },
|
||||
@@ -651,6 +657,13 @@ static const std::map<llm_tensor, const char *> LLM_TENSOR_NAMES = {
|
||||
{ LLM_TENSOR_DSPARK_MARKOV_W1, "markov_w1" },
|
||||
{ LLM_TENSOR_DSPARK_MARKOV_W2, "markov_w2" },
|
||||
{ LLM_TENSOR_DSPARK_CONF_PROJ, "conf_proj" },
|
||||
{ LLM_TENSOR_DFLASH_ATTN_CONV_BASE, "blk.%d.attn_conv_base" },
|
||||
{ LLM_TENSOR_DFLASH_ATTN_CONV_PROJ, "blk.%d.attn_conv_proj" },
|
||||
{ LLM_TENSOR_DFLASH_FFN_CONV_BASE, "blk.%d.ffn_conv_base" },
|
||||
{ LLM_TENSOR_DFLASH_FFN_CONV_PROJ, "blk.%d.ffn_conv_proj" },
|
||||
{ LLM_TENSOR_DFLASH_SELECTOR_PREV, "selector_predecessor" },
|
||||
{ LLM_TENSOR_DFLASH_SELECTOR_NEXT, "selector_successor" },
|
||||
{ LLM_TENSOR_DFLASH_SELECTOR_HIDDEN, "selector_hidden" },
|
||||
};
|
||||
|
||||
// declare information about the model weight tensors:
|
||||
@@ -916,6 +929,13 @@ static const std::map<llm_tensor, llm_tensor_info> LLM_TENSOR_INFOS = {
|
||||
{LLM_TENSOR_DSPARK_MARKOV_W1, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_GET_ROWS}},
|
||||
{LLM_TENSOR_DSPARK_MARKOV_W2, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_DSPARK_CONF_PROJ, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_DFLASH_ATTN_CONV_BASE, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_DFLASH_ATTN_CONV_PROJ, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_DFLASH_FFN_CONV_BASE, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_DFLASH_FFN_CONV_PROJ, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_DFLASH_SELECTOR_PREV, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_GET_ROWS}},
|
||||
{LLM_TENSOR_DFLASH_SELECTOR_NEXT, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_GET_ROWS}},
|
||||
{LLM_TENSOR_DFLASH_SELECTOR_HIDDEN, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
|
||||
};
|
||||
|
||||
LLM_KV::LLM_KV(llm_arch arch, const char * suffix) : arch(arch), suffix(suffix) {}
|
||||
|
||||
@@ -387,6 +387,11 @@ enum llm_kv {
|
||||
|
||||
LLM_KV_TARGET_LAYERS,
|
||||
LLM_KV_TARGET_HIDDEN_SIZE,
|
||||
LLM_KV_DFLASH_BLOCK_SIZE,
|
||||
LLM_KV_DFLASH_CONV_KERNEL_SIZE,
|
||||
LLM_KV_DFLASH_CONV_GROUP_SIZE,
|
||||
LLM_KV_DFLASH_SELECTOR_RANK,
|
||||
LLM_KV_DFLASH_SELECTOR_TOP_K,
|
||||
LLM_KV_NORM_BEFORE_RESIDUAL,
|
||||
LLM_KV_NORM_BEFORE_FC,
|
||||
|
||||
@@ -659,6 +664,13 @@ enum llm_tensor {
|
||||
LLM_TENSOR_DSPARK_MARKOV_W1,
|
||||
LLM_TENSOR_DSPARK_MARKOV_W2,
|
||||
LLM_TENSOR_DSPARK_CONF_PROJ,
|
||||
LLM_TENSOR_DFLASH_ATTN_CONV_BASE,
|
||||
LLM_TENSOR_DFLASH_ATTN_CONV_PROJ,
|
||||
LLM_TENSOR_DFLASH_FFN_CONV_BASE,
|
||||
LLM_TENSOR_DFLASH_FFN_CONV_PROJ,
|
||||
LLM_TENSOR_DFLASH_SELECTOR_PREV,
|
||||
LLM_TENSOR_DFLASH_SELECTOR_NEXT,
|
||||
LLM_TENSOR_DFLASH_SELECTOR_HIDDEN,
|
||||
};
|
||||
|
||||
|
||||
|
||||
@@ -2307,6 +2307,10 @@ uint32_t llama_context::graph_max_nodes(uint32_t n_tokens) const {
|
||||
model.arch == LLM_ARCH_MINIMAX_01 ||
|
||||
model.arch == LLM_ARCH_MINIMAX_M3) {
|
||||
res = std::max<uint32_t>(n_tokens * 40, 32u * model.n_tensors());
|
||||
} else if (model.arch == LLM_ARCH_DFLASH && model.hparams.dflash_selector_rank > 0) {
|
||||
// DFlash2's convolutions and selector are shape work rather than matmuls,
|
||||
// so they cost ~8.6 nodes per tensor against ~5.9 for a plain DFlash draft
|
||||
res = std::max<uint32_t>(1024u, 12u*model.n_tensors());
|
||||
} else {
|
||||
res = std::max<uint32_t>(1024u, 8u*model.n_tensors());
|
||||
for (const auto & lora : model.loras) {
|
||||
|
||||
@@ -120,6 +120,8 @@ LLAMA_API llama_context * llama_get_ctx_other(struct llama_context * ctx);
|
||||
// model/context data extraction
|
||||
//
|
||||
|
||||
LLAMA_API int32_t llama_model_dflash_selector_top_k(const struct llama_model * model);
|
||||
|
||||
// returns pointer to the target-model layer indices
|
||||
LLAMA_API const int32_t * llama_model_target_layer_ids (const struct llama_model * model);
|
||||
// returns the number of extracted layers from target model
|
||||
|
||||
@@ -223,6 +223,12 @@ struct llama_hparams {
|
||||
// output embedding dimension (0 = use n_embd)
|
||||
uint32_t n_embd_out_impl = 0;
|
||||
|
||||
uint32_t dflash_block_size = 0;
|
||||
uint32_t dflash_conv_kernel_size = 0;
|
||||
uint32_t dflash_conv_group_size = 0;
|
||||
uint32_t dflash_selector_rank = 0;
|
||||
uint32_t dflash_selector_top_k = 0;
|
||||
|
||||
// llama4 smallthinker
|
||||
uint32_t n_moe_layer_step = 0;
|
||||
uint32_t n_no_rope_layer_step = 4;
|
||||
|
||||
@@ -2684,6 +2684,10 @@ int32_t llama_model_n_layer_nextn(const llama_model * model) {
|
||||
return model->hparams.n_layer_nextn;
|
||||
}
|
||||
|
||||
int32_t llama_model_dflash_selector_top_k(const llama_model * model) {
|
||||
return model->hparams.dflash_selector_top_k;
|
||||
}
|
||||
|
||||
int32_t llama_model_n_head(const llama_model * model) {
|
||||
return model->hparams.n_head();
|
||||
}
|
||||
@@ -2882,6 +2886,10 @@ llama_rope_type llama_model_rope_type(const llama_model * model) {
|
||||
return LLAMA_ROPE_TYPE_NEOX;
|
||||
|
||||
case LLM_ARCH_DFLASH:
|
||||
// drafts for M-RoPE targets carry rope sections and follow the target's temporal dim
|
||||
if (const auto & s = model->hparams.rope_sections; s[0] || s[1] || s[2] || s[3]) {
|
||||
return LLAMA_ROPE_TYPE_MROPE;
|
||||
}
|
||||
// DSV4 DSpark drafters use DeepSeek-V4's normal RoPE; legacy DFlash backbones are NeoX
|
||||
return model->hparams.dsv4_hc_mult > 0 ? LLAMA_ROPE_TYPE_NORM : LLAMA_ROPE_TYPE_NEOX;
|
||||
|
||||
|
||||
@@ -363,6 +363,11 @@ struct llama_layer {
|
||||
struct ggml_tensor * ffn_exp_probs_b = nullptr;
|
||||
struct ggml_tensor * ffn_gate_tid2eid = nullptr;
|
||||
|
||||
struct ggml_tensor * dflash_attn_conv_base = nullptr;
|
||||
struct ggml_tensor * dflash_attn_conv_proj = nullptr;
|
||||
struct ggml_tensor * dflash_ffn_conv_base = nullptr;
|
||||
struct ggml_tensor * dflash_ffn_conv_proj = nullptr;
|
||||
|
||||
// mamba proj
|
||||
struct ggml_tensor * ssm_in = nullptr;
|
||||
struct ggml_tensor * ssm_x = nullptr;
|
||||
@@ -649,6 +654,10 @@ struct llama_model {
|
||||
struct ggml_tensor * dspark_conf_proj = nullptr;
|
||||
struct ggml_tensor * dspark_conf_proj_b = nullptr;
|
||||
|
||||
struct ggml_tensor * dflash_selector_prev = nullptr;
|
||||
struct ggml_tensor * dflash_selector_next = nullptr;
|
||||
struct ggml_tensor * dflash_selector_hidden = nullptr;
|
||||
|
||||
// unified vector to store target-model extracted layer ids in eagle3, dflash, etc.
|
||||
std::vector<int32_t> target_layer_ids;
|
||||
|
||||
|
||||
+265
-15
@@ -7,6 +7,18 @@
|
||||
void llama_model_dflash::load_arch_hparams(llama_model_loader & ml) {
|
||||
|
||||
ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps);
|
||||
ml.get_key(LLM_KV_LOGIT_SCALE, hparams.f_logit_scale, false);
|
||||
hparams.f_final_logit_softcapping = 0.0f;
|
||||
ml.get_key(LLM_KV_FINAL_LOGIT_SOFTCAPPING, hparams.f_final_logit_softcapping, false);
|
||||
|
||||
// drafts for M-RoPE targets carry degenerate sections [n_rot/2, 0, 0, 0]
|
||||
ml.get_key_or_arr(LLM_KV_ROPE_DIMENSION_SECTIONS, hparams.rope_sections, 4, false);
|
||||
|
||||
ml.get_key(LLM_KV_DFLASH_BLOCK_SIZE, hparams.dflash_block_size, false);
|
||||
ml.get_key(LLM_KV_DFLASH_CONV_KERNEL_SIZE, hparams.dflash_conv_kernel_size, false);
|
||||
ml.get_key(LLM_KV_DFLASH_CONV_GROUP_SIZE, hparams.dflash_conv_group_size, false);
|
||||
ml.get_key(LLM_KV_DFLASH_SELECTOR_RANK, hparams.dflash_selector_rank, false);
|
||||
ml.get_key(LLM_KV_DFLASH_SELECTOR_TOP_K, hparams.dflash_selector_top_k, false);
|
||||
|
||||
if (!ml.get_arr(LLM_KV_TARGET_LAYERS, target_layer_ids, false)) {
|
||||
throw std::runtime_error("DFlash model requires 'target_layers' in GGUF metadata");
|
||||
@@ -112,6 +124,29 @@ void llama_model_dflash::load_arch_tensors(llama_model_loader &) {
|
||||
LLAMA_LOG_INFO("%s: DFlash with DSpark markov head (rank = %lld)\n", __func__, (long long) dspark_markov_rank);
|
||||
}
|
||||
|
||||
const struct ggml_tensor * selector_meta = ml->get_tensor_meta("selector_hidden.weight");
|
||||
if (selector_meta) {
|
||||
const int64_t rank = hparams.dflash_selector_rank;
|
||||
if (rank <= 0 || hparams.dflash_block_size <= 0 || hparams.dflash_selector_top_k <= 0 ||
|
||||
hparams.dflash_conv_kernel_size <= 0 || hparams.dflash_conv_group_size <= 0) {
|
||||
throw std::runtime_error("DFlash2 model is missing conv/selector metadata");
|
||||
}
|
||||
if (n_embd % hparams.dflash_conv_group_size != 0) {
|
||||
throw std::runtime_error("DFlash2 hidden size must be divisible by conv_group_size");
|
||||
}
|
||||
if (n_embd < hparams.dflash_selector_top_k * (hparams.dflash_selector_top_k + 1)) {
|
||||
throw std::runtime_error("DFlash2 hidden size is too small for the selector lattice");
|
||||
}
|
||||
|
||||
dflash_selector_prev = create_tensor(tn(LLM_TENSOR_DFLASH_SELECTOR_PREV, "weight"), { rank, n_vocab }, 0);
|
||||
dflash_selector_next = create_tensor(tn(LLM_TENSOR_DFLASH_SELECTOR_NEXT, "weight"), { rank, n_vocab }, 0);
|
||||
dflash_selector_hidden = create_tensor(tn(LLM_TENSOR_DFLASH_SELECTOR_HIDDEN, "weight"), { n_embd, rank }, 0);
|
||||
|
||||
LLAMA_LOG_INFO("%s: DFlash2 conv kernel = %u, group = %u, selector rank = %u, top-k = %u\n", __func__,
|
||||
hparams.dflash_conv_kernel_size, hparams.dflash_conv_group_size,
|
||||
hparams.dflash_selector_rank, hparams.dflash_selector_top_k);
|
||||
}
|
||||
|
||||
fc = create_tensor(tn(LLM_TENSOR_FC, "weight"), { n_embd_inp, n_embd }, 0);
|
||||
fc_s = create_tensor(tn(LLM_TENSOR_FC, "scale"), { 1 }, TENSOR_NOT_REQUIRED);
|
||||
output_norm_enc = create_tensor(tn(LLM_TENSOR_ENC_OUTPUT_NORM, "weight"), { n_embd }, 0); // encoder hidden_norm (after fc)
|
||||
@@ -188,6 +223,16 @@ void llama_model_dflash::load_arch_tensors(llama_model_loader &) {
|
||||
layer.ffn_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "weight", i), { n_embd, n_ff }, 0);
|
||||
layer.ffn_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "weight", i), { n_ff, n_embd }, 0);
|
||||
layer.ffn_up = create_tensor(tn(LLM_TENSOR_FFN_UP, "weight", i), { n_embd, n_ff }, 0);
|
||||
|
||||
if (selector_meta) {
|
||||
const int64_t kernel = hparams.dflash_conv_kernel_size;
|
||||
const int64_t groups = n_embd / hparams.dflash_conv_group_size;
|
||||
const int64_t projected = 2 * kernel * groups;
|
||||
layer.dflash_attn_conv_base = create_tensor(tn(LLM_TENSOR_DFLASH_ATTN_CONV_BASE, i), { n_embd, kernel, 2 }, 0);
|
||||
layer.dflash_attn_conv_proj = create_tensor(tn(LLM_TENSOR_DFLASH_ATTN_CONV_PROJ, "weight", i), { n_embd, projected }, 0);
|
||||
layer.dflash_ffn_conv_base = create_tensor(tn(LLM_TENSOR_DFLASH_FFN_CONV_BASE, i), { n_embd, kernel, 2 }, 0);
|
||||
layer.dflash_ffn_conv_proj = create_tensor(tn(LLM_TENSOR_DFLASH_FFN_CONV_PROJ, "weight", i), { n_embd, projected }, 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -346,6 +391,167 @@ static void build_dspark_markov_head(llm_graph_context & g, const llama_model &
|
||||
ggml_build_forward_expand(g.gf, out);
|
||||
}
|
||||
|
||||
static ggml_tensor * build_dflash2_conv(
|
||||
llm_graph_context & g,
|
||||
ggml_tensor * hidden,
|
||||
ggml_tensor * dynamic,
|
||||
ggml_tensor * base,
|
||||
int side) {
|
||||
const auto & hparams = g.hparams;
|
||||
const int64_t hidden_size = hidden->ne[0];
|
||||
const int64_t n_tokens = hidden->ne[1];
|
||||
const int64_t n_blocks = g.ubatch.n_seqs_unq;
|
||||
const int64_t kernel_size = hparams.dflash_conv_kernel_size;
|
||||
const int64_t group_size = hparams.dflash_conv_group_size;
|
||||
const int64_t n_groups = hidden_size / group_size;
|
||||
|
||||
GGML_ASSERT(n_blocks > 0 && n_tokens % n_blocks == 0);
|
||||
GGML_ASSERT(dynamic && base && side >= 0 && side < 2);
|
||||
|
||||
const int64_t block_size = n_tokens / n_blocks;
|
||||
ggml_context * ctx0 = g.ctx0;
|
||||
// ggml_cont copies even when the tensor is already contiguous
|
||||
if (!ggml_is_contiguous(hidden) || hidden->ne[1] != n_tokens) {
|
||||
hidden = ggml_cont_2d(ctx0, hidden, hidden_size, n_tokens);
|
||||
}
|
||||
if (!ggml_is_contiguous(dynamic) || dynamic->ne[1] != n_tokens) {
|
||||
dynamic = ggml_cont_2d(ctx0, dynamic, dynamic->ne[0], n_tokens);
|
||||
}
|
||||
ggml_tensor * blocks = ggml_reshape_3d(ctx0, hidden, hidden_size, block_size, n_blocks);
|
||||
ggml_tensor * coeffs = ggml_reshape_4d(ctx0, dynamic, n_groups, kernel_size, 2, n_tokens);
|
||||
ggml_tensor * coeffs_side = ggml_view_3d(ctx0, coeffs, n_groups, kernel_size, n_tokens,
|
||||
coeffs->nb[1], coeffs->nb[3], side * coeffs->nb[2]);
|
||||
|
||||
ggml_tensor * coeff_all = ggml_cont(ctx0, coeffs_side);
|
||||
coeff_all = ggml_reshape_4d(ctx0, coeff_all, 1, n_groups, kernel_size, n_tokens);
|
||||
coeff_all = ggml_repeat_4d(ctx0, coeff_all, group_size, n_groups, kernel_size, n_tokens);
|
||||
|
||||
ggml_tensor * base_side = ggml_reshape_4d(ctx0,
|
||||
ggml_view_1d(ctx0, base, hidden_size * kernel_size, side * base->nb[2]),
|
||||
group_size, n_groups, kernel_size, 1);
|
||||
|
||||
ggml_tensor * weight_all = ggml_add(ctx0, coeff_all, base_side);
|
||||
|
||||
ggml_tensor * result = nullptr;
|
||||
for (int64_t tap = 0; tap < kernel_size; ++tap) {
|
||||
ggml_tensor * values = blocks;
|
||||
if (tap > 0) {
|
||||
ggml_tensor * zeros = ggml_fill(ctx0,
|
||||
ggml_new_tensor_3d(ctx0, hidden->type, hidden_size, std::min(tap, block_size), n_blocks), 0.0f);
|
||||
if (tap < block_size) {
|
||||
ggml_tensor * previous = ggml_view_3d(ctx0, blocks, hidden_size, block_size - tap, n_blocks,
|
||||
blocks->nb[1], blocks->nb[2], 0);
|
||||
values = ggml_concat(ctx0, zeros, previous, 1);
|
||||
} else {
|
||||
values = zeros;
|
||||
}
|
||||
}
|
||||
values = ggml_reshape_2d(ctx0, values, hidden_size, n_tokens);
|
||||
|
||||
ggml_tensor * weight = ggml_reshape_2d(ctx0,
|
||||
ggml_cont(ctx0, ggml_view_4d(ctx0, weight_all, group_size, n_groups, 1, n_tokens,
|
||||
weight_all->nb[1], weight_all->nb[2], weight_all->nb[3], tap * weight_all->nb[2])),
|
||||
hidden_size, n_tokens);
|
||||
|
||||
ggml_tensor * term = ggml_mul(ctx0, weight, values);
|
||||
result = result ? ggml_add(ctx0, result, term) : term;
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
// DFlash2 selector: top-k candidates per block position plus the pairwise
|
||||
// transition scores, packed into the nextn output slot for the CPU-side walk.
|
||||
static void build_dflash2_selector(llm_graph_context & g, const llama_model & model, ggml_tensor * tokens) {
|
||||
ggml_context * ctx0 = g.ctx0;
|
||||
auto & res = g.res;
|
||||
|
||||
const auto & hparams = g.hparams;
|
||||
const int64_t n_tokens = g.n_tokens;
|
||||
const int64_t n_embd = g.n_embd;
|
||||
|
||||
const int64_t top_k = hparams.dflash_selector_top_k;
|
||||
const int64_t rank = hparams.dflash_selector_rank;
|
||||
const int64_t n_blocks = g.ubatch.n_seqs_unq;
|
||||
GGML_ASSERT(n_blocks > 0 && n_tokens % n_blocks == 0);
|
||||
GGML_ASSERT(res->t_logits->ne[1] == n_tokens);
|
||||
if (!tokens) {
|
||||
return;
|
||||
}
|
||||
|
||||
const int64_t tokens_per_block = n_tokens / n_blocks;
|
||||
const int64_t block_size = std::min<int64_t>(tokens_per_block, hparams.dflash_block_size);
|
||||
const int64_t row_used = top_k + top_k * top_k;
|
||||
|
||||
ggml_tensor * candidates = ggml_top_k(ctx0, res->t_logits, top_k);
|
||||
ggml_tensor * logits_rows = ggml_reshape_3d(ctx0, res->t_logits, 1, res->t_logits->ne[0], n_tokens);
|
||||
ggml_tensor * unary = ggml_reshape_2d(ctx0,
|
||||
ggml_get_rows(ctx0, logits_rows, candidates), top_k, n_tokens);
|
||||
ggml_tensor * gate = g.build_lora_mm(model.dflash_selector_hidden, res->t_embd);
|
||||
|
||||
// Everything below indexes [.., tokens_per_block, n_blocks]: the block
|
||||
// position varies fastest, sequences are the outer dimension.
|
||||
ggml_tensor * cand_blk = ggml_reshape_3d(ctx0, candidates, top_k, tokens_per_block, n_blocks);
|
||||
ggml_tensor * unary_blk = ggml_reshape_3d(ctx0, unary, top_k, tokens_per_block, n_blocks);
|
||||
ggml_tensor * gate_blk = ggml_reshape_3d(ctx0, gate, rank, tokens_per_block, n_blocks);
|
||||
|
||||
// a position's score reads only the candidate sets at pos-1 and pos, so a run
|
||||
// of positions has no internal dependency and scores in one batched matmul
|
||||
auto score_run = [&](int64_t beg_pos, int64_t n_pos, ggml_tensor * pred_ids) {
|
||||
ggml_tensor * cand_run = ggml_cont(ctx0, ggml_view_3d(ctx0, cand_blk, top_k, n_pos, n_blocks,
|
||||
cand_blk->nb[1], cand_blk->nb[2], beg_pos * cand_blk->nb[1]));
|
||||
ggml_tensor * unary_run = ggml_cont(ctx0, ggml_view_3d(ctx0, unary_blk, top_k, n_pos, n_blocks,
|
||||
unary_blk->nb[1], unary_blk->nb[2], beg_pos * unary_blk->nb[1]));
|
||||
ggml_tensor * gate_run = ggml_cont(ctx0, ggml_view_3d(ctx0, gate_blk, rank, n_pos, n_blocks,
|
||||
gate_blk->nb[1], gate_blk->nb[2], beg_pos * gate_blk->nb[1]));
|
||||
|
||||
const int64_t n_pred = pred_ids->ne[0] / (n_pos * n_blocks);
|
||||
|
||||
ggml_tensor * successor = ggml_reshape_4d(ctx0,
|
||||
ggml_get_rows(ctx0, model.dflash_selector_next, ggml_reshape_1d(ctx0, cand_run, top_k * n_pos * n_blocks)),
|
||||
rank, top_k, n_pos, n_blocks);
|
||||
ggml_tensor * predecessor = ggml_reshape_4d(ctx0,
|
||||
ggml_get_rows(ctx0, model.dflash_selector_prev, pred_ids),
|
||||
rank, n_pred, n_pos, n_blocks);
|
||||
|
||||
ggml_tensor * gate_bcast = ggml_reshape_4d(ctx0, gate_run, rank, 1, n_pos, n_blocks);
|
||||
ggml_tensor * cond = ggml_mul(ctx0, predecessor, ggml_repeat(ctx0, gate_bcast, predecessor));
|
||||
ggml_tensor * score = ggml_mul_mat(ctx0, successor, cond);
|
||||
if (n_pred == 1) {
|
||||
score = ggml_repeat_4d(ctx0, score, top_k, top_k, n_pos, n_blocks);
|
||||
}
|
||||
ggml_tensor * unary_bcast = ggml_reshape_4d(ctx0, unary_run, top_k, 1, n_pos, n_blocks);
|
||||
score = ggml_add(ctx0, score, ggml_repeat(ctx0, unary_bcast, score));
|
||||
|
||||
ggml_tensor * row = ggml_concat(ctx0,
|
||||
ggml_cast(ctx0, cand_run, GGML_TYPE_F32),
|
||||
ggml_reshape_3d(ctx0, score, top_k * top_k, n_pos, n_blocks), 0);
|
||||
return ggml_pad(ctx0, row, n_embd - row_used, 0, 0, 0);
|
||||
};
|
||||
|
||||
ggml_tensor * packed = ggml_fill(ctx0,
|
||||
ggml_new_tensor_3d(ctx0, GGML_TYPE_F32, n_embd, 1, n_blocks), 0.0f);
|
||||
|
||||
if (block_size > 1) {
|
||||
// Position 1 alone: its predecessor is the anchor token, one id per
|
||||
// sequence rather than a candidate set.
|
||||
ggml_tensor * anchor_ids = ggml_cont_1d(ctx0,
|
||||
ggml_view_2d(ctx0, tokens, 1, n_blocks, tokens_per_block * tokens->nb[0], 0), n_blocks);
|
||||
packed = ggml_concat(ctx0, packed, score_run(1, 1, anchor_ids), 1);
|
||||
}
|
||||
if (block_size > 2) {
|
||||
ggml_tensor * prev_ids = ggml_reshape_1d(ctx0,
|
||||
ggml_cont(ctx0, ggml_view_3d(ctx0, cand_blk, top_k, block_size - 2, n_blocks,
|
||||
cand_blk->nb[1], cand_blk->nb[2], cand_blk->nb[1])),
|
||||
top_k * (block_size - 2) * n_blocks);
|
||||
packed = ggml_concat(ctx0, packed, score_run(2, block_size - 2, prev_ids), 1);
|
||||
}
|
||||
|
||||
packed = ggml_reshape_2d(ctx0, packed, n_embd, block_size * n_blocks);
|
||||
g.cb(packed, "dflash2_lattice", -1);
|
||||
res->t_h_nextn = packed;
|
||||
ggml_build_forward_expand(g.gf, packed);
|
||||
}
|
||||
|
||||
// DFlash decoder, dual-mode by batch type:
|
||||
// * embd batch -> fused target features: project + inject K/V into the cache.
|
||||
// * token batch -> noise-block diffusion: attend over [committed, MASK...] to generate draft tokens
|
||||
@@ -370,6 +576,20 @@ llama_model_dflash::graph<false>::graph(const llama_model & model, const llm_gra
|
||||
|
||||
const float kq_scale = 1.0f/sqrtf(float(n_embd_head));
|
||||
|
||||
// drafts for M-RoPE targets use degenerate sections (temporal dim only)
|
||||
int sections[4];
|
||||
std::copy(std::begin(hparams.rope_sections), std::begin(hparams.rope_sections) + 4, sections);
|
||||
|
||||
auto build_rope = [&](ggml_tensor * cur, ggml_tensor * pos) {
|
||||
return rope_type == GGML_ROPE_TYPE_MROPE
|
||||
? ggml_rope_multi(ctx0, cur, pos, nullptr,
|
||||
n_rot, sections, rope_type, n_ctx_orig, freq_base, freq_scale,
|
||||
ext_factor, attn_factor, beta_fast, beta_slow)
|
||||
: ggml_rope_ext(ctx0, cur, pos, nullptr,
|
||||
n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,
|
||||
ext_factor, attn_factor, beta_fast, beta_slow);
|
||||
};
|
||||
|
||||
// KV cache injection
|
||||
if (ubatch.embd) {
|
||||
auto inp = std::make_unique<llm_graph_input_embd>(n_embd);
|
||||
@@ -392,11 +612,7 @@ llama_model_dflash::graph<false>::graph(const llama_model & model, const llm_gra
|
||||
Vcur = ggml_reshape_3d(ctx0, Vcur, n_embd_head, n_head_kv, n_tokens);
|
||||
|
||||
Kcur = build_norm(Kcur, layer.attn_k_norm, NULL, LLM_NORM_RMS, il);
|
||||
Kcur = ggml_rope_ext(
|
||||
ctx0, Kcur, inp_pos, nullptr,
|
||||
n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,
|
||||
ext_factor, attn_factor, beta_fast, beta_slow
|
||||
);
|
||||
Kcur = build_rope(Kcur, inp_pos);
|
||||
cb(Kcur, "Kcur_injected", il);
|
||||
cb(Vcur, "Vcur_injected", il);
|
||||
|
||||
@@ -450,6 +666,7 @@ llama_model_dflash::graph<false>::graph(const llama_model & model, const llm_gra
|
||||
|
||||
inp->tokens = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_tokens);
|
||||
ggml_set_input(inp->tokens);
|
||||
res->t_inp_tokens = inp->tokens;
|
||||
|
||||
ggml_tensor * inp_tokens = inp->tokens;
|
||||
|
||||
@@ -464,6 +681,13 @@ llama_model_dflash::graph<false>::graph(const llama_model & model, const llm_gra
|
||||
ggml_tensor * noise_norm = build_norm(inpL, layer.attn_norm, NULL, LLM_NORM_RMS, il);
|
||||
cb(noise_norm, "noise_norm", il);
|
||||
|
||||
ggml_tensor * attn_dynamic = nullptr;
|
||||
if (layer.dflash_attn_conv_proj) {
|
||||
attn_dynamic = build_lora_mm(layer.dflash_attn_conv_proj, noise_norm);
|
||||
noise_norm = build_dflash2_conv(*this, noise_norm, attn_dynamic, layer.dflash_attn_conv_base, 0);
|
||||
cb(noise_norm, "attn_conv_in", il);
|
||||
}
|
||||
|
||||
ggml_tensor * Qcur = build_lora_mm(layer.wq, noise_norm);
|
||||
ggml_tensor * Kcur = build_lora_mm(layer.wk, noise_norm);
|
||||
ggml_tensor * Vcur = build_lora_mm(layer.wv, noise_norm);
|
||||
@@ -475,16 +699,8 @@ llama_model_dflash::graph<false>::graph(const llama_model & model, const llm_gra
|
||||
Qcur = build_norm(Qcur, layer.attn_q_norm, NULL, LLM_NORM_RMS, il);
|
||||
Kcur = build_norm(Kcur, layer.attn_k_norm, NULL, LLM_NORM_RMS, il);
|
||||
|
||||
Qcur = ggml_rope_ext(
|
||||
ctx0, Qcur, inp_pos, nullptr,
|
||||
n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,
|
||||
ext_factor, attn_factor, beta_fast, beta_slow
|
||||
);
|
||||
Kcur = ggml_rope_ext(
|
||||
ctx0, Kcur, inp_pos, nullptr,
|
||||
n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,
|
||||
ext_factor, attn_factor, beta_fast, beta_slow
|
||||
);
|
||||
Qcur = build_rope(Qcur, inp_pos);
|
||||
Kcur = build_rope(Kcur, inp_pos);
|
||||
cb(Qcur, "Qcur", il);
|
||||
cb(Kcur, "Kcur", il);
|
||||
cb(Vcur, "Vcur", il);
|
||||
@@ -494,12 +710,24 @@ llama_model_dflash::graph<false>::graph(const llama_model & model, const llm_gra
|
||||
? build_attn(inp_attn_iswa, layer.wo, NULL, NULL, Qcur, Kcur, Vcur, nullptr, nullptr, nullptr, kq_scale, il)
|
||||
: build_attn(inp_attn, layer.wo, NULL, NULL, Qcur, Kcur, Vcur, nullptr, nullptr, nullptr, kq_scale, il);
|
||||
|
||||
if (attn_dynamic) {
|
||||
cur = build_dflash2_conv(*this, cur, attn_dynamic, layer.dflash_attn_conv_base, 1);
|
||||
cb(cur, "attn_conv_out", il);
|
||||
}
|
||||
|
||||
ggml_tensor * ffn_inp = ggml_add(ctx0, cur, inpL);
|
||||
cb(ffn_inp, "ffn_inp", il);
|
||||
|
||||
cur = build_norm(ffn_inp, layer.ffn_norm, NULL, LLM_NORM_RMS, il);
|
||||
cb(cur, "ffn_norm", il);
|
||||
|
||||
ggml_tensor * ffn_dynamic = nullptr;
|
||||
if (layer.dflash_ffn_conv_proj) {
|
||||
ffn_dynamic = build_lora_mm(layer.dflash_ffn_conv_proj, cur);
|
||||
cur = build_dflash2_conv(*this, cur, ffn_dynamic, layer.dflash_ffn_conv_base, 0);
|
||||
cb(cur, "ffn_conv_in", il);
|
||||
}
|
||||
|
||||
cur = build_ffn(cur,
|
||||
layer.ffn_up, NULL, layer.ffn_up_s,
|
||||
layer.ffn_gate, NULL, layer.ffn_gate_s,
|
||||
@@ -508,6 +736,11 @@ llama_model_dflash::graph<false>::graph(const llama_model & model, const llm_gra
|
||||
LLM_FFN_SILU, LLM_FFN_PAR, il);
|
||||
cb(cur, "ffn_out", il);
|
||||
|
||||
if (ffn_dynamic) {
|
||||
cur = build_dflash2_conv(*this, cur, ffn_dynamic, layer.dflash_ffn_conv_base, 1);
|
||||
cb(cur, "ffn_conv_out", il);
|
||||
}
|
||||
|
||||
cur = ggml_add(ctx0, cur, ffn_inp);
|
||||
cb(cur, "l_out", il);
|
||||
|
||||
@@ -532,6 +765,19 @@ llama_model_dflash::graph<false>::graph(const llama_model & model, const llm_gra
|
||||
|
||||
cur = build_lora_mm(output, cur, output_s);
|
||||
|
||||
// DFlash2 feeds these logits to the selector, so they need the target's output
|
||||
// transforms; DFlash1 and DSpark read them through the sampler instead
|
||||
if (model.dflash_selector_hidden) {
|
||||
if (hparams.f_logit_scale != 0.0f) {
|
||||
cur = ggml_scale(ctx0, cur, hparams.f_logit_scale);
|
||||
}
|
||||
if (hparams.f_final_logit_softcapping > 0.0f) {
|
||||
cur = ggml_scale(ctx0, cur, 1.0f / hparams.f_final_logit_softcapping);
|
||||
cur = ggml_tanh(ctx0, cur);
|
||||
cur = ggml_scale(ctx0, cur, hparams.f_final_logit_softcapping);
|
||||
}
|
||||
}
|
||||
|
||||
// reduced-draft-vocab exports: scatter the draft logits to the target vocabulary via d2t
|
||||
if (model.d2t) {
|
||||
const int64_t n_draft_vocab = cur->ne[0];
|
||||
@@ -556,6 +802,10 @@ llama_model_dflash::graph<false>::graph(const llama_model & model, const llm_gra
|
||||
if (model.dspark_markov_w1) {
|
||||
build_dspark_markov_head(*this, model, inp_tokens);
|
||||
}
|
||||
|
||||
if (model.dflash_selector_hidden) {
|
||||
build_dflash2_selector(*this, model, inp_tokens);
|
||||
}
|
||||
}
|
||||
|
||||
// DSV4 DSpark decoder, dual-mode by batch type (see the DFlash decoder above):
|
||||
|
||||
Reference in New Issue
Block a user