ggml : adjust logic for offloading ops to weight's backend (#25832)
* ggml : adjust logic for offloading ops to weight's backend * llama : dsv4 graph fixes
This commit is contained in:
@@ -906,14 +906,22 @@ static int ggml_backend_sched_backend_id_from_cur(ggml_backend_sched_t sched, st
|
|||||||
}
|
}
|
||||||
|
|
||||||
// operations with weights are preferably run on the same backend as the weights
|
// operations with weights are preferably run on the same backend as the weights
|
||||||
|
// TODO: there are exceptions (see below) - not an ideal solution
|
||||||
|
bool allow = true;
|
||||||
|
|
||||||
|
// skip ROPE since the rope freqs tensor is too small to choose a backend based on it
|
||||||
|
allow = allow && tensor->op != GGML_OP_ROPE;
|
||||||
|
|
||||||
|
// skip FLASH_ATTN_EXT since the sinks tensor is too small to choose a based based on it
|
||||||
|
allow = allow && tensor->op != GGML_OP_FLASH_ATTN_EXT;
|
||||||
|
|
||||||
|
if (allow) {
|
||||||
for (int i = 0; i < GGML_MAX_SRC; i++) {
|
for (int i = 0; i < GGML_MAX_SRC; i++) {
|
||||||
const struct ggml_tensor * src = tensor->src[i];
|
const struct ggml_tensor * src = tensor->src[i];
|
||||||
if (src == NULL) {
|
if (src == NULL) {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
// skip ROPE since the rope freqs tensor is too small to choose a backend based on it
|
if (src->buffer != NULL && src->buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS) {
|
||||||
// not an ideal solution
|
|
||||||
if (tensor->op != GGML_OP_ROPE && src->buffer != NULL && src->buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS) {
|
|
||||||
int src_backend_id = ggml_backend_sched_backend_from_buffer(sched, src, tensor);
|
int src_backend_id = ggml_backend_sched_backend_from_buffer(sched, src, tensor);
|
||||||
// check if a backend with higher prio wants to offload the op
|
// check if a backend with higher prio wants to offload the op
|
||||||
if (sched->op_offload && src_backend_id == sched->n_backends - 1 && ggml_backend_buffer_is_host(src->buffer)) {
|
if (sched->op_offload && src_backend_id == sched->n_backends - 1 && ggml_backend_buffer_is_host(src->buffer)) {
|
||||||
@@ -928,6 +936,7 @@ static int ggml_backend_sched_backend_id_from_cur(ggml_backend_sched_t sched, st
|
|||||||
return src_backend_id;
|
return src_backend_id;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
return -1;
|
return -1;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -2473,11 +2473,12 @@ llm_graph_cb llama_context::graph_get_cb() const {
|
|||||||
ggml_set_name(cur, name);
|
ggml_set_name(cur, name);
|
||||||
}
|
}
|
||||||
|
|
||||||
// norm may be automatically assigned to the backend of the previous layer, increasing data transfer between backends
|
// - norm may be automatically assigned to the backend of the previous layer, increasing data transfer between backends
|
||||||
|
// - force the last op of the layer on the specified backend to avoid running it on the backend of the next layer due to scheduling
|
||||||
// FIXME: fix in ggml_backend_sched
|
// FIXME: fix in ggml_backend_sched
|
||||||
const bool full_offload = model.n_gpu_layers() > model.hparams.n_layer_all;
|
const bool full_offload = model.n_gpu_layers() > model.hparams.n_layer_all;
|
||||||
if (ubatch.n_tokens < 32 || full_offload) {
|
if (ubatch.n_tokens < 32 || full_offload) {
|
||||||
if (il != -1 && strcmp(name, "norm") == 0) {
|
if (il != -1 && (strcmp(name, "norm") == 0 || strcmp(name, "l_last") == 0)) {
|
||||||
const auto & dev_layer = model.dev_layer(il);
|
const auto & dev_layer = model.dev_layer(il);
|
||||||
for (const auto & backend : backends) {
|
for (const auto & backend : backends) {
|
||||||
if (ggml_backend_get_device(backend.get()) == dev_layer) {
|
if (ggml_backend_get_device(backend.get()) == dev_layer) {
|
||||||
|
|||||||
@@ -1133,6 +1133,10 @@ llama_model_deepseek4::graph::graph(const llama_model & model, const llm_graph_p
|
|||||||
&post, &comb, il);
|
&post, &comb, il);
|
||||||
cb(cur, "hc_ffn_pre", il);
|
cb(cur, "hc_ffn_pre", il);
|
||||||
|
|
||||||
|
ggml_build_forward_expand(gf, residual);
|
||||||
|
ggml_build_forward_expand(gf, post);
|
||||||
|
ggml_build_forward_expand(gf, comb);
|
||||||
|
|
||||||
cur = build_norm(cur, model.layers[il].ffn_norm, nullptr, LLM_NORM_RMS, il);
|
cur = build_norm(cur, model.layers[il].ffn_norm, nullptr, LLM_NORM_RMS, il);
|
||||||
cb(cur, "ffn_norm", il);
|
cb(cur, "ffn_norm", il);
|
||||||
|
|
||||||
@@ -1175,7 +1179,7 @@ llama_model_deepseek4::graph::graph(const llama_model & model, const llm_graph_p
|
|||||||
|
|
||||||
inpL = build_hc_post(cur, residual, post, comb, il);
|
inpL = build_hc_post(cur, residual, post, comb, il);
|
||||||
inpL = build_cvec(inpL, il);
|
inpL = build_cvec(inpL, il);
|
||||||
cb(inpL, "l_out", il);
|
cb(inpL, "l_last", il);
|
||||||
}
|
}
|
||||||
|
|
||||||
if (inp_out_ids) {
|
if (inp_out_ids) {
|
||||||
|
|||||||
Reference in New Issue
Block a user