llama-batch: add n_keep_tail in split_equal for recurrent models (#25278)
This commit is contained in:
@@ -496,8 +496,8 @@ ggml_tensor * llm_build_delta_net_base::build_conv_state(
|
||||
ggml_build_forward_expand(gf, ggml_cpy(ctx0, conv_state_last, conv_state_update));
|
||||
} else {
|
||||
// [TAG_RECURRENT_ROLLBACK_SPLITS]
|
||||
// TODO: this logic incorrectly assumes that the last (n_rs_seq + 1) tokens of a sequence in a batch are
|
||||
// inside the same ubatch. currently with `split_equal()` this is not correct
|
||||
// this logic assumes that the last (n_rs_seq + 1) tokens of a sequence in a batch are inside
|
||||
// the same ubatch, which `split_equal()` guarantees via its n_keep_tail argument
|
||||
|
||||
const int64_t K = (int64_t) cparams.n_rs_seq + 1;
|
||||
|
||||
|
||||
Reference in New Issue
Block a user