* ggml : process data in smaller chunks in CUDA ggml_top_k() implementation to reduce temporary buffers memory usage * ggml : allocate tmp_dst only only once before the loop * chore : whitespaces Co-authored-by: Georgi Gerganov <ggerganov@gmail.com> * ggml : use chunked processing in both CUDA CUB top-k and argsort implementations * chore : separate argsort_f32_i32_cuda_bitonic() call from return statement Co-authored-by: Johannes Gäßler <johannesg@5d6.de> * chore : replace ternary operators with min/max --------- Co-authored-by: Stanisław Szymczyk <sszymczy@gmail.com> Co-authored-by: Georgi Gerganov <ggerganov@gmail.com> Co-authored-by: Johannes Gäßler <johannesg@5d6.de>
21 lines
950 B
Plaintext
21 lines
950 B
Plaintext
#include "common.cuh"
|
|
|
|
void ggml_cuda_op_argsort(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
|
|
|
|
#ifdef GGML_CUDA_USE_CUB
|
|
int argsort_f32_i32_cuda_cub_chunk_nrows(const size_t nb01, const int64_t nrows);
|
|
void argsort_f32_i32_cuda_cub(ggml_cuda_pool & pool,
|
|
const float * x,
|
|
int * dst,
|
|
const int ncols,
|
|
const int nrows,
|
|
ggml_sort_order order,
|
|
cudaStream_t stream);
|
|
#endif // GGML_CUDA_USE_CUB
|
|
void argsort_f32_i32_cuda_bitonic(const float * x,
|
|
int * dst,
|
|
const int ncols,
|
|
const int nrows,
|
|
ggml_sort_order order,
|
|
cudaStream_t stream);
|