* vulkan : added PAD_REFLECT_1D operation Implemented the GGML_OP_PAD_REFLECT_1D operation for the Vulkan backend Changes: - pad_reflect_1d.comp: implemented the GLSL compute shader with reflection logic - vulkan-shaders-gen.cpp: register the shader for SPIR-V compilation - ggml-vulkan.cpp: pushed constants struct, pipeline creation, supports_op, dispatch function, compute switch and debug validation Tested the PAD_REFLECT_1D on Intel Iris Xe (Vulkan 1.4, Mesa 25.2.8): Correctness: PAD_REFLECT_1D(type=f32,ne_a=[512,34,2,1],pad_0=10,pad_1=9) = Pass PAD_REFLECT_1D(type=f32,ne_a=[3000,384,4,1],pad_0=10,pad_1=9) = Pass 2/2 tests passed - All test are passed Performance: ne_a=[512,34,2,1] -> 5.38 us/run, 24.55 GB/s ne_a=[3000,80,1,1] -> 30.09 us/run, 59.62 GB/s ne_a=[3000,384,4,1] -> 158.31 us/run, 54.39 GB/s * Update ggml/src/ggml-vulkan/vulkan-shaders/pad_reflect_1d.comp Co-authored-by: Jeff Bolz <jbolz@nvidia.com> --------- Co-authored-by: Jeff Bolz <jbolz@nvidia.com>
44 lines
1.3 KiB
Plaintext
44 lines
1.3 KiB
Plaintext
#version 450
|
|
|
|
#include "types.glsl"
|
|
#include "generic_unary_head.glsl" // included to use functions like fastdiv etc.
|
|
|
|
layout(local_size_x = 512, local_size_y = 1, local_size_z = 1) in;
|
|
|
|
void main() {
|
|
|
|
const uint idx = get_idx();
|
|
|
|
if (idx >= p.ne) {
|
|
return;
|
|
}
|
|
|
|
const uint p0 = floatBitsToUint(p.param1);
|
|
const uint p1 = floatBitsToUint(p.param2);
|
|
|
|
const uint i3 = fastdiv(idx, p.ne1_012mp, fastdiv_L(p.ne1_Ls, 0));
|
|
const uint i3_offset = i3 * p.ne12 * p.ne11 * p.ne10;
|
|
|
|
const uint i2 = fastdiv(idx - i3_offset, p.ne1_01mp, fastdiv_L(p.ne1_Ls, 1));
|
|
const uint i2_offset = i2 * p.ne11 * p.ne10;
|
|
|
|
const uint i1 = fastdiv(idx - i3_offset - i2_offset, p.ne1_0mp, fastdiv_L(p.ne1_Ls, 2));
|
|
const uint i0 = idx - i3_offset - i2_offset - i1 * p.ne10;
|
|
|
|
uint src_col;
|
|
|
|
if (i0 < p0) {
|
|
src_col = p0 - i0; // left pad area
|
|
} else if (i0 < p0 + p.ne00) {
|
|
src_col = i0 - p0; // center area
|
|
} else {
|
|
src_col = 2u * p.ne00 - 2u - (i0 - p0); // right pad area
|
|
}
|
|
|
|
const uint src_idx = i3 * p.nb03 + i2 * p.nb02 + i1 * p.nb01 + src_col * p.nb00;
|
|
const uint d_idx = i3 * p.nb13 + i2 * p.nb12 + i1 * p.nb11 + i0 * p.nb10;
|
|
|
|
// copy the computed value to the destination tensor
|
|
data_d[get_doffset() + d_idx] = D_TYPE(data_a[get_aoffset() + src_idx]);
|
|
}
|