* vulkan (DRAFT): split shader generation by GLSL source file, to improve incremental build times * support dep-files so shaders are recompiled if their included files change * rename shader files which are used as "headers" to use .glsl extension * move glslc extension detection shaders to separate folders * the above is to prevent them from getting glob'd with the actual compute shaders that need to be compiled * vulkan : only write embedded shader .hpp/.cpp when they change * avoid recompiling ggml-vulkan.cpp when editing shaders * pass single --source argument instead of --input-dir & --filter to shader gen * check for source file match earlier * fix hang in vulkan-shaders-gen when there are compilation errors * early out did not decrement compile_count * clean up * fix glslc integer dot product test * unconditionally write the embedded shader cpp output * replace output filepath in generated dep-files to match output in CMakeLists --------- Co-authored-by: Jeff Bolz <jbolz@nvidia.com>
35 lines
1.0 KiB
Plaintext
35 lines
1.0 KiB
Plaintext
#version 450
|
|
|
|
#include "dequant_head.glsl"
|
|
|
|
layout(local_size_x = 256, local_size_y = 1, local_size_z = 1) in;
|
|
|
|
layout (binding = 0) readonly buffer A {block_q5_0 data_a[];};
|
|
layout (binding = 1) writeonly buffer D {D_TYPE data_b[];};
|
|
|
|
void main() {
|
|
const uint i = gl_WorkGroupID.x * 4 + gl_LocalInvocationID.x / 64;
|
|
|
|
const uint tid = gl_LocalInvocationID.x % 64;
|
|
const uint il = tid/32;
|
|
const uint ir = tid%32;
|
|
const uint ib = 32*i + ir;
|
|
if (ib >= p.nel / 32) {
|
|
return;
|
|
}
|
|
|
|
const uint b_idx = 1024*i + 32*ir + 8*il;
|
|
|
|
const float d = float(data_a[ib].d);
|
|
const uint qh = uint(data_a[ib].qh[1]) << 16 | data_a[ib].qh[0];
|
|
|
|
const uint q_idx = 8*il;
|
|
|
|
[[unroll]] for (uint l = 0; l < 8; ++l) {
|
|
const uint iqs = q_idx + l;
|
|
const uint vui = uint(data_a[ib].qs[iqs]);
|
|
data_b[b_idx + l + 0] = D_TYPE(d * (((vui & 0xF) | (((qh >> iqs) << 4) & 0x10)) - 16.0f));
|
|
data_b[b_idx + l + 16] = D_TYPE(d * (((vui >> 4) | ((qh >> (iqs + 12)) & 0x10)) - 16.0f));
|
|
}
|
|
}
|