vulkan: copy iq4_nl LUT into shared memory (#10409)

2024-11-20 01:40:18 -06:00
parent 1bacb9f625
commit 8fd4b7fa29
6 changed files with 29 additions and 4 deletions
--- a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec.comp
+++ b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec.comp
@@ -161,6 +161,10 @@ void compute_outputs(const uint32_t first_row, const uint32_t num_rows) {
 void main() {
    const uint first_row = NUM_ROWS * (gl_WorkGroupID.x + gl_NumWorkGroups.x * gl_WorkGroupID.z);

+#if defined(DATA_A_IQ4_NL)
+    init_iq4nl_shmem();
+#endif
+
    // do NUM_ROWS at a time, unless there aren't enough remaining rows
    if (first_row + NUM_ROWS <= p.stride_d) {
        compute_outputs(first_row, NUM_ROWS);