ggml : drop support for QK_K=64 (#7473)

* ggml : drop support for QK_K=64 ggml-ci * opencl : restore QK_K=256 define
2025-07-26 03:03:25 -04:00 · 2024-05-23 10:00:21 +03:00
parent 1b1e27cb49
commit e84b71c2c6
16 changed files with 26 additions and 4049 deletions
--- a/ggml-cuda/mmq.cu
+++ b/ggml-cuda/mmq.cu
@ -826,11 +826,7 @@ template <int mmq_y, int nwarps, bool need_check> static __device__ __forceinlin

        const block_q4_K * bxi = bx0 + i*blocks_per_row + kbxd;

-#if QK_K == 256
        x_dm[i * (WARP_SIZE/QI4_K) + i / QI4_K + kbxd] = bxi->dm;
-#else
-        x_dm[i * (WARP_SIZE/QI4_K) + i / QI4_K + kbxd] = {bxi->dm[0], bxi->dm[1]};
-#endif
    }

 #pragma unroll
@ -933,9 +929,7 @@ template <int mmq_y, int nwarps, bool need_check> static __device__ __forceinlin

        const block_q5_K * bxi = bx0 + i*blocks_per_row + kbxd;

-#if QK_K == 256
        x_dm[i * (WARP_SIZE/QI5_K) + i / QI5_K + kbxd] = bxi->dm;
-#endif
    }

 #pragma unroll