Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
dequantize.cuh453 linesDownload Raw Back to ggml-cuda
1#include "common.cuh"2#include "convert.cuh"3 4static __device__ __forceinline__ void dequantize_q1_0(const void * vx, const int64_t ib, const int iqs, float2 & v){5    const block_q1_0 * x = (const block_q1_0 *) vx;6 7    const float d = x[ib].d;8 9    const int bit_index_0 = iqs;10    const int bit_index_1 = iqs + 1;11 12    const int byte_index_0 = bit_index_0 / 8;13    const int bit_offset_0 = bit_index_0 % 8;14 15    const int byte_index_1 = bit_index_1 / 8;16    const int bit_offset_1 = bit_index_1 % 8;17 18    // Extract bits: 1 = +d, 0 = -d (branchless)19    const int bit_0 = (x[ib].qs[byte_index_0] >> bit_offset_0) & 1;20    const int bit_1 = (x[ib].qs[byte_index_1] >> bit_offset_1) & 1;21 22    v.x = (2*bit_0 - 1) * d;23    v.y = (2*bit_1 - 1) * d;24}25 26static __device__ __forceinline__ void dequantize_q2_0(const void * vx, const int64_t ib, const int iqs, float2 & v){27    const block_q2_0 * x = (const block_q2_0 *) vx;28 29    const float d = x[ib].d;30 31    // Q2_0: 2 bits per element, 4 elements per byte.32    // Stored code c in {0,1,2,3} maps to symbol s = c - 1 in {-1, 0, +1, +2}.33    const int byte_index_0 = iqs / 4;34    const int bit_offset_0 = (iqs % 4) * 2;35 36    const int byte_index_1 = (iqs + 1) / 4;37    const int bit_offset_1 = ((iqs + 1) % 4) * 2;38 39    const int c0 = (x[ib].qs[byte_index_0] >> bit_offset_0) & 0x3;40    const int c1 = (x[ib].qs[byte_index_1] >> bit_offset_1) & 0x3;41 42    v.x = (c0 - 1) * d;43    v.y = (c1 - 1) * d;44}45 46static __device__ __forceinline__ void dequantize_q4_0(const void * vx, const int64_t ib, const int iqs, float2 & v){47    const block_q4_0 * x = (const block_q4_0 *) vx;48 49    const float d = x[ib].d;50 51    const int vui = x[ib].qs[iqs];52 53    v.x = vui & 0xF;54    v.y = vui >> 4;55 56    v.x = (v.x - 8.0f) * d;57    v.y = (v.y - 8.0f) * d;58}59 60static __device__ __forceinline__ void dequantize_q4_1(const void * vx, const int64_t ib, const int iqs, float2 & v){61    const block_q4_1 * x = (const block_q4_1 *) vx;62 63    const float2 dm = __half22float2(x[ib].dm);64 65    const int vui = x[ib].qs[iqs];66 67    v.x = vui & 0xF;68    v.y = vui >> 4;69 70    v.x = (v.x * dm.x) + dm.y;71    v.y = (v.y * dm.x) + dm.y;72}73 74static __device__ __forceinline__ void dequantize_q5_0(const void * vx, const int64_t ib, const int iqs, float2 & v){75    const block_q5_0 * x = (const block_q5_0 *) vx;76 77    const float d = x[ib].d;78 79    uint32_t qh;80    memcpy(&qh, x[ib].qh, sizeof(qh));81 82    const int xh_0 = ((qh >> (iqs +  0)) << 4) & 0x10;83    const int xh_1 = ((qh >> (iqs + 12))     ) & 0x10;84 85    v.x = ((x[ib].qs[iqs] & 0xf) | xh_0);86    v.y = ((x[ib].qs[iqs] >>  4) | xh_1);87 88    v.x = (v.x - 16.0f) * d;89    v.y = (v.y - 16.0f) * d;90}91 92static __device__ __forceinline__ void dequantize_q5_1(const void * vx, const int64_t ib, const int iqs, float2 & v){93    const block_q5_1 * x = (const block_q5_1 *) vx;94 95    const float2 dm = __half22float2(x[ib].dm);96 97    uint32_t qh;98    memcpy(&qh, x[ib].qh, sizeof(qh));99 100    const int xh_0 = ((qh >> (iqs +  0)) << 4) & 0x10;101    const int xh_1 = ((qh >> (iqs + 12))     ) & 0x10;102 103    v.x = ((x[ib].qs[iqs] & 0xf) | xh_0);104    v.y = ((x[ib].qs[iqs] >>  4) | xh_1);105 106    v.x = (v.x * dm.x) + dm.y;107    v.y = (v.y * dm.x) + dm.y;108}109 110static __device__ __forceinline__ void dequantize_q8_0(const void * vx, const int64_t ib, const int iqs, float2 & v){111    const block_q8_0 * x = (const block_q8_0 *) vx;112 113    const float d = x[ib].d;114 115    v.x = x[ib].qs[iqs + 0];116    v.y = x[ib].qs[iqs + 1];117 118    v.x *= d;119    v.y *= d;120}121 122//================================== k-quants123 124// Each call dequantizes one super-block of QK_K values into y using the125// thread layout of the caller: 32 threads for q4_K, 64 threads otherwise.126 127template<typename dst_t>128static __device__ __forceinline__ void dequantize_q2_K(const void * vx, const int64_t ib, dst_t * yy, const int tid) {129    const block_q2_K * x = (const block_q2_K *) vx;130 131    const int64_t n   = tid/32;132    const int64_t l   = tid - 32*n;133    const int64_t is  = 8*n + l/16;134 135    const uint8_t q = x[ib].qs[32*n + l];136    dst_t * y = yy + 128*n;137 138    float dall = __low2half(x[ib].dm);139    float dmin = __high2half(x[ib].dm);140    y[l+ 0] = ggml_cuda_cast<dst_t>(dall * (x[ib].scales[is+0] & 0xF) * ((q >> 0) & 3) - dmin * (x[ib].scales[is+0] >> 4));141    y[l+32] = ggml_cuda_cast<dst_t>(dall * (x[ib].scales[is+2] & 0xF) * ((q >> 2) & 3) - dmin * (x[ib].scales[is+2] >> 4));142    y[l+64] = ggml_cuda_cast<dst_t>(dall * (x[ib].scales[is+4] & 0xF) * ((q >> 4) & 3) - dmin * (x[ib].scales[is+4] >> 4));143    y[l+96] = ggml_cuda_cast<dst_t>(dall * (x[ib].scales[is+6] & 0xF) * ((q >> 6) & 3) - dmin * (x[ib].scales[is+6] >> 4));144}145 146template<typename dst_t>147static __device__ __forceinline__ void dequantize_q3_K(const void * vx, const int64_t ib, dst_t * yy, const int tid) {148    const block_q3_K * x = (const block_q3_K *) vx;149 150    const int64_t r = tid/4;151    const int64_t t = r/2;152    const int64_t is0 = r%2;153    const int64_t l0 = 16*is0 + 4*(tid%4);154    const int64_t n = t / 4;155    const int64_t j = t - 4*n;156 157    uint8_t m = 1 << (4*n + j);158    int64_t is = 8*n + 2*j + is0;159    int shift = 2*j;160 161    int8_t us = is <  4 ? (x[ib].scales[is-0] & 0xF) | (((x[ib].scales[is+8] >> 0) & 3) << 4) :162                is <  8 ? (x[ib].scales[is-0] & 0xF) | (((x[ib].scales[is+4] >> 2) & 3) << 4) :163                is < 12 ? (x[ib].scales[is-8] >>  4) | (((x[ib].scales[is+0] >> 4) & 3) << 4) :164                          (x[ib].scales[is-8] >>  4) | (((x[ib].scales[is-4] >> 6) & 3) << 4);165    float d_all = x[ib].d;166    float dl = d_all * (us - 32);167 168    dst_t * y = yy + 128*n + 32*j;169    const uint8_t * q = x[ib].qs + 32*n;170    const uint8_t * hm = x[ib].hmask;171 172    for (int l = l0; l < l0+4; ++l) {173        y[l] = ggml_cuda_cast<dst_t>(dl * ((int8_t)((q[l] >> shift) & 3) - ((hm[l] & m) ? 0 : 4)));174    }175}176 177static inline __device__ void get_scale_min_k4(int j, const uint8_t * q, uint8_t & d, uint8_t & m) {178    if (j < 4) {179        d = q[j] & 63; m = q[j + 4] & 63;180    } else {181        d = (q[j+4] & 0xF) | ((q[j-4] >> 6) << 4);182        m = (q[j+4] >>  4) | ((q[j-0] >> 6) << 4);183    }184}185 186template<typename dst_t>187static __device__ __forceinline__ void dequantize_q4_K(const void * vx, const int64_t ib, dst_t * yy, const int tid) {188    const block_q4_K * x = (const block_q4_K *) vx;189 190    // assume 32 threads191    const int64_t il  = tid/8;192    const int64_t ir  = tid%8;193    const int64_t is  = 2*il;194    const int64_t n   = 4;195 196    dst_t * y = yy + 64*il + n*ir;197 198    const float dall = __low2half(x[ib].dm);199    const float dmin = __high2half(x[ib].dm);200 201    const uint8_t * q = x[ib].qs + 32*il + n*ir;202 203    uint8_t sc, m;204    get_scale_min_k4(is + 0, x[ib].scales, sc, m);205    const float d1 = dall * sc; const float m1 = dmin * m;206    get_scale_min_k4(is + 1, x[ib].scales, sc, m);207    const float d2 = dall * sc; const float m2 = dmin * m;208    for (int l = 0; l < n; ++l) {209        y[l + 0] = ggml_cuda_cast<dst_t>(d1 * (q[l] & 0xF) - m1);210        y[l +32] = ggml_cuda_cast<dst_t>(d2 * (q[l] >>  4) - m2);211    }212}213 214template<typename dst_t>215static __device__ __forceinline__ void dequantize_q5_K(const void * vx, const int64_t ib, dst_t * yy, const int tid) {216    const block_q5_K * x = (const block_q5_K *) vx;217 218    // assume 64 threads - this is very slightly better than the one below219    const int64_t il  = tid/16;   // il is in 0...3220    const int64_t ir  = tid%16;   // ir is in 0...15221    const int64_t is  = 2*il;     // is is in 0...6222 223    dst_t * y = yy + 64*il + 2*ir;224 225    const float dall = __low2half(x[ib].dm);226    const float dmin = __high2half(x[ib].dm);227 228    const uint8_t * ql = x[ib].qs + 32*il + 2*ir;229    const uint8_t * qh = x[ib].qh + 2*ir;230 231    uint8_t sc, m;232    get_scale_min_k4(is + 0, x[ib].scales, sc, m);233    const float d1 = dall * sc; const float m1 = dmin * m;234    get_scale_min_k4(is + 1, x[ib].scales, sc, m);235    const float d2 = dall * sc; const float m2 = dmin * m;236 237    uint8_t   hm  = 1 << (2*il);238    y[ 0] = ggml_cuda_cast<dst_t>(d1 * ((ql[ 0] & 0xF) + (qh[ 0] & hm ? 16 : 0)) - m1);239    y[ 1] = ggml_cuda_cast<dst_t>(d1 * ((ql[ 1] & 0xF) + (qh[ 1] & hm ? 16 : 0)) - m1);240    hm <<= 1;241    y[32] = ggml_cuda_cast<dst_t>(d2 * ((ql[ 0] >>  4) + (qh[ 0] & hm ? 16 : 0)) - m2);242    y[33] = ggml_cuda_cast<dst_t>(d2 * ((ql[ 1] >>  4) + (qh[ 1] & hm ? 16 : 0)) - m2);243}244 245template<typename dst_t>246static __device__ __forceinline__ void dequantize_q6_K(const void * vx, const int64_t ib, dst_t * yy, const int tid) {247    const block_q6_K * x = (const block_q6_K *) vx;248 249    // assume 64 threads - this is very slightly better than the one below250    const int64_t ip  = tid/32;   // ip is 0 or 1251    const int64_t il  = tid - 32*ip; // 0...32252    const int64_t is  = 8*ip + il/16;253 254    dst_t * y = yy + 128*ip + il;255 256    const float d = x[ib].d;257 258    const uint8_t * ql = x[ib].ql + 64*ip + il;259    const uint8_t   qh = x[ib].qh[32*ip + il];260    const int8_t  * sc = x[ib].scales + is;261 262    y[ 0] = ggml_cuda_cast<dst_t>(d * sc[0] * ((int8_t)((ql[ 0] & 0xF) | (((qh >> 0) & 3) << 4)) - 32));263    y[32] = ggml_cuda_cast<dst_t>(d * sc[2] * ((int8_t)((ql[32] & 0xF) | (((qh >> 2) & 3) << 4)) - 32));264    y[64] = ggml_cuda_cast<dst_t>(d * sc[4] * ((int8_t)((ql[ 0]  >> 4) | (((qh >> 4) & 3) << 4)) - 32));265    y[96] = ggml_cuda_cast<dst_t>(d * sc[6] * ((int8_t)((ql[32]  >> 4) | (((qh >> 6) & 3) << 4)) - 32));266}267 268//================================== i-quants269 270// Each call dequantizes one super-block of QK_K values into y with 32271// threads; iq4_nl packs QK_K/QK4_NL sub-blocks per super-block.272 273template<typename dst_t>274static __device__ __forceinline__ void dequantize_iq2_xxs(const void * vx, const int64_t ibs, dst_t * yy, const int tid) {275 276    const block_iq2_xxs * x = (const block_iq2_xxs  *) vx;277 278    const int64_t il = tid/8; // 0...3279    const int64_t ib = tid%8; // 0...7280    dst_t * y = yy + 32*ib + 8*il;281    const uint16_t * q2 = x[ibs].qs + 4*ib;282    const uint8_t  * aux8 = (const uint8_t *)q2;283    const uint8_t  * grid = (const uint8_t *)(iq2xxs_grid + aux8[il]);284    const uint32_t aux32 = q2[2] | (q2[3] << 16);285    const float d = (float)x[ibs].d * (0.5f + (aux32 >> 28)) * 0.25f;286    const uint8_t signs = ksigns_iq2xs[(aux32 >> 7*il) & 127];287    for (int j = 0; j < 8; ++j) {288        y[j] = ggml_cuda_cast<dst_t>(d * grid[j] * (signs & kmask_iq2xs[j] ? -1.f : 1.f));289    }290}291 292template<typename dst_t>293static __device__ __forceinline__ void dequantize_iq2_xs(const void * vx, const int64_t ibs, dst_t * yy, const int tid) {294 295    const block_iq2_xs * x = (const block_iq2_xs *) vx;296 297    const int64_t il = tid/8; // 0...3298    const int64_t ib = tid%8; // 0...7299    dst_t * y = yy + 32*ib + 8*il;300    const uint16_t * q2 = x[ibs].qs + 4*ib;301    const uint8_t  * grid = (const uint8_t *)(iq2xs_grid + (q2[il] & 511));302    const float d = (float)x[ibs].d * (0.5f + ((x[ibs].scales[ib] >> 4*(il/2)) & 0xf)) * 0.25f;303    const uint8_t signs = ksigns_iq2xs[q2[il] >> 9];304    for (int j = 0; j < 8; ++j) {305        y[j] = ggml_cuda_cast<dst_t>(d * grid[j] * (signs & kmask_iq2xs[j] ? -1.f : 1.f));306    }307}308 309template<typename dst_t>310static __device__ __forceinline__ void dequantize_iq2_s(const void * vx, const int64_t ibs, dst_t * yy, const int tid) {311 312    const block_iq2_s * x = (const block_iq2_s *) vx;313 314    const int64_t il = tid/8; // 0...3315    const int64_t ib = tid%8; // 0...7316    dst_t * y = yy + 32*ib + 8*il;317    const uint8_t * grid = (const uint8_t *)(iq2s_grid + (x[ibs].qs[4*ib+il] | ((x[ibs].qh[ib] << (8-2*il)) & 0x300)));318    const float d = (float)x[ibs].d * (0.5f + ((x[ibs].scales[ib] >> 4*(il/2)) & 0xf)) * 0.25f;319    const uint8_t signs = x[ibs].qs[QK_K/8+4*ib+il];320    for (int j = 0; j < 8; ++j) {321        y[j] = ggml_cuda_cast<dst_t>(d * grid[j] * (signs & kmask_iq2xs[j] ? -1.f : 1.f));322    }323}324 325template<typename dst_t>326static __device__ __forceinline__ void dequantize_iq3_xxs(const void * vx, const int64_t ibs, dst_t * yy, const int tid) {327 328    const block_iq3_xxs * x = (const block_iq3_xxs  *) vx;329 330    const int64_t il = tid/8; // 0...3331    const int64_t ib = tid%8; // 0...7332    dst_t * y = yy + 32*ib + 8*il;333    const uint8_t  * q3 = x[ibs].qs + 8*ib;334    const uint16_t * gas = (const uint16_t *)(x[ibs].qs + QK_K/4) + 2*ib;335    const uint8_t  * grid1 = (const uint8_t *)(iq3xxs_grid + q3[2*il+0]);336    const uint8_t  * grid2 = (const uint8_t *)(iq3xxs_grid + q3[2*il+1]);337    const uint32_t aux32 = gas[0] | (gas[1] << 16);338    const float d = (float)x[ibs].d * (0.5f + (aux32 >> 28)) * 0.5f;339    const uint8_t signs = ksigns_iq2xs[(aux32 >> 7*il) & 127];340    for (int j = 0; j < 4; ++j) {341        y[j+0] = ggml_cuda_cast<dst_t>(d * grid1[j] * (signs & kmask_iq2xs[j+0] ? -1.f : 1.f));342        y[j+4] = ggml_cuda_cast<dst_t>(d * grid2[j] * (signs & kmask_iq2xs[j+4] ? -1.f : 1.f));343    }344}345 346template<typename dst_t>347static __device__ __forceinline__ void dequantize_iq3_s(const void * vx, const int64_t ibs, dst_t * yy, const int tid) {348 349    const block_iq3_s * x = (const block_iq3_s *) vx;350 351    const int64_t il = tid/8; // 0...3352    const int64_t ib = tid%8; // 0...7353    dst_t * y = yy + 32*ib + 8*il;354    const uint8_t * qs = x[ibs].qs + 8*ib;355    const uint8_t * grid1 = (const uint8_t *)(iq3s_grid + (qs[2*il+0] | ((x[ibs].qh[ib] << (8-2*il)) & 256)));356    const uint8_t * grid2 = (const uint8_t *)(iq3s_grid + (qs[2*il+1] | ((x[ibs].qh[ib] << (7-2*il)) & 256)));357    const float d = (float)x[ibs].d * (1 + 2*((x[ibs].scales[ib/2] >> 4*(ib%2)) & 0xf));358    const uint8_t signs = x[ibs].signs[4*ib + il];359    for (int j = 0; j < 4; ++j) {360        y[j+0] = ggml_cuda_cast<dst_t>(d * grid1[j] * (signs & kmask_iq2xs[j+0] ? -1.f : 1.f));361        y[j+4] = ggml_cuda_cast<dst_t>(d * grid2[j] * (signs & kmask_iq2xs[j+4] ? -1.f : 1.f));362    }363}364 365template<typename dst_t>366static __device__ __forceinline__ void dequantize_iq1_s(const void * vx, const int64_t ibs, dst_t * yy, const int tid) {367 368    const block_iq1_s * x = (const block_iq1_s  *) vx;369 370    const int64_t il = tid/8; // 0...3371    const int64_t ib = tid%8; // 0...7372    dst_t * y = yy + 32*ib + 8*il;373    const float delta = x[ibs].qh[ib] & 0x8000 ? -1 - IQ1S_DELTA : -1 + IQ1S_DELTA;374    const float d = (float)x[ibs].d * (2*((x[ibs].qh[ib] >> 12) & 7) + 1);375    uint32_t grid32[2]; const int8_t * q = (const int8_t *)grid32;376    grid32[0] = iq1s_grid_gpu[x[ibs].qs[4*ib+il] | (((x[ibs].qh[ib] >> 3*il) & 7) << 8)];377    grid32[1] = (grid32[0] >> 4) & 0x0f0f0f0f;378    grid32[0] &= 0x0f0f0f0f;379    for (int j = 0; j < 8; ++j) {380        y[j] = ggml_cuda_cast<dst_t>(d * (q[j] + delta));381    }382}383 384template<typename dst_t>385static __device__ __forceinline__ void dequantize_iq1_m(const void * vx, const int64_t ibs, dst_t * yy, const int tid) {386 387    const block_iq1_m * x = (const block_iq1_m  *) vx;388 389    const int64_t il = tid/8; // 0...3390    const int64_t ib = tid%8; // 0...7391    dst_t * y = yy + 32*ib + 8*il;392    const uint16_t * sc = (const uint16_t *)x[ibs].scales;393    iq1m_scale_t scale;394    scale.u16 = (sc[0] >> 12) | ((sc[1] >> 8) & 0x00f0) | ((sc[2] >> 4) & 0x0f00) | (sc[3] & 0xf000);395    const int64_t ib16 = 2*ib + il/2; // sc[ib16/4] >> 3*(ib16%4) -> sc[ib/2] >> 3*((2*ib+il/2)%4);396    const float d = (float)scale.f16 * (2*((sc[ib16/4] >> 3*(ib16%4)) & 0x7) + 1);397    const float delta = x[ibs].qh[2*ib+il/2] & (0x08 << 4*(il%2)) ? -1 - IQ1M_DELTA : -1 + IQ1M_DELTA;398    uint32_t grid32[2]; const int8_t * q = (const int8_t *)grid32;399    grid32[0] = iq1s_grid_gpu[x[ibs].qs[4*ib+il] | (((x[ibs].qh[2*ib+il/2] >> 4*(il%2)) & 7) << 8)];400    grid32[1] = (grid32[0] >> 4) & 0x0f0f0f0f;401    grid32[0] &= 0x0f0f0f0f;402    for (int j = 0; j < 8; ++j) {403        y[j] = ggml_cuda_cast<dst_t>(d * (q[j] + delta));404    }405}406 407template<typename dst_t>408static __device__ __forceinline__ void dequantize_iq4_nl(const void * vx, const int64_t ibs, dst_t * yy, const int tid) {409 410    const block_iq4_nl * x = (const block_iq4_nl *) vx + ibs*(QK_K/QK4_NL);411 412    const int64_t il = tid/8; // 0...3413    const int64_t ib = tid%8; // 0...7414    dst_t * y = yy + 32*ib + 4*il;415    const uint8_t  * q4 = x[ib].qs + 4*il;416    const float d = (float)x[ib].d;417    for (int j = 0; j < 4; ++j) {418        y[j+ 0] = ggml_cuda_cast<dst_t>(d * kvalues_iq4nl[q4[j] & 0xf]);419        y[j+16] = ggml_cuda_cast<dst_t>(d * kvalues_iq4nl[q4[j] >>  4]);420    }421}422 423template<typename dst_t>424static __device__ __forceinline__ void dequantize_iq4_xs(const void * vx, const int64_t ibs, dst_t * yy, const int tid) {425    const block_iq4_xs * x = (const block_iq4_xs *)vx;426 427    const int64_t il = tid/8; // 0...3428    const int64_t ib = tid%8; // 0...7429    dst_t * y = yy + 32*ib + 4*il;430    const uint8_t  * q4 = x[ibs].qs + 16*ib + 4*il;431    const float d = (float)x[ibs].d * ((((x[ibs].scales_l[ib/2] >> 4*(ib%2)) & 0xf) | (((x[ibs].scales_h >> 2*ib) & 3) << 4)) - 32);432    for (int j = 0; j < 4; ++j) {433        y[j+ 0] = ggml_cuda_cast<dst_t>(d * kvalues_iq4nl[q4[j] & 0xf]);434        y[j+16] = ggml_cuda_cast<dst_t>(d * kvalues_iq4nl[q4[j] >>  4]);435    }436}437 438template<typename dst_t>439static __device__ __forceinline__ void dequantize_mxfp4(const void * vx, const int64_t ibs, dst_t * yy, const int tid) {440 441    const block_mxfp4 * x = (const block_mxfp4 *) vx + ibs*(QK_K/QK_MXFP4);442 443    const int64_t il = tid/8; // 0...3444    const int64_t ib = tid%8; // 0...7445    dst_t * y = yy + 32*ib + 4*il;446    const uint8_t  * q4 = x[ib].qs + 4*il;447    const float d = ggml_cuda_e8m0_to_fp32(x[ib].e);448    for (int j = 0; j < 4; ++j) {449        y[j+ 0] = ggml_cuda_cast<dst_t>(d * kvalues_mxfp4[q4[j] & 0xf]*0.5f);450        y[j+16] = ggml_cuda_cast<dst_t>(d * kvalues_mxfp4[q4[j] >>  4]*0.5f);451    }452}453 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai