Commit 1e6f04a75 for llama.cpp
commit 1e6f04a75e91da4bb66b5248f3fd4bdeff169d41
Author: Captain-Tripps <jstaples2@hotmail.com>
Date: Fri Oct 9 21:56:23 2026 -0500
sycl : accelerate MXFP4 MoE with arithmetic decoding and weight reordering (#29809)
diff --git a/ggml/src/ggml-sycl/convert.cpp b/ggml/src/ggml-sycl/convert.cpp
index b660b56ab..b02353a18 100644
--- a/ggml/src/ggml-sycl/convert.cpp
+++ b/ggml/src/ggml-sycl/convert.cpp
@@ -537,6 +537,18 @@ static void dequantize_row_mxfp4_sycl(const void * vx, dst_t * y, const int64_t
});
}
+template <typename dst_t>
+static void dequantize_row_mxfp4_sycl_reorder(const void * vx, dst_t * y, const int64_t k, dpct::queue_ptr stream) {
+ GGML_ASSERT(k % QK_MXFP4 == 0);
+ const int n_warp = (k / QK_MXFP4 + WARP_SIZE - 1) / WARP_SIZE;
+ stream->parallel_for(
+ sycl::nd_range<3>(sycl::range<3>(1, 1, n_warp) * sycl::range<3>(1, 1, WARP_SIZE),
+ sycl::range<3>(1, 1, WARP_SIZE)),
+ [=](sycl::nd_item<3> item_ct1) [[sycl::reqd_sub_group_size(WARP_SIZE)]] {
+ dequantize_block_mxfp4_reorder(vx, y, k, item_ct1);
+ });
+}
+
template <typename dst_t>
static void dequantize_row_nvfp4_sycl(const void * vx, dst_t * y, const int64_t k, dpct::queue_ptr stream) {
GGML_ASSERT(k % QK_NVFP4 == 0);
@@ -728,6 +740,9 @@ to_fp16_sycl_t ggml_get_to_fp16_sycl(ggml_type type, ggml_tensor * dst) {
case GGML_TYPE_IQ4_NL:
return dequantize_row_iq4_nl_sycl;
case GGML_TYPE_MXFP4:
+ if (dst->src[0]->extra && ((ggml_tensor_extra_gpu *) dst->src[0]->extra)->optimized_feature.reorder) {
+ return dequantize_row_mxfp4_sycl_reorder;
+ }
return dequantize_row_mxfp4_sycl;
case GGML_TYPE_NVFP4:
return dequantize_row_nvfp4_sycl;
@@ -819,6 +834,9 @@ to_fp32_sycl_t ggml_get_to_fp32_sycl(ggml_type type, ggml_tensor *dst) {
case GGML_TYPE_IQ4_NL:
return dequantize_row_iq4_nl_sycl;
case GGML_TYPE_MXFP4:
+ if (dst->src[0]->extra && ((ggml_tensor_extra_gpu *) dst->src[0]->extra)->optimized_feature.reorder) {
+ return dequantize_row_mxfp4_sycl_reorder;
+ }
return dequantize_row_mxfp4_sycl;
case GGML_TYPE_NVFP4:
return dequantize_row_nvfp4_sycl;
diff --git a/ggml/src/ggml-sycl/dequantize.hpp b/ggml/src/ggml-sycl/dequantize.hpp
index 1b13e0f1a..8a2fd1f0e 100644
--- a/ggml/src/ggml-sycl/dequantize.hpp
+++ b/ggml/src/ggml-sycl/dequantize.hpp
@@ -1646,6 +1646,26 @@ static void dequantize_block_mxfp4(const void * __restrict__ vx, dst_t * __restr
}
}
+// Reordered MXFP4 ([qs...][e...], see ggml_sycl_reordered::block_q_t<MXFP4>): one work-item per block.
+template <typename dst_t>
+static void dequantize_block_mxfp4_reorder(const void * __restrict__ vx, dst_t * __restrict__ yy, int64_t k,
+ const sycl::nd_item<3> & item_ct1) {
+ const int64_t ib = (int64_t) item_ct1.get_group(2) * WARP_SIZE + item_ct1.get_local_id(2);
+ if (ib >= k / QK_MXFP4) {
+ return;
+ }
+
+ const uint8_t * qs = (const uint8_t *) vx + ib * (QK_MXFP4 / 2);
+ const float d = ggml_sycl_e8m0_to_fp32(((const uint8_t *) vx)[k / 2 + ib]) * 0.5f;
+ dst_t * y = yy + ib * QK_MXFP4;
+
+#pragma unroll
+ for (int j = 0; j < QK_MXFP4 / 2; ++j) {
+ y[j] = d * kvalues_mxfp4[qs[j] & 0xf];
+ y[j + QK_MXFP4 / 2] = d * kvalues_mxfp4[qs[j] >> 4];
+ }
+}
+
template <typename dst_t>
static void dequantize_block_nvfp4(
diff --git a/ggml/src/ggml-sycl/ggml-sycl.cpp b/ggml/src/ggml-sycl/ggml-sycl.cpp
index d2c1a1a44..8557c1160 100644
--- a/ggml/src/ggml-sycl/ggml-sycl.cpp
+++ b/ggml/src/ggml-sycl/ggml-sycl.cpp
@@ -773,7 +773,8 @@ ggml_backend_sycl_buffer_init_tensor(ggml_backend_buffer_t buffer,
case GGML_TYPE_Q3_K:
case GGML_TYPE_Q4_K:
case GGML_TYPE_Q5_K:
- case GGML_TYPE_Q6_K:{
+ case GGML_TYPE_Q6_K:
+ case GGML_TYPE_MXFP4:{
ggml_tensor_extra_gpu * extra = new ggml_tensor_extra_gpu{};
tensor->extra = extra;
ctx->tensor_extras.push_back(extra);
@@ -4623,6 +4624,58 @@ static bool reorder_qw_q6_k_moe(uint8_t * data_device, size_t expert_bytes, int6
return true;
}
+// Reorder each MXFP4 expert slice into [qs][e]: 16-byte nibble blocks, then one E8M0 byte per block.
+// Experts are self-contained, so the tensor is reordered a few experts at a time through a small
+// temporary: a whole-tensor temporary (hundreds of MB) can exceed the VRAM left on a nearly full card,
+// and on Windows the driver then pages device memory out to host RAM instead of failing.
+static bool reorder_qw_mxfp4_moe(uint8_t * data_device, size_t expert_bytes, int64_t n_expert, dpct::queue_ptr stream) {
+ GGML_ASSERT(expert_bytes % sizeof(block_mxfp4) == 0);
+ const int blocks_per_expert = (int) (expert_bytes / sizeof(block_mxfp4));
+ const size_t max_chunk_bytes = 32u << 20;
+ const int64_t chunk_experts = std::max<int64_t>(1, std::min<int64_t>(n_expert, (int64_t) (max_chunk_bytes / expert_bytes)));
+
+ sycl_reorder_temp_buffer tmp(stream, (size_t) chunk_experts * expert_bytes);
+ if (!tmp) {
+ GGML_LOG_WARN("%s: failed to allocate %zu bytes for reorder temp buffer, skipping reorder\n", __func__,
+ (size_t) chunk_experts * expert_bytes);
+ return false;
+ }
+ uint8_t * tmp_buf = static_cast<uint8_t *>(tmp.ptr);
+
+ // the queue is in-order: each chunk's copy into tmp_buf waits for the previous chunk's kernel
+ for (int64_t e0 = 0; e0 < n_expert; e0 += chunk_experts) {
+ const int64_t n_chunk = std::min(chunk_experts, n_expert - e0);
+ uint8_t * chunk = data_device + (size_t) e0 * expert_bytes;
+
+ sycl::event copy_event;
+ SYCL_CHECK(CHECK_TRY_ERROR(copy_event = stream->memcpy(tmp_buf, chunk, (size_t) n_chunk * expert_bytes)));
+ if (!g_ggml_sycl_use_async_mem_op) {
+ copy_event.wait();
+ }
+
+ const int total_blocks = blocks_per_expert * (int) n_chunk;
+ auto reorder_event = stream->parallel_for(total_blocks, [=](auto gb_) {
+ const int gb = gb_;
+ const int e = gb / blocks_per_expert;
+ const int ib = gb % blocks_per_expert;
+ const block_mxfp4 * x = (const block_mxfp4 *) (tmp_buf + (size_t) e * expert_bytes);
+ uint8_t * base = chunk + (size_t) e * expert_bytes;
+
+ uint8_t * qs_ptr = base;
+ uint8_t * e_ptr = qs_ptr + (QK_MXFP4 / 2) * (size_t) blocks_per_expert;
+
+ for (int j = 0; j < QK_MXFP4 / 2; ++j) {
+ qs_ptr[(size_t) ib * (QK_MXFP4 / 2) + j] = x[ib].qs[j];
+ }
+ e_ptr[ib] = x[ib].e;
+ });
+ if (!g_ggml_sycl_use_async_mem_op) {
+ reorder_event.wait_and_throw();
+ }
+ }
+ return true;
+}
+
static bool reorder_qw_q2_k(uint8_t * data_device, size_t size, size_t offset, dpct::queue_ptr stream) {
GGML_ASSERT(size % sizeof(block_q2_K) == 0);
GGML_ASSERT(offset % sizeof(block_q2_K) == 0);
@@ -4832,6 +4885,8 @@ static bool reorder_qw(const ggml_tensor * src0, dpct::queue_ptr stream) {
return reorder_qw_q5_k_moe(data_device, src0->nb[2], src0->ne[2], stream);
case GGML_TYPE_Q6_K:
return reorder_qw_q6_k_moe(data_device, src0->nb[2], src0->ne[2], stream);
+ case GGML_TYPE_MXFP4:
+ return reorder_qw_mxfp4_moe(data_device, src0->nb[2], src0->ne[2], stream);
default:
return false;
}
@@ -4905,7 +4960,12 @@ static void opt_for_reorder_id(ggml_backend_sycl_context * ctx, const ggml_tenso
if (!g_ggml_sycl_enable_optimize || !ctx->opt_feature.reorder) {
return;
}
- if (src0->type != GGML_TYPE_Q4_K && src0->type != GGML_TYPE_Q5_K && src0->type != GGML_TYPE_Q6_K) {
+ if (src0->type != GGML_TYPE_Q4_K && src0->type != GGML_TYPE_Q5_K && src0->type != GGML_TYPE_Q6_K &&
+ src0->type != GGML_TYPE_MXFP4) {
+ return;
+ }
+ // The MXFP4 reorder kernels use 8-byte vector loads, so every expert slice must stay aligned.
+ if (src0->type == GGML_TYPE_MXFP4 && (src0->nb[2] % 16 != 0 || (uintptr_t) src0->data % 16 != 0)) {
return;
}
ggml_tensor_extra_gpu * extra = static_cast<ggml_tensor_extra_gpu *>(src0->extra);
@@ -5388,6 +5448,11 @@ static void ggml_sycl_mul_mat_id(ggml_backend_sycl_context & ctx,
}
}
+ // The per-expert loop below reads the experts in whatever layout they have: reorder MXFP4 here as well, so prompt processing does not depend on a single-token decode having run first.
+ if (src0->type == GGML_TYPE_MXFP4) {
+ opt_for_reorder_id(&ctx, src0);
+ }
+
std::vector<char> ids_host(ggml_nbytes(ids));
const char * ids_dev = (const char *) ids->data;
diff --git a/ggml/src/ggml-sycl/mmvq.cpp b/ggml/src/ggml-sycl/mmvq.cpp
index e8b293a97..8a46967d9 100644
--- a/ggml/src/ggml-sycl/mmvq.cpp
+++ b/ggml/src/ggml-sycl/mmvq.cpp
@@ -1285,6 +1285,65 @@ static void reorder_mul_mat_vec_q8_0_q8_1_sycl_switch_ncols(
}
}
+// MXFP4 reorder GEMV. Only MoE expert slices are reordered (opt_for_reorder_id); these dense entry
+// points serve per-expert ggml_sycl_mul_mat calls from multi-token MUL_MAT_ID after that reorder.
+static void reorder_mul_mat_vec_mxfp4_q8_1_sycl(const void * vx, const void * vy, float * dst, const int ncols,
+ const int nrows, dpct::queue_ptr stream) {
+ GGML_ASSERT(ncols % QK_MXFP4 == 0);
+ constexpr size_t num_subgroups = WARP_SIZE;
+ const int block_num_y = ceil_div(nrows, GGML_SYCL_MMV_Y * (int) num_subgroups);
+ const sycl::range<3> block_nums(1, 1, block_num_y);
+ const sycl::range<3> block_dims(1, GGML_SYCL_MMV_Y, num_subgroups * WARP_SIZE);
+
+ stream->submit([&](sycl::handler & cgh) {
+ cgh.parallel_for(sycl::nd_range<3>(block_nums * block_dims, block_dims),
+ [=](sycl::nd_item<3> nd_item) [[sycl::reqd_sub_group_size(WARP_SIZE)]] {
+ mul_mat_vec_q_reorder<reorder_vec_dot_q_sycl<GGML_TYPE_MXFP4>>(vx, vy, dst, ncols, nrows,
+ nd_item);
+ });
+ });
+}
+
+template <int ncols_dst>
+static void reorder_mul_mat_vec_mxfp4_q8_1_sycl_ncols(
+ const void * vx, const void * vy, float * dst,
+ const int ncols, const int nrows,
+ const int stride_col_y_bytes, const int stride_col_dst,
+ dpct::queue_ptr stream) {
+ GGML_ASSERT(ncols % QK_MXFP4 == 0);
+ constexpr size_t num_subgroups = WARP_SIZE;
+ const int block_num_y = ceil_div(nrows, GGML_SYCL_MMV_Y * (int) num_subgroups);
+ const sycl::range<3> block_nums(1, 1, block_num_y);
+ const sycl::range<3> block_dims(1, GGML_SYCL_MMV_Y, num_subgroups * WARP_SIZE);
+
+ stream->submit([&](sycl::handler & cgh) {
+ cgh.parallel_for(sycl::nd_range<3>(block_nums * block_dims, block_dims),
+ [=](sycl::nd_item<3> nd_item) [[sycl::reqd_sub_group_size(WARP_SIZE)]] {
+ mul_mat_vec_q_reorder_ncols<reorder_vec_dot_q_sycl<GGML_TYPE_MXFP4>, ncols_dst>(
+ vx, /*vgate=*/ nullptr, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst,
+ /*glu_op=*/ GGML_GLU_OP_SWIGLU, nd_item);
+ });
+ });
+}
+
+static void reorder_mul_mat_vec_mxfp4_q8_1_sycl_switch_ncols(
+ const void * vx, const void * vy, float * dst,
+ const int ncols, const int nrows, const int ncols_dst,
+ const int stride_col_y_bytes, const int stride_col_dst,
+ dpct::queue_ptr stream) {
+ switch (ncols_dst) {
+ case 1: reorder_mul_mat_vec_mxfp4_q8_1_sycl(vx, vy, dst, ncols, nrows, stream); break;
+ case 2: reorder_mul_mat_vec_mxfp4_q8_1_sycl_ncols<2>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); break;
+ case 3: reorder_mul_mat_vec_mxfp4_q8_1_sycl_ncols<3>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); break;
+ case 4: reorder_mul_mat_vec_mxfp4_q8_1_sycl_ncols<4>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); break;
+ case 5: reorder_mul_mat_vec_mxfp4_q8_1_sycl_ncols<5>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); break;
+ case 6: reorder_mul_mat_vec_mxfp4_q8_1_sycl_ncols<6>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); break;
+ case 7: reorder_mul_mat_vec_mxfp4_q8_1_sycl_ncols<7>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); break;
+ case 8: reorder_mul_mat_vec_mxfp4_q8_1_sycl_ncols<8>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); break;
+ default: GGML_ABORT("unsupported ncols_dst=%d for MXFP4 reorder multi-col MMVQ", ncols_dst);
+ }
+}
+
static void mul_mat_vec_q8_0_q8_1_sycl(const void *vx, const void *vy,
float *dst, const int ncols,
const int nrows,
@@ -2765,7 +2824,21 @@ void ggml_sycl_op_mul_mat_vec_q(ggml_backend_sycl_context & ctx, const ggml_tens
}
break;
case GGML_TYPE_MXFP4:
- if (i == 0 && src1_ncols > 1 && src1_ncols <= 8) {
+ if ((ggml_tensor_extra_gpu *) dst->src[0]->extra &&
+ ((ggml_tensor_extra_gpu *) dst->src[0]->extra)->optimized_feature.reorder) {
+ if (i == 0 && src1_ncols > 1 && src1_ncols <= 8) {
+ const int stride_col_y_bytes = src1_padded_col_size * q8_1_ts / q8_1_bs;
+ const int stride_col_dst = dst->ne[0];
+ GGML_SYCL_DEBUG("Calling reorder_mul_mat_vec_mxfp4_q8_1_sycl_switch_ncols ncols=%d\n", (int)src1_ncols);
+ reorder_mul_mat_vec_mxfp4_q8_1_sycl_switch_ncols(
+ src0_dd_i, src1_ddq_i, dst_dd_i, ne00, row_diff,
+ src1_ncols, stride_col_y_bytes, stride_col_dst, stream);
+ return;
+ } else {
+ GGML_SYCL_DEBUG("Calling reorder_mul_mat_vec_mxfp4_q8_1_sycl\n");
+ reorder_mul_mat_vec_mxfp4_q8_1_sycl(src0_dd_i, src1_ddq_i_bs, dst_dd_i_bs, ne00, row_diff, stream);
+ }
+ } else if (i == 0 && src1_ncols > 1 && src1_ncols <= 8) {
const int stride_col_y = src1_padded_col_size / QK8_1;
const int stride_col_dst = dst->ne[0];
GGML_SYCL_DEBUG("Calling mul_mat_vec_mxfp4_q8_1_sycl_switch_ncols ncols=%d\n", (int)src1_ncols);
@@ -3111,6 +3184,11 @@ bool ggml_sycl_mul_mat_vec_q_id_reorder(
vx_base, vy, ids_dev, dst_base, ncols, nrows, n_experts_used,
expert_weight_stride, dst_row_stride, src1_row_stride, stream);
return true;
+ case GGML_TYPE_MXFP4:
+ launch_mul_mat_vec_q_moe_reorder<reorder_vec_dot_q_sycl<GGML_TYPE_MXFP4>>(
+ vx_base, vy, ids_dev, dst_base, ncols, nrows, n_experts_used,
+ expert_weight_stride, dst_row_stride, src1_row_stride, stream);
+ return true;
default:
return false;
}
diff --git a/ggml/src/ggml-sycl/quants.hpp b/ggml/src/ggml-sycl/quants.hpp
index a26a6ce6e..e6d7765b6 100644
--- a/ggml/src/ggml-sycl/quants.hpp
+++ b/ggml/src/ggml-sycl/quants.hpp
@@ -199,6 +199,27 @@ template <> struct block_q_t<GGML_TYPE_Q8_0> {
static constexpr int block_to_q8_1_ratio() { return traits::qk / QK8_1; } // 1
};
+template <> struct block_q_t<GGML_TYPE_MXFP4> {
+ struct traits {
+ static constexpr uint32_t qk = QK_MXFP4; // 32
+ static constexpr uint32_t qi = QI_MXFP4; // 4
+ static constexpr uint32_t qr = QR_MXFP4; // 2
+ static constexpr uint32_t vdr_mmvq = 2;
+ };
+
+ // MXFP4 reorder layout: [qs0|qs1|...|qsN][e0|e1|...|eN]
+ // The 17-byte AoS block leaves qs unaligned; split out, every 16-byte nibble block is aligned.
+ static constexpr std::pair<int, int> get_block_offset(const int block_index, const int /* nblocks */) {
+ return { block_index * (QK_MXFP4 / 2), 0 };
+ }
+
+ static constexpr std::pair<int, int> get_d_offset(int nrows, int ncols, const int block_index) {
+ return { (ncols / 2 * nrows) + block_index, 0 };
+ }
+
+ static constexpr int block_to_q8_1_ratio() { return traits::qk / QK8_1; } // 1
+};
+
} // namespace ggml_sycl_reordered
#endif // GGML_SYCL_QUANTS_HPP
diff --git a/ggml/src/ggml-sycl/vecdotq.hpp b/ggml/src/ggml-sycl/vecdotq.hpp
index 6ae951525..075995662 100644
--- a/ggml/src/ggml-sycl/vecdotq.hpp
+++ b/ggml/src/ggml-sycl/vecdotq.hpp
@@ -148,6 +148,28 @@ static __dpct_inline__ sycl::int2 get_int_from_table_16(
dpct::byte_level_permute(tmp[0], tmp[1], 0x7531));
}
+// Four E2M1 codes (one per byte, bits 0..3) to their kvalues_mxfp4 int8 values. SWAR arithmetic
+// replaces get_int_from_table_16 for MXFP4: dpct::byte_level_permute is emulated with 64-bit shifts,
+// eight per int, which made the MXFP4 GEMV compute-bound on Intel GPUs.
+// Magnitudes 0,1,2,3,4,6,8,12 = m + [m>=5] + [m>=6] + 3*[m>=7]; each byte stays below 256, so the
+// byte-wise adds never carry. -0 (code 8) is left as 0 so the two's-complement +1 cannot carry either.
+static __dpct_inline__ int mxfp4_codes_to_int8(const uint32_t x) {
+ const uint32_t m = x & 0x07070707u;
+ const uint32_t ge5 = ((m + 0x03030303u) >> 3) & 0x01010101u;
+ const uint32_t ge6 = ((m + 0x02020202u) >> 3) & 0x01010101u;
+ const uint32_t ge7 = ((m + 0x01010101u) >> 3) & 0x01010101u;
+ const uint32_t mag = m + ge5 + ge6 + 3u * ge7;
+ const uint32_t nz = ((mag + 0x7f7f7f7fu) >> 7) & 0x01010101u;
+ const uint32_t neg = (x >> 3) & nz & 0x01010101u;
+ return (int) ((mag ^ (neg * 0xffu)) + neg);
+}
+
+// Same result as get_int_from_table_16(q4, kvalues_mxfp4): x = low nibbles, y = high nibbles.
+static __dpct_inline__ sycl::int2 get_int_from_mxfp4(const int q4) {
+ return sycl::int2(mxfp4_codes_to_int8((uint32_t) q4 & 0x0f0f0f0fu),
+ mxfp4_codes_to_int8(((uint32_t) q4 >> 4) & 0x0f0f0f0fu));
+}
+
#define VDR_Q2_K_Q8_1_MMVQ 1
// contiguous v/x values
@@ -795,6 +817,41 @@ template <> struct reorder_vec_dot_q_sycl<GGML_TYPE_Q6_K> {
vl, vh, u0, u1, scs[0], scs[4], *d, d80, d81);
}
};
+
+template <> struct reorder_vec_dot_q_sycl<GGML_TYPE_MXFP4> {
+ static constexpr ggml_type gtype = GGML_TYPE_MXFP4;
+
+ using mxfp4_block = ggml_sycl_reordered::block_q_t<GGML_TYPE_MXFP4>;
+ using mxfp4_traits = typename mxfp4_block::traits;
+
+ __dpct_inline__ float operator()(const void * __restrict__ vbq, const std::pair<int, int> ibx_offset,
+ const std::pair<int, int> d_offset, const int8_t * q8_1_quant_ptr,
+ const sycl::half2 * q8_1_ds, const int & iqs) {
+ static_assert(mxfp4_traits::vdr_mmvq == 2, "vector load assumes vdr_mmvq == 2");
+ const uint8_t * base = static_cast<const uint8_t *>(vbq);
+
+ // Reordered nibble blocks are 16 contiguous bytes and iqs is 0 or 2, so each lane's two
+ // weight ints are one aligned 8-byte load (the AoS layout needed eight byte loads).
+ const sycl::int2 q4 = *reinterpret_cast<const sycl::int2 *>(base + ibx_offset.first + sizeof(int) * iqs);
+ const uint8_t e = base[d_offset.first];
+
+ // Low nibbles pair with q8_1 ints iqs..iqs+1, high nibbles with iqs+4..iqs+5.
+ const sycl::int2 u_lo = *reinterpret_cast<const sycl::int2 *>(q8_1_quant_ptr + sizeof(int) * iqs);
+ const sycl::int2 u_hi = *reinterpret_cast<const sycl::int2 *>(q8_1_quant_ptr + sizeof(int) * (iqs + 4));
+
+ const sycl::int2 v0 = get_int_from_mxfp4(q4.x());
+ const sycl::int2 v1 = get_int_from_mxfp4(q4.y());
+
+ int sumi = 0;
+ sumi = ggml_sycl_dp4a(v0.x(), u_lo.x(), sumi);
+ sumi = ggml_sycl_dp4a(v0.y(), u_hi.x(), sumi);
+ sumi = ggml_sycl_dp4a(v1.x(), u_lo.y(), sumi);
+ sumi = ggml_sycl_dp4a(v1.y(), u_hi.y(), sumi);
+
+ const float d = ggml_sycl_e8m0_to_fp32(e) * 0.5f * static_cast<float>((*q8_1_ds)[0]);
+ return d * sumi;
+ }
+};
#define VDR_Q4_0_Q8_1_MMVQ 2
#define VDR_Q4_0_Q8_1_MMQ 4
@@ -1124,7 +1181,7 @@ static __dpct_inline__ float vec_dot_mxfp4_q8_1(const void * __restrict__ vbq,
#pragma unroll
for (int l = 0; l < VDR_MXFP4_Q8_1_MMVQ; ++l) {
const int aux_q4 = get_int_b1(bq4->qs, iqs + l);
- const sycl::int2 v = get_int_from_table_16(aux_q4, kvalues_mxfp4);
+ const sycl::int2 v = get_int_from_mxfp4(aux_q4);
sumi = ggml_sycl_dp4a(v.x(), q8[l + 0], sumi);
sumi = ggml_sycl_dp4a(v.y(), q8[l + 4], sumi);
}