Commit ff30363a0 for llama.cpp
commit ff30363a0e3e2630828309ef5a37da5d1d40396c
Author: Titaniumtown <titaniumtown@proton.me>
Date: Wed Oct 7 23:45:21 2026 -0700
sycl: fuse the delta-net alpha gate (add + unary + mul) (#29687)
* sycl: fuse the delta-net alpha gate (add + unary + mul)
* tests: cover the fused add + unary + mul chain
* sycl: give the fused alpha gate a flat path and pin the node skip
diff --git a/ggml/src/ggml-sycl/element_wise.cpp b/ggml/src/ggml-sycl/element_wise.cpp
index 2e926abea..0d263aa76 100644
--- a/ggml/src/ggml-sycl/element_wise.cpp
+++ b/ggml/src/ggml-sycl/element_wise.cpp
@@ -452,6 +452,46 @@ static void unary_mul_sycl(const T * x, const T * g, T * dst, const int64_t k, c
});
}
+// ADD(bias) + UNARY + MUL(scale) with both broadcast over dim 0, the delta-net alpha gate:
+// dst[i] = op(a[i] + bias[i % ne0]) * scale[i % ne0]. k == ne0 makes that the flat index.
+template<typename F>
+static void add_unary_mul_flat_kernel(const float * a, const float * bias, const float * scale, float * dst,
+ const int64_t k, const sycl::nd_item<1> &item_ct1, F op) {
+ SYCL_GLOBAL_ID_LOOP(k, item_ct1) {
+ dst[i] = op(a[i] + bias[i]) * scale[i];
+ }
+}
+
+template<typename F>
+static void add_unary_mul_bcast_kernel(const float * a, const float * bias, const float * scale, float * dst,
+ const int64_t k, const sycl::uint3 ne0_fd, const sycl::nd_item<1> &item_ct1, F op) {
+ SYCL_GLOBAL_ID_LOOP(k, item_ct1) {
+ const uint32_t h = fastmodulo((uint32_t) i, ne0_fd);
+ dst[i] = op(a[i] + bias[h]) * scale[h];
+ }
+}
+
+template<typename F>
+static void add_unary_mul_sycl(const float * a, const float * bias, const float * scale, float * dst,
+ const int64_t k, const int64_t ne0, queue_ptr main_stream, F op) {
+ const size_t num_blocks = ceil_div((size_t) k, (size_t) SYCL_GLU_BLOCK_SIZE);
+ const sycl::nd_range<1> range(num_blocks * sycl::range<1>(SYCL_GLU_BLOCK_SIZE), sycl::range<1>(SYCL_GLU_BLOCK_SIZE));
+
+ if (k == ne0) {
+ main_stream->parallel_for(range, [=](sycl::nd_item<1> item_ct1) [[sycl::reqd_sub_group_size(WARP_SIZE)]] {
+ add_unary_mul_flat_kernel(a, bias, scale, dst, k, item_ct1, op);
+ });
+ return;
+ }
+
+ // 32-bit fastdiv, exact only below 2^31; ggml_sycl_can_fuse() already declined past that
+ GGML_ASSERT(k < ((int64_t) 1 << 31));
+ const sycl::uint3 ne0_fd = init_fastdiv_values((uint32_t) ne0);
+ main_stream->parallel_for(range, [=](sycl::nd_item<1> item_ct1) [[sycl::reqd_sub_group_size(WARP_SIZE)]] {
+ add_unary_mul_bcast_kernel(a, bias, scale, dst, k, ne0_fd, item_ct1, op);
+ });
+}
+
namespace ggml_sycl_detail {
static void acc_f32_sycl(const char *x, const char *y, float *dst,
const int64_t n_elements,
@@ -995,6 +1035,19 @@ static inline void ggml_sycl_op_swiglu(ggml_backend_sycl_context & ctx, ggml_ten
});
}
+// Hands `launch` the functor for the unary op of a fused unary chain. Anything else
+// ggml_sycl_can_fuse() has already declined, so the default is a dispatcher bug.
+template<typename F>
+static void dispatch_fused_unary_op(ggml_unary_op uop, F && launch) {
+ switch (uop) {
+ case GGML_UNARY_OP_SILU: launch([](float v) { return op_silu(v); }); break;
+ case GGML_UNARY_OP_SIGMOID: launch([](float v) { return op_sigmoid(v); }); break;
+ case GGML_UNARY_OP_SOFTPLUS: launch([](float v) { return op_softplus(v); }); break;
+ default:
+ GGML_ABORT("fused unary chain: unsupported unary op %s", ggml_unary_op_name(uop));
+ }
+}
+
// dst = op(unary_node->src[0]) * other, written straight to the MUL output, saving the
// standalone unary launch. Preconditions come from ggml_sycl_can_fuse(); re-asserted here.
void ggml_sycl_op_unary_mul_fused(ggml_backend_sycl_context & ctx, ggml_tensor * unary_node, ggml_tensor * mul_node) {
@@ -1032,13 +1085,41 @@ void ggml_sycl_op_unary_mul_fused(ggml_backend_sycl_context & ctx, ggml_tensor *
}
};
- switch (ggml_get_unary_op(unary_node)) {
- case GGML_UNARY_OP_SILU: dispatch_type([](float v) { return op_silu(v); }); break;
- case GGML_UNARY_OP_SIGMOID: dispatch_type([](float v) { return op_sigmoid(v); }); break;
- case GGML_UNARY_OP_SOFTPLUS: dispatch_type([](float v) { return op_softplus(v); }); break;
- default:
- GGML_ABORT("fused unary+mul: unsupported unary op %s", ggml_unary_op_name(ggml_get_unary_op(unary_node)));
- }
+ dispatch_fused_unary_op(ggml_get_unary_op(unary_node), dispatch_type);
+}
+
+// dst = op(a + bias) * scale for an ADD + UNARY + MUL chain whose bias and scale broadcast
+// over dim 0. Preconditions come from ggml_sycl_can_fuse(); re-asserted here.
+void ggml_sycl_op_add_unary_mul_fused(ggml_backend_sycl_context & ctx, ggml_tensor * add_node,
+ ggml_tensor * unary_node, ggml_tensor * mul_node) {
+ // the dst-arity convention the other fusions follow; a and bias live on add_node
+ scope_op_debug_print scope_dbg_print(__func__, mul_node, /*num_src=*/2);
+
+ const ggml_tensor * a = add_node->src[0];
+ const ggml_tensor * bias = add_node->src[1];
+ const ggml_tensor * scale = (mul_node->src[0] == unary_node) ? mul_node->src[1] : mul_node->src[0];
+
+ // scale is picked by elimination; ggml_can_fuse()'s single-use rule rules out MUL(unary, unary)
+ GGML_ASSERT(scale != unary_node);
+ GGML_ASSERT(a->type == GGML_TYPE_F32 && bias->type == GGML_TYPE_F32);
+ GGML_ASSERT(scale->type == GGML_TYPE_F32 && mul_node->type == GGML_TYPE_F32);
+ GGML_ASSERT(ggml_are_same_shape(a, mul_node));
+ // a and dst are indexed flat
+ GGML_ASSERT(ggml_is_contiguous(a) && ggml_is_contiguous(mul_node));
+ // bias and scale are one contiguous ne0-length row each, broadcast over the outer dims
+ GGML_ASSERT(bias->ne[0] == a->ne[0] && scale->ne[0] == a->ne[0]);
+ GGML_ASSERT(ggml_nrows(bias) == 1 && ggml_nrows(scale) == 1);
+ GGML_ASSERT(ggml_is_contiguous(bias) && ggml_is_contiguous(scale));
+
+ queue_ptr main_stream = ctx.stream();
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
+
+ const auto dispatch_op = [&](auto op) {
+ add_unary_mul_sycl((const float *) a->data, (const float *) bias->data, (const float *) scale->data,
+ (float *) mul_node->data, ggml_nelements(mul_node), mul_node->ne[0], main_stream, op);
+ };
+
+ dispatch_fused_unary_op(ggml_get_unary_op(unary_node), dispatch_op);
}
__dpct_inline__ float ggml_sycl_op_swiglu_oai_single(float x, float g, float alpha = 1.702f, float limit = 7.0f) {
diff --git a/ggml/src/ggml-sycl/element_wise.hpp b/ggml/src/ggml-sycl/element_wise.hpp
index d280066ef..0a5bfa956 100644
--- a/ggml/src/ggml-sycl/element_wise.hpp
+++ b/ggml/src/ggml-sycl/element_wise.hpp
@@ -132,4 +132,9 @@ void ggml_sycl_arange(ggml_backend_sycl_context & ctx, ggml_tensor * dst);
// fused UNARY(silu|sigmoid|softplus) + MUL; see ggml_sycl_can_fuse() for the accepted shapes
void ggml_sycl_op_unary_mul_fused(ggml_backend_sycl_context & ctx, ggml_tensor * unary_node, ggml_tensor * mul_node);
+// fused f32 ADD + UNARY(silu|sigmoid|softplus) + MUL with the bias and the scale broadcast
+// over dim 0; see ggml_sycl_can_fuse() for the accepted shapes
+void ggml_sycl_op_add_unary_mul_fused(ggml_backend_sycl_context & ctx, ggml_tensor * add_node,
+ ggml_tensor * unary_node, ggml_tensor * mul_node);
+
#endif // GGML_SYCL_ELEMENTWISE_HPP
diff --git a/ggml/src/ggml-sycl/fusion.cpp b/ggml/src/ggml-sycl/fusion.cpp
index d3e995233..79fe13a1d 100644
--- a/ggml/src/ggml-sycl/fusion.cpp
+++ b/ggml/src/ggml-sycl/fusion.cpp
@@ -64,6 +64,12 @@ static bool ggml_sycl_should_fuse_mul_mat_glu(const ggml_tensor * gate, const gg
return true;
}
+// the unary ops the fused unary chains in element_wise.cpp have a functor for
+static bool ggml_sycl_fused_unary_has_kernel(ggml_unary_op unary_op) {
+ return unary_op == GGML_UNARY_OP_SILU || unary_op == GGML_UNARY_OP_SIGMOID ||
+ unary_op == GGML_UNARY_OP_SOFTPLUS;
+}
+
bool ggml_sycl_can_fuse(const ggml_cgraph * cgraph, int node_idx, std::initializer_list<enum ggml_op> ops,
std::initializer_list<enum ggml_unary_op> unary_ops) {
#ifndef NDEBUG
@@ -184,9 +190,7 @@ bool ggml_sycl_can_fuse(const ggml_cgraph * cgraph, int node_idx, std::initializ
return false;
}
- // the ops ggml_sycl_op_unary_mul_fused() has a kernel for
- if (unary_op != GGML_UNARY_OP_SILU && unary_op != GGML_UNARY_OP_SIGMOID &&
- unary_op != GGML_UNARY_OP_SOFTPLUS) {
+ if (!ggml_sycl_fused_unary_has_kernel(unary_op)) {
return false;
}
@@ -233,6 +237,55 @@ bool ggml_sycl_can_fuse(const ggml_cgraph * cgraph, int node_idx, std::initializ
return true;
}
+ // ADD(bias) + UNARY + MUL(scale): the delta-net alpha gate, softplus(alpha + dt) * a.
+ // The broadcast is what stops the same-shape UNARY + MUL branch above firing past one token.
+ if (ops.size() == 3 && ops.begin()[0] == GGML_OP_ADD && ops.begin()[1] == GGML_OP_UNARY &&
+ ops.begin()[2] == GGML_OP_MUL && unary_ops.size() == 1) {
+ const ggml_tensor * add = cgraph->nodes[node_idx];
+ const ggml_tensor * unary = cgraph->nodes[node_idx + 1];
+ const ggml_tensor * mul = cgraph->nodes[node_idx + 2];
+
+ const ggml_unary_op unary_op = ggml_get_unary_op(unary);
+ if (unary_op != unary_ops.begin()[0]) {
+ return false;
+ }
+
+ if (!ggml_sycl_fused_unary_has_kernel(unary_op)) {
+ return false;
+ }
+
+ // ggml_can_fuse() has already pinned the chain: unary consumes add, mul consumes
+ // unary, add and unary have one use each, and all three have the same shape
+ const ggml_tensor * a = add->src[0];
+ const ggml_tensor * bias = add->src[1];
+ const ggml_tensor * scale = (mul->src[0] == unary) ? mul->src[1] : mul->src[0];
+
+ if (a->type != GGML_TYPE_F32 || bias->type != GGML_TYPE_F32 ||
+ scale->type != GGML_TYPE_F32 || mul->type != GGML_TYPE_F32) {
+ return false;
+ }
+
+ // the activation and the destination are indexed flat
+ if (!ggml_is_contiguous(a) || !ggml_is_contiguous(mul) || !ggml_are_same_shape(a, mul)) {
+ return false;
+ }
+
+ // the kernel reads the bias and the scale as v[col], so each must be a single
+ // contiguous row spanning ne0
+ if (bias->ne[0] != a->ne[0] || scale->ne[0] != a->ne[0] ||
+ ggml_nrows(bias) != 1 || ggml_nrows(scale) != 1 ||
+ !ggml_is_contiguous(bias) || !ggml_is_contiguous(scale)) {
+ return false;
+ }
+
+ // the 32-bit fastdiv is inexact past 2^31; decline, the unfused path handles it
+ if (ggml_nelements(mul) >= ((int64_t) 1 << 31)) {
+ return false;
+ }
+
+ return true;
+ }
+
if (ops.size() == 3 && ops.begin()[0] == GGML_OP_SSM_CONV && ops.begin()[1] == GGML_OP_ADD &&
ops.begin()[2] == GGML_OP_UNARY && unary_ops.size() == 1 && unary_ops.begin()[0] == GGML_UNARY_OP_SILU) {
const ggml_tensor * ssm_conv = cgraph->nodes[node_idx];
diff --git a/ggml/src/ggml-sycl/ggml-sycl.cpp b/ggml/src/ggml-sycl/ggml-sycl.cpp
index 13fc02325..2a5184e01 100644
--- a/ggml/src/ggml-sycl/ggml-sycl.cpp
+++ b/ggml/src/ggml-sycl/ggml-sycl.cpp
@@ -6312,6 +6312,16 @@ static void ggml_backend_sycl_graph_compute_impl(ggml_backend_sycl_context * syc
i++;
continue;
}
+ // ADD(bias) + UNARY + MUL(scale) with both broadcast over dim 0, the form the branch
+ // above cannot take; ggml_get_unary_op() asserts, so check the op first.
+ if (node->op == GGML_OP_ADD && i + 2 < cgraph->n_nodes &&
+ cgraph->nodes[i + 1]->op == GGML_OP_UNARY &&
+ ggml_sycl_can_fuse(cgraph, i, { GGML_OP_ADD, GGML_OP_UNARY, GGML_OP_MUL },
+ { ggml_get_unary_op(cgraph->nodes[i + 1]) })) {
+ ggml_sycl_op_add_unary_mul_fused(*sycl_ctx, node, cgraph->nodes[i + 1], cgraph->nodes[i + 2]);
+ i += 2;
+ continue;
+ }
// Batch consecutive independent same-shape F32 L2_NORM siblings (the GDN q/k
// norms) into one launch; sources are strided views of the fused qkv buffer, so
diff --git a/tests/test-backend-ops.cpp b/tests/test-backend-ops.cpp
index cfe5b73e7..f31016bbb 100644
--- a/tests/test-backend-ops.cpp
+++ b/tests/test-backend-ops.cpp
@@ -4159,6 +4159,93 @@ struct test_unary_mul : public test_case {
}
};
+// GGML_OP_ADD + GGML_OP_UNARY(SILU|SIGMOID|SOFTPLUS) + GGML_OP_MUL with the ADD's bias and
+// the MUL's scale broadcast over dim 0: the delta-net alpha gate, softplus(alpha + dt) * a_coeff.
+struct test_add_unary_mul : public test_case {
+ const ggml_unary_op op;
+ const ggml_type type;
+ const std::array<int64_t, 4> ne;
+ const bool swap; // unary result is the second MUL operand
+ const std::string layout; // bias/scale layout, see build_graph()
+ const std::string tail; // extra consumer past the MUL, see build_graph()
+
+ std::string op_desc(ggml_tensor * t) override {
+ GGML_UNUSED(t);
+ return "ADD_" + std::string(ggml_unary_op_name(op)) + "_MUL";
+ }
+ bool run_whole_graph() override { return true; }
+
+ double max_nmse_err() override {
+ switch (type) {
+ // f16 never fuses (the kernel is f32-only), so this bound is the unfused
+ // chain's own f16 rounding drift, as in test_unary_mul
+ case GGML_TYPE_F16: return 5e-5;
+ // gelu never fuses either, and the backends' exp form drifts from the CPU's tanhf
+ default: return op == GGML_UNARY_OP_GELU ? 5e-7 : 1e-7;
+ }
+ }
+
+ std::string vars() override {
+ return VARS_TO_STR5(type, ne, swap, layout, tail);
+ }
+
+ test_add_unary_mul(ggml_unary_op op,
+ ggml_type type = GGML_TYPE_F32,
+ std::array<int64_t, 4> ne = {32, 7, 1, 1},
+ bool swap = false,
+ std::string layout = "bcast",
+ std::string tail = "")
+ : op(op), type(type), ne(ne), swap(swap), layout(std::move(layout)), tail(std::move(tail)) {}
+
+ ggml_tensor * build_graph(ggml_context * ctx) override {
+ ggml_tensor * a = ggml_new_tensor(ctx, type, 4, ne.data());
+ ggml_set_name(a, "a");
+
+ std::array<int64_t, 4> ne_v = { ne[0], 1, 1, 1 };
+ if (layout == "bcast") {
+ // one ne0 row each, broadcast over the outer dims, which is the alpha-gate form
+ } else if (layout == "same_shape") {
+ // no broadcast at all; fuses only while the activation is a single row
+ ne_v = ne;
+ } else if (layout == "rep_ne0") {
+ // repeat on dim 0, which bias[col] cannot address, so this must not fuse
+ ne_v[0] = ne[0] / 4;
+ } else {
+ GGML_ABORT("unknown layout %s", layout.c_str());
+ }
+
+ ggml_tensor * bias = ggml_new_tensor(ctx, type, 4, ne_v.data());
+ ggml_set_name(bias, "bias");
+
+ ggml_tensor * scale = ggml_new_tensor(ctx, type, 4, ne_v.data());
+ ggml_set_name(scale, "scale");
+
+ ggml_tensor * s = ggml_add(ctx, a, bias);
+ ggml_set_name(s, "add");
+
+ ggml_tensor * u = ggml_unary(ctx, s, op);
+ ggml_set_name(u, "unary");
+
+ // a broadcasting operand can only be the second one, so swap needs same-shape operands
+ ggml_tensor * out = swap ? ggml_mul(ctx, scale, u) : ggml_mul(ctx, u, scale);
+
+ if (tail == "reuse") {
+ // a second read of the add result must block the fusion
+ ggml_set_name(out, "mul");
+ out = ggml_add(ctx, out, s);
+ } else if (tail == "consumer") {
+ // fusion still applies; catches a dispatcher that skips one node too many
+ ggml_set_name(out, "mul");
+ out = ggml_add(ctx, out, scale);
+ } else if (!tail.empty()) {
+ GGML_ABORT("unknown tail %s", tail.c_str());
+ }
+ ggml_set_name(out, "out");
+
+ return out;
+ }
+};
+
// SNAKE activation fusion: y = x + sin(a*x)^2 * inv_b
// CUDA backend matches the naive 5-op chain (mul, sin, sqr, mul, add)
// and dispatches a single fused kernel.
@@ -9393,6 +9480,24 @@ static std::vector<std::unique_ptr<test_case>> make_test_cases_eval() {
}
}
+ // fused add + unary + mul: the delta-net alpha gate, bias and scale broadcast over dim 0
+ for (ggml_unary_op op : { GGML_UNARY_OP_SILU, GGML_UNARY_OP_SIGMOID, GGML_UNARY_OP_SOFTPLUS }) {
+ test_cases.emplace_back(new test_add_unary_mul(op, GGML_TYPE_F32, { 32, 7, 1, 1 }));
+ test_cases.emplace_back(new test_add_unary_mul(op, GGML_TYPE_F32, { 32, 512, 1, 1 }));
+ test_cases.emplace_back(new test_add_unary_mul(op, GGML_TYPE_F32, { 5, 7, 11, 13 }));
+ // one token: no broadcast left, and the unary result may be either MUL operand
+ test_cases.emplace_back(new test_add_unary_mul(op, GGML_TYPE_F32, { 32, 1, 1, 1 }, false, "same_shape"));
+ test_cases.emplace_back(new test_add_unary_mul(op, GGML_TYPE_F32, { 32, 1, 1, 1 }, true, "same_shape"));
+ test_cases.emplace_back(new test_add_unary_mul(op, GGML_TYPE_F32, { 32, 7, 1, 1 }, false, "bcast", "consumer"));
+ // must not fuse
+ test_cases.emplace_back(new test_add_unary_mul(op, GGML_TYPE_F32, { 32, 7, 1, 1 }, false, "same_shape"));
+ test_cases.emplace_back(new test_add_unary_mul(op, GGML_TYPE_F32, { 32, 7, 1, 1 }, false, "rep_ne0"));
+ test_cases.emplace_back(new test_add_unary_mul(op, GGML_TYPE_F32, { 32, 7, 1, 1 }, false, "bcast", "reuse"));
+ test_cases.emplace_back(new test_add_unary_mul(op, GGML_TYPE_F16, { 32, 7, 1, 1 }));
+ }
+ // a unary op with no fused kernel must fall back to the three-op chain
+ test_cases.emplace_back(new test_add_unary_mul(GGML_UNARY_OP_GELU, GGML_TYPE_F32, { 32, 7, 1, 1 }));
+
// SNAKE activation fusion: x + sin(a*x)^2 * inv_b
for (ggml_type type : { GGML_TYPE_F32, GGML_TYPE_F16, GGML_TYPE_BF16 }) {
test_cases.emplace_back(new test_snake_fuse(type, { 5, 7, 1, 1})); // primes sub-block