Commit 65840ed53 for llama.cpp
commit 65840ed53c8653bfbf3e9014d9cbf71ac8c08725
Author: Pascal <admin@serveurperso.com>
Date: Tue Oct 6 16:18:34 2026 +0200
ggml: fix CLAMP on non-contiguous views (CPU, CUDA) (#29517)
* ggml: fix CLAMP on non-contiguous views (CPU, CUDA)
CUDA clamped ggml_nelements values flat and ignored the view strides.
CPU addressed row j as j*nb01 and ignored nb02/nb03. Both now follow the
strides of dims 1..3; CUDA supports_op requires contiguous rows, like
Metal. test_clamp gains a non-contiguous view case.
* cuda: clamp kernel uses fastdiv for the view strides
diff --git a/ggml/src/ggml-cpu/ops.cpp b/ggml/src/ggml-cpu/ops.cpp
index 16a916d20..c7e7c47a7 100644
--- a/ggml/src/ggml-cpu/ops.cpp
+++ b/ggml/src/ggml-cpu/ops.cpp
@@ -6063,18 +6063,18 @@ static void ggml_compute_forward_clamp_f32(
const int n = ggml_nrows(src0);
const int nc = src0->ne[0];
- const size_t nb00 = src0->nb[0];
- const size_t nb01 = src0->nb[1];
-
- const size_t nb0 = dst->nb[0];
- const size_t nb1 = dst->nb[1];
+ GGML_TENSOR_UNARY_OP_LOCALS
GGML_ASSERT( nb0 == sizeof(float));
GGML_ASSERT(nb00 == sizeof(float));
for (int j = ith; j < n; j += nth) {
- float * dst_ptr = (float *) ((char *) dst->data + j*nb1);
- float * src0_ptr = (float *) ((char *) src0->data + j*nb01);
+ const int64_t i1 = j % ne01;
+ const int64_t i2 = (j / ne01) % ne02;
+ const int64_t i3 = j / (ne01*ne02);
+
+ float * dst_ptr = (float *) ((char *) dst->data + i1*nb1 + i2*nb2 + i3*nb3);
+ float * src0_ptr = (float *) ((char *) src0->data + i1*nb01 + i2*nb02 + i3*nb03);
for (int i = 0; i < nc; i++) {
dst_ptr[i] = MAX(MIN(src0_ptr[i], max), min);
@@ -6099,18 +6099,18 @@ static void ggml_compute_forward_clamp_f16(
const int n = ggml_nrows(src0);
const int nc = src0->ne[0];
- const size_t nb00 = src0->nb[0];
- const size_t nb01 = src0->nb[1];
-
- const size_t nb0 = dst->nb[0];
- const size_t nb1 = dst->nb[1];
+ GGML_TENSOR_UNARY_OP_LOCALS
GGML_ASSERT( nb0 == sizeof(ggml_fp16_t));
GGML_ASSERT(nb00 == sizeof(ggml_fp16_t));
for (int j = ith; j < n; j += nth) {
- ggml_fp16_t * dst_ptr = (ggml_fp16_t *) ((char *) dst->data + j*nb1);
- ggml_fp16_t * src0_ptr = (ggml_fp16_t *) ((char *) src0->data + j*nb01);
+ const int64_t i1 = j % ne01;
+ const int64_t i2 = (j / ne01) % ne02;
+ const int64_t i3 = j / (ne01*ne02);
+
+ ggml_fp16_t * dst_ptr = (ggml_fp16_t *) ((char *) dst->data + i1*nb1 + i2*nb2 + i3*nb3);
+ ggml_fp16_t * src0_ptr = (ggml_fp16_t *) ((char *) src0->data + i1*nb01 + i2*nb02 + i3*nb03);
for (int i = 0; i < nc; i++) {
float v = GGML_CPU_FP16_TO_FP32(src0_ptr[i]);
diff --git a/ggml/src/ggml-cuda/clamp.cu b/ggml/src/ggml-cuda/clamp.cu
index fe415e7f7..2727b7ec9 100644
--- a/ggml/src/ggml-cuda/clamp.cu
+++ b/ggml/src/ggml-cuda/clamp.cu
@@ -4,21 +4,42 @@ static __device__ __forceinline__ float op_clamp(float x, float min, float max)
return fminf(fmaxf(x, min), max);
}
+// src and dst may be views: rows are contiguous, dims 1..3 follow the strides (in elements).
template <class T>
-static __global__ void op_clamp_kernel(const T * x, T * dst, const T min, const T max, const int k) {
- const int i = blockDim.x*blockIdx.x + threadIdx.x;
+static __global__ void op_clamp_kernel(const T * x, T * dst, const T min, const T max, const uint32_t k,
+ const uint3 ne0, const uint3 ne1, const uint3 ne2,
+ const uint32_t s01, const uint32_t s02, const uint32_t s03,
+ const uint32_t s1, const uint32_t s2, const uint32_t s3) {
+ const uint32_t i = blockDim.x*blockIdx.x + threadIdx.x;
if (i >= k) {
return;
}
- dst[i] = (T)op_clamp((float)x[i], (float)min, (float)max);
+ const uint2 d0 = fast_div_modulo(i, ne0); // <i / ne0, i0>
+ const uint2 d1 = fast_div_modulo(d0.x, ne1); // <i / (ne0*ne1), i1>
+ const uint2 d2 = fast_div_modulo(d1.x, ne2); // <i3, i2>
+
+ const size_t i_src = d0.y + size_t(d1.y)*s01 + size_t(d2.y)*s02 + size_t(d2.x)*s03;
+ const size_t i_dst = d0.y + size_t(d1.y)*s1 + size_t(d2.y)*s2 + size_t(d2.x)*s3;
+
+ dst[i_dst] = (T)op_clamp((float)x[i_src], (float)min, (float)max);
}
template <class T>
-static void clamp_cuda(const T * x, T * dst, const T min, const T max, const int k, cudaStream_t stream) {
- const int num_blocks = (k + CUDA_CLAMP_BLOCK_SIZE - 1) / CUDA_CLAMP_BLOCK_SIZE;
- op_clamp_kernel<<<num_blocks, CUDA_CLAMP_BLOCK_SIZE, 0, stream>>>(x, dst, min, max, k);
+static void clamp_cuda(const T * x, T * dst, const T min, const T max, const ggml_tensor * src0, const ggml_tensor * t, cudaStream_t stream) {
+ const int64_t k = ggml_nelements(src0);
+ const size_t ts = sizeof(T);
+ GGML_ASSERT(k <= std::numeric_limits<uint32_t>::max());
+
+ const uint3 ne0 = init_fastdiv_values(src0->ne[0]);
+ const uint3 ne1 = init_fastdiv_values(src0->ne[1]);
+ const uint3 ne2 = init_fastdiv_values(src0->ne[2]);
+
+ const int64_t num_blocks = (k + CUDA_CLAMP_BLOCK_SIZE - 1) / CUDA_CLAMP_BLOCK_SIZE;
+ op_clamp_kernel<<<num_blocks, CUDA_CLAMP_BLOCK_SIZE, 0, stream>>>(x, dst, min, max, (uint32_t) k, ne0, ne1, ne2,
+ src0->nb[1]/ts, src0->nb[2]/ts, src0->nb[3]/ts,
+ t->nb[1]/ts, t->nb[2]/ts, t->nb[3]/ts);
}
@@ -31,6 +52,7 @@ void ggml_cuda_op_clamp(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16);
GGML_ASSERT( dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
GGML_ASSERT(src0->type == dst->type);
+ GGML_ASSERT(ggml_is_contiguous_rows(src0) && ggml_is_contiguous_rows(dst));
float min;
float max;
@@ -38,8 +60,8 @@ void ggml_cuda_op_clamp(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
memcpy(&max, (float *) dst->op_params + 1, sizeof(float));
if (src0->type == GGML_TYPE_F16) {
- clamp_cuda((const half *)src0_d, (half *)dst_d, (half)min, (half)max, ggml_nelements(src0), stream);
+ clamp_cuda((const half *)src0_d, (half *)dst_d, (half)min, (half)max, src0, dst, stream);
} else {
- clamp_cuda((const float *)src0_d, (float *)dst_d, (float)min, (float)max, ggml_nelements(src0), stream);
+ clamp_cuda((const float *)src0_d, (float *)dst_d, (float)min, (float)max, src0, dst, stream);
}
}
diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu
index 1122232a9..67b5e16d8 100644
--- a/ggml/src/ggml-cuda/ggml-cuda.cu
+++ b/ggml/src/ggml-cuda/ggml-cuda.cu
@@ -5609,11 +5609,12 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g
case GGML_OP_SQRT:
case GGML_OP_SIN:
case GGML_OP_COS:
- case GGML_OP_CLAMP:
case GGML_OP_LOG:
return true;
case GGML_OP_SCALE:
return (op->src[0]->type == GGML_TYPE_F32 || op->src[0]->type == GGML_TYPE_BF16) && op->type == op->src[0]->type;
+ case GGML_OP_CLAMP:
+ return ggml_is_contiguous_rows(op->src[0]);
case GGML_OP_ADD:
case GGML_OP_SUB:
case GGML_OP_MUL:
diff --git a/tests/test-backend-ops.cpp b/tests/test-backend-ops.cpp
index 51d309b46..cbe59178d 100644
--- a/tests/test-backend-ops.cpp
+++ b/tests/test-backend-ops.cpp
@@ -5710,19 +5710,34 @@ struct test_clamp : public test_case {
const std::array<int64_t, 4> ne;
float min;
float max;
+ int v; // view (1 : non-contiguous a)
std::string vars() override {
- return VARS_TO_STR4(type, ne, min, max);
+ return VARS_TO_STR5(type, ne, min, max, v);
}
test_clamp(ggml_type type = GGML_TYPE_F32,
std::array<int64_t, 4> ne = {10, 5, 4, 3},
- float min = -0.5f, float max = 0.5f)
- : type(type), ne(ne), min(min), max(max) {}
+ float min = -0.5f, float max = 0.5f, int v = 0)
+ : type(type), ne(ne), min(min), max(max), v(v) {}
ggml_tensor * build_graph(ggml_context * ctx) override {
- ggml_tensor * a = ggml_new_tensor(ctx, type, 4, ne.data());
- ggml_set_name(a, "a");
+ ggml_tensor * a;
+ if (v & 1) {
+ auto ne_a = ne;
+ ne_a[0] *= 3;
+ ne_a[1] *= 2;
+ ne_a[2] *= 5;
+ ne_a[3] *= 4;
+ a = ggml_new_tensor(ctx, type, 4, ne_a.data());
+ ggml_set_name(a, "a");
+
+ a = ggml_view_4d(ctx, a, ne[0], ne[1], ne[2], ne[3], a->nb[1], a->nb[2], a->nb[3], 0);
+ ggml_set_name(a, "view_of_a");
+ } else {
+ a = ggml_new_tensor(ctx, type, 4, ne.data());
+ ggml_set_name(a, "a");
+ }
ggml_tensor * out = ggml_clamp(ctx, a, min, max);
ggml_set_name(out, "out");
@@ -10772,6 +10787,7 @@ static std::vector<std::unique_ptr<test_case>> make_test_cases_eval() {
test_cases.emplace_back(new test_sin (type));
test_cases.emplace_back(new test_cos (type));
test_cases.emplace_back(new test_clamp (type));
+ test_cases.emplace_back(new test_clamp (type, {10, 5, 4, 3}, -0.5f, 0.5f, 1));
test_cases.emplace_back(new test_leaky_relu(type));
test_cases.emplace_back(new test_floor (type));
test_cases.emplace_back(new test_ceil (type));