Commit b23701f77 for llama.cpp

commit b23701f77d47dad9de834d59ebfcbe25c9e8b46f
Author: TheArchitectit <roger@vroger.com>
Date:   Sat Sep 19 00:32:52 2026 -0500

    cuda : fix CUB argsort corruption caused by in-place keys (#28389)

    argsort_f32_i32_cuda_cub called the one-shot DeviceRadixSort::SortPairs
    API with d_keys_in == d_keys_out (temp_keys, temp_keys). CUB's internal
    double-buffer ping-pong requires distinct key buffers: with aliased
    buffers the sort partially overwrites its own input mid-pass and emits a
    corrupted permutation, surfacing as intermittent garbage indices (e.g.
    backend top_k over a 248k-column vocab on Maxwell/CUDA 12.5/CCCL 2.x,
    which then triggered out-of-bounds gathers in downstream get_rows).

    Use a distinct keys-out buffer for all six call sites (plain and
    segmented, ascending and descending, size-query and execute).

    ---------

    Co-authored-by: Claude Opus 4.6 <noreply@anthropic.com>
    Co-authored-by: Oliver Simons <osimons@nvidia.com>

diff --git a/ggml/src/ggml-cuda/argsort.cu b/ggml/src/ggml-cuda/argsort.cu
index 26af90025..24115da09 100644
--- a/ggml/src/ggml-cuda/argsort.cu
+++ b/ggml/src/ggml-cuda/argsort.cu
@@ -51,9 +51,12 @@ void argsort_f32_i32_cuda_cub(ggml_cuda_pool & pool,
                               cudaStream_t     stream) {
     ggml_cuda_pool_alloc<int>   temp_indices_alloc(pool, ncols * nrows);
     ggml_cuda_pool_alloc<float> temp_keys_alloc(pool, ncols * nrows);
+    // Device*Sort algorithms currently do not allow for in-place sorting/aliasing of input/outputs
+    ggml_cuda_pool_alloc<float> temp_keys_out_alloc(pool, ncols * nrows);

     int *   temp_indices = temp_indices_alloc.get();
     float * temp_keys    = temp_keys_alloc.get();
+    float * temp_keys_out = temp_keys_out_alloc.get();

     static const int block_size = 256;
     const dim3 grid_size((ncols + block_size - 1) / block_size, nrows);
@@ -85,18 +88,18 @@ void argsort_f32_i32_cuda_cub(ggml_cuda_pool & pool,

     if (order == GGML_SORT_ORDER_ASC) {
         if (nrows == 1) {
-            CUDA_CHECK(DeviceRadixSort::SortPairs(nullptr, temp_storage_bytes, temp_keys, temp_keys,  // keys (in-place)
+            CUDA_CHECK(DeviceRadixSort::SortPairs(nullptr, temp_storage_bytes, temp_keys, temp_keys_out,  // keys in, keys out
                                                   temp_indices, dst,  // values (indices)
                                                   ncols, 0, sizeof(float) * 8, stream));
         } else if (is_capturing) {
             CUDA_CHECK(DeviceSegmentedRadixSort::SortPairs(
-                nullptr, temp_storage_bytes, temp_keys, temp_keys,  // keys (in-place)
+                nullptr, temp_storage_bytes, temp_keys, temp_keys_out,  // keys in, keys out
                 temp_indices, dst,                                  // values (indices)
                 ncols * nrows, nrows,                               // num items, num segments
                 offset_iterator, offset_iterator + 1, 0, sizeof(float) * 8, stream));
         } else {
             CUDA_CHECK(DeviceSegmentedSort::SortPairs(nullptr, temp_storage_bytes, temp_keys,
-                                                      temp_keys,             // keys (in-place)
+                                                      temp_keys_out, // keys out
                                                       temp_indices, dst,     // values (indices)
                                                       ncols * nrows, nrows,  // num items, num segments
                                                       offset_iterator, offset_iterator + 1, stream));
@@ -104,15 +107,15 @@ void argsort_f32_i32_cuda_cub(ggml_cuda_pool & pool,
     } else {
         if (nrows == 1) {
             CUDA_CHECK(DeviceRadixSort::SortPairsDescending(nullptr, temp_storage_bytes, temp_keys,
-                                                            temp_keys,          // keys (in-place)
+                                                            temp_keys_out, // keys out
                                                             temp_indices, dst,  // values (indices)
                                                             ncols, 0, sizeof(float) * 8, stream));
         } else if (is_capturing) {
             CUDA_CHECK(DeviceSegmentedRadixSort::SortPairsDescending(
-                nullptr, temp_storage_bytes, temp_keys, temp_keys, temp_indices, dst, ncols * nrows, nrows,
+                nullptr, temp_storage_bytes, temp_keys, temp_keys_out, temp_indices, dst, ncols * nrows, nrows,
                 offset_iterator, offset_iterator + 1, 0, sizeof(float) * 8, stream));
         } else {
-            CUDA_CHECK(DeviceSegmentedSort::SortPairsDescending(nullptr, temp_storage_bytes, temp_keys, temp_keys,
+            CUDA_CHECK(DeviceSegmentedSort::SortPairsDescending(nullptr, temp_storage_bytes, temp_keys, temp_keys_out,
                                                                 temp_indices, dst, ncols * nrows, nrows,
                                                                 offset_iterator, offset_iterator + 1, stream));
         }
@@ -124,31 +127,31 @@ void argsort_f32_i32_cuda_cub(ggml_cuda_pool & pool,
     if (order == GGML_SORT_ORDER_ASC) {
         if (nrows == 1) {
             CUDA_CHECK(DeviceRadixSort::SortPairs(d_temp_storage, temp_storage_bytes, temp_keys,
-                                                  temp_keys,          // keys (in-place)
+                                                  temp_keys_out, // keys out
                                                   temp_indices, dst,  // values (indices)
                                                   ncols, 0, sizeof(float) * 8, stream));
         } else if (is_capturing) {
-            CUDA_CHECK(DeviceSegmentedRadixSort::SortPairs(d_temp_storage, temp_storage_bytes, temp_keys, temp_keys,
+            CUDA_CHECK(DeviceSegmentedRadixSort::SortPairs(d_temp_storage, temp_storage_bytes, temp_keys, temp_keys_out,
                                                            temp_indices, dst, ncols * nrows, nrows, offset_iterator,
                                                            offset_iterator + 1, 0, sizeof(float) * 8, stream));
         } else {
-            CUDA_CHECK(DeviceSegmentedSort::SortPairs(d_temp_storage, temp_storage_bytes, temp_keys, temp_keys,
+            CUDA_CHECK(DeviceSegmentedSort::SortPairs(d_temp_storage, temp_storage_bytes, temp_keys, temp_keys_out,
                                                       temp_indices, dst, ncols * nrows, nrows, offset_iterator,
                                                       offset_iterator + 1, stream));
         }
     } else {
         if (nrows == 1) {
             CUDA_CHECK(DeviceRadixSort::SortPairsDescending(d_temp_storage, temp_storage_bytes, temp_keys,
-                                                            temp_keys,          // keys (in-place)
+                                                            temp_keys_out, // keys out
                                                             temp_indices, dst,  // values (indices)
                                                             ncols, 0, sizeof(float) * 8, stream));
         } else if (is_capturing) {
             CUDA_CHECK(DeviceSegmentedRadixSort::SortPairsDescending(
-                d_temp_storage, temp_storage_bytes, temp_keys, temp_keys, temp_indices, dst, ncols * nrows, nrows,
+                d_temp_storage, temp_storage_bytes, temp_keys, temp_keys_out, temp_indices, dst, ncols * nrows, nrows,
                 offset_iterator, offset_iterator + 1, 0, sizeof(float) * 8, stream));
         } else {
             CUDA_CHECK(DeviceSegmentedSort::SortPairsDescending(d_temp_storage, temp_storage_bytes, temp_keys,
-                                                                temp_keys, temp_indices, dst, ncols * nrows, nrows,
+                                                                temp_keys_out, temp_indices, dst, ncols * nrows, nrows,
                                                                 offset_iterator, offset_iterator + 1, stream));
         }
     }