Commit 81aeaeb74 for llama.cpp

commit 81aeaeb74b205121d14475b0353e3749884a04b9
Author: Yuri Khrustalev <ykhrustalev@users.noreply.github.com>
Date:   Thu Sep 17 03:19:44 2026 -0400

    gguf : align the data section relative to the GGUF start, not the file (#28993)

    * gguf : align the data section relative to the GGUF start, not the file

    gguf_init_from_file_ptr reads a GGUF from the current file position, but padded
    the data section from file offset 0, so a GGUF embedded at an offset that is not
    a multiple of the alignment loaded without error and returned wrong tensor data.

    Also adds llama_adapter_lora_init_from_file_ptr, and disables mmap with a warning
    when an embedded data section is not aligned, instead of asserting in ggml.

    Assisted-by: Claude Opus 5

    * llama : load lora from path through the FILE* variant

    The test now checks that mmap is disabled only for an unaligned offset.

    Assisted-by: Claude Fable 5.1

    * Update ggml/src/gguf.cpp

    Co-authored-by: Johannes Gäßler <johannesg@5d6.de>

    * Update include/llama.h

    Co-authored-by: Johannes Gäßler <johannesg@5d6.de>

    * llama : error on unaligned mmap of an embedded GGUF, drop test-load-file-ptr

    ---------

    Co-authored-by: Johannes Gäßler <johannesg@5d6.de>

diff --git a/ggml/src/gguf.cpp b/ggml/src/gguf.cpp
index 144a8edf8..0eb9fb744 100644
--- a/ggml/src/gguf.cpp
+++ b/ggml/src/gguf.cpp
@@ -238,6 +238,7 @@ struct gguf_reader {
         : callback(callback),
           userdata(userdata),
           max_chunk_read(max_chunk_read),
+          start_offset(data_offset),
           data_offset(data_offset),
           nbytes_remain(nbytes_remain) {
         GGML_ASSERT(max_chunk_read > 0);
@@ -366,6 +367,11 @@ struct gguf_reader {
         return data_offset;
     }

+    // position in the file where the GGUF data starts, alignment is relative to it, not to the file
+    uint64_t start() const {
+        return start_offset;
+    }
+
     bool seek(uint64_t absolute_offset) const {
         const uint64_t end_offset = uint64_t(data_offset) + nbytes_remain;
         if (absolute_offset > end_offset) {
@@ -415,6 +421,7 @@ private:
     gguf_reader_callback_t callback = nullptr;
     void * userdata = nullptr;
     size_t max_chunk_read = 0;
+    uint64_t start_offset = 0;
     mutable uint64_t data_offset = 0;
     mutable uint64_t nbytes_remain = 0;
 };
@@ -763,7 +770,7 @@ static struct gguf_context * gguf_init_from_reader(const struct gguf_reader & gr
     GGML_ASSERT(int64_t(ctx->info.size()) == n_tensors);

     // we require the data section to be aligned, so take into account any padding
-    if (n_tensors > 0 && !gr.seek(GGML_PAD(gr.tell(), ctx->alignment))) {
+    if (n_tensors > 0 && !gr.seek(gr.start() + GGML_PAD(gr.tell() - gr.start(), ctx->alignment))) {
         GGML_LOG_ERROR("%s: failed to seek to beginning of data section\n", __func__);
         gguf_free(ctx);
         return nullptr;
diff --git a/include/llama.h b/include/llama.h
index 3ab935939..ac2215dc7 100644
--- a/include/llama.h
+++ b/include/llama.h
@@ -518,6 +518,8 @@ extern "C" {
               struct llama_model_params   params);

     // Load a model from an open FILE pointer
+    // The GGUF is read from the current position, so it can be embedded in a larger file
+    // mmap needs the GGUF data section at a file offset to be aligned to the CPU tensor alignment (32 bytes)
     LLAMA_API struct llama_model * llama_model_load_from_file_ptr(
                                    FILE * file,
               struct llama_model_params   params);
@@ -681,6 +683,11 @@ extern "C" {
             struct llama_model * model,
             const char * path_lora);

+    // Load a LoRA adapter from an open FILE pointer, reading from its current position
+    LLAMA_API struct llama_adapter_lora * llama_adapter_lora_init_from_file_ptr(
+            struct llama_model * model,
+            FILE * file);
+
     // Functions to access the adapter's GGUF metadata scalar values
     // - The functions return the length of the string on success, or -1 on failure
     // - The output string is always null-terminated and cleared on failure
diff --git a/src/llama-adapter.cpp b/src/llama-adapter.cpp
index e6678a66d..df3654d86 100644
--- a/src/llama-adapter.cpp
+++ b/src/llama-adapter.cpp
@@ -6,6 +6,8 @@

 #include <map>
 #include <cassert>
+#include <cerrno>
+#include <cstring>
 #include <sstream>
 #include <stdexcept>

@@ -146,22 +148,23 @@ llama_adapter_lora_weight * llama_adapter_lora::get_weight(ggml_tensor * w) {
     return nullptr;
 }

-static void llama_adapter_lora_init_impl(llama_model & model, const char * path_lora, llama_adapter_lora & adapter) {
-    LLAMA_LOG_INFO("%s: loading lora adapter from '%s' ...\n", __func__, path_lora);
-
+static void llama_adapter_lora_init_impl(llama_model & model, FILE * file, llama_adapter_lora & adapter) {
     ggml_context * ctx_init;
     gguf_init_params meta_gguf_params = {
         /* .no_alloc = */ true,
         /* .ctx      = */ &ctx_init,
     };

-    gguf_context_ptr ctx_gguf { gguf_init_from_file(path_lora, meta_gguf_params) };
+    gguf_context_ptr ctx_gguf { gguf_init_from_file_ptr(file, meta_gguf_params) };
     if (!ctx_gguf) {
-        throw std::runtime_error("failed to load lora adapter file from " + std::string(path_lora));
+        throw std::runtime_error("failed to load lora adapter from file");
     }

     ggml_context_ptr ctx { ctx_init };

+    // must come after gguf_init_from_file_ptr, the llama_file constructor moves the file position
+    llama_file gguf_file(file);
+
     // check metadata
     {
         const gguf_context * gguf_ctx = ctx_gguf.get();
@@ -393,7 +396,6 @@ static void llama_adapter_lora_init_impl(llama_model & model, const char * path_

     // set tensor data
     {
-        llama_file gguf_file(path_lora, "rb");
         std::vector<uint8_t> read_buf;
         auto set_tensor = [&](ggml_tensor * orig, ggml_tensor * dev) {
             const size_t offs = gguf_get_data_offset(ctx_gguf.get()) + gguf_get_tensor_offset(ctx_gguf.get(), gguf_find_tensor(ctx_gguf.get(), orig->name));
@@ -421,10 +423,30 @@ static void llama_adapter_lora_init_impl(llama_model & model, const char * path_
 }

 llama_adapter_lora * llama_adapter_lora_init(llama_model * model, const char * path_lora) {
+    LLAMA_LOG_INFO("%s: loading lora adapter from '%s' ...\n", __func__, path_lora);
+
+    FILE * file = ggml_fopen(path_lora, "rb");
+    if (!file) {
+        LLAMA_LOG_ERROR("%s: failed to open '%s': %s\n", __func__, path_lora, strerror(errno));
+        return nullptr;
+    }
+
+    llama_adapter_lora * adapter = llama_adapter_lora_init_from_file_ptr(model, file);
+    fclose(file);
+
+    return adapter;
+}
+
+llama_adapter_lora * llama_adapter_lora_init_from_file_ptr(llama_model * model, FILE * file) {
+    if (!file) {
+        LLAMA_LOG_ERROR("%s: file is NULL\n", __func__);
+        return nullptr;
+    }
+
     llama_adapter_lora * adapter = new llama_adapter_lora(model);

     try {
-        llama_adapter_lora_init_impl(*model, path_lora, *adapter);
+        llama_adapter_lora_init_impl(*model, file, *adapter);
         return adapter;
     } catch (const std::exception & err) {
         LLAMA_LOG_ERROR("%s: failed to apply lora adapter: %s\n", __func__, err.what());
diff --git a/src/llama-model-loader.cpp b/src/llama-model-loader.cpp
index 91bb5e7cc..43c396f15 100644
--- a/src/llama-model-loader.cpp
+++ b/src/llama-model-loader.cpp
@@ -685,6 +685,13 @@ llama_model_loader::llama_model_loader(
             throw std::runtime_error(format("%s: failed to load model from file pointer", __func__));
         }

+        // mmap places tensors at their file offsets, so an embedded GGUF must be aligned in the file too
+        const size_t tensor_align = ggml_backend_buft_get_alignment(ggml_backend_cpu_buffer_type());
+        if (use_mmap && gguf_get_data_offset(metadata) % tensor_align != 0) {
+            throw std::runtime_error(format("%s: GGUF data section at file offset %zu is not %zu byte aligned, cannot mmap",
+                __func__, gguf_get_data_offset(metadata), tensor_align));
+        }
+
         get_key(llm_kv(LLM_KV_GENERAL_ARCHITECTURE), arch_name, false);
         llm_kv = LLM_KV(llm_arch_from_string(arch_name));

diff --git a/tests/test-gguf.cpp b/tests/test-gguf.cpp
index fc636186f..f40d6984b 100644
--- a/tests/test-gguf.cpp
+++ b/tests/test-gguf.cpp
@@ -1167,15 +1167,17 @@ static bool same_tensor_data(const struct ggml_context * orig, const struct ggml

 enum roundtrip_read_mode {
     ROUNDTRIP_READ_MODE_FILE,
+    ROUNDTRIP_READ_MODE_FILE_OFFSET, // GGUF embedded after some bytes of a bigger file
     ROUNDTRIP_READ_MODE_BUFFER,
     ROUNDTRIP_READ_MODE_CALLBACK,
 };

 static const char * roundtrip_read_mode_name(const roundtrip_read_mode mode) {
     switch (mode) {
-        case ROUNDTRIP_READ_MODE_FILE:     return "file";
-        case ROUNDTRIP_READ_MODE_BUFFER:   return "buffer";
-        case ROUNDTRIP_READ_MODE_CALLBACK: return "callback";
+        case ROUNDTRIP_READ_MODE_FILE:        return "file";
+        case ROUNDTRIP_READ_MODE_FILE_OFFSET: return "file_offset";
+        case ROUNDTRIP_READ_MODE_BUFFER:      return "buffer";
+        case ROUNDTRIP_READ_MODE_CALLBACK:    return "callback";
     }

     GGML_ABORT("fatal error");
@@ -1214,6 +1216,12 @@ static std::pair<int, int> test_roundtrip(
     GGML_ASSERT(file);
 #endif // _WIN32

+    // not a multiple of any alignment, so the data section padding must be relative to the GGUF start
+    const long prefix = read_mode == ROUNDTRIP_READ_MODE_FILE_OFFSET ? 7 : 0;
+    for (long i = 0; i < prefix; ++i) {
+        fputc(0xAB, file);
+    }
+
     gguf_write_to_file_ptr(gguf_ctx_0, file, only_meta);
     rewind(file);

@@ -1236,6 +1244,7 @@ static std::pair<int, int> test_roundtrip(
         };
         gguf_ctx_1 = gguf_init_from_callback(read_buffer_callback, &reader, 4096, 4ull << 30 /* 4GB */, gguf_params);
     } else {
+        GGML_ASSERT(fseek(file, prefix, SEEK_SET) == 0);
         gguf_ctx_1 = gguf_init_from_file_ptr(file, gguf_params);
     }

@@ -1451,6 +1460,11 @@ int main(int argc, char ** argv) {
             npass += result.first;
             ntest += result.second;
         }
+        {
+            std::pair<int, int> result = test_roundtrip(dev, seed, /*only_meta=*/false, ROUNDTRIP_READ_MODE_FILE_OFFSET);
+            npass += result.first;
+            ntest += result.second;
+        }
         {
             std::pair<int, int> result = test_roundtrip(dev, seed, /*only_meta=*/false, ROUNDTRIP_READ_MODE_BUFFER);
             npass += result.first;