From 78264220d9407a1e7a6bca3b7a9c4191e24f3895 Mon Sep 17 00:00:00 2001 From: Yuri Khrustalev Date: Thu, 17 Sep 2026 03:19:44 -0400 Subject: [PATCH] gguf : align the data section relative to the GGUF start, not the file (llama/28993) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * gguf : align the data section relative to the GGUF start, not the file gguf_init_from_file_ptr reads a GGUF from the current file position, but padded the data section from file offset 0, so a GGUF embedded at an offset that is not a multiple of the alignment loaded without error and returned wrong tensor data. Also adds llama_adapter_lora_init_from_file_ptr, and disables mmap with a warning when an embedded data section is not aligned, instead of asserting in ggml. Assisted-by: Claude Opus 5 * llama : load lora from path through the FILE* variant The test now checks that mmap is disabled only for an unaligned offset. Assisted-by: Claude Fable 5.1 * Update ggml/src/gguf.cpp Co-authored-by: Johannes Gäßler * Update include/llama.h Co-authored-by: Johannes Gäßler * llama : error on unaligned mmap of an embedded GGUF, drop test-load-file-ptr --------- Co-authored-by: Johannes Gäßler --- ggml/src/gguf.cpp | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/ggml/src/gguf.cpp b/ggml/src/gguf.cpp index 144a8edf8..0eb9fb744 100644 --- a/ggml/src/gguf.cpp +++ b/ggml/src/gguf.cpp @@ -238,6 +238,7 @@ struct gguf_reader { : callback(callback), userdata(userdata), max_chunk_read(max_chunk_read), + start_offset(data_offset), data_offset(data_offset), nbytes_remain(nbytes_remain) { GGML_ASSERT(max_chunk_read > 0); @@ -366,6 +367,11 @@ struct gguf_reader { return data_offset; } + // position in the file where the GGUF data starts, alignment is relative to it, not to the file + uint64_t start() const { + return start_offset; + } + bool seek(uint64_t absolute_offset) const { const uint64_t end_offset = uint64_t(data_offset) + nbytes_remain; if (absolute_offset > end_offset) { @@ -415,6 +421,7 @@ private: gguf_reader_callback_t callback = nullptr; void * userdata = nullptr; size_t max_chunk_read = 0; + uint64_t start_offset = 0; mutable uint64_t data_offset = 0; mutable uint64_t nbytes_remain = 0; }; @@ -763,7 +770,7 @@ static struct gguf_context * gguf_init_from_reader(const struct gguf_reader & gr GGML_ASSERT(int64_t(ctx->info.size()) == n_tensors); // we require the data section to be aligned, so take into account any padding - if (n_tensors > 0 && !gr.seek(GGML_PAD(gr.tell(), ctx->alignment))) { + if (n_tensors > 0 && !gr.seek(gr.start() + GGML_PAD(gr.tell() - gr.start(), ctx->alignment))) { GGML_LOG_ERROR("%s: failed to seek to beginning of data section\n", __func__); gguf_free(ctx); return nullptr;