From 32dd62ee6dfa80ada846551fefec215cefc5ae1c Mon Sep 17 00:00:00 2001 From: Pranesh Gonegandla Date: Thu, 1 Oct 2026 13:14:31 +0530 Subject: [PATCH] llama-mmap : avoid a second full-size copy of each tensor with direct-io (#29749) Assisted-by: Claude Co-authored-by: Pranesh Gonegandla --- src/llama-mmap.cpp | 21 +++++++++++++++++---- src/llama-mmap.h | 3 +++ src/llama-model-loader.cpp | 3 +-- 3 files changed, 21 insertions(+), 6 deletions(-) diff --git a/src/llama-mmap.cpp b/src/llama-mmap.cpp index 6047c10616..045489df7a 100644 --- a/src/llama-mmap.cpp +++ b/src/llama-mmap.cpp @@ -323,8 +323,11 @@ struct llama_file::impl { off_t offset_from_alignment = offset - aligned_offset; size_t bytes_to_read = (offset_from_alignment + size + alignment - 1) & ~(alignment - 1); + // stage through a bounded buffer, so that a large tensor is not held in memory twice while it loads + const size_t buffer_size = std::min(bytes_to_read, LLAMA_DIRECT_IO_BUFFER_SIZE); + void * raw_buffer = nullptr; - int ret = posix_memalign(&raw_buffer, alignment, bytes_to_read); + int ret = posix_memalign(&raw_buffer, alignment, buffer_size); if (ret != 0) { throw std::runtime_error(format("posix_memalign failed with error %d", ret)); } @@ -335,10 +338,20 @@ struct llama_file::impl { std::unique_ptr buffer(raw_buffer); seek(aligned_offset, SEEK_SET); - read_raw_unsafe(buffer.get(), bytes_to_read); - uintptr_t actual_data = reinterpret_cast(buffer.get()) + offset_from_alignment; - memcpy(dest, reinterpret_cast(actual_data), size); + size_t skip = offset_from_alignment; + size_t copied = 0; + for (size_t done = 0; done < bytes_to_read; ) { + const size_t n = std::min(buffer_size, bytes_to_read - done); + read_raw_unsafe(buffer.get(), n); + + const size_t count = std::min(n - skip, size - copied); + memcpy(reinterpret_cast(dest) + copied, reinterpret_cast(buffer.get()) + skip, count); + + copied += count; + skip = 0; + done += n; + } } void read_raw(void * ptr, size_t len) { diff --git a/src/llama-mmap.h b/src/llama-mmap.h index e75945286d..6c6393d573 100644 --- a/src/llama-mmap.h +++ b/src/llama-mmap.h @@ -6,6 +6,9 @@ #include #include +// staging buffer size for direct I/O reads, 64MB works well for NVMe drives +#define LLAMA_DIRECT_IO_BUFFER_SIZE (64 * 1024 * 1024) + struct llama_file; struct llama_mmap; struct llama_mlock; diff --git a/src/llama-model-loader.cpp b/src/llama-model-loader.cpp index 43c396f15a..6cc0b20288 100644 --- a/src/llama-model-loader.cpp +++ b/src/llama-model-loader.cpp @@ -1516,8 +1516,7 @@ bool llama_model_loader::load_all_data( } // Buffer size: balance between memory usage and I/O efficiency - // 64MB works well for NVMe drives - const size_t buffer_size = alignment != 1 ? 64 * 1024 * 1024 + 2 * alignment : 1 * 1024 * 1024; + const size_t buffer_size = alignment != 1 ? LLAMA_DIRECT_IO_BUFFER_SIZE + 2 * alignment : 1 * 1024 * 1024; std::vector host_buffers; std::vector events;