llama-mmap : avoid a second full-size copy of each tensor with direct-io (#29749)

Assisted-by: Claude

Co-authored-by: Pranesh Gonegandla <pgonegandla@nvidia.com>
This commit is contained in:
Pranesh Gonegandla
2026-10-01 09:44:31 +02:00
committed by GitHub
co-authored by Pranesh Gonegandla
parent f11d642a27
commit 32dd62ee6d
3 changed files with 21 additions and 6 deletions
+17 -4
View File
@@ -323,8 +323,11 @@ struct llama_file::impl {
off_t offset_from_alignment = offset - aligned_offset;
size_t bytes_to_read = (offset_from_alignment + size + alignment - 1) & ~(alignment - 1);
// stage through a bounded buffer, so that a large tensor is not held in memory twice while it loads
const size_t buffer_size = std::min<size_t>(bytes_to_read, LLAMA_DIRECT_IO_BUFFER_SIZE);
void * raw_buffer = nullptr;
int ret = posix_memalign(&raw_buffer, alignment, bytes_to_read);
int ret = posix_memalign(&raw_buffer, alignment, buffer_size);
if (ret != 0) {
throw std::runtime_error(format("posix_memalign failed with error %d", ret));
}
@@ -335,10 +338,20 @@ struct llama_file::impl {
std::unique_ptr<void, aligned_buffer_deleter> buffer(raw_buffer);
seek(aligned_offset, SEEK_SET);
read_raw_unsafe(buffer.get(), bytes_to_read);
uintptr_t actual_data = reinterpret_cast<uintptr_t>(buffer.get()) + offset_from_alignment;
memcpy(dest, reinterpret_cast<void *>(actual_data), size);
size_t skip = offset_from_alignment;
size_t copied = 0;
for (size_t done = 0; done < bytes_to_read; ) {
const size_t n = std::min(buffer_size, bytes_to_read - done);
read_raw_unsafe(buffer.get(), n);
const size_t count = std::min(n - skip, size - copied);
memcpy(reinterpret_cast<char *>(dest) + copied, reinterpret_cast<char *>(buffer.get()) + skip, count);
copied += count;
skip = 0;
done += n;
}
}
void read_raw(void * ptr, size_t len) {
+3
View File
@@ -6,6 +6,9 @@
#include <vector>
#include <cstdio>
// staging buffer size for direct I/O reads, 64MB works well for NVMe drives
#define LLAMA_DIRECT_IO_BUFFER_SIZE (64 * 1024 * 1024)
struct llama_file;
struct llama_mmap;
struct llama_mlock;
+1 -2
View File
@@ -1516,8 +1516,7 @@ bool llama_model_loader::load_all_data(
}
// Buffer size: balance between memory usage and I/O efficiency
// 64MB works well for NVMe drives
const size_t buffer_size = alignment != 1 ? 64 * 1024 * 1024 + 2 * alignment : 1 * 1024 * 1024;
const size_t buffer_size = alignment != 1 ? LLAMA_DIRECT_IO_BUFFER_SIZE + 2 * alignment : 1 * 1024 * 1024;
std::vector<ggml_backend_buffer_t> host_buffers;
std::vector<ggml_backend_event_t> events;