mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-02 02:47:26 -05:00
llama-mmap : avoid a second full-size copy of each tensor with direct-io (#29749)
Assisted-by: Claude Co-authored-by: Pranesh Gonegandla <pgonegandla@nvidia.com>
This commit is contained in:
co-authored by
Pranesh Gonegandla
parent
f11d642a27
commit
32dd62ee6d
+17
-4
@@ -323,8 +323,11 @@ struct llama_file::impl {
|
||||
off_t offset_from_alignment = offset - aligned_offset;
|
||||
size_t bytes_to_read = (offset_from_alignment + size + alignment - 1) & ~(alignment - 1);
|
||||
|
||||
// stage through a bounded buffer, so that a large tensor is not held in memory twice while it loads
|
||||
const size_t buffer_size = std::min<size_t>(bytes_to_read, LLAMA_DIRECT_IO_BUFFER_SIZE);
|
||||
|
||||
void * raw_buffer = nullptr;
|
||||
int ret = posix_memalign(&raw_buffer, alignment, bytes_to_read);
|
||||
int ret = posix_memalign(&raw_buffer, alignment, buffer_size);
|
||||
if (ret != 0) {
|
||||
throw std::runtime_error(format("posix_memalign failed with error %d", ret));
|
||||
}
|
||||
@@ -335,10 +338,20 @@ struct llama_file::impl {
|
||||
std::unique_ptr<void, aligned_buffer_deleter> buffer(raw_buffer);
|
||||
|
||||
seek(aligned_offset, SEEK_SET);
|
||||
read_raw_unsafe(buffer.get(), bytes_to_read);
|
||||
|
||||
uintptr_t actual_data = reinterpret_cast<uintptr_t>(buffer.get()) + offset_from_alignment;
|
||||
memcpy(dest, reinterpret_cast<void *>(actual_data), size);
|
||||
size_t skip = offset_from_alignment;
|
||||
size_t copied = 0;
|
||||
for (size_t done = 0; done < bytes_to_read; ) {
|
||||
const size_t n = std::min(buffer_size, bytes_to_read - done);
|
||||
read_raw_unsafe(buffer.get(), n);
|
||||
|
||||
const size_t count = std::min(n - skip, size - copied);
|
||||
memcpy(reinterpret_cast<char *>(dest) + copied, reinterpret_cast<char *>(buffer.get()) + skip, count);
|
||||
|
||||
copied += count;
|
||||
skip = 0;
|
||||
done += n;
|
||||
}
|
||||
}
|
||||
|
||||
void read_raw(void * ptr, size_t len) {
|
||||
|
||||
@@ -6,6 +6,9 @@
|
||||
#include <vector>
|
||||
#include <cstdio>
|
||||
|
||||
// staging buffer size for direct I/O reads, 64MB works well for NVMe drives
|
||||
#define LLAMA_DIRECT_IO_BUFFER_SIZE (64 * 1024 * 1024)
|
||||
|
||||
struct llama_file;
|
||||
struct llama_mmap;
|
||||
struct llama_mlock;
|
||||
|
||||
@@ -1516,8 +1516,7 @@ bool llama_model_loader::load_all_data(
|
||||
}
|
||||
|
||||
// Buffer size: balance between memory usage and I/O efficiency
|
||||
// 64MB works well for NVMe drives
|
||||
const size_t buffer_size = alignment != 1 ? 64 * 1024 * 1024 + 2 * alignment : 1 * 1024 * 1024;
|
||||
const size_t buffer_size = alignment != 1 ? LLAMA_DIRECT_IO_BUFFER_SIZE + 2 * alignment : 1 * 1024 * 1024;
|
||||
|
||||
std::vector<ggml_backend_buffer_t> host_buffers;
|
||||
std::vector<ggml_backend_event_t> events;
|
||||
|
||||
Reference in New Issue
Block a user