diff --git a/src/llama-mmap.cpp b/src/llama-mmap.cpp index 6047c10616..045489df7a 100644 --- a/src/llama-mmap.cpp +++ b/src/llama-mmap.cpp @@ -323,8 +323,11 @@ struct llama_file::impl { off_t offset_from_alignment = offset - aligned_offset; size_t bytes_to_read = (offset_from_alignment + size + alignment - 1) & ~(alignment - 1); + // stage through a bounded buffer, so that a large tensor is not held in memory twice while it loads + const size_t buffer_size = std::min(bytes_to_read, LLAMA_DIRECT_IO_BUFFER_SIZE); + void * raw_buffer = nullptr; - int ret = posix_memalign(&raw_buffer, alignment, bytes_to_read); + int ret = posix_memalign(&raw_buffer, alignment, buffer_size); if (ret != 0) { throw std::runtime_error(format("posix_memalign failed with error %d", ret)); } @@ -335,10 +338,20 @@ struct llama_file::impl { std::unique_ptr buffer(raw_buffer); seek(aligned_offset, SEEK_SET); - read_raw_unsafe(buffer.get(), bytes_to_read); - uintptr_t actual_data = reinterpret_cast(buffer.get()) + offset_from_alignment; - memcpy(dest, reinterpret_cast(actual_data), size); + size_t skip = offset_from_alignment; + size_t copied = 0; + for (size_t done = 0; done < bytes_to_read; ) { + const size_t n = std::min(buffer_size, bytes_to_read - done); + read_raw_unsafe(buffer.get(), n); + + const size_t count = std::min(n - skip, size - copied); + memcpy(reinterpret_cast(dest) + copied, reinterpret_cast(buffer.get()) + skip, count); + + copied += count; + skip = 0; + done += n; + } } void read_raw(void * ptr, size_t len) { diff --git a/src/llama-mmap.h b/src/llama-mmap.h index e75945286d..6c6393d573 100644 --- a/src/llama-mmap.h +++ b/src/llama-mmap.h @@ -6,6 +6,9 @@ #include #include +// staging buffer size for direct I/O reads, 64MB works well for NVMe drives +#define LLAMA_DIRECT_IO_BUFFER_SIZE (64 * 1024 * 1024) + struct llama_file; struct llama_mmap; struct llama_mlock; diff --git a/src/llama-model-loader.cpp b/src/llama-model-loader.cpp index 43c396f15a..6cc0b20288 100644 --- a/src/llama-model-loader.cpp +++ b/src/llama-model-loader.cpp @@ -1516,8 +1516,7 @@ bool llama_model_loader::load_all_data( } // Buffer size: balance between memory usage and I/O efficiency - // 64MB works well for NVMe drives - const size_t buffer_size = alignment != 1 ? 64 * 1024 * 1024 + 2 * alignment : 1 * 1024 * 1024; + const size_t buffer_size = alignment != 1 ? LLAMA_DIRECT_IO_BUFFER_SIZE + 2 * alignment : 1 * 1024 * 1024; std::vector host_buffers; std::vector events;