mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-02 10:57:33 -05:00
llama-mmap : avoid a second full-size copy of each tensor with direct-io (#29749)
Assisted-by: Claude Co-authored-by: Pranesh Gonegandla <pgonegandla@nvidia.com>
This commit is contained in:
co-authored by
Pranesh Gonegandla
parent
f11d642a27
commit
32dd62ee6d
@@ -1516,8 +1516,7 @@ bool llama_model_loader::load_all_data(
|
||||
}
|
||||
|
||||
// Buffer size: balance between memory usage and I/O efficiency
|
||||
// 64MB works well for NVMe drives
|
||||
const size_t buffer_size = alignment != 1 ? 64 * 1024 * 1024 + 2 * alignment : 1 * 1024 * 1024;
|
||||
const size_t buffer_size = alignment != 1 ? LLAMA_DIRECT_IO_BUFFER_SIZE + 2 * alignment : 1 * 1024 * 1024;
|
||||
|
||||
std::vector<ggml_backend_buffer_t> host_buffers;
|
||||
std::vector<ggml_backend_event_t> events;
|
||||
|
||||
Reference in New Issue
Block a user