llama-mmap : avoid a second full-size copy of each tensor with direct-io (#29749)

Assisted-by: Claude

Co-authored-by: Pranesh Gonegandla <pgonegandla@nvidia.com>
This commit is contained in:
Pranesh Gonegandla
2026-10-01 09:44:31 +02:00
committed by GitHub
co-authored by Pranesh Gonegandla
parent f11d642a27
commit 32dd62ee6d
3 changed files with 21 additions and 6 deletions
+1 -2
View File
@@ -1516,8 +1516,7 @@ bool llama_model_loader::load_all_data(
}
// Buffer size: balance between memory usage and I/O efficiency
// 64MB works well for NVMe drives
const size_t buffer_size = alignment != 1 ? 64 * 1024 * 1024 + 2 * alignment : 1 * 1024 * 1024;
const size_t buffer_size = alignment != 1 ? LLAMA_DIRECT_IO_BUFFER_SIZE + 2 * alignment : 1 * 1024 * 1024;
std::vector<ggml_backend_buffer_t> host_buffers;
std::vector<ggml_backend_event_t> events;